ethoscopy 2.3.0__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. ethoscopy-2.4.0/.claude/settings.local.json +48 -0
  2. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/Dockerfile +34 -8
  3. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/README.md +35 -0
  4. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/PKG-INFO +3 -1
  5. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/README.md +2 -0
  6. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/pyproject.toml +1 -1
  7. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/behavpy_core.py +271 -39
  8. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/behavpy_draw.py +83 -0
  9. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/behavpy_plotly.py +267 -0
  10. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/behavpy_seaborn.py +267 -0
  11. ethoscopy-2.4.0/src/ethoscopy/survival.py +335 -0
  12. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tasks/todo.md +100 -0
  13. ethoscopy-2.4.0/tests/test_survival.py +530 -0
  14. ethoscopy-2.4.0/uv.lock +1922 -0
  15. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/.codecov.yml +0 -0
  16. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/.github/workflows/ci.yml +0 -0
  17. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/.github/workflows/release.yml +0 -0
  18. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/.gitignore +0 -0
  19. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/.pre-commit-config.yaml +0 -0
  20. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/CLAUDE.md +0 -0
  21. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/.env.dummy.example +0 -0
  22. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/.env.github.example +0 -0
  23. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/.env.gitlab.example +0 -0
  24. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/.env.google.example +0 -0
  25. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/.env.keycloak.example +0 -0
  26. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/README_AUTH.md +0 -0
  27. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/config/jupyterhub_config.py +0 -0
  28. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/config/users.py +0 -0
  29. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/docker-compose.yml +0 -0
  30. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/install_r_packages.r +0 -0
  31. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/Docker/jupyterhub_data/jupyterhub_config.py +0 -0
  32. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/LICENSE +0 -0
  33. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/TESTING.md +0 -0
  34. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/_config.yml +0 -0
  35. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/pytest.ini +0 -0
  36. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/run_tests.py +0 -0
  37. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/scripts/README.md +0 -0
  38. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/scripts/convert_databases.sh +0 -0
  39. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/scripts/convert_wal_to_delete.py +0 -0
  40. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/scripts/publish_tutorials.py +0 -0
  41. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/setup.py +0 -0
  42. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/__init__.py +0 -0
  43. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/analyse.py +0 -0
  44. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/behavpy.py +0 -0
  45. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/behavpy_HMM_class.py +0 -0
  46. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/behavpy_class.py +0 -0
  47. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/behavpy_periodogram_class.py +0 -0
  48. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/load.py +0 -0
  49. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/metadata_db.py +0 -0
  50. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/misc/__init__.py +0 -0
  51. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/misc/circadian_bars.py +0 -0
  52. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/misc/general_functions.py +0 -0
  53. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/misc/get_HMM.py +0 -0
  54. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/misc/get_tutorials.py +0 -0
  55. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/misc/hmm_functions.py +0 -0
  56. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/misc/periodogram_functions.py +0 -0
  57. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/src/ethoscopy/misc/validate_datetime.py +0 -0
  58. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/__init__.py +0 -0
  59. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/conftest.py +0 -0
  60. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/data/README.md +0 -0
  61. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/data/test_ethoscope.db +0 -0
  62. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_analyse.py +0 -0
  63. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_baseline_enhancements.py +0 -0
  64. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_behavpy.py +0 -0
  65. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_behavpy_core_simple.py +0 -0
  66. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_compatibility_classes.py +0 -0
  67. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_general_functions.py +0 -0
  68. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_get_tutorials.py +0 -0
  69. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_load.py +0 -0
  70. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_load_comprehensive.py +0 -0
  71. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_load_diagnostics.py +0 -0
  72. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_load_light.py +0 -0
  73. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_load_metadata_fixes.py +0 -0
  74. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_load_optimizations.py +0 -0
  75. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tests/test_load_wal.py +0 -0
  76. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/1_Overview_tutorial.ipynb +0 -0
  77. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/2_HMM_tutorial.ipynb +0 -0
  78. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/3_Circadian_tutorial.ipynb +0 -0
  79. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/4_Navigating_db_tutorial.ipynb +0 -0
  80. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/5_Ethoscopy_catch22_tutorial.ipynb +0 -0
  81. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/6_Ethoscopy_to_hctsa_tutorial.ipynb +0 -0
  82. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/ethoscope_db.csv +0 -0
  83. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/jones_et_al_metadata.csv +0 -0
  84. {ethoscopy-2.3.0 → ethoscopy-2.4.0}/tutorial_notebook/notebook_paper.ipynb +0 -0
@@ -0,0 +1,48 @@
1
+ {
2
+ "permissions": {
3
+ "allow": [
4
+ "Bash(wc:*)",
5
+ "Bash(paplay:*)",
6
+ "Bash(ETHOSCOPE_DATA_PATH=/mnt/ethoscope_data docker compose:*)",
7
+ "Bash(docker ps:*)",
8
+ "Bash(docker logs:*)",
9
+ "Bash(docker compose:*)",
10
+ "Bash(curl:*)",
11
+ "Bash(docker exec:*)",
12
+ "Bash(mkdir -p /tmp/ethoscopy_wheel_inspect)",
13
+ "Read(//tmp/**)",
14
+ "Bash(pip download *)",
15
+ "Bash(unzip -l ethoscopy-2.0.4-py3-none-any.whl)",
16
+ "Bash(git check-ignore *)",
17
+ "Bash(/home/gg/Code/ethoscope_project/ethoscopy/.venv/bin/pip install *)",
18
+ "Bash(.venv/bin/python -m pytest tests/test_get_tutorials.py -v)",
19
+ "Bash(.venv/bin/python *)",
20
+ "Bash(.venv/bin/pip install *)",
21
+ "Bash(gh auth *)",
22
+ "Bash(gh repo *)",
23
+ "Bash(docker info *)",
24
+ "mcp__bookstack__bookstack_search",
25
+ "Bash(.venv/bin/ruff check *)",
26
+ "Bash(.venv/bin/black --check src/ tests/)",
27
+ "Bash(.venv/bin/black src/ethoscopy/misc/get_tutorials.py tests/test_get_tutorials.py)",
28
+ "mcp__bookstack__bookstack_pages_read",
29
+ "mcp__bookstack__bookstack_pages_update",
30
+ "Bash(gh issue create --repo gilestrolab/ethoscopy --title 'Tutorial data pickles missing from PyPI wheel \\(2.0.0 – 2.0.4\\)' --body ' *)",
31
+ "Bash(git add *)",
32
+ "Bash(git commit *)",
33
+ "Bash(git push *)",
34
+ "Bash(git tag *)",
35
+ "Bash(gh release create v2.0.5 --repo gilestrolab/ethoscopy --title 'v2.0.5 — package tutorial data separately' --notes ' *)",
36
+ "Bash(gh run *)",
37
+ "Bash(gh issue *)",
38
+ "Bash(.venv/bin/twine check *)",
39
+ "Bash(.venv/bin/twine upload *)",
40
+ "Bash(.venv/bin/pip index *)",
41
+ "Bash(ETHOSCOPE_LAB_TAG=1.2 docker compose build)",
42
+ "Bash(gh release *)",
43
+ "Bash(ETHOSCOPE_LAB_TAG=1.2 docker compose build --no-cache --progress=plain)",
44
+ "Bash(grep -E --line-buffered '^#[0-9]+ \\(ERROR|CANCELED\\)|^#[0-9]+ DONE [0-9.]+s$|^ERROR|^failed to solve|Execution halted|^Error: Required R packages|non-zero code|^naming to|^ *=> .* done *$')",
45
+ "Bash(grep -E --line-buffered '^#[0-9]+ \\(ERROR|CANCELED\\)$|^#[0-9]+ DONE [0-9.]+s$|^failed to solve|Execution halted|^Error: Required R packages|Successfully built|naming to docker\\\\.io')"
46
+ ]
47
+ }
48
+ }
@@ -105,19 +105,13 @@ RUN pip3 install \
105
105
  openpyxl \
106
106
  pyarrow \
107
107
  tqdm \
108
- jupyterlab-git \
109
- ethoscopy==2.2.1
108
+ jupyterlab-git
110
109
  # pycatch22
111
110
 
112
- # Pre-populate tutorial datasets inside the installed package so non-root
113
- # JupyterHub users don't have to download them on first notebook run. The
114
- # pickles (~36 MB total) live in the ethoscopy repo, not the PyPI wheel.
115
- RUN python3 -c "from ethoscopy.misc.get_tutorials import download_tutorial_data, package_tutorial_data_dir; download_tutorial_data(dest_dir=package_tutorial_data_dir(), verbose=True)"
116
-
117
111
  # Install Ethoscope from source
118
112
  RUN git clone https://github.com/gilestrolab/ethoscope.git /opt/ethoscope && \
119
113
  cd /opt/ethoscope/ && git checkout dev && \
120
- cd /opt/ethoscope/src/ethoscope && pip install .
114
+ cd /opt/ethoscope/src/ethoscope && pip install .
121
115
 
122
116
  # ============================================================================
123
117
  # USER SETUP
@@ -127,6 +121,38 @@ RUN git clone https://github.com/gilestrolab/ethoscope.git /opt/ethoscope && \
127
121
  RUN useradd -m ethoscopelab && \
128
122
  echo 'ethoscopelab:ethoscope' | chpasswd
129
123
 
124
+ # ============================================================================
125
+ # ETHOSCOPY (kept last on purpose)
126
+ # ============================================================================
127
+ # Reason: ethoscopy is the fastest-moving pin in this image and the one most
128
+ # often bumped for a bugfix. Everything above — the ~1 h R-package compile, the
129
+ # scientific pip stack, ethoscope-from-source — is stable, so installing
130
+ # ethoscopy in its own late layer means a version bump rebuilds only these two
131
+ # steps (seconds) instead of invalidating the whole pip block. Keep this the
132
+ # last software install; put nothing slow after it. Bumping the pin below is the
133
+ # only change needed to ship a new ethoscopy into the image.
134
+ RUN pip3 install ethoscopy==2.3.0
135
+
136
+ # Pre-populate tutorial datasets inside the installed package so non-root
137
+ # JupyterHub users don't have to download them on first notebook run. The
138
+ # pickles (~36 MB total) live in the ethoscopy repo, not the PyPI wheel.
139
+ #
140
+ # This is best-effort. urlretrieve() has no timeout, so a stalled connection to
141
+ # GitHub would hang the build forever (observed: a 36 MB / ~8 s download stuck
142
+ # >25 min); cap each connection with socket.setdefaulttimeout and retry, using
143
+ # overwrite so a socket killed mid-write cannot leave a partial file the next
144
+ # attempt skips as "already present". If it still cannot reach GitHub (e.g. a
145
+ # build host whose Docker bridge network is firewalled off from GitHub while the
146
+ # container's runtime network is not), warn and continue rather than failing the
147
+ # whole image: ethoscopy fetches the tutorial data on demand at first use, so a
148
+ # missing prefetch only costs a one-time download inside the user's session.
149
+ RUN for attempt in 1 2 3; do \
150
+ python3 -c "import socket; socket.setdefaulttimeout(60); from ethoscopy.misc.get_tutorials import download_tutorial_data, package_tutorial_data_dir; download_tutorial_data(dest_dir=package_tutorial_data_dir(), overwrite=True, verbose=True)" && exit 0; \
151
+ echo "tutorial download attempt $attempt failed; retrying in 5s..."; \
152
+ sleep 5; \
153
+ done; \
154
+ echo "WARNING: could not pre-populate tutorial data at build time; it will download on first use at runtime." >&2
155
+
130
156
  # ============================================================================
131
157
  # FINAL CONFIGURATION
132
158
  # ============================================================================
@@ -12,6 +12,41 @@ The command to use to recreate that image is `JUPYTER_HUB_TAG=5.3.0 ETHOSCOPE_LA
12
12
 
13
13
  After creation, the image can be run using the enclosed `docker-compose.yml` file, replacing values as fit.
14
14
 
15
+ ### The GPU does not speed up the build (it is a *runtime* resource)
16
+
17
+ A recurring confusion: **`docker compose build` cannot use the GPU, no matter what.**
18
+ The host does have an NVIDIA GPU (e.g. an RTX A4000) and the *running* container is
19
+ given access to it via the `deploy.resources.reservations.devices: capabilities: ["gpu"]`
20
+ block in `docker-compose.yml` — that is what GPU-aware analysis inside the notebook uses.
21
+ But building the image is an entirely separate phase that runs on CPU: BuildKit has no
22
+ GPU path, and every slow step here is CPU/IO-bound anyway.
23
+
24
+ What actually makes the build slow (~1–1.5 h from cold) is, in order:
25
+
26
+ 1. **Compiling the 8 R packages** from source (`Rscript install_r_packages.r`) — by far
27
+ the biggest cost, and pure CPU.
28
+ 2. `pip install` of the JupyterLab / notebook-intelligence / scientific stack.
29
+ 3. `apt-get` and the `git clone` of ethoscope.
30
+
31
+ None of these is GPU-accelerable. If a rebuild is taking the full hour+ it is almost
32
+ always because the **build cache was lost**, forcing the R compile to re-run. To avoid
33
+ re-paying it:
34
+
35
+ - **Do not kill a build mid-layer.** BuildKit only commits a layer's cache when that
36
+ layer *finishes*; a build killed during the R step discards it and the next build
37
+ recompiles from scratch.
38
+ - **Bumping `ethoscopy==X.Y.Z` only invalidates the pip layer and below.** The R compile
39
+ sits earlier in the `Dockerfile`, so a version-pin bump should reuse the cached R layer
40
+ and finish in minutes — *provided the cache still exists*. Old cache is eventually
41
+ evicted (`docker buildx du` shows what is reclaimable), which is the usual reason an
42
+ "only changed the pin" rebuild still takes an hour.
43
+ - **Prefer pulling over building.** Regular deployments should `docker pull
44
+ ggilestro/ethoscope-lab:<tag>` rather than rebuild; only rebuild when the recipe itself
45
+ changes, then push the result so nobody else has to.
46
+
47
+ Rule of thumb: reach for the GPU to make *analysis* faster, never to make the *image build*
48
+ faster — there is nothing to accelerate there.
49
+
15
50
  ## Authentication and User Management
16
51
 
17
52
  This JupyterHub instance supports multiple authentication methods including:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: ethoscopy
3
- Version: 2.3.0
3
+ Version: 2.4.0
4
4
  Summary: "A python based toolkit to download and anlyse data from the Ethoscope hardware system."
5
5
  Author-email: Lblackhurst29 <lblackhurst29@gmail.com>
6
6
  License-File: LICENSE
@@ -56,6 +56,8 @@ At its core ethoscopy is a subclass of the data manipulation tool Pandas. The da
56
56
 
57
57
  Ethoscopy contains methods to perform common analytical techniques per specimen in the data table, such as removing dead specimens, interpolating missing values, or calculating sleep from movement. Addtionally, specialist anlysing tools have been implemented for analysing circadian rhythm, such as periodograms, and for generating hidden Markov models (HMM) to understand latent behavioural states. HMMs are trained utilising hmmlearn in the background and come accompanied with a range of visualisation tools to understand the generated model.
58
58
 
59
+ For lifespan experiments, `survival_table()` and `km_survival_plot()` estimate Kaplan-Meier survival curves with Greenwood confidence bands. Death is detected from sustained immobility using the same rule as `curate_dead_animals()`, and specimens still moving when their recording ends are censored rather than counted as deaths. Pass `subject_cols` when one animal was recorded across several sessions so its recordings are treated as one individual.
60
+
59
61
  ### -- Update to 2.0 --
60
62
 
61
63
  This new update sees a whole refactoring of the code base to make everything more streamline and keep the package up to date with the new versions of pandas and numpy. Gone are seperate classes for periodograms and HMM based analysis, all are under one class behavpy(). Addtioanlly, now the user can choose between plotter packages, Seaborn and Plotly, and choose a desired colour pallete. The previous used package Plotly can balloon the size of jupyter notebooks, putting a strain on storage, despite being great for data exploration. If you just want static plots, use Seaborn. But be wary of comparison, the backend for Plotly plots is all calculated in ethoscopy applying z-score and bootstrapping to quantification plots, whereas Seaborn based plots will use the Seaborn internal tools for errors and averaging.
@@ -17,6 +17,8 @@ At its core ethoscopy is a subclass of the data manipulation tool Pandas. The da
17
17
 
18
18
  Ethoscopy contains methods to perform common analytical techniques per specimen in the data table, such as removing dead specimens, interpolating missing values, or calculating sleep from movement. Addtionally, specialist anlysing tools have been implemented for analysing circadian rhythm, such as periodograms, and for generating hidden Markov models (HMM) to understand latent behavioural states. HMMs are trained utilising hmmlearn in the background and come accompanied with a range of visualisation tools to understand the generated model.
19
19
 
20
+ For lifespan experiments, `survival_table()` and `km_survival_plot()` estimate Kaplan-Meier survival curves with Greenwood confidence bands. Death is detected from sustained immobility using the same rule as `curate_dead_animals()`, and specimens still moving when their recording ends are censored rather than counted as deaths. Pass `subject_cols` when one animal was recorded across several sessions so its recordings are treated as one individual.
21
+
20
22
  ### -- Update to 2.0 --
21
23
 
22
24
  This new update sees a whole refactoring of the code base to make everything more streamline and keep the package up to date with the new versions of pandas and numpy. Gone are seperate classes for periodograms and HMM based analysis, all are under one class behavpy(). Addtioanlly, now the user can choose between plotter packages, Seaborn and Plotly, and choose a desired colour pallete. The previous used package Plotly can balloon the size of jupyter notebooks, putting a strain on storage, despite being great for data exploration. If you just want static plots, use Seaborn. But be wary of comparison, the backend for Plotly plots is all calculated in ethoscopy applying z-score and bootstrapping to quantification plots, whereas Seaborn based plots will use the Seaborn internal tools for errors and averaging.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "ethoscopy"
3
- version = "2.3.0"
3
+ version = "2.4.0"
4
4
  description = "\"A python based toolkit to download and anlyse data from the Ethoscope hardware system.\""
5
5
  authors = [{name = "Lblackhurst29",email = "lblackhurst29@gmail.com"}]
6
6
  readme = "README.md"
@@ -13,6 +13,11 @@ from tqdm.auto import tqdm
13
13
 
14
14
  from ethoscopy.analyse import max_velocity_detector
15
15
  from ethoscopy.misc.general_functions import concat, rle
16
+ from ethoscopy.survival import (
17
+ kaplan_meier,
18
+ sliding_window_death,
19
+ )
20
+ from ethoscopy.survival import survival_table as build_survival_table
16
21
  from ethoscopy.misc.periodogram_functions import ( # noqa: F401 — resolved by eval() in _check_periodogram_input
17
22
  chi_squared,
18
23
  fourier,
@@ -688,24 +693,31 @@ class behavpy_core(pd.DataFrame):
688
693
  # Already numeric
689
694
  return float(val) if val is not None else 0
690
695
 
691
- # Process data by specimen ID groups to avoid loading entire dataset in memory
692
- result_list = []
693
- for specimen_id, group in self.groupby(level="id"):
694
- if specimen_id in day_dict:
695
- shift_days = convert_baseline_value(day_dict[specimen_id])
696
- if shift_days != 0: # Only copy and modify if there's actually a shift
697
- group = group.copy()
698
- group[t_column] = group[t_column] + (
699
- shift_days * day_length_seconds
700
- )
701
- result_list.append(group)
696
+ # Vectorised shift: one pass over the data instead of a copy per specimen.
697
+ # Reason: specimens absent from the metadata column keep an integer 0 so
698
+ # that a dataset shifted only by whole-number days keeps an integer
699
+ # timestamp column, as the per-specimen implementation did.
700
+ shift_map = {
701
+ specimen_id: (
702
+ convert_baseline_value(day_dict[specimen_id]) * day_length_seconds
703
+ if specimen_id in day_dict
704
+ else 0
705
+ )
706
+ for specimen_id in self.index.unique()
707
+ }
708
+ shifts = self.index.map(shift_map)
702
709
 
703
- # Concatenate results
704
- if result_list:
705
- new_data = pd.concat(result_list, axis=0)
706
- else:
707
- # Return empty DataFrame with same structure
708
- new_data = self.iloc[:0].copy()
710
+ new_data = self.copy(deep=False)
711
+ if (shifts != 0).any():
712
+ # Adding the mapped array reproduces the dtype the previous
713
+ # per-specimen concatenation produced: integer shifts keep an
714
+ # integer column, fractional ones promote it to float.
715
+ new_data[t_column] = self[t_column].to_numpy() + shifts.to_numpy()
716
+
717
+ # Grouping by specimen used to sort the frame by id as a side effect;
718
+ # preserve that ordering for data that does not already arrive sorted.
719
+ if not new_data.index.is_monotonic_increasing:
720
+ new_data = new_data.sort_index(kind="stable")
709
721
 
710
722
  # Return new behavpy object
711
723
  return self.__class__(
@@ -1143,6 +1155,7 @@ class behavpy_core(pd.DataFrame):
1143
1155
  prop_immobile: float,
1144
1156
  resolution: int,
1145
1157
  time_dict: Optional[Dict[str, List[int]]] = None,
1158
+ min_coverage: Optional[float] = None,
1146
1159
  ) -> pd.DataFrame:
1147
1160
  """
1148
1161
  Internal method to detect and remove data after presumed death for a single specimen.
@@ -1159,6 +1172,10 @@ class behavpy_core(pd.DataFrame):
1159
1172
  resolution (int): Number of segments to divide each time window into
1160
1173
  time_dict (Optional[Dict[str, List[int]]], optional): Dictionary to store valid time ranges.
1161
1174
  Keys are specimen IDs, values are [start_time, death_time]. Defaults to None.
1175
+ min_coverage (Optional[float], optional): Fraction of the expected samples a window
1176
+ must contain before it can be judged, expected samples being derived from the
1177
+ specimen's own median sampling interval. None judges every window, as previous
1178
+ versions did. Defaults to None.
1162
1179
 
1163
1180
  Returns:
1164
1181
  pd.DataFrame: Filtered DataFrame containing only data before detected death point
@@ -1174,28 +1191,17 @@ class behavpy_core(pd.DataFrame):
1174
1191
  raise ValueError("resolution cannot be larger than time_window")
1175
1192
 
1176
1193
  time_window = 60 * 60 * time_window
1177
- d = data[[time_var, moving_var]].copy(deep=True)
1178
- target_t = np.array(
1179
- list(
1180
- range(
1181
- d[time_var].min().astype(int),
1182
- d[time_var].max().astype(int),
1183
- floor(time_window / resolution),
1184
- )
1185
- )
1186
- )
1187
- local_means = np.array(
1188
- [
1189
- d[d[time_var].between(i, i + time_window)][moving_var].mean()
1190
- for i in target_t
1191
- ]
1194
+ first_death_time = sliding_window_death(
1195
+ data[time_var].to_numpy(),
1196
+ data[moving_var].to_numpy(),
1197
+ time_window_s=time_window,
1198
+ step=floor(time_window / resolution),
1199
+ prop_immobile=prop_immobile,
1200
+ min_coverage=min_coverage,
1192
1201
  )
1193
1202
 
1194
- # Find indices where animal is considered dead
1195
- death_points = np.where(local_means <= prop_immobile)[0]
1196
-
1197
1203
  # If no death points found, return original data
1198
- if len(death_points) == 0:
1204
+ if first_death_time is None:
1199
1205
  if time_dict is not None:
1200
1206
  time_dict[data["id"].iloc[0]] = [
1201
1207
  data[time_var].min(),
@@ -1203,9 +1209,6 @@ class behavpy_core(pd.DataFrame):
1203
1209
  ]
1204
1210
  return data
1205
1211
 
1206
- # Get first death point
1207
- first_death_time = target_t[death_points[0]]
1208
-
1209
1212
  if time_dict is not None:
1210
1213
  time_dict[data["id"].iloc[0]] = [data[time_var].min(), first_death_time]
1211
1214
 
@@ -1218,6 +1221,7 @@ class behavpy_core(pd.DataFrame):
1218
1221
  time_window: int = 24,
1219
1222
  prop_immobile: float = 0.01,
1220
1223
  resolution: int = 24,
1224
+ min_coverage: Optional[float] = None,
1221
1225
  ) -> "behavpy_core":
1222
1226
  """
1223
1227
  Detect and remove data after specimens are presumed dead based on extended immobility.
@@ -1237,6 +1241,11 @@ class behavpy_core(pd.DataFrame):
1237
1241
  to consider specimen dead (0-1). Defaults to 0.01 (1%).
1238
1242
  resolution (int, optional): Number of segments to divide each time window into.
1239
1243
  Controls overlap between windows. Defaults to 24.
1244
+ min_coverage (float, optional): Fraction of expected samples a window must hold
1245
+ before it is judged, expected samples being derived from each specimen's own
1246
+ median sampling interval. Use it when recordings contain gaps, where a short
1247
+ stretch of data at a machine stop can otherwise read as death. None judges
1248
+ every window, matching previous versions. Defaults to None.
1240
1249
 
1241
1250
  Returns:
1242
1251
  behavpy_core: Filtered behavpy object containing only data before detected death points.
@@ -1260,6 +1269,7 @@ class behavpy_core(pd.DataFrame):
1260
1269
  time_window=time_window,
1261
1270
  prop_immobile=prop_immobile,
1262
1271
  resolution=resolution,
1272
+ min_coverage=min_coverage,
1263
1273
  )
1264
1274
  ),
1265
1275
  tdf.meta,
@@ -1276,6 +1286,7 @@ class behavpy_core(pd.DataFrame):
1276
1286
  time_window: int = 24,
1277
1287
  prop_immobile: float = 0.01,
1278
1288
  resolution: int = 24,
1289
+ min_coverage: Optional[float] = None,
1279
1290
  ) -> Tuple["behavpy_core", "behavpy_core"]:
1280
1291
  """
1281
1292
  Remove interaction data after specimens are presumed dead based on movement data.
@@ -1297,6 +1308,11 @@ class behavpy_core(pd.DataFrame):
1297
1308
  to consider specimen dead (0-1). Defaults to 0.01 (1%).
1298
1309
  resolution (int, optional): Number of segments to divide each time window into.
1299
1310
  Controls overlap between windows. Defaults to 24.
1311
+ min_coverage (float, optional): Fraction of expected samples a window must hold
1312
+ before it is judged, expected samples being derived from each specimen's own
1313
+ median sampling interval. Use it when recordings contain gaps, where a short
1314
+ stretch of data at a machine stop can otherwise read as death. None judges
1315
+ every window, matching previous versions. Defaults to None.
1300
1316
 
1301
1317
  Returns:
1302
1318
  Tuple[behavpy_core, behavpy_core]: Tuple containing:
@@ -1340,6 +1356,7 @@ class behavpy_core(pd.DataFrame):
1340
1356
  prop_immobile=prop_immobile,
1341
1357
  resolution=resolution,
1342
1358
  time_dict=time_dict,
1359
+ min_coverage=min_coverage,
1343
1360
  )
1344
1361
  )
1345
1362
 
@@ -2966,3 +2983,218 @@ class behavpy_core(pd.DataFrame):
2966
2983
  long_palette=self.attrs["lg_pal"],
2967
2984
  check=True,
2968
2985
  )
2986
+
2987
+ # ------------------------------------------------------------------
2988
+ # Kaplan-Meier survival analysis
2989
+ # ------------------------------------------------------------------
2990
+
2991
+ def _subject_key(self, subject_cols: Optional[List[str]]) -> pd.Series:
2992
+ """
2993
+ Map every specimen id to the subject it belongs to.
2994
+
2995
+ Args:
2996
+ subject_cols (Optional[List[str]]): Metadata columns that together
2997
+ identify one animal, typically ['machine_name', 'region_id'] when
2998
+ the same animal was recorded in several sessions. None gives every
2999
+ specimen id its own subject.
3000
+
3001
+ Returns:
3002
+ pd.Series: Subject key per specimen id, indexed by specimen id.
3003
+
3004
+ Raises:
3005
+ KeyError: If any column in subject_cols is missing from the metadata.
3006
+ """
3007
+ if subject_cols is None:
3008
+ return pd.Series(self.meta.index, index=self.meta.index)
3009
+
3010
+ missing = [c for c in subject_cols if c not in self.meta.columns]
3011
+ if missing:
3012
+ raise KeyError(f"Columns {missing} are not metadata columns")
3013
+
3014
+ return self.meta[subject_cols].astype(str).agg("|".join, axis=1)
3015
+
3016
+ def survival_table(
3017
+ self,
3018
+ t_column: str = "t",
3019
+ mov_column: str = "moving",
3020
+ time_window: int = 24,
3021
+ prop_immobile: float = 0.01,
3022
+ resolution: int = 24,
3023
+ subject_cols: Optional[List[str]] = None,
3024
+ restart_gap: float = 1.0,
3025
+ min_coverage: Optional[float] = None,
3026
+ zero_run_hours: Optional[float] = None,
3027
+ second_mov_column: Optional[str] = None,
3028
+ ) -> pd.DataFrame:
3029
+ """
3030
+ Build a per-subject time-to-death table for Kaplan-Meier analysis.
3031
+
3032
+ Death is detected with the same sliding window of immobility that
3033
+ curate_dead_animals() uses, so both agree on when an animal died.
3034
+ Subjects still moving when their recording ends are censored rather
3035
+ than counted as deaths, which is what separates this from
3036
+ survival_plot().
3037
+
3038
+ Args:
3039
+ t_column (str, optional): Column containing timestamps in seconds.
3040
+ Defaults to 't'.
3041
+ mov_column (str, optional): Column containing movement data.
3042
+ Defaults to 'moving'.
3043
+ time_window (int, optional): Size of the immobility window in hours.
3044
+ Defaults to 24.
3045
+ prop_immobile (float, optional): Mean movement at or below which the
3046
+ animal is considered dead. Defaults to 0.01.
3047
+ resolution (int, optional): Number of window starts per window length.
3048
+ Defaults to 24.
3049
+ subject_cols (List[str], optional): Metadata columns identifying one
3050
+ animal across several recordings, e.g. ['machine_name', 'region_id'].
3051
+ Sessions merged this way must not overlap in time. None treats every
3052
+ specimen id as its own animal. Defaults to None.
3053
+ restart_gap (float, optional): Gap in hours above which a jump in the
3054
+ time axis is read as a recording restart rather than missing data.
3055
+ Defaults to 1.0.
3056
+ min_coverage (float, optional): Fraction of expected samples a window
3057
+ must hold before it is judged. Defaults to None.
3058
+ zero_run_hours (float, optional): If set, a contiguous run of zero
3059
+ movement lasting this many hours also counts as death.
3060
+ Defaults to None.
3061
+ second_mov_column (str, optional): Second movement column; death in
3062
+ either column counts. Defaults to None.
3063
+
3064
+ Returns:
3065
+ pd.DataFrame: One row per subject, indexed by subject key, with columns
3066
+ 'id' (a specimen id belonging to the subject), 'T' (hours from first
3067
+ sample to death or censoring), 'E' (1 dead, 0 censored), 'start_time',
3068
+ 'end_time' and 'n_segments'.
3069
+
3070
+ Raises:
3071
+ KeyError: If t_column or mov_column are missing from the data.
3072
+ ValueError: If specimens merged into one subject overlap in time.
3073
+
3074
+ Examples:
3075
+ # Time to death per specimen
3076
+ df.survival_table()
3077
+
3078
+ # Merge recording sessions of the same animal
3079
+ df.survival_table(subject_cols=['machine_name', 'region_id'])
3080
+ """
3081
+ for col in [t_column, mov_column] + (
3082
+ [second_mov_column] if second_mov_column else []
3083
+ ):
3084
+ if col not in self.columns:
3085
+ raise KeyError(f'Column "{col}" is not in the data')
3086
+
3087
+ return build_survival_table(
3088
+ pd.DataFrame(self),
3089
+ self._subject_key(subject_cols),
3090
+ t_column=t_column,
3091
+ mov_column=mov_column,
3092
+ time_window=time_window,
3093
+ prop_immobile=prop_immobile,
3094
+ resolution=resolution,
3095
+ restart_gap=restart_gap,
3096
+ min_coverage=min_coverage,
3097
+ zero_run_hours=zero_run_hours,
3098
+ second_mov_column=second_mov_column,
3099
+ )
3100
+
3101
+ def km_death_table(
3102
+ self,
3103
+ meta_cols: Optional[List[str]] = None,
3104
+ time_unit: str = "hours",
3105
+ **kwargs,
3106
+ ) -> pd.DataFrame:
3107
+ """
3108
+ List the subjects detected as dead, with their time of death.
3109
+
3110
+ Args:
3111
+ meta_cols (List[str], optional): Metadata columns to report alongside
3112
+ each death, e.g. ['species', 'machine_name']. Defaults to None.
3113
+ time_unit (str, optional): 'hours' or 'days'. Defaults to 'hours'.
3114
+ **kwargs: Passed to survival_table(), e.g. time_window, prop_immobile,
3115
+ subject_cols.
3116
+
3117
+ Returns:
3118
+ pd.DataFrame: Dead subjects ordered by time of death, with columns
3119
+ 'id', the requested metadata columns, and 'T'.
3120
+
3121
+ Raises:
3122
+ KeyError: If a requested metadata column does not exist.
3123
+ ValueError: If time_unit is not 'hours' or 'days'.
3124
+
3125
+ Examples:
3126
+ df.km_death_table(meta_cols=['species'], time_unit='days')
3127
+ """
3128
+ if time_unit not in ("hours", "days"):
3129
+ raise ValueError("time_unit must be 'hours' or 'days'")
3130
+
3131
+ table = self.survival_table(**kwargs)
3132
+ dead = table[table["E"] == 1].sort_values("T")
3133
+ out = dead[["id"]].copy()
3134
+
3135
+ for col in meta_cols or []:
3136
+ if col not in self.meta.columns:
3137
+ raise KeyError(f'Column "{col}" is not a metadata column')
3138
+ out[col] = self.meta[col].reindex(dead["id"]).to_numpy()
3139
+
3140
+ out["T"] = dead["T"] / (24.0 if time_unit == "days" else 1.0)
3141
+ return out.reset_index(drop=True)
3142
+
3143
+ def _km_curves(
3144
+ self,
3145
+ facet_col: Optional[str] = None,
3146
+ facet_arg: Optional[List] = None,
3147
+ facet_labels: Optional[List] = None,
3148
+ **kwargs,
3149
+ ) -> Tuple[List[Tuple[str, pd.DataFrame, int]], pd.DataFrame]:
3150
+ """
3151
+ Compute one Kaplan-Meier curve per facet group.
3152
+
3153
+ Shared by the plotly and seaborn km_survival_plot() implementations.
3154
+
3155
+ Args:
3156
+ facet_col (str, optional): Metadata column to group by. Defaults to None.
3157
+ facet_arg (list, optional): Values of facet_col to include. Defaults to None.
3158
+ facet_labels (list, optional): Display labels for each group. Defaults to None.
3159
+ **kwargs: Passed to survival_table().
3160
+
3161
+ Returns:
3162
+ Tuple[List[Tuple[str, pd.DataFrame, int]], pd.DataFrame]: A list of
3163
+ (label, Kaplan-Meier estimate, number of subjects) per group, and
3164
+ the underlying survival table.
3165
+ """
3166
+ facet_arg, facet_labels = self._check_lists(facet_col, facet_arg, facet_labels)
3167
+ table = self.survival_table(**kwargs)
3168
+
3169
+ if facet_col is not None:
3170
+ table = table.assign(
3171
+ **{facet_col: self.meta[facet_col].reindex(table["id"]).to_numpy()}
3172
+ )
3173
+
3174
+ curves = []
3175
+ for arg, label in zip(facet_arg, facet_labels):
3176
+ group = table if facet_col is None else table[table[facet_col] == arg]
3177
+ if len(group) == 0:
3178
+ print(f"Group '{label}' has no subjects and cannot be plotted")
3179
+ continue
3180
+ curves.append((label, kaplan_meier(group["T"], group["E"]), len(group)))
3181
+
3182
+ return curves, table
3183
+
3184
+ @staticmethod
3185
+ def _km_steps(km_df: pd.DataFrame, time_div: float) -> Tuple[np.ndarray, ...]:
3186
+ """
3187
+ Turn a Kaplan-Meier estimate into step-plot coordinates starting at (0, 1).
3188
+
3189
+ Args:
3190
+ km_df (pd.DataFrame): Estimate as returned by kaplan_meier().
3191
+ time_div (float): Divisor applied to time, 24 to plot days.
3192
+
3193
+ Returns:
3194
+ Tuple[np.ndarray, ...]: Time, survival, lower and upper confidence arrays.
3195
+ """
3196
+ time = np.concatenate([[0.0], km_df["time"].to_numpy()]) / time_div
3197
+ survival = np.concatenate([[1.0], km_df["survival"].to_numpy()])
3198
+ lower = np.concatenate([[1.0], km_df["ci_lower"].to_numpy()])
3199
+ upper = np.concatenate([[1.0], km_df["ci_upper"].to_numpy()])
3200
+ return time, survival, lower, upper