educelab-hercdb 0.1.0.dev2__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. educelab_hercdb-0.2.2/PKG-INFO +202 -0
  2. educelab_hercdb-0.2.2/README.md +177 -0
  3. {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/pyproject.toml +27 -8
  4. educelab_hercdb-0.2.2/src/educelab/hercdb/__init__.py +16 -0
  5. educelab_hercdb-0.2.2/src/educelab/hercdb/api.py +14 -0
  6. educelab_hercdb-0.2.2/src/educelab/hercdb/cli/__init__.py +1 -0
  7. educelab_hercdb-0.2.2/src/educelab/hercdb/cli/cleanup_pipelines.py +101 -0
  8. educelab_hercdb-0.2.2/src/educelab/hercdb/cli/sample_uuid_check.py +354 -0
  9. educelab_hercdb-0.2.2/src/educelab/hercdb/cli/scan_completeness.py +335 -0
  10. educelab_hercdb-0.2.2/src/educelab/hercdb/client/__init__.py +3 -0
  11. educelab_hercdb-0.2.2/src/educelab/hercdb/client/herc_client.py +283 -0
  12. {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/src/educelab/hercdb/config.py +21 -4
  13. educelab_hercdb-0.2.2/src/educelab/hercdb/db/__init__.py +13 -0
  14. educelab_hercdb-0.2.2/src/educelab/hercdb/db/connection.py +1425 -0
  15. educelab_hercdb-0.2.2/src/educelab/hercdb/loader/__init__.py +7 -0
  16. educelab_hercdb-0.2.2/src/educelab/hercdb/loader/graph_loader.py +931 -0
  17. educelab_hercdb-0.2.2/src/educelab/hercdb/loader/metadata_loader.py +369 -0
  18. educelab_hercdb-0.2.2/src/educelab/hercdb/loader/scan_loader.py +130 -0
  19. educelab_hercdb-0.2.2/src/educelab/hercdb/rest/__init__.py +1 -0
  20. educelab_hercdb-0.2.2/src/educelab/hercdb/rest/server.py +435 -0
  21. educelab_hercdb-0.2.2/src/educelab_hercdb.egg-info/PKG-INFO +202 -0
  22. educelab_hercdb-0.2.2/src/educelab_hercdb.egg-info/SOURCES.txt +25 -0
  23. educelab_hercdb-0.2.2/src/educelab_hercdb.egg-info/entry_points.txt +4 -0
  24. educelab_hercdb-0.2.2/src/educelab_hercdb.egg-info/requires.txt +11 -0
  25. educelab_hercdb-0.1.0.dev2/PKG-INFO +0 -22
  26. educelab_hercdb-0.1.0.dev2/README.md +0 -3
  27. educelab_hercdb-0.1.0.dev2/src/educelab/hercdb/__init__.py +0 -3
  28. educelab_hercdb-0.1.0.dev2/src/educelab/hercdb/api.py +0 -186
  29. educelab_hercdb-0.1.0.dev2/src/educelab_hercdb.egg-info/PKG-INFO +0 -22
  30. educelab_hercdb-0.1.0.dev2/src/educelab_hercdb.egg-info/SOURCES.txt +0 -11
  31. educelab_hercdb-0.1.0.dev2/src/educelab_hercdb.egg-info/entry_points.txt +0 -2
  32. educelab_hercdb-0.1.0.dev2/src/educelab_hercdb.egg-info/requires.txt +0 -4
  33. {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/setup.cfg +0 -0
  34. {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/src/educelab_hercdb.egg-info/dependency_links.txt +0 -0
  35. {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/src/educelab_hercdb.egg-info/top_level.txt +0 -0
@@ -0,0 +1,202 @@
1
+ Metadata-Version: 2.4
2
+ Name: educelab-hercdb
3
+ Version: 0.2.2
4
+ Summary: Graph database API for Herculaneum data
5
+ Author-email: Mami Hayashida <mami.hayashida@uky.edu>, Seth Parker <c.seth.parker@uky.edu>
6
+ Project-URL: Repository, https://github.com/educelab/educelab-hercdb
7
+ Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
8
+ Classifier: Operating System :: OS Independent
9
+ Classifier: Programming Language :: Python :: 3
10
+ Classifier: Programming Language :: Python :: 3.10
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Requires-Python: >=3.10
14
+ Description-Content-Type: text/markdown
15
+ Requires-Dist: requests>=2.32.5
16
+ Provides-Extra: server
17
+ Requires-Dist: fastapi>=0.124.4; extra == "server"
18
+ Requires-Dist: neo4j>=5.20; extra == "server"
19
+ Requires-Dist: numpy>=2.0; extra == "server"
20
+ Requires-Dist: pandas>=2.2; extra == "server"
21
+ Requires-Dist: prompt-toolkit; extra == "server"
22
+ Requires-Dist: pydantic>=2.0; extra == "server"
23
+ Requires-Dist: rapidfuzz>=3.0; extra == "server"
24
+ Requires-Dist: uvicorn>=0.40.0; extra == "server"
25
+
26
+ # EduceLab Herculaneum Graph Database API
27
+
28
+ This API is considered a work in progress and can change at any moment.
29
+
30
+ ## Architecture
31
+
32
+ ![hercdb architecture](docs/hercdb_architecture.svg)
33
+
34
+ ## Setup Quick Start
35
+
36
+ For a visual overview of the full setup process (beyond just this repo), see the quick start guide:
37
+
38
+ ![setup quick start](docs/hercdb_setup_quickstart.svg)
39
+
40
+ For step-by-step server setup instructions, see [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md).
41
+
42
+ ## Installation
43
+
44
+ This package supports two install modes:
45
+
46
+ ### Client only (lightweight)
47
+
48
+ For remote machines that only need to call the REST API:
49
+
50
+ ```shell
51
+ pip install educelab-hercdb
52
+ ```
53
+
54
+ This installs only the `requests` library. See [src/educelab/hercdb/client/README.md](src/educelab/hercdb/client/README.md) for client usage.
55
+
56
+ ### Server (full)
57
+
58
+ For running the REST API server, loading data, or querying Neo4j directly:
59
+
60
+ ```shell
61
+ pip install educelab-hercdb[server]
62
+ ```
63
+
64
+ This adds `fastapi`, `neo4j`, `numpy`, `pandas`, `prompt-toolkit`, and `uvicorn`.
65
+
66
+ ## Development Setup
67
+
68
+ ```shell
69
+ # Install base dependencies
70
+ uv sync
71
+
72
+ # Or with server extras (fastapi, neo4j, etc.)
73
+ uv sync --extra server
74
+
75
+ # Run commands in the environment
76
+ uv run python -c "from educelab.hercdb.client import HercClient"
77
+
78
+ # Or activate the venv directly
79
+ source .venv/bin/activate
80
+ ```
81
+
82
+ ## Client Library
83
+
84
+ ```python
85
+ from educelab.hercdb.client import HercClient
86
+
87
+ client = HercClient(host="api.example.com", token="my-token")
88
+ pherc = client.get_artifact_by_name("211")
89
+ ```
90
+
91
+ See [src/educelab/hercdb/client/README.md](src/educelab/hercdb/client/README.md) for the full API reference.
92
+
93
+ ## Direct Database Connection
94
+
95
+ For environments with the `server` extra installed, you can connect to Neo4j directly:
96
+
97
+ ```python
98
+ from educelab import hercdb
99
+
100
+ uri = "neo4j://localhost:7687"
101
+ user = "foo"
102
+ password = "bar"
103
+ db = hercdb.connect(uri, user, password)
104
+ if db.verify_connection():
105
+ print("Connected!")
106
+ ```
107
+
108
+ If credentials are not passed directly, the package reads them from `~/.educedb` or environment variables. See [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md) for configuration details.
109
+
110
+ ## REST API
111
+
112
+ A FastAPI-based REST API is available for querying the database over HTTP. All endpoints require Bearer token authentication.
113
+
114
+ ```shell
115
+ uv run uvicorn educelab.hercdb.rest.server:app --reload
116
+ ```
117
+
118
+ Interactive API docs are available at `/docs` (Swagger UI) and `/redoc` (ReDoc) once the server is running.
119
+
120
+ See [src/educelab/hercdb/rest/README.md](src/educelab/hercdb/rest/README.md) for endpoint documentation and [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md) for production deployment.
121
+
122
+ ## Loading Data
123
+
124
+ Data loading is done in two steps using the loader scripts. Both read CSV files from `input_data/`.
125
+
126
+ ### 1. Load metadata and UUIDs
127
+
128
+ ```shell
129
+ uv run python src/educelab/hercdb/loader/metadata_loader.py
130
+ ```
131
+
132
+ Reads (defaults):
133
+ - `input_data/metadata_file.csv` - Pre-processed metadata file. (PHerc, Cornice, Pezzo, Disegni nodes and properties.)
134
+ - `input_data/uuid_file.csv` - Pre-processed uuid file. (all EduceLabID added)
135
+
136
+ Optional arguments:
137
+ ```shell
138
+ uv run python src/educelab/hercdb/loader/metadata_loader.py \
139
+ --metadata path/to/metadata.csv \
140
+ --uuid path/to/uuid.csv
141
+ ```
142
+
143
+ ### 2. Load scan data
144
+
145
+ ```shell
146
+ uv run python src/educelab/hercdb/loader/scan_loader.py
147
+ ```
148
+
149
+ Reads (defaults):
150
+ - `input_data/negatives.csv` - FlatbedScanDataset nodes
151
+ - `input_data/pgs_datasets_20260601(in).csv` - PGSRaw nodes
152
+ - `input_data/spectral_datasets_20260601_reconciled.csv` - SpectralRaw nodes
153
+
154
+ By default (`--replace`) it deletes all existing PGSRaw/SpectralRaw nodes and reloads from scratch (FlatbedScanDataset is untouched); pass `--no-replace` to merge into existing data instead. Nodes are keyed on the scan `uuid`, so re-running is idempotent.
155
+
156
+ Optional arguments:
157
+ ```shell
158
+ uv run python src/educelab/hercdb/loader/scan_loader.py \
159
+ --negatives path/to/negatives.csv \
160
+ --photogrammetry path/to/pgs.csv \
161
+ --spectral path/to/spectral.csv \
162
+ --no-replace
163
+ ```
164
+
165
+ **Note:** Run metadata_loader first since scan data links to EduceLabID nodes.
166
+
167
+ ## Reporting Tools
168
+
169
+ ### Scan completeness report
170
+
171
+ `el-hercdb-scan-report` walks every PHerc and its hierarchy and writes two CSVs that flag artifacts needing first-time scans or re-scans.
172
+
173
+ ```shell
174
+ # After `uv sync --extra server`, the entry point is on PATH:
175
+ uv run el-hercdb-scan-report --out-dir ./tmp
176
+
177
+ # Equivalent fallback without re-syncing:
178
+ uv run python -m educelab.hercdb.cli.scan_completeness --out-dir ./tmp
179
+ ```
180
+
181
+ Outputs (default names):
182
+
183
+ - `scan_completeness_full.csv` — one row per (artifact, UUID) pair, plus blank-UUID sentinel rows for artifacts that have no EduceLabID assigned. Columns: PHerc, Cornice, Pezzo, UUID, PGS Status, PGS Latest Complete Date, PGS Path, Spectral Status, Spectral Latest Complete Date, Spectral Path, Institution.
184
+ - `scan_completeness_issues.csv` — same shape, filtered to rows where PGS or Spectral is missing/incomplete, or the artifact has no UUID at all.
185
+
186
+ Status values are `complete` / `incomplete` / `missing` when the artifact has a UUID, and **blank** when it does not (so "UUID assigned but no scan" stays distinguishable from "no UUID even assigned"). Files are written with `utf-8-sig` so Excel opens them with correct character encoding.
187
+
188
+ The report walks `REPLACES` relationships between EduceLabIDs, so pre-replacement scans on a retired predecessor UUID still count toward the artifact's coverage.
189
+
190
+ ## Temporary Scripts and Notes
191
+
192
+ The `tmp/` directory contains temporary scripts, notes, and other informal resources shared among the team. Contents are version controlled but considered ephemeral — they may be rewritten or deleted at any time and should not be relied upon as stable code.
193
+
194
+ ### Delete all data
195
+
196
+ To clear the database before reloading:
197
+
198
+ ```python
199
+ from educelab.hercdb.loader import PhercGraphDatabaseLoader
200
+ loader = PhercGraphDatabaseLoader()
201
+ loader._delete_all_nodes()
202
+ ```
@@ -0,0 +1,177 @@
1
+ # EduceLab Herculaneum Graph Database API
2
+
3
+ This API is considered a work in progress and can change at any moment.
4
+
5
+ ## Architecture
6
+
7
+ ![hercdb architecture](docs/hercdb_architecture.svg)
8
+
9
+ ## Setup Quick Start
10
+
11
+ For a visual overview of the full setup process (beyond just this repo), see the quick start guide:
12
+
13
+ ![setup quick start](docs/hercdb_setup_quickstart.svg)
14
+
15
+ For step-by-step server setup instructions, see [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md).
16
+
17
+ ## Installation
18
+
19
+ This package supports two install modes:
20
+
21
+ ### Client only (lightweight)
22
+
23
+ For remote machines that only need to call the REST API:
24
+
25
+ ```shell
26
+ pip install educelab-hercdb
27
+ ```
28
+
29
+ This installs only the `requests` library. See [src/educelab/hercdb/client/README.md](src/educelab/hercdb/client/README.md) for client usage.
30
+
31
+ ### Server (full)
32
+
33
+ For running the REST API server, loading data, or querying Neo4j directly:
34
+
35
+ ```shell
36
+ pip install educelab-hercdb[server]
37
+ ```
38
+
39
+ This adds `fastapi`, `neo4j`, `numpy`, `pandas`, `prompt-toolkit`, and `uvicorn`.
40
+
41
+ ## Development Setup
42
+
43
+ ```shell
44
+ # Install base dependencies
45
+ uv sync
46
+
47
+ # Or with server extras (fastapi, neo4j, etc.)
48
+ uv sync --extra server
49
+
50
+ # Run commands in the environment
51
+ uv run python -c "from educelab.hercdb.client import HercClient"
52
+
53
+ # Or activate the venv directly
54
+ source .venv/bin/activate
55
+ ```
56
+
57
+ ## Client Library
58
+
59
+ ```python
60
+ from educelab.hercdb.client import HercClient
61
+
62
+ client = HercClient(host="api.example.com", token="my-token")
63
+ pherc = client.get_artifact_by_name("211")
64
+ ```
65
+
66
+ See [src/educelab/hercdb/client/README.md](src/educelab/hercdb/client/README.md) for the full API reference.
67
+
68
+ ## Direct Database Connection
69
+
70
+ For environments with the `server` extra installed, you can connect to Neo4j directly:
71
+
72
+ ```python
73
+ from educelab import hercdb
74
+
75
+ uri = "neo4j://localhost:7687"
76
+ user = "foo"
77
+ password = "bar"
78
+ db = hercdb.connect(uri, user, password)
79
+ if db.verify_connection():
80
+ print("Connected!")
81
+ ```
82
+
83
+ If credentials are not passed directly, the package reads them from `~/.educedb` or environment variables. See [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md) for configuration details.
84
+
85
+ ## REST API
86
+
87
+ A FastAPI-based REST API is available for querying the database over HTTP. All endpoints require Bearer token authentication.
88
+
89
+ ```shell
90
+ uv run uvicorn educelab.hercdb.rest.server:app --reload
91
+ ```
92
+
93
+ Interactive API docs are available at `/docs` (Swagger UI) and `/redoc` (ReDoc) once the server is running.
94
+
95
+ See [src/educelab/hercdb/rest/README.md](src/educelab/hercdb/rest/README.md) for endpoint documentation and [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md) for production deployment.
96
+
97
+ ## Loading Data
98
+
99
+ Data loading is done in two steps using the loader scripts. Both read CSV files from `input_data/`.
100
+
101
+ ### 1. Load metadata and UUIDs
102
+
103
+ ```shell
104
+ uv run python src/educelab/hercdb/loader/metadata_loader.py
105
+ ```
106
+
107
+ Reads (defaults):
108
+ - `input_data/metadata_file.csv` - Pre-processed metadata file. (PHerc, Cornice, Pezzo, Disegni nodes and properties.)
109
+ - `input_data/uuid_file.csv` - Pre-processed uuid file. (all EduceLabID added)
110
+
111
+ Optional arguments:
112
+ ```shell
113
+ uv run python src/educelab/hercdb/loader/metadata_loader.py \
114
+ --metadata path/to/metadata.csv \
115
+ --uuid path/to/uuid.csv
116
+ ```
117
+
118
+ ### 2. Load scan data
119
+
120
+ ```shell
121
+ uv run python src/educelab/hercdb/loader/scan_loader.py
122
+ ```
123
+
124
+ Reads (defaults):
125
+ - `input_data/negatives.csv` - FlatbedScanDataset nodes
126
+ - `input_data/pgs_datasets_20260601(in).csv` - PGSRaw nodes
127
+ - `input_data/spectral_datasets_20260601_reconciled.csv` - SpectralRaw nodes
128
+
129
+ By default (`--replace`) it deletes all existing PGSRaw/SpectralRaw nodes and reloads from scratch (FlatbedScanDataset is untouched); pass `--no-replace` to merge into existing data instead. Nodes are keyed on the scan `uuid`, so re-running is idempotent.
130
+
131
+ Optional arguments:
132
+ ```shell
133
+ uv run python src/educelab/hercdb/loader/scan_loader.py \
134
+ --negatives path/to/negatives.csv \
135
+ --photogrammetry path/to/pgs.csv \
136
+ --spectral path/to/spectral.csv \
137
+ --no-replace
138
+ ```
139
+
140
+ **Note:** Run metadata_loader first since scan data links to EduceLabID nodes.
141
+
142
+ ## Reporting Tools
143
+
144
+ ### Scan completeness report
145
+
146
+ `el-hercdb-scan-report` walks every PHerc and its hierarchy and writes two CSVs that flag artifacts needing first-time scans or re-scans.
147
+
148
+ ```shell
149
+ # After `uv sync --extra server`, the entry point is on PATH:
150
+ uv run el-hercdb-scan-report --out-dir ./tmp
151
+
152
+ # Equivalent fallback without re-syncing:
153
+ uv run python -m educelab.hercdb.cli.scan_completeness --out-dir ./tmp
154
+ ```
155
+
156
+ Outputs (default names):
157
+
158
+ - `scan_completeness_full.csv` — one row per (artifact, UUID) pair, plus blank-UUID sentinel rows for artifacts that have no EduceLabID assigned. Columns: PHerc, Cornice, Pezzo, UUID, PGS Status, PGS Latest Complete Date, PGS Path, Spectral Status, Spectral Latest Complete Date, Spectral Path, Institution.
159
+ - `scan_completeness_issues.csv` — same shape, filtered to rows where PGS or Spectral is missing/incomplete, or the artifact has no UUID at all.
160
+
161
+ Status values are `complete` / `incomplete` / `missing` when the artifact has a UUID, and **blank** when it does not (so "UUID assigned but no scan" stays distinguishable from "no UUID even assigned"). Files are written with `utf-8-sig` so Excel opens them with correct character encoding.
162
+
163
+ The report walks `REPLACES` relationships between EduceLabIDs, so pre-replacement scans on a retired predecessor UUID still count toward the artifact's coverage.
164
+
165
+ ## Temporary Scripts and Notes
166
+
167
+ The `tmp/` directory contains temporary scripts, notes, and other informal resources shared among the team. Contents are version controlled but considered ephemeral — they may be rewritten or deleted at any time and should not be relied upon as stable code.
168
+
169
+ ### Delete all data
170
+
171
+ To clear the database before reloading:
172
+
173
+ ```python
174
+ from educelab.hercdb.loader import PhercGraphDatabaseLoader
175
+ loader = PhercGraphDatabaseLoader()
176
+ loader._delete_all_nodes()
177
+ ```
@@ -4,16 +4,13 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [tool.setuptools.packages.find]
6
6
  where = ["src/"]
7
- include = ["educelab.hercdb"]
7
+ include = ["educelab.hercdb*"]
8
8
 
9
9
  [project]
10
10
  name = "educelab-hercdb"
11
- version = "0.1.0.dev2"
11
+ version = "0.2.2"
12
12
  dependencies = [
13
- "neo4j>=5.20",
14
- "numpy>=2.0",
15
- "pandas>=2.2",
16
- "prompt-toolkit"
13
+ "requests>=2.32.5",
17
14
  ]
18
15
  requires-python = ">= 3.10"
19
16
  authors = [
@@ -32,8 +29,30 @@ classifiers = [
32
29
  "Programming Language :: Python :: 3.12",
33
30
  ]
34
31
 
32
+ [project.optional-dependencies]
33
+ server = [
34
+ "fastapi>=0.124.4",
35
+ "neo4j>=5.20",
36
+ "numpy>=2.0",
37
+ "pandas>=2.2",
38
+ "prompt-toolkit",
39
+ "pydantic>=2.0",
40
+ "rapidfuzz>=3.0",
41
+ "uvicorn>=0.40.0",
42
+ ]
43
+
35
44
  [project.urls]
36
- Repository = "https://gitlab.com/educelab/herculaneum-graph-db"
45
+ Repository = "https://github.com/educelab/educelab-hercdb"
37
46
 
38
47
  [project.scripts]
39
- el-hercdb-search = "educelab.hercdb.apps.search:main"
48
+ el-hercdb-scan-report = "educelab.hercdb.cli.scan_completeness:main"
49
+ el-hercdb-sample-uuid-check = "educelab.hercdb.cli.sample_uuid_check:main"
50
+ el-hercdb-pipeline-cleanup = "educelab.hercdb.cli.cleanup_pipelines:main"
51
+
52
+ [dependency-groups]
53
+ dev = [
54
+ "ipykernel>=7.1.0",
55
+ "matplotlib>=3.10.9",
56
+ "networkx>=3.4.2",
57
+ "pyyaml>=6.0.3",
58
+ ]
@@ -0,0 +1,16 @@
1
+ try:
2
+ from educelab.hercdb import config
3
+ from educelab.hercdb.db import (
4
+ connect,
5
+ GraphDBConnection,
6
+ DatasetType,
7
+ )
8
+
9
+ __all__ = [
10
+ "config",
11
+ "connect",
12
+ "GraphDBConnection",
13
+ "DatasetType",
14
+ ]
15
+ except ImportError:
16
+ pass
@@ -0,0 +1,14 @@
1
+ """Backward compatibility shim - use educelab.hercdb.db instead."""
2
+
3
+ # Re-export everything from db module for backward compatibility
4
+ from educelab.hercdb.db import (
5
+ GraphDBConnection,
6
+ connect,
7
+ DatasetType,
8
+ )
9
+
10
+ __all__ = [
11
+ "GraphDBConnection",
12
+ "connect",
13
+ "DatasetType",
14
+ ]
@@ -0,0 +1 @@
1
+ """Command-line interface tools for hercdb."""
@@ -0,0 +1,101 @@
1
+ """Delete pipeline records from Neo4j by pipeline id (DB-admin tool).
2
+
3
+ Companion to the pipeline-recording write path in the acquisition-workflow
4
+ ``submit_uber_pipeline.py``: while that path is being tested, every (dry-run or
5
+ real) submission writes a Pipeline node keyed by its ``uber_job_id``. This tool
6
+ lets a database administrator remove those test records again, straight against
7
+ Neo4j (no REST server needed).
8
+
9
+ Deleting a pipeline removes the Pipeline node, its Process nodes, and the output
10
+ dataset nodes (PGSProcessed/SpectralProcessed/Registered/WebProcessed) those
11
+ processes produced. Input/raw datasets (PGSRaw/SpectralRaw/...) are left
12
+ untouched.
13
+
14
+ Wired up as the ``el-hercdb-pipeline-cleanup`` shell command (see
15
+ ``[project.scripts]`` in pyproject.toml). Usage:
16
+
17
+ el-hercdb-pipeline-cleanup uber-1a2b3c4d [uber-...] [-y]
18
+ uv run el-hercdb-pipeline-cleanup --list # show what exists first
19
+ """
20
+ import argparse
21
+ import sys
22
+
23
+ from educelab import hercdb
24
+ from educelab.hercdb import config
25
+
26
+
27
+ def main():
28
+ parser = argparse.ArgumentParser(
29
+ description=(
30
+ "Delete pipeline records (Pipeline + Process + output datasets) "
31
+ "from Neo4j by pipeline id. Input/raw datasets are left untouched."
32
+ )
33
+ )
34
+ parser.add_argument(
35
+ "pipeline_ids", nargs="*",
36
+ help="Pipeline id(s) to delete (the uber_job_id printed by "
37
+ "submit_uber_pipeline.py).",
38
+ )
39
+ parser.add_argument(
40
+ "--list", action="store_true",
41
+ help="List all pipelines currently in the database and exit (no "
42
+ "deletion). Useful for finding ids to clean up.",
43
+ )
44
+ parser.add_argument(
45
+ "--yes", "-y", action="store_true",
46
+ help="Skip the confirmation prompt.",
47
+ )
48
+ args = parser.parse_args()
49
+
50
+ if not args.list and not args.pipeline_ids:
51
+ parser.error("provide one or more pipeline ids, or use --list.")
52
+
53
+ config.request_required()
54
+ db = hercdb.connect()
55
+ if not db.verify_connection():
56
+ print("Failed to connect to Neo4j.", file=sys.stderr)
57
+ sys.exit(1)
58
+
59
+ if args.list:
60
+ pipelines = db.get_all_pipeline_summaries()
61
+ if not pipelines:
62
+ print("No pipelines in the database.")
63
+ return
64
+ print(f"{len(pipelines)} pipeline(s):")
65
+ for p in pipelines:
66
+ pid = p.get("pipeline_id", "?")
67
+ uuid = p.get("artifact_uuid", "")
68
+ print(f" - {pid}" + (f" (artifact {uuid})" if uuid else ""))
69
+ return
70
+
71
+ if not args.yes:
72
+ print("About to delete the following pipeline(s):")
73
+ for pid in args.pipeline_ids:
74
+ print(f" - {pid}")
75
+ confirm = input("Proceed? [y/N]: ").strip().lower()
76
+ if confirm not in ("y", "yes"):
77
+ print("Aborted.")
78
+ return
79
+
80
+ failed = False
81
+ for pid in args.pipeline_ids:
82
+ result = db.delete_pipeline(pid)
83
+ if result is None:
84
+ print(f"{pid}: not found")
85
+ failed = True
86
+ else:
87
+ print(
88
+ f"{pid}: deleted "
89
+ f"({result.get('processes_deleted', 0)} process(es), "
90
+ f"{result.get('output_datasets_deleted', 0)} output dataset(s))"
91
+ )
92
+
93
+ if failed:
94
+ sys.exit(1)
95
+
96
+
97
+ if __name__ == "__main__":
98
+ try:
99
+ main()
100
+ except KeyboardInterrupt:
101
+ sys.exit(0)