educelab-hercdb 0.1.0.dev2__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- educelab_hercdb-0.2.2/PKG-INFO +202 -0
- educelab_hercdb-0.2.2/README.md +177 -0
- {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/pyproject.toml +27 -8
- educelab_hercdb-0.2.2/src/educelab/hercdb/__init__.py +16 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/api.py +14 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/cli/__init__.py +1 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/cli/cleanup_pipelines.py +101 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/cli/sample_uuid_check.py +354 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/cli/scan_completeness.py +335 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/client/__init__.py +3 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/client/herc_client.py +283 -0
- {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/src/educelab/hercdb/config.py +21 -4
- educelab_hercdb-0.2.2/src/educelab/hercdb/db/__init__.py +13 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/db/connection.py +1425 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/loader/__init__.py +7 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/loader/graph_loader.py +931 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/loader/metadata_loader.py +369 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/loader/scan_loader.py +130 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/rest/__init__.py +1 -0
- educelab_hercdb-0.2.2/src/educelab/hercdb/rest/server.py +435 -0
- educelab_hercdb-0.2.2/src/educelab_hercdb.egg-info/PKG-INFO +202 -0
- educelab_hercdb-0.2.2/src/educelab_hercdb.egg-info/SOURCES.txt +25 -0
- educelab_hercdb-0.2.2/src/educelab_hercdb.egg-info/entry_points.txt +4 -0
- educelab_hercdb-0.2.2/src/educelab_hercdb.egg-info/requires.txt +11 -0
- educelab_hercdb-0.1.0.dev2/PKG-INFO +0 -22
- educelab_hercdb-0.1.0.dev2/README.md +0 -3
- educelab_hercdb-0.1.0.dev2/src/educelab/hercdb/__init__.py +0 -3
- educelab_hercdb-0.1.0.dev2/src/educelab/hercdb/api.py +0 -186
- educelab_hercdb-0.1.0.dev2/src/educelab_hercdb.egg-info/PKG-INFO +0 -22
- educelab_hercdb-0.1.0.dev2/src/educelab_hercdb.egg-info/SOURCES.txt +0 -11
- educelab_hercdb-0.1.0.dev2/src/educelab_hercdb.egg-info/entry_points.txt +0 -2
- educelab_hercdb-0.1.0.dev2/src/educelab_hercdb.egg-info/requires.txt +0 -4
- {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/setup.cfg +0 -0
- {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/src/educelab_hercdb.egg-info/dependency_links.txt +0 -0
- {educelab_hercdb-0.1.0.dev2 → educelab_hercdb-0.2.2}/src/educelab_hercdb.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: educelab-hercdb
|
|
3
|
+
Version: 0.2.2
|
|
4
|
+
Summary: Graph database API for Herculaneum data
|
|
5
|
+
Author-email: Mami Hayashida <mami.hayashida@uky.edu>, Seth Parker <c.seth.parker@uky.edu>
|
|
6
|
+
Project-URL: Repository, https://github.com/educelab/educelab-hercdb
|
|
7
|
+
Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Requires-Python: >=3.10
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: requests>=2.32.5
|
|
16
|
+
Provides-Extra: server
|
|
17
|
+
Requires-Dist: fastapi>=0.124.4; extra == "server"
|
|
18
|
+
Requires-Dist: neo4j>=5.20; extra == "server"
|
|
19
|
+
Requires-Dist: numpy>=2.0; extra == "server"
|
|
20
|
+
Requires-Dist: pandas>=2.2; extra == "server"
|
|
21
|
+
Requires-Dist: prompt-toolkit; extra == "server"
|
|
22
|
+
Requires-Dist: pydantic>=2.0; extra == "server"
|
|
23
|
+
Requires-Dist: rapidfuzz>=3.0; extra == "server"
|
|
24
|
+
Requires-Dist: uvicorn>=0.40.0; extra == "server"
|
|
25
|
+
|
|
26
|
+
# EduceLab Herculaneum Graph Database API
|
|
27
|
+
|
|
28
|
+
This API is considered a work in progress and can change at any moment.
|
|
29
|
+
|
|
30
|
+
## Architecture
|
|
31
|
+
|
|
32
|
+

|
|
33
|
+
|
|
34
|
+
## Setup Quick Start
|
|
35
|
+
|
|
36
|
+
For a visual overview of the full setup process (beyond just this repo), see the quick start guide:
|
|
37
|
+
|
|
38
|
+

|
|
39
|
+
|
|
40
|
+
For step-by-step server setup instructions, see [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md).
|
|
41
|
+
|
|
42
|
+
## Installation
|
|
43
|
+
|
|
44
|
+
This package supports two install modes:
|
|
45
|
+
|
|
46
|
+
### Client only (lightweight)
|
|
47
|
+
|
|
48
|
+
For remote machines that only need to call the REST API:
|
|
49
|
+
|
|
50
|
+
```shell
|
|
51
|
+
pip install educelab-hercdb
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
This installs only the `requests` library. See [src/educelab/hercdb/client/README.md](src/educelab/hercdb/client/README.md) for client usage.
|
|
55
|
+
|
|
56
|
+
### Server (full)
|
|
57
|
+
|
|
58
|
+
For running the REST API server, loading data, or querying Neo4j directly:
|
|
59
|
+
|
|
60
|
+
```shell
|
|
61
|
+
pip install educelab-hercdb[server]
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
This adds `fastapi`, `neo4j`, `numpy`, `pandas`, `prompt-toolkit`, and `uvicorn`.
|
|
65
|
+
|
|
66
|
+
## Development Setup
|
|
67
|
+
|
|
68
|
+
```shell
|
|
69
|
+
# Install base dependencies
|
|
70
|
+
uv sync
|
|
71
|
+
|
|
72
|
+
# Or with server extras (fastapi, neo4j, etc.)
|
|
73
|
+
uv sync --extra server
|
|
74
|
+
|
|
75
|
+
# Run commands in the environment
|
|
76
|
+
uv run python -c "from educelab.hercdb.client import HercClient"
|
|
77
|
+
|
|
78
|
+
# Or activate the venv directly
|
|
79
|
+
source .venv/bin/activate
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Client Library
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from educelab.hercdb.client import HercClient
|
|
86
|
+
|
|
87
|
+
client = HercClient(host="api.example.com", token="my-token")
|
|
88
|
+
pherc = client.get_artifact_by_name("211")
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
See [src/educelab/hercdb/client/README.md](src/educelab/hercdb/client/README.md) for the full API reference.
|
|
92
|
+
|
|
93
|
+
## Direct Database Connection
|
|
94
|
+
|
|
95
|
+
For environments with the `server` extra installed, you can connect to Neo4j directly:
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
from educelab import hercdb
|
|
99
|
+
|
|
100
|
+
uri = "neo4j://localhost:7687"
|
|
101
|
+
user = "foo"
|
|
102
|
+
password = "bar"
|
|
103
|
+
db = hercdb.connect(uri, user, password)
|
|
104
|
+
if db.verify_connection():
|
|
105
|
+
print("Connected!")
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
If credentials are not passed directly, the package reads them from `~/.educedb` or environment variables. See [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md) for configuration details.
|
|
109
|
+
|
|
110
|
+
## REST API
|
|
111
|
+
|
|
112
|
+
A FastAPI-based REST API is available for querying the database over HTTP. All endpoints require Bearer token authentication.
|
|
113
|
+
|
|
114
|
+
```shell
|
|
115
|
+
uv run uvicorn educelab.hercdb.rest.server:app --reload
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Interactive API docs are available at `/docs` (Swagger UI) and `/redoc` (ReDoc) once the server is running.
|
|
119
|
+
|
|
120
|
+
See [src/educelab/hercdb/rest/README.md](src/educelab/hercdb/rest/README.md) for endpoint documentation and [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md) for production deployment.
|
|
121
|
+
|
|
122
|
+
## Loading Data
|
|
123
|
+
|
|
124
|
+
Data loading is done in two steps using the loader scripts. Both read CSV files from `input_data/`.
|
|
125
|
+
|
|
126
|
+
### 1. Load metadata and UUIDs
|
|
127
|
+
|
|
128
|
+
```shell
|
|
129
|
+
uv run python src/educelab/hercdb/loader/metadata_loader.py
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
Reads (defaults):
|
|
133
|
+
- `input_data/metadata_file.csv` - Pre-processed metadata file. (PHerc, Cornice, Pezzo, Disegni nodes and properties.)
|
|
134
|
+
- `input_data/uuid_file.csv` - Pre-processed uuid file. (all EduceLabID added)
|
|
135
|
+
|
|
136
|
+
Optional arguments:
|
|
137
|
+
```shell
|
|
138
|
+
uv run python src/educelab/hercdb/loader/metadata_loader.py \
|
|
139
|
+
--metadata path/to/metadata.csv \
|
|
140
|
+
--uuid path/to/uuid.csv
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
### 2. Load scan data
|
|
144
|
+
|
|
145
|
+
```shell
|
|
146
|
+
uv run python src/educelab/hercdb/loader/scan_loader.py
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
Reads (defaults):
|
|
150
|
+
- `input_data/negatives.csv` - FlatbedScanDataset nodes
|
|
151
|
+
- `input_data/pgs_datasets_20260601(in).csv` - PGSRaw nodes
|
|
152
|
+
- `input_data/spectral_datasets_20260601_reconciled.csv` - SpectralRaw nodes
|
|
153
|
+
|
|
154
|
+
By default (`--replace`) it deletes all existing PGSRaw/SpectralRaw nodes and reloads from scratch (FlatbedScanDataset is untouched); pass `--no-replace` to merge into existing data instead. Nodes are keyed on the scan `uuid`, so re-running is idempotent.
|
|
155
|
+
|
|
156
|
+
Optional arguments:
|
|
157
|
+
```shell
|
|
158
|
+
uv run python src/educelab/hercdb/loader/scan_loader.py \
|
|
159
|
+
--negatives path/to/negatives.csv \
|
|
160
|
+
--photogrammetry path/to/pgs.csv \
|
|
161
|
+
--spectral path/to/spectral.csv \
|
|
162
|
+
--no-replace
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
**Note:** Run metadata_loader first since scan data links to EduceLabID nodes.
|
|
166
|
+
|
|
167
|
+
## Reporting Tools
|
|
168
|
+
|
|
169
|
+
### Scan completeness report
|
|
170
|
+
|
|
171
|
+
`el-hercdb-scan-report` walks every PHerc and its hierarchy and writes two CSVs that flag artifacts needing first-time scans or re-scans.
|
|
172
|
+
|
|
173
|
+
```shell
|
|
174
|
+
# After `uv sync --extra server`, the entry point is on PATH:
|
|
175
|
+
uv run el-hercdb-scan-report --out-dir ./tmp
|
|
176
|
+
|
|
177
|
+
# Equivalent fallback without re-syncing:
|
|
178
|
+
uv run python -m educelab.hercdb.cli.scan_completeness --out-dir ./tmp
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Outputs (default names):
|
|
182
|
+
|
|
183
|
+
- `scan_completeness_full.csv` — one row per (artifact, UUID) pair, plus blank-UUID sentinel rows for artifacts that have no EduceLabID assigned. Columns: PHerc, Cornice, Pezzo, UUID, PGS Status, PGS Latest Complete Date, PGS Path, Spectral Status, Spectral Latest Complete Date, Spectral Path, Institution.
|
|
184
|
+
- `scan_completeness_issues.csv` — same shape, filtered to rows where PGS or Spectral is missing/incomplete, or the artifact has no UUID at all.
|
|
185
|
+
|
|
186
|
+
Status values are `complete` / `incomplete` / `missing` when the artifact has a UUID, and **blank** when it does not (so "UUID assigned but no scan" stays distinguishable from "no UUID even assigned"). Files are written with `utf-8-sig` so Excel opens them with correct character encoding.
|
|
187
|
+
|
|
188
|
+
The report walks `REPLACES` relationships between EduceLabIDs, so pre-replacement scans on a retired predecessor UUID still count toward the artifact's coverage.
|
|
189
|
+
|
|
190
|
+
## Temporary Scripts and Notes
|
|
191
|
+
|
|
192
|
+
The `tmp/` directory contains temporary scripts, notes, and other informal resources shared among the team. Contents are version controlled but considered ephemeral — they may be rewritten or deleted at any time and should not be relied upon as stable code.
|
|
193
|
+
|
|
194
|
+
### Delete all data
|
|
195
|
+
|
|
196
|
+
To clear the database before reloading:
|
|
197
|
+
|
|
198
|
+
```python
|
|
199
|
+
from educelab.hercdb.loader import PhercGraphDatabaseLoader
|
|
200
|
+
loader = PhercGraphDatabaseLoader()
|
|
201
|
+
loader._delete_all_nodes()
|
|
202
|
+
```
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
# EduceLab Herculaneum Graph Database API
|
|
2
|
+
|
|
3
|
+
This API is considered a work in progress and can change at any moment.
|
|
4
|
+
|
|
5
|
+
## Architecture
|
|
6
|
+
|
|
7
|
+

|
|
8
|
+
|
|
9
|
+
## Setup Quick Start
|
|
10
|
+
|
|
11
|
+
For a visual overview of the full setup process (beyond just this repo), see the quick start guide:
|
|
12
|
+
|
|
13
|
+

|
|
14
|
+
|
|
15
|
+
For step-by-step server setup instructions, see [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md).
|
|
16
|
+
|
|
17
|
+
## Installation
|
|
18
|
+
|
|
19
|
+
This package supports two install modes:
|
|
20
|
+
|
|
21
|
+
### Client only (lightweight)
|
|
22
|
+
|
|
23
|
+
For remote machines that only need to call the REST API:
|
|
24
|
+
|
|
25
|
+
```shell
|
|
26
|
+
pip install educelab-hercdb
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
This installs only the `requests` library. See [src/educelab/hercdb/client/README.md](src/educelab/hercdb/client/README.md) for client usage.
|
|
30
|
+
|
|
31
|
+
### Server (full)
|
|
32
|
+
|
|
33
|
+
For running the REST API server, loading data, or querying Neo4j directly:
|
|
34
|
+
|
|
35
|
+
```shell
|
|
36
|
+
pip install educelab-hercdb[server]
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
This adds `fastapi`, `neo4j`, `numpy`, `pandas`, `prompt-toolkit`, and `uvicorn`.
|
|
40
|
+
|
|
41
|
+
## Development Setup
|
|
42
|
+
|
|
43
|
+
```shell
|
|
44
|
+
# Install base dependencies
|
|
45
|
+
uv sync
|
|
46
|
+
|
|
47
|
+
# Or with server extras (fastapi, neo4j, etc.)
|
|
48
|
+
uv sync --extra server
|
|
49
|
+
|
|
50
|
+
# Run commands in the environment
|
|
51
|
+
uv run python -c "from educelab.hercdb.client import HercClient"
|
|
52
|
+
|
|
53
|
+
# Or activate the venv directly
|
|
54
|
+
source .venv/bin/activate
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Client Library
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
from educelab.hercdb.client import HercClient
|
|
61
|
+
|
|
62
|
+
client = HercClient(host="api.example.com", token="my-token")
|
|
63
|
+
pherc = client.get_artifact_by_name("211")
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
See [src/educelab/hercdb/client/README.md](src/educelab/hercdb/client/README.md) for the full API reference.
|
|
67
|
+
|
|
68
|
+
## Direct Database Connection
|
|
69
|
+
|
|
70
|
+
For environments with the `server` extra installed, you can connect to Neo4j directly:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from educelab import hercdb
|
|
74
|
+
|
|
75
|
+
uri = "neo4j://localhost:7687"
|
|
76
|
+
user = "foo"
|
|
77
|
+
password = "bar"
|
|
78
|
+
db = hercdb.connect(uri, user, password)
|
|
79
|
+
if db.verify_connection():
|
|
80
|
+
print("Connected!")
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
If credentials are not passed directly, the package reads them from `~/.educedb` or environment variables. See [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md) for configuration details.
|
|
84
|
+
|
|
85
|
+
## REST API
|
|
86
|
+
|
|
87
|
+
A FastAPI-based REST API is available for querying the database over HTTP. All endpoints require Bearer token authentication.
|
|
88
|
+
|
|
89
|
+
```shell
|
|
90
|
+
uv run uvicorn educelab.hercdb.rest.server:app --reload
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Interactive API docs are available at `/docs` (Swagger UI) and `/redoc` (ReDoc) once the server is running.
|
|
94
|
+
|
|
95
|
+
See [src/educelab/hercdb/rest/README.md](src/educelab/hercdb/rest/README.md) for endpoint documentation and [docs/SERVER_SETUP.md](docs/SERVER_SETUP.md) for production deployment.
|
|
96
|
+
|
|
97
|
+
## Loading Data
|
|
98
|
+
|
|
99
|
+
Data loading is done in two steps using the loader scripts. Both read CSV files from `input_data/`.
|
|
100
|
+
|
|
101
|
+
### 1. Load metadata and UUIDs
|
|
102
|
+
|
|
103
|
+
```shell
|
|
104
|
+
uv run python src/educelab/hercdb/loader/metadata_loader.py
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Reads (defaults):
|
|
108
|
+
- `input_data/metadata_file.csv` - Pre-processed metadata file. (PHerc, Cornice, Pezzo, Disegni nodes and properties.)
|
|
109
|
+
- `input_data/uuid_file.csv` - Pre-processed uuid file. (all EduceLabID added)
|
|
110
|
+
|
|
111
|
+
Optional arguments:
|
|
112
|
+
```shell
|
|
113
|
+
uv run python src/educelab/hercdb/loader/metadata_loader.py \
|
|
114
|
+
--metadata path/to/metadata.csv \
|
|
115
|
+
--uuid path/to/uuid.csv
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
### 2. Load scan data
|
|
119
|
+
|
|
120
|
+
```shell
|
|
121
|
+
uv run python src/educelab/hercdb/loader/scan_loader.py
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Reads (defaults):
|
|
125
|
+
- `input_data/negatives.csv` - FlatbedScanDataset nodes
|
|
126
|
+
- `input_data/pgs_datasets_20260601(in).csv` - PGSRaw nodes
|
|
127
|
+
- `input_data/spectral_datasets_20260601_reconciled.csv` - SpectralRaw nodes
|
|
128
|
+
|
|
129
|
+
By default (`--replace`) it deletes all existing PGSRaw/SpectralRaw nodes and reloads from scratch (FlatbedScanDataset is untouched); pass `--no-replace` to merge into existing data instead. Nodes are keyed on the scan `uuid`, so re-running is idempotent.
|
|
130
|
+
|
|
131
|
+
Optional arguments:
|
|
132
|
+
```shell
|
|
133
|
+
uv run python src/educelab/hercdb/loader/scan_loader.py \
|
|
134
|
+
--negatives path/to/negatives.csv \
|
|
135
|
+
--photogrammetry path/to/pgs.csv \
|
|
136
|
+
--spectral path/to/spectral.csv \
|
|
137
|
+
--no-replace
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
**Note:** Run metadata_loader first since scan data links to EduceLabID nodes.
|
|
141
|
+
|
|
142
|
+
## Reporting Tools
|
|
143
|
+
|
|
144
|
+
### Scan completeness report
|
|
145
|
+
|
|
146
|
+
`el-hercdb-scan-report` walks every PHerc and its hierarchy and writes two CSVs that flag artifacts needing first-time scans or re-scans.
|
|
147
|
+
|
|
148
|
+
```shell
|
|
149
|
+
# After `uv sync --extra server`, the entry point is on PATH:
|
|
150
|
+
uv run el-hercdb-scan-report --out-dir ./tmp
|
|
151
|
+
|
|
152
|
+
# Equivalent fallback without re-syncing:
|
|
153
|
+
uv run python -m educelab.hercdb.cli.scan_completeness --out-dir ./tmp
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Outputs (default names):
|
|
157
|
+
|
|
158
|
+
- `scan_completeness_full.csv` — one row per (artifact, UUID) pair, plus blank-UUID sentinel rows for artifacts that have no EduceLabID assigned. Columns: PHerc, Cornice, Pezzo, UUID, PGS Status, PGS Latest Complete Date, PGS Path, Spectral Status, Spectral Latest Complete Date, Spectral Path, Institution.
|
|
159
|
+
- `scan_completeness_issues.csv` — same shape, filtered to rows where PGS or Spectral is missing/incomplete, or the artifact has no UUID at all.
|
|
160
|
+
|
|
161
|
+
Status values are `complete` / `incomplete` / `missing` when the artifact has a UUID, and **blank** when it does not (so "UUID assigned but no scan" stays distinguishable from "no UUID even assigned"). Files are written with `utf-8-sig` so Excel opens them with correct character encoding.
|
|
162
|
+
|
|
163
|
+
The report walks `REPLACES` relationships between EduceLabIDs, so pre-replacement scans on a retired predecessor UUID still count toward the artifact's coverage.
|
|
164
|
+
|
|
165
|
+
## Temporary Scripts and Notes
|
|
166
|
+
|
|
167
|
+
The `tmp/` directory contains temporary scripts, notes, and other informal resources shared among the team. Contents are version controlled but considered ephemeral — they may be rewritten or deleted at any time and should not be relied upon as stable code.
|
|
168
|
+
|
|
169
|
+
### Delete all data
|
|
170
|
+
|
|
171
|
+
To clear the database before reloading:
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
from educelab.hercdb.loader import PhercGraphDatabaseLoader
|
|
175
|
+
loader = PhercGraphDatabaseLoader()
|
|
176
|
+
loader._delete_all_nodes()
|
|
177
|
+
```
|
|
@@ -4,16 +4,13 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[tool.setuptools.packages.find]
|
|
6
6
|
where = ["src/"]
|
|
7
|
-
include = ["educelab.hercdb"]
|
|
7
|
+
include = ["educelab.hercdb*"]
|
|
8
8
|
|
|
9
9
|
[project]
|
|
10
10
|
name = "educelab-hercdb"
|
|
11
|
-
version = "0.
|
|
11
|
+
version = "0.2.2"
|
|
12
12
|
dependencies = [
|
|
13
|
-
"
|
|
14
|
-
"numpy>=2.0",
|
|
15
|
-
"pandas>=2.2",
|
|
16
|
-
"prompt-toolkit"
|
|
13
|
+
"requests>=2.32.5",
|
|
17
14
|
]
|
|
18
15
|
requires-python = ">= 3.10"
|
|
19
16
|
authors = [
|
|
@@ -32,8 +29,30 @@ classifiers = [
|
|
|
32
29
|
"Programming Language :: Python :: 3.12",
|
|
33
30
|
]
|
|
34
31
|
|
|
32
|
+
[project.optional-dependencies]
|
|
33
|
+
server = [
|
|
34
|
+
"fastapi>=0.124.4",
|
|
35
|
+
"neo4j>=5.20",
|
|
36
|
+
"numpy>=2.0",
|
|
37
|
+
"pandas>=2.2",
|
|
38
|
+
"prompt-toolkit",
|
|
39
|
+
"pydantic>=2.0",
|
|
40
|
+
"rapidfuzz>=3.0",
|
|
41
|
+
"uvicorn>=0.40.0",
|
|
42
|
+
]
|
|
43
|
+
|
|
35
44
|
[project.urls]
|
|
36
|
-
Repository = "https://
|
|
45
|
+
Repository = "https://github.com/educelab/educelab-hercdb"
|
|
37
46
|
|
|
38
47
|
[project.scripts]
|
|
39
|
-
el-hercdb-
|
|
48
|
+
el-hercdb-scan-report = "educelab.hercdb.cli.scan_completeness:main"
|
|
49
|
+
el-hercdb-sample-uuid-check = "educelab.hercdb.cli.sample_uuid_check:main"
|
|
50
|
+
el-hercdb-pipeline-cleanup = "educelab.hercdb.cli.cleanup_pipelines:main"
|
|
51
|
+
|
|
52
|
+
[dependency-groups]
|
|
53
|
+
dev = [
|
|
54
|
+
"ipykernel>=7.1.0",
|
|
55
|
+
"matplotlib>=3.10.9",
|
|
56
|
+
"networkx>=3.4.2",
|
|
57
|
+
"pyyaml>=6.0.3",
|
|
58
|
+
]
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Backward compatibility shim - use educelab.hercdb.db instead."""
|
|
2
|
+
|
|
3
|
+
# Re-export everything from db module for backward compatibility
|
|
4
|
+
from educelab.hercdb.db import (
|
|
5
|
+
GraphDBConnection,
|
|
6
|
+
connect,
|
|
7
|
+
DatasetType,
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"GraphDBConnection",
|
|
12
|
+
"connect",
|
|
13
|
+
"DatasetType",
|
|
14
|
+
]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Command-line interface tools for hercdb."""
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Delete pipeline records from Neo4j by pipeline id (DB-admin tool).
|
|
2
|
+
|
|
3
|
+
Companion to the pipeline-recording write path in the acquisition-workflow
|
|
4
|
+
``submit_uber_pipeline.py``: while that path is being tested, every (dry-run or
|
|
5
|
+
real) submission writes a Pipeline node keyed by its ``uber_job_id``. This tool
|
|
6
|
+
lets a database administrator remove those test records again, straight against
|
|
7
|
+
Neo4j (no REST server needed).
|
|
8
|
+
|
|
9
|
+
Deleting a pipeline removes the Pipeline node, its Process nodes, and the output
|
|
10
|
+
dataset nodes (PGSProcessed/SpectralProcessed/Registered/WebProcessed) those
|
|
11
|
+
processes produced. Input/raw datasets (PGSRaw/SpectralRaw/...) are left
|
|
12
|
+
untouched.
|
|
13
|
+
|
|
14
|
+
Wired up as the ``el-hercdb-pipeline-cleanup`` shell command (see
|
|
15
|
+
``[project.scripts]`` in pyproject.toml). Usage:
|
|
16
|
+
|
|
17
|
+
el-hercdb-pipeline-cleanup uber-1a2b3c4d [uber-...] [-y]
|
|
18
|
+
uv run el-hercdb-pipeline-cleanup --list # show what exists first
|
|
19
|
+
"""
|
|
20
|
+
import argparse
|
|
21
|
+
import sys
|
|
22
|
+
|
|
23
|
+
from educelab import hercdb
|
|
24
|
+
from educelab.hercdb import config
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def main():
|
|
28
|
+
parser = argparse.ArgumentParser(
|
|
29
|
+
description=(
|
|
30
|
+
"Delete pipeline records (Pipeline + Process + output datasets) "
|
|
31
|
+
"from Neo4j by pipeline id. Input/raw datasets are left untouched."
|
|
32
|
+
)
|
|
33
|
+
)
|
|
34
|
+
parser.add_argument(
|
|
35
|
+
"pipeline_ids", nargs="*",
|
|
36
|
+
help="Pipeline id(s) to delete (the uber_job_id printed by "
|
|
37
|
+
"submit_uber_pipeline.py).",
|
|
38
|
+
)
|
|
39
|
+
parser.add_argument(
|
|
40
|
+
"--list", action="store_true",
|
|
41
|
+
help="List all pipelines currently in the database and exit (no "
|
|
42
|
+
"deletion). Useful for finding ids to clean up.",
|
|
43
|
+
)
|
|
44
|
+
parser.add_argument(
|
|
45
|
+
"--yes", "-y", action="store_true",
|
|
46
|
+
help="Skip the confirmation prompt.",
|
|
47
|
+
)
|
|
48
|
+
args = parser.parse_args()
|
|
49
|
+
|
|
50
|
+
if not args.list and not args.pipeline_ids:
|
|
51
|
+
parser.error("provide one or more pipeline ids, or use --list.")
|
|
52
|
+
|
|
53
|
+
config.request_required()
|
|
54
|
+
db = hercdb.connect()
|
|
55
|
+
if not db.verify_connection():
|
|
56
|
+
print("Failed to connect to Neo4j.", file=sys.stderr)
|
|
57
|
+
sys.exit(1)
|
|
58
|
+
|
|
59
|
+
if args.list:
|
|
60
|
+
pipelines = db.get_all_pipeline_summaries()
|
|
61
|
+
if not pipelines:
|
|
62
|
+
print("No pipelines in the database.")
|
|
63
|
+
return
|
|
64
|
+
print(f"{len(pipelines)} pipeline(s):")
|
|
65
|
+
for p in pipelines:
|
|
66
|
+
pid = p.get("pipeline_id", "?")
|
|
67
|
+
uuid = p.get("artifact_uuid", "")
|
|
68
|
+
print(f" - {pid}" + (f" (artifact {uuid})" if uuid else ""))
|
|
69
|
+
return
|
|
70
|
+
|
|
71
|
+
if not args.yes:
|
|
72
|
+
print("About to delete the following pipeline(s):")
|
|
73
|
+
for pid in args.pipeline_ids:
|
|
74
|
+
print(f" - {pid}")
|
|
75
|
+
confirm = input("Proceed? [y/N]: ").strip().lower()
|
|
76
|
+
if confirm not in ("y", "yes"):
|
|
77
|
+
print("Aborted.")
|
|
78
|
+
return
|
|
79
|
+
|
|
80
|
+
failed = False
|
|
81
|
+
for pid in args.pipeline_ids:
|
|
82
|
+
result = db.delete_pipeline(pid)
|
|
83
|
+
if result is None:
|
|
84
|
+
print(f"{pid}: not found")
|
|
85
|
+
failed = True
|
|
86
|
+
else:
|
|
87
|
+
print(
|
|
88
|
+
f"{pid}: deleted "
|
|
89
|
+
f"({result.get('processes_deleted', 0)} process(es), "
|
|
90
|
+
f"{result.get('output_datasets_deleted', 0)} output dataset(s))"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
if failed:
|
|
94
|
+
sys.exit(1)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
if __name__ == "__main__":
|
|
98
|
+
try:
|
|
99
|
+
main()
|
|
100
|
+
except KeyboardInterrupt:
|
|
101
|
+
sys.exit(0)
|