marinerg-data 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- marinerg_data/__init__.py +14 -0
- marinerg_data/bundle.py +159 -0
- marinerg_data/ckan.py +272 -0
- marinerg_data/cli.py +324 -0
- marinerg_data/converters/__init__.py +80 -0
- marinerg_data/converters/wave_tank.py +215 -0
- marinerg_data/data/__init__.py +0 -0
- marinerg_data/data/capture_formats.yaml +92 -0
- marinerg_data/data/run_manifest.schema.json +252 -0
- marinerg_data/inspect_dir.py +151 -0
- marinerg_data/kerchunk_refs.py +132 -0
- marinerg_data/manifest.py +59 -0
- marinerg_data/mock.py +1652 -0
- marinerg_data/publish.py +120 -0
- marinerg_data/validation.py +317 -0
- marinerg_data/zenodo.py +187 -0
- marinerg_data-0.1.0.dist-info/METADATA +87 -0
- marinerg_data-0.1.0.dist-info/RECORD +21 -0
- marinerg_data-0.1.0.dist-info/WHEEL +5 -0
- marinerg_data-0.1.0.dist-info/entry_points.txt +2 -0
- marinerg_data-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""marinerg-data: facility-side convert / validate / publish tool.
|
|
2
|
+
|
|
3
|
+
Design: docs/domain-metadata-roadmap.md "Submission & ingest architecture" in
|
|
4
|
+
the data-access-service repo. Validation runs locally at the facility before any
|
|
5
|
+
upload; publication is thin orchestration over the facility's own Zenodo account
|
|
6
|
+
plus CKAN registration. Bulk data never transits the e-infrastructure.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
10
|
+
|
|
11
|
+
try:
|
|
12
|
+
__version__ = version("marinerg-data")
|
|
13
|
+
except PackageNotFoundError: # not installed (e.g. running from a source tree)
|
|
14
|
+
__version__ = "0.0.0"
|
marinerg_data/bundle.py
ADDED
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""A publishable bundle: one campaign directory.
|
|
2
|
+
|
|
3
|
+
Layout expected by ``marinerg-data publish <dir>``:
|
|
4
|
+
|
|
5
|
+
<dir>/dataset.yaml campaign-level metadata (this module's model)
|
|
6
|
+
<dir>/*.manifest.json one run manifest per run (schema v0.1)
|
|
7
|
+
<dir>/... data files referenced by the manifests
|
|
8
|
+
|
|
9
|
+
The CKAN dataset represents the campaign; runs are structured resources
|
|
10
|
+
within it (domain-metadata-roadmap.md design principle 6).
|
|
11
|
+
|
|
12
|
+
Publish state (Zenodo record id + DOIs) is written to
|
|
13
|
+
``<dir>/.marinerg-publish.json`` so ``publish --new-version`` knows which
|
|
14
|
+
record to version. The state file is local bookkeeping, not metadata — it
|
|
15
|
+
never gets uploaded.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
from datetime import UTC, datetime
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
import yaml
|
|
26
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
27
|
+
|
|
28
|
+
from .manifest import (
|
|
29
|
+
ManifestError,
|
|
30
|
+
find_dataset_metadata,
|
|
31
|
+
find_manifests,
|
|
32
|
+
load_manifest,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
STATE_FILE_NAME = ".marinerg-publish.json"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class Creator(BaseModel):
|
|
39
|
+
model_config = ConfigDict(extra="forbid")
|
|
40
|
+
|
|
41
|
+
name: str
|
|
42
|
+
affiliation: str | None = None
|
|
43
|
+
orcid: str | None = None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class DatasetMetadata(BaseModel):
|
|
47
|
+
"""Campaign-level metadata; the manifest carries per-run detail."""
|
|
48
|
+
|
|
49
|
+
model_config = ConfigDict(extra="forbid")
|
|
50
|
+
|
|
51
|
+
name: str = Field(description="CKAN dataset slug, e.g. 'gwk-floatx-2026'")
|
|
52
|
+
title: str
|
|
53
|
+
description: str
|
|
54
|
+
creators: list[Creator] = Field(min_length=1)
|
|
55
|
+
license: str = "CC-BY-4.0"
|
|
56
|
+
keywords: list[str] = Field(default_factory=list)
|
|
57
|
+
visibility: str = "public"
|
|
58
|
+
community: str = "marinerg-i"
|
|
59
|
+
|
|
60
|
+
facility_ref: str | None = None
|
|
61
|
+
facility_name: str | None = None
|
|
62
|
+
equipment_name: str | None = None
|
|
63
|
+
owner_org: str | None = None
|
|
64
|
+
|
|
65
|
+
site: str | None = None
|
|
66
|
+
feature_type: str | None = None
|
|
67
|
+
processing_level: str | None = None
|
|
68
|
+
coordinate_reference_system: str | None = None
|
|
69
|
+
data_mode: str | None = None
|
|
70
|
+
|
|
71
|
+
# Test Context (domain-metadata-roadmap short term). Passed to CKAN as
|
|
72
|
+
# extras until the scheming field group lands, then as top-level fields.
|
|
73
|
+
test_type: str | None = None
|
|
74
|
+
device_type: str | None = None
|
|
75
|
+
device_scale: str | None = None
|
|
76
|
+
campaign_name: str | None = None
|
|
77
|
+
test_report_ref: str | None = None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class PublishState(BaseModel):
|
|
81
|
+
model_config = ConfigDict(extra="allow")
|
|
82
|
+
|
|
83
|
+
record_id: str
|
|
84
|
+
doi: str
|
|
85
|
+
concept_doi: str | None = None
|
|
86
|
+
ckan_id: str | None = None
|
|
87
|
+
published_at: str | None = None
|
|
88
|
+
sandbox: bool = False
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class BundleError(Exception):
|
|
92
|
+
pass
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class Bundle:
|
|
96
|
+
def __init__(
|
|
97
|
+
self,
|
|
98
|
+
directory: Path,
|
|
99
|
+
metadata: DatasetMetadata,
|
|
100
|
+
manifests: dict[Path, dict[str, Any]],
|
|
101
|
+
):
|
|
102
|
+
self.directory = directory
|
|
103
|
+
self.metadata = metadata
|
|
104
|
+
self.manifests = manifests
|
|
105
|
+
|
|
106
|
+
@classmethod
|
|
107
|
+
def load(cls, directory: Path) -> Bundle:
|
|
108
|
+
if not directory.is_dir():
|
|
109
|
+
raise BundleError(f"Not a directory: {directory}")
|
|
110
|
+
metadata_path = find_dataset_metadata(directory)
|
|
111
|
+
if metadata_path is None:
|
|
112
|
+
raise BundleError(
|
|
113
|
+
f"No dataset.yaml in {directory} — the bundle needs "
|
|
114
|
+
"campaign-level metadata (title, creators, license...)"
|
|
115
|
+
)
|
|
116
|
+
raw = yaml.safe_load(metadata_path.read_text(encoding="utf-8")) or {}
|
|
117
|
+
metadata = DatasetMetadata(**raw)
|
|
118
|
+
manifests = {}
|
|
119
|
+
try:
|
|
120
|
+
for manifest_path in find_manifests(directory):
|
|
121
|
+
manifests[manifest_path] = load_manifest(manifest_path)
|
|
122
|
+
except ManifestError as exc:
|
|
123
|
+
raise BundleError(str(exc)) from exc
|
|
124
|
+
if not manifests:
|
|
125
|
+
raise BundleError(f"No *.manifest.json files in {directory}")
|
|
126
|
+
return cls(directory, metadata, manifests)
|
|
127
|
+
|
|
128
|
+
def upload_files(self) -> list[Path]:
|
|
129
|
+
"""Files that belong in the Zenodo record: every manifest, plus all
|
|
130
|
+
zenodo-hosted resources, media, documents, and reference sidecars they
|
|
131
|
+
reference."""
|
|
132
|
+
files: list[Path] = sorted(self.manifests)
|
|
133
|
+
seen = set(files)
|
|
134
|
+
for manifest_path, manifest in sorted(self.manifests.items()):
|
|
135
|
+
base = manifest_path.parent
|
|
136
|
+
for section in ("resources", "media", "documents", "references"):
|
|
137
|
+
for entry in manifest.get(section, []):
|
|
138
|
+
if entry.get("hosted", "zenodo") != "zenodo":
|
|
139
|
+
continue
|
|
140
|
+
local = base / entry["path"]
|
|
141
|
+
if local not in seen:
|
|
142
|
+
files.append(local)
|
|
143
|
+
seen.add(local)
|
|
144
|
+
return files
|
|
145
|
+
|
|
146
|
+
@property
|
|
147
|
+
def state_path(self) -> Path:
|
|
148
|
+
return self.directory / STATE_FILE_NAME
|
|
149
|
+
|
|
150
|
+
def read_state(self) -> PublishState | None:
|
|
151
|
+
if not self.state_path.is_file():
|
|
152
|
+
return None
|
|
153
|
+
return PublishState(**json.loads(self.state_path.read_text(encoding="utf-8")))
|
|
154
|
+
|
|
155
|
+
def write_state(self, state: PublishState) -> None:
|
|
156
|
+
state.published_at = datetime.now(UTC).isoformat()
|
|
157
|
+
self.state_path.write_text(
|
|
158
|
+
json.dumps(state.model_dump(), indent=2) + "\n", encoding="utf-8"
|
|
159
|
+
)
|
marinerg_data/ckan.py
ADDED
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
"""CKAN catalogue registration.
|
|
2
|
+
|
|
3
|
+
Maps a published bundle onto the marinerg scheming schema. Conventions
|
|
4
|
+
follow infra/ckan/scripts/seed_ckan.py: scheming fields go as top-level
|
|
5
|
+
package keys, everything else into the extras list (scheming rejects
|
|
6
|
+
extras whose key collides with a schema field).
|
|
7
|
+
|
|
8
|
+
Record semantics (data-access-service/CLAUDE.md): the version DOI is the
|
|
9
|
+
record's primary identifier; the concept DOI travels as an extra until the
|
|
10
|
+
record-semantics scheming fields land. Test Context and Forcing values are
|
|
11
|
+
extras for the same reason — one home each, promoted when the scheming
|
|
12
|
+
groups are added.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
import requests
|
|
21
|
+
|
|
22
|
+
from .bundle import Bundle
|
|
23
|
+
from .zenodo import PublishedRecord
|
|
24
|
+
|
|
25
|
+
# Keep in sync with ckanext-marinerg/ckanext/marinerg/schemas/dataset.yaml
|
|
26
|
+
# (same list as infra/ckan/scripts/seed_ckan.py).
|
|
27
|
+
SCHEMING_FIELDS = {
|
|
28
|
+
"zenodo_doi",
|
|
29
|
+
"publication_date",
|
|
30
|
+
"visibility",
|
|
31
|
+
"access_scope",
|
|
32
|
+
"data_mode",
|
|
33
|
+
"source_kind",
|
|
34
|
+
"facility_ref",
|
|
35
|
+
"facility_name",
|
|
36
|
+
"equipment_name",
|
|
37
|
+
"feature_type",
|
|
38
|
+
"site",
|
|
39
|
+
"time_coverage_start",
|
|
40
|
+
"time_coverage_end",
|
|
41
|
+
"processing_level",
|
|
42
|
+
"coordinate_reference_system",
|
|
43
|
+
# Test Context + Forcing groups
|
|
44
|
+
"test_type",
|
|
45
|
+
"device_type",
|
|
46
|
+
"device_scale",
|
|
47
|
+
"device_trl",
|
|
48
|
+
"campaign_name",
|
|
49
|
+
"campaign_id",
|
|
50
|
+
"test_report_ref",
|
|
51
|
+
"wave_type",
|
|
52
|
+
"wave_height_m",
|
|
53
|
+
"wave_period_s",
|
|
54
|
+
"water_depth_m",
|
|
55
|
+
"current_speed_ms",
|
|
56
|
+
"wind_speed_ms",
|
|
57
|
+
"sample_frequencies_hz",
|
|
58
|
+
"run_count",
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class CkanError(Exception):
|
|
63
|
+
pass
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class CkanConflict(CkanError): # noqa: N818 — established exception name
|
|
67
|
+
"""package_create hit an existing dataset with the same name."""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _json_value(value: Any) -> str:
|
|
71
|
+
if value is None:
|
|
72
|
+
return ""
|
|
73
|
+
if isinstance(value, (str, int, float, bool)):
|
|
74
|
+
return str(value)
|
|
75
|
+
return json.dumps(value, ensure_ascii=False)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _single_or_list(values: list[Any]) -> Any:
|
|
79
|
+
unique = sorted({value for value in values if value is not None}, key=str)
|
|
80
|
+
if not unique:
|
|
81
|
+
return None
|
|
82
|
+
if len(unique) == 1:
|
|
83
|
+
return unique[0]
|
|
84
|
+
return unique
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _aggregate_manifests(bundle: Bundle) -> dict[str, Any]:
|
|
88
|
+
"""Campaign-level summary of the per-run manifests."""
|
|
89
|
+
manifests = list(bundle.manifests.values())
|
|
90
|
+
|
|
91
|
+
starts = sorted(
|
|
92
|
+
manifest.get("run", {}).get("started")
|
|
93
|
+
for manifest in manifests
|
|
94
|
+
if manifest.get("run", {}).get("started")
|
|
95
|
+
)
|
|
96
|
+
values: dict[str, Any] = {"run_count": len(manifests)}
|
|
97
|
+
if starts:
|
|
98
|
+
values["time_coverage_start"] = starts[0]
|
|
99
|
+
values["time_coverage_end"] = starts[-1]
|
|
100
|
+
|
|
101
|
+
for key in (
|
|
102
|
+
"wave_type",
|
|
103
|
+
"wave_height_m",
|
|
104
|
+
"wave_period_s",
|
|
105
|
+
"water_depth_m",
|
|
106
|
+
"current_speed_ms",
|
|
107
|
+
"wind_speed_ms",
|
|
108
|
+
):
|
|
109
|
+
value = _single_or_list(
|
|
110
|
+
[manifest.get("forcing", {}).get(key) for manifest in manifests]
|
|
111
|
+
)
|
|
112
|
+
if value is not None:
|
|
113
|
+
values[key] = value
|
|
114
|
+
|
|
115
|
+
rates = sorted(
|
|
116
|
+
{
|
|
117
|
+
channel.get("sample_rate_hz")
|
|
118
|
+
for manifest in manifests
|
|
119
|
+
for channel in manifest.get("channels", [])
|
|
120
|
+
if channel.get("sample_rate_hz")
|
|
121
|
+
}
|
|
122
|
+
)
|
|
123
|
+
if rates:
|
|
124
|
+
values["sample_frequencies_hz"] = rates
|
|
125
|
+
|
|
126
|
+
parameters = sorted(
|
|
127
|
+
{
|
|
128
|
+
channel.get("name")
|
|
129
|
+
for manifest in manifests
|
|
130
|
+
for channel in manifest.get("channels", [])
|
|
131
|
+
if channel.get("name")
|
|
132
|
+
}
|
|
133
|
+
)
|
|
134
|
+
if parameters:
|
|
135
|
+
values["measured_parameters"] = parameters
|
|
136
|
+
return values
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _resources(bundle: Bundle, record: PublishedRecord) -> list[dict[str, Any]]:
|
|
140
|
+
resources = []
|
|
141
|
+
for manifest_path, manifest in sorted(bundle.manifests.items()):
|
|
142
|
+
run_id = manifest.get("run", {}).get("id", manifest_path.stem)
|
|
143
|
+
for entry in manifest.get("resources", []):
|
|
144
|
+
if entry.get("hosted", "zenodo") != "zenodo":
|
|
145
|
+
continue
|
|
146
|
+
name = entry["path"].rsplit("/", 1)[-1]
|
|
147
|
+
url = (
|
|
148
|
+
f"{record.html_url}/files/{name}"
|
|
149
|
+
if record.html_url
|
|
150
|
+
else f"https://doi.org/{record.doi}"
|
|
151
|
+
)
|
|
152
|
+
resource = {
|
|
153
|
+
"name": name,
|
|
154
|
+
"url": url,
|
|
155
|
+
"format": entry.get("format", ""),
|
|
156
|
+
"description": f"Run {run_id} ({entry.get('record_type', 'data')})",
|
|
157
|
+
}
|
|
158
|
+
resources.append({k: v for k, v in resource.items() if v})
|
|
159
|
+
manifest_url = (
|
|
160
|
+
f"{record.html_url}/files/{manifest_path.name}"
|
|
161
|
+
if record.html_url
|
|
162
|
+
else f"https://doi.org/{record.doi}"
|
|
163
|
+
)
|
|
164
|
+
resources.append(
|
|
165
|
+
{
|
|
166
|
+
"name": manifest_path.name,
|
|
167
|
+
"url": manifest_url,
|
|
168
|
+
"format": "JSON",
|
|
169
|
+
"description": f"Run manifest for run {run_id} "
|
|
170
|
+
"(MARINERG-i run-manifest schema)",
|
|
171
|
+
}
|
|
172
|
+
)
|
|
173
|
+
return resources
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def build_package(bundle: Bundle, record: PublishedRecord) -> dict[str, Any]:
|
|
177
|
+
metadata = bundle.metadata
|
|
178
|
+
package: dict[str, Any] = {
|
|
179
|
+
"name": metadata.name,
|
|
180
|
+
"title": metadata.title,
|
|
181
|
+
"notes": metadata.description.strip(),
|
|
182
|
+
"license_id": metadata.license,
|
|
183
|
+
"private": metadata.visibility == "private",
|
|
184
|
+
"visibility": metadata.visibility,
|
|
185
|
+
"url": f"https://doi.org/{record.doi}",
|
|
186
|
+
"tags": [{"name": keyword} for keyword in metadata.keywords],
|
|
187
|
+
"zenodo_doi": record.doi,
|
|
188
|
+
"source_kind": "zenodo",
|
|
189
|
+
"author": metadata.creators[0].name,
|
|
190
|
+
"resources": _resources(bundle, record),
|
|
191
|
+
}
|
|
192
|
+
if metadata.owner_org:
|
|
193
|
+
package["owner_org"] = metadata.owner_org
|
|
194
|
+
|
|
195
|
+
fields = {
|
|
196
|
+
"facility_ref": metadata.facility_ref,
|
|
197
|
+
"facility_name": metadata.facility_name,
|
|
198
|
+
"equipment_name": metadata.equipment_name,
|
|
199
|
+
"site": metadata.site,
|
|
200
|
+
"feature_type": metadata.feature_type,
|
|
201
|
+
"processing_level": metadata.processing_level,
|
|
202
|
+
"coordinate_reference_system": metadata.coordinate_reference_system,
|
|
203
|
+
"data_mode": metadata.data_mode,
|
|
204
|
+
"test_type": metadata.test_type,
|
|
205
|
+
"device_type": metadata.device_type,
|
|
206
|
+
"device_scale": metadata.device_scale,
|
|
207
|
+
"campaign_name": metadata.campaign_name,
|
|
208
|
+
"test_report_ref": metadata.test_report_ref,
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
extras: dict[str, Any] = {}
|
|
212
|
+
if record.concept_doi:
|
|
213
|
+
extras["concept_doi"] = record.concept_doi
|
|
214
|
+
|
|
215
|
+
for key, value in _aggregate_manifests(bundle).items():
|
|
216
|
+
# Aggregates that differ across runs come back as lists; a scheming
|
|
217
|
+
# numeric field holds one value, so lists stay in extras (except
|
|
218
|
+
# sample_frequencies_hz, whose scheming field is a list).
|
|
219
|
+
is_field_shaped = not isinstance(value, list) or key == "sample_frequencies_hz"
|
|
220
|
+
if key in SCHEMING_FIELDS and is_field_shaped:
|
|
221
|
+
fields[key] = value
|
|
222
|
+
else:
|
|
223
|
+
extras[key] = value
|
|
224
|
+
|
|
225
|
+
for key, value in fields.items():
|
|
226
|
+
if value is not None:
|
|
227
|
+
package[key] = _json_value(value)
|
|
228
|
+
package["extras"] = [
|
|
229
|
+
{"key": key, "value": _json_value(value)} for key, value in extras.items()
|
|
230
|
+
]
|
|
231
|
+
return package
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
class CkanRegistrar:
|
|
235
|
+
def __init__(self, url: str, token: str, session: requests.Session | None = None):
|
|
236
|
+
if not url:
|
|
237
|
+
raise CkanError("A CKAN URL is required (set CKAN_URL)")
|
|
238
|
+
if not token:
|
|
239
|
+
raise CkanError("A CKAN API token is required (set CKAN_TOKEN)")
|
|
240
|
+
self.base_url = url.rstrip("/")
|
|
241
|
+
self.session = session or requests.Session()
|
|
242
|
+
self.session.headers.update({"Authorization": token})
|
|
243
|
+
|
|
244
|
+
def _call(self, action: str, payload: dict[str, Any]) -> dict[str, Any]:
|
|
245
|
+
url = f"{self.base_url}/api/3/action/{action}"
|
|
246
|
+
response = self.session.post(url, json=payload, timeout=60)
|
|
247
|
+
if response.status_code >= 400 and response.status_code != 409:
|
|
248
|
+
try:
|
|
249
|
+
detail = response.json().get("error")
|
|
250
|
+
except ValueError:
|
|
251
|
+
detail = response.text
|
|
252
|
+
raise CkanError(f"{action} -> {response.status_code}: {detail}")
|
|
253
|
+
body = response.json()
|
|
254
|
+
if not body.get("success", False):
|
|
255
|
+
if response.status_code == 409:
|
|
256
|
+
raise CkanConflict(body.get("error", {}))
|
|
257
|
+
raise CkanError(body.get("error", {}).get("message", f"{action} failed"))
|
|
258
|
+
return body["result"]
|
|
259
|
+
|
|
260
|
+
def register(self, package: dict[str, Any]) -> dict[str, Any]:
|
|
261
|
+
try:
|
|
262
|
+
return self._call("package_create", package)
|
|
263
|
+
except CkanConflict:
|
|
264
|
+
existing = self._call("package_show", {"id": package["name"]})
|
|
265
|
+
payload = dict(package)
|
|
266
|
+
payload["id"] = existing["id"]
|
|
267
|
+
return self._call("package_update", payload)
|
|
268
|
+
|
|
269
|
+
def update(self, ckan_id: str, package: dict[str, Any]) -> dict[str, Any]:
|
|
270
|
+
payload = dict(package)
|
|
271
|
+
payload["id"] = ckan_id
|
|
272
|
+
return self._call("package_update", payload)
|