marinerg-data 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. marinerg_data-0.1.0/PKG-INFO +87 -0
  2. marinerg_data-0.1.0/README.md +56 -0
  3. marinerg_data-0.1.0/pyproject.toml +111 -0
  4. marinerg_data-0.1.0/setup.cfg +4 -0
  5. marinerg_data-0.1.0/src/marinerg_data/__init__.py +14 -0
  6. marinerg_data-0.1.0/src/marinerg_data/bundle.py +159 -0
  7. marinerg_data-0.1.0/src/marinerg_data/ckan.py +272 -0
  8. marinerg_data-0.1.0/src/marinerg_data/cli.py +324 -0
  9. marinerg_data-0.1.0/src/marinerg_data/converters/__init__.py +80 -0
  10. marinerg_data-0.1.0/src/marinerg_data/converters/wave_tank.py +215 -0
  11. marinerg_data-0.1.0/src/marinerg_data/data/__init__.py +0 -0
  12. marinerg_data-0.1.0/src/marinerg_data/data/capture_formats.yaml +92 -0
  13. marinerg_data-0.1.0/src/marinerg_data/data/run_manifest.schema.json +252 -0
  14. marinerg_data-0.1.0/src/marinerg_data/inspect_dir.py +151 -0
  15. marinerg_data-0.1.0/src/marinerg_data/kerchunk_refs.py +132 -0
  16. marinerg_data-0.1.0/src/marinerg_data/manifest.py +59 -0
  17. marinerg_data-0.1.0/src/marinerg_data/mock.py +1652 -0
  18. marinerg_data-0.1.0/src/marinerg_data/publish.py +120 -0
  19. marinerg_data-0.1.0/src/marinerg_data/validation.py +317 -0
  20. marinerg_data-0.1.0/src/marinerg_data/zenodo.py +187 -0
  21. marinerg_data-0.1.0/src/marinerg_data.egg-info/PKG-INFO +87 -0
  22. marinerg_data-0.1.0/src/marinerg_data.egg-info/SOURCES.txt +34 -0
  23. marinerg_data-0.1.0/src/marinerg_data.egg-info/dependency_links.txt +1 -0
  24. marinerg_data-0.1.0/src/marinerg_data.egg-info/entry_points.txt +2 -0
  25. marinerg_data-0.1.0/src/marinerg_data.egg-info/requires.txt +18 -0
  26. marinerg_data-0.1.0/src/marinerg_data.egg-info/top_level.txt +1 -0
  27. marinerg_data-0.1.0/tests/test_bundle.py +103 -0
  28. marinerg_data-0.1.0/tests/test_ckan.py +164 -0
  29. marinerg_data-0.1.0/tests/test_cli.py +153 -0
  30. marinerg_data-0.1.0/tests/test_converters.py +153 -0
  31. marinerg_data-0.1.0/tests/test_inspect.py +92 -0
  32. marinerg_data-0.1.0/tests/test_kerchunk_refs.py +158 -0
  33. marinerg_data-0.1.0/tests/test_mock.py +243 -0
  34. marinerg_data-0.1.0/tests/test_publish.py +156 -0
  35. marinerg_data-0.1.0/tests/test_validation.py +214 -0
  36. marinerg_data-0.1.0/tests/test_zenodo.py +179 -0
@@ -0,0 +1,87 @@
1
+ Metadata-Version: 2.4
2
+ Name: marinerg-data
3
+ Version: 0.1.0
4
+ Summary: Facility-side MARINERG-i data client: inspect, convert, validate, publish, and fetch marine test-facility datasets.
5
+ Author-email: Irish Centre for High End Computing <james.grogan@ichec.ie>
6
+ License: AGPL-3.0-or-later
7
+ Project-URL: Repository, https://git.ichec.ie/marinerg-i/marinerg-data
8
+ Project-URL: Homepage, https://git.ichec.ie/marinerg-i/marinerg-data
9
+ Keywords: Marine Renewable Energy,Data,EOSC,Zenodo,CKAN,kerchunk
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.14
13
+ Classifier: Operating System :: OS Independent
14
+ Requires-Python: >=3.14
15
+ Description-Content-Type: text/markdown
16
+ Requires-Dist: jsonschema>=4
17
+ Requires-Dist: requests
18
+ Requires-Dist: PyYAML
19
+ Requires-Dist: pydantic>=2
20
+ Provides-Extra: mock
21
+ Requires-Dist: numpy; extra == "mock"
22
+ Provides-Extra: kerchunk
23
+ Requires-Dist: kerchunk>=0.2.6; extra == "kerchunk"
24
+ Requires-Dist: h5py>=3.10; extra == "kerchunk"
25
+ Requires-Dist: fsspec>=2024.2; extra == "kerchunk"
26
+ Requires-Dist: zarr>=3; extra == "kerchunk"
27
+ Requires-Dist: xarray>=2024.1; extra == "kerchunk"
28
+ Provides-Extra: types
29
+ Requires-Dist: types-requests; extra == "types"
30
+ Requires-Dist: types-PyYAML; extra == "types"
31
+
32
+ # marinerg-data
33
+
34
+ Facility-side data client for the MARINERG-i e-infrastructure: **inspect →
35
+ convert → validate → publish**, plus VRE consumption (search, resolve, fetch,
36
+ lazy streaming). Django-free and pip-installable, so it runs both at a test
37
+ facility and inside a Blue-Cloud / D4Science VRE Jupyter image.
38
+
39
+ The invariant that shapes everything: **bulk data never transits the
40
+ e-infrastructure.** Facilities publish to their own Zenodo; this tool registers
41
+ the record in CKAN; the server only re-validates metadata.
42
+
43
+ ## Install
44
+
45
+ ```sh
46
+ pip install marinerg-data # core client (inspect/validate/publish)
47
+ pip install "marinerg-data[kerchunk]" # + kerchunk reference sidecars
48
+ pip install "marinerg-data[mock]" # + synthetic scenario generator
49
+ pip install marinerg-wave-tank # optional wave-tank converter family
50
+ ```
51
+
52
+ ## Commands
53
+
54
+ | Command | Purpose |
55
+ |---|---|
56
+ | `inspect <dir>` | Sniff capture formats, report converter families; `--draft-manifest` writes a skeleton |
57
+ | `convert <run-dir>` | Run a converter family (`--family wave-tank`, else auto-detect); writes open-format outputs + `<run>.manifest.json`, then validates. With `[kerchunk]`, also emits a `<file>.nc.kerchunk.json` reference sidecar per NetCDF |
58
+ | `validate <manifest-or-dir>` | JSON Schema (run-manifest v0.1) + referential integrity + file existence; `--verify-checksums`, `--no-files` |
59
+ | `publish <bundle-dir>` | Facility Zenodo record + CKAN registration; `--sandbox`, `--new-version`, `--dry-run` |
60
+ | `mock --scenario <name>` | Synthetic NetCDF + manifest for fixtures/seeding/demos (`--scenario list`) |
61
+
62
+ Environment: `ZENODO_TOKEN` (a sandbox token with `--sandbox`), `CKAN_URL`,
63
+ `CKAN_TOKEN`.
64
+
65
+ ## Cloud-optimised access (VRE)
66
+
67
+ With the `[kerchunk]` extra, `convert` emits a kerchunk reference sidecar per
68
+ NetCDF. The sidecar lets `xarray` lazily stream individual chunks straight from
69
+ Zenodo over HTTP range requests — open a 10 GB campaign, plot five minutes of one
70
+ gauge, transfer megabytes not gigabytes. The reference URL is templated (`{{u}}`)
71
+ so the sidecar is portable; it is retargeted to the Zenodo file URL at publish.
72
+
73
+ ## Develop
74
+
75
+ ```sh
76
+ uv sync --group dev
77
+ uv run pytest # Django-free, no network (HTTP is faked)
78
+ uv run ruff check src tests
79
+ ```
80
+
81
+ Python 3.14, uv-native toolchain. See `CLAUDE.md` for architecture pointers and
82
+ the token-free release flow.
83
+
84
+ ## Licence
85
+
86
+ Copyright ICHEC. GNU AGPL v3 or later. Exemptions available for MARINERG-i
87
+ project partners.
@@ -0,0 +1,56 @@
1
+ # marinerg-data
2
+
3
+ Facility-side data client for the MARINERG-i e-infrastructure: **inspect →
4
+ convert → validate → publish**, plus VRE consumption (search, resolve, fetch,
5
+ lazy streaming). Django-free and pip-installable, so it runs both at a test
6
+ facility and inside a Blue-Cloud / D4Science VRE Jupyter image.
7
+
8
+ The invariant that shapes everything: **bulk data never transits the
9
+ e-infrastructure.** Facilities publish to their own Zenodo; this tool registers
10
+ the record in CKAN; the server only re-validates metadata.
11
+
12
+ ## Install
13
+
14
+ ```sh
15
+ pip install marinerg-data # core client (inspect/validate/publish)
16
+ pip install "marinerg-data[kerchunk]" # + kerchunk reference sidecars
17
+ pip install "marinerg-data[mock]" # + synthetic scenario generator
18
+ pip install marinerg-wave-tank # optional wave-tank converter family
19
+ ```
20
+
21
+ ## Commands
22
+
23
+ | Command | Purpose |
24
+ |---|---|
25
+ | `inspect <dir>` | Sniff capture formats, report converter families; `--draft-manifest` writes a skeleton |
26
+ | `convert <run-dir>` | Run a converter family (`--family wave-tank`, else auto-detect); writes open-format outputs + `<run>.manifest.json`, then validates. With `[kerchunk]`, also emits a `<file>.nc.kerchunk.json` reference sidecar per NetCDF |
27
+ | `validate <manifest-or-dir>` | JSON Schema (run-manifest v0.1) + referential integrity + file existence; `--verify-checksums`, `--no-files` |
28
+ | `publish <bundle-dir>` | Facility Zenodo record + CKAN registration; `--sandbox`, `--new-version`, `--dry-run` |
29
+ | `mock --scenario <name>` | Synthetic NetCDF + manifest for fixtures/seeding/demos (`--scenario list`) |
30
+
31
+ Environment: `ZENODO_TOKEN` (a sandbox token with `--sandbox`), `CKAN_URL`,
32
+ `CKAN_TOKEN`.
33
+
34
+ ## Cloud-optimised access (VRE)
35
+
36
+ With the `[kerchunk]` extra, `convert` emits a kerchunk reference sidecar per
37
+ NetCDF. The sidecar lets `xarray` lazily stream individual chunks straight from
38
+ Zenodo over HTTP range requests — open a 10 GB campaign, plot five minutes of one
39
+ gauge, transfer megabytes not gigabytes. The reference URL is templated (`{{u}}`)
40
+ so the sidecar is portable; it is retargeted to the Zenodo file URL at publish.
41
+
42
+ ## Develop
43
+
44
+ ```sh
45
+ uv sync --group dev
46
+ uv run pytest # Django-free, no network (HTTP is faked)
47
+ uv run ruff check src tests
48
+ ```
49
+
50
+ Python 3.14, uv-native toolchain. See `CLAUDE.md` for architecture pointers and
51
+ the token-free release flow.
52
+
53
+ ## Licence
54
+
55
+ Copyright ICHEC. GNU AGPL v3 or later. Exemptions available for MARINERG-i
56
+ project partners.
@@ -0,0 +1,111 @@
1
+ [project]
2
+ name = "marinerg-data"
3
+ dynamic = ["version"]
4
+ authors = [
5
+ { name = "Irish Centre for High End Computing", email = "james.grogan@ichec.ie" },
6
+ ]
7
+ description = "Facility-side MARINERG-i data client: inspect, convert, validate, publish, and fetch marine test-facility datasets."
8
+ readme = "README.md"
9
+ requires-python = ">=3.14"
10
+ license = { text = "AGPL-3.0-or-later" }
11
+ classifiers = [
12
+ "Development Status :: 3 - Alpha",
13
+ "Programming Language :: Python :: 3",
14
+ "Programming Language :: Python :: 3.14",
15
+ "Operating System :: OS Independent",
16
+ ]
17
+ keywords = ["Marine Renewable Energy", "Data", "EOSC", "Zenodo", "CKAN", "kerchunk"]
18
+
19
+ # Django-free client. Core deps are what the CLI itself needs; heavier,
20
+ # workflow-specific stacks are optional extras.
21
+ dependencies = [
22
+ "jsonschema>=4",
23
+ "requests",
24
+ "PyYAML",
25
+ "pydantic>=2",
26
+ ]
27
+
28
+ [project.urls]
29
+ Repository = "https://git.ichec.ie/marinerg-i/marinerg-data"
30
+ Homepage = "https://git.ichec.ie/marinerg-i/marinerg-data"
31
+
32
+ [project.optional-dependencies]
33
+ # `marinerg-data mock` generates synthetic scenario datasets.
34
+ mock = ["numpy"]
35
+
36
+ # Cloud-optimised access: emit kerchunk reference sidecars per NetCDF for lazy
37
+ # HTTP-range streaming from Zenodo (see the converter / VRE docs).
38
+ kerchunk = [
39
+ "kerchunk>=0.2.6",
40
+ "h5py>=3.10",
41
+ "fsspec>=2024.2",
42
+ "zarr>=3",
43
+ "xarray>=2024.1",
44
+ ]
45
+
46
+ types = ["types-requests", "types-PyYAML"]
47
+
48
+ [project.scripts]
49
+ marinerg-data = "marinerg_data.cli:main"
50
+
51
+ [build-system]
52
+ requires = ["setuptools>=64", "setuptools-scm>=8", "wheel"]
53
+ build-backend = "setuptools.build_meta"
54
+
55
+ # Version derived from the git tag (setuptools-scm); fallback keeps non-tag
56
+ # builds importable. Token-free release: compute-version stamps
57
+ # SETUPTOOLS_SCM_PRETEND_VERSION, create-release makes the tag (planning#132).
58
+ [tool.setuptools_scm]
59
+ fallback_version = "0.0.0"
60
+
61
+ # Development toolchain (uv-native, PEP 735). CI installs it with
62
+ # `uv sync --group dev`. Toolchain set canonical (see ichec-platform-core).
63
+ [dependency-groups]
64
+ dev = [
65
+ "ruff",
66
+ "mypy",
67
+ "pytest",
68
+ "pytest-cov",
69
+ "jsonschema>=4",
70
+ "requests",
71
+ "types-requests",
72
+ "PyYAML",
73
+ "types-PyYAML",
74
+ "numpy",
75
+ # kerchunk sidecar tests (round-trip lazy open); mirror the [kerchunk] extra.
76
+ "kerchunk>=0.2.6",
77
+ "h5py>=3.10",
78
+ "fsspec>=2024.2",
79
+ "zarr>=3",
80
+ "xarray>=2024.1",
81
+ "h5netcdf>=1.3",
82
+ ]
83
+
84
+ [tool.mypy]
85
+ ignore_missing_imports = true
86
+ # strict baseline (target — enable as the codebase reaches it):
87
+ # warn_return_any = true
88
+ # disallow_untyped_defs = true
89
+
90
+ [tool.setuptools.package-data]
91
+ "marinerg_data.data" = ["run_manifest.schema.json", "capture_formats.yaml"]
92
+
93
+ [tool.setuptools.packages.find]
94
+ where = ["src"]
95
+
96
+ # --- ruff: format + lint (supersedes black, isort, flake8, pylint) ---
97
+ [tool.ruff]
98
+ line-length = 88
99
+ target-version = "py314"
100
+ src = ["src", "tests"]
101
+
102
+ [tool.ruff.lint]
103
+ select = ["E", "W", "F", "I", "N", "UP", "B"]
104
+
105
+ [tool.ruff.lint.isort]
106
+ known-first-party = ["marinerg_data"]
107
+
108
+ [tool.pytest.ini_options]
109
+ testpaths = ["tests"]
110
+ log_cli_level = "warning"
111
+ addopts = "--cov=marinerg_data --cov-report term --cov-report xml:coverage.xml --cov-report html"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,14 @@
1
+ """marinerg-data: facility-side convert / validate / publish tool.
2
+
3
+ Design: docs/domain-metadata-roadmap.md "Submission & ingest architecture" in
4
+ the data-access-service repo. Validation runs locally at the facility before any
5
+ upload; publication is thin orchestration over the facility's own Zenodo account
6
+ plus CKAN registration. Bulk data never transits the e-infrastructure.
7
+ """
8
+
9
+ from importlib.metadata import PackageNotFoundError, version
10
+
11
+ try:
12
+ __version__ = version("marinerg-data")
13
+ except PackageNotFoundError: # not installed (e.g. running from a source tree)
14
+ __version__ = "0.0.0"
@@ -0,0 +1,159 @@
1
+ """A publishable bundle: one campaign directory.
2
+
3
+ Layout expected by ``marinerg-data publish <dir>``:
4
+
5
+ <dir>/dataset.yaml campaign-level metadata (this module's model)
6
+ <dir>/*.manifest.json one run manifest per run (schema v0.1)
7
+ <dir>/... data files referenced by the manifests
8
+
9
+ The CKAN dataset represents the campaign; runs are structured resources
10
+ within it (domain-metadata-roadmap.md design principle 6).
11
+
12
+ Publish state (Zenodo record id + DOIs) is written to
13
+ ``<dir>/.marinerg-publish.json`` so ``publish --new-version`` knows which
14
+ record to version. The state file is local bookkeeping, not metadata — it
15
+ never gets uploaded.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ from datetime import UTC, datetime
22
+ from pathlib import Path
23
+ from typing import Any
24
+
25
+ import yaml
26
+ from pydantic import BaseModel, ConfigDict, Field
27
+
28
+ from .manifest import (
29
+ ManifestError,
30
+ find_dataset_metadata,
31
+ find_manifests,
32
+ load_manifest,
33
+ )
34
+
35
+ STATE_FILE_NAME = ".marinerg-publish.json"
36
+
37
+
38
+ class Creator(BaseModel):
39
+ model_config = ConfigDict(extra="forbid")
40
+
41
+ name: str
42
+ affiliation: str | None = None
43
+ orcid: str | None = None
44
+
45
+
46
+ class DatasetMetadata(BaseModel):
47
+ """Campaign-level metadata; the manifest carries per-run detail."""
48
+
49
+ model_config = ConfigDict(extra="forbid")
50
+
51
+ name: str = Field(description="CKAN dataset slug, e.g. 'gwk-floatx-2026'")
52
+ title: str
53
+ description: str
54
+ creators: list[Creator] = Field(min_length=1)
55
+ license: str = "CC-BY-4.0"
56
+ keywords: list[str] = Field(default_factory=list)
57
+ visibility: str = "public"
58
+ community: str = "marinerg-i"
59
+
60
+ facility_ref: str | None = None
61
+ facility_name: str | None = None
62
+ equipment_name: str | None = None
63
+ owner_org: str | None = None
64
+
65
+ site: str | None = None
66
+ feature_type: str | None = None
67
+ processing_level: str | None = None
68
+ coordinate_reference_system: str | None = None
69
+ data_mode: str | None = None
70
+
71
+ # Test Context (domain-metadata-roadmap short term). Passed to CKAN as
72
+ # extras until the scheming field group lands, then as top-level fields.
73
+ test_type: str | None = None
74
+ device_type: str | None = None
75
+ device_scale: str | None = None
76
+ campaign_name: str | None = None
77
+ test_report_ref: str | None = None
78
+
79
+
80
+ class PublishState(BaseModel):
81
+ model_config = ConfigDict(extra="allow")
82
+
83
+ record_id: str
84
+ doi: str
85
+ concept_doi: str | None = None
86
+ ckan_id: str | None = None
87
+ published_at: str | None = None
88
+ sandbox: bool = False
89
+
90
+
91
+ class BundleError(Exception):
92
+ pass
93
+
94
+
95
+ class Bundle:
96
+ def __init__(
97
+ self,
98
+ directory: Path,
99
+ metadata: DatasetMetadata,
100
+ manifests: dict[Path, dict[str, Any]],
101
+ ):
102
+ self.directory = directory
103
+ self.metadata = metadata
104
+ self.manifests = manifests
105
+
106
+ @classmethod
107
+ def load(cls, directory: Path) -> Bundle:
108
+ if not directory.is_dir():
109
+ raise BundleError(f"Not a directory: {directory}")
110
+ metadata_path = find_dataset_metadata(directory)
111
+ if metadata_path is None:
112
+ raise BundleError(
113
+ f"No dataset.yaml in {directory} — the bundle needs "
114
+ "campaign-level metadata (title, creators, license...)"
115
+ )
116
+ raw = yaml.safe_load(metadata_path.read_text(encoding="utf-8")) or {}
117
+ metadata = DatasetMetadata(**raw)
118
+ manifests = {}
119
+ try:
120
+ for manifest_path in find_manifests(directory):
121
+ manifests[manifest_path] = load_manifest(manifest_path)
122
+ except ManifestError as exc:
123
+ raise BundleError(str(exc)) from exc
124
+ if not manifests:
125
+ raise BundleError(f"No *.manifest.json files in {directory}")
126
+ return cls(directory, metadata, manifests)
127
+
128
+ def upload_files(self) -> list[Path]:
129
+ """Files that belong in the Zenodo record: every manifest, plus all
130
+ zenodo-hosted resources, media, documents, and reference sidecars they
131
+ reference."""
132
+ files: list[Path] = sorted(self.manifests)
133
+ seen = set(files)
134
+ for manifest_path, manifest in sorted(self.manifests.items()):
135
+ base = manifest_path.parent
136
+ for section in ("resources", "media", "documents", "references"):
137
+ for entry in manifest.get(section, []):
138
+ if entry.get("hosted", "zenodo") != "zenodo":
139
+ continue
140
+ local = base / entry["path"]
141
+ if local not in seen:
142
+ files.append(local)
143
+ seen.add(local)
144
+ return files
145
+
146
+ @property
147
+ def state_path(self) -> Path:
148
+ return self.directory / STATE_FILE_NAME
149
+
150
+ def read_state(self) -> PublishState | None:
151
+ if not self.state_path.is_file():
152
+ return None
153
+ return PublishState(**json.loads(self.state_path.read_text(encoding="utf-8")))
154
+
155
+ def write_state(self, state: PublishState) -> None:
156
+ state.published_at = datetime.now(UTC).isoformat()
157
+ self.state_path.write_text(
158
+ json.dumps(state.model_dump(), indent=2) + "\n", encoding="utf-8"
159
+ )