rddac 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rddac/__init__.py +66 -0
- rddac/cli.py +166 -0
- rddac/croissant.py +81 -0
- rddac/h5_tools.py +50 -0
- rddac/pytorch.py +78 -0
- rddac/spec.py +47 -0
- rddac/streaming.py +159 -0
- rddac/visualization.py +266 -0
- rddac-1.0.0.dist-info/METADATA +155 -0
- rddac-1.0.0.dist-info/RECORD +14 -0
- rddac-1.0.0.dist-info/WHEEL +5 -0
- rddac-1.0.0.dist-info/entry_points.txt +2 -0
- rddac-1.0.0.dist-info/licenses/LICENSE +21 -0
- rddac-1.0.0.dist-info/top_level.txt +1 -0
rddac/__init__.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""RDDAC — Real Deep Drawing and Cutting Dataset.
|
|
2
|
+
|
|
3
|
+
Python interface for the RDDAC dataset (experimental measurements of sheet
|
|
4
|
+
metal forming; the physical counterpart to the DDACS simulations). Built on a
|
|
5
|
+
Croissant 1.1 manifest: `rddac.load()` returns an `mlcroissant.Dataset` whose
|
|
6
|
+
`records(view)` streams the data; `add_view`, `open_h5`, `inspect_h5` are
|
|
7
|
+
convenience helpers around the same manifest.
|
|
8
|
+
|
|
9
|
+
The public surface mirrors the `ddacs` package, so DDACS code ports by
|
|
10
|
+
swapping the import.
|
|
11
|
+
|
|
12
|
+
Examples:
|
|
13
|
+
>>> import rddac
|
|
14
|
+
>>> ds = rddac.load(data_dir="./data")
|
|
15
|
+
>>> for record in rddac.streaming.iter_view("force-curve", data_dir="./data"):
|
|
16
|
+
... ...
|
|
17
|
+
|
|
18
|
+
>>> with rddac.open_h5(42, data_dir="./data") as f:
|
|
19
|
+
... rddac.inspect_h5(f)
|
|
20
|
+
|
|
21
|
+
Note: prefer ``rddac.streaming.iter_view`` (or ``RDDACDataset``) over
|
|
22
|
+
``ds.records(view)`` for the h5-backed views — mlcroissant's own records()
|
|
23
|
+
walks the full multi-GB zips per view and is impractically slow there.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
__version__ = "1.0.0"
|
|
27
|
+
|
|
28
|
+
from . import streaming
|
|
29
|
+
from .croissant import add_view, load
|
|
30
|
+
from .h5_tools import inspect_h5, open_h5
|
|
31
|
+
from .spec import RDDAC_SPEC, DatasetSpec
|
|
32
|
+
from .visualization import (
|
|
33
|
+
plot_force,
|
|
34
|
+
plot_point_cloud,
|
|
35
|
+
plot_scan,
|
|
36
|
+
plot_traverse,
|
|
37
|
+
scan_to_pointcloud,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
try:
|
|
41
|
+
from .pytorch import RDDACDataset
|
|
42
|
+
except ImportError:
|
|
43
|
+
pass
|
|
44
|
+
|
|
45
|
+
__all__ = [
|
|
46
|
+
"__version__",
|
|
47
|
+
# Dataset identity (consumed by the ddacs machinery via spec=)
|
|
48
|
+
"RDDAC_SPEC",
|
|
49
|
+
"DatasetSpec",
|
|
50
|
+
# Croissant entry point + helpers
|
|
51
|
+
"load",
|
|
52
|
+
"add_view",
|
|
53
|
+
# HDF5 helpers
|
|
54
|
+
"open_h5",
|
|
55
|
+
"inspect_h5",
|
|
56
|
+
# Streaming pipeline (offline iteration + numpy export)
|
|
57
|
+
"streaming",
|
|
58
|
+
# PyTorch (optional — only available if torch is installed)
|
|
59
|
+
"RDDACDataset",
|
|
60
|
+
# Visualization
|
|
61
|
+
"plot_scan",
|
|
62
|
+
"plot_point_cloud",
|
|
63
|
+
"plot_force",
|
|
64
|
+
"plot_traverse",
|
|
65
|
+
"scan_to_pointcloud",
|
|
66
|
+
]
|
rddac/cli.py
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""RDDAC dataset CLI — a thin front-end over the ``ddacs`` CLI machinery.
|
|
2
|
+
|
|
3
|
+
Provides commands to view dataset information and download files from the RDDAC
|
|
4
|
+
(Real Deep Drawing and Cutting) dataset hosted on DaRUS. Because RDDAC is the
|
|
5
|
+
experimental counterpart to DDACS, the full download also fetches the matching
|
|
6
|
+
DDACS simulations (skip with --no-sim).
|
|
7
|
+
|
|
8
|
+
The info/download implementation is `ddacs.cli`'s, called with
|
|
9
|
+
``spec=RDDAC_SPEC`` (requires ddacs >= 3.2.1). Only the parser (prog,
|
|
10
|
+
--no-sim) and the simulation leg live here.
|
|
11
|
+
|
|
12
|
+
Usage:
|
|
13
|
+
rddac info # Show dataset info and versions
|
|
14
|
+
rddac download # Real measurements + DDACS simulations
|
|
15
|
+
rddac download --no-sim # Real measurements only (skip simulations)
|
|
16
|
+
rddac download --small # Small sample bundle (quick start)
|
|
17
|
+
rddac download --files a.zip # Download specific files
|
|
18
|
+
rddac download --extract # Also extract zips next to the zip
|
|
19
|
+
rddac download --extract --remove-zip
|
|
20
|
+
rddac download --quiet # No output/progress; implies --yes
|
|
21
|
+
|
|
22
|
+
Zip files are kept by default so they remain readable in place via mlcroissant
|
|
23
|
+
(the Croissant manifest references zip members directly).
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import argparse
|
|
29
|
+
import importlib.util
|
|
30
|
+
import os
|
|
31
|
+
import subprocess
|
|
32
|
+
import sys
|
|
33
|
+
|
|
34
|
+
from ddacs import cli as _ddacs_cli
|
|
35
|
+
from ddacs.cli import _dataset_title # noqa: F401 — identical helper, re-exported for tests
|
|
36
|
+
from rich.panel import Panel
|
|
37
|
+
|
|
38
|
+
from . import __version__
|
|
39
|
+
from .spec import (
|
|
40
|
+
DDACS_DATASET_DOI,
|
|
41
|
+
DDACS_SIM_FILE,
|
|
42
|
+
RDDAC_SPEC,
|
|
43
|
+
SIM_SUBDIR,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
DEFAULT_VERSION = RDDAC_SPEC.default_version
|
|
47
|
+
DEFAULT_DATA_DIR = RDDAC_SPEC.default_data_dir
|
|
48
|
+
SMALL_TEST_FILES = list(RDDAC_SPEC.small_test_files)
|
|
49
|
+
|
|
50
|
+
console = _ddacs_cli.console # shared console: ddacs's --quiet handling applies
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
# ── small helpers kept for tests / tooling (dataset-agnostic) ─────────────────
|
|
54
|
+
def _file_info(file_meta: dict) -> tuple[str, int]:
|
|
55
|
+
"""Original filename + size from a DaRUS file metadata entry."""
|
|
56
|
+
df = file_meta["dataFile"]
|
|
57
|
+
if "originalFileName" in df:
|
|
58
|
+
return df["originalFileName"], df.get("originalFileSize", df["filesize"])
|
|
59
|
+
return df["filename"], df["filesize"]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _matches(file_meta: dict, names: list[str]) -> bool:
|
|
63
|
+
return _file_info(file_meta)[0] in names or file_meta["dataFile"]["filename"] in names
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
# ── commands ──────────────────────────────────────────────────────────────────
|
|
67
|
+
def cmd_info(args: argparse.Namespace) -> None:
|
|
68
|
+
"""Display dataset information and available versions (via ddacs.cli)."""
|
|
69
|
+
_ddacs_cli.cmd_info(args, spec=RDDAC_SPEC)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def cmd_download(args: argparse.Namespace) -> None:
|
|
73
|
+
"""Download RDDAC measurements (and, by default, the DDACS simulations)."""
|
|
74
|
+
_ddacs_cli.cmd_download(args, spec=RDDAC_SPEC)
|
|
75
|
+
|
|
76
|
+
# The DDACS simulations come along only on a full download (not --small /
|
|
77
|
+
# --files), unless explicitly skipped.
|
|
78
|
+
if not args.small and not args.files and not args.no_sim:
|
|
79
|
+
_download_simulations(args)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _download_simulations(args: argparse.Namespace) -> int:
|
|
83
|
+
"""Fetch the matching DDACS simulations by delegating to the `ddacs` CLI.
|
|
84
|
+
|
|
85
|
+
The download machinery is not duplicated here: if the `ddacs` package is
|
|
86
|
+
installed, its own CLI downloads `rddac.zip` into ``<out>/simulation``;
|
|
87
|
+
otherwise the user gets the exact command to run after installing it.
|
|
88
|
+
|
|
89
|
+
Returns the number of files fetched (0 when skipped or delegated-and-failed).
|
|
90
|
+
"""
|
|
91
|
+
sim_dir = os.path.join(args.out, SIM_SUBDIR)
|
|
92
|
+
console.print()
|
|
93
|
+
console.print(Panel(
|
|
94
|
+
f"[bold]Source:[/bold] DDACS {DDACS_DATASET_DOI}\n[bold]File:[/bold] {DDACS_SIM_FILE}\n"
|
|
95
|
+
f"[bold]Destination:[/bold] {os.path.abspath(sim_dir)}\n"
|
|
96
|
+
"[dim]The matching FEM simulations. Skip with --no-sim.[/dim]",
|
|
97
|
+
title="DDACS simulation reference data", border_style="cyan"))
|
|
98
|
+
|
|
99
|
+
if importlib.util.find_spec("ddacs") is None:
|
|
100
|
+
console.print(
|
|
101
|
+
"[yellow]The `ddacs` package is not installed — skipping the simulations.[/yellow]\n"
|
|
102
|
+
"To fetch them later:\n"
|
|
103
|
+
" [bold]pip install ddacs[/bold]\n"
|
|
104
|
+
f" [bold]ddacs download --files {DDACS_SIM_FILE} metadata.json process_parameters.csv --out {sim_dir} -y[/bold]"
|
|
105
|
+
)
|
|
106
|
+
return 0
|
|
107
|
+
|
|
108
|
+
# Also fetch DDACS's manifest + parameter table so <out>/simulation is a
|
|
109
|
+
# self-contained DDACS data dir: ddacs.load(data_dir="<out>/simulation")
|
|
110
|
+
# resolves the DDACS manifest locally and cannot pick up RDDAC's
|
|
111
|
+
# metadata.json from the parent directory.
|
|
112
|
+
cmd = [sys.executable, "-m", "ddacs.cli", "download",
|
|
113
|
+
"--files", DDACS_SIM_FILE, "metadata.json", "process_parameters.csv",
|
|
114
|
+
"--out", sim_dir]
|
|
115
|
+
if args.yes:
|
|
116
|
+
cmd.append("-y")
|
|
117
|
+
if getattr(args, "quiet", False):
|
|
118
|
+
cmd.append("--quiet")
|
|
119
|
+
if args.extract:
|
|
120
|
+
cmd.append("--extract")
|
|
121
|
+
if args.remove_zip:
|
|
122
|
+
cmd.append("--remove-zip")
|
|
123
|
+
console.print(f"[dim]delegating to: {' '.join(cmd[2:])}[/dim]")
|
|
124
|
+
result = subprocess.run(cmd)
|
|
125
|
+
if result.returncode != 0:
|
|
126
|
+
console.print("[red]ddacs download failed.[/red]")
|
|
127
|
+
return 0
|
|
128
|
+
return 1
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def main() -> None:
|
|
132
|
+
"""CLI entry point for RDDAC dataset commands."""
|
|
133
|
+
parser = argparse.ArgumentParser(
|
|
134
|
+
prog="rddac", description="RDDAC Dataset CLI - Download experimental data from DaRUS")
|
|
135
|
+
parser.add_argument("-V", "--version", action="version", version=f"%(prog)s {__version__}")
|
|
136
|
+
parser.add_argument("--token", help="DaRUS API token (for draft access)")
|
|
137
|
+
sub = parser.add_subparsers(dest="command", help="Command")
|
|
138
|
+
|
|
139
|
+
sub.add_parser("info", help="Show dataset info and versions")
|
|
140
|
+
|
|
141
|
+
dl = sub.add_parser("download", help="Download dataset files")
|
|
142
|
+
dl.add_argument("version", nargs="?", default=DEFAULT_VERSION,
|
|
143
|
+
help=f"Dataset version (default: {DEFAULT_VERSION})")
|
|
144
|
+
dl.add_argument("--files", nargs="+", help="Specific filenames to download")
|
|
145
|
+
dl.add_argument("--small", action="store_true", help="Download the small sample bundle")
|
|
146
|
+
dl.add_argument("--no-sim", action="store_true",
|
|
147
|
+
help="Download only the real measurements (skip the DDACS simulations)")
|
|
148
|
+
dl.add_argument("--out", default=DEFAULT_DATA_DIR,
|
|
149
|
+
help=f"Output directory (default: {DEFAULT_DATA_DIR})")
|
|
150
|
+
dl.add_argument("-y", "--yes", action="store_true", help="Skip confirmation prompt")
|
|
151
|
+
dl.add_argument("-q", "--quiet", action="store_true",
|
|
152
|
+
help="No output or progress bars; implies --yes")
|
|
153
|
+
dl.add_argument("--extract", action="store_true", help="Extract downloaded zips into their directory")
|
|
154
|
+
dl.add_argument("--remove-zip", action="store_true", help="Delete zips after extraction (with --extract)")
|
|
155
|
+
|
|
156
|
+
args = parser.parse_args()
|
|
157
|
+
if args.command == "info":
|
|
158
|
+
cmd_info(args)
|
|
159
|
+
elif args.command == "download":
|
|
160
|
+
cmd_download(args)
|
|
161
|
+
else:
|
|
162
|
+
parser.print_help()
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
if __name__ == "__main__":
|
|
166
|
+
main()
|
rddac/croissant.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Croissant manifest access for RDDAC — thin wrappers over :mod:`ddacs.croissant`.
|
|
2
|
+
|
|
3
|
+
The machinery lives in the ``ddacs`` package (shared with the DDACS simulation
|
|
4
|
+
dataset); everything here just injects :data:`rddac.spec.RDDAC_SPEC` so the
|
|
5
|
+
manifest resolution targets the RDDAC dataset. The manifest conventions
|
|
6
|
+
(``field-map`` RecordSet, ``process-parameters`` join on ``index``) are
|
|
7
|
+
identical between the datasets, so ``add_view`` & friends pass through as-is.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
import mlcroissant as mlc
|
|
16
|
+
from ddacs import croissant as _ddacs_croissant
|
|
17
|
+
from ddacs.croissant import ( # noqa: F401 — shared, dataset-agnostic API
|
|
18
|
+
FieldSpec,
|
|
19
|
+
TimestepSpec,
|
|
20
|
+
add_view,
|
|
21
|
+
dataset_name,
|
|
22
|
+
field_map,
|
|
23
|
+
process_parameters_descriptions,
|
|
24
|
+
)
|
|
25
|
+
from ddacs.croissant import ( # noqa: F401 — internals used by tests/tools
|
|
26
|
+
_build_mapping,
|
|
27
|
+
_load_jsonld_dict,
|
|
28
|
+
_lookup_data_type,
|
|
29
|
+
_normalize_field_spec,
|
|
30
|
+
_record_set,
|
|
31
|
+
_resolve_field_id,
|
|
32
|
+
_slicing_to_jsonpath,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
from .spec import RDDAC_SPEC
|
|
36
|
+
|
|
37
|
+
DEFAULT_DATA_DIR = RDDAC_SPEC.default_data_dir
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def metadata_url() -> str:
|
|
41
|
+
"""Return the DaRUS download URL for the published RDDAC ``metadata.json``.
|
|
42
|
+
|
|
43
|
+
Resolved via the DaRUS API (numeric file id — DaRUS has no per-file
|
|
44
|
+
persistent ids) and cached for the process lifetime.
|
|
45
|
+
"""
|
|
46
|
+
return _ddacs_croissant.metadata_url(RDDAC_SPEC)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def __getattr__(name: str) -> Any:
|
|
50
|
+
# Lazy module attribute (PEP 562): METADATA_URL mirrors the ddacs API.
|
|
51
|
+
if name == "METADATA_URL":
|
|
52
|
+
return metadata_url()
|
|
53
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def resolve_source(source: str | Path | None = None, data_dir: str | Path | None = None) -> str:
|
|
57
|
+
"""Return the source string (path or URL) that :func:`load` would use.
|
|
58
|
+
|
|
59
|
+
Resolution order:
|
|
60
|
+
1. ``source`` if given (local path or HTTP(S) URL).
|
|
61
|
+
2. ``<data_dir>/metadata.json`` if it exists locally.
|
|
62
|
+
3. The DaRUS download URL from :func:`metadata_url`.
|
|
63
|
+
"""
|
|
64
|
+
return _ddacs_croissant.resolve_source(source, data_dir, spec=RDDAC_SPEC)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def load(
|
|
68
|
+
source: str | Path | None = None,
|
|
69
|
+
data_dir: str | Path | None = DEFAULT_DATA_DIR,
|
|
70
|
+
) -> mlc.Dataset:
|
|
71
|
+
"""Return an :class:`mlcroissant.Dataset` for the RDDAC manifest.
|
|
72
|
+
|
|
73
|
+
Local-first, URL-fallback resolution — see :func:`resolve_source`. When
|
|
74
|
+
``data_dir`` points at a directory that contains files referenced by the
|
|
75
|
+
manifest (e.g. zips written by ``rddac download``), `mlcroissant` is told
|
|
76
|
+
to use those local copies instead of refetching from DaRUS.
|
|
77
|
+
|
|
78
|
+
Pass ``data_dir=None`` to opt out of local-file discovery and force
|
|
79
|
+
`mlcroissant` to download via its own cache.
|
|
80
|
+
"""
|
|
81
|
+
return _ddacs_croissant.load(source, data_dir, spec=RDDAC_SPEC)
|
rddac/h5_tools.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""HDF5 access helpers for RDDAC — thin wrappers over :mod:`ddacs.h5_tools`.
|
|
2
|
+
|
|
3
|
+
`open_h5` resolves the RDDAC manifest, locates the zip member matching the
|
|
4
|
+
requested experiment (zero-padded: ``42`` -> ``0042.h5``) and returns an
|
|
5
|
+
`h5py.File`. `inspect_h5` pretty-prints any `h5py.File` or path.
|
|
6
|
+
|
|
7
|
+
Both are re-exported as `rddac.open_h5` and `rddac.inspect_h5`.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
import h5py
|
|
15
|
+
from ddacs import h5_tools as _ddacs_h5_tools
|
|
16
|
+
from ddacs.h5_tools import inspect_h5 # noqa: F401 — dataset-agnostic
|
|
17
|
+
|
|
18
|
+
from .spec import RDDAC_SPEC
|
|
19
|
+
|
|
20
|
+
DEFAULT_DATA_DIR = RDDAC_SPEC.default_data_dir
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def open_h5(
|
|
24
|
+
experiment_id: int,
|
|
25
|
+
source: str | Path | None = None,
|
|
26
|
+
data_dir: str | Path | None = DEFAULT_DATA_DIR,
|
|
27
|
+
dataset=None,
|
|
28
|
+
) -> h5py.File:
|
|
29
|
+
"""Return an `h5py.File` for the requested RDDAC experiment.
|
|
30
|
+
|
|
31
|
+
Looks the manifest up, walks the locally mapped zips and reads the h5
|
|
32
|
+
member matching the zero-padded ``<experiment_id>.h5`` (e.g. ``0042.h5``)
|
|
33
|
+
into a `BytesIO`. The returned object is read-only, supports the `with`
|
|
34
|
+
idiom and can be indexed like any other `h5py.File`.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
experiment_id: The experiment index (matches the h5 filename inside
|
|
38
|
+
the zip; ``42`` -> ``0042.h5``).
|
|
39
|
+
source: Override the Croissant manifest URL / path.
|
|
40
|
+
data_dir: Directory searched for already-downloaded zips. Pass `None`
|
|
41
|
+
to skip the local lookup entirely.
|
|
42
|
+
dataset: A pre-loaded `mlcroissant.Dataset` (e.g. from `rddac.load`).
|
|
43
|
+
When given, `source` and `data_dir` are ignored.
|
|
44
|
+
|
|
45
|
+
Raises:
|
|
46
|
+
FileNotFoundError: No locally mapped zip contained the requested h5.
|
|
47
|
+
"""
|
|
48
|
+
return _ddacs_h5_tools.open_h5(
|
|
49
|
+
experiment_id, source=source, data_dir=data_dir, dataset=dataset, spec=RDDAC_SPEC
|
|
50
|
+
)
|
rddac/pytorch.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""PyTorch IterableDataset adapter for RDDAC — a thin subclass of :class:`ddacs.pytorch.DDACSDataset`.
|
|
2
|
+
|
|
3
|
+
All machinery (view/field resolution, zip streaming, DataLoader-worker and DDP
|
|
4
|
+
sharding, seeded shuffle, `MissingDataWarning`) is inherited; the subclass only
|
|
5
|
+
injects :data:`rddac.spec.RDDAC_SPEC` so member names are zero-padded and the
|
|
6
|
+
manifest resolves to the RDDAC dataset.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
try:
|
|
17
|
+
from ddacs.pytorch import DDACSDataset
|
|
18
|
+
except ImportError as exc:
|
|
19
|
+
raise ImportError(
|
|
20
|
+
"PyTorch is required for RDDACDataset. Install with `pip install rddac[torch]` "
|
|
21
|
+
"or install a flavour from https://pytorch.org/get-started/locally/."
|
|
22
|
+
) from exc
|
|
23
|
+
|
|
24
|
+
from .spec import RDDAC_SPEC
|
|
25
|
+
|
|
26
|
+
DEFAULT_DATA_DIR = RDDAC_SPEC.default_data_dir
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class RDDACDataset(DDACSDataset):
|
|
30
|
+
"""Streaming PyTorch dataset for a single RDDAC Croissant view.
|
|
31
|
+
|
|
32
|
+
Yields a `dict[str, numpy.ndarray]` per experiment. Field selection is
|
|
33
|
+
derived from the Croissant view + field-map; sharding across DataLoader
|
|
34
|
+
workers and DDP ranks is decided inside `__iter__`, so the same instance
|
|
35
|
+
works under `num_workers=0`, `num_workers=N` and DDP.
|
|
36
|
+
|
|
37
|
+
Views must source `field-map` (HDF5) fields only; for views that include
|
|
38
|
+
`process-parameters` metadata columns use `rddac.streaming.iter_view`, or
|
|
39
|
+
build the view with `with_metadata=False`.
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
view: Name of the RecordSet to stream (e.g. "force-curve").
|
|
43
|
+
source: Override the manifest URL / path.
|
|
44
|
+
data_dir: Local data directory (default "./data"). Pass `None` to skip
|
|
45
|
+
local-file discovery.
|
|
46
|
+
dataset: A pre-loaded `mlcroissant.Dataset` (e.g. one mutated by
|
|
47
|
+
`rddac.add_view`) — the way to stream a custom view.
|
|
48
|
+
sim_ids: Explicit allowlist of experiment ids (name kept for drop-in
|
|
49
|
+
DDACS compatibility). Requested ids that cannot be served warn via
|
|
50
|
+
`rddac.streaming.MissingDataWarning`.
|
|
51
|
+
where: Predicate applied to each `process_parameters.csv` row before
|
|
52
|
+
any zip is opened.
|
|
53
|
+
shuffle: Per-shard seeded shuffle; call `set_epoch` between epochs.
|
|
54
|
+
seed: Base seed for the per-shard shuffle.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
def __init__(
|
|
58
|
+
self,
|
|
59
|
+
view: str,
|
|
60
|
+
source: str | Path | None = None,
|
|
61
|
+
data_dir: str | Path | None = DEFAULT_DATA_DIR,
|
|
62
|
+
dataset=None,
|
|
63
|
+
sim_ids: list[int] | None = None,
|
|
64
|
+
where: Callable[[pd.Series], bool] | None = None,
|
|
65
|
+
shuffle: bool = False,
|
|
66
|
+
seed: int = 0,
|
|
67
|
+
):
|
|
68
|
+
super().__init__(
|
|
69
|
+
view,
|
|
70
|
+
source=source,
|
|
71
|
+
data_dir=data_dir,
|
|
72
|
+
dataset=dataset,
|
|
73
|
+
sim_ids=sim_ids,
|
|
74
|
+
where=where,
|
|
75
|
+
shuffle=shuffle,
|
|
76
|
+
seed=seed,
|
|
77
|
+
spec=RDDAC_SPEC,
|
|
78
|
+
)
|
rddac/spec.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""RDDAC dataset specification — single source of truth for the dataset identity.
|
|
2
|
+
|
|
3
|
+
The identity lives in :data:`RDDAC_SPEC` (a :class:`ddacs.spec.DatasetSpec`);
|
|
4
|
+
the machinery in the ``ddacs`` package consumes it via the ``spec=`` keyword —
|
|
5
|
+
see the thin wrappers in :mod:`rddac.croissant`, :mod:`rddac.h5_tools`,
|
|
6
|
+
:mod:`rddac.streaming` and :mod:`rddac.pytorch`. Consuming modules derive any
|
|
7
|
+
module-level constants they need directly from the spec.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from ddacs.spec import DatasetSpec
|
|
11
|
+
|
|
12
|
+
__all__ = ["DatasetSpec", "RDDAC_SPEC"]
|
|
13
|
+
|
|
14
|
+
# ── RDDAC dataset identity ────────────────────────────────────────────────────
|
|
15
|
+
RDDAC_SPEC = DatasetSpec(
|
|
16
|
+
name="RDDAC",
|
|
17
|
+
prog="rddac",
|
|
18
|
+
dataset_doi="doi:10.18419/DARUS-5589",
|
|
19
|
+
default_version="1.0",
|
|
20
|
+
# Experiment ids are zero-padded in the HDF5 member names: 42 -> "0042.h5".
|
|
21
|
+
id_format="{:04d}",
|
|
22
|
+
small_test_files=(
|
|
23
|
+
"process_parameters.csv",
|
|
24
|
+
"metadata.json",
|
|
25
|
+
"sample.zip",
|
|
26
|
+
),
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
# ── Hyperlinks (kept here so a URL change touches one file) ───────────────────
|
|
30
|
+
DATASET_URL = (
|
|
31
|
+
f"{RDDAC_SPEC.darus_base_url}/dataset.xhtml?persistentId={RDDAC_SPEC.dataset_doi}"
|
|
32
|
+
)
|
|
33
|
+
DOI_URL = f"https://doi.org/{RDDAC_SPEC.dataset_doi.replace('doi:', '')}"
|
|
34
|
+
GITHUB_URL = "https://github.com/BaumSebastian/RDDAC"
|
|
35
|
+
DOCS_URL = "https://rddac.readthedocs.io"
|
|
36
|
+
|
|
37
|
+
# ── DDACS simulation reference data (RDDAC-specific, not spec material) ───────
|
|
38
|
+
# RDDAC is the experimental counterpart to DDACS; `rddac download` fetches the
|
|
39
|
+
# matching FEM simulations alongside the measurements (skip with --no-sim) by
|
|
40
|
+
# delegating to the installed `ddacs` CLI. They are the RDDAC sub-study subset
|
|
41
|
+
# published in the DDACS dataset as a single zip.
|
|
42
|
+
DDACS_DATASET_DOI = "doi:10.18419/DARUS-4801"
|
|
43
|
+
DDACS_DATASET_URL = (
|
|
44
|
+
f"{RDDAC_SPEC.darus_base_url}/dataset.xhtml?persistentId={DDACS_DATASET_DOI}"
|
|
45
|
+
)
|
|
46
|
+
DDACS_SIM_FILE = "rddac.zip"
|
|
47
|
+
SIM_SUBDIR = "simulation" # local subdirectory the simulations download into
|
rddac/streaming.py
ADDED
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Streaming iteration and numpy export for RDDAC — thin wrappers over :mod:`ddacs.streaming`.
|
|
2
|
+
|
|
3
|
+
`iter_view` is the plain-Python counterpart to `RDDACDataset.__iter__`: it
|
|
4
|
+
yields one ``dict[str, numpy.ndarray]`` per experiment with no torch
|
|
5
|
+
dependency. `export_to_numpy` materializes a view as flat ``.npy`` memmaps
|
|
6
|
+
(fixed shapes); `export_to_numpy_per_sim` writes one ``.npz`` per experiment
|
|
7
|
+
(ragged shapes, e.g. the raw force tables); `load_export` reads an export back.
|
|
8
|
+
|
|
9
|
+
The ``sim_ids=`` keyword names are kept identical to ``ddacs`` so DDACS code
|
|
10
|
+
ports by swapping the import. Requested ids that cannot be served locally
|
|
11
|
+
raise a suppressible :class:`MissingDataWarning`.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from collections.abc import Callable, Iterator
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
from ddacs import streaming as _ddacs_streaming
|
|
23
|
+
from ddacs.streaming import ( # noqa: F401 — dataset-agnostic API + internals
|
|
24
|
+
MissingDataWarning,
|
|
25
|
+
_LoadedExport,
|
|
26
|
+
_apply_transforms,
|
|
27
|
+
_as_array,
|
|
28
|
+
_build_field_specs,
|
|
29
|
+
_build_unified_index,
|
|
30
|
+
_extract_record,
|
|
31
|
+
_parse_jsonpath,
|
|
32
|
+
_progress_iter,
|
|
33
|
+
_resolve_sim_ids,
|
|
34
|
+
_warn_missing,
|
|
35
|
+
load_export,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
from .spec import RDDAC_SPEC
|
|
39
|
+
|
|
40
|
+
DEFAULT_DATA_DIR = RDDAC_SPEC.default_data_dir
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"iter_view",
|
|
44
|
+
"export_to_numpy",
|
|
45
|
+
"export_to_numpy_per_sim",
|
|
46
|
+
"load_export",
|
|
47
|
+
"MissingDataWarning",
|
|
48
|
+
]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def iter_view(
|
|
52
|
+
view: str,
|
|
53
|
+
*,
|
|
54
|
+
source: str | Path | None = None,
|
|
55
|
+
data_dir: str | Path | None = DEFAULT_DATA_DIR,
|
|
56
|
+
dataset=None,
|
|
57
|
+
sim_ids: list[int] | None = None,
|
|
58
|
+
where: Callable[[pd.Series], bool] | None = None,
|
|
59
|
+
) -> Iterator[dict[str, np.ndarray]]:
|
|
60
|
+
"""Yield one record per RDDAC experiment for a Croissant view.
|
|
61
|
+
|
|
62
|
+
Args:
|
|
63
|
+
view: Name of the RecordSet to stream (published, e.g. ``force-curve``,
|
|
64
|
+
or added via :func:`rddac.add_view`).
|
|
65
|
+
source: Override the Croissant manifest URL/path.
|
|
66
|
+
data_dir: Directory holding ``metadata.json``, ``process_parameters.csv``
|
|
67
|
+
and either loose ``h5/<id>.h5`` files or the dataset zips.
|
|
68
|
+
dataset: A pre-loaded ``mlcroissant.Dataset`` (carries ``add_view``
|
|
69
|
+
mutations).
|
|
70
|
+
sim_ids: Optional allowlist of experiment ids (name kept for drop-in
|
|
71
|
+
DDACS compatibility). Requested ids that cannot be served warn via
|
|
72
|
+
:class:`MissingDataWarning`.
|
|
73
|
+
where: Predicate applied to each ``process_parameters.csv`` row before
|
|
74
|
+
any HDF5 file is touched.
|
|
75
|
+
|
|
76
|
+
Yields:
|
|
77
|
+
A ``dict[str, np.ndarray]`` per experiment, keyed by view-field aliases
|
|
78
|
+
(plus the private ``_sim_id`` scratch key).
|
|
79
|
+
"""
|
|
80
|
+
return _ddacs_streaming.iter_view(
|
|
81
|
+
view,
|
|
82
|
+
source=source,
|
|
83
|
+
data_dir=data_dir,
|
|
84
|
+
dataset=dataset,
|
|
85
|
+
sim_ids=sim_ids,
|
|
86
|
+
where=where,
|
|
87
|
+
spec=RDDAC_SPEC,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def export_to_numpy(
|
|
92
|
+
view: str,
|
|
93
|
+
out_dir: str | Path,
|
|
94
|
+
*,
|
|
95
|
+
source: str | Path | None = None,
|
|
96
|
+
data_dir: str | Path | None = DEFAULT_DATA_DIR,
|
|
97
|
+
dataset=None,
|
|
98
|
+
sim_ids: list[int] | None = None,
|
|
99
|
+
where: Callable[[pd.Series], bool] | None = None,
|
|
100
|
+
transforms: dict[str, Callable[[Any], Any]] | None = None,
|
|
101
|
+
record_transform: Callable[[dict[str, Any]], dict[str, Any]] | None = None,
|
|
102
|
+
show_progress: bool = False,
|
|
103
|
+
) -> dict[str, Path]:
|
|
104
|
+
"""Materialize a Croissant view as flat ``.npy`` memmaps on disk.
|
|
105
|
+
|
|
106
|
+
Requires every record to share the same shape per field — raw RDDAC force
|
|
107
|
+
and traverse tables vary per experiment, so either slice/resample them via
|
|
108
|
+
``record_transform`` or use :func:`export_to_numpy_per_sim`. See
|
|
109
|
+
:func:`ddacs.streaming.export_to_numpy` for the full parameter reference.
|
|
110
|
+
"""
|
|
111
|
+
return _ddacs_streaming.export_to_numpy(
|
|
112
|
+
view,
|
|
113
|
+
out_dir,
|
|
114
|
+
source=source,
|
|
115
|
+
data_dir=data_dir,
|
|
116
|
+
dataset=dataset,
|
|
117
|
+
sim_ids=sim_ids,
|
|
118
|
+
where=where,
|
|
119
|
+
transforms=transforms,
|
|
120
|
+
record_transform=record_transform,
|
|
121
|
+
show_progress=show_progress,
|
|
122
|
+
spec=RDDAC_SPEC,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def export_to_numpy_per_sim(
|
|
127
|
+
view: str,
|
|
128
|
+
out_dir: str | Path,
|
|
129
|
+
*,
|
|
130
|
+
source: str | Path | None = None,
|
|
131
|
+
data_dir: str | Path | None = DEFAULT_DATA_DIR,
|
|
132
|
+
dataset=None,
|
|
133
|
+
sim_ids: list[int] | None = None,
|
|
134
|
+
where: Callable[[pd.Series], bool] | None = None,
|
|
135
|
+
transforms: dict[str, Callable[[Any], Any]] | None = None,
|
|
136
|
+
record_transform: Callable[[dict[str, Any]], dict[str, Any]] | None = None,
|
|
137
|
+
compressed: bool = False,
|
|
138
|
+
show_progress: bool = False,
|
|
139
|
+
) -> Path:
|
|
140
|
+
"""Write one ``<experiment_id>.npz`` per experiment under ``out_dir``.
|
|
141
|
+
|
|
142
|
+
Same pipeline as :func:`export_to_numpy` but fields may have
|
|
143
|
+
experiment-dependent shapes (the natural fit for RDDAC's raw tables). See
|
|
144
|
+
:func:`ddacs.streaming.export_to_numpy_per_sim` for details.
|
|
145
|
+
"""
|
|
146
|
+
return _ddacs_streaming.export_to_numpy_per_sim(
|
|
147
|
+
view,
|
|
148
|
+
out_dir,
|
|
149
|
+
source=source,
|
|
150
|
+
data_dir=data_dir,
|
|
151
|
+
dataset=dataset,
|
|
152
|
+
sim_ids=sim_ids,
|
|
153
|
+
where=where,
|
|
154
|
+
transforms=transforms,
|
|
155
|
+
record_transform=record_transform,
|
|
156
|
+
compressed=compressed,
|
|
157
|
+
show_progress=show_progress,
|
|
158
|
+
spec=RDDAC_SPEC,
|
|
159
|
+
)
|
rddac/visualization.py
ADDED
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
"""Visualization utilities for RDDAC experimental data.
|
|
2
|
+
|
|
3
|
+
Plotting helpers operate on numpy arrays the caller already pulled out of an
|
|
4
|
+
HDF5 file. Pair them with `rddac.open_h5` for end-to-end inspection:
|
|
5
|
+
|
|
6
|
+
>>> import rddac
|
|
7
|
+
>>> with rddac.open_h5(42, data_dir="./data") as f:
|
|
8
|
+
... z = f["pointcloud/op10/z"][:]
|
|
9
|
+
... lumi = f["pointcloud/op10/luminescence"][:]
|
|
10
|
+
... force = f["force/data"][:]
|
|
11
|
+
>>> ax, cbar = rddac.plot_scan(z) # heightmap image
|
|
12
|
+
>>> ax = rddac.plot_force(force) # force time series
|
|
13
|
+
>>> pts = rddac.scan_to_pointcloud(z, lumi) # flat buffer -> (N, 3)
|
|
14
|
+
>>> ax, cbar = rddac.plot_point_cloud(pts, values=pts[:, 2])
|
|
15
|
+
|
|
16
|
+
The raw scans are flattened row-major buffers over a ``y_shape x x_shape``
|
|
17
|
+
(2000 x 3200) pixel grid — see the group attrs in the h5 file. Pixel values
|
|
18
|
+
are sensor units; the calibrated mm conversion is part of the (optional)
|
|
19
|
+
preprocessing step.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import matplotlib.pyplot as plt
|
|
25
|
+
import numpy as np
|
|
26
|
+
|
|
27
|
+
# Default scan grid (Keyence line scanner): see pointcloud group attrs.
|
|
28
|
+
SCAN_X_SHAPE = 3200
|
|
29
|
+
SCAN_Y_SHAPE = 2000
|
|
30
|
+
|
|
31
|
+
# Force table column layout (see the `columns` attr on the h5 `force` group).
|
|
32
|
+
FORCE_COLUMNS = (
|
|
33
|
+
"time", "load_cell_1", "load_cell_2", "load_cell_3", "load_cell_4",
|
|
34
|
+
"punch_temp", "punch_pos", "total_force",
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
DEFAULT_DPI = 150
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _scan_to_2d(buffer: np.ndarray, x_shape: int, y_shape: int) -> np.ndarray:
|
|
41
|
+
buffer = np.asarray(buffer)
|
|
42
|
+
if buffer.ndim == 1:
|
|
43
|
+
return buffer.reshape(y_shape, x_shape)
|
|
44
|
+
if buffer.ndim == 2:
|
|
45
|
+
return buffer
|
|
46
|
+
raise ValueError(f"expected a flat or 2D scan buffer, got shape {buffer.shape}")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def plot_scan(
|
|
50
|
+
z: np.ndarray,
|
|
51
|
+
ax=None,
|
|
52
|
+
figsize: tuple[float, float] | None = None,
|
|
53
|
+
x_shape: int = SCAN_X_SHAPE,
|
|
54
|
+
y_shape: int = SCAN_Y_SHAPE,
|
|
55
|
+
cmap: str = "viridis",
|
|
56
|
+
vmin: float | None = None,
|
|
57
|
+
vmax: float | None = None,
|
|
58
|
+
mask_zero: bool = True,
|
|
59
|
+
colorbar: bool = True,
|
|
60
|
+
title: str | None = None,
|
|
61
|
+
):
|
|
62
|
+
"""Show a raw laser scan buffer (height or luminescence) as a 2D image.
|
|
63
|
+
|
|
64
|
+
Args:
|
|
65
|
+
z: Flat ``(x_shape*y_shape,)`` or already-reshaped ``(y_shape, x_shape)``
|
|
66
|
+
scan buffer, e.g. ``f["pointcloud/op10/z"][:]``.
|
|
67
|
+
ax: Optional existing matplotlib Axes.
|
|
68
|
+
figsize: Figure size when a new figure is created.
|
|
69
|
+
x_shape / y_shape: Grid dimensions (defaults match the group attrs).
|
|
70
|
+
cmap: Matplotlib colormap name.
|
|
71
|
+
vmin / vmax: Color range; defaults to the data range of valid pixels.
|
|
72
|
+
mask_zero: Render 0-valued pixels (no measurement) as transparent.
|
|
73
|
+
colorbar: Attach a colorbar.
|
|
74
|
+
title: Optional axes title.
|
|
75
|
+
|
|
76
|
+
Returns:
|
|
77
|
+
``(ax, colorbar)`` — the colorbar is ``None`` when ``colorbar=False``.
|
|
78
|
+
"""
|
|
79
|
+
img = _scan_to_2d(z, x_shape, y_shape).astype(float)
|
|
80
|
+
if mask_zero:
|
|
81
|
+
img = np.where(img == 0, np.nan, img)
|
|
82
|
+
|
|
83
|
+
if ax is None:
|
|
84
|
+
_, ax = plt.subplots(figsize=figsize or (12, 7.5), dpi=DEFAULT_DPI)
|
|
85
|
+
im = ax.imshow(img, origin="lower", cmap=cmap, vmin=vmin, vmax=vmax, aspect="equal")
|
|
86
|
+
ax.set_xlabel("x in px")
|
|
87
|
+
ax.set_ylabel("y in px")
|
|
88
|
+
if title:
|
|
89
|
+
ax.set_title(title)
|
|
90
|
+
cbar = ax.figure.colorbar(im, ax=ax, shrink=0.8) if colorbar else None
|
|
91
|
+
return ax, cbar
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def scan_to_pointcloud(
|
|
95
|
+
z: np.ndarray,
|
|
96
|
+
luminescence: np.ndarray | None = None,
|
|
97
|
+
x_shape: int = SCAN_X_SHAPE,
|
|
98
|
+
y_shape: int = SCAN_Y_SHAPE,
|
|
99
|
+
drop_invalid: bool = True,
|
|
100
|
+
stride: int = 1,
|
|
101
|
+
) -> np.ndarray:
|
|
102
|
+
"""Convert a flat scan buffer into an ``(N, 3)`` point cloud ``[x, y, z]``.
|
|
103
|
+
|
|
104
|
+
Coordinates are pixel indices and raw z units (uncalibrated); the mm
|
|
105
|
+
conversion, outlier removal and simulation alignment belong to the
|
|
106
|
+
(optional) preprocessing step.
|
|
107
|
+
|
|
108
|
+
Args:
|
|
109
|
+
z: Flat or 2D height buffer.
|
|
110
|
+
luminescence: Optional matching buffer; when given, pixels with zero
|
|
111
|
+
luminescence are treated as invalid too.
|
|
112
|
+
x_shape / y_shape: Grid dimensions.
|
|
113
|
+
drop_invalid: Drop pixels with ``z == 0`` (no measurement).
|
|
114
|
+
stride: Keep every ``stride``-th pixel in both axes (fast previews;
|
|
115
|
+
``stride=4`` reduces 6.4M points to ~400k).
|
|
116
|
+
|
|
117
|
+
Returns:
|
|
118
|
+
``(N, 3)`` float32 array of ``[x_px, y_px, z_raw]``.
|
|
119
|
+
"""
|
|
120
|
+
z2d = _scan_to_2d(z, x_shape, y_shape)[::stride, ::stride]
|
|
121
|
+
valid = np.ones_like(z2d, dtype=bool)
|
|
122
|
+
if drop_invalid:
|
|
123
|
+
valid &= z2d != 0
|
|
124
|
+
if luminescence is not None:
|
|
125
|
+
lumi2d = _scan_to_2d(luminescence, x_shape, y_shape)[::stride, ::stride]
|
|
126
|
+
valid &= lumi2d != 0
|
|
127
|
+
yy, xx = np.nonzero(valid)
|
|
128
|
+
pts = np.column_stack([xx * stride, yy * stride, z2d[yy, xx]]).astype(np.float32)
|
|
129
|
+
return pts
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def plot_point_cloud(
|
|
133
|
+
points: np.ndarray,
|
|
134
|
+
values: np.ndarray | None = None,
|
|
135
|
+
ax=None,
|
|
136
|
+
figsize: tuple[float, float] | None = None,
|
|
137
|
+
cmap: str = "viridis",
|
|
138
|
+
vmin: float | None = None,
|
|
139
|
+
vmax: float | None = None,
|
|
140
|
+
point_size: float = 1.0,
|
|
141
|
+
max_points: int | None = 200_000,
|
|
142
|
+
colorbar: bool = True,
|
|
143
|
+
title: str | None = None,
|
|
144
|
+
):
|
|
145
|
+
"""Scatter an ``(N, 3)`` point cloud in 3D, optionally colored by ``values``.
|
|
146
|
+
|
|
147
|
+
Signature mirrors ``ddacs.plot_point_cloud`` so DDACS code ports by
|
|
148
|
+
swapping the import. Use :func:`scan_to_pointcloud` to build ``points``
|
|
149
|
+
from a raw scan buffer, or plot processed clouds directly.
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
points: ``(N, 3)`` array of xyz positions.
|
|
153
|
+
values: Optional per-point scalars for coloring (default: z).
|
|
154
|
+
ax: Optional existing 3D Axes.
|
|
155
|
+
figsize: Figure size when a new figure is created.
|
|
156
|
+
cmap / vmin / vmax: Color mapping.
|
|
157
|
+
point_size: Scatter marker size.
|
|
158
|
+
max_points: Random-subsample cap (``None`` = plot everything).
|
|
159
|
+
colorbar: Attach a colorbar.
|
|
160
|
+
title: Optional axes title.
|
|
161
|
+
|
|
162
|
+
Returns:
|
|
163
|
+
``(ax, colorbar)`` — the colorbar is ``None`` when ``colorbar=False``.
|
|
164
|
+
"""
|
|
165
|
+
points = np.asarray(points)
|
|
166
|
+
if points.ndim != 2 or points.shape[1] != 3:
|
|
167
|
+
raise ValueError(f"expected (N, 3) points, got {points.shape}")
|
|
168
|
+
if values is None:
|
|
169
|
+
values = points[:, 2]
|
|
170
|
+
values = np.asarray(values)
|
|
171
|
+
|
|
172
|
+
if max_points is not None and len(points) > max_points:
|
|
173
|
+
idx = np.random.default_rng(0).choice(len(points), max_points, replace=False)
|
|
174
|
+
points, values = points[idx], values[idx]
|
|
175
|
+
|
|
176
|
+
if ax is None:
|
|
177
|
+
fig = plt.figure(figsize=figsize or (10, 8), dpi=DEFAULT_DPI)
|
|
178
|
+
ax = fig.add_subplot(projection="3d")
|
|
179
|
+
sc = ax.scatter(points[:, 0], points[:, 1], points[:, 2],
|
|
180
|
+
c=values, cmap=cmap, vmin=vmin, vmax=vmax, s=point_size)
|
|
181
|
+
ax.set_xlabel("x")
|
|
182
|
+
ax.set_ylabel("y")
|
|
183
|
+
ax.set_zlabel("z")
|
|
184
|
+
if title:
|
|
185
|
+
ax.set_title(title)
|
|
186
|
+
cbar = ax.figure.colorbar(sc, ax=ax, shrink=0.6) if colorbar else None
|
|
187
|
+
return ax, cbar
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def plot_force(
|
|
191
|
+
force: np.ndarray,
|
|
192
|
+
columns: tuple[str, ...] = FORCE_COLUMNS,
|
|
193
|
+
signals: tuple[str, ...] = ("load_cell_1", "load_cell_2", "load_cell_3", "load_cell_4", "total_force"),
|
|
194
|
+
ax=None,
|
|
195
|
+
figsize: tuple[float, float] | None = None,
|
|
196
|
+
title: str | None = None,
|
|
197
|
+
):
|
|
198
|
+
"""Plot press force / process signals over time from a ``force/data`` table.
|
|
199
|
+
|
|
200
|
+
Args:
|
|
201
|
+
force: ``(n, 8)`` array as stored under ``force/data``.
|
|
202
|
+
columns: Column layout (defaults to the raw layout; pass the h5 group's
|
|
203
|
+
``columns`` attr when it differs, e.g. for preprocessed data).
|
|
204
|
+
signals: Which columns to draw against ``time``.
|
|
205
|
+
ax: Optional existing Axes.
|
|
206
|
+
figsize: Figure size when a new figure is created.
|
|
207
|
+
title: Optional axes title.
|
|
208
|
+
|
|
209
|
+
Returns:
|
|
210
|
+
The matplotlib Axes.
|
|
211
|
+
"""
|
|
212
|
+
force = np.asarray(force)
|
|
213
|
+
col = {name: i for i, name in enumerate(columns)}
|
|
214
|
+
if "time" not in col:
|
|
215
|
+
raise ValueError(f"columns must contain 'time', got {columns}")
|
|
216
|
+
if ax is None:
|
|
217
|
+
_, ax = plt.subplots(figsize=figsize or (10, 5), dpi=DEFAULT_DPI)
|
|
218
|
+
t = force[:, col["time"]]
|
|
219
|
+
for name in signals:
|
|
220
|
+
if name not in col:
|
|
221
|
+
raise ValueError(f"unknown signal {name!r}; available: {sorted(col)}")
|
|
222
|
+
ax.plot(t, force[:, col[name]], label=name, linewidth=1)
|
|
223
|
+
ax.set_xlabel("time in s")
|
|
224
|
+
ax.set_ylabel("force in kN")
|
|
225
|
+
ax.legend(loc="best", fontsize="small")
|
|
226
|
+
ax.grid(True, alpha=0.3)
|
|
227
|
+
if title:
|
|
228
|
+
ax.set_title(title)
|
|
229
|
+
return ax
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def plot_traverse(
|
|
233
|
+
data: np.ndarray,
|
|
234
|
+
ax=None,
|
|
235
|
+
figsize: tuple[float, float] | None = None,
|
|
236
|
+
ylabel: str = "value",
|
|
237
|
+
label: str | None = None,
|
|
238
|
+
title: str | None = None,
|
|
239
|
+
):
|
|
240
|
+
"""Plot a sensor traverse table (``sheet_thickness/data`` or ``oil_thickness/data``).
|
|
241
|
+
|
|
242
|
+
Args:
|
|
243
|
+
data: ``(n, 2)`` array of ``[sensor_position, value]``.
|
|
244
|
+
ax: Optional existing Axes.
|
|
245
|
+
figsize: Figure size when a new figure is created.
|
|
246
|
+
ylabel: Y axis label (e.g. ``"sheet thickness in um"``, ``"oil in g/m^2"``).
|
|
247
|
+
label: Optional line label (for overlaying multiple traverses).
|
|
248
|
+
title: Optional axes title.
|
|
249
|
+
|
|
250
|
+
Returns:
|
|
251
|
+
The matplotlib Axes.
|
|
252
|
+
"""
|
|
253
|
+
data = np.asarray(data)
|
|
254
|
+
if data.ndim != 2 or data.shape[1] != 2:
|
|
255
|
+
raise ValueError(f"expected (n, 2) traverse, got {data.shape}")
|
|
256
|
+
if ax is None:
|
|
257
|
+
_, ax = plt.subplots(figsize=figsize or (10, 4), dpi=DEFAULT_DPI)
|
|
258
|
+
ax.plot(data[:, 0], data[:, 1], linewidth=1, label=label)
|
|
259
|
+
ax.set_xlabel("sensor position in mm")
|
|
260
|
+
ax.set_ylabel(ylabel)
|
|
261
|
+
ax.grid(True, alpha=0.3)
|
|
262
|
+
if label:
|
|
263
|
+
ax.legend(loc="best", fontsize="small")
|
|
264
|
+
if title:
|
|
265
|
+
ax.set_title(title)
|
|
266
|
+
return ax
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: rddac
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Python package for the Real Deep Drawing and Cutting (RDDAC) Dataset
|
|
5
|
+
Author: Sebastian Baum, Pascal Heinzelmann
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/BaumSebastian/RDDAC
|
|
8
|
+
Project-URL: Documentation, https://rddac.readthedocs.io
|
|
9
|
+
Project-URL: Repository, https://github.com/BaumSebastian/RDDAC
|
|
10
|
+
Keywords: deep-drawing,sheet-metal,forming,dataset,experimental,measurement
|
|
11
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: numpy>=1.24.0
|
|
23
|
+
Requires-Dist: pandas>=2.0.0
|
|
24
|
+
Requires-Dist: h5py>=3.8.0
|
|
25
|
+
Requires-Dist: matplotlib>=3.7.0
|
|
26
|
+
Requires-Dist: requests>=2.28.0
|
|
27
|
+
Requires-Dist: humanfriendly>=10.0
|
|
28
|
+
Requires-Dist: rich>=13.0.0
|
|
29
|
+
Requires-Dist: mlcroissant>=1.1.0
|
|
30
|
+
Requires-Dist: ddacs<4,>=3.2.1
|
|
31
|
+
Provides-Extra: torch
|
|
32
|
+
Requires-Dist: torch>=2.0; extra == "torch"
|
|
33
|
+
Provides-Extra: dev
|
|
34
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
35
|
+
Requires-Dist: black>=24.0.0; extra == "dev"
|
|
36
|
+
Requires-Dist: ruff>=0.3.0; extra == "dev"
|
|
37
|
+
Requires-Dist: bumpver>=2023.1129; extra == "dev"
|
|
38
|
+
Requires-Dist: pre-commit>=3.6.0; extra == "dev"
|
|
39
|
+
Provides-Extra: docs
|
|
40
|
+
Requires-Dist: mkdocs>=1.5.0; extra == "docs"
|
|
41
|
+
Requires-Dist: mkdocs-material>=9.5.0; extra == "docs"
|
|
42
|
+
Requires-Dist: mkdocstrings[python]>=0.24.0; extra == "docs"
|
|
43
|
+
Requires-Dist: mkdocs-macros-plugin>=1.0.0; extra == "docs"
|
|
44
|
+
Dynamic: license-file
|
|
45
|
+
|
|
46
|
+
<div align="center">
|
|
47
|
+
<img src="https://raw.githubusercontent.com/BaumSebastian/RDDAC/main/docs/images/icon/icon.png" width="150"/>
|
|
48
|
+
<h1>Real Deep Drawing and Cutting (RDDAC) Dataset</h1>
|
|
49
|
+
</div>
|
|
50
|
+
|
|
51
|
+
[](LICENSE)
|
|
52
|
+
[](https://creativecommons.org/licenses/by/4.0/)
|
|
53
|
+
[](https://www.python.org/downloads/)
|
|
54
|
+
[](https://rddac.readthedocs.io)
|
|
55
|
+
[](https://darus.uni-stuttgart.de/dataset.xhtml?persistentId=doi:10.18419/DARUS-5589)
|
|
56
|
+
[](https://doi.org/10.18419/DARUS-5589)
|
|
57
|
+
|
|
58
|
+
<div align="center">
|
|
59
|
+
|
|
60
|
+

|
|
61
|
+
|
|
62
|
+
*Measured point clouds of one experiment after deep drawing (OP10, left) and cutting (OP20, right), colored by the deviation from the matching DDACS simulation.*
|
|
63
|
+
|
|
64
|
+
</div>
|
|
65
|
+
|
|
66
|
+
**A large-scale experimental dataset of 9,000 physical deep-drawing and cutting experiments — the real-world counterpart to the [DDACS](https://ddacs.readthedocs.io) FEM simulations.** Each experiment forms a modified quadratic cup from DP600 dual-phase steel (deep drawing in OP10, cutting in OP20) and records press force signals, sheet-thickness and oil-film traverses, and high-resolution 3D laser scans of the part after each operation. Use it to quantify the simulation-to-reality gap, train models on real process data, or validate DDACS-trained surrogates against physical measurements.
|
|
67
|
+
|
|
68
|
+
| | |
|
|
69
|
+
|---|---|
|
|
70
|
+
| **Experiments** | 9,000 |
|
|
71
|
+
| **Total size** | ~87 GB (HDF5, lossless) |
|
|
72
|
+
| **Process steps per experiment** | 2 (OP10 deep drawing, OP20 cutting) |
|
|
73
|
+
| **Parameter space** | 2 geometries x 3 blankholder forces x 3 oil types (18 categories) |
|
|
74
|
+
| **Repetitions** | up to 500 per category |
|
|
75
|
+
| **Train / val / test** | 7,200 / 900 / 900 (predefined, seed 42) |
|
|
76
|
+
| **Matching simulations** | DDACS `rddac.zip` (~9 GB), fetched by `rddac download` |
|
|
77
|
+
|
|
78
|
+
**[Documentation](https://rddac.readthedocs.io)** · **[Dataset DOI](https://doi.org/10.18419/DARUS-5589)** · **[Paper](https://doi.org/10.1007/s12666-026-03870-5)**
|
|
79
|
+
|
|
80
|
+
The `rddac` package ships with the dataset and provides a Croissant native interface: one CLI for the download, one Python module for access, and an optional PyTorch `IterableDataset` for training.
|
|
81
|
+
|
|
82
|
+
## Installation
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install rddac
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
The PyTorch adapter is an optional extra. For hardware specific PyTorch builds (CUDA, ROCm, MPS), install PyTorch first from [pytorch.org](https://pytorch.org/get-started/locally/), then install the extra:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
pip install 'rddac[torch]'
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## Download the dataset
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
# Small sample bundle (~174 MB): manifest, CSV, and one experiment per category.
|
|
98
|
+
rddac download --small -y
|
|
99
|
+
|
|
100
|
+
# Full release (~87 GB), including the matching DDACS simulations (~9 GB).
|
|
101
|
+
rddac download
|
|
102
|
+
|
|
103
|
+
# Real measurements only (skip the simulations).
|
|
104
|
+
rddac download --no-sim
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
## Basic usage
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
import rddac
|
|
111
|
+
|
|
112
|
+
with rddac.open_h5(0) as f: # one experiment by id
|
|
113
|
+
force = f["force/data"][:] # (n, 8): time, load cells, temp, position, total force
|
|
114
|
+
sheet = f["sheet_thickness/data"][:] # (n, 2): sensor position, thickness
|
|
115
|
+
z10 = f["pointcloud/op10/z"][:] # (6400000,) flat scan buffer
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
The public surface mirrors the [`ddacs`](https://ddacs.readthedocs.io) package one to one — `load`, `add_view`, `open_h5`, `inspect_h5`, `streaming.iter_view` / `export_to_numpy` / `load_export`, and the PyTorch `IterableDataset` share names, signatures, and semantics. Code written against DDACS ports by swapping the import:
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
# import ddacs as dataset_pkg # simulations
|
|
122
|
+
import rddac as dataset_pkg # real experiments
|
|
123
|
+
|
|
124
|
+
ds = dataset_pkg.load(data_dir="./data")
|
|
125
|
+
for record in dataset_pkg.streaming.iter_view("force-curve", data_dir="./data", dataset=ds):
|
|
126
|
+
...
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
See the [documentation](https://rddac.readthedocs.io) for the dataset reference (parameter space, HDF5 structure, Croissant manifest) and step-by-step tutorials from a first plot to PyTorch training.
|
|
130
|
+
|
|
131
|
+
## Citation
|
|
132
|
+
|
|
133
|
+
```bibtex
|
|
134
|
+
@dataset{baum2026rddac,
|
|
135
|
+
title={Real Deep Drawing and Cutting Dataset},
|
|
136
|
+
author={Baum, Sebastian and Heinzelmann, Pascal},
|
|
137
|
+
year={2026},
|
|
138
|
+
publisher={DaRUS},
|
|
139
|
+
doi={10.18419/DARUS-5589}
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
@article{baum2026deviation,
|
|
143
|
+
title={Statistical Analysis of Simulation to Reality Deviation in Deep Drawing with a Benchmark Dataset},
|
|
144
|
+
author={Baum, Sebastian and Heinzelmann, Pascal and Clau{\ss}, P. and others},
|
|
145
|
+
journal={Transactions of the Indian Institute of Metals},
|
|
146
|
+
volume={79},
|
|
147
|
+
pages={176},
|
|
148
|
+
year={2026},
|
|
149
|
+
doi={10.1007/s12666-026-03870-5}
|
|
150
|
+
}
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
## License
|
|
154
|
+
|
|
155
|
+
The dataset on DaRUS is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/). The `rddac` software is licensed under the MIT License — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
rddac/__init__.py,sha256=69uu0Bc2Gh9EhZ7Xt69q4J_bBFPGN1-OQtyYcMXUmHs,1898
|
|
2
|
+
rddac/cli.py,sha256=CqKPmy1y2NHAnqarH7OBbpuzb81TW3B0rMhcDu4MWG0,7244
|
|
3
|
+
rddac/croissant.py,sha256=LxQsuFrsQKJSfzYuNUZMxdPRrvEjqj97fdVfA5cbM0E,2845
|
|
4
|
+
rddac/h5_tools.py,sha256=mUrMo72lAeKuz8gMPC1d1CHgC_zT6OqfBUySjggyZCM,1835
|
|
5
|
+
rddac/pytorch.py,sha256=t8MHThEObdzhlH7MZv3c0XbP_mKP82AfrrJMn7cPjN8,2900
|
|
6
|
+
rddac/spec.py,sha256=pNwRuwWr1duHo16807MjdAX7CjxvCKTSDfHagYX92kE,2115
|
|
7
|
+
rddac/streaming.py,sha256=HjmGL1xFttEex1POu5nvnZIW9gdUcWq_2aRd-SIX0zM,5269
|
|
8
|
+
rddac/visualization.py,sha256=y6XXOWZJ6ZQPo_y5eEOzX70UFhfy0DsdFCuTvHzkr7c,9653
|
|
9
|
+
rddac-1.0.0.dist-info/licenses/LICENSE,sha256=htWii6DUxgmJYXwJcRQJvrOrQVQah0dysFFpSE8litg,1071
|
|
10
|
+
rddac-1.0.0.dist-info/METADATA,sha256=CD8-2YflQii72BZqkTLHBLgNE6t0DO7-0Ql73lctiDU,7078
|
|
11
|
+
rddac-1.0.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
12
|
+
rddac-1.0.0.dist-info/entry_points.txt,sha256=uAwbYPNp8KxNg8vfIn6K6cxqORiwzW8kkbwHu-L2AkE,41
|
|
13
|
+
rddac-1.0.0.dist-info/top_level.txt,sha256=a82Qw357Rc8hZLYLMFZ-e6OQ2SQtoV5iRM34c7Ffif8,6
|
|
14
|
+
rddac-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Sebastian Baum
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
rddac
|