vnuli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vnuli/__init__.py +3 -0
- vnuli/adapters/__init__.py +1 -0
- vnuli/adapters/base.py +41 -0
- vnuli/adapters/csv.py +176 -0
- vnuli/adapters/nasa_milling.py +434 -0
- vnuli/analysis/__init__.py +2 -0
- vnuli/analysis/plotting.py +89 -0
- vnuli/analysis/statistics.py +93 -0
- vnuli/benchmarks/__init__.py +31 -0
- vnuli/benchmarks/controlled.py +1075 -0
- vnuli/cli.py +53 -0
- vnuli/comparison/__init__.py +15 -0
- vnuli/comparison/pointwise.py +1027 -0
- vnuli/config.py +177 -0
- vnuli/domain/__init__.py +1 -0
- vnuli/domain/channel.py +104 -0
- vnuli/domain/run.py +98 -0
- vnuli/domain/test.py +52 -0
- vnuli/domain/timeseries.py +109 -0
- vnuli/domain/units.py +40 -0
- vnuli/evidence/__init__.py +1 -0
- vnuli/evidence/provenance.py +141 -0
- vnuli/evidence/result.py +98 -0
- vnuli/evidence/warnings.py +46 -0
- vnuli/interfaces/__init__.py +1 -0
- vnuli/interfaces/mcp/__init__.py +1 -0
- vnuli/interfaces/mcp/server.py +205 -0
- vnuli/interfaces/mcp/toy_server.py +30 -0
- vnuli/measurement/__init__.py +45 -0
- vnuli/measurement/phase11.py +427 -0
- vnuli/measurement/phase12.py +436 -0
- vnuli/model/__init__.py +27 -0
- vnuli/model/query.py +440 -0
- vnuli/reduction/__init__.py +69 -0
- vnuli/reduction/derived_comparison.py +177 -0
- vnuli/reduction/derived_results.py +241 -0
- vnuli/reduction/exact_max_argmax.py +763 -0
- vnuli/reduction/summary_statistics.py +238 -0
- vnuli/reduction/trend_observations.py +219 -0
- vnuli/representation/__init__.py +25 -0
- vnuli/representation/time_history.py +900 -0
- vnuli/representation/time_series_history.py +370 -0
- vnuli/resources/__init__.py +23 -0
- vnuli/resources/model.py +394 -0
- vnuli/runtime.py +42 -0
- vnuli/sample_data/synthetic_test_001.csv +10001 -0
- vnuli/sample_data/synthetic_test_001_metadata.json +186 -0
- vnuli/search/__init__.py +25 -0
- vnuli/search/predicate_search.py +994 -0
- vnuli/search/threshold_events.py +302 -0
- vnuli/selection/__init__.py +11 -0
- vnuli/selection/point_lookup.py +444 -0
- vnuli/service/__init__.py +2 -0
- vnuli/service/vnuli.py +1339 -0
- vnuli/synthetic/__init__.py +61 -0
- vnuli/synthetic/config.py +337 -0
- vnuli/synthetic/export.py +162 -0
- vnuli/synthetic/generator.py +218 -0
- vnuli/version.py +23 -0
- vnuli-0.1.0.dist-info/METADATA +98 -0
- vnuli-0.1.0.dist-info/RECORD +64 -0
- vnuli-0.1.0.dist-info/WHEEL +5 -0
- vnuli-0.1.0.dist-info/entry_points.txt +2 -0
- vnuli-0.1.0.dist-info/top_level.txt +1 -0
vnuli/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""External data source adapter contracts."""
|
vnuli/adapters/base.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Source-independent adapter contract for Vnuli data."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
|
+
from typing import Sequence
|
|
7
|
+
|
|
8
|
+
from vnuli.domain.channel import Channel
|
|
9
|
+
from vnuli.domain.run import Run
|
|
10
|
+
from vnuli.domain.test import Test
|
|
11
|
+
from vnuli.domain.timeseries import TimeSeries
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class DataAdapter(ABC):
|
|
15
|
+
"""Boundary that normalizes external data sources into domain objects."""
|
|
16
|
+
|
|
17
|
+
@abstractmethod
|
|
18
|
+
def list_tests(self) -> Sequence[Test]:
|
|
19
|
+
"""Return available tests as domain objects."""
|
|
20
|
+
|
|
21
|
+
@abstractmethod
|
|
22
|
+
def list_runs(self, test_id: str | None = None) -> Sequence[Run]:
|
|
23
|
+
"""Return available runs, optionally filtered by stable parent test ID."""
|
|
24
|
+
|
|
25
|
+
@abstractmethod
|
|
26
|
+
def list_channels(self, run_id: str) -> Sequence[Channel]:
|
|
27
|
+
"""Return channels available for a stable run ID."""
|
|
28
|
+
|
|
29
|
+
@abstractmethod
|
|
30
|
+
def load_timeseries(
|
|
31
|
+
self,
|
|
32
|
+
run_id: str,
|
|
33
|
+
channel_ids: Sequence[str],
|
|
34
|
+
start: float,
|
|
35
|
+
end: float,
|
|
36
|
+
) -> Sequence[TimeSeries]:
|
|
37
|
+
"""Return requested channel samples where start <= timestamp <= end.
|
|
38
|
+
|
|
39
|
+
Concrete adapters normalize source-specific storage and boundary behavior
|
|
40
|
+
to this inclusive Vnuli contract.
|
|
41
|
+
"""
|
vnuli/adapters/csv.py
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""CSV-backed adapter for the synthetic Prototype 0.1 fixture."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import json
|
|
7
|
+
from math import isfinite
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Mapping, Sequence
|
|
10
|
+
|
|
11
|
+
from vnuli.adapters.base import DataAdapter
|
|
12
|
+
from vnuli.domain.channel import Channel
|
|
13
|
+
from vnuli.domain.run import Run
|
|
14
|
+
from vnuli.domain.test import Test
|
|
15
|
+
from vnuli.domain.timeseries import TimeSeries
|
|
16
|
+
|
|
17
|
+
METADATA_FILENAME = "synthetic_test_001_metadata.json"
|
|
18
|
+
SAMPLES_FILENAME = "synthetic_test_001.csv"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class CSVAdapter(DataAdapter):
|
|
22
|
+
"""Read AI-visible CSV fixture metadata into domain objects."""
|
|
23
|
+
|
|
24
|
+
def __init__(self, sample_data_dir: Path | str = "sample_data") -> None:
|
|
25
|
+
self._sample_data_dir = Path(sample_data_dir)
|
|
26
|
+
self._samples_path = self._sample_data_dir / SAMPLES_FILENAME
|
|
27
|
+
metadata = _load_metadata(self._sample_data_dir / METADATA_FILENAME)
|
|
28
|
+
|
|
29
|
+
self._test = _load_test(metadata)
|
|
30
|
+
self._run = _load_run(metadata)
|
|
31
|
+
self._channels = _load_channels(metadata)
|
|
32
|
+
self._channel_ids = frozenset(channel.id for channel in self._channels)
|
|
33
|
+
|
|
34
|
+
def list_tests(self) -> Sequence[Test]:
|
|
35
|
+
return (self._test,)
|
|
36
|
+
|
|
37
|
+
def list_runs(self, test_id: str | None = None) -> Sequence[Run]:
|
|
38
|
+
if test_id is None or test_id == self._run.test_id:
|
|
39
|
+
return (self._run,)
|
|
40
|
+
return ()
|
|
41
|
+
|
|
42
|
+
def list_channels(self, run_id: str) -> Sequence[Channel]:
|
|
43
|
+
if run_id != self._run.id:
|
|
44
|
+
raise ValueError(f"Unknown run_id: {run_id}")
|
|
45
|
+
return self._channels
|
|
46
|
+
|
|
47
|
+
def load_timeseries(
|
|
48
|
+
self,
|
|
49
|
+
run_id: str,
|
|
50
|
+
channel_ids: Sequence[str],
|
|
51
|
+
start: float,
|
|
52
|
+
end: float,
|
|
53
|
+
) -> Sequence[TimeSeries]:
|
|
54
|
+
if run_id != self._run.id:
|
|
55
|
+
raise ValueError(f"Unknown run_id: {run_id}")
|
|
56
|
+
|
|
57
|
+
selected_channel_ids = _validate_channel_ids(channel_ids)
|
|
58
|
+
unknown_channel_ids = [
|
|
59
|
+
channel_id
|
|
60
|
+
for channel_id in selected_channel_ids
|
|
61
|
+
if channel_id not in self._channel_ids
|
|
62
|
+
]
|
|
63
|
+
if unknown_channel_ids:
|
|
64
|
+
raise ValueError(f"Unknown channel_id: {unknown_channel_ids[0]}")
|
|
65
|
+
|
|
66
|
+
start_bound = _validate_bound("start", start)
|
|
67
|
+
end_bound = _validate_bound("end", end)
|
|
68
|
+
if end_bound < start_bound:
|
|
69
|
+
raise ValueError("end must be >= start")
|
|
70
|
+
|
|
71
|
+
timestamps: list[float] = []
|
|
72
|
+
values_by_channel = {
|
|
73
|
+
channel_id: []
|
|
74
|
+
for channel_id in selected_channel_ids
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
with self._samples_path.open(newline="", encoding="utf-8") as file:
|
|
78
|
+
reader = csv.DictReader(file)
|
|
79
|
+
for row in reader:
|
|
80
|
+
timestamp = float(row["time_seconds"])
|
|
81
|
+
if start_bound <= timestamp <= end_bound:
|
|
82
|
+
timestamps.append(timestamp)
|
|
83
|
+
for channel_id in selected_channel_ids:
|
|
84
|
+
values_by_channel[channel_id].append(float(row[channel_id]))
|
|
85
|
+
|
|
86
|
+
return tuple(
|
|
87
|
+
TimeSeries(
|
|
88
|
+
channel_id=channel_id,
|
|
89
|
+
timestamps=timestamps,
|
|
90
|
+
values=values_by_channel[channel_id],
|
|
91
|
+
)
|
|
92
|
+
for channel_id in selected_channel_ids
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _load_metadata(path: Path) -> Mapping[str, object]:
|
|
97
|
+
with path.open(encoding="utf-8") as file:
|
|
98
|
+
metadata = json.load(file)
|
|
99
|
+
|
|
100
|
+
if not isinstance(metadata, Mapping):
|
|
101
|
+
raise TypeError("CSV metadata must be a JSON object")
|
|
102
|
+
return metadata
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _load_test(metadata: Mapping[str, object]) -> Test:
|
|
106
|
+
test_data = metadata["test"]
|
|
107
|
+
if not isinstance(test_data, Mapping):
|
|
108
|
+
raise TypeError("CSV metadata test must be an object")
|
|
109
|
+
|
|
110
|
+
return Test(
|
|
111
|
+
id=test_data["id"], # type: ignore[arg-type]
|
|
112
|
+
name=test_data["name"], # type: ignore[arg-type]
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _load_run(metadata: Mapping[str, object]) -> Run:
|
|
117
|
+
run_data = metadata["run"]
|
|
118
|
+
if not isinstance(run_data, Mapping):
|
|
119
|
+
raise TypeError("CSV metadata run must be an object")
|
|
120
|
+
|
|
121
|
+
return Run(
|
|
122
|
+
id=run_data["id"], # type: ignore[arg-type]
|
|
123
|
+
test_id=run_data["test_id"], # type: ignore[arg-type]
|
|
124
|
+
name=run_data["name"], # type: ignore[arg-type]
|
|
125
|
+
channel_ids=run_data["channel_ids"], # type: ignore[arg-type]
|
|
126
|
+
duration=run_data["duration_seconds"], # type: ignore[arg-type]
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _load_channels(metadata: Mapping[str, object]) -> tuple[Channel, ...]:
|
|
131
|
+
channels_data = metadata["channels"]
|
|
132
|
+
if not isinstance(channels_data, Sequence) or isinstance(channels_data, (str, bytes)):
|
|
133
|
+
raise TypeError("CSV metadata channels must be a sequence")
|
|
134
|
+
|
|
135
|
+
channels = []
|
|
136
|
+
for channel_data in channels_data:
|
|
137
|
+
if not isinstance(channel_data, Mapping):
|
|
138
|
+
raise TypeError("CSV metadata channel entries must be objects")
|
|
139
|
+
channels.append(Channel.from_dict(channel_data))
|
|
140
|
+
|
|
141
|
+
return tuple(channels)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _validate_channel_ids(channel_ids: object) -> tuple[str, ...]:
|
|
145
|
+
if isinstance(channel_ids, (str, bytes)) or not isinstance(channel_ids, Sequence):
|
|
146
|
+
raise TypeError("channel_ids must be a non-empty sequence of strings")
|
|
147
|
+
|
|
148
|
+
normalized = tuple(_validate_channel_id(channel_id) for channel_id in channel_ids)
|
|
149
|
+
if not normalized:
|
|
150
|
+
raise ValueError("channel_ids must not be empty")
|
|
151
|
+
if len(set(normalized)) != len(normalized):
|
|
152
|
+
raise ValueError("channel_ids must be unique")
|
|
153
|
+
|
|
154
|
+
return normalized
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _validate_channel_id(channel_id: object) -> str:
|
|
158
|
+
if not isinstance(channel_id, str):
|
|
159
|
+
raise TypeError("channel_ids must contain strings")
|
|
160
|
+
|
|
161
|
+
normalized = channel_id.strip()
|
|
162
|
+
if not normalized:
|
|
163
|
+
raise ValueError("channel_ids must not contain empty values")
|
|
164
|
+
|
|
165
|
+
return normalized
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _validate_bound(name: str, value: object) -> float:
|
|
169
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
170
|
+
raise TypeError(f"{name} must be a finite number")
|
|
171
|
+
|
|
172
|
+
bound = float(value)
|
|
173
|
+
if not isfinite(bound):
|
|
174
|
+
raise ValueError(f"{name} must be a finite number")
|
|
175
|
+
|
|
176
|
+
return bound
|
|
@@ -0,0 +1,434 @@
|
|
|
1
|
+
"""NASA Milling Dataset adapter for the original MATLAB source file."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping, Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from math import isfinite, isnan
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
from scipy.io import loadmat
|
|
12
|
+
|
|
13
|
+
from vnuli.adapters.base import DataAdapter
|
|
14
|
+
from vnuli.config import (
|
|
15
|
+
SIGNAL_UNIT_POLICY_SOURCE_NATIVE,
|
|
16
|
+
TIME_BASIS_GENERATED_ZERO,
|
|
17
|
+
NasaMillingMatConfig,
|
|
18
|
+
)
|
|
19
|
+
from vnuli.domain.channel import Channel
|
|
20
|
+
from vnuli.domain.run import Run
|
|
21
|
+
from vnuli.domain.test import Test
|
|
22
|
+
from vnuli.domain.timeseries import TimeSeries
|
|
23
|
+
from vnuli.domain.units import Unit
|
|
24
|
+
|
|
25
|
+
SIGNAL_FIELDS = (
|
|
26
|
+
"smcAC",
|
|
27
|
+
"smcDC",
|
|
28
|
+
"vib_table",
|
|
29
|
+
"vib_spindle",
|
|
30
|
+
"AE_table",
|
|
31
|
+
"AE_spindle",
|
|
32
|
+
)
|
|
33
|
+
SCALAR_FIELDS = ("case", "run", "VB", "time", "DOC", "feed", "material")
|
|
34
|
+
REQUIRED_FIELDS = (*SCALAR_FIELDS, *SIGNAL_FIELDS)
|
|
35
|
+
|
|
36
|
+
CHANNEL_DESCRIPTIONS: Mapping[str, Mapping[str, str]] = {
|
|
37
|
+
"smcAC": {
|
|
38
|
+
"name": "AC spindle motor current",
|
|
39
|
+
"quantity": "spindle motor current",
|
|
40
|
+
"location": "spindle motor",
|
|
41
|
+
},
|
|
42
|
+
"smcDC": {
|
|
43
|
+
"name": "DC spindle motor current",
|
|
44
|
+
"quantity": "spindle motor current",
|
|
45
|
+
"location": "spindle motor",
|
|
46
|
+
},
|
|
47
|
+
"vib_table": {
|
|
48
|
+
"name": "Table vibration",
|
|
49
|
+
"quantity": "vibration",
|
|
50
|
+
"location": "table",
|
|
51
|
+
},
|
|
52
|
+
"vib_spindle": {
|
|
53
|
+
"name": "Spindle vibration",
|
|
54
|
+
"quantity": "vibration",
|
|
55
|
+
"location": "spindle",
|
|
56
|
+
},
|
|
57
|
+
"AE_table": {
|
|
58
|
+
"name": "Table acoustic emission",
|
|
59
|
+
"quantity": "acoustic emission",
|
|
60
|
+
"location": "table",
|
|
61
|
+
},
|
|
62
|
+
"AE_spindle": {
|
|
63
|
+
"name": "Spindle acoustic emission",
|
|
64
|
+
"quantity": "acoustic emission",
|
|
65
|
+
"location": "spindle",
|
|
66
|
+
},
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(frozen=True)
|
|
71
|
+
class _Record:
|
|
72
|
+
run: Run
|
|
73
|
+
signal_values: Mapping[str, tuple[float, ...]]
|
|
74
|
+
sample_count: int
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class NasaMillingMatAdapter(DataAdapter):
|
|
78
|
+
"""Read the NASA Milling Dataset MATLAB file into domain objects."""
|
|
79
|
+
|
|
80
|
+
def __init__(self, config: NasaMillingMatConfig) -> None:
|
|
81
|
+
if not isinstance(config, NasaMillingMatConfig):
|
|
82
|
+
raise TypeError("config must be a NasaMillingMatConfig")
|
|
83
|
+
if config.time_basis != TIME_BASIS_GENERATED_ZERO:
|
|
84
|
+
raise ValueError(
|
|
85
|
+
f"time_basis must be {TIME_BASIS_GENERATED_ZERO!r}"
|
|
86
|
+
)
|
|
87
|
+
if config.signal_unit_policy != SIGNAL_UNIT_POLICY_SOURCE_NATIVE:
|
|
88
|
+
raise ValueError(
|
|
89
|
+
f"signal_unit_policy must be {SIGNAL_UNIT_POLICY_SOURCE_NATIVE!r}"
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
self._config = config
|
|
93
|
+
self._test = Test(
|
|
94
|
+
id=config.dataset_id,
|
|
95
|
+
name=config.dataset_name,
|
|
96
|
+
metadata={
|
|
97
|
+
"adapter": config.adapter,
|
|
98
|
+
"source_format": "matlab_v5",
|
|
99
|
+
"source_path": str(config.mat_path),
|
|
100
|
+
},
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
records = _load_records(config)
|
|
104
|
+
self._records_by_run_id = {record.run.id: record for record in records}
|
|
105
|
+
self._runs = tuple(record.run for record in records)
|
|
106
|
+
self._channels = _build_channels(config)
|
|
107
|
+
self._channel_ids = frozenset(channel.id for channel in self._channels)
|
|
108
|
+
|
|
109
|
+
self._test = Test(
|
|
110
|
+
id=config.dataset_id,
|
|
111
|
+
name=config.dataset_name,
|
|
112
|
+
metadata={
|
|
113
|
+
"adapter": config.adapter,
|
|
114
|
+
"source_format": "matlab_v5",
|
|
115
|
+
"source_path": str(config.mat_path),
|
|
116
|
+
"record_count": len(self._runs),
|
|
117
|
+
},
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
def list_tests(self) -> Sequence[Test]:
|
|
121
|
+
return (self._test,)
|
|
122
|
+
|
|
123
|
+
def list_runs(self, test_id: str | None = None) -> Sequence[Run]:
|
|
124
|
+
if test_id is None or test_id == self._test.id:
|
|
125
|
+
return self._runs
|
|
126
|
+
return ()
|
|
127
|
+
|
|
128
|
+
def list_channels(self, run_id: str) -> Sequence[Channel]:
|
|
129
|
+
self._record_for_run_id(run_id)
|
|
130
|
+
return self._channels
|
|
131
|
+
|
|
132
|
+
def load_timeseries(
|
|
133
|
+
self,
|
|
134
|
+
run_id: str,
|
|
135
|
+
channel_ids: Sequence[str],
|
|
136
|
+
start: float,
|
|
137
|
+
end: float,
|
|
138
|
+
) -> Sequence[TimeSeries]:
|
|
139
|
+
record = self._record_for_run_id(run_id)
|
|
140
|
+
selected_channel_ids = _validate_channel_ids(channel_ids)
|
|
141
|
+
for channel_id in selected_channel_ids:
|
|
142
|
+
if channel_id not in self._channel_ids:
|
|
143
|
+
raise ValueError(f"Unknown channel_id: {channel_id}")
|
|
144
|
+
|
|
145
|
+
start_bound = _validate_bound("start", start)
|
|
146
|
+
end_bound = _validate_bound("end", end)
|
|
147
|
+
if end_bound < start_bound:
|
|
148
|
+
raise ValueError("end must be >= start")
|
|
149
|
+
|
|
150
|
+
timestamps = np.arange(record.sample_count, dtype=float) / self._config.sample_rate_hz
|
|
151
|
+
mask = (timestamps >= start_bound) & (timestamps <= end_bound)
|
|
152
|
+
bounded_timestamps = tuple(float(timestamp) for timestamp in timestamps[mask])
|
|
153
|
+
|
|
154
|
+
return tuple(
|
|
155
|
+
TimeSeries(
|
|
156
|
+
channel_id=channel_id,
|
|
157
|
+
timestamps=bounded_timestamps,
|
|
158
|
+
values=tuple(
|
|
159
|
+
record.signal_values[channel_id][index]
|
|
160
|
+
for index in np.nonzero(mask)[0]
|
|
161
|
+
),
|
|
162
|
+
)
|
|
163
|
+
for channel_id in selected_channel_ids
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
def _record_for_run_id(self, run_id: str) -> _Record:
|
|
167
|
+
try:
|
|
168
|
+
return self._records_by_run_id[run_id]
|
|
169
|
+
except KeyError as error:
|
|
170
|
+
raise ValueError(f"Unknown run_id: {run_id}") from error
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _load_records(config: NasaMillingMatConfig) -> tuple[_Record, ...]:
|
|
174
|
+
mat_data = loadmat(config.mat_path, squeeze_me=True, struct_as_record=False)
|
|
175
|
+
if "mill" not in mat_data:
|
|
176
|
+
raise ValueError("NASA Milling MAT file must contain top-level variable 'mill'")
|
|
177
|
+
|
|
178
|
+
source_records = _normalize_records(mat_data["mill"])
|
|
179
|
+
if not source_records:
|
|
180
|
+
raise ValueError("NASA Milling 'mill' variable must contain at least one record")
|
|
181
|
+
|
|
182
|
+
return tuple(
|
|
183
|
+
_build_record(config, source_record, index)
|
|
184
|
+
for index, source_record in enumerate(source_records)
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _normalize_records(value: object) -> tuple[object, ...]:
|
|
189
|
+
if _is_struct_like(value):
|
|
190
|
+
return (value,)
|
|
191
|
+
|
|
192
|
+
if isinstance(value, np.ndarray):
|
|
193
|
+
flattened = tuple(item for item in value.ravel())
|
|
194
|
+
if flattened and all(_is_struct_like(item) for item in flattened):
|
|
195
|
+
return flattened
|
|
196
|
+
|
|
197
|
+
raise TypeError("NASA Milling 'mill' variable must be a MATLAB struct array")
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _build_record(
|
|
201
|
+
config: NasaMillingMatConfig,
|
|
202
|
+
source_record: object,
|
|
203
|
+
index: int,
|
|
204
|
+
) -> _Record:
|
|
205
|
+
_require_consumed_fields(source_record)
|
|
206
|
+
|
|
207
|
+
source_case = _required_integer_scalar(source_record, "case", index)
|
|
208
|
+
source_run = _required_integer_scalar(source_record, "run", index)
|
|
209
|
+
scalar_time = _required_numeric_scalar(source_record, "time", index)
|
|
210
|
+
doc = _required_numeric_scalar(source_record, "DOC", index)
|
|
211
|
+
feed = _required_numeric_scalar(source_record, "feed", index)
|
|
212
|
+
material = _required_integer_scalar(source_record, "material", index)
|
|
213
|
+
vb = _required_numeric_scalar(source_record, "VB", index, allow_nan=True)
|
|
214
|
+
|
|
215
|
+
signal_values = {
|
|
216
|
+
field_name: _required_signal_values(source_record, field_name, index)
|
|
217
|
+
for field_name in SIGNAL_FIELDS
|
|
218
|
+
}
|
|
219
|
+
sample_counts = {len(values) for values in signal_values.values()}
|
|
220
|
+
if len(sample_counts) != 1:
|
|
221
|
+
raise ValueError(
|
|
222
|
+
f"NASA Milling record {index} signal fields must have matching lengths"
|
|
223
|
+
)
|
|
224
|
+
sample_count = sample_counts.pop()
|
|
225
|
+
if sample_count == 0:
|
|
226
|
+
raise ValueError(f"NASA Milling record {index} signal fields must not be empty")
|
|
227
|
+
|
|
228
|
+
vb_missing = isnan(vb)
|
|
229
|
+
run_id = (
|
|
230
|
+
f"{config.dataset_id}_RECORD_{index:03d}_"
|
|
231
|
+
f"CASE_{source_case:02d}_RUN_{source_run:02d}"
|
|
232
|
+
)
|
|
233
|
+
generated_time_end = (sample_count - 1) / config.sample_rate_hz
|
|
234
|
+
run = Run(
|
|
235
|
+
id=run_id,
|
|
236
|
+
test_id=config.dataset_id,
|
|
237
|
+
name=f"NASA Milling record {index:03d} case {source_case} run {source_run}",
|
|
238
|
+
channel_ids=SIGNAL_FIELDS,
|
|
239
|
+
duration=None,
|
|
240
|
+
metadata={
|
|
241
|
+
"source_record_index": index,
|
|
242
|
+
"source_case": source_case,
|
|
243
|
+
"source_run": source_run,
|
|
244
|
+
"VB": None if vb_missing else vb,
|
|
245
|
+
"VB_missing": vb_missing,
|
|
246
|
+
"time": scalar_time,
|
|
247
|
+
"time_unit_status": "unknown",
|
|
248
|
+
"DOC": doc,
|
|
249
|
+
"feed": feed,
|
|
250
|
+
"material": material,
|
|
251
|
+
"sample_count": sample_count,
|
|
252
|
+
"generated_time_start": 0.0,
|
|
253
|
+
"generated_time_end": generated_time_end,
|
|
254
|
+
"sample_rate_hz": config.sample_rate_hz,
|
|
255
|
+
"time_basis": config.time_basis,
|
|
256
|
+
"time_basis_source": "configuration",
|
|
257
|
+
"source_path": str(config.mat_path),
|
|
258
|
+
},
|
|
259
|
+
)
|
|
260
|
+
return _Record(run=run, signal_values=signal_values, sample_count=sample_count)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _build_channels(config: NasaMillingMatConfig) -> tuple[Channel, ...]:
|
|
264
|
+
channels = []
|
|
265
|
+
for field_name in SIGNAL_FIELDS:
|
|
266
|
+
description = CHANNEL_DESCRIPTIONS[field_name]
|
|
267
|
+
channels.append(
|
|
268
|
+
Channel(
|
|
269
|
+
id=field_name,
|
|
270
|
+
name=description["name"],
|
|
271
|
+
quantity=description["quantity"],
|
|
272
|
+
units=Unit(f"source_native:{field_name}"),
|
|
273
|
+
sample_rate=config.sample_rate_hz,
|
|
274
|
+
location=description["location"],
|
|
275
|
+
metadata={
|
|
276
|
+
"unit_status": "not_encoded_in_source",
|
|
277
|
+
"unit_token_semantics": "opaque_source_native_placeholder",
|
|
278
|
+
"unit_policy": config.signal_unit_policy,
|
|
279
|
+
"source_field": field_name,
|
|
280
|
+
"time_basis": config.time_basis,
|
|
281
|
+
"sample_rate_hz": config.sample_rate_hz,
|
|
282
|
+
"time_basis_source": "configuration",
|
|
283
|
+
"source_path": str(config.mat_path),
|
|
284
|
+
},
|
|
285
|
+
)
|
|
286
|
+
)
|
|
287
|
+
return tuple(channels)
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _require_consumed_fields(source_record: object) -> None:
|
|
291
|
+
missing_fields = [
|
|
292
|
+
field_name
|
|
293
|
+
for field_name in REQUIRED_FIELDS
|
|
294
|
+
if not _has_field(source_record, field_name)
|
|
295
|
+
]
|
|
296
|
+
if missing_fields:
|
|
297
|
+
raise ValueError(
|
|
298
|
+
"NASA Milling record is missing required field: "
|
|
299
|
+
f"{missing_fields[0]}"
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _required_integer_scalar(
|
|
304
|
+
source_record: object,
|
|
305
|
+
field_name: str,
|
|
306
|
+
record_index: int,
|
|
307
|
+
) -> int:
|
|
308
|
+
value = _required_numeric_scalar(source_record, field_name, record_index)
|
|
309
|
+
if isnan(value) or not value.is_integer():
|
|
310
|
+
raise ValueError(
|
|
311
|
+
f"NASA Milling record {record_index} field '{field_name}' must be an integer"
|
|
312
|
+
)
|
|
313
|
+
return int(value)
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def _required_numeric_scalar(
|
|
317
|
+
source_record: object,
|
|
318
|
+
field_name: str,
|
|
319
|
+
record_index: int,
|
|
320
|
+
*,
|
|
321
|
+
allow_nan: bool = False,
|
|
322
|
+
) -> float:
|
|
323
|
+
raw_value = _get_field(source_record, field_name)
|
|
324
|
+
array = np.asarray(raw_value)
|
|
325
|
+
if array.size != 1:
|
|
326
|
+
raise ValueError(
|
|
327
|
+
f"NASA Milling record {record_index} field '{field_name}' must be scalar"
|
|
328
|
+
)
|
|
329
|
+
try:
|
|
330
|
+
value = float(array.reshape(-1)[0])
|
|
331
|
+
except (TypeError, ValueError) as error:
|
|
332
|
+
raise TypeError(
|
|
333
|
+
f"NASA Milling record {record_index} field '{field_name}' must be numeric"
|
|
334
|
+
) from error
|
|
335
|
+
|
|
336
|
+
if allow_nan and isnan(value):
|
|
337
|
+
return value
|
|
338
|
+
if not isfinite(value):
|
|
339
|
+
raise ValueError(
|
|
340
|
+
f"NASA Milling record {record_index} field '{field_name}' must be finite"
|
|
341
|
+
)
|
|
342
|
+
return value
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _required_signal_values(
|
|
346
|
+
source_record: object,
|
|
347
|
+
field_name: str,
|
|
348
|
+
record_index: int,
|
|
349
|
+
) -> tuple[float, ...]:
|
|
350
|
+
raw_value = _get_field(source_record, field_name)
|
|
351
|
+
array = np.asarray(raw_value)
|
|
352
|
+
if array.ndim == 0:
|
|
353
|
+
raise ValueError(
|
|
354
|
+
f"NASA Milling record {record_index} field '{field_name}' must be a signal array"
|
|
355
|
+
)
|
|
356
|
+
if array.ndim > 1:
|
|
357
|
+
non_singleton_dims = [dimension for dimension in array.shape if dimension != 1]
|
|
358
|
+
if len(non_singleton_dims) != 1:
|
|
359
|
+
raise ValueError(
|
|
360
|
+
f"NASA Milling record {record_index} field '{field_name}' "
|
|
361
|
+
"must be one-dimensional"
|
|
362
|
+
)
|
|
363
|
+
try:
|
|
364
|
+
values = tuple(float(value) for value in array.reshape(-1))
|
|
365
|
+
except (TypeError, ValueError) as error:
|
|
366
|
+
raise TypeError(
|
|
367
|
+
f"NASA Milling record {record_index} field '{field_name}' must be numeric"
|
|
368
|
+
) from error
|
|
369
|
+
|
|
370
|
+
if any(not isfinite(value) and not isnan(value) for value in values):
|
|
371
|
+
raise ValueError(
|
|
372
|
+
f"NASA Milling record {record_index} field '{field_name}' "
|
|
373
|
+
"contains a value the TimeSeries domain cannot represent"
|
|
374
|
+
)
|
|
375
|
+
return values
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def _is_struct_like(value: object) -> bool:
|
|
379
|
+
if hasattr(value, "_fieldnames"):
|
|
380
|
+
return True
|
|
381
|
+
if isinstance(value, np.void) and value.dtype.names:
|
|
382
|
+
return True
|
|
383
|
+
return False
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _has_field(source_record: object, field_name: str) -> bool:
|
|
387
|
+
if hasattr(source_record, "_fieldnames"):
|
|
388
|
+
return field_name in source_record._fieldnames # noqa: SLF001
|
|
389
|
+
if isinstance(source_record, np.void) and source_record.dtype.names:
|
|
390
|
+
return field_name in source_record.dtype.names
|
|
391
|
+
return False
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _get_field(source_record: object, field_name: str) -> Any:
|
|
395
|
+
if hasattr(source_record, "_fieldnames"):
|
|
396
|
+
return getattr(source_record, field_name)
|
|
397
|
+
if isinstance(source_record, np.void) and source_record.dtype.names:
|
|
398
|
+
return source_record[field_name]
|
|
399
|
+
raise TypeError("NASA Milling record must be a MATLAB struct")
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def _validate_channel_ids(channel_ids: object) -> tuple[str, ...]:
|
|
403
|
+
if isinstance(channel_ids, (str, bytes)) or not isinstance(channel_ids, Sequence):
|
|
404
|
+
raise TypeError("channel_ids must be a non-empty sequence of strings")
|
|
405
|
+
|
|
406
|
+
normalized = tuple(_validate_channel_id(channel_id) for channel_id in channel_ids)
|
|
407
|
+
if not normalized:
|
|
408
|
+
raise ValueError("channel_ids must not be empty")
|
|
409
|
+
if len(set(normalized)) != len(normalized):
|
|
410
|
+
raise ValueError("channel_ids must be unique")
|
|
411
|
+
|
|
412
|
+
return normalized
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def _validate_channel_id(channel_id: object) -> str:
|
|
416
|
+
if not isinstance(channel_id, str):
|
|
417
|
+
raise TypeError("channel_ids must contain strings")
|
|
418
|
+
|
|
419
|
+
normalized = channel_id.strip()
|
|
420
|
+
if not normalized:
|
|
421
|
+
raise ValueError("channel_ids must not contain empty values")
|
|
422
|
+
|
|
423
|
+
return normalized
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _validate_bound(name: str, value: object) -> float:
|
|
427
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
428
|
+
raise TypeError(f"{name} must be a finite number")
|
|
429
|
+
|
|
430
|
+
bound = float(value)
|
|
431
|
+
if not isfinite(bound):
|
|
432
|
+
raise ValueError(f"{name} must be a finite number")
|
|
433
|
+
|
|
434
|
+
return bound
|