sdfio 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sdfio/__init__.py +64 -0
- sdfio/__main__.py +12 -0
- sdfio/_ascii.py +287 -0
- sdfio/_binary.py +221 -0
- sdfio/_numeric.py +35 -0
- sdfio/cli.py +140 -0
- sdfio/datatypes.py +251 -0
- sdfio/exceptions.py +21 -0
- sdfio/file.py +436 -0
- sdfio/header.py +420 -0
- sdfio/py.typed +0 -0
- sdfio-1.0.0.dist-info/METADATA +117 -0
- sdfio-1.0.0.dist-info/RECORD +16 -0
- sdfio-1.0.0.dist-info/WHEEL +4 -0
- sdfio-1.0.0.dist-info/entry_points.txt +2 -0
- sdfio-1.0.0.dist-info/licenses/LICENSE +21 -0
sdfio/__init__.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: 2026 Thomas Ascher <thomas.ascher@gmx.at>
|
|
2
|
+
#
|
|
3
|
+
# SPDX-License-Identifier: MIT
|
|
4
|
+
|
|
5
|
+
"""sdfio: read and write ISO 25178-71 SDF surface data files.
|
|
6
|
+
|
|
7
|
+
The Surface Data File (SDF) format is defined by ISO 25178-71, "Geometrical
|
|
8
|
+
product specifications (GPS) -- Surface texture: Areal -- Part 71: Software
|
|
9
|
+
measurement standards". It stores areal (or profile) surface topography
|
|
10
|
+
measurements as a rectangular grid of height values, in either a
|
|
11
|
+
human-readable ASCII or a compact binary format.
|
|
12
|
+
|
|
13
|
+
Basic usage::
|
|
14
|
+
|
|
15
|
+
>>> import os
|
|
16
|
+
>>> import tempfile
|
|
17
|
+
>>> import numpy as np
|
|
18
|
+
>>> import sdfio
|
|
19
|
+
>>> path = os.path.join(tempfile.mkdtemp(), "example.sdf")
|
|
20
|
+
>>> sdfio.write(path, np.zeros((2, 3)), x_scale=1e-6, y_scale=1e-6)
|
|
21
|
+
>>> sdf = sdfio.read(path)
|
|
22
|
+
>>> sdf.data.shape
|
|
23
|
+
(2, 3)
|
|
24
|
+
>>> sdf.header.z_scale
|
|
25
|
+
1e-06
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
from .datatypes import (
|
|
31
|
+
DATA_TYPES,
|
|
32
|
+
DataType,
|
|
33
|
+
SdfDataType,
|
|
34
|
+
decode_raw,
|
|
35
|
+
encode_raw,
|
|
36
|
+
get_data_type,
|
|
37
|
+
suggest_z_scale,
|
|
38
|
+
)
|
|
39
|
+
from .exceptions import SdfError, SdfFormatError, SdfVersionError
|
|
40
|
+
from .file import FileFormat, SdfFile, read, write
|
|
41
|
+
from .header import SdfDialect, SdfHeader, SdfMetadata
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"DATA_TYPES",
|
|
45
|
+
"DataType",
|
|
46
|
+
"FileFormat",
|
|
47
|
+
"SdfDataType",
|
|
48
|
+
"SdfDialect",
|
|
49
|
+
"SdfError",
|
|
50
|
+
"SdfFile",
|
|
51
|
+
"SdfFormatError",
|
|
52
|
+
"SdfHeader",
|
|
53
|
+
"SdfMetadata",
|
|
54
|
+
"SdfVersionError",
|
|
55
|
+
"__version__",
|
|
56
|
+
"decode_raw",
|
|
57
|
+
"encode_raw",
|
|
58
|
+
"get_data_type",
|
|
59
|
+
"read",
|
|
60
|
+
"suggest_z_scale",
|
|
61
|
+
"write",
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
__version__ = "1.0.0"
|
sdfio/__main__.py
ADDED
sdfio/_ascii.py
ADDED
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: 2026 Thomas Ascher <thomas.ascher@gmx.at>
|
|
2
|
+
#
|
|
3
|
+
# SPDX-License-Identifier: MIT
|
|
4
|
+
|
|
5
|
+
"""ASCII (text) representation of the SDF format."""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from collections.abc import Mapping
|
|
11
|
+
from typing import IO
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
|
|
15
|
+
from ._numeric import format_scientific
|
|
16
|
+
from .datatypes import DataType, SdfDataType, require_supported_data_type, validate_data_range
|
|
17
|
+
from .exceptions import SdfFormatError
|
|
18
|
+
from .header import (
|
|
19
|
+
ASCII_PREFIX,
|
|
20
|
+
BINARY_PREFIX,
|
|
21
|
+
SdfDialect,
|
|
22
|
+
SdfHeader,
|
|
23
|
+
format_sdf_datetime,
|
|
24
|
+
parse_sdf_datetime,
|
|
25
|
+
validate_check_type,
|
|
26
|
+
validate_compression,
|
|
27
|
+
validate_manufacturer_id_ascii,
|
|
28
|
+
validate_trailer_ascii,
|
|
29
|
+
validate_trailer_xml,
|
|
30
|
+
validate_z_scale,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
__all__ = ["dump", "dumps", "load", "loads"]
|
|
34
|
+
|
|
35
|
+
_MAGIC_RE = re.compile(
|
|
36
|
+
rf"^(?P<prefix>[{ASCII_PREFIX}{BINARY_PREFIX}])(?P<dialect>[A-Z]{{3}})-(?P<version>\d\.\d)$"
|
|
37
|
+
)
|
|
38
|
+
_FIELD_RE = re.compile(r"^(?P<name>\w+)\s*=\s*(?P<value>.*)$")
|
|
39
|
+
_INVALID_MARKER = "BAD"
|
|
40
|
+
# The standard mandates <CRLF> line endings; dumps() always writes them.
|
|
41
|
+
# \r? here is a read-side leniency to also accept bare LF.
|
|
42
|
+
_LINE_SPLIT_RE = re.compile(r"\r?\n")
|
|
43
|
+
_TERMINATOR_RE = re.compile(r"^[ \t]*\*[ \t]*$")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _format_scale_field(value: float) -> str:
|
|
47
|
+
# Xscale/Yscale/Zscale/Zresolution are always written at binary64
|
|
48
|
+
# precision, regardless of the data area's own DataType.
|
|
49
|
+
return format_scientific(value, 14, 3)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _split_records(remainder: str) -> list[str]:
|
|
53
|
+
# Splitting by matching the "*" delimiter together with its surrounding
|
|
54
|
+
# newlines (as one combined pattern) can't tell two delimiters directly
|
|
55
|
+
# adjacent to each other (an explicit empty record, e.g. an empty
|
|
56
|
+
# trailer written as "*<CRLF>*<CRLF>") apart from a single delimiter --
|
|
57
|
+
# the newline between them would need to be consumed by both matches at
|
|
58
|
+
# once. Splitting into lines first and testing each one for being a bare
|
|
59
|
+
# "*" avoids that ambiguity.
|
|
60
|
+
records = []
|
|
61
|
+
current: list[str] = []
|
|
62
|
+
for line in _LINE_SPLIT_RE.split(remainder):
|
|
63
|
+
if _TERMINATOR_RE.match(line):
|
|
64
|
+
records.append("\n".join(current))
|
|
65
|
+
current = []
|
|
66
|
+
else:
|
|
67
|
+
current.append(line)
|
|
68
|
+
records.append("\n".join(current))
|
|
69
|
+
return records
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _read_fields(header_text: str) -> dict[str, str]:
|
|
73
|
+
fields: dict[str, str] = {}
|
|
74
|
+
for raw_line in header_text.splitlines():
|
|
75
|
+
line = raw_line.strip()
|
|
76
|
+
if not line:
|
|
77
|
+
continue
|
|
78
|
+
match = _FIELD_RE.match(line)
|
|
79
|
+
if not match:
|
|
80
|
+
raise SdfFormatError(f"Malformed SDF header line: {line!r}")
|
|
81
|
+
fields[match.group("name")] = match.group("value").strip()
|
|
82
|
+
return fields
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _field(fields: Mapping[str, str], name: str) -> str:
|
|
86
|
+
for key, value in fields.items():
|
|
87
|
+
if key.lower() == name.lower():
|
|
88
|
+
return value
|
|
89
|
+
raise SdfFormatError(f"Missing required SDF header field {name!r}")
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _parse_int(fields: Mapping[str, str], name: str) -> int:
|
|
93
|
+
value = _field(fields, name)
|
|
94
|
+
try:
|
|
95
|
+
return int(value)
|
|
96
|
+
except ValueError:
|
|
97
|
+
raise SdfFormatError(
|
|
98
|
+
f"Invalid integer value for SDF header field {name!r}: {value!r}"
|
|
99
|
+
) from None
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _parse_float(fields: Mapping[str, str], name: str) -> float:
|
|
103
|
+
value = _field(fields, name)
|
|
104
|
+
try:
|
|
105
|
+
return float(value)
|
|
106
|
+
except ValueError:
|
|
107
|
+
raise SdfFormatError(
|
|
108
|
+
f"Invalid float value for SDF header field {name!r}: {value!r}"
|
|
109
|
+
) from None
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def loads(text: str) -> tuple[SdfHeader, np.ndarray, str]:
|
|
113
|
+
"""Parse an in-memory ASCII SDF document.
|
|
114
|
+
|
|
115
|
+
:param text: Full contents of an ASCII-encoded ``.sdf`` file.
|
|
116
|
+
:returns: A ``(header, data, trailer)`` tuple. ``data`` is a
|
|
117
|
+
``(num_profiles, num_points)`` array of height values in metres,
|
|
118
|
+
with ``NaN`` marking non-measured or spurious points.
|
|
119
|
+
:raises SdfFormatError: If the file is malformed, e.g. a bad magic,
|
|
120
|
+
unsupported dialect or data type, missing records, or a trailer
|
|
121
|
+
that isn't 7-bit ASCII.
|
|
122
|
+
"""
|
|
123
|
+
text = text.lstrip("\ufeff")
|
|
124
|
+
first_newline = re.search(r"\r?\n", text)
|
|
125
|
+
if first_newline is None:
|
|
126
|
+
raise SdfFormatError("SDF file is missing the header and data records")
|
|
127
|
+
magic_line = text[: first_newline.start()].strip()
|
|
128
|
+
remainder = text[first_newline.end() :]
|
|
129
|
+
|
|
130
|
+
magic_match = _MAGIC_RE.match(magic_line)
|
|
131
|
+
if not magic_match:
|
|
132
|
+
raise SdfFormatError(f"Not an ASCII SDF file, unexpected magic: {magic_line!r}")
|
|
133
|
+
if magic_match.group("prefix") != ASCII_PREFIX:
|
|
134
|
+
raise SdfFormatError("Binary magic found while parsing an ASCII SDF file")
|
|
135
|
+
dialect_text = magic_match.group("dialect")
|
|
136
|
+
version_text = magic_match.group("version")
|
|
137
|
+
dialect = SdfDialect.resolve(dialect_text, version_text, magic_line)
|
|
138
|
+
|
|
139
|
+
records = _split_records(remainder)
|
|
140
|
+
if len(records) < 3:
|
|
141
|
+
raise SdfFormatError("SDF file must contain header, data and trailer records")
|
|
142
|
+
# The trailer is itself terminated by its own "*" record; drop
|
|
143
|
+
# a resulting trailing empty segment instead of re-joining it back in.
|
|
144
|
+
header_text, data_text, *trailer_parts = records
|
|
145
|
+
if trailer_parts and trailer_parts[-1] == "":
|
|
146
|
+
trailer_parts = trailer_parts[:-1]
|
|
147
|
+
trailer_text = "*".join(trailer_parts)
|
|
148
|
+
|
|
149
|
+
fields = _read_fields(header_text)
|
|
150
|
+
data_type_code = _parse_int(fields, "DataType")
|
|
151
|
+
data_type = require_supported_data_type(data_type_code, dialect)
|
|
152
|
+
validate_compression(_parse_int(fields, "Compression"))
|
|
153
|
+
validate_check_type(_parse_int(fields, "CheckType"))
|
|
154
|
+
|
|
155
|
+
header = SdfHeader(
|
|
156
|
+
dialect=dialect,
|
|
157
|
+
binary=False,
|
|
158
|
+
manufacturer_id=_field(fields, "ManufacID").strip(),
|
|
159
|
+
create_date=parse_sdf_datetime(_field(fields, "CreateDate"), dialect),
|
|
160
|
+
mod_date=parse_sdf_datetime(_field(fields, "ModDate"), dialect),
|
|
161
|
+
num_points=_parse_int(fields, "NumPoints"),
|
|
162
|
+
num_profiles=_parse_int(fields, "NumProfiles"),
|
|
163
|
+
x_scale=_parse_float(fields, "Xscale"),
|
|
164
|
+
y_scale=_parse_float(fields, "Yscale"),
|
|
165
|
+
z_scale=_parse_float(fields, "Zscale"),
|
|
166
|
+
z_resolution=_parse_float(fields, "Zresolution"),
|
|
167
|
+
data_type=data_type_code,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
data = _parse_data(data_text, header, data_type)
|
|
171
|
+
trailer = trailer_text.strip()
|
|
172
|
+
validate_trailer_ascii(trailer)
|
|
173
|
+
return header, data, trailer
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _parse_data(data_text: str, header: SdfHeader, data_type: SdfDataType) -> np.ndarray:
|
|
177
|
+
tokens = data_text.split()
|
|
178
|
+
expected = header.num_points * header.num_profiles
|
|
179
|
+
if len(tokens) != expected:
|
|
180
|
+
raise SdfFormatError(f"Expected {expected} data values, found {len(tokens)}")
|
|
181
|
+
|
|
182
|
+
is_float = data_type.type in (DataType.BINARY32, DataType.BINARY64)
|
|
183
|
+
values = np.empty(header.num_points * header.num_profiles, dtype=np.float64)
|
|
184
|
+
for index, token in enumerate(tokens):
|
|
185
|
+
# The standard only specifies the literal "BAD"; matching
|
|
186
|
+
# case-insensitively is a deliberate leniency, not mandated.
|
|
187
|
+
if token.upper() == _INVALID_MARKER:
|
|
188
|
+
values[index] = np.nan
|
|
189
|
+
else:
|
|
190
|
+
try:
|
|
191
|
+
# int(token) first for non-float types so a fractional token
|
|
192
|
+
# (e.g. "1.5" for an int16 field) is rejected outright,
|
|
193
|
+
# rather than silently truncated by float()'s wider parsing.
|
|
194
|
+
values[index] = float(token) if is_float else float(int(token))
|
|
195
|
+
except ValueError:
|
|
196
|
+
raise SdfFormatError(f"Invalid data value at position {index}: {token!r}") from None
|
|
197
|
+
|
|
198
|
+
invalid_mask = np.isnan(values)
|
|
199
|
+
scaled = values * header.z_scale
|
|
200
|
+
scaled[invalid_mask] = np.nan
|
|
201
|
+
data = scaled.reshape(header.num_profiles, header.num_points)
|
|
202
|
+
return data[::-1] if header.dialect.reverses_profile_order else data
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def load(fp: IO[str]) -> tuple[SdfHeader, np.ndarray, str]:
|
|
206
|
+
"""Parse an ASCII SDF file object opened in text mode.
|
|
207
|
+
|
|
208
|
+
Equivalent to :func:`loads`, reading the text from ``fp`` first.
|
|
209
|
+
"""
|
|
210
|
+
return loads(fp.read())
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _format_value(value: float, data_type: SdfDataType) -> str:
|
|
214
|
+
if np.isnan(value):
|
|
215
|
+
return _INVALID_MARKER
|
|
216
|
+
if data_type.type == DataType.BINARY32:
|
|
217
|
+
return format_scientific(value, 6, 2)
|
|
218
|
+
if data_type.type == DataType.BINARY64:
|
|
219
|
+
return format_scientific(value, 14, 3)
|
|
220
|
+
return str(round(value))
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def dumps(header: SdfHeader, data: np.ndarray, trailer: str = "") -> str:
|
|
224
|
+
"""Serialize a header/data pair to the ASCII SDF text representation.
|
|
225
|
+
|
|
226
|
+
:param header: Header describing ``data``; ``header.binary`` is ignored.
|
|
227
|
+
:param data: ``(num_profiles, num_points)`` array of height values in
|
|
228
|
+
metres. ``NaN`` marks non-measured or spurious points.
|
|
229
|
+
:param trailer: Optional record 3 trailer content.
|
|
230
|
+
|
|
231
|
+
For :attr:`~sdfio.DataType.BINARY64`, this is lossy at the ~1e-15
|
|
232
|
+
relative level: the standard's mandated 15 significant digits are one
|
|
233
|
+
short of the 17 an IEEE 754 double needs to round-trip exactly (see
|
|
234
|
+
:func:`sdfio._numeric.format_scientific`). Use the binary format to
|
|
235
|
+
avoid this.
|
|
236
|
+
"""
|
|
237
|
+
if data.shape != header.shape:
|
|
238
|
+
raise SdfFormatError(f"Data shape {data.shape} does not match header shape {header.shape}")
|
|
239
|
+
validate_z_scale(header.z_scale)
|
|
240
|
+
validate_manufacturer_id_ascii(header.manufacturer_id)
|
|
241
|
+
validate_trailer_ascii(trailer)
|
|
242
|
+
validate_trailer_xml(header.dialect, trailer)
|
|
243
|
+
data_type = require_supported_data_type(header.data_type, header.dialect)
|
|
244
|
+
|
|
245
|
+
lines = [f"{ASCII_PREFIX}{header.dialect}"]
|
|
246
|
+
fields = (
|
|
247
|
+
("ManufacID", header.manufacturer_id),
|
|
248
|
+
("CreateDate", format_sdf_datetime(header.create_date, header.dialect)),
|
|
249
|
+
("ModDate", format_sdf_datetime(header.mod_date, header.dialect)),
|
|
250
|
+
("NumPoints", str(header.num_points)),
|
|
251
|
+
("NumProfiles", str(header.num_profiles)),
|
|
252
|
+
("Xscale", _format_scale_field(header.x_scale)),
|
|
253
|
+
("Yscale", _format_scale_field(header.y_scale)),
|
|
254
|
+
("Zscale", _format_scale_field(header.z_scale)),
|
|
255
|
+
("Zresolution", _format_scale_field(header.z_resolution)),
|
|
256
|
+
("Compression", "0"), # Never supported, see SdfHeader's docstring.
|
|
257
|
+
("DataType", str(header.data_type)),
|
|
258
|
+
("CheckType", "0"), # Never written, see SdfHeader's docstring.
|
|
259
|
+
)
|
|
260
|
+
for name, value in fields:
|
|
261
|
+
lines.append(f"{name} = {value}")
|
|
262
|
+
lines.append("*")
|
|
263
|
+
|
|
264
|
+
write_data = data[::-1] if header.dialect.reverses_profile_order else data
|
|
265
|
+
raw = write_data / header.z_scale
|
|
266
|
+
validate_data_range(raw, np.isnan(write_data), data_type, header.dialect)
|
|
267
|
+
for row in raw:
|
|
268
|
+
lines.append(" ".join(_format_value(value, data_type) for value in row))
|
|
269
|
+
lines.append("*")
|
|
270
|
+
|
|
271
|
+
# The standard doesn't unambiguously specify how a missing/empty
|
|
272
|
+
# trailer should be represented on disk; this always terminates the
|
|
273
|
+
# trailer record, even when empty. loads() tells that apart from an
|
|
274
|
+
# actual trailer whose content happens to be empty by record position,
|
|
275
|
+
# not by counting "*" markers.
|
|
276
|
+
if trailer:
|
|
277
|
+
lines.append(trailer)
|
|
278
|
+
lines.append("*")
|
|
279
|
+
return "\r\n".join(lines) + "\r\n"
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def dump(header: SdfHeader, data: np.ndarray, fp: IO[str], trailer: str = "") -> None:
|
|
283
|
+
"""Serialize a header/data pair as ASCII SDF text to a file object.
|
|
284
|
+
|
|
285
|
+
Equivalent to :func:`dumps`, writing the result to ``fp``.
|
|
286
|
+
"""
|
|
287
|
+
fp.write(dumps(header, data, trailer))
|
sdfio/_binary.py
ADDED
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: 2026 Thomas Ascher <thomas.ascher@gmx.at>
|
|
2
|
+
#
|
|
3
|
+
# SPDX-License-Identifier: MIT
|
|
4
|
+
|
|
5
|
+
"""Binary (fixed-width) representation of the SDF format."""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import io
|
|
10
|
+
import struct
|
|
11
|
+
from typing import IO
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
|
|
15
|
+
from .datatypes import decode_raw, encode_raw, require_supported_data_type
|
|
16
|
+
from .exceptions import SdfFormatError
|
|
17
|
+
from .header import (
|
|
18
|
+
ASCII_PREFIX,
|
|
19
|
+
BINARY_PREFIX,
|
|
20
|
+
MAGIC_SIZE,
|
|
21
|
+
SdfDialect,
|
|
22
|
+
SdfHeader,
|
|
23
|
+
format_sdf_datetime,
|
|
24
|
+
parse_sdf_datetime,
|
|
25
|
+
validate_check_type,
|
|
26
|
+
validate_compression,
|
|
27
|
+
validate_manufacturer_id_ascii,
|
|
28
|
+
validate_trailer_ascii,
|
|
29
|
+
validate_trailer_xml,
|
|
30
|
+
validate_z_scale,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
__all__ = ["HEADER_SIZE", "dump", "dumps", "load", "loads"]
|
|
34
|
+
|
|
35
|
+
# Field widths from the standard's record 1 header field table.
|
|
36
|
+
_DIALECT_SIZE = len("ISO") # == len("BCR")
|
|
37
|
+
_MAGIC_PREFIX_SIZE = 1 + _DIALECT_SIZE + 1 # a/b + dialect + '-'
|
|
38
|
+
_MANUFACTURER_ID_SIZE = 10
|
|
39
|
+
_DATE_SIZE = 12
|
|
40
|
+
|
|
41
|
+
_STRINGS_STRUCT = struct.Struct(f"<{_MANUFACTURER_ID_SIZE}s{_DATE_SIZE}s{_DATE_SIZE}s")
|
|
42
|
+
_TAIL_STRUCT = struct.Struct("<ddddBBB")
|
|
43
|
+
_COUNTS_STRUCT = {
|
|
44
|
+
SdfDialect.ISO_1_0: struct.Struct("<HH"),
|
|
45
|
+
SdfDialect.ISO_2_0: struct.Struct("<II"),
|
|
46
|
+
SdfDialect.BCR_1_0: struct.Struct("<HH"),
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
#: Total header size in bytes for each SDF dialect.
|
|
50
|
+
HEADER_SIZE = {
|
|
51
|
+
dialect: MAGIC_SIZE + _STRINGS_STRUCT.size + counts_struct.size + _TAIL_STRUCT.size
|
|
52
|
+
for dialect, counts_struct in _COUNTS_STRUCT.items()
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _pad(value: str, length: int) -> bytes:
|
|
57
|
+
# Oversized values are silently truncated (not rejected) to fit the
|
|
58
|
+
# fixed-width field; only ManufacID can realistically be long enough to
|
|
59
|
+
# hit this, since the date fields are always exactly formatted per the
|
|
60
|
+
# standard's header field table. Callers already validate ASCII-ness
|
|
61
|
+
# before this point.
|
|
62
|
+
return value.encode("ascii")[:length].ljust(length, b" ")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def load(fp: IO[bytes]) -> tuple[SdfHeader, np.ndarray, bytes]:
|
|
66
|
+
"""Parse a binary SDF file object opened in binary mode.
|
|
67
|
+
|
|
68
|
+
:param fp: File object to read from, positioned at the start of the file.
|
|
69
|
+
:returns: A ``(header, data, trailer)`` tuple. ``data`` is a
|
|
70
|
+
``(num_profiles, num_points)`` array of height values in metres,
|
|
71
|
+
with ``NaN`` marking non-measured or spurious points.
|
|
72
|
+
:raises SdfFormatError: If the file is malformed, e.g. a bad magic,
|
|
73
|
+
unsupported dialect or data type, wrong data length, or a trailer
|
|
74
|
+
that isn't 7-bit ASCII.
|
|
75
|
+
"""
|
|
76
|
+
magic = fp.read(MAGIC_SIZE)
|
|
77
|
+
if len(magic) != MAGIC_SIZE:
|
|
78
|
+
raise SdfFormatError("File is too short to contain an SDF header")
|
|
79
|
+
magic_text = magic.decode("ascii", errors="replace")
|
|
80
|
+
if (
|
|
81
|
+
len(magic_text) != MAGIC_SIZE
|
|
82
|
+
or magic_text[0] not in (ASCII_PREFIX, BINARY_PREFIX)
|
|
83
|
+
or magic_text[4] != "-"
|
|
84
|
+
):
|
|
85
|
+
raise SdfFormatError(f"Not an SDF file, unexpected magic: {magic_text!r}")
|
|
86
|
+
dialect_text = magic_text[1:4]
|
|
87
|
+
version_text = magic_text[_MAGIC_PREFIX_SIZE:]
|
|
88
|
+
dialect = SdfDialect.resolve(dialect_text, version_text, magic_text)
|
|
89
|
+
if magic_text[0] != BINARY_PREFIX:
|
|
90
|
+
raise SdfFormatError("ASCII magic found while parsing a binary SDF file")
|
|
91
|
+
|
|
92
|
+
manufacturer_raw, create_raw, mod_raw = _STRINGS_STRUCT.unpack(fp.read(_STRINGS_STRUCT.size))
|
|
93
|
+
counts_struct = _COUNTS_STRUCT[dialect]
|
|
94
|
+
num_points, num_profiles = counts_struct.unpack(fp.read(counts_struct.size))
|
|
95
|
+
(
|
|
96
|
+
x_scale,
|
|
97
|
+
y_scale,
|
|
98
|
+
z_scale,
|
|
99
|
+
z_resolution,
|
|
100
|
+
compression,
|
|
101
|
+
data_type_code,
|
|
102
|
+
check_type,
|
|
103
|
+
) = _TAIL_STRUCT.unpack(fp.read(_TAIL_STRUCT.size))
|
|
104
|
+
validate_compression(compression)
|
|
105
|
+
validate_check_type(check_type)
|
|
106
|
+
|
|
107
|
+
data_type = require_supported_data_type(data_type_code, dialect)
|
|
108
|
+
|
|
109
|
+
# errors="replace" here is a deliberate read-side leniency (unlike the
|
|
110
|
+
# trailer, ManufacID/dates are not re-validated as ASCII on read) so a
|
|
111
|
+
# file that's merely non-compliant here can still be opened and
|
|
112
|
+
# inspected; only *writing* one is rejected (validate_manufacturer_id_ascii).
|
|
113
|
+
header = SdfHeader(
|
|
114
|
+
dialect=dialect,
|
|
115
|
+
binary=True,
|
|
116
|
+
manufacturer_id=manufacturer_raw.decode("ascii", errors="replace").rstrip(),
|
|
117
|
+
create_date=parse_sdf_datetime(create_raw.decode("ascii", errors="replace"), dialect),
|
|
118
|
+
mod_date=parse_sdf_datetime(mod_raw.decode("ascii", errors="replace"), dialect),
|
|
119
|
+
num_points=num_points,
|
|
120
|
+
num_profiles=num_profiles,
|
|
121
|
+
x_scale=x_scale,
|
|
122
|
+
y_scale=y_scale,
|
|
123
|
+
z_scale=z_scale,
|
|
124
|
+
z_resolution=z_resolution,
|
|
125
|
+
data_type=data_type_code,
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
count = header.num_points * header.num_profiles
|
|
129
|
+
byte_count = count * data_type.dtype.itemsize
|
|
130
|
+
raw_bytes = fp.read(byte_count)
|
|
131
|
+
if len(raw_bytes) != byte_count:
|
|
132
|
+
raise SdfFormatError("Unexpected end of file while reading the SDF data area")
|
|
133
|
+
raw = np.frombuffer(raw_bytes, dtype=data_type.dtype)
|
|
134
|
+
data = decode_raw(raw, data_type, header.z_scale, dialect).reshape(
|
|
135
|
+
header.num_profiles, header.num_points
|
|
136
|
+
)
|
|
137
|
+
if dialect.reverses_profile_order:
|
|
138
|
+
data = data[::-1]
|
|
139
|
+
|
|
140
|
+
trailer = fp.read()
|
|
141
|
+
validate_trailer_ascii(trailer)
|
|
142
|
+
return header, data, trailer
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def dump(header: SdfHeader, data: np.ndarray, fp: IO[bytes], trailer: bytes = b"") -> None:
|
|
146
|
+
"""Serialize a header/data pair to the binary SDF representation.
|
|
147
|
+
|
|
148
|
+
:param header: Header describing ``data``; ``header.binary`` is ignored.
|
|
149
|
+
:param data: ``(num_profiles, num_points)`` array of height values in
|
|
150
|
+
metres. ``NaN`` marks non-measured or spurious points.
|
|
151
|
+
:param trailer: Optional raw record 3 trailer content.
|
|
152
|
+
"""
|
|
153
|
+
if data.shape != header.shape:
|
|
154
|
+
raise SdfFormatError(f"Data shape {data.shape} does not match header shape {header.shape}")
|
|
155
|
+
if header.dialect not in _COUNTS_STRUCT:
|
|
156
|
+
raise SdfFormatError(f"Unsupported SDF dialect {header.dialect!r}")
|
|
157
|
+
validate_z_scale(header.z_scale)
|
|
158
|
+
validate_manufacturer_id_ascii(header.manufacturer_id)
|
|
159
|
+
validate_trailer_ascii(trailer)
|
|
160
|
+
validate_trailer_xml(header.dialect, trailer)
|
|
161
|
+
data_type = require_supported_data_type(header.data_type, header.dialect)
|
|
162
|
+
|
|
163
|
+
fp.write(f"{BINARY_PREFIX}{header.dialect}".encode("ascii"))
|
|
164
|
+
fp.write(_pad(header.manufacturer_id, _MANUFACTURER_ID_SIZE))
|
|
165
|
+
fp.write(_pad(format_sdf_datetime(header.create_date, header.dialect), _DATE_SIZE))
|
|
166
|
+
fp.write(_pad(format_sdf_datetime(header.mod_date, header.dialect), _DATE_SIZE))
|
|
167
|
+
counts_struct = _COUNTS_STRUCT[header.dialect]
|
|
168
|
+
# counts_struct.size is NumPoints+NumProfiles combined (e.g. 4 bytes for
|
|
169
|
+
# ISO-1.0/BCR-1.0's two uint16 fields); halving it gives one field's byte
|
|
170
|
+
# width, and from that its max unsigned value (65535 for uint16, 2**32-1
|
|
171
|
+
# for uint32).
|
|
172
|
+
max_count = 2 ** (8 * (counts_struct.size // 2)) - 1
|
|
173
|
+
if not (0 <= header.num_points <= max_count and 0 <= header.num_profiles <= max_count):
|
|
174
|
+
raise SdfFormatError(
|
|
175
|
+
f"NumPoints/NumProfiles must be between 0 and {max_count} for SDF "
|
|
176
|
+
f"dialect {header.dialect} (binary format)"
|
|
177
|
+
)
|
|
178
|
+
fp.write(counts_struct.pack(header.num_points, header.num_profiles))
|
|
179
|
+
fp.write(
|
|
180
|
+
_TAIL_STRUCT.pack(
|
|
181
|
+
header.x_scale,
|
|
182
|
+
header.y_scale,
|
|
183
|
+
header.z_scale,
|
|
184
|
+
header.z_resolution,
|
|
185
|
+
0, # Compression: never supported, always written as 0.
|
|
186
|
+
header.data_type,
|
|
187
|
+
0, # CheckType: never written, always 0 (see SdfHeader's docstring).
|
|
188
|
+
)
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
write_data = data[::-1] if header.dialect.reverses_profile_order else data
|
|
192
|
+
raw = encode_raw(write_data, data_type, header.z_scale, header.dialect)
|
|
193
|
+
fp.write(raw.tobytes())
|
|
194
|
+
|
|
195
|
+
if trailer:
|
|
196
|
+
fp.write(trailer)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def loads(data: bytes) -> tuple[SdfHeader, np.ndarray, bytes]:
|
|
200
|
+
"""Parse an in-memory binary SDF blob.
|
|
201
|
+
|
|
202
|
+
Equivalent to :func:`load`, reading from an in-memory ``data`` buffer.
|
|
203
|
+
|
|
204
|
+
:returns: A ``(header, data, trailer)`` tuple. ``data`` is a
|
|
205
|
+
``(num_profiles, num_points)`` array of height values in metres,
|
|
206
|
+
with ``NaN`` marking non-measured or spurious points.
|
|
207
|
+
"""
|
|
208
|
+
return load(io.BytesIO(data))
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def dumps(header: SdfHeader, data: np.ndarray, trailer: bytes = b"") -> bytes:
|
|
212
|
+
"""Serialize a header/data pair to an in-memory binary SDF blob.
|
|
213
|
+
|
|
214
|
+
:param header: Header describing ``data``; ``header.binary`` is ignored.
|
|
215
|
+
:param data: ``(num_profiles, num_points)`` array of height values in
|
|
216
|
+
metres. ``NaN`` marks non-measured or spurious points.
|
|
217
|
+
:param trailer: Optional raw record 3 trailer content.
|
|
218
|
+
"""
|
|
219
|
+
fp = io.BytesIO()
|
|
220
|
+
dump(header, data, fp, trailer)
|
|
221
|
+
return fp.getvalue()
|
sdfio/_numeric.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: 2026 Thomas Ascher <thomas.ascher@gmx.at>
|
|
2
|
+
#
|
|
3
|
+
# SPDX-License-Identifier: MIT
|
|
4
|
+
|
|
5
|
+
"""Shared numeric formatting helpers for the SDF ASCII representation."""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
__all__ = ["format_scientific"]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def format_scientific(value: float, precision: int, exponent_width: int) -> str:
|
|
13
|
+
"""Format ``value`` in scientific notation with a fixed-width exponent.
|
|
14
|
+
|
|
15
|
+
Produces a normalized signed floating point number in scientific
|
|
16
|
+
notation with a fixed number of digits after the decimal point and a
|
|
17
|
+
zero-padded, signed exponent of fixed width, e.g. ``"3.141593e+00"``.
|
|
18
|
+
|
|
19
|
+
:param value: The number to format.
|
|
20
|
+
:param precision: Digits after the decimal point (6 for ``binary32``,
|
|
21
|
+
14 for ``binary64``).
|
|
22
|
+
:param exponent_width: Digits used for the exponent (2 for ``binary32``,
|
|
23
|
+
3 for ``binary64``).
|
|
24
|
+
|
|
25
|
+
The standard mandates 7 significant digits for ``binary32`` and 15 for
|
|
26
|
+
``binary64``. ``binary64`` (IEEE 754 double) needs 17 to round-trip
|
|
27
|
+
exactly, so formatting a ``binary64`` value this way and parsing it back
|
|
28
|
+
can differ by up to ~1 ULP at the 15th significant digit.
|
|
29
|
+
"""
|
|
30
|
+
# value + 0.0 normalizes -0.0 to +0.0 (IEEE 754: -0.0 + 0.0 == 0.0), since
|
|
31
|
+
# the standard's zero format ("0.000000e+00") has no sign to preserve.
|
|
32
|
+
mantissa, exponent = f"{value + 0.0:.{precision}e}".split("e")
|
|
33
|
+
exponent_value = int(exponent)
|
|
34
|
+
sign = "+" if exponent_value >= 0 else "-"
|
|
35
|
+
return f"{mantissa}e{sign}{abs(exponent_value):0{exponent_width}d}"
|