specmod 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. specmod/__init__.py +17 -0
  2. specmod/_vendor/__init__.py +21 -0
  3. specmod/_vendor/qiinv.py +243 -0
  4. specmod/acquire.py +358 -0
  5. specmod/api.py +480 -0
  6. specmod/cli.py +139 -0
  7. specmod/config/__init__.py +44 -0
  8. specmod/config/layers.py +168 -0
  9. specmod/config/provenance.py +77 -0
  10. specmod/config/sections.py +385 -0
  11. specmod/config/serialize.py +58 -0
  12. specmod/core/__init__.py +41 -0
  13. specmod/core/bandwidth.py +187 -0
  14. specmod/core/collection.py +549 -0
  15. specmod/core/noise.py +478 -0
  16. specmod/core/scalogram.py +234 -0
  17. specmod/core/spectrum.py +326 -0
  18. specmod/core/units.py +116 -0
  19. specmod/datasets.py +316 -0
  20. specmod/distance.py +190 -0
  21. specmod/exceptions.py +58 -0
  22. specmod/fitting/__init__.py +58 -0
  23. specmod/fitting/base.py +50 -0
  24. specmod/fitting/event.py +284 -0
  25. specmod/fitting/guess.py +170 -0
  26. specmod/fitting/spectrum.py +330 -0
  27. specmod/io.py +241 -0
  28. specmod/magnitude.py +312 -0
  29. specmod/picks/__init__.py +182 -0
  30. specmod/picks/base.py +250 -0
  31. specmod/picks/delimited.py +224 -0
  32. specmod/picks/events.py +157 -0
  33. specmod/picks/resolution.py +149 -0
  34. specmod/picks/snuffler.py +92 -0
  35. specmod/pipeline.py +280 -0
  36. specmod/plotting.py +203 -0
  37. specmod/preprocess.py +554 -0
  38. specmod/smoothing/__init__.py +50 -0
  39. specmod/smoothing/base.py +56 -0
  40. specmod/smoothing/konno_ohmachi.py +83 -0
  41. specmod/smoothing/log_bins.py +171 -0
  42. specmod/sources/__init__.py +65 -0
  43. specmod/sources/attenuation.py +110 -0
  44. specmod/sources/composite.py +135 -0
  45. specmod/sources/motion.py +40 -0
  46. specmod/sources/source.py +147 -0
  47. specmod/spreading.py +209 -0
  48. specmod/staged.py +523 -0
  49. specmod/tables.py +110 -0
  50. specmod/transforms/__init__.py +50 -0
  51. specmod/transforms/base.py +242 -0
  52. specmod/transforms/cwt.py +219 -0
  53. specmod/transforms/fft.py +157 -0
  54. specmod/transforms/multitaper.py +357 -0
  55. specmod/transforms/prieto.py +272 -0
  56. specmod/transforms/quadratic.py +221 -0
  57. specmod/utils.py +305 -0
  58. specmod-0.2.0.dist-info/METADATA +294 -0
  59. specmod-0.2.0.dist-info/RECORD +62 -0
  60. specmod-0.2.0.dist-info/WHEEL +4 -0
  61. specmod-0.2.0.dist-info/entry_points.txt +2 -0
  62. specmod-0.2.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,168 @@
1
+ """Layered configuration resolution.
2
+
3
+ Precedence, lowest to highest:
4
+
5
+ 1. Package defaults (:mod:`specmod.config.sections`)
6
+ 2. A committed project config, ``specmod.toml``
7
+ 3. A local override, ``specmod.local.toml`` — gitignored by default
8
+ 4. Environment variables, ``SPECMOD_<SECTION>__<KEY>``
9
+ 5. Explicit keyword arguments
10
+
11
+ Local overrides are deliberately uncommitted, which would make a run
12
+ irreproducible from the repository alone. That is resolved by recording the
13
+ *resolved* configuration in every output rather than by forbidding local files;
14
+ see :mod:`specmod.config.provenance`.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import ast
20
+ import os
21
+ import tomllib
22
+ from dataclasses import fields
23
+ from pathlib import Path
24
+ from typing import Any
25
+
26
+ from .sections import Config
27
+
28
+ __all__ = [
29
+ "LAYER_NAMES",
30
+ "ResolvedConfig",
31
+ "load_config",
32
+ ]
33
+
34
+ PROJECT_FILE = "specmod.toml"
35
+ LOCAL_FILE = "specmod.local.toml"
36
+ ENV_PREFIX = "SPECMOD_"
37
+
38
+ LAYER_NAMES = ("default", PROJECT_FILE, LOCAL_FILE, "environment", "arguments")
39
+
40
+
41
+ class ResolvedConfig:
42
+ """A :class:`Config` plus the layer each value came from.
43
+
44
+ The provenance is what makes ``specmod config show`` able to answer "why
45
+ did this run differ", which is otherwise guesswork once local overrides and
46
+ environment variables are in play.
47
+ """
48
+
49
+ def __init__(self, config: Config, sources: dict[str, str]) -> None:
50
+ self.config = config
51
+ #: Maps ``"section.key"`` to the name of the layer that set it.
52
+ self.sources = sources
53
+
54
+ def source_of(self, dotted: str) -> str:
55
+ return self.sources.get(dotted, "default")
56
+
57
+ def explain(self) -> str:
58
+ """Render the resolved config with the origin of every value."""
59
+ lines: list[str] = []
60
+ data = self.config.to_dict()
61
+ for section in sorted(data):
62
+ lines.append(f"[{section}]")
63
+ for key in sorted(data[section]):
64
+ dotted = f"{section}.{key}"
65
+ origin = self.source_of(dotted)
66
+ marker = "" if origin == "default" else f" <- {origin}"
67
+ lines.append(f" {key} = {data[section][key]!r}{marker}")
68
+ lines.append("")
69
+ return "\n".join(lines)
70
+
71
+
72
+ def _read_toml(path: Path) -> dict[str, Any]:
73
+ with path.open("rb") as fh:
74
+ return tomllib.load(fh)
75
+
76
+
77
+ def _env_overrides() -> dict[str, dict[str, Any]]:
78
+ """Collect ``SPECMOD_SNR__TOLERANCE=4`` style overrides.
79
+
80
+ Values are parsed as Python literals where possible so numbers and booleans
81
+ do not arrive as strings; anything else is left as text.
82
+ """
83
+ out: dict[str, dict[str, Any]] = {}
84
+ valid = {f.name for f in fields(Config)}
85
+ for raw_key, raw_value in os.environ.items():
86
+ if not raw_key.startswith(ENV_PREFIX) or "__" not in raw_key:
87
+ continue
88
+ section, _, key = raw_key[len(ENV_PREFIX) :].partition("__")
89
+ section, key = section.lower(), key.lower()
90
+ if section not in valid:
91
+ continue
92
+ try:
93
+ value: Any = ast.literal_eval(raw_value)
94
+ except (ValueError, SyntaxError):
95
+ value = raw_value
96
+ out.setdefault(section, {})[key] = value
97
+ return out
98
+
99
+
100
+ def _merge(
101
+ base: dict[str, dict[str, Any]],
102
+ incoming: dict[str, Any],
103
+ layer: str,
104
+ sources: dict[str, str],
105
+ ) -> None:
106
+ for section, values in incoming.items():
107
+ if not isinstance(values, dict):
108
+ raise ValueError(
109
+ f"Configuration section [{section}] must be a table, got "
110
+ f"{type(values).__name__}."
111
+ )
112
+ for key, value in values.items():
113
+ base.setdefault(section, {})[key] = value
114
+ sources[f"{section}.{key}"] = layer
115
+
116
+
117
+ def load_config(
118
+ start: Path | str | None = None,
119
+ *,
120
+ project_file: Path | str | None = None,
121
+ use_local: bool = True,
122
+ use_env: bool = True,
123
+ **overrides: dict[str, Any],
124
+ ) -> ResolvedConfig:
125
+ """Resolve configuration through all layers.
126
+
127
+ Parameters
128
+ ----------
129
+ start
130
+ Directory to search for config files. Defaults to the current
131
+ directory. The search does not walk upwards — an implicit parent search
132
+ makes it unclear which file a run actually used.
133
+ project_file
134
+ Explicit path to a committed config, bypassing the search. This is how
135
+ a study config (``studies/magna_2020_paper.toml``) is applied, and how
136
+ tests pin an explicit configuration rather than inheriting defaults.
137
+ use_local, use_env
138
+ Disable the local-file and environment layers. Tests set both to False
139
+ so a developer's machine cannot influence a result.
140
+ **overrides
141
+ Section-keyed dicts, e.g. ``snr={"tolerance": 4}``. Highest precedence.
142
+ """
143
+ root = Path(start) if start is not None else Path.cwd()
144
+ merged: dict[str, dict[str, Any]] = {}
145
+ sources: dict[str, str] = {}
146
+
147
+ if project_file is not None:
148
+ path = Path(project_file)
149
+ if not path.is_file():
150
+ raise FileNotFoundError(f"Config file not found: {path}")
151
+ _merge(merged, _read_toml(path), str(path), sources)
152
+ else:
153
+ candidate = root / PROJECT_FILE
154
+ if candidate.is_file():
155
+ _merge(merged, _read_toml(candidate), PROJECT_FILE, sources)
156
+
157
+ if use_local:
158
+ local = root / LOCAL_FILE
159
+ if local.is_file():
160
+ _merge(merged, _read_toml(local), LOCAL_FILE, sources)
161
+
162
+ if use_env:
163
+ _merge(merged, _env_overrides(), "environment", sources)
164
+
165
+ if overrides:
166
+ _merge(merged, overrides, "arguments", sources)
167
+
168
+ return ResolvedConfig(Config.from_dict(merged), sources)
@@ -0,0 +1,77 @@
1
+ """Provenance stamping: what makes a locally-overridden run reproducible.
2
+
3
+ Local config files are uncommitted by design, so the repository alone cannot
4
+ say how a given result was produced. The output can. Every artifact SpecMod
5
+ writes carries the fully resolved configuration, a short hash of it, and the
6
+ SpecMod version.
7
+
8
+ The version matters as much as the config. Defaults move between releases, so
9
+ without it "reproducible" fails silently across an upgrade — which is exactly
10
+ what made identifying the code behind the published Magna run so difficult (see
11
+ ``docs/REFACTOR_PLAN.md`` §5.2.5).
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import hashlib
17
+ import json
18
+ from dataclasses import dataclass
19
+ from datetime import UTC, datetime
20
+ from typing import Any
21
+
22
+ from .. import __version__
23
+ from .sections import Config
24
+
25
+ __all__ = ["Provenance", "config_hash"]
26
+
27
+
28
+ def _canonical(config: Config) -> str:
29
+ """Serialise deterministically so the hash is stable across runs."""
30
+ return json.dumps(config.to_dict(), sort_keys=True, separators=(",", ":"))
31
+
32
+
33
+ def config_hash(config: Config, *, length: int = 12) -> str:
34
+ """Short, stable digest of a configuration.
35
+
36
+ Comparing two runs starts here: same hash means same settings, so any
37
+ difference is in the data or the code, not the configuration.
38
+ """
39
+ return hashlib.sha256(_canonical(config).encode()).hexdigest()[:length]
40
+
41
+
42
+ @dataclass(frozen=True, slots=True)
43
+ class Provenance:
44
+ """The record attached to every output."""
45
+
46
+ specmod_version: str
47
+ config: dict[str, Any]
48
+ config_hash: str
49
+ created_at: str
50
+ sources: dict[str, str]
51
+
52
+ @classmethod
53
+ def capture(
54
+ cls,
55
+ config: Config,
56
+ *,
57
+ sources: dict[str, str] | None = None,
58
+ ) -> Provenance:
59
+ return cls(
60
+ specmod_version=__version__,
61
+ config=config.to_dict(),
62
+ config_hash=config_hash(config),
63
+ created_at=datetime.now(UTC).isoformat(timespec="seconds"),
64
+ sources=dict(sources or {}),
65
+ )
66
+
67
+ def to_dict(self) -> dict[str, Any]:
68
+ return {
69
+ "specmod_version": self.specmod_version,
70
+ "config_hash": self.config_hash,
71
+ "created_at": self.created_at,
72
+ "config": self.config,
73
+ "config_sources": self.sources,
74
+ }
75
+
76
+ def to_json(self, *, indent: int = 2) -> str:
77
+ return json.dumps(self.to_dict(), indent=indent, sort_keys=True)
@@ -0,0 +1,385 @@
1
+ """Configuration sections, grouped semantically by pipeline stage.
2
+
3
+ Each section is a frozen dataclass owned by the stage it configures, replacing
4
+ the flat dicts in the old ``config.py`` and the parameters that were previously
5
+ reachable only as function defaults — or, in the case of the multitaper
6
+ time-bandwidth product, not reachable at all.
7
+
8
+ Defaults here reproduce the behaviour shipped before this refactor, not the
9
+ values used for the published Magna run. Those live in
10
+ ``studies/magna_2020_paper.toml``. See ``docs/REFACTOR_PLAN.md`` §4.7 and §5.2.5.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from dataclasses import dataclass, field, fields, is_dataclass
16
+ from typing import Any, Literal, Self
17
+
18
+ __all__ = [
19
+ "AcquireConfig",
20
+ "Config",
21
+ "FittingConfig",
22
+ "ModelConfig",
23
+ "SmoothingConfig",
24
+ "SnrConfig",
25
+ "TransformConfig",
26
+ "VizConfig",
27
+ "WindowsConfig",
28
+ ]
29
+
30
+
31
+ @dataclass(frozen=True, slots=True)
32
+ class AcquireConfig:
33
+ """Waveform acquisition. Consumed by :mod:`specmod.acquire`."""
34
+
35
+ #: FDSN data centre. Different centres serve different holdings for the
36
+ #: same event, so this is part of the provenance record, not a detail.
37
+ client: str = "IRIS"
38
+ event_id: str | None = None
39
+ #: Fallback when ``event_id`` is not given: explicit origin.
40
+ origin_time: str | None = None
41
+ latitude: float | None = None
42
+ longitude: float | None = None
43
+ depth_km: float | None = None
44
+ magnitude: float | None = None
45
+
46
+ networks: tuple[str, ...] = ("*",)
47
+ stations: tuple[str, ...] = ("*",)
48
+ locations: tuple[str, ...] = ("*",)
49
+ channels: tuple[str, ...] = ("HH?", "BH?", "EN?")
50
+ max_radius_km: float = 400.0
51
+
52
+ #: Seconds either side of origin to request.
53
+ seconds_before: float = 60.0
54
+ seconds_after: float = 300.0
55
+
56
+ #: Store raw counts plus the response rather than a deconvolved trace, so
57
+ #: response removal stays under test and no ObsPy version is baked in.
58
+ remove_response: bool = False
59
+
60
+
61
+ @dataclass(frozen=True, slots=True)
62
+ class WindowsConfig:
63
+ """Phase arrivals and signal/noise window construction."""
64
+
65
+ #: Group velocities (km/s) for theoretical arrivals. The published Magna
66
+ #: run used s=3.4; 2.9 is the shipped default and is kept as such.
67
+ p_velocity: float = 5.9
68
+ s_velocity: float = 2.9
69
+
70
+ #: Used when an S pick is missing: s_time = p_time + emergency_ratio * (p - o).
71
+ emergency_ratio: float = 1.7
72
+
73
+ #: S window: opens at ``s_start_ratio`` of the P-S time, runs ``s_length``.
74
+ s_start_ratio: float = 0.8
75
+ s_length: float = 20.0
76
+ s_length_mode: Literal["absolute_time", "relative_ps"] = "absolute_time"
77
+
78
+ #: P window.
79
+ p_before: float = 0.0
80
+ p_length: float = 0.8
81
+ p_length_mode: Literal["absolute_time", "relative_time"] = "relative_time"
82
+
83
+ #: Refine windows to percentiles of the cumulative squared-amplitude
84
+ #: integral. This is step 5 of the published Magna workflow.
85
+ refine: bool = True
86
+ refine_percentiles: tuple[float, float] = (1.0, 99.0)
87
+
88
+ #: Noise window ends this many seconds before the P arrival. The published
89
+ #: run used 0.5; 0.2 is the shipped default.
90
+ noise_shift: float = 0.2
91
+ noise_length: float = 1.0
92
+
93
+ pad_seconds: float = 0.0
94
+ pad_value: float = 0.0
95
+ station_shifts: dict[str, float] = field(default_factory=dict)
96
+
97
+
98
+ @dataclass(frozen=True, slots=True)
99
+ class TransformConfig:
100
+ """Time-to-frequency conversion. Consumed by :mod:`specmod.transforms`."""
101
+
102
+ estimator: Literal[
103
+ "multitaper", "fft", "welch", "cwt", "prieto", "quadratic", "mtspec"
104
+ ] = "multitaper"
105
+
106
+ #: Multitaper. ``time_bandwidth`` was previously the literal 3 passed
107
+ #: positionally to mtspec, with no way to configure it.
108
+ time_bandwidth: float = 3.0
109
+ n_tapers: int = 5
110
+ #: On by default: leakage suppression is the point of multitaper, and flat
111
+ #: weighting leaves the high-frequency floor ~287x high under a strong
112
+ #: low-frequency peak. See specmod.transforms.multitaper.
113
+ adaptive: bool = True
114
+ #: Rescale the spectrum to integrate to the record variance, as mtspec and
115
+ #: Prieto's multitaper do. Needed to reproduce pre-refactor results; off by
116
+ #: default because it makes the Parseval check circular.
117
+ normalize_to_variance: bool = False
118
+
119
+ #: FFT / Welch.
120
+ taper: Literal["hann", "tukey", "boxcar"] = "tukey"
121
+ taper_alpha: float = 0.05
122
+ #: ``None`` for no padding, an integer, or "fast"/"pow2". Padding is a pure
123
+ #: interpolation here -- the normalisation is keyed off duration, not
124
+ #: len(freq), which is what the old psd_to_amp got wrong. Use "fast" to
125
+ #: avoid the slow FFT path on prime-length cut windows.
126
+ n_fft: int | str | None = None
127
+ welch_segment_length: int | None = None
128
+
129
+ #: Continuous wavelet transform.
130
+ wavelet: Literal["morlet"] = "morlet"
131
+ omega0: float = 6.0
132
+ #: Scale resolution: voices per octave.
133
+ dj: float = 0.125
134
+ mask_coi: bool = True
135
+
136
+ #: Drop the DC bin. The old code did this unconditionally and *before*
137
+ #: using len(freq) for normalisation, biasing amplitudes slightly.
138
+ drop_dc: bool = True
139
+
140
+
141
+ @dataclass(frozen=True, slots=True)
142
+ class SmoothingConfig:
143
+ """Spectral smoothing and log-space binning."""
144
+
145
+ method: Literal["log_bins", "konno_ohmachi", "none"] = "log_bins"
146
+
147
+ #: Log bin edges. ``None`` derives them from the record: fmin from 1/T,
148
+ #: fmax from Nyquist. The old code hardcoded 0.001-200 Hz regardless.
149
+ f_min: float | None = 0.001
150
+ f_max: float | None = 200.0
151
+ n_bins: int = 151
152
+
153
+ #: Konno-Ohmachi bandwidth ``b``. Smaller smooths harder.
154
+ konno_ohmachi_bandwidth: float = 40.0
155
+
156
+
157
+ @dataclass(frozen=True, slots=True)
158
+ class SnrConfig:
159
+ """Signal-to-noise assessment and usable bandwidth selection."""
160
+
161
+ tolerance: float = 3.0
162
+ min_points: int = 10
163
+
164
+ #: Require SNR above ``tolerance`` in every band. The published Magna run
165
+ #: used this as its selection criterion; it ships disabled.
166
+ assert_bandwidths: bool = False
167
+ bands: tuple[tuple[float, float], ...] = ((2.0, 4.0), (4.0, 6.0), (6.0, 8.0))
168
+
169
+ #: Scale noise amplitude by sqrt(len(signal)/len(noise)) when the noise
170
+ #: window is shorter than the signal window.
171
+ scale_parseval: bool = True
172
+ interpolate_noise: bool = True
173
+
174
+ #: Names come from :data:`specmod.core.bandwidth.BANDWIDTH_SELECTORS`.
175
+ #: This said ``"integral"`` while the registry said ``"widest"``, from the
176
+ #: period when the selector was still a percentile of a sign integral —
177
+ #: a config value that named nothing the code would accept.
178
+ bandwidth_method: Literal["widest", "peak"] = "peak"
179
+
180
+ #: Impose a low-frequency floor from the window length (~1/T), or the cone
181
+ #: of influence when the spectrum came from a CWT. Nothing enforced this
182
+ #: before, so a short window could report bandwidth it could not resolve.
183
+ resolution_floor: bool = True
184
+
185
+ rotate_noise: bool = True
186
+ rotation_method: Literal["rotate", "boost"] = "boost"
187
+ rotation_increment: float = 0.05
188
+ rotation_space: tuple[float, float] = (1e-3, 1.001)
189
+
190
+
191
+ @dataclass(frozen=True, slots=True)
192
+ class ModelConfig:
193
+ """Source and attenuation model."""
194
+
195
+ source: Literal["brune", "boatwright"] = "brune"
196
+ #: The motion the *model* is expressed in. Once Spectrum carries its own
197
+ #: motion (§4.2) this is a default rather than a global that must be kept
198
+ #: in sync by hand.
199
+ motion: Literal["displacement", "velocity", "acceleration"] = "velocity"
200
+ frequency_dependent_attenuation: bool = False
201
+
202
+
203
+ @dataclass(frozen=True, slots=True)
204
+ class FittingConfig:
205
+ """Minimisation."""
206
+
207
+ method: str = "powell"
208
+ fit_bins: bool = False
209
+ weight_method: Literal["none", "log"] = "none"
210
+
211
+ #: Initial guesses that were hardcoded in ModelGuess.
212
+ initial_t_star: float = 0.01
213
+ initial_alpha: float = 1e-5
214
+
215
+ #: Lower bounds on the fitted parameters that have one physically.
216
+ #:
217
+ #: A negative ``t*`` says the wave gained energy travelling; a corner
218
+ #: frequency at or below zero is not a poor measurement but a meaningless
219
+ #: one. Neither is prevented by the misfit surface, and lmfit will return
220
+ #: either if the surface leans that way — with the shipped multitaper
221
+ #: default it returned ``fc = -4.45 Hz`` on one PNR station.
222
+ #:
223
+ #: Zero rather than a small positive number, deliberately: a parameter that
224
+ #: lands *on* its bound is flagged by ``pass_fitting``, so the fit is
225
+ #: rejected rather than reported as a corner frequency of nothing.
226
+ t_star_min: float = 1e-4
227
+ corner_frequency_min: float = 0.0
228
+
229
+ #: The two-stage event fit; see :mod:`specmod.staged`.
230
+ #:
231
+ #: One spectrum cannot separate the source corner from the path
232
+ #: attenuation — they trade off on the falling limb — so the corner is
233
+ #: determined by the ensemble and then held fixed while each station
234
+ #: refits the rest. ``event_parameter`` is what the ensemble decides.
235
+ #: ``"fc"`` because that is the term belonging to the source; ``"ts"`` is
236
+ #: the meaningful alternative for a study with an independent handle on Q.
237
+ event_parameter: str = "fc"
238
+ #: How stations are weighted into the event value. The published choice is
239
+ #: inverse hypocentral distance: the nearer station has less path, so less
240
+ #: of its falloff can be attenuation. See ``specmod.staged.WEIGHT_MODELS``.
241
+ event_weighting: str = "inverse_distance"
242
+
243
+ #: Which channels contribute to the event value, as shell globs matched
244
+ #: against the trace id and each of its SEED components — so ``"AQ07"``
245
+ #: means the station, ``"HHE"`` means the component, ``"UR"`` means the
246
+ #: network. Empty ``include`` means "everything not excluded".
247
+ #:
248
+ #: These exist to be edited *after* looking at a first pass. Quality
249
+ #: control is a judgement — a clipped record, a bad response, a pick on
250
+ #: the wrong phase — and a station that is confidently wrong moves the
251
+ #: event value for every other station. Putting the decision in the study
252
+ #: file is what makes it part of the record rather than something done in
253
+ #: a notebook and forgotten.
254
+ include: tuple[str, ...] = ()
255
+ exclude: tuple[str, ...] = ()
256
+ #: Drop a station whose stage-1 fit ended with a parameter pinned against
257
+ #: one of its bounds. The value reported there is the bound rather than a
258
+ #: measurement, so averaging it in is averaging in a constant.
259
+ require_pass: bool = True
260
+
261
+
262
+ @dataclass(frozen=True, slots=True)
263
+ class VizConfig:
264
+ """Plotting.
265
+
266
+ ``PLOT_COLUMNS`` was previously defined in *both* the SPECTRAL and FITTING
267
+ dicts, and the two copies could disagree. One home makes that impossible.
268
+ """
269
+
270
+ plot_columns: int = 3
271
+
272
+
273
+ @dataclass(frozen=True, slots=True)
274
+ class GeometryConfig:
275
+ """Source-to-site geometry.
276
+
277
+ Its own section because more than one stage needs it. Distance feeds the
278
+ ensemble weighting of the two-stage fit (:mod:`specmod.staged`) and the
279
+ geometric spreading a moment calculation corrects for, and a setting two
280
+ consumers each keep their own copy of is how the two come to disagree.
281
+
282
+ It lived in ``[windows]`` until there was a second reader, which was the
283
+ wrong home even then: cutting a window does not depend on how distance is
284
+ measured.
285
+ """
286
+
287
+ #: Which distance, resolved through :data:`specmod.distance.DISTANCE_MEASURES`.
288
+ #:
289
+ #: ``repi`` is the default and is the honest one wherever sensor depths are
290
+ #: not known. ``rhyp`` is built from the source depth and the station
291
+ #: *elevation*, so it assumes every sensor sits at the surface — for a
292
+ #: borehole deployment that is wrong by the burial depth, and nothing in
293
+ #: the metadata says so.
294
+ #:
295
+ #: The choice is not a detail at short range: on the PNR data the nearest
296
+ #: station is 1.02 km epicentral against 2.30 km hypocentral, a factor of
297
+ #: 2.24, while the farthest agree to 1.004 — so anything weighted by inverse
298
+ #: distance is most sensitive to it exactly where it matters most.
299
+ #:
300
+ #: ``rrup`` and ``rjb`` are registered and raise: both need a rupture
301
+ #: surface, and for a point source they degenerate to ``rhyp`` and ``repi``.
302
+ distance_measure: str = "repi"
303
+
304
+
305
+ @dataclass(frozen=True, slots=True)
306
+ class Config:
307
+ """The whole resolved configuration."""
308
+
309
+ acquire: AcquireConfig = field(default_factory=AcquireConfig)
310
+ windows: WindowsConfig = field(default_factory=WindowsConfig)
311
+ geometry: GeometryConfig = field(default_factory=GeometryConfig)
312
+ transform: TransformConfig = field(default_factory=TransformConfig)
313
+ smoothing: SmoothingConfig = field(default_factory=SmoothingConfig)
314
+ snr: SnrConfig = field(default_factory=SnrConfig)
315
+ model: ModelConfig = field(default_factory=ModelConfig)
316
+ fitting: FittingConfig = field(default_factory=FittingConfig)
317
+ viz: VizConfig = field(default_factory=VizConfig)
318
+
319
+ def to_dict(self) -> dict[str, Any]:
320
+ """Return a plain nested dict, suitable for TOML or JSON."""
321
+ result: dict[str, Any] = _as_dict(self)
322
+ return result
323
+
324
+ @classmethod
325
+ def from_dict(cls, data: dict[str, Any]) -> Self:
326
+ """Build a Config from a nested mapping, validating section names."""
327
+ known = {f.name: f.type for f in fields(cls)}
328
+ unknown = set(data) - set(known)
329
+ if unknown:
330
+ raise ValueError(
331
+ f"Unknown configuration section(s): {sorted(unknown)}. "
332
+ f"Valid sections are {sorted(known)}."
333
+ )
334
+ kwargs: dict[str, Any] = {}
335
+ for name, f in ((f.name, f) for f in fields(cls)):
336
+ if name in data:
337
+ kwargs[name] = _build_section(f.default_factory(), data[name], name) # type: ignore[misc]
338
+ return cls(**kwargs)
339
+
340
+
341
+ def _build_section(default: Any, values: dict[str, Any], section: str) -> Any:
342
+ """Construct one section, rejecting unknown keys loudly.
343
+
344
+ A silently ignored typo in a config file is a reproducibility bug: the run
345
+ looks configured and is not.
346
+ """
347
+ valid = {f.name for f in fields(default)}
348
+ unknown = set(values) - valid
349
+ if unknown:
350
+ raise ValueError(
351
+ f"Unknown key(s) in [{section}]: {sorted(unknown)}. "
352
+ f"Valid keys are {sorted(valid)}."
353
+ )
354
+ # Use the live attribute values, not _as_dict: that flattens tuples to
355
+ # lists for serialisation, and feeding them back in would rebuild the
356
+ # section with list-valued fields that compare unequal and are unhashable.
357
+ defaults = {f.name: getattr(default, f.name) for f in fields(default)}
358
+ coerced = {k: _coerce(defaults.get(k), v) for k, v in values.items()}
359
+ return type(default)(**{**defaults, **coerced})
360
+
361
+
362
+ def _coerce(default_value: Any, incoming: Any) -> Any:
363
+ """Reshape a TOML value to match the default's container types.
364
+
365
+ TOML has arrays but no tuples, so every sequence arrives as a list. The
366
+ dataclasses use tuples for immutability, and the nesting has to be matched
367
+ all the way down — ``bands`` is a tuple *of tuples*, and coercing only the
368
+ outer level leaves inner lists that compare unequal and break hashing.
369
+ """
370
+ if isinstance(default_value, tuple) and isinstance(incoming, (list, tuple)):
371
+ inner = default_value[0] if default_value else None
372
+ return tuple(_coerce(inner, item) for item in incoming)
373
+ if isinstance(default_value, dict) and isinstance(incoming, dict):
374
+ return dict(incoming)
375
+ return incoming
376
+
377
+
378
+ def _as_dict(obj: Any) -> Any:
379
+ if is_dataclass(obj) and not isinstance(obj, type):
380
+ return {f.name: _as_dict(getattr(obj, f.name)) for f in fields(obj)}
381
+ if isinstance(obj, tuple):
382
+ return [_as_dict(v) for v in obj]
383
+ if isinstance(obj, dict):
384
+ return {k: _as_dict(v) for k, v in obj.items()}
385
+ return obj
@@ -0,0 +1,58 @@
1
+ """Minimal TOML writer for configuration.
2
+
3
+ Python has a TOML reader in the standard library (``tomllib``) but no writer,
4
+ and the configuration schema is narrow enough — sections of scalars, flat
5
+ sequences, and one string-keyed float mapping — that a dependency is not worth
6
+ it.
7
+
8
+ TOML has no null. Keys whose value is ``None`` are written as commented-out
9
+ placeholders so a frozen file still documents that the option exists.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from typing import Any
15
+
16
+ from .sections import Config
17
+
18
+ __all__ = ["to_toml"]
19
+
20
+
21
+ def _fmt(value: Any) -> str:
22
+ if isinstance(value, bool): # before int — bool is a subclass of int
23
+ return "true" if value else "false"
24
+ if isinstance(value, str):
25
+ return '"' + value.replace("\\", "\\\\").replace('"', '\\"') + '"'
26
+ if isinstance(value, (int, float)):
27
+ return repr(value)
28
+ if isinstance(value, (list, tuple)):
29
+ return "[" + ", ".join(_fmt(v) for v in value) + "]"
30
+ if isinstance(value, dict):
31
+ return "{" + ", ".join(f"{k} = {_fmt(v)}" for k, v in value.items()) + "}"
32
+ raise TypeError(f"Cannot serialise {type(value).__name__} to TOML: {value!r}")
33
+
34
+
35
+ def to_toml(config: Config, *, header: str | None = None) -> str:
36
+ """Render a configuration as TOML.
37
+
38
+ The output round-trips through :meth:`Config.from_dict`, so a frozen file
39
+ reproduces the configuration it was frozen from.
40
+ """
41
+ lines: list[str] = []
42
+ if header:
43
+ lines.extend(f"# {line}" for line in header.splitlines())
44
+ lines.append("")
45
+
46
+ data = config.to_dict()
47
+ for section in sorted(data):
48
+ lines.append(f"[{section}]")
49
+ for key in sorted(data[section]):
50
+ value = data[section][key]
51
+ if value is None:
52
+ lines.append(f"# {key} = ") # TOML has no null
53
+ elif isinstance(value, dict) and not value:
54
+ lines.append(f"{key} = {{}}")
55
+ else:
56
+ lines.append(f"{key} = {_fmt(value)}")
57
+ lines.append("")
58
+ return "\n".join(lines).rstrip() + "\n"