audio-as-code 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- audio_as_code/__init__.py +44 -0
- audio_as_code/__main__.py +3 -0
- audio_as_code/_audio.py +87 -0
- audio_as_code/_export_rules.py +32 -0
- audio_as_code/_orchestra_profiles.py +188 -0
- audio_as_code/_paths.py +36 -0
- audio_as_code/_voices.py +173 -0
- audio_as_code/acoustics.py +126 -0
- audio_as_code/automation.py +40 -0
- audio_as_code/cli.py +171 -0
- audio_as_code/demo.py +106 -0
- audio_as_code/effects.py +85 -0
- audio_as_code/extended.py +283 -0
- audio_as_code/inspection.py +351 -0
- audio_as_code/instruments.py +923 -0
- audio_as_code/midi.py +192 -0
- audio_as_code/model.py +292 -0
- audio_as_code/orchestra.py +481 -0
- audio_as_code/pattern.py +151 -0
- audio_as_code/physical.py +189 -0
- audio_as_code/py.typed +0 -0
- audio_as_code/render.py +230 -0
- audio_as_code-0.1.0.dist-info/METADATA +340 -0
- audio_as_code-0.1.0.dist-info/RECORD +27 -0
- audio_as_code-0.1.0.dist-info/WHEEL +4 -0
- audio_as_code-0.1.0.dist-info/entry_points.txt +2 -0
- audio_as_code-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
"""Sample-free physical/modal approximations, implemented with NumPy only.
|
|
2
|
+
|
|
3
|
+
The string uses a passive Karplus-Strong feedback loop with phase-compensated
|
|
4
|
+
fractional delay. Bar and bell presets sum the impulse responses of damped modes.
|
|
5
|
+
All excitations are generated here; no recordings or measured impulse responses.
|
|
6
|
+
See docs/synthesis.md for equations, scope, and references.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import math
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from types import MappingProxyType
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
from numpy.typing import NDArray
|
|
17
|
+
|
|
18
|
+
from .acoustics import colored_noise, nyquist_gain, resonant_body, struck_mode
|
|
19
|
+
from .instruments import require_instrument
|
|
20
|
+
from .model import Tone
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True)
|
|
24
|
+
class ModalProfile:
|
|
25
|
+
ratios: tuple[float, ...]
|
|
26
|
+
amplitudes: tuple[float, ...]
|
|
27
|
+
lifetimes: tuple[float, ...]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
_MODAL_PROFILES = MappingProxyType(
|
|
31
|
+
{
|
|
32
|
+
# Approximate tuned wooden bar; higher modes die away much sooner.
|
|
33
|
+
"marimba": ModalProfile((1, 4, 10, 17), (1, 0.32, 0.12, 0.04), (1, 0.32, 0.14, 0.08)),
|
|
34
|
+
# Designed inharmonic chime, not a model fitted to a particular bell.
|
|
35
|
+
"bell": ModalProfile(
|
|
36
|
+
(1, 2, 2.756, 4.07, 5.404, 6.81),
|
|
37
|
+
(1, 0.38, 0.32, 0.22, 0.12, 0.08),
|
|
38
|
+
(1, 0.8, 0.65, 0.5, 0.4, 0.3),
|
|
39
|
+
),
|
|
40
|
+
}
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _string(
|
|
45
|
+
frequency: float,
|
|
46
|
+
frames: int,
|
|
47
|
+
rate: int,
|
|
48
|
+
seed: int,
|
|
49
|
+
brightness: float,
|
|
50
|
+
decay: float,
|
|
51
|
+
position: float,
|
|
52
|
+
) -> NDArray[np.float64]:
|
|
53
|
+
omega = 2 * math.pi * frequency / rate
|
|
54
|
+
# A passive one-zero filter dissipates high frequencies on every round trip.
|
|
55
|
+
damping = 0.3 - brightness * 0.25
|
|
56
|
+
damping_response = (1 - damping) + damping * np.exp(-1j * omega)
|
|
57
|
+
damping_phase = -float(np.angle(damping_response))
|
|
58
|
+
delay = max(1, math.floor((2 * math.pi - damping_phase) / omega))
|
|
59
|
+
residual = max(0.0, 2 * math.pi - damping_phase - delay * omega)
|
|
60
|
+
# Solve the linear interpolator's phase at the fundamental, rather than
|
|
61
|
+
# assuming its group delay equals its interpolation weight at every pitch.
|
|
62
|
+
fractional = math.sin(residual) / (math.sin(omega - residual) + math.sin(residual))
|
|
63
|
+
fractional = min(1.0, max(0.0, fractional))
|
|
64
|
+
interpolation_response = (1 - fractional) + fractional * np.exp(-1j * omega)
|
|
65
|
+
target_loss = math.exp(-math.log(1000) / (frequency * decay))
|
|
66
|
+
filter_loss = abs(damping_response * interpolation_response)
|
|
67
|
+
loop_gain = min(0.9999, target_loss / max(filter_loss, 1e-12))
|
|
68
|
+
coefficients = loop_gain * np.array(
|
|
69
|
+
[
|
|
70
|
+
(1 - damping) * (1 - fractional),
|
|
71
|
+
damping * (1 - fractional) + (1 - damping) * fractional,
|
|
72
|
+
damping * fractional,
|
|
73
|
+
]
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
rng = np.random.Generator(np.random.PCG64(seed))
|
|
77
|
+
excitation = rng.uniform(-1, 1, delay)
|
|
78
|
+
excitation -= np.roll(excitation, max(1, round(position * delay)))
|
|
79
|
+
# Shape a code-generated pluck, emphasizing low string modes while retaining
|
|
80
|
+
# the noise burst's transient. This spectrum is generated afresh per note.
|
|
81
|
+
harmonics = np.arange(delay // 2 + 1)
|
|
82
|
+
spectrum = np.fft.rfft(excitation)
|
|
83
|
+
spectrum *= 1 / (1 + (harmonics / (5 + brightness * 24)) ** 2)
|
|
84
|
+
excitation = np.fft.irfft(spectrum, n=delay)
|
|
85
|
+
location = np.arange(delay) / delay
|
|
86
|
+
displacement = np.where(
|
|
87
|
+
location <= position, location / position, (1 - location) / (1 - position)
|
|
88
|
+
)
|
|
89
|
+
excitation = 0.65 * excitation + 0.55 * (displacement - np.mean(displacement))
|
|
90
|
+
excitation += 0.3 * np.sin(2 * np.pi * location)
|
|
91
|
+
excitation -= np.mean(excitation)
|
|
92
|
+
peak = float(np.max(np.abs(excitation)))
|
|
93
|
+
if peak:
|
|
94
|
+
excitation *= 0.8 / peak
|
|
95
|
+
|
|
96
|
+
padding = delay + 2
|
|
97
|
+
history = np.zeros(frames + padding)
|
|
98
|
+
initial = min(delay, frames)
|
|
99
|
+
history[padding : padding + initial] = excitation[:initial]
|
|
100
|
+
# A block never exceeds the shortest delay, so every feedback read refers
|
|
101
|
+
# to an already-computed block. No Python loop per individual audio sample.
|
|
102
|
+
for start in range(padding, len(history), delay):
|
|
103
|
+
stop = min(len(history), start + delay)
|
|
104
|
+
history[start:stop] += (
|
|
105
|
+
coefficients[0] * history[start - delay : stop - delay]
|
|
106
|
+
+ coefficients[1] * history[start - delay - 1 : stop - delay - 1]
|
|
107
|
+
+ coefficients[2] * history[start - delay - 2 : stop - delay - 2]
|
|
108
|
+
)
|
|
109
|
+
signal = history[padding:]
|
|
110
|
+
# Body modes are driven by the vibrating string, not an independent knock.
|
|
111
|
+
t = np.arange(frames) / rate
|
|
112
|
+
signal += (
|
|
113
|
+
0.009 * brightness * colored_noise(frames, rate, seed + 1, 1600, 8000) * np.exp(-t / 0.006)
|
|
114
|
+
)
|
|
115
|
+
return resonant_body(
|
|
116
|
+
signal, rate, ((110, 0.19, 0.45), (205, 0.12, 0.35), (430, 0.075, 0.2)), wet=0.22
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _modes(
|
|
121
|
+
instrument: str,
|
|
122
|
+
frequency: float,
|
|
123
|
+
frames: int,
|
|
124
|
+
rate: int,
|
|
125
|
+
brightness: float,
|
|
126
|
+
decay: float,
|
|
127
|
+
) -> NDArray[np.float64]:
|
|
128
|
+
t = np.arange(frames, dtype=np.float64) / rate
|
|
129
|
+
profile = _MODAL_PROFILES[instrument]
|
|
130
|
+
signal = np.zeros(frames)
|
|
131
|
+
total_weight = 0.0
|
|
132
|
+
for index, (ratio, amplitude, lifetime) in enumerate(
|
|
133
|
+
zip(profile.ratios, profile.amplitudes, profile.lifetimes, strict=True)
|
|
134
|
+
):
|
|
135
|
+
amplitude *= 1 if index == 0 else 0.15 + brightness * 1.7
|
|
136
|
+
total_weight += amplitude
|
|
137
|
+
if frequency * ratio >= rate * 0.49:
|
|
138
|
+
continue
|
|
139
|
+
if instrument == "marimba":
|
|
140
|
+
contact = min(0.65 / frequency, 0.0035 * (1.2 - 0.8 * brightness))
|
|
141
|
+
response = struck_mode(
|
|
142
|
+
t, frequency * ratio, math.log(1000) / (decay * lifetime), contact
|
|
143
|
+
)
|
|
144
|
+
else:
|
|
145
|
+
envelope = np.exp(-math.log(1000) * t / (decay * lifetime))
|
|
146
|
+
partial = np.sin(2 * np.pi * frequency * ratio * t)
|
|
147
|
+
split = frequency * ratio * (1 + 0.0006 * (index + 1))
|
|
148
|
+
partial = 0.8 * partial + 0.2 * np.sin(2 * np.pi * split * t) * nyquist_gain(
|
|
149
|
+
split, rate
|
|
150
|
+
)
|
|
151
|
+
response = partial * envelope * (1 - np.exp(-t / 0.001))
|
|
152
|
+
signal += amplitude * response * nyquist_gain(frequency * ratio, rate)
|
|
153
|
+
return signal * (0.9 / total_weight) if total_weight else signal
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def synthesize(
|
|
157
|
+
instrument: str,
|
|
158
|
+
frequency: float,
|
|
159
|
+
frames: int,
|
|
160
|
+
rate: int,
|
|
161
|
+
seed: int,
|
|
162
|
+
velocity: float,
|
|
163
|
+
tone: Tone | None,
|
|
164
|
+
) -> NDArray[np.float64]:
|
|
165
|
+
instrument_info = require_instrument(instrument)
|
|
166
|
+
if instrument not in {"guitar", "marimba", "bell"}:
|
|
167
|
+
raise ValueError(f"{instrument!r} does not use the original physical/modal engine")
|
|
168
|
+
settings = tone or Tone()
|
|
169
|
+
if frequency >= rate / 2:
|
|
170
|
+
return np.zeros(frames)
|
|
171
|
+
brightness = min(1.0, settings.brightness * 0.75 + velocity * 0.25)
|
|
172
|
+
decay = (
|
|
173
|
+
settings.decay_seconds
|
|
174
|
+
if settings.decay_seconds is not None
|
|
175
|
+
else instrument_info.default_decay_seconds
|
|
176
|
+
)
|
|
177
|
+
if decay is None:
|
|
178
|
+
raise ValueError(f"{instrument!r} has no default decay configured")
|
|
179
|
+
if instrument_info.engine == "string":
|
|
180
|
+
return _string(
|
|
181
|
+
frequency,
|
|
182
|
+
frames,
|
|
183
|
+
rate,
|
|
184
|
+
seed,
|
|
185
|
+
brightness,
|
|
186
|
+
decay,
|
|
187
|
+
settings.pluck_position if settings.pluck_position is not None else 0.22,
|
|
188
|
+
)
|
|
189
|
+
return _modes(instrument, frequency, frames, rate, brightness, decay)
|
audio_as_code/py.typed
ADDED
|
File without changes
|
audio_as_code/render.py
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
"""Deterministic, in-process synthesis. No audio device, API key, or DAW needed."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import math
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from importlib.metadata import version
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
|
|
14
|
+
from ._audio import Audio, _metrics, _write_wav, analyze_wav
|
|
15
|
+
from ._export_rules import MAX_RENDER_SECONDS
|
|
16
|
+
from ._export_rules import note_frame as _note_frame
|
|
17
|
+
from ._paths import check_paths
|
|
18
|
+
from ._voices import _voice
|
|
19
|
+
from .automation import automation_values
|
|
20
|
+
from .effects import apply_effects
|
|
21
|
+
from .model import Song, Track, midi_pitch
|
|
22
|
+
|
|
23
|
+
PEAK_CEILING = 0.95
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True)
|
|
27
|
+
class RenderResult:
|
|
28
|
+
audio: Audio
|
|
29
|
+
report: dict
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _note_seed(song_seed: int, track_name: str, note_index: int) -> int:
|
|
33
|
+
"""Track-local seeds keep noise unchanged when an unrelated track is added."""
|
|
34
|
+
data = f"{song_seed}:{track_name}:{note_index}".encode()
|
|
35
|
+
return int.from_bytes(hashlib.sha256(data).digest()[:8], "little")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _track_audio(song: Song, track: Track, frames: int) -> Audio:
|
|
39
|
+
if (
|
|
40
|
+
song.tempo_map
|
|
41
|
+
or song.automation
|
|
42
|
+
or track.automation
|
|
43
|
+
or any(effect.mix for effect in track.effects)
|
|
44
|
+
or track.release_seconds
|
|
45
|
+
or track.pedal
|
|
46
|
+
or any(note.release_seconds for note in track.notes)
|
|
47
|
+
):
|
|
48
|
+
return _expressive_track_audio(song, track, frames)
|
|
49
|
+
result = np.zeros((frames, 2), dtype=np.float32)
|
|
50
|
+
angle = (track.pan + 1) * np.pi / 4
|
|
51
|
+
left, right = math.cos(angle), math.sin(angle)
|
|
52
|
+
for index, note in enumerate(track.notes):
|
|
53
|
+
start = _note_frame(song, note.start)
|
|
54
|
+
end = min(frames, _note_frame(song, note.start + note.duration))
|
|
55
|
+
if end <= start or track.gain == 0:
|
|
56
|
+
continue
|
|
57
|
+
seed = _note_seed(song.seed, track.name, index)
|
|
58
|
+
voice = _voice(
|
|
59
|
+
track.instrument,
|
|
60
|
+
midi_pitch(note.pitch),
|
|
61
|
+
end - start,
|
|
62
|
+
song.sample_rate,
|
|
63
|
+
seed,
|
|
64
|
+
note.velocity,
|
|
65
|
+
track.tone,
|
|
66
|
+
)
|
|
67
|
+
voice *= note.velocity * track.gain * song.master_gain
|
|
68
|
+
result[start:end, 0] += voice * left
|
|
69
|
+
result[start:end, 1] += voice * right
|
|
70
|
+
return result
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _expressive_track_audio(song: Song, track: Track, frames: int) -> Audio:
|
|
74
|
+
mono = np.zeros(frames, dtype=np.float32)
|
|
75
|
+
for index, note in enumerate(track.notes):
|
|
76
|
+
start = _note_frame(song, note.start)
|
|
77
|
+
held_end = _note_frame(song, song.note_gate_end(track, note))
|
|
78
|
+
release = song.note_release_seconds(track, note)
|
|
79
|
+
end = min(frames, held_end + round(release * song.sample_rate))
|
|
80
|
+
if held_end <= start:
|
|
81
|
+
continue
|
|
82
|
+
seed = _note_seed(song.seed, track.name, index)
|
|
83
|
+
voice = _voice(
|
|
84
|
+
track.instrument,
|
|
85
|
+
midi_pitch(note.pitch),
|
|
86
|
+
end - start,
|
|
87
|
+
song.sample_rate,
|
|
88
|
+
seed,
|
|
89
|
+
note.velocity,
|
|
90
|
+
track.tone,
|
|
91
|
+
held_frames=held_end - start if end > held_end else None,
|
|
92
|
+
)
|
|
93
|
+
mono[start:end] += voice * note.velocity
|
|
94
|
+
result = np.zeros((frames, 2), dtype=np.float32)
|
|
95
|
+
lanes = {lane.parameter: lane for lane in track.automation}
|
|
96
|
+
for start in range(0, frames, 65536):
|
|
97
|
+
stop = min(frames, start + 65536)
|
|
98
|
+
gain = (
|
|
99
|
+
automation_values(song, lanes["gain"], start, stop) if "gain" in lanes else track.gain
|
|
100
|
+
)
|
|
101
|
+
pan = automation_values(song, lanes["pan"], start, stop) if "pan" in lanes else track.pan
|
|
102
|
+
angle = (pan + 1) * np.pi / 4
|
|
103
|
+
result[start:stop, 0] = mono[start:stop] * gain * np.cos(angle)
|
|
104
|
+
result[start:stop, 1] = mono[start:stop] * gain * np.sin(angle)
|
|
105
|
+
del mono
|
|
106
|
+
result = apply_effects(result, track.effects, song.sample_rate)
|
|
107
|
+
for start in range(0, frames, 65536):
|
|
108
|
+
stop = min(frames, start + 65536)
|
|
109
|
+
if song.automation:
|
|
110
|
+
gain = automation_values(song, song.automation[0], start, stop)
|
|
111
|
+
result[start:stop] *= gain[:, None]
|
|
112
|
+
else:
|
|
113
|
+
result[start:stop] *= song.master_gain
|
|
114
|
+
return result
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _canonical_score(song: Song) -> dict:
|
|
118
|
+
"""Omit additive defaults from hashing to preserve existing score identities."""
|
|
119
|
+
data = song.model_dump(mode="json")
|
|
120
|
+
for key in ("tempo_map", "automation", "effects"):
|
|
121
|
+
if not data[key]:
|
|
122
|
+
del data[key]
|
|
123
|
+
for track in data["tracks"]:
|
|
124
|
+
for key in ("automation", "effects", "release_seconds", "pedal"):
|
|
125
|
+
if not track[key]:
|
|
126
|
+
del track[key]
|
|
127
|
+
for note in track["notes"]:
|
|
128
|
+
if note["release_seconds"] is None:
|
|
129
|
+
del note["release_seconds"]
|
|
130
|
+
return data
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def render_audio(song: Song, *, normalize: bool = True) -> RenderResult:
|
|
134
|
+
"""Render stereo float32 audio and measurements. Limit: five minutes per call.
|
|
135
|
+
|
|
136
|
+
Normalization only attenuates to a 0.95 peak ceiling; it never boosts a quiet mix.
|
|
137
|
+
Floating-point output remains unclipped when normalization is disabled.
|
|
138
|
+
"""
|
|
139
|
+
# Revalidate even if a caller used Pydantic's unchecked construction/copy escape hatches.
|
|
140
|
+
song = Song.model_validate(song.model_dump())
|
|
141
|
+
if song.render_seconds > MAX_RENDER_SECONDS:
|
|
142
|
+
raise ValueError(f"offline renders are limited to {MAX_RENDER_SECONDS} seconds")
|
|
143
|
+
frames = round(song.render_seconds * song.sample_rate)
|
|
144
|
+
if frames < 1:
|
|
145
|
+
raise ValueError("song is shorter than one audio sample")
|
|
146
|
+
mix = np.zeros((frames, 2), dtype=np.float32)
|
|
147
|
+
for track in song.tracks:
|
|
148
|
+
mix += _track_audio(song, track, frames)
|
|
149
|
+
mix = apply_effects(mix, song.effects, song.sample_rate)
|
|
150
|
+
if not np.all(np.isfinite(mix)):
|
|
151
|
+
raise ValueError("render produced non-finite samples; lower score gains or effect levels")
|
|
152
|
+
before = _metrics(mix, song.sample_rate)
|
|
153
|
+
attenuation = min(1.0, PEAK_CEILING / before["peak"]) if normalize and before["peak"] else 1.0
|
|
154
|
+
mix *= attenuation
|
|
155
|
+
warnings = []
|
|
156
|
+
if before["silent"]:
|
|
157
|
+
warnings.append("The render is silent; check notes, pitch range, and track/master gains.")
|
|
158
|
+
if attenuation < 1:
|
|
159
|
+
warnings.append(
|
|
160
|
+
"The mix exceeded the peak ceiling and was attenuated; consider lowering gains."
|
|
161
|
+
)
|
|
162
|
+
if not normalize and before["clipped_samples"]:
|
|
163
|
+
warnings.append("The mix exceeds full scale; WAV export will hard-clip these samples.")
|
|
164
|
+
if any(
|
|
165
|
+
_note_frame(song, song.note_gate_end(track, note)) - _note_frame(song, note.start) < 3
|
|
166
|
+
for track in song.tracks
|
|
167
|
+
for note in track.notes
|
|
168
|
+
):
|
|
169
|
+
warnings.append("Some notes are too short for the audio sample grid and may be silent.")
|
|
170
|
+
canonical = json.dumps(_canonical_score(song), sort_keys=True, separators=(",", ":"))
|
|
171
|
+
report = {
|
|
172
|
+
"engine_version": version("audio-as-code"),
|
|
173
|
+
"numpy_version": np.__version__,
|
|
174
|
+
"score_sha256": hashlib.sha256(canonical.encode()).hexdigest(),
|
|
175
|
+
"title": song.title,
|
|
176
|
+
"tracks": len(song.tracks),
|
|
177
|
+
"notes": sum(len(track.notes) for track in song.tracks),
|
|
178
|
+
"seed": song.seed,
|
|
179
|
+
"normalize": normalize,
|
|
180
|
+
"score_duration_seconds": song.seconds,
|
|
181
|
+
"tail_seconds": max(0.0, song.render_seconds - song.seconds),
|
|
182
|
+
"effects": {
|
|
183
|
+
"master": [effect.model_dump() for effect in song.effects],
|
|
184
|
+
"tracks": {
|
|
185
|
+
track.name: [effect.model_dump() for effect in track.effects]
|
|
186
|
+
for track in song.tracks
|
|
187
|
+
if track.effects
|
|
188
|
+
},
|
|
189
|
+
},
|
|
190
|
+
"gain_applied": attenuation,
|
|
191
|
+
"before_gain": before,
|
|
192
|
+
"audio": _metrics(mix, song.sample_rate),
|
|
193
|
+
"warnings": warnings,
|
|
194
|
+
}
|
|
195
|
+
return RenderResult(mix, report)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def render(
|
|
199
|
+
song: Song, path: str | Path, *, normalize: bool = True, stems_dir: str | Path | None = None
|
|
200
|
+
) -> dict:
|
|
201
|
+
"""Write a 16-bit stereo WAV. Optional stems share the mix's attenuation."""
|
|
202
|
+
song = Song.model_validate(song.model_dump())
|
|
203
|
+
destination = Path(path)
|
|
204
|
+
stem_paths = []
|
|
205
|
+
if stems_dir is not None:
|
|
206
|
+
# Use numeric filenames; track names never become filesystem paths.
|
|
207
|
+
stem_paths = [Path(stems_dir) / f"{index + 1:02d}.wav" for index in range(len(song.tracks))]
|
|
208
|
+
check_paths(
|
|
209
|
+
[destination, *stem_paths],
|
|
210
|
+
[stems_dir] if stems_dir is not None else [],
|
|
211
|
+
conflict_message="mix and stem outputs must not have the same path or file",
|
|
212
|
+
)
|
|
213
|
+
result = render_audio(song, normalize=normalize)
|
|
214
|
+
_write_wav(destination, result.audio, song.sample_rate)
|
|
215
|
+
report = {**result.report, "output": str(destination), "stems": []}
|
|
216
|
+
report["wav"] = analyze_wav(destination)
|
|
217
|
+
if stem_paths and song.effects:
|
|
218
|
+
report["warnings"].append(
|
|
219
|
+
"Stems include track effects and master gain automation, but omit master effects; "
|
|
220
|
+
"their sum will differ from the processed mix."
|
|
221
|
+
)
|
|
222
|
+
for track, stem_path in zip(song.tracks, stem_paths, strict=False):
|
|
223
|
+
stem = _track_audio(song, track, len(result.audio))
|
|
224
|
+
stem *= result.report["gain_applied"]
|
|
225
|
+
_write_wav(stem_path, stem, song.sample_rate)
|
|
226
|
+
metrics = _metrics(stem, song.sample_rate)
|
|
227
|
+
report["stems"].append({"track": track.name, "path": str(stem_path), "audio": metrics})
|
|
228
|
+
if metrics["clipped_samples"]:
|
|
229
|
+
report["warnings"].append(f"Stem {track.name!r} exceeds full scale and was clipped.")
|
|
230
|
+
return report
|