audio-as-code 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,189 @@
1
+ """Sample-free physical/modal approximations, implemented with NumPy only.
2
+
3
+ The string uses a passive Karplus-Strong feedback loop with phase-compensated
4
+ fractional delay. Bar and bell presets sum the impulse responses of damped modes.
5
+ All excitations are generated here; no recordings or measured impulse responses.
6
+ See docs/synthesis.md for equations, scope, and references.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import math
12
+ from dataclasses import dataclass
13
+ from types import MappingProxyType
14
+
15
+ import numpy as np
16
+ from numpy.typing import NDArray
17
+
18
+ from .acoustics import colored_noise, nyquist_gain, resonant_body, struck_mode
19
+ from .instruments import require_instrument
20
+ from .model import Tone
21
+
22
+
23
+ @dataclass(frozen=True)
24
+ class ModalProfile:
25
+ ratios: tuple[float, ...]
26
+ amplitudes: tuple[float, ...]
27
+ lifetimes: tuple[float, ...]
28
+
29
+
30
+ _MODAL_PROFILES = MappingProxyType(
31
+ {
32
+ # Approximate tuned wooden bar; higher modes die away much sooner.
33
+ "marimba": ModalProfile((1, 4, 10, 17), (1, 0.32, 0.12, 0.04), (1, 0.32, 0.14, 0.08)),
34
+ # Designed inharmonic chime, not a model fitted to a particular bell.
35
+ "bell": ModalProfile(
36
+ (1, 2, 2.756, 4.07, 5.404, 6.81),
37
+ (1, 0.38, 0.32, 0.22, 0.12, 0.08),
38
+ (1, 0.8, 0.65, 0.5, 0.4, 0.3),
39
+ ),
40
+ }
41
+ )
42
+
43
+
44
+ def _string(
45
+ frequency: float,
46
+ frames: int,
47
+ rate: int,
48
+ seed: int,
49
+ brightness: float,
50
+ decay: float,
51
+ position: float,
52
+ ) -> NDArray[np.float64]:
53
+ omega = 2 * math.pi * frequency / rate
54
+ # A passive one-zero filter dissipates high frequencies on every round trip.
55
+ damping = 0.3 - brightness * 0.25
56
+ damping_response = (1 - damping) + damping * np.exp(-1j * omega)
57
+ damping_phase = -float(np.angle(damping_response))
58
+ delay = max(1, math.floor((2 * math.pi - damping_phase) / omega))
59
+ residual = max(0.0, 2 * math.pi - damping_phase - delay * omega)
60
+ # Solve the linear interpolator's phase at the fundamental, rather than
61
+ # assuming its group delay equals its interpolation weight at every pitch.
62
+ fractional = math.sin(residual) / (math.sin(omega - residual) + math.sin(residual))
63
+ fractional = min(1.0, max(0.0, fractional))
64
+ interpolation_response = (1 - fractional) + fractional * np.exp(-1j * omega)
65
+ target_loss = math.exp(-math.log(1000) / (frequency * decay))
66
+ filter_loss = abs(damping_response * interpolation_response)
67
+ loop_gain = min(0.9999, target_loss / max(filter_loss, 1e-12))
68
+ coefficients = loop_gain * np.array(
69
+ [
70
+ (1 - damping) * (1 - fractional),
71
+ damping * (1 - fractional) + (1 - damping) * fractional,
72
+ damping * fractional,
73
+ ]
74
+ )
75
+
76
+ rng = np.random.Generator(np.random.PCG64(seed))
77
+ excitation = rng.uniform(-1, 1, delay)
78
+ excitation -= np.roll(excitation, max(1, round(position * delay)))
79
+ # Shape a code-generated pluck, emphasizing low string modes while retaining
80
+ # the noise burst's transient. This spectrum is generated afresh per note.
81
+ harmonics = np.arange(delay // 2 + 1)
82
+ spectrum = np.fft.rfft(excitation)
83
+ spectrum *= 1 / (1 + (harmonics / (5 + brightness * 24)) ** 2)
84
+ excitation = np.fft.irfft(spectrum, n=delay)
85
+ location = np.arange(delay) / delay
86
+ displacement = np.where(
87
+ location <= position, location / position, (1 - location) / (1 - position)
88
+ )
89
+ excitation = 0.65 * excitation + 0.55 * (displacement - np.mean(displacement))
90
+ excitation += 0.3 * np.sin(2 * np.pi * location)
91
+ excitation -= np.mean(excitation)
92
+ peak = float(np.max(np.abs(excitation)))
93
+ if peak:
94
+ excitation *= 0.8 / peak
95
+
96
+ padding = delay + 2
97
+ history = np.zeros(frames + padding)
98
+ initial = min(delay, frames)
99
+ history[padding : padding + initial] = excitation[:initial]
100
+ # A block never exceeds the shortest delay, so every feedback read refers
101
+ # to an already-computed block. No Python loop per individual audio sample.
102
+ for start in range(padding, len(history), delay):
103
+ stop = min(len(history), start + delay)
104
+ history[start:stop] += (
105
+ coefficients[0] * history[start - delay : stop - delay]
106
+ + coefficients[1] * history[start - delay - 1 : stop - delay - 1]
107
+ + coefficients[2] * history[start - delay - 2 : stop - delay - 2]
108
+ )
109
+ signal = history[padding:]
110
+ # Body modes are driven by the vibrating string, not an independent knock.
111
+ t = np.arange(frames) / rate
112
+ signal += (
113
+ 0.009 * brightness * colored_noise(frames, rate, seed + 1, 1600, 8000) * np.exp(-t / 0.006)
114
+ )
115
+ return resonant_body(
116
+ signal, rate, ((110, 0.19, 0.45), (205, 0.12, 0.35), (430, 0.075, 0.2)), wet=0.22
117
+ )
118
+
119
+
120
+ def _modes(
121
+ instrument: str,
122
+ frequency: float,
123
+ frames: int,
124
+ rate: int,
125
+ brightness: float,
126
+ decay: float,
127
+ ) -> NDArray[np.float64]:
128
+ t = np.arange(frames, dtype=np.float64) / rate
129
+ profile = _MODAL_PROFILES[instrument]
130
+ signal = np.zeros(frames)
131
+ total_weight = 0.0
132
+ for index, (ratio, amplitude, lifetime) in enumerate(
133
+ zip(profile.ratios, profile.amplitudes, profile.lifetimes, strict=True)
134
+ ):
135
+ amplitude *= 1 if index == 0 else 0.15 + brightness * 1.7
136
+ total_weight += amplitude
137
+ if frequency * ratio >= rate * 0.49:
138
+ continue
139
+ if instrument == "marimba":
140
+ contact = min(0.65 / frequency, 0.0035 * (1.2 - 0.8 * brightness))
141
+ response = struck_mode(
142
+ t, frequency * ratio, math.log(1000) / (decay * lifetime), contact
143
+ )
144
+ else:
145
+ envelope = np.exp(-math.log(1000) * t / (decay * lifetime))
146
+ partial = np.sin(2 * np.pi * frequency * ratio * t)
147
+ split = frequency * ratio * (1 + 0.0006 * (index + 1))
148
+ partial = 0.8 * partial + 0.2 * np.sin(2 * np.pi * split * t) * nyquist_gain(
149
+ split, rate
150
+ )
151
+ response = partial * envelope * (1 - np.exp(-t / 0.001))
152
+ signal += amplitude * response * nyquist_gain(frequency * ratio, rate)
153
+ return signal * (0.9 / total_weight) if total_weight else signal
154
+
155
+
156
+ def synthesize(
157
+ instrument: str,
158
+ frequency: float,
159
+ frames: int,
160
+ rate: int,
161
+ seed: int,
162
+ velocity: float,
163
+ tone: Tone | None,
164
+ ) -> NDArray[np.float64]:
165
+ instrument_info = require_instrument(instrument)
166
+ if instrument not in {"guitar", "marimba", "bell"}:
167
+ raise ValueError(f"{instrument!r} does not use the original physical/modal engine")
168
+ settings = tone or Tone()
169
+ if frequency >= rate / 2:
170
+ return np.zeros(frames)
171
+ brightness = min(1.0, settings.brightness * 0.75 + velocity * 0.25)
172
+ decay = (
173
+ settings.decay_seconds
174
+ if settings.decay_seconds is not None
175
+ else instrument_info.default_decay_seconds
176
+ )
177
+ if decay is None:
178
+ raise ValueError(f"{instrument!r} has no default decay configured")
179
+ if instrument_info.engine == "string":
180
+ return _string(
181
+ frequency,
182
+ frames,
183
+ rate,
184
+ seed,
185
+ brightness,
186
+ decay,
187
+ settings.pluck_position if settings.pluck_position is not None else 0.22,
188
+ )
189
+ return _modes(instrument, frequency, frames, rate, brightness, decay)
audio_as_code/py.typed ADDED
File without changes
@@ -0,0 +1,230 @@
1
+ """Deterministic, in-process synthesis. No audio device, API key, or DAW needed."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import math
8
+ from dataclasses import dataclass
9
+ from importlib.metadata import version
10
+ from pathlib import Path
11
+
12
+ import numpy as np
13
+
14
+ from ._audio import Audio, _metrics, _write_wav, analyze_wav
15
+ from ._export_rules import MAX_RENDER_SECONDS
16
+ from ._export_rules import note_frame as _note_frame
17
+ from ._paths import check_paths
18
+ from ._voices import _voice
19
+ from .automation import automation_values
20
+ from .effects import apply_effects
21
+ from .model import Song, Track, midi_pitch
22
+
23
+ PEAK_CEILING = 0.95
24
+
25
+
26
+ @dataclass(frozen=True)
27
+ class RenderResult:
28
+ audio: Audio
29
+ report: dict
30
+
31
+
32
+ def _note_seed(song_seed: int, track_name: str, note_index: int) -> int:
33
+ """Track-local seeds keep noise unchanged when an unrelated track is added."""
34
+ data = f"{song_seed}:{track_name}:{note_index}".encode()
35
+ return int.from_bytes(hashlib.sha256(data).digest()[:8], "little")
36
+
37
+
38
+ def _track_audio(song: Song, track: Track, frames: int) -> Audio:
39
+ if (
40
+ song.tempo_map
41
+ or song.automation
42
+ or track.automation
43
+ or any(effect.mix for effect in track.effects)
44
+ or track.release_seconds
45
+ or track.pedal
46
+ or any(note.release_seconds for note in track.notes)
47
+ ):
48
+ return _expressive_track_audio(song, track, frames)
49
+ result = np.zeros((frames, 2), dtype=np.float32)
50
+ angle = (track.pan + 1) * np.pi / 4
51
+ left, right = math.cos(angle), math.sin(angle)
52
+ for index, note in enumerate(track.notes):
53
+ start = _note_frame(song, note.start)
54
+ end = min(frames, _note_frame(song, note.start + note.duration))
55
+ if end <= start or track.gain == 0:
56
+ continue
57
+ seed = _note_seed(song.seed, track.name, index)
58
+ voice = _voice(
59
+ track.instrument,
60
+ midi_pitch(note.pitch),
61
+ end - start,
62
+ song.sample_rate,
63
+ seed,
64
+ note.velocity,
65
+ track.tone,
66
+ )
67
+ voice *= note.velocity * track.gain * song.master_gain
68
+ result[start:end, 0] += voice * left
69
+ result[start:end, 1] += voice * right
70
+ return result
71
+
72
+
73
+ def _expressive_track_audio(song: Song, track: Track, frames: int) -> Audio:
74
+ mono = np.zeros(frames, dtype=np.float32)
75
+ for index, note in enumerate(track.notes):
76
+ start = _note_frame(song, note.start)
77
+ held_end = _note_frame(song, song.note_gate_end(track, note))
78
+ release = song.note_release_seconds(track, note)
79
+ end = min(frames, held_end + round(release * song.sample_rate))
80
+ if held_end <= start:
81
+ continue
82
+ seed = _note_seed(song.seed, track.name, index)
83
+ voice = _voice(
84
+ track.instrument,
85
+ midi_pitch(note.pitch),
86
+ end - start,
87
+ song.sample_rate,
88
+ seed,
89
+ note.velocity,
90
+ track.tone,
91
+ held_frames=held_end - start if end > held_end else None,
92
+ )
93
+ mono[start:end] += voice * note.velocity
94
+ result = np.zeros((frames, 2), dtype=np.float32)
95
+ lanes = {lane.parameter: lane for lane in track.automation}
96
+ for start in range(0, frames, 65536):
97
+ stop = min(frames, start + 65536)
98
+ gain = (
99
+ automation_values(song, lanes["gain"], start, stop) if "gain" in lanes else track.gain
100
+ )
101
+ pan = automation_values(song, lanes["pan"], start, stop) if "pan" in lanes else track.pan
102
+ angle = (pan + 1) * np.pi / 4
103
+ result[start:stop, 0] = mono[start:stop] * gain * np.cos(angle)
104
+ result[start:stop, 1] = mono[start:stop] * gain * np.sin(angle)
105
+ del mono
106
+ result = apply_effects(result, track.effects, song.sample_rate)
107
+ for start in range(0, frames, 65536):
108
+ stop = min(frames, start + 65536)
109
+ if song.automation:
110
+ gain = automation_values(song, song.automation[0], start, stop)
111
+ result[start:stop] *= gain[:, None]
112
+ else:
113
+ result[start:stop] *= song.master_gain
114
+ return result
115
+
116
+
117
+ def _canonical_score(song: Song) -> dict:
118
+ """Omit additive defaults from hashing to preserve existing score identities."""
119
+ data = song.model_dump(mode="json")
120
+ for key in ("tempo_map", "automation", "effects"):
121
+ if not data[key]:
122
+ del data[key]
123
+ for track in data["tracks"]:
124
+ for key in ("automation", "effects", "release_seconds", "pedal"):
125
+ if not track[key]:
126
+ del track[key]
127
+ for note in track["notes"]:
128
+ if note["release_seconds"] is None:
129
+ del note["release_seconds"]
130
+ return data
131
+
132
+
133
+ def render_audio(song: Song, *, normalize: bool = True) -> RenderResult:
134
+ """Render stereo float32 audio and measurements. Limit: five minutes per call.
135
+
136
+ Normalization only attenuates to a 0.95 peak ceiling; it never boosts a quiet mix.
137
+ Floating-point output remains unclipped when normalization is disabled.
138
+ """
139
+ # Revalidate even if a caller used Pydantic's unchecked construction/copy escape hatches.
140
+ song = Song.model_validate(song.model_dump())
141
+ if song.render_seconds > MAX_RENDER_SECONDS:
142
+ raise ValueError(f"offline renders are limited to {MAX_RENDER_SECONDS} seconds")
143
+ frames = round(song.render_seconds * song.sample_rate)
144
+ if frames < 1:
145
+ raise ValueError("song is shorter than one audio sample")
146
+ mix = np.zeros((frames, 2), dtype=np.float32)
147
+ for track in song.tracks:
148
+ mix += _track_audio(song, track, frames)
149
+ mix = apply_effects(mix, song.effects, song.sample_rate)
150
+ if not np.all(np.isfinite(mix)):
151
+ raise ValueError("render produced non-finite samples; lower score gains or effect levels")
152
+ before = _metrics(mix, song.sample_rate)
153
+ attenuation = min(1.0, PEAK_CEILING / before["peak"]) if normalize and before["peak"] else 1.0
154
+ mix *= attenuation
155
+ warnings = []
156
+ if before["silent"]:
157
+ warnings.append("The render is silent; check notes, pitch range, and track/master gains.")
158
+ if attenuation < 1:
159
+ warnings.append(
160
+ "The mix exceeded the peak ceiling and was attenuated; consider lowering gains."
161
+ )
162
+ if not normalize and before["clipped_samples"]:
163
+ warnings.append("The mix exceeds full scale; WAV export will hard-clip these samples.")
164
+ if any(
165
+ _note_frame(song, song.note_gate_end(track, note)) - _note_frame(song, note.start) < 3
166
+ for track in song.tracks
167
+ for note in track.notes
168
+ ):
169
+ warnings.append("Some notes are too short for the audio sample grid and may be silent.")
170
+ canonical = json.dumps(_canonical_score(song), sort_keys=True, separators=(",", ":"))
171
+ report = {
172
+ "engine_version": version("audio-as-code"),
173
+ "numpy_version": np.__version__,
174
+ "score_sha256": hashlib.sha256(canonical.encode()).hexdigest(),
175
+ "title": song.title,
176
+ "tracks": len(song.tracks),
177
+ "notes": sum(len(track.notes) for track in song.tracks),
178
+ "seed": song.seed,
179
+ "normalize": normalize,
180
+ "score_duration_seconds": song.seconds,
181
+ "tail_seconds": max(0.0, song.render_seconds - song.seconds),
182
+ "effects": {
183
+ "master": [effect.model_dump() for effect in song.effects],
184
+ "tracks": {
185
+ track.name: [effect.model_dump() for effect in track.effects]
186
+ for track in song.tracks
187
+ if track.effects
188
+ },
189
+ },
190
+ "gain_applied": attenuation,
191
+ "before_gain": before,
192
+ "audio": _metrics(mix, song.sample_rate),
193
+ "warnings": warnings,
194
+ }
195
+ return RenderResult(mix, report)
196
+
197
+
198
+ def render(
199
+ song: Song, path: str | Path, *, normalize: bool = True, stems_dir: str | Path | None = None
200
+ ) -> dict:
201
+ """Write a 16-bit stereo WAV. Optional stems share the mix's attenuation."""
202
+ song = Song.model_validate(song.model_dump())
203
+ destination = Path(path)
204
+ stem_paths = []
205
+ if stems_dir is not None:
206
+ # Use numeric filenames; track names never become filesystem paths.
207
+ stem_paths = [Path(stems_dir) / f"{index + 1:02d}.wav" for index in range(len(song.tracks))]
208
+ check_paths(
209
+ [destination, *stem_paths],
210
+ [stems_dir] if stems_dir is not None else [],
211
+ conflict_message="mix and stem outputs must not have the same path or file",
212
+ )
213
+ result = render_audio(song, normalize=normalize)
214
+ _write_wav(destination, result.audio, song.sample_rate)
215
+ report = {**result.report, "output": str(destination), "stems": []}
216
+ report["wav"] = analyze_wav(destination)
217
+ if stem_paths and song.effects:
218
+ report["warnings"].append(
219
+ "Stems include track effects and master gain automation, but omit master effects; "
220
+ "their sum will differ from the processed mix."
221
+ )
222
+ for track, stem_path in zip(song.tracks, stem_paths, strict=False):
223
+ stem = _track_audio(song, track, len(result.audio))
224
+ stem *= result.report["gain_applied"]
225
+ _write_wav(stem_path, stem, song.sample_rate)
226
+ metrics = _metrics(stem, song.sample_rate)
227
+ report["stems"].append({"track": track.name, "path": str(stem_path), "audio": metrics})
228
+ if metrics["clipped_samples"]:
229
+ report["warnings"].append(f"Stem {track.name!r} exceeds full scale and was clipped.")
230
+ return report