audiocompose 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- audiocompose/__init__.py +89 -0
- audiocompose/__main__.py +3 -0
- audiocompose/_version.py +24 -0
- audiocompose/alignment.py +94 -0
- audiocompose/analysis.py +179 -0
- audiocompose/cli.py +293 -0
- audiocompose/composer.py +170 -0
- audiocompose/diagnostics.py +25 -0
- audiocompose/errors.py +14 -0
- audiocompose/job.py +399 -0
- audiocompose/loudness.py +224 -0
- audiocompose/model.py +179 -0
- audiocompose/operations.py +185 -0
- audiocompose/py.typed +0 -0
- audiocompose/resampling.py +23 -0
- audiocompose/sources.py +68 -0
- audiocompose/timeline.py +19 -0
- audiocompose/wav.py +131 -0
- audiocompose-0.1.0.dist-info/METADATA +342 -0
- audiocompose-0.1.0.dist-info/RECORD +24 -0
- audiocompose-0.1.0.dist-info/WHEEL +5 -0
- audiocompose-0.1.0.dist-info/entry_points.txt +2 -0
- audiocompose-0.1.0.dist-info/licenses/LICENSE +201 -0
- audiocompose-0.1.0.dist-info/top_level.txt +1 -0
audiocompose/__init__.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
from ._version import __version__
|
|
2
|
+
from .alignment import AudioAnchor, AudioSpan, ComposedMarker, ComposedSpan, Marker
|
|
3
|
+
from .analysis import (
|
|
4
|
+
AcousticGap,
|
|
5
|
+
ActivityConfig,
|
|
6
|
+
ActivityRegion,
|
|
7
|
+
ActivityReport,
|
|
8
|
+
analyze_activity,
|
|
9
|
+
measure_gap_near,
|
|
10
|
+
)
|
|
11
|
+
from .composer import Composer
|
|
12
|
+
from .diagnostics import CompositionDiagnostic, DiagnosticSeverity
|
|
13
|
+
from .errors import AudioComposeError, AudioValidationError, CompositionError
|
|
14
|
+
from .job import load_job, save_job, validate_job
|
|
15
|
+
from .loudness import (
|
|
16
|
+
LoudnessMetrics,
|
|
17
|
+
LoudnessPolicy,
|
|
18
|
+
LoudnessResult,
|
|
19
|
+
apply_complete_output_loudness,
|
|
20
|
+
)
|
|
21
|
+
from .model import AudioClip, AudioJob, ComposedItem, CompositionResult, OutputPolicy, Silence
|
|
22
|
+
from .operations import (
|
|
23
|
+
AudioOperation,
|
|
24
|
+
FadeIn,
|
|
25
|
+
FadeOut,
|
|
26
|
+
Gain,
|
|
27
|
+
PitchShift,
|
|
28
|
+
Tempo,
|
|
29
|
+
apply_operation,
|
|
30
|
+
operation_from_dict,
|
|
31
|
+
)
|
|
32
|
+
from .resampling import resample_audio
|
|
33
|
+
from .sources import AudioBufferSource, AudioFileSource, AudioSource
|
|
34
|
+
from .timeline import samples_for_duration, silence
|
|
35
|
+
from .wav import prepare_output, read_wav, sha256_file, wav_info, write_intermediate_wav, write_wav
|
|
36
|
+
|
|
37
|
+
__all__ = [
|
|
38
|
+
"ActivityConfig",
|
|
39
|
+
"ActivityRegion",
|
|
40
|
+
"ActivityReport",
|
|
41
|
+
"AcousticGap",
|
|
42
|
+
"analyze_activity",
|
|
43
|
+
"measure_gap_near",
|
|
44
|
+
"__version__",
|
|
45
|
+
"AudioAnchor",
|
|
46
|
+
"AudioBufferSource",
|
|
47
|
+
"AudioClip",
|
|
48
|
+
"AudioComposeError",
|
|
49
|
+
"AudioFileSource",
|
|
50
|
+
"AudioJob",
|
|
51
|
+
"AudioOperation",
|
|
52
|
+
"AudioSource",
|
|
53
|
+
"AudioSpan",
|
|
54
|
+
"CompositionDiagnostic",
|
|
55
|
+
"CompositionError",
|
|
56
|
+
"CompositionResult",
|
|
57
|
+
"ComposedItem",
|
|
58
|
+
"ComposedMarker",
|
|
59
|
+
"ComposedSpan",
|
|
60
|
+
"Composer",
|
|
61
|
+
"DiagnosticSeverity",
|
|
62
|
+
"FadeIn",
|
|
63
|
+
"FadeOut",
|
|
64
|
+
"LoudnessMetrics",
|
|
65
|
+
"Gain",
|
|
66
|
+
"LoudnessPolicy",
|
|
67
|
+
"LoudnessResult",
|
|
68
|
+
"Marker",
|
|
69
|
+
"OutputPolicy",
|
|
70
|
+
"PitchShift",
|
|
71
|
+
"Silence",
|
|
72
|
+
"Tempo",
|
|
73
|
+
"apply_complete_output_loudness",
|
|
74
|
+
"apply_operation",
|
|
75
|
+
"load_job",
|
|
76
|
+
"operation_from_dict",
|
|
77
|
+
"prepare_output",
|
|
78
|
+
"read_wav",
|
|
79
|
+
"resample_audio",
|
|
80
|
+
"samples_for_duration",
|
|
81
|
+
"save_job",
|
|
82
|
+
"sha256_file",
|
|
83
|
+
"silence",
|
|
84
|
+
"validate_job",
|
|
85
|
+
"wav_info",
|
|
86
|
+
"write_intermediate_wav",
|
|
87
|
+
"write_wav",
|
|
88
|
+
"AudioValidationError",
|
|
89
|
+
]
|
audiocompose/__main__.py
ADDED
audiocompose/_version.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# file generated by vcs-versioning
|
|
2
|
+
# don't change, don't track in version control
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"__version__",
|
|
7
|
+
"__version_tuple__",
|
|
8
|
+
"version",
|
|
9
|
+
"version_tuple",
|
|
10
|
+
"__commit_id__",
|
|
11
|
+
"commit_id",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
version: str
|
|
15
|
+
__version__: str
|
|
16
|
+
__version_tuple__: tuple[int | str, ...]
|
|
17
|
+
version_tuple: tuple[int | str, ...]
|
|
18
|
+
commit_id: str | None
|
|
19
|
+
__commit_id__: str | None
|
|
20
|
+
|
|
21
|
+
__version__ = version = '0.1.0'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 1, 0)
|
|
23
|
+
|
|
24
|
+
__commit_id__ = commit_id = 'g88a3e01d9'
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from .errors import AudioValidationError
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True, slots=True)
|
|
11
|
+
class AudioAnchor:
|
|
12
|
+
id: str
|
|
13
|
+
sample_offset: int
|
|
14
|
+
name: str | None = None
|
|
15
|
+
|
|
16
|
+
def __post_init__(self) -> None:
|
|
17
|
+
if not isinstance(self.id, str) or not self.id:
|
|
18
|
+
raise AudioValidationError("anchor id must not be empty")
|
|
19
|
+
if (
|
|
20
|
+
isinstance(self.sample_offset, bool)
|
|
21
|
+
or not isinstance(self.sample_offset, int)
|
|
22
|
+
or self.sample_offset < 0
|
|
23
|
+
):
|
|
24
|
+
raise AudioValidationError("sample_offset must be a non-negative integer")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True, slots=True)
|
|
28
|
+
class AudioSpan:
|
|
29
|
+
source_start: int
|
|
30
|
+
source_end: int
|
|
31
|
+
sample_start: int
|
|
32
|
+
sample_end: int
|
|
33
|
+
id: str | None = None
|
|
34
|
+
metadata: Mapping[str, Any] = field(default_factory=dict)
|
|
35
|
+
|
|
36
|
+
def __post_init__(self) -> None:
|
|
37
|
+
values = (
|
|
38
|
+
("source_start", self.source_start),
|
|
39
|
+
("source_end", self.source_end),
|
|
40
|
+
("sample_start", self.sample_start),
|
|
41
|
+
("sample_end", self.sample_end),
|
|
42
|
+
)
|
|
43
|
+
for name, value in values:
|
|
44
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
|
45
|
+
raise AudioValidationError(f"{name} must be a non-negative integer")
|
|
46
|
+
if self.source_end < self.source_start:
|
|
47
|
+
raise AudioValidationError("source_end must be >= source_start")
|
|
48
|
+
if self.sample_end < self.sample_start:
|
|
49
|
+
raise AudioValidationError("sample_end must be >= sample_start")
|
|
50
|
+
if self.id is not None and (not isinstance(self.id, str) or not self.id):
|
|
51
|
+
raise AudioValidationError("span id must be a non-empty string or None")
|
|
52
|
+
if not isinstance(self.metadata, Mapping):
|
|
53
|
+
raise AudioValidationError("span metadata must be an object")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True, slots=True)
|
|
57
|
+
class ComposedSpan:
|
|
58
|
+
item_id: str
|
|
59
|
+
source_start: int
|
|
60
|
+
source_end: int
|
|
61
|
+
sample_start: int
|
|
62
|
+
sample_end: int
|
|
63
|
+
id: str | None = None
|
|
64
|
+
metadata: Mapping[str, Any] = field(default_factory=dict)
|
|
65
|
+
|
|
66
|
+
def __post_init__(self) -> None:
|
|
67
|
+
values = (
|
|
68
|
+
("source_start", self.source_start),
|
|
69
|
+
("source_end", self.source_end),
|
|
70
|
+
("sample_start", self.sample_start),
|
|
71
|
+
("sample_end", self.sample_end),
|
|
72
|
+
)
|
|
73
|
+
for name, value in values:
|
|
74
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
|
75
|
+
raise AudioValidationError(f"{name} must be a non-negative integer")
|
|
76
|
+
if self.source_end < self.source_start:
|
|
77
|
+
raise AudioValidationError("source_end must be >= source_start")
|
|
78
|
+
if self.sample_end < self.sample_start:
|
|
79
|
+
raise AudioValidationError("sample_end must be >= sample_start")
|
|
80
|
+
if self.id is not None and (not isinstance(self.id, str) or not self.id):
|
|
81
|
+
raise AudioValidationError("span id must be a non-empty string or None")
|
|
82
|
+
if not isinstance(self.metadata, Mapping):
|
|
83
|
+
raise AudioValidationError("composed span metadata must be an object")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclass(frozen=True, slots=True)
|
|
87
|
+
class ComposedMarker:
|
|
88
|
+
id: str
|
|
89
|
+
sample_offset: int
|
|
90
|
+
name: str | None = None
|
|
91
|
+
item_id: str | None = None
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
Marker = AudioAnchor
|
audiocompose/analysis.py
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True, slots=True)
|
|
10
|
+
class ActivityConfig:
|
|
11
|
+
frame_ms: float = 20.0
|
|
12
|
+
hop_ms: float = 5.0
|
|
13
|
+
active_threshold_dbfs: float = -40.0
|
|
14
|
+
release_threshold_dbfs: float = -45.0
|
|
15
|
+
min_active_ms: float = 30.0
|
|
16
|
+
min_gap_ms: float = 60.0
|
|
17
|
+
|
|
18
|
+
def __post_init__(self) -> None:
|
|
19
|
+
if self.frame_ms <= 0 or self.hop_ms <= 0:
|
|
20
|
+
raise ValueError("frame_ms and hop_ms must be positive")
|
|
21
|
+
if self.release_threshold_dbfs > self.active_threshold_dbfs:
|
|
22
|
+
raise ValueError("release threshold must not exceed active threshold")
|
|
23
|
+
if self.min_active_ms < 0 or self.min_gap_ms < 0:
|
|
24
|
+
raise ValueError("minimum durations must be non-negative")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True, slots=True)
|
|
28
|
+
class ActivityRegion:
|
|
29
|
+
start_sample: int
|
|
30
|
+
end_sample: int
|
|
31
|
+
rms_dbfs: float
|
|
32
|
+
peak_dbfs: float
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def duration_samples(self) -> int:
|
|
36
|
+
return self.end_sample - self.start_sample
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True, slots=True)
|
|
40
|
+
class AcousticGap:
|
|
41
|
+
start_sample: int
|
|
42
|
+
end_sample: int
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def duration_samples(self) -> int:
|
|
46
|
+
return self.end_sample - self.start_sample
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(frozen=True, slots=True)
|
|
50
|
+
class ActivityReport:
|
|
51
|
+
sample_rate: int
|
|
52
|
+
frames: int
|
|
53
|
+
activity: tuple[ActivityRegion, ...]
|
|
54
|
+
gaps: tuple[AcousticGap, ...]
|
|
55
|
+
leading_gap: AcousticGap | None
|
|
56
|
+
trailing_gap: AcousticGap | None
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def duration_seconds(self) -> float:
|
|
60
|
+
return self.frames / self.sample_rate
|
|
61
|
+
|
|
62
|
+
def seconds(self, samples: int) -> float:
|
|
63
|
+
return samples / self.sample_rate
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _dbfs(values: np.ndarray) -> tuple[float, float]:
|
|
67
|
+
if values.size == 0:
|
|
68
|
+
return -math.inf, -math.inf
|
|
69
|
+
rms = float(np.sqrt(np.mean(values.astype(np.float64) ** 2)))
|
|
70
|
+
peak = float(np.max(np.abs(values)))
|
|
71
|
+
return (
|
|
72
|
+
20.0 * math.log10(rms) if rms > 0 else -math.inf,
|
|
73
|
+
20.0 * math.log10(peak) if peak > 0 else -math.inf,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _raw_regions(
|
|
78
|
+
values: np.ndarray,
|
|
79
|
+
sample_rate: int,
|
|
80
|
+
config: ActivityConfig,
|
|
81
|
+
) -> list[tuple[int, int]]:
|
|
82
|
+
frame_size = max(1, round(config.frame_ms * sample_rate / 1000.0))
|
|
83
|
+
hop_size = max(1, round(config.hop_ms * sample_rate / 1000.0))
|
|
84
|
+
active = False
|
|
85
|
+
regions: list[tuple[int, int]] = []
|
|
86
|
+
start = 0
|
|
87
|
+
for frame_start in range(0, len(values), hop_size):
|
|
88
|
+
frame = values[frame_start : frame_start + frame_size]
|
|
89
|
+
rms_dbfs, _ = _dbfs(frame)
|
|
90
|
+
if not active and rms_dbfs >= config.active_threshold_dbfs:
|
|
91
|
+
active = True
|
|
92
|
+
start = frame_start
|
|
93
|
+
elif active and rms_dbfs < config.release_threshold_dbfs:
|
|
94
|
+
end = min(len(values), frame_start + frame_size)
|
|
95
|
+
regions.append((start, end))
|
|
96
|
+
active = False
|
|
97
|
+
if active:
|
|
98
|
+
regions.append((start, len(values)))
|
|
99
|
+
minimum = round(config.min_active_ms * sample_rate / 1000.0)
|
|
100
|
+
return [region for region in regions if region[1] - region[0] >= minimum]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _merge_regions(regions: list[tuple[int, int]], minimum_gap: int) -> list[tuple[int, int]]:
|
|
104
|
+
merged: list[tuple[int, int]] = []
|
|
105
|
+
for start, end in regions:
|
|
106
|
+
if merged and start - merged[-1][1] < minimum_gap:
|
|
107
|
+
merged[-1] = (merged[-1][0], end)
|
|
108
|
+
else:
|
|
109
|
+
merged.append((start, end))
|
|
110
|
+
return merged
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def analyze_activity(
|
|
114
|
+
audio: np.ndarray,
|
|
115
|
+
sample_rate: int,
|
|
116
|
+
config: ActivityConfig | None = None,
|
|
117
|
+
) -> ActivityReport:
|
|
118
|
+
if isinstance(sample_rate, bool) or not isinstance(sample_rate, int) or sample_rate <= 0:
|
|
119
|
+
raise ValueError("sample_rate must be a positive integer")
|
|
120
|
+
if config is None:
|
|
121
|
+
config = ActivityConfig()
|
|
122
|
+
values = np.asarray(audio, dtype=np.float32)
|
|
123
|
+
if values.ndim != 1 or not np.all(np.isfinite(values)):
|
|
124
|
+
raise ValueError("audio must be a finite one-dimensional waveform")
|
|
125
|
+
raw = _raw_regions(values, sample_rate, config)
|
|
126
|
+
minimum_gap = round(config.min_gap_ms * sample_rate / 1000.0)
|
|
127
|
+
merged = _merge_regions(raw, minimum_gap)
|
|
128
|
+
activity = tuple(ActivityRegion(start, end, *_dbfs(values[start:end])) for start, end in merged)
|
|
129
|
+
gaps: list[AcousticGap] = []
|
|
130
|
+
if activity:
|
|
131
|
+
for left, right in zip(activity, activity[1:], strict=False):
|
|
132
|
+
gap = AcousticGap(left.end_sample, right.start_sample)
|
|
133
|
+
if gap.duration_samples >= minimum_gap:
|
|
134
|
+
gaps.append(gap)
|
|
135
|
+
leading = (
|
|
136
|
+
AcousticGap(0, activity[0].start_sample)
|
|
137
|
+
if activity[0].start_sample >= minimum_gap
|
|
138
|
+
else None
|
|
139
|
+
)
|
|
140
|
+
trailing = (
|
|
141
|
+
AcousticGap(activity[-1].end_sample, len(values))
|
|
142
|
+
if len(values) - activity[-1].end_sample >= minimum_gap
|
|
143
|
+
else None
|
|
144
|
+
)
|
|
145
|
+
else:
|
|
146
|
+
whole = AcousticGap(0, len(values))
|
|
147
|
+
leading = whole if whole.duration_samples >= minimum_gap else None
|
|
148
|
+
trailing = None
|
|
149
|
+
if activity:
|
|
150
|
+
gaps_with_edges = ([leading] if leading else []) + gaps + ([trailing] if trailing else [])
|
|
151
|
+
else:
|
|
152
|
+
gaps_with_edges = [leading] if leading else []
|
|
153
|
+
return ActivityReport(
|
|
154
|
+
sample_rate, len(values), activity, tuple(gaps_with_edges), leading, trailing
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def measure_gap_near(
|
|
159
|
+
report: ActivityReport,
|
|
160
|
+
sample_offset: int,
|
|
161
|
+
*,
|
|
162
|
+
search_window_ms: float = 500.0,
|
|
163
|
+
) -> AcousticGap | None:
|
|
164
|
+
window = round(search_window_ms * report.sample_rate / 1000.0)
|
|
165
|
+
candidates = [
|
|
166
|
+
gap
|
|
167
|
+
for gap in report.gaps
|
|
168
|
+
if gap.end_sample >= sample_offset - window and gap.start_sample <= sample_offset + window
|
|
169
|
+
]
|
|
170
|
+
if not candidates:
|
|
171
|
+
return None
|
|
172
|
+
return min(
|
|
173
|
+
candidates,
|
|
174
|
+
key=lambda gap: (
|
|
175
|
+
0
|
|
176
|
+
if gap.start_sample <= sample_offset <= gap.end_sample
|
|
177
|
+
else min(abs(sample_offset - gap.start_sample), abs(sample_offset - gap.end_sample))
|
|
178
|
+
),
|
|
179
|
+
)
|
audiocompose/cli.py
ADDED
|
@@ -0,0 +1,293 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import html
|
|
5
|
+
import json
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from ._version import __version__
|
|
11
|
+
from .analysis import ActivityConfig, ActivityReport, analyze_activity, measure_gap_near
|
|
12
|
+
from .composer import Composer
|
|
13
|
+
from .errors import AudioComposeError
|
|
14
|
+
from .model import AudioClip, AudioJob, Silence
|
|
15
|
+
from .wav import read_wav
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _add_json(parser: argparse.ArgumentParser) -> None:
|
|
19
|
+
parser.add_argument("--json", action="store_true", help="write machine-readable JSON")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _add_analysis_options(parser: argparse.ArgumentParser) -> None:
|
|
23
|
+
parser.add_argument("--frame-ms", type=float, default=20.0)
|
|
24
|
+
parser.add_argument("--hop-ms", type=float, default=5.0)
|
|
25
|
+
parser.add_argument("--active-threshold-dbfs", type=float, default=-40.0)
|
|
26
|
+
parser.add_argument("--release-threshold-dbfs", type=float, default=-45.0)
|
|
27
|
+
parser.add_argument("--min-active-ms", type=float, default=30.0)
|
|
28
|
+
parser.add_argument("--min-gap-ms", type=float, default=60.0)
|
|
29
|
+
parser.add_argument("--near-marker")
|
|
30
|
+
parser.add_argument("--window-ms", type=float, default=500.0)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _parser() -> argparse.ArgumentParser:
|
|
34
|
+
parser = argparse.ArgumentParser(
|
|
35
|
+
prog="audiocompose", description="Compose declarative AudioJob bundles"
|
|
36
|
+
)
|
|
37
|
+
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
38
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
39
|
+
|
|
40
|
+
validate = subparsers.add_parser("validate")
|
|
41
|
+
validate.add_argument("manifest")
|
|
42
|
+
_add_json(validate)
|
|
43
|
+
|
|
44
|
+
inspect = subparsers.add_parser("inspect")
|
|
45
|
+
inspect.add_argument("manifest")
|
|
46
|
+
_add_json(inspect)
|
|
47
|
+
|
|
48
|
+
compose = subparsers.add_parser("compose")
|
|
49
|
+
compose.add_argument("manifest")
|
|
50
|
+
compose.add_argument("output")
|
|
51
|
+
|
|
52
|
+
timeline = subparsers.add_parser("timeline")
|
|
53
|
+
timeline.add_argument("manifest")
|
|
54
|
+
_add_json(timeline)
|
|
55
|
+
|
|
56
|
+
analyze = subparsers.add_parser("analyze")
|
|
57
|
+
analyze.add_argument("input")
|
|
58
|
+
_add_analysis_options(analyze)
|
|
59
|
+
_add_json(analyze)
|
|
60
|
+
|
|
61
|
+
report = subparsers.add_parser("report")
|
|
62
|
+
report.add_argument("input")
|
|
63
|
+
report.add_argument("-o", "--output", required=True)
|
|
64
|
+
_add_analysis_options(report)
|
|
65
|
+
|
|
66
|
+
return parser
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _analysis_config(args: argparse.Namespace) -> ActivityConfig:
|
|
70
|
+
return ActivityConfig(
|
|
71
|
+
frame_ms=args.frame_ms,
|
|
72
|
+
hop_ms=args.hop_ms,
|
|
73
|
+
active_threshold_dbfs=args.active_threshold_dbfs,
|
|
74
|
+
release_threshold_dbfs=args.release_threshold_dbfs,
|
|
75
|
+
min_active_ms=args.min_active_ms,
|
|
76
|
+
min_gap_ms=args.min_gap_ms,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _report_dict(report: ActivityReport) -> dict[str, Any]:
|
|
81
|
+
def gap_dict(gap: Any) -> dict[str, Any]:
|
|
82
|
+
return {
|
|
83
|
+
"start_sample": gap.start_sample,
|
|
84
|
+
"end_sample": gap.end_sample,
|
|
85
|
+
"duration_samples": gap.duration_samples,
|
|
86
|
+
"duration_seconds": report.seconds(gap.duration_samples),
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
return {
|
|
90
|
+
"sample_rate": report.sample_rate,
|
|
91
|
+
"frames": report.frames,
|
|
92
|
+
"duration_seconds": report.duration_seconds,
|
|
93
|
+
"activity": [
|
|
94
|
+
{
|
|
95
|
+
"start_sample": region.start_sample,
|
|
96
|
+
"end_sample": region.end_sample,
|
|
97
|
+
"duration_samples": region.duration_samples,
|
|
98
|
+
"rms_dbfs": region.rms_dbfs,
|
|
99
|
+
"peak_dbfs": region.peak_dbfs,
|
|
100
|
+
}
|
|
101
|
+
for region in report.activity
|
|
102
|
+
],
|
|
103
|
+
"gaps": [gap_dict(gap) for gap in report.gaps],
|
|
104
|
+
"leading_gap": gap_dict(report.leading_gap) if report.leading_gap else None,
|
|
105
|
+
"trailing_gap": gap_dict(report.trailing_gap) if report.trailing_gap else None,
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _load_input(path: str) -> tuple[Any, int, AudioJob | None]:
|
|
110
|
+
if Path(path).suffix.lower() == ".wav":
|
|
111
|
+
audio, sample_rate = read_wav(path)
|
|
112
|
+
return audio, sample_rate, None
|
|
113
|
+
job = AudioJob.load(path)
|
|
114
|
+
result = Composer().compose(job)
|
|
115
|
+
return result.audio, result.sample_rate, job
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _inspect(job: AudioJob) -> dict[str, Any]:
|
|
119
|
+
clips = [item for item in job.items if isinstance(item, AudioClip)]
|
|
120
|
+
silence = [item for item in job.items if isinstance(item, Silence)]
|
|
121
|
+
loaded = [(item, *item.source.load()) for item in clips]
|
|
122
|
+
rates = sorted({rate for _, _, rate in loaded})
|
|
123
|
+
operations = sorted({operation.type for item in clips for operation in item.operations})
|
|
124
|
+
raw_duration = sum(len(audio) / rate for _, audio, rate in loaded) + sum(
|
|
125
|
+
item.seconds for item in silence
|
|
126
|
+
)
|
|
127
|
+
return {
|
|
128
|
+
"job_id": job.job_id,
|
|
129
|
+
"producer": dict(job.producer),
|
|
130
|
+
"clips": len(clips),
|
|
131
|
+
"silence": len(silence),
|
|
132
|
+
"source_sample_rates": rates,
|
|
133
|
+
"operations": operations,
|
|
134
|
+
"expected_output_rate": job.output.sample_rate,
|
|
135
|
+
"total_raw_duration": raw_duration,
|
|
136
|
+
"loudness": {
|
|
137
|
+
"target_lufs": job.output.loudness.target_lufs,
|
|
138
|
+
"true_peak_ceiling_dbtp": job.output.loudness.true_peak_ceiling_dbtp,
|
|
139
|
+
},
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _timeline(job: AudioJob) -> dict[str, Any]:
|
|
144
|
+
result = Composer().compose(job)
|
|
145
|
+
return {
|
|
146
|
+
"sample_rate": result.sample_rate,
|
|
147
|
+
"items": [
|
|
148
|
+
{
|
|
149
|
+
"id": item.item_id,
|
|
150
|
+
"kind": item.kind,
|
|
151
|
+
"start_sample": item.start_sample,
|
|
152
|
+
"end_sample": item.end_sample,
|
|
153
|
+
}
|
|
154
|
+
for item in result.items
|
|
155
|
+
],
|
|
156
|
+
"markers": [
|
|
157
|
+
{
|
|
158
|
+
"id": marker.id,
|
|
159
|
+
"item_id": marker.item_id,
|
|
160
|
+
"sample_offset": marker.sample_offset,
|
|
161
|
+
"name": marker.name,
|
|
162
|
+
}
|
|
163
|
+
for marker in result.markers
|
|
164
|
+
],
|
|
165
|
+
"spans": [
|
|
166
|
+
{
|
|
167
|
+
"item_id": span.item_id,
|
|
168
|
+
"source_start": span.source_start,
|
|
169
|
+
"source_end": span.source_end,
|
|
170
|
+
"sample_start": span.sample_start,
|
|
171
|
+
"sample_end": span.sample_end,
|
|
172
|
+
}
|
|
173
|
+
for span in result.spans
|
|
174
|
+
],
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _svg_report(audio: Any, report: ActivityReport, job: AudioJob | None) -> str:
|
|
179
|
+
width = 1000
|
|
180
|
+
height = 260
|
|
181
|
+
scale = width / max(1, len(audio))
|
|
182
|
+
lines: list[str] = [f'<svg viewBox="0 0 {width} {height}" role="img">']
|
|
183
|
+
lines.append('<rect width="100%" height="100%" fill="white"/>')
|
|
184
|
+
for gap in report.gaps:
|
|
185
|
+
x = gap.start_sample * scale
|
|
186
|
+
w = max(1.0, gap.duration_samples * scale)
|
|
187
|
+
lines.append(f'<rect x="{x:.2f}" y="20" width="{w:.2f}" height="180" fill="#fee2e2"/>')
|
|
188
|
+
if len(audio):
|
|
189
|
+
step = max(1, len(audio) // width)
|
|
190
|
+
points = []
|
|
191
|
+
for index in range(0, len(audio), step):
|
|
192
|
+
values = audio[index : index + step]
|
|
193
|
+
points.append(f"{index * scale:.2f},{130 - float(max(abs(values))) * 100:.2f}")
|
|
194
|
+
lines.append(
|
|
195
|
+
f'<polyline points="{" ".join(points)}" fill="none" stroke="#1d4ed8" stroke-width="1"/>'
|
|
196
|
+
)
|
|
197
|
+
for region in report.activity:
|
|
198
|
+
x = region.start_sample * scale
|
|
199
|
+
w = max(1.0, region.duration_samples * scale)
|
|
200
|
+
lines.append(f'<rect x="{x:.2f}" y="205" width="{w:.2f}" height="20" fill="#86efac"/>')
|
|
201
|
+
lines.append("</svg>")
|
|
202
|
+
payload = json.dumps(_report_dict(report), indent=2, sort_keys=True)
|
|
203
|
+
title = "AudioCompose activity report"
|
|
204
|
+
if job is not None:
|
|
205
|
+
title += f" ({job.job_id})"
|
|
206
|
+
return (
|
|
207
|
+
'<!doctype html><html><head><meta charset="utf-8"><title>'
|
|
208
|
+
+ html.escape(title)
|
|
209
|
+
+ "</title></head><body>"
|
|
210
|
+
+ f"<h1>{html.escape(title)}</h1>"
|
|
211
|
+
+ "<p>Green regions are detected audio activity. Red regions are acoustic gaps.</p>"
|
|
212
|
+
+ "<section>"
|
|
213
|
+
+ "".join(lines)
|
|
214
|
+
+ "</section><pre>"
|
|
215
|
+
+ html.escape(payload)
|
|
216
|
+
+ "</pre></body></html>\n"
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def main(argv: list[str] | None = None) -> int:
|
|
221
|
+
args = _parser().parse_args(argv)
|
|
222
|
+
try:
|
|
223
|
+
if args.command == "validate":
|
|
224
|
+
validated_job = AudioJob.load(args.manifest)
|
|
225
|
+
result = {
|
|
226
|
+
"valid": True,
|
|
227
|
+
"format": "audiojob",
|
|
228
|
+
"schema_version": 1,
|
|
229
|
+
"job_id": validated_job.job_id,
|
|
230
|
+
}
|
|
231
|
+
if args.json:
|
|
232
|
+
print(json.dumps(result, indent=2, sort_keys=True))
|
|
233
|
+
else:
|
|
234
|
+
print(f"valid AudioJob: {args.manifest}")
|
|
235
|
+
elif args.command == "inspect":
|
|
236
|
+
result = _inspect(AudioJob.load(args.manifest))
|
|
237
|
+
print(json.dumps(result, indent=2, sort_keys=True))
|
|
238
|
+
elif args.command == "compose":
|
|
239
|
+
Composer().to_wav(AudioJob.load(args.manifest), args.output)
|
|
240
|
+
print(f"wrote {args.output}")
|
|
241
|
+
elif args.command == "timeline":
|
|
242
|
+
result = _timeline(AudioJob.load(args.manifest))
|
|
243
|
+
print(json.dumps(result, indent=2, sort_keys=True))
|
|
244
|
+
elif args.command in {"analyze", "report"}:
|
|
245
|
+
audio, sample_rate, job = _load_input(args.input)
|
|
246
|
+
report = analyze_activity(audio, sample_rate, _analysis_config(args))
|
|
247
|
+
if args.command == "analyze":
|
|
248
|
+
result = _report_dict(report)
|
|
249
|
+
if args.near_marker:
|
|
250
|
+
if job is None:
|
|
251
|
+
raise AudioComposeError("--near-marker requires an AudioJob input")
|
|
252
|
+
timeline = Composer().compose(job)
|
|
253
|
+
marker = next(
|
|
254
|
+
(marker for marker in timeline.markers if marker.id == args.near_marker),
|
|
255
|
+
None,
|
|
256
|
+
)
|
|
257
|
+
if marker is None:
|
|
258
|
+
raise AudioComposeError(f"unknown marker: {args.near_marker}")
|
|
259
|
+
gap = measure_gap_near(
|
|
260
|
+
report, marker.sample_offset, search_window_ms=args.window_ms
|
|
261
|
+
)
|
|
262
|
+
result["near_marker"] = args.near_marker
|
|
263
|
+
result["near_gap"] = (
|
|
264
|
+
_report_dict(
|
|
265
|
+
ActivityReport(
|
|
266
|
+
report.sample_rate,
|
|
267
|
+
report.frames,
|
|
268
|
+
(),
|
|
269
|
+
(gap,) if gap else (),
|
|
270
|
+
gap,
|
|
271
|
+
gap,
|
|
272
|
+
)
|
|
273
|
+
)["leading_gap"]
|
|
274
|
+
if gap
|
|
275
|
+
else None
|
|
276
|
+
)
|
|
277
|
+
print(json.dumps(result, indent=2, sort_keys=True))
|
|
278
|
+
else:
|
|
279
|
+
Path(args.output).write_text(_svg_report(audio, report, job), encoding="utf-8")
|
|
280
|
+
print(f"wrote {args.output}")
|
|
281
|
+
return 0
|
|
282
|
+
except (AudioComposeError, ValueError, OSError) as exc:
|
|
283
|
+
if getattr(args, "json", False):
|
|
284
|
+
print(
|
|
285
|
+
json.dumps(
|
|
286
|
+
{"valid": False, "errors": [{"code": "validation_error", "message": str(exc)}]},
|
|
287
|
+
indent=2,
|
|
288
|
+
),
|
|
289
|
+
file=sys.stderr,
|
|
290
|
+
)
|
|
291
|
+
else:
|
|
292
|
+
print(f"audiocompose: {exc}", file=sys.stderr)
|
|
293
|
+
return 2
|