diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/tasks/audio_utils.py
ADDED
|
@@ -0,0 +1,1862 @@
|
|
|
1
|
+
"""Waveform utilities for audio tasks and segment-chained video generation.
|
|
2
|
+
|
|
3
|
+
Waveforms are handled as (channels, samples) float32 numpy arrays throughout -
|
|
4
|
+
as_channels_samples normalizes the shapes pipelines and files actually produce
|
|
5
|
+
into that layout.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import io
|
|
9
|
+
import os
|
|
10
|
+
import logging
|
|
11
|
+
from fractions import Fraction
|
|
12
|
+
|
|
13
|
+
import numpy
|
|
14
|
+
import soundfile
|
|
15
|
+
import torch
|
|
16
|
+
|
|
17
|
+
from ..events import emit_log, emit_warning
|
|
18
|
+
from ..loudness import integrated_lufs
|
|
19
|
+
from ..task_domains import as_number, check_arguments
|
|
20
|
+
from ..security import (
|
|
21
|
+
validate_file_extension,
|
|
22
|
+
ALLOWED_AUDIO_EXTENSIONS,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
logger = logging.getLogger("dw")
|
|
26
|
+
|
|
27
|
+
# A few milliseconds of fade applied on each side of a butt-joined seam so the
|
|
28
|
+
# discontinuity does not click
|
|
29
|
+
DECLICK_MS = 3.0
|
|
30
|
+
|
|
31
|
+
# Padding shorter than this at the end of a slice is the rounding that
|
|
32
|
+
# frame-aligned slicing produces, not a slice that overran its source
|
|
33
|
+
SLICE_PAD_WARN_MS = 10.0
|
|
34
|
+
|
|
35
|
+
# A dropped tail is only the "almost reached the end" signature this warning
|
|
36
|
+
# exists for when it is both short next to the slice and short in absolute
|
|
37
|
+
# terms - a deliberate excerpt out of a long recording drops most of the
|
|
38
|
+
# source and should not warn
|
|
39
|
+
SLICE_TRIM_WARN_SECONDS = 10.0
|
|
40
|
+
SLICE_TRIM_WARN_FRACTION = 0.05
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def as_channels_samples(audio):
|
|
44
|
+
"""Normalize a waveform to a (channels, samples) float32 numpy array.
|
|
45
|
+
|
|
46
|
+
Accepts torch tensors or numpy arrays shaped (samples,), (channels, samples),
|
|
47
|
+
(samples, channels), or a one-item batch (1, channels, samples). Channel
|
|
48
|
+
position is decided the way normalize_audio in result.py decides it: there
|
|
49
|
+
are always more samples than channels.
|
|
50
|
+
"""
|
|
51
|
+
if torch.is_tensor(audio):
|
|
52
|
+
audio = audio.detach().cpu().float().numpy()
|
|
53
|
+
audio = numpy.asarray(audio, dtype=numpy.float32)
|
|
54
|
+
|
|
55
|
+
if audio.ndim == 1:
|
|
56
|
+
return audio[numpy.newaxis, :]
|
|
57
|
+
|
|
58
|
+
if audio.ndim == 3:
|
|
59
|
+
if audio.shape[0] != 1:
|
|
60
|
+
raise ValueError(f"Cannot normalize a waveform batch of {audio.shape[0]}")
|
|
61
|
+
audio = audio[0]
|
|
62
|
+
|
|
63
|
+
if audio.ndim != 2:
|
|
64
|
+
raise ValueError(f"A waveform must have 1-3 dimensions, got {audio.ndim}")
|
|
65
|
+
|
|
66
|
+
if audio.shape[0] > audio.shape[1]: # (samples, channels) -> transpose
|
|
67
|
+
audio = audio.T
|
|
68
|
+
|
|
69
|
+
return numpy.ascontiguousarray(audio)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def frames_to_samples(frames, fps, sample_rate):
|
|
73
|
+
"""The number of audio samples spanning a run of video frames."""
|
|
74
|
+
return int(round(frames / fps * sample_rate))
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def fit_audio_to_frames(audio, sample_rate, total_frames, fps, command):
|
|
78
|
+
"""Pad a joined track that falls short of its frame grid, and warn when
|
|
79
|
+
the gap is a frame or more.
|
|
80
|
+
|
|
81
|
+
concat_videos and dissolve_videos each build their joined track by
|
|
82
|
+
measuring and concatenating/crossfading the actual input waveforms, with
|
|
83
|
+
nothing reconciling a shortfall against total_frames - so an input that
|
|
84
|
+
is itself short of its own frame grid (#435 traced this to an
|
|
85
|
+
ltx2/keyframes clip short of its 121-frame bucket) propagates its
|
|
86
|
+
shortfall into the join, and the shortfall compounds across further
|
|
87
|
+
joins that each take the previous join's output as an input. The same
|
|
88
|
+
remedy #428 gave pair_audio's 'fit: video' for a shortfall, applied here
|
|
89
|
+
at the one place every join's audio passes through before its shot map
|
|
90
|
+
is measured.
|
|
91
|
+
|
|
92
|
+
A track *longer* than its frame grid is left alone: concat_videos has
|
|
93
|
+
measured such an overrun deliberately since #378 (its own shot keeps the
|
|
94
|
+
samples it actually took, not a count derived from frame/fps
|
|
95
|
+
arithmetic), and trimming it here would silently reverse that contract
|
|
96
|
+
for the whole joined track.
|
|
97
|
+
"""
|
|
98
|
+
if audio is None or not total_frames or not fps or not sample_rate:
|
|
99
|
+
return audio
|
|
100
|
+
|
|
101
|
+
wanted = frames_to_samples(total_frames, fps, sample_rate)
|
|
102
|
+
have = audio.shape[1]
|
|
103
|
+
if have >= wanted:
|
|
104
|
+
return audio
|
|
105
|
+
|
|
106
|
+
audio_seconds = have / float(sample_rate)
|
|
107
|
+
video_seconds = total_frames / float(fps)
|
|
108
|
+
pad_samples = wanted - have
|
|
109
|
+
audio = numpy.pad(audio, ((0, 0), (0, pad_samples)))
|
|
110
|
+
if pad_samples < sample_rate / fps:
|
|
111
|
+
# Less than one frame is rounding between the rate and the frame
|
|
112
|
+
# grid, the gap pair_audio's own unfitted check leaves unwarned
|
|
113
|
+
# (LENGTH_WARN_MS): padded, and logged, but not a warning on every
|
|
114
|
+
# stock join (#454)
|
|
115
|
+
emit_log(
|
|
116
|
+
f"{command}: padded the joined track with {pad_samples} sample"
|
|
117
|
+
f"{'s' if pad_samples != 1 else ''} of silence to the frame grid",
|
|
118
|
+
command=command,
|
|
119
|
+
pad_samples=pad_samples,
|
|
120
|
+
)
|
|
121
|
+
return audio
|
|
122
|
+
emit_warning(
|
|
123
|
+
f"{command}: the joined track is {audio_seconds:.3f} s and the "
|
|
124
|
+
f"joined video is {video_seconds:.3f} s ({total_frames} frames at "
|
|
125
|
+
f"{fps:g} fps) - padded the track with {pad_samples} sample"
|
|
126
|
+
f"{'s' if pad_samples != 1 else ''} of silence to reach the frame "
|
|
127
|
+
"grid, so the shortfall does not carry into a later join.",
|
|
128
|
+
kind="joined_audio_padded_to_frames",
|
|
129
|
+
command=command,
|
|
130
|
+
audio_seconds=audio_seconds,
|
|
131
|
+
video_seconds=video_seconds,
|
|
132
|
+
pad_samples=pad_samples,
|
|
133
|
+
)
|
|
134
|
+
return audio
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def slice_samples(waveform, start, length):
|
|
138
|
+
"""Cut length samples out of a (channels, samples) waveform from start.
|
|
139
|
+
|
|
140
|
+
A slice reaching past the end of the waveform is zero-padded to the
|
|
141
|
+
requested length, so frame-aligned slicing near the end of a track always
|
|
142
|
+
yields full-size chunks.
|
|
143
|
+
"""
|
|
144
|
+
channels, total = waveform.shape
|
|
145
|
+
piece = waveform[:, start : start + length]
|
|
146
|
+
if piece.shape[1] < length:
|
|
147
|
+
padding = numpy.zeros((channels, length - piece.shape[1]), dtype=waveform.dtype)
|
|
148
|
+
piece = numpy.concatenate([piece, padding], axis=1)
|
|
149
|
+
return piece
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def equal_power_crossfade_join(
|
|
153
|
+
previous, head, following, sample_rate, crossfade_ms, seam_fade_ms=None
|
|
154
|
+
):
|
|
155
|
+
"""Join two segments' audio at a seam without changing the total duration.
|
|
156
|
+
|
|
157
|
+
previous ends at the seam. head is the audio trimmed off the next segment's
|
|
158
|
+
start - it covers the same stretch of time as the tail of previous, so the
|
|
159
|
+
two are blended with an equal-power crossfade over the last
|
|
160
|
+
min(crossfade_ms, len(head)) of that stretch. following is the next
|
|
161
|
+
segment's on-timeline audio and is appended unchanged.
|
|
162
|
+
|
|
163
|
+
With no head material (nothing was trimmed), the seam gets a fade-out and
|
|
164
|
+
fade-in in place instead, of seam_fade_ms - a few milliseconds by default,
|
|
165
|
+
just enough not to click.
|
|
166
|
+
"""
|
|
167
|
+
previous, head, following = _matched_channels(previous, head, following)
|
|
168
|
+
|
|
169
|
+
window = min(
|
|
170
|
+
int(crossfade_ms / 1000.0 * sample_rate),
|
|
171
|
+
head.shape[1],
|
|
172
|
+
previous.shape[1],
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
if window == 0:
|
|
176
|
+
return _declick_join(previous, following, sample_rate, seam_fade_ms)
|
|
177
|
+
|
|
178
|
+
fade_out, fade_in = _equal_power_ramps(window)
|
|
179
|
+
blended = previous[:, -window:] * fade_out + head[:, -window:] * fade_in
|
|
180
|
+
return numpy.concatenate([previous[:, :-window], blended, following], axis=1)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
# Below this, the reversed tail's spectral energy is concentrated in a few
|
|
184
|
+
# bins rather than spread across the band - speech or a pitched/tonal element
|
|
185
|
+
# rather than room tone or crowd noise, the direction-agnostic material a
|
|
186
|
+
# bleed is meant for
|
|
187
|
+
TONAL_FLATNESS_THRESHOLD = 0.3
|
|
188
|
+
|
|
189
|
+
# Above this, the tail's waveform repeats closely enough within a plausible
|
|
190
|
+
# pitch period to be voiced speech or a pitched note rather than noise -
|
|
191
|
+
# a bandwidth-insensitive companion to flatness, since a resample's own
|
|
192
|
+
# band limiting does not touch how periodic the waveform is (#198)
|
|
193
|
+
HARMONICITY_THRESHOLD = 0.45
|
|
194
|
+
|
|
195
|
+
# Typical fundamental range for a human voice or a pitched instrument note;
|
|
196
|
+
# the periodicity search only looks at lags in this range so a slow room-tone
|
|
197
|
+
# swell or hum near DC cannot register as a pitch
|
|
198
|
+
_PERIODICITY_MIN_HZ = 60.0
|
|
199
|
+
_PERIODICITY_MAX_HZ = 500.0
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _spectral_flatness(waveform, sample_rate=None, native_sample_rate=None):
|
|
203
|
+
"""Geometric-mean-over-arithmetic-mean of the magnitude spectrum, averaged
|
|
204
|
+
across channels - near 0 for tonal/speech material, near 1 for noise-like
|
|
205
|
+
material (see TONAL_FLATNESS_THRESHOLD).
|
|
206
|
+
|
|
207
|
+
When the material was upsampled, band-limited interpolation leaves near
|
|
208
|
+
zero energy above the original Nyquist - a large near-silent band that
|
|
209
|
+
depresses the geometric mean relative to the arithmetic one regardless of
|
|
210
|
+
what the material actually is, reading as spuriously tonal (#198). Given
|
|
211
|
+
both rates, the spectrum is limited to bins below the native Nyquist so an
|
|
212
|
+
upsampled tail is measured the same as it would be at its own rate.
|
|
213
|
+
"""
|
|
214
|
+
spectrum = numpy.abs(numpy.fft.rfft(waveform, axis=1))
|
|
215
|
+
if sample_rate and native_sample_rate and native_sample_rate < sample_rate:
|
|
216
|
+
native_bins = max(
|
|
217
|
+
2,
|
|
218
|
+
int(spectrum.shape[1] * native_sample_rate / sample_rate),
|
|
219
|
+
)
|
|
220
|
+
spectrum = spectrum[:, :native_bins]
|
|
221
|
+
spectrum = numpy.maximum(spectrum, 1e-10)
|
|
222
|
+
geometric_mean = numpy.exp(numpy.mean(numpy.log(spectrum), axis=1))
|
|
223
|
+
arithmetic_mean = numpy.mean(spectrum, axis=1)
|
|
224
|
+
return float(numpy.mean(geometric_mean / arithmetic_mean))
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _harmonicity(waveform, sample_rate):
|
|
228
|
+
"""Normalized autocorrelation peak within a plausible pitch range,
|
|
229
|
+
averaged across channels - near 1 for a strongly periodic signal (voiced
|
|
230
|
+
speech, a pitched note), near 0 for noise (see HARMONICITY_THRESHOLD).
|
|
231
|
+
|
|
232
|
+
Spectral flatness alone missed real speech (#198): a vowel's formants
|
|
233
|
+
spread its energy broadly enough across the band that flatness reads
|
|
234
|
+
similar to noise, even though the waveform itself repeats every pitch
|
|
235
|
+
period. Autocorrelation measures that repetition directly and is
|
|
236
|
+
insensitive to how the spectrum happens to be shaped, so it catches what
|
|
237
|
+
flatness cannot.
|
|
238
|
+
"""
|
|
239
|
+
min_lag = max(int(sample_rate / _PERIODICITY_MAX_HZ), 1)
|
|
240
|
+
max_lag = min(int(sample_rate / _PERIODICITY_MIN_HZ), waveform.shape[1] - 1)
|
|
241
|
+
if max_lag <= min_lag:
|
|
242
|
+
return 0.0
|
|
243
|
+
|
|
244
|
+
scores = []
|
|
245
|
+
for channel in waveform:
|
|
246
|
+
centered = channel - channel.mean()
|
|
247
|
+
energy = float(numpy.dot(centered, centered))
|
|
248
|
+
if energy <= 1e-12:
|
|
249
|
+
continue
|
|
250
|
+
correlation = numpy.correlate(centered, centered, mode="full")
|
|
251
|
+
zero_lag = correlation.shape[0] // 2
|
|
252
|
+
window = correlation[zero_lag + min_lag : zero_lag + max_lag + 1]
|
|
253
|
+
if window.size == 0:
|
|
254
|
+
continue
|
|
255
|
+
scores.append(float(numpy.max(window) / energy))
|
|
256
|
+
return max(scores) if scores else 0.0
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def bleed_join(
|
|
260
|
+
previous,
|
|
261
|
+
following,
|
|
262
|
+
sample_rate,
|
|
263
|
+
bleed_ms,
|
|
264
|
+
seam_fade_ms=None,
|
|
265
|
+
gain_db=0.0,
|
|
266
|
+
native_sample_rate=None,
|
|
267
|
+
):
|
|
268
|
+
"""Butt-join two waveforms, ringing the outgoing tail on across the seam.
|
|
269
|
+
|
|
270
|
+
Cut-based workflows generate every shot independently, so nothing overlaps at
|
|
271
|
+
a seam and there is no trimmed material to crossfade. Generated shots also
|
|
272
|
+
tend to open on near-silence and end mid-sound - a laugh track still rolling,
|
|
273
|
+
a room still ringing - so a plain butt-join drops a wall of sound into a hole.
|
|
274
|
+
|
|
275
|
+
This lays a decaying copy of the outgoing tail over the head of the incoming
|
|
276
|
+
waveform, the way an audience carries across a picture cut. The copy is
|
|
277
|
+
time-reversed so it starts on the outgoing waveform's own last sample and the
|
|
278
|
+
seam stays continuous without a declick fade; crowd noise and room tone are
|
|
279
|
+
direction-agnostic, so the reversal itself is not audible - speech or a
|
|
280
|
+
tonal/musical tail is not, which is what gets a warning below rather than a
|
|
281
|
+
refusal, since a caller who has already listened to the material may still
|
|
282
|
+
want the bleed.
|
|
283
|
+
|
|
284
|
+
The tail is added to whatever the incoming waveform already carries, and
|
|
285
|
+
neither side is shortened, so frames and samples stay in step.
|
|
286
|
+
|
|
287
|
+
Args:
|
|
288
|
+
previous: Waveform ending at the seam
|
|
289
|
+
following: Waveform starting at the seam
|
|
290
|
+
sample_rate: Sample rate of both waveforms
|
|
291
|
+
bleed_ms: How long the tail rings on, clamped to the material available
|
|
292
|
+
seam_fade_ms: Fade applied on each side of the seam when there is no
|
|
293
|
+
material to bleed at all
|
|
294
|
+
gain_db: Gain applied to the bled copy before it is added, in dB -
|
|
295
|
+
negative ducks a tail that would otherwise push the seam over
|
|
296
|
+
0 dBFS; 0 (the default) is unchanged, full-scale, the prior
|
|
297
|
+
behavior
|
|
298
|
+
native_sample_rate: The rate the outgoing tail was actually recorded
|
|
299
|
+
or generated at, when that differs from sample_rate because the
|
|
300
|
+
caller upsampled it to join. Band-limits the flatness check to
|
|
301
|
+
below the tail's own Nyquist, so upsampling's near-silent high
|
|
302
|
+
band cannot itself read as tonal (#198). Omit when the tail is
|
|
303
|
+
already at its native rate
|
|
304
|
+
|
|
305
|
+
Returns:
|
|
306
|
+
The two waveforms joined, of their full combined length
|
|
307
|
+
"""
|
|
308
|
+
previous, following = _matched_channels(previous, following)
|
|
309
|
+
|
|
310
|
+
window = min(
|
|
311
|
+
int(bleed_ms / 1000.0 * sample_rate),
|
|
312
|
+
previous.shape[1],
|
|
313
|
+
following.shape[1],
|
|
314
|
+
)
|
|
315
|
+
if window <= 0:
|
|
316
|
+
return _declick_join(previous, following, sample_rate, seam_fade_ms)
|
|
317
|
+
|
|
318
|
+
tail = previous[:, ::-1][:, :window]
|
|
319
|
+
|
|
320
|
+
tail_source = previous[:, -window:]
|
|
321
|
+
flatness = _spectral_flatness(tail_source, sample_rate, native_sample_rate)
|
|
322
|
+
# sample_rate, not native_sample_rate: _harmonicity turns a rate into lag
|
|
323
|
+
# bounds in samples of the waveform it is handed, and that waveform is at
|
|
324
|
+
# sample_rate however it got there. Passing the native rate of an upsampled
|
|
325
|
+
# tail searched the wrong lag range (16k against a 48k track: 180-1500 Hz
|
|
326
|
+
# rather than 60-500) and could miss the voiced speech #198 added it for.
|
|
327
|
+
# Only _spectral_flatness wants the native rate, to band-limit its window.
|
|
328
|
+
harmonicity = _harmonicity(tail_source, sample_rate)
|
|
329
|
+
if flatness < TONAL_FLATNESS_THRESHOLD or harmonicity > HARMONICITY_THRESHOLD:
|
|
330
|
+
emit_warning(
|
|
331
|
+
f"bleed_join: the tail being reversed onto the seam looks tonal or "
|
|
332
|
+
f"speech-like (spectral flatness {flatness:.2f}, harmonicity "
|
|
333
|
+
f"{harmonicity:.2f}) rather than the room tone or crowd noise a "
|
|
334
|
+
f"bleed is meant for - the reversal is likely to be audible as a "
|
|
335
|
+
f"stutter or a note running backwards. Pass 'audio_bleed_ms': 0 "
|
|
336
|
+
f"for a hard cut on this material instead - seam_fade_ms has no "
|
|
337
|
+
f"effect while audio_bleed_ms is non-zero.",
|
|
338
|
+
kind="bleed_tonal_material",
|
|
339
|
+
command="bleed_join",
|
|
340
|
+
flatness=round(flatness, 3),
|
|
341
|
+
harmonicity=round(harmonicity, 3),
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
decay, _ = _equal_power_ramps(window) # cos: 1 down to ~0
|
|
345
|
+
gain = 10.0 ** (gain_db / 20.0) if gain_db else 1.0
|
|
346
|
+
following = following.copy()
|
|
347
|
+
following[:, :window] += tail * decay * gain
|
|
348
|
+
|
|
349
|
+
peak = numpy.abs(following[:, :window]).max()
|
|
350
|
+
if peak > 1.0:
|
|
351
|
+
logger.warning(
|
|
352
|
+
f"Audio bleed pushed the seam to {peak:.2f} - it is added to the "
|
|
353
|
+
f"incoming track, which was not silent enough to absorb it. Pass "
|
|
354
|
+
f"a negative gain_db to duck the bled copy."
|
|
355
|
+
)
|
|
356
|
+
return numpy.concatenate([previous, following], axis=1)
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def crossfade_concat(waveforms, sample_rate, crossfade_ms, starts=None):
|
|
360
|
+
"""Concatenate waveforms, overlapping each seam by an equal-power crossfade.
|
|
361
|
+
|
|
362
|
+
The classic crossfade: each seam overlaps the two waveforms by the fade
|
|
363
|
+
window, so the result is shorter than the plain sum by one window per seam.
|
|
364
|
+
|
|
365
|
+
`starts`, when given a list, is filled with the sample each waveform
|
|
366
|
+
begins at in the result - where its crossfade opens - measured as the
|
|
367
|
+
result grows rather than worked out from the lengths (#378).
|
|
368
|
+
"""
|
|
369
|
+
waveforms = [as_channels_samples(waveform) for waveform in waveforms]
|
|
370
|
+
if not waveforms:
|
|
371
|
+
raise ValueError("No waveforms to concatenate")
|
|
372
|
+
|
|
373
|
+
result = waveforms[0]
|
|
374
|
+
if starts is not None:
|
|
375
|
+
starts.append(0)
|
|
376
|
+
for following in waveforms[1:]:
|
|
377
|
+
result, following = _matched_channels(result, following)
|
|
378
|
+
window = min(
|
|
379
|
+
int(round(crossfade_ms / 1000.0 * sample_rate)),
|
|
380
|
+
result.shape[1],
|
|
381
|
+
following.shape[1],
|
|
382
|
+
)
|
|
383
|
+
if starts is not None:
|
|
384
|
+
starts.append(result.shape[1] - window)
|
|
385
|
+
if window == 0:
|
|
386
|
+
result = _declick_join(result, following, sample_rate)
|
|
387
|
+
continue
|
|
388
|
+
|
|
389
|
+
fade_out, fade_in = _equal_power_ramps(window)
|
|
390
|
+
blended = result[:, -window:] * fade_out + following[:, :window] * fade_in
|
|
391
|
+
result = numpy.concatenate(
|
|
392
|
+
[result[:, :-window], blended, following[:, window:]], axis=1
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
return result
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def load_audio(location, base_dir=None):
|
|
399
|
+
"""Load an audio file from a local path or http(s) URL.
|
|
400
|
+
|
|
401
|
+
A video file loads too, and contributes the track muxed into it: the cut
|
|
402
|
+
an earlier run wrote is exactly what a scoring pass wants to mix under,
|
|
403
|
+
and refusing its extension made an agent extract the audio by hand
|
|
404
|
+
(2026-09-08). A video without an audio stream is an error, not silence.
|
|
405
|
+
|
|
406
|
+
Returns:
|
|
407
|
+
Tuple of a (channels, samples) float32 waveform and its sample rate
|
|
408
|
+
"""
|
|
409
|
+
from ..security import ALLOWED_VIDEO_EXTENSIONS
|
|
410
|
+
|
|
411
|
+
extension = os.path.splitext(location.split("?", 1)[0])[1].lower()
|
|
412
|
+
if extension in ALLOWED_VIDEO_EXTENSIONS:
|
|
413
|
+
from .video_utils import load_audio_video
|
|
414
|
+
|
|
415
|
+
video = load_audio_video(location, base_dir=base_dir)
|
|
416
|
+
if video.audio is None:
|
|
417
|
+
raise ValueError(
|
|
418
|
+
f"{location} carries no audio track - a video's soundtrack is "
|
|
419
|
+
"what an audio task takes from it"
|
|
420
|
+
)
|
|
421
|
+
return as_channels_samples(video.audio), video.sample_rate
|
|
422
|
+
|
|
423
|
+
if location.startswith(("http://", "https://")):
|
|
424
|
+
from ..locations import safe_get
|
|
425
|
+
|
|
426
|
+
logger.debug(f"Downloading audio from {location}")
|
|
427
|
+
response = safe_get(location, "an audio argument", timeout=60)
|
|
428
|
+
data, sample_rate = soundfile.read(
|
|
429
|
+
io.BytesIO(response.content), dtype="float32"
|
|
430
|
+
)
|
|
431
|
+
else:
|
|
432
|
+
from ..locations import validate_media_path
|
|
433
|
+
|
|
434
|
+
validated_path = validate_media_path(location, base_dir, "an audio argument")
|
|
435
|
+
validate_file_extension(validated_path, ALLOWED_AUDIO_EXTENSIONS)
|
|
436
|
+
logger.debug(f"Reading audio from {validated_path}")
|
|
437
|
+
data, sample_rate = soundfile.read(validated_path, dtype="float32")
|
|
438
|
+
|
|
439
|
+
# soundfile returns (samples,) or (samples, channels)
|
|
440
|
+
return as_channels_samples(data), sample_rate
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def _as_track(waveform, sample_rate, command="an audio task", source_mean_dbfs=None):
|
|
444
|
+
"""An audio task's return value: the waveform with the rate it is at.
|
|
445
|
+
|
|
446
|
+
Every one of these commands already knows the rate - it was given, or it
|
|
447
|
+
came off the file or the video the track was taken from - and dropping it
|
|
448
|
+
on the way out made the next command in the chain ask for it again. A
|
|
449
|
+
resample fed straight from a slice failed for want of a number the slice
|
|
450
|
+
had read and thrown away (2026-09-11). An AudioTrack carries it, and
|
|
451
|
+
everything downstream of audio reads '.audio'/'.sample_rate' already; a
|
|
452
|
+
'sample_rate' the workflow declares on the result still wins at save.
|
|
453
|
+
|
|
454
|
+
A rate that is not a rate stops here. Saving falls back to 44100 Hz for a
|
|
455
|
+
track that carries none (DEFAULT_AUDIO_SAMPLE_RATE in result.py), so a
|
|
456
|
+
zero handed through would have been written as a 44100 Hz header over
|
|
457
|
+
samples at some other rate - the same audio at the wrong speed and pitch,
|
|
458
|
+
reported as a success (#140). There is no waveform whose rate is zero, so
|
|
459
|
+
the only thing to do with one is refuse it.
|
|
460
|
+
"""
|
|
461
|
+
from ..result import AudioTrack
|
|
462
|
+
|
|
463
|
+
rate = int(sample_rate) if sample_rate is not None else 0
|
|
464
|
+
if rate <= 0:
|
|
465
|
+
raise ValueError(
|
|
466
|
+
f"{command} ended up with a sample rate of {sample_rate!r}, which "
|
|
467
|
+
f"is not a rate. Labelling a waveform with a rate it is not at "
|
|
468
|
+
f"changes its speed and pitch, so it is refused rather than "
|
|
469
|
+
f"written"
|
|
470
|
+
)
|
|
471
|
+
return AudioTrack(
|
|
472
|
+
numpy.ascontiguousarray(waveform), rate, source_mean_dbfs=source_mean_dbfs
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _as_number(value, kind, name, command="slice_audio"):
|
|
477
|
+
"""Coerce a numeric task argument given as a string, leaving None alone."""
|
|
478
|
+
if not isinstance(value, str):
|
|
479
|
+
return value
|
|
480
|
+
try:
|
|
481
|
+
return kind(value)
|
|
482
|
+
except ValueError as e:
|
|
483
|
+
raise ValueError(f"{command} needs a number for '{name}', got {value!r}") from e
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def slice_audio(
|
|
487
|
+
audio,
|
|
488
|
+
start_seconds=None,
|
|
489
|
+
duration_seconds=None,
|
|
490
|
+
start_frame=None,
|
|
491
|
+
num_frames=None,
|
|
492
|
+
fps=None,
|
|
493
|
+
sample_rate=None,
|
|
494
|
+
):
|
|
495
|
+
"""Task command: cut a slice out of an audio track.
|
|
496
|
+
|
|
497
|
+
The slice is addressed either in seconds (start_seconds + duration_seconds)
|
|
498
|
+
or in video frames (start_frame + num_frames + fps).
|
|
499
|
+
|
|
500
|
+
A slice reaching past the end of the track is zero-padded to the length
|
|
501
|
+
asked for - it does not fail and it is not shortened - and the padding is
|
|
502
|
+
digital silence, so asking for more than the source holds returns a track
|
|
503
|
+
that is partly empty. Anything past a few milliseconds of that is
|
|
504
|
+
reported as a 'slice_past_end' warning on the job. To fill a cut longer
|
|
505
|
+
than the recording, make a bed with the 'loop_audio' task first
|
|
506
|
+
('target_frames' + 'fps' matches one exactly) and slice that.
|
|
507
|
+
|
|
508
|
+
Either half of a pair may be left out: with no start the slice begins at the
|
|
509
|
+
head of the track, and with no duration it runs to the end of it. A workflow
|
|
510
|
+
that trims only when it is told a length therefore still produces the track
|
|
511
|
+
rather than failing.
|
|
512
|
+
|
|
513
|
+
Args:
|
|
514
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
515
|
+
soundtrack is taken), a video generated with a
|
|
516
|
+
soundtrack (which brings its sample rate along), or a waveform
|
|
517
|
+
(which needs sample_rate alongside it)
|
|
518
|
+
sample_rate: Sample rate of a waveform passed directly; given for a
|
|
519
|
+
file or a video it overrides the rate they carry
|
|
520
|
+
|
|
521
|
+
Returns:
|
|
522
|
+
An AudioTrack holding the slice and the rate it is at, so the next
|
|
523
|
+
audio command in the chain does not have to be told the rate again
|
|
524
|
+
"""
|
|
525
|
+
# A variable a workflow declares null carries no type, so a value given for
|
|
526
|
+
# it on the command line arrives as a string - the same coercion the upscale
|
|
527
|
+
# and interpolation tasks do on their numeric arguments
|
|
528
|
+
start_seconds = _as_number(start_seconds, float, "start_seconds")
|
|
529
|
+
duration_seconds = _as_number(duration_seconds, float, "duration_seconds")
|
|
530
|
+
start_frame = _as_number(start_frame, int, "start_frame")
|
|
531
|
+
num_frames = _as_number(num_frames, int, "num_frames")
|
|
532
|
+
fps = _as_number(fps, Fraction, "fps")
|
|
533
|
+
|
|
534
|
+
# A count or an offset outside its domain is refused rather than handed to
|
|
535
|
+
# Python's slice semantics, which answered a negative 'num_frames' with
|
|
536
|
+
# the track minus its last N frames and called it a success (#139).
|
|
537
|
+
# validate_workflow refuses a literal one for free; this is the same
|
|
538
|
+
# refusal for a value that arrived from a variable or an earlier step
|
|
539
|
+
check_arguments(
|
|
540
|
+
"slice_audio",
|
|
541
|
+
start_seconds=start_seconds,
|
|
542
|
+
duration_seconds=duration_seconds,
|
|
543
|
+
start_frame=start_frame,
|
|
544
|
+
num_frames=num_frames,
|
|
545
|
+
fps=fps,
|
|
546
|
+
sample_rate=sample_rate,
|
|
547
|
+
)
|
|
548
|
+
|
|
549
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "slice_audio")
|
|
550
|
+
total = waveform.shape[1]
|
|
551
|
+
|
|
552
|
+
if start_seconds is not None or duration_seconds is not None:
|
|
553
|
+
start = int(round((start_seconds or 0) * sample_rate))
|
|
554
|
+
length = (
|
|
555
|
+
max(total - start, 0)
|
|
556
|
+
if duration_seconds is None
|
|
557
|
+
else int(round(duration_seconds * sample_rate))
|
|
558
|
+
)
|
|
559
|
+
elif start_frame is not None or num_frames is not None:
|
|
560
|
+
if fps is None:
|
|
561
|
+
raise ValueError("slice_audio needs 'fps' to address a slice in frames")
|
|
562
|
+
start = frames_to_samples(start_frame or 0, fps, sample_rate)
|
|
563
|
+
length = (
|
|
564
|
+
max(total - start, 0)
|
|
565
|
+
if num_frames is None
|
|
566
|
+
else frames_to_samples(num_frames, fps, sample_rate)
|
|
567
|
+
)
|
|
568
|
+
else:
|
|
569
|
+
raise ValueError(
|
|
570
|
+
"slice_audio needs either 'start_seconds'/'duration_seconds' or "
|
|
571
|
+
"'start_frame'/'num_frames'/'fps'"
|
|
572
|
+
)
|
|
573
|
+
|
|
574
|
+
_warn_on_slice_past_end(total, start, length, sample_rate)
|
|
575
|
+
_warn_on_slice_trims_tail(waveform, total, start, length, sample_rate)
|
|
576
|
+
# #309: a cut out of a source that was already near-silent (room tone,
|
|
577
|
+
# a deliberate quiet bed) is not a defect the slice introduced - measure
|
|
578
|
+
# the source before cutting it down, so save can tell the two apart from
|
|
579
|
+
# a track that arrived at a normal level and something upstream lost
|
|
580
|
+
source_mean_dbfs = level_dbfs(waveform, "rms")
|
|
581
|
+
return _as_track(
|
|
582
|
+
slice_samples(waveform, start, length),
|
|
583
|
+
sample_rate,
|
|
584
|
+
"slice_audio",
|
|
585
|
+
source_mean_dbfs=source_mean_dbfs,
|
|
586
|
+
)
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
def gain_audio(
|
|
590
|
+
audio,
|
|
591
|
+
gain_db,
|
|
592
|
+
start_seconds=None,
|
|
593
|
+
duration_seconds=None,
|
|
594
|
+
start_frame=None,
|
|
595
|
+
num_frames=None,
|
|
596
|
+
fps=None,
|
|
597
|
+
sample_rate=None,
|
|
598
|
+
):
|
|
599
|
+
"""Task command: apply a gain to a region of an audio track.
|
|
600
|
+
|
|
601
|
+
The region is addressed the same way slice_audio's is - either in
|
|
602
|
+
seconds (start_seconds + duration_seconds) or in video frames
|
|
603
|
+
(start_frame + num_frames + fps). Everything outside the region is
|
|
604
|
+
passed through unchanged, so ducking a scene under another is one step
|
|
605
|
+
rather than the slice/gain/mix/rejoin/pair_audio chain that was
|
|
606
|
+
previously the only way to apply a gain to part of a track rather than
|
|
607
|
+
all of it (#187). With no region given at all, the gain applies to the
|
|
608
|
+
whole track - the same "no region means everything" reading mix_audio's
|
|
609
|
+
gains use, and the obvious meaning of "duck this clip by 8 dB" (#395).
|
|
610
|
+
To gain everything from some point on, give just start_seconds=0 (or
|
|
611
|
+
start_frame=0 + fps) and leave duration_seconds/num_frames unset, which
|
|
612
|
+
runs to the end of the track without the caller needing to already know
|
|
613
|
+
how long that is.
|
|
614
|
+
|
|
615
|
+
Unlike slice_audio, a region reaching past the end of the track is
|
|
616
|
+
clipped to it rather than zero-padded: there is no silence there to
|
|
617
|
+
gain, only the end of the real material.
|
|
618
|
+
|
|
619
|
+
A file's or video's own sample rate is read automatically; sample_rate
|
|
620
|
+
is for a waveform passed directly, or to override what a file carries -
|
|
621
|
+
which relabels the waveform at that rate rather than resampling it, the
|
|
622
|
+
same caveat slice_audio's sample_rate carries (#180).
|
|
623
|
+
|
|
624
|
+
Args:
|
|
625
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
626
|
+
soundtrack is taken), a generated video carrying its own
|
|
627
|
+
soundtrack, or a waveform (which needs sample_rate alongside it)
|
|
628
|
+
gain_db: Gain to apply within the region, in decibels - negative
|
|
629
|
+
ducks it, positive boosts it
|
|
630
|
+
start_seconds: Start of the region, in seconds. Omitted along with
|
|
631
|
+
every other region argument, the gain applies to the whole track
|
|
632
|
+
duration_seconds: Length of the region, in seconds
|
|
633
|
+
start_frame: Start of the region, in video frames
|
|
634
|
+
num_frames: Length of the region, in video frames
|
|
635
|
+
fps: Frame rate used to convert start_frame/num_frames to samples
|
|
636
|
+
sample_rate: Sample rate of a waveform passed directly; given for a
|
|
637
|
+
file or a video it overrides the rate they carry
|
|
638
|
+
|
|
639
|
+
Returns:
|
|
640
|
+
An AudioTrack holding the whole track with the region's gain
|
|
641
|
+
applied, and the rate it is at
|
|
642
|
+
"""
|
|
643
|
+
start_seconds = _as_number(
|
|
644
|
+
start_seconds, float, "start_seconds", command="gain_audio"
|
|
645
|
+
)
|
|
646
|
+
duration_seconds = _as_number(
|
|
647
|
+
duration_seconds, float, "duration_seconds", command="gain_audio"
|
|
648
|
+
)
|
|
649
|
+
start_frame = _as_number(start_frame, int, "start_frame", command="gain_audio")
|
|
650
|
+
num_frames = _as_number(num_frames, int, "num_frames", command="gain_audio")
|
|
651
|
+
fps = _as_number(fps, Fraction, "fps", command="gain_audio")
|
|
652
|
+
gain_db = _as_number(gain_db, float, "gain_db", command="gain_audio")
|
|
653
|
+
|
|
654
|
+
check_arguments(
|
|
655
|
+
"gain_audio",
|
|
656
|
+
start_seconds=start_seconds,
|
|
657
|
+
duration_seconds=duration_seconds,
|
|
658
|
+
start_frame=start_frame,
|
|
659
|
+
num_frames=num_frames,
|
|
660
|
+
fps=fps,
|
|
661
|
+
sample_rate=sample_rate,
|
|
662
|
+
)
|
|
663
|
+
|
|
664
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "gain_audio")
|
|
665
|
+
total = waveform.shape[1]
|
|
666
|
+
|
|
667
|
+
if start_seconds is not None or duration_seconds is not None:
|
|
668
|
+
start = int(round((start_seconds or 0) * sample_rate))
|
|
669
|
+
length = (
|
|
670
|
+
max(total - start, 0)
|
|
671
|
+
if duration_seconds is None
|
|
672
|
+
else int(round(duration_seconds * sample_rate))
|
|
673
|
+
)
|
|
674
|
+
elif start_frame is not None or num_frames is not None:
|
|
675
|
+
if fps is None:
|
|
676
|
+
raise ValueError("gain_audio needs 'fps' to address a region in frames")
|
|
677
|
+
start = frames_to_samples(start_frame or 0, fps, sample_rate)
|
|
678
|
+
length = (
|
|
679
|
+
max(total - start, 0)
|
|
680
|
+
if num_frames is None
|
|
681
|
+
else frames_to_samples(num_frames, fps, sample_rate)
|
|
682
|
+
)
|
|
683
|
+
else:
|
|
684
|
+
start = 0
|
|
685
|
+
length = total
|
|
686
|
+
|
|
687
|
+
region_start = max(0, min(start, total))
|
|
688
|
+
region_end = max(region_start, min(start + max(length, 0), total))
|
|
689
|
+
|
|
690
|
+
gained = waveform.copy()
|
|
691
|
+
if region_end > region_start:
|
|
692
|
+
gain = 10 ** (gain_db / 20)
|
|
693
|
+
gained[:, region_start:region_end] = (
|
|
694
|
+
gained[:, region_start:region_end] * gain
|
|
695
|
+
).astype(waveform.dtype)
|
|
696
|
+
|
|
697
|
+
region_start_seconds = region_start / float(sample_rate)
|
|
698
|
+
region_end_seconds = region_end / float(sample_rate)
|
|
699
|
+
emit_log(
|
|
700
|
+
f"gain_audio: {gain_db:.1f} dB over "
|
|
701
|
+
f"{region_start_seconds:.3f}-{region_end_seconds:.3f} s "
|
|
702
|
+
f"(samples {region_start}-{region_end} @ {sample_rate} Hz)",
|
|
703
|
+
command="gain_audio",
|
|
704
|
+
gain_db=gain_db,
|
|
705
|
+
start_seconds=region_start_seconds,
|
|
706
|
+
duration_seconds=region_end_seconds - region_start_seconds,
|
|
707
|
+
start_sample=region_start,
|
|
708
|
+
end_sample=region_end,
|
|
709
|
+
sample_rate=sample_rate,
|
|
710
|
+
)
|
|
711
|
+
|
|
712
|
+
return _as_track(gained, sample_rate, "gain_audio")
|
|
713
|
+
|
|
714
|
+
|
|
715
|
+
def _warn_on_slice_past_end(total, start, length, sample_rate):
|
|
716
|
+
"""Say when a slice asked for more material than its source holds.
|
|
717
|
+
|
|
718
|
+
slice_samples zero-pads the shortfall, which is what makes frame-aligned
|
|
719
|
+
chunking near the end of a track work at all - but the same padding is
|
|
720
|
+
how a score shorter than the film it is laid under leaves the film
|
|
721
|
+
unscored for the rest of its length, with nothing anywhere saying so
|
|
722
|
+
(#126). emit_warning rather than logger.warning for the reason the
|
|
723
|
+
concat_videos resample warning is emitted: silently substituting silence
|
|
724
|
+
for four fifths of a track is an audio decision made on the caller's
|
|
725
|
+
behalf, and a caller reading the job over the API or MCP sees the
|
|
726
|
+
warnings list and nothing else (#82, #108).
|
|
727
|
+
"""
|
|
728
|
+
available = max(0, min(total - start, length))
|
|
729
|
+
padded = length - available
|
|
730
|
+
if padded <= 0 or not sample_rate:
|
|
731
|
+
return
|
|
732
|
+
padded_seconds = padded / float(sample_rate)
|
|
733
|
+
if padded_seconds * 1000.0 < SLICE_PAD_WARN_MS:
|
|
734
|
+
# Frame-aligned slicing lands a sample or two past the end routinely;
|
|
735
|
+
# that is rounding, not a decision anyone can act on
|
|
736
|
+
return
|
|
737
|
+
emit_warning(
|
|
738
|
+
f"slice_audio: the requested slice runs "
|
|
739
|
+
f"{padded_seconds:.2f} s past the end of a "
|
|
740
|
+
f"{total / float(sample_rate):.2f} s source, so that much of the "
|
|
741
|
+
f"{length / float(sample_rate):.2f} s returned is digital silence. "
|
|
742
|
+
f"If you meant to fill a cut of this length, make a bed with the "
|
|
743
|
+
f"'loop_audio' task ('target_frames' + 'fps' matches one exactly) "
|
|
744
|
+
f"and slice that; if you meant the tail pad, nothing is wrong.",
|
|
745
|
+
kind="slice_past_end",
|
|
746
|
+
command="slice_audio",
|
|
747
|
+
source_seconds=round(total / float(sample_rate), 3),
|
|
748
|
+
requested_seconds=round(length / float(sample_rate), 3),
|
|
749
|
+
padded_seconds=round(padded_seconds, 3),
|
|
750
|
+
sample_rate=sample_rate,
|
|
751
|
+
)
|
|
752
|
+
|
|
753
|
+
|
|
754
|
+
def _warn_on_slice_trims_tail(waveform, total, start, length, sample_rate):
|
|
755
|
+
"""Say when a slice left material behind that the caller likely wanted.
|
|
756
|
+
|
|
757
|
+
slice_audio is a slice, so most unused remainders are deliberate excerpts
|
|
758
|
+
and warning on every one would be noise. What #342 found is a narrower
|
|
759
|
+
signature: a cut landing a few seconds short of a source's natural end
|
|
760
|
+
(a frame-lattice total that cannot land exactly on the score's length)
|
|
761
|
+
silently drops the source's tail, including whatever is loudest there.
|
|
762
|
+
Only fires when the dropped remainder is both short in absolute terms
|
|
763
|
+
and small next to the slice itself, and only when that remainder is not
|
|
764
|
+
already silence - a track that legitimately ends in a fade should not
|
|
765
|
+
warn just because its last seconds are quiet.
|
|
766
|
+
"""
|
|
767
|
+
if not sample_rate or length <= 0:
|
|
768
|
+
return
|
|
769
|
+
slice_end = start + length
|
|
770
|
+
remainder = total - slice_end
|
|
771
|
+
if remainder <= 0:
|
|
772
|
+
return
|
|
773
|
+
remainder_seconds = remainder / float(sample_rate)
|
|
774
|
+
if remainder_seconds >= SLICE_TRIM_WARN_SECONDS:
|
|
775
|
+
return
|
|
776
|
+
if remainder_seconds / (length / float(sample_rate)) >= SLICE_TRIM_WARN_FRACTION:
|
|
777
|
+
return
|
|
778
|
+
dropped = waveform[:, slice_end:total]
|
|
779
|
+
peak_dbfs = level_dbfs(dropped, "peak")
|
|
780
|
+
if peak_dbfs is None:
|
|
781
|
+
# No level at all is silence - nothing was lost
|
|
782
|
+
return
|
|
783
|
+
emit_warning(
|
|
784
|
+
f"slice_audio: the slice ends {remainder_seconds:.2f} s before the "
|
|
785
|
+
f"{total / float(sample_rate):.2f} s source does, dropping its tail "
|
|
786
|
+
f"(peak {peak_dbfs:.1f} dBFS in the dropped {remainder_seconds:.2f} s) "
|
|
787
|
+
f"- if the slice was meant to reach the source's end, adjust "
|
|
788
|
+
f"start/length to land there, or fade the source's own tail first",
|
|
789
|
+
kind="slice_trimmed_tail",
|
|
790
|
+
command="slice_audio",
|
|
791
|
+
source_seconds=round(total / float(sample_rate), 3),
|
|
792
|
+
dropped_seconds=round(remainder_seconds, 3),
|
|
793
|
+
dropped_peak_dbfs=round(peak_dbfs, 1),
|
|
794
|
+
sample_rate=sample_rate,
|
|
795
|
+
)
|
|
796
|
+
|
|
797
|
+
|
|
798
|
+
def resample_waveform(waveform, sample_rate, target_sample_rate):
|
|
799
|
+
"""A waveform at a different rate, as a plain (channels, samples) array.
|
|
800
|
+
|
|
801
|
+
The conversion resample_audio performs, without the task's argument
|
|
802
|
+
handling or its AudioTrack return, so a task that has waveforms in hand
|
|
803
|
+
already can reach the rate conversion directly.
|
|
804
|
+
"""
|
|
805
|
+
# PyAV's resampler accepts a zero rate and answers with the samples
|
|
806
|
+
# unchanged, which is indistinguishable from a conversion that happened
|
|
807
|
+
# (#140) - so neither rate is allowed to be one that cannot be a rate
|
|
808
|
+
for name, rate in (("sample_rate", sample_rate), ("target", target_sample_rate)):
|
|
809
|
+
if as_number(rate) is None or as_number(rate) <= 0:
|
|
810
|
+
raise ValueError(
|
|
811
|
+
f"resample_waveform needs a {name} above zero, got {rate!r}"
|
|
812
|
+
)
|
|
813
|
+
if sample_rate == target_sample_rate:
|
|
814
|
+
return waveform
|
|
815
|
+
|
|
816
|
+
import av
|
|
817
|
+
from av.audio.resampler import AudioResampler
|
|
818
|
+
|
|
819
|
+
channels = waveform.shape[0]
|
|
820
|
+
layout = {1: "mono", 2: "stereo"}.get(channels, f"{channels}c")
|
|
821
|
+
frame = av.AudioFrame.from_ndarray(
|
|
822
|
+
numpy.ascontiguousarray(waveform, dtype=numpy.float32),
|
|
823
|
+
format="fltp",
|
|
824
|
+
layout=layout,
|
|
825
|
+
)
|
|
826
|
+
frame.sample_rate = sample_rate
|
|
827
|
+
frame.pts = 0
|
|
828
|
+
frame.time_base = Fraction(1, sample_rate)
|
|
829
|
+
|
|
830
|
+
resampler = AudioResampler(format="fltp", layout=layout, rate=target_sample_rate)
|
|
831
|
+
converted = [f.to_ndarray() for f in resampler.resample(frame)]
|
|
832
|
+
converted += [f.to_ndarray() for f in resampler.resample(None)]
|
|
833
|
+
logger.debug(
|
|
834
|
+
f"Resampled {waveform.shape[1]} samples at {sample_rate}Hz "
|
|
835
|
+
f"to {target_sample_rate}Hz"
|
|
836
|
+
)
|
|
837
|
+
return numpy.concatenate(converted, axis=1).astype(numpy.float32)
|
|
838
|
+
|
|
839
|
+
|
|
840
|
+
def resample_audio(audio, target_sample_rate, sample_rate=None):
|
|
841
|
+
"""Task command: resample an audio track to a different sample rate.
|
|
842
|
+
|
|
843
|
+
MiniMax H3 conditions on audio at its audio VAE's own rate and resamples
|
|
844
|
+
anything else with torchaudio, which dw does not depend on. Resampling a
|
|
845
|
+
supplied recording once, up front, feeds the pipeline what it already wants
|
|
846
|
+
and keeps the dependency out - PyAV, which dw needs for video anyway, does
|
|
847
|
+
the conversion.
|
|
848
|
+
|
|
849
|
+
Args:
|
|
850
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
851
|
+
soundtrack is taken), a video generated with a
|
|
852
|
+
soundtrack (which brings its sample rate along), or a waveform
|
|
853
|
+
(which needs sample_rate alongside it)
|
|
854
|
+
target_sample_rate: Rate to convert to
|
|
855
|
+
sample_rate: Sample rate of a waveform passed directly; given for a
|
|
856
|
+
file or a video it overrides the rate they carry
|
|
857
|
+
|
|
858
|
+
Returns:
|
|
859
|
+
An AudioTrack holding the resampled waveform and its new rate
|
|
860
|
+
"""
|
|
861
|
+
target_sample_rate = _as_number(
|
|
862
|
+
target_sample_rate, int, "target_sample_rate", "resample_audio"
|
|
863
|
+
)
|
|
864
|
+
# A zero or negative rate is not a rate. It used to reach PyAV's resampler,
|
|
865
|
+
# which left the samples alone, and then the save, which fell back to
|
|
866
|
+
# 44100 Hz - the original audio under a header 38% off, reported as a
|
|
867
|
+
# success (#140)
|
|
868
|
+
check_arguments(
|
|
869
|
+
"resample_audio",
|
|
870
|
+
target_sample_rate=target_sample_rate,
|
|
871
|
+
sample_rate=sample_rate,
|
|
872
|
+
)
|
|
873
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "resample_audio")
|
|
874
|
+
resampled = resample_waveform(waveform, sample_rate, target_sample_rate)
|
|
875
|
+
emit_log(
|
|
876
|
+
f"resample_audio: {sample_rate} → {target_sample_rate} Hz, "
|
|
877
|
+
f"{resampled.shape[-1] / target_sample_rate:.2f} s",
|
|
878
|
+
command="resample_audio",
|
|
879
|
+
source_sample_rate=sample_rate,
|
|
880
|
+
target_sample_rate=target_sample_rate,
|
|
881
|
+
seconds=round(resampled.shape[-1] / target_sample_rate, 2),
|
|
882
|
+
)
|
|
883
|
+
return _as_track(
|
|
884
|
+
resampled,
|
|
885
|
+
target_sample_rate,
|
|
886
|
+
"resample_audio",
|
|
887
|
+
)
|
|
888
|
+
|
|
889
|
+
|
|
890
|
+
def _track_names(audios):
|
|
891
|
+
"""A name per audio track, for a warning that has to say which one.
|
|
892
|
+
|
|
893
|
+
Mirrors concat_videos' video_names: a caller passes a path, or a
|
|
894
|
+
previous step's result; only the path says anything by itself, so the
|
|
895
|
+
rest are named by position.
|
|
896
|
+
"""
|
|
897
|
+
return [
|
|
898
|
+
original if isinstance(original, str) else f"track {index + 1}"
|
|
899
|
+
for index, original in enumerate(audios)
|
|
900
|
+
]
|
|
901
|
+
|
|
902
|
+
|
|
903
|
+
def _load_tracks_matching_rate(audios, sample_rate, command):
|
|
904
|
+
"""Load a list of audio tracks, resampling any that disagree on rate.
|
|
905
|
+
|
|
906
|
+
A mismatch among tracks that bring their own rate has no editorial
|
|
907
|
+
meaning - the same reasoning concat_videos (#108) and dissolve_videos
|
|
908
|
+
(#287) apply to shots - so when the caller has not pinned a rate the
|
|
909
|
+
highest one found is chosen as the target and the rest are converted,
|
|
910
|
+
with a warning naming each track's rate. When the caller *does* pin
|
|
911
|
+
`sample_rate`, _waveform_and_rate's own per-track relabel warning
|
|
912
|
+
(#180) still applies and this function changes nothing about it.
|
|
913
|
+
"""
|
|
914
|
+
names = _track_names(audios)
|
|
915
|
+
waveforms, native_rates, bare = [], [], False
|
|
916
|
+
for audio in audios:
|
|
917
|
+
if isinstance(audio, str) or hasattr(audio, "audio"):
|
|
918
|
+
waveform, rate = _waveform_and_rate(audio, sample_rate, command)
|
|
919
|
+
native_rates.append(rate)
|
|
920
|
+
else:
|
|
921
|
+
waveform, bare = as_channels_samples(audio), True
|
|
922
|
+
native_rates.append(None)
|
|
923
|
+
waveforms.append(waveform)
|
|
924
|
+
|
|
925
|
+
if sample_rate is not None:
|
|
926
|
+
return waveforms, sample_rate
|
|
927
|
+
|
|
928
|
+
resolved = [rate for rate in native_rates if rate is not None]
|
|
929
|
+
if bare or not resolved:
|
|
930
|
+
raise ValueError(f"{command} needs 'sample_rate' with a raw waveform")
|
|
931
|
+
target_rate = max(resolved)
|
|
932
|
+
if len(set(resolved)) > 1:
|
|
933
|
+
per_track = {
|
|
934
|
+
name: rate for name, rate in zip(names, native_rates) if rate is not None
|
|
935
|
+
}
|
|
936
|
+
emit_warning(
|
|
937
|
+
f"{command}: tracks carry audio at different sample rates ("
|
|
938
|
+
+ ", ".join(f"{name}: {rate} Hz" for name, rate in per_track.items())
|
|
939
|
+
+ f") - resampling them all to {target_rate} Hz. Pass "
|
|
940
|
+
"'sample_rate' to pin a different target, or resample ahead of "
|
|
941
|
+
"this step with the 'resample_audio' task.",
|
|
942
|
+
kind="sample_rate_mismatch",
|
|
943
|
+
command=command,
|
|
944
|
+
sample_rate=target_rate,
|
|
945
|
+
sample_rates=per_track,
|
|
946
|
+
)
|
|
947
|
+
waveforms = [
|
|
948
|
+
(
|
|
949
|
+
waveform
|
|
950
|
+
if rate is None or rate == target_rate
|
|
951
|
+
else resample_waveform(waveform, rate, target_rate)
|
|
952
|
+
)
|
|
953
|
+
for waveform, rate in zip(waveforms, native_rates)
|
|
954
|
+
]
|
|
955
|
+
return waveforms, target_rate
|
|
956
|
+
|
|
957
|
+
|
|
958
|
+
def crossfade_audio(audios, crossfade_ms=75, sample_rate=None):
|
|
959
|
+
"""Task command: join audio tracks with an equal-power crossfade.
|
|
960
|
+
|
|
961
|
+
Each seam overlaps the two tracks by the fade window, so the result is
|
|
962
|
+
shorter than the plain sum by one window per seam.
|
|
963
|
+
|
|
964
|
+
Args:
|
|
965
|
+
audios: The tracks to join, in order - waveforms, audio or video file
|
|
966
|
+
paths, or videos generated with a soundtrack
|
|
967
|
+
crossfade_ms: Length of each crossfade
|
|
968
|
+
sample_rate: Sample rate of the joined track. Required unless every
|
|
969
|
+
track brings its own. Left unset, tracks at different rates are
|
|
970
|
+
not a constraint - the highest rate found is used and the rest
|
|
971
|
+
are resampled up to it, with a warning naming which (#108, #287,
|
|
972
|
+
#293). Given here instead, it *relabels* rather than resamples
|
|
973
|
+
any track whose real rate disagrees - changing its speed and
|
|
974
|
+
pitch, not just its rate - which warns separately (#180); use
|
|
975
|
+
resample_audio ahead of this step if conversion is what is
|
|
976
|
+
wanted at a pinned rate
|
|
977
|
+
|
|
978
|
+
Returns:
|
|
979
|
+
An AudioTrack holding the joined waveform and its rate
|
|
980
|
+
"""
|
|
981
|
+
if not isinstance(audios, list) or not audios:
|
|
982
|
+
raise ValueError("crossfade_audio needs a non-empty list of audio tracks")
|
|
983
|
+
waveforms, sample_rate = _load_tracks_matching_rate(
|
|
984
|
+
audios, sample_rate, "crossfade_audio"
|
|
985
|
+
)
|
|
986
|
+
return _as_track(
|
|
987
|
+
crossfade_concat(waveforms, sample_rate, crossfade_ms), sample_rate
|
|
988
|
+
)
|
|
989
|
+
|
|
990
|
+
|
|
991
|
+
# #306: templates/assemble-and-score and templates/dissolve-between-shots
|
|
992
|
+
# both ship a stock world_gain of 1.8 - a deliberate multiplier, not a dB
|
|
993
|
+
# figure typed into the wrong unit - so the not-dB heuristic below has to sit
|
|
994
|
+
# above it
|
|
995
|
+
GAIN_LOOKS_LIKE_DB_ABOVE = 3.0
|
|
996
|
+
|
|
997
|
+
|
|
998
|
+
def mix_audio(audios, gains=None, sample_rate=None):
|
|
999
|
+
"""Task command: layer audio tracks on top of one another.
|
|
1000
|
+
|
|
1001
|
+
crossfade_audio puts tracks one after another; this puts them on top of
|
|
1002
|
+
each other. It is what a score laid under a film's own sound needs: the
|
|
1003
|
+
music runs unbroken while the world underneath it is replaced at every cut.
|
|
1004
|
+
|
|
1005
|
+
Tracks of different lengths are padded with silence to the longest, so a
|
|
1006
|
+
score shorter than the picture leaves the tail dry rather than cutting the
|
|
1007
|
+
picture down to fit.
|
|
1008
|
+
|
|
1009
|
+
Summing can push peaks past full scale. This returns the plain weighted sum
|
|
1010
|
+
and does not rescale it, since quietening a mix is a decision about how it
|
|
1011
|
+
should sound - follow it with normalize_audio to bring the peak back down.
|
|
1012
|
+
|
|
1013
|
+
Args:
|
|
1014
|
+
audios: The tracks to layer - waveforms, audio or video file paths, or videos
|
|
1015
|
+
generated with a soundtrack
|
|
1016
|
+
gains: One plain multiplier per track, in the same order - not decibels.
|
|
1017
|
+
Defaults to unity on every track
|
|
1018
|
+
sample_rate: Sample rate of the joined mix. Required unless every
|
|
1019
|
+
track brings its own. Left unset, tracks at different rates are
|
|
1020
|
+
not a constraint - the highest rate found is used and the rest
|
|
1021
|
+
are resampled up to it, with a warning naming which (#108, #287,
|
|
1022
|
+
#293). Given here instead, it *relabels* rather than resamples
|
|
1023
|
+
any track whose real rate disagrees - changing its speed and
|
|
1024
|
+
pitch, not just its rate - which warns separately (#180); use
|
|
1025
|
+
resample_audio ahead of this step if conversion is what is
|
|
1026
|
+
wanted at a pinned rate
|
|
1027
|
+
|
|
1028
|
+
Returns:
|
|
1029
|
+
An AudioTrack holding the mixed waveform and its rate
|
|
1030
|
+
"""
|
|
1031
|
+
if not isinstance(audios, list) or not audios:
|
|
1032
|
+
raise ValueError("mix_audio needs a non-empty list of audio tracks")
|
|
1033
|
+
if gains is not None and len(gains) != len(audios):
|
|
1034
|
+
raise ValueError(
|
|
1035
|
+
f"mix_audio needs one gain per track - got {len(gains)} for "
|
|
1036
|
+
f"{len(audios)} tracks"
|
|
1037
|
+
)
|
|
1038
|
+
check_arguments("mix_audio", gains=gains, sample_rate=sample_rate)
|
|
1039
|
+
if gains is not None:
|
|
1040
|
+
# #306: a modest boost (a stock template's world_gain: 1.8 among them)
|
|
1041
|
+
# is a legitimate multiplier a caller chose on purpose, not a typo -
|
|
1042
|
+
# only a gain loud enough that a caller almost certainly meant it as
|
|
1043
|
+
# dB (12, 6, 20, ...) is worth flagging. GAIN_LOOKS_LIKE_DB_ABOVE sits
|
|
1044
|
+
# above any observed catalog default and below the smallest figure a
|
|
1045
|
+
# dB-as-multiplier typo would produce (a "6 dB" or "12 dB" boost)
|
|
1046
|
+
loud = [
|
|
1047
|
+
g
|
|
1048
|
+
for g in gains
|
|
1049
|
+
if as_number(g) is not None and as_number(g) > GAIN_LOOKS_LIKE_DB_ABOVE
|
|
1050
|
+
]
|
|
1051
|
+
if loud:
|
|
1052
|
+
emit_warning(
|
|
1053
|
+
f"mix_audio: gain(s) {loud} are a multiplier, not decibels - "
|
|
1054
|
+
f"a value like 12, 6 or -3 is almost always a dB figure typed "
|
|
1055
|
+
f"into the wrong unit. A multiplier above 1 boosts the track; "
|
|
1056
|
+
f"convert a dB figure with 10 ** (db / 20) if that was intended.",
|
|
1057
|
+
kind="mix_audio_gain_not_db",
|
|
1058
|
+
command="mix_audio",
|
|
1059
|
+
gains=gains,
|
|
1060
|
+
)
|
|
1061
|
+
|
|
1062
|
+
waveforms, sample_rate = _load_tracks_matching_rate(
|
|
1063
|
+
audios, sample_rate, "mix_audio"
|
|
1064
|
+
)
|
|
1065
|
+
|
|
1066
|
+
waveforms = _matched_channels(*waveforms)
|
|
1067
|
+
channels = waveforms[0].shape[0]
|
|
1068
|
+
length = max(waveform.shape[1] for waveform in waveforms)
|
|
1069
|
+
|
|
1070
|
+
mixed = numpy.zeros((channels, length), dtype=numpy.float32)
|
|
1071
|
+
applied = []
|
|
1072
|
+
for index, waveform in enumerate(waveforms):
|
|
1073
|
+
gain = 1.0 if gains is None else float(gains[index])
|
|
1074
|
+
applied.append(gain)
|
|
1075
|
+
mixed[:, : waveform.shape[1]] += waveform * gain
|
|
1076
|
+
emit_log(
|
|
1077
|
+
f"mix_audio: {len(waveforms)} tracks, gains {applied}",
|
|
1078
|
+
command="mix_audio",
|
|
1079
|
+
gains=applied,
|
|
1080
|
+
)
|
|
1081
|
+
return _as_track(mixed, sample_rate, "mix_audio")
|
|
1082
|
+
|
|
1083
|
+
|
|
1084
|
+
def loop_audio(
|
|
1085
|
+
audio,
|
|
1086
|
+
duration_seconds=None,
|
|
1087
|
+
target_frames=None,
|
|
1088
|
+
fps=None,
|
|
1089
|
+
crossfade_ms=250,
|
|
1090
|
+
sample_rate=None,
|
|
1091
|
+
):
|
|
1092
|
+
"""Task command: make a bed of a given length out of a short recording.
|
|
1093
|
+
|
|
1094
|
+
A cut between two independently generated shots has a hole in it: each
|
|
1095
|
+
shot carries its own room, and nothing runs underneath the seam. A
|
|
1096
|
+
continuous bed laid under the whole cut is what fills it - the way a
|
|
1097
|
+
location's room tone is laid under a dialogue scene so the edits stop
|
|
1098
|
+
being audible - and a bed is made by looping a few seconds of tone to
|
|
1099
|
+
the length of the picture.
|
|
1100
|
+
|
|
1101
|
+
Laps are joined with an equal-power crossfade rather than butted
|
|
1102
|
+
together, so the loop point itself is not a click. That only smooths the
|
|
1103
|
+
seam, though: a transient in the source (a hit, a swell) still recurs
|
|
1104
|
+
once per lap at full strength, so the loop still reads as a level pulse
|
|
1105
|
+
at the lap rate - measured at 9.3 dB on a source with one such transient.
|
|
1106
|
+
Picking a source with even internal level avoids the pulse; the
|
|
1107
|
+
crossfade does not. The source is used whole
|
|
1108
|
+
every lap; only the last one is trimmed, to land exactly on the
|
|
1109
|
+
requested length. A source longer than the request is trimmed to it.
|
|
1110
|
+
|
|
1111
|
+
Args:
|
|
1112
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
1113
|
+
soundtrack is taken), a video generated with a soundtrack, or a
|
|
1114
|
+
waveform (which needs sample_rate alongside it)
|
|
1115
|
+
duration_seconds: How long the bed should be, in seconds
|
|
1116
|
+
target_frames: How long the bed should be, in video frames - needs
|
|
1117
|
+
'fps', and is how a bed is matched to a cut exactly
|
|
1118
|
+
fps: Frame rate 'target_frames' is counted at
|
|
1119
|
+
crossfade_ms: Length of the crossfade at each loop point, clamped to
|
|
1120
|
+
the material available
|
|
1121
|
+
sample_rate: Sample rate of a waveform passed directly; given for a
|
|
1122
|
+
file or a video it overrides the rate they carry
|
|
1123
|
+
|
|
1124
|
+
Returns:
|
|
1125
|
+
An AudioTrack holding the bed and the rate it is at
|
|
1126
|
+
"""
|
|
1127
|
+
duration_seconds = _as_number(duration_seconds, float, "duration_seconds")
|
|
1128
|
+
target_frames = _as_number(target_frames, int, "target_frames")
|
|
1129
|
+
fps = _as_number(fps, Fraction, "fps")
|
|
1130
|
+
crossfade_ms = _as_number(crossfade_ms, float, "crossfade_ms")
|
|
1131
|
+
|
|
1132
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "loop_audio")
|
|
1133
|
+
if waveform.size == 0:
|
|
1134
|
+
raise ValueError("loop_audio needs a source with samples in it")
|
|
1135
|
+
|
|
1136
|
+
if duration_seconds is not None:
|
|
1137
|
+
length = int(round(duration_seconds * sample_rate))
|
|
1138
|
+
elif target_frames is not None:
|
|
1139
|
+
if fps is None:
|
|
1140
|
+
raise ValueError("loop_audio needs 'fps' to count a length in frames")
|
|
1141
|
+
length = frames_to_samples(target_frames, fps, sample_rate)
|
|
1142
|
+
else:
|
|
1143
|
+
raise ValueError(
|
|
1144
|
+
"loop_audio needs either 'duration_seconds' or 'target_frames'/'fps' "
|
|
1145
|
+
"to know how long a bed to make"
|
|
1146
|
+
)
|
|
1147
|
+
if length <= 0:
|
|
1148
|
+
raise ValueError(f"loop_audio needs a length above zero, got {length} samples")
|
|
1149
|
+
if crossfade_ms < 0:
|
|
1150
|
+
raise ValueError("loop_audio 'crossfade_ms' cannot be negative")
|
|
1151
|
+
|
|
1152
|
+
window = min(int(crossfade_ms / 1000.0 * sample_rate), waveform.shape[1] // 2)
|
|
1153
|
+
bed = waveform
|
|
1154
|
+
laps = 1
|
|
1155
|
+
# Each lap after the first overlaps the one before it by the crossfade, so
|
|
1156
|
+
# a lap adds (source - window) samples rather than a whole source
|
|
1157
|
+
while bed.shape[1] < length:
|
|
1158
|
+
bed = crossfade_concat(
|
|
1159
|
+
[bed, waveform], sample_rate, window / sample_rate * 1000.0
|
|
1160
|
+
)
|
|
1161
|
+
laps += 1
|
|
1162
|
+
emit_log(
|
|
1163
|
+
f"loop_audio: {waveform.shape[1]} samples at {sample_rate}Hz looped "
|
|
1164
|
+
f"{laps}x to {length} samples ({length / sample_rate:.2f} s)",
|
|
1165
|
+
command="loop_audio",
|
|
1166
|
+
laps=laps,
|
|
1167
|
+
output_samples=length,
|
|
1168
|
+
output_seconds=round(length / sample_rate, 2),
|
|
1169
|
+
)
|
|
1170
|
+
return _as_track(bed[:, :length], sample_rate, "loop_audio")
|
|
1171
|
+
|
|
1172
|
+
|
|
1173
|
+
# Shots generated independently land at whatever level the model chose, and
|
|
1174
|
+
# joining two of them butts one loudness against another - the one seam
|
|
1175
|
+
# artifact no fade can hide, because it is not at the seam, it is either side
|
|
1176
|
+
# of it. These are the levels a matched join targets, and the spread at which
|
|
1177
|
+
# an unmatched one is worth warning about
|
|
1178
|
+
MATCH_MEASURES = ("peak", "rms")
|
|
1179
|
+
DEFAULT_MATCH_DBFS = {"peak": -1.0, "rms": -20.0}
|
|
1180
|
+
# Matching to an rms target can ask for a gain that would clip; the peak is
|
|
1181
|
+
# held here instead, which keeps a loud shot's relative level honest rather
|
|
1182
|
+
# than squaring off its transients
|
|
1183
|
+
MATCH_CEILING_DBFS = -0.5
|
|
1184
|
+
LEVEL_SPREAD_WARN_DB = 6.0
|
|
1185
|
+
# Mirrors result.py's NEAR_SILENT_WARN_DBFS: the same mean/rms level a job's
|
|
1186
|
+
# own near-silent check treats as having no real content. Gaining an input
|
|
1187
|
+
# already this quiet up to the target raises a noise floor rather than
|
|
1188
|
+
# leveling a performance, and #434 found a +29.9 dB case that only reached
|
|
1189
|
+
# the log, never job.warnings
|
|
1190
|
+
MATCH_NEAR_SILENT_DBFS = -40.0
|
|
1191
|
+
MATCH_LARGE_GAIN_WARN_DB = 20.0
|
|
1192
|
+
|
|
1193
|
+
|
|
1194
|
+
def level_dbfs(waveform, measure="peak"):
|
|
1195
|
+
"""A waveform's level in dBFS, measured as `peak` or `rms`.
|
|
1196
|
+
|
|
1197
|
+
`rms` is the same measurement `get_gallery_metadata` reports as
|
|
1198
|
+
`mean_dbfs`, so a matched join can be checked against what the gallery
|
|
1199
|
+
said about the shots going into it. A silent track has no level: None.
|
|
1200
|
+
"""
|
|
1201
|
+
if measure not in MATCH_MEASURES:
|
|
1202
|
+
raise ValueError(
|
|
1203
|
+
f"level measure must be one of {MATCH_MEASURES}, got '{measure}'"
|
|
1204
|
+
)
|
|
1205
|
+
if waveform is None or waveform.size == 0:
|
|
1206
|
+
return None
|
|
1207
|
+
if measure == "peak":
|
|
1208
|
+
value = float(numpy.abs(waveform).max())
|
|
1209
|
+
else:
|
|
1210
|
+
value = float(
|
|
1211
|
+
numpy.sqrt(numpy.mean(numpy.square(waveform, dtype=numpy.float64)))
|
|
1212
|
+
)
|
|
1213
|
+
if value <= 0.0:
|
|
1214
|
+
return None
|
|
1215
|
+
return 20.0 * numpy.log10(value)
|
|
1216
|
+
|
|
1217
|
+
|
|
1218
|
+
def match_levels(waveforms, measure, target_dbfs=None, command="concat_videos"):
|
|
1219
|
+
"""Scale each waveform so its level sits at one shared target.
|
|
1220
|
+
|
|
1221
|
+
Returns a new list in the same order and shape; a None entry (a video
|
|
1222
|
+
with no soundtrack) and a silent track pass through untouched, since
|
|
1223
|
+
neither has a level to move. A gain that would push the peak past
|
|
1224
|
+
MATCH_CEILING_DBFS is held there and said so in the log - the shot is
|
|
1225
|
+
then quieter than the target rather than clipped.
|
|
1226
|
+
"""
|
|
1227
|
+
if measure not in MATCH_MEASURES:
|
|
1228
|
+
raise ValueError(
|
|
1229
|
+
f"{command} 'match_levels' must be one of {MATCH_MEASURES}, got '{measure}'"
|
|
1230
|
+
)
|
|
1231
|
+
if target_dbfs is None:
|
|
1232
|
+
target_dbfs = DEFAULT_MATCH_DBFS[measure]
|
|
1233
|
+
if target_dbfs > 0:
|
|
1234
|
+
raise ValueError(
|
|
1235
|
+
f"{command} 'match_levels_dbfs' cannot be above full scale (0)"
|
|
1236
|
+
)
|
|
1237
|
+
|
|
1238
|
+
matched = []
|
|
1239
|
+
for index, waveform in enumerate(waveforms):
|
|
1240
|
+
level = level_dbfs(waveform, measure)
|
|
1241
|
+
if level is None:
|
|
1242
|
+
matched.append(waveform)
|
|
1243
|
+
continue
|
|
1244
|
+
target_gain_db = target_dbfs - level
|
|
1245
|
+
gain_db = target_gain_db
|
|
1246
|
+
peak = level_dbfs(waveform, "peak")
|
|
1247
|
+
held = False
|
|
1248
|
+
if peak is not None and peak + gain_db > MATCH_CEILING_DBFS:
|
|
1249
|
+
gain_db = MATCH_CEILING_DBFS - peak
|
|
1250
|
+
held = True
|
|
1251
|
+
shortfall_db = target_gain_db - gain_db
|
|
1252
|
+
# emit_warning rather than logger.warning: a clip-held shot stays
|
|
1253
|
+
# off the target and the residual spread is exactly the level
|
|
1254
|
+
# jump match_levels exists to remove (#214) - a caller reading
|
|
1255
|
+
# the job's warnings list is the one who can act on it (#82)
|
|
1256
|
+
emit_warning(
|
|
1257
|
+
f"{command}: video {index + 1} would clip at the {measure} target "
|
|
1258
|
+
f"({peak + target_gain_db:+.1f} dBFS peak) - held to "
|
|
1259
|
+
f"{MATCH_CEILING_DBFS} dBFS, {shortfall_db:.1f} dB short of target",
|
|
1260
|
+
kind="match_levels_held",
|
|
1261
|
+
command=command,
|
|
1262
|
+
index=index,
|
|
1263
|
+
measure_dbfs=round(level, 1),
|
|
1264
|
+
target_dbfs=target_dbfs,
|
|
1265
|
+
gain_db=round(gain_db, 1),
|
|
1266
|
+
shortfall_db=round(shortfall_db, 1),
|
|
1267
|
+
ceiling_dbfs=MATCH_CEILING_DBFS,
|
|
1268
|
+
)
|
|
1269
|
+
elif level <= MATCH_NEAR_SILENT_DBFS or gain_db >= MATCH_LARGE_GAIN_WARN_DB:
|
|
1270
|
+
# The other end of the range `held` covers (#434): an input this
|
|
1271
|
+
# quiet is noise floor, not a performance at a lower level, and
|
|
1272
|
+
# matching it up to the target passes that noise off as content -
|
|
1273
|
+
# a consumer reading job.warnings sees nothing was wrong
|
|
1274
|
+
emit_warning(
|
|
1275
|
+
f"{command}: video {index + 1} {measure} {level:.1f} dBFS is "
|
|
1276
|
+
f"near-silent - matched up to the target with a {gain_db:+.1f} dB "
|
|
1277
|
+
"gain, raising its noise floor rather than leveling content",
|
|
1278
|
+
kind="match_levels_near_silent",
|
|
1279
|
+
command=command,
|
|
1280
|
+
index=index,
|
|
1281
|
+
measure_dbfs=round(level, 1),
|
|
1282
|
+
target_dbfs=target_dbfs,
|
|
1283
|
+
gain_db=round(gain_db, 1),
|
|
1284
|
+
)
|
|
1285
|
+
emit_log(
|
|
1286
|
+
f"{command}: video {index + 1} {measure} {level:.1f} dBFS, "
|
|
1287
|
+
f"gain {gain_db:+.1f} dB{' (held)' if held else ''}",
|
|
1288
|
+
index=index,
|
|
1289
|
+
measure_dbfs=round(level, 1),
|
|
1290
|
+
gain_db=round(gain_db, 1),
|
|
1291
|
+
held=held,
|
|
1292
|
+
)
|
|
1293
|
+
matched.append((waveform * (10 ** (gain_db / 20.0))).astype(numpy.float32))
|
|
1294
|
+
return matched
|
|
1295
|
+
|
|
1296
|
+
|
|
1297
|
+
def warn_on_level_spread(waveforms, command="concat_videos", measure="rms"):
|
|
1298
|
+
"""Say something when shots about to be joined are levels apart.
|
|
1299
|
+
|
|
1300
|
+
Independently generated shots drift by 10 dB and more, and each one reads
|
|
1301
|
+
as fine on its own - it is only wrong relative to what it is cut against,
|
|
1302
|
+
and nothing else compares them.
|
|
1303
|
+
"""
|
|
1304
|
+
levels = [level for level in (level_dbfs(w, measure) for w in waveforms) if level]
|
|
1305
|
+
if len(levels) < 2:
|
|
1306
|
+
return None
|
|
1307
|
+
spread = max(levels) - min(levels)
|
|
1308
|
+
if spread >= LEVEL_SPREAD_WARN_DB:
|
|
1309
|
+
# emit_warning rather than logger.warning: this is a property of the
|
|
1310
|
+
# file the run is about to write, and the caller reading the job is
|
|
1311
|
+
# the one who can act on it (#82)
|
|
1312
|
+
emit_warning(
|
|
1313
|
+
f"{command}: the tracks being joined span {spread:.1f} dB "
|
|
1314
|
+
f"({measure} {min(levels):.1f} to {max(levels):.1f} dBFS) - "
|
|
1315
|
+
"audible as a level jump unless the difference is intended (a "
|
|
1316
|
+
"shot written silent against the score). If it is not, pass "
|
|
1317
|
+
"match_levels to even them out",
|
|
1318
|
+
kind="level_spread",
|
|
1319
|
+
command=command,
|
|
1320
|
+
spread_db=round(spread, 1),
|
|
1321
|
+
measure=measure,
|
|
1322
|
+
)
|
|
1323
|
+
return spread
|
|
1324
|
+
|
|
1325
|
+
|
|
1326
|
+
def _equal_power_ramps(window):
|
|
1327
|
+
"""Cosine/sine fade curves that sum to constant power across the window."""
|
|
1328
|
+
theta = numpy.linspace(0.0, numpy.pi / 2.0, window, endpoint=False)
|
|
1329
|
+
return numpy.cos(theta, dtype=numpy.float32), numpy.sin(theta, dtype=numpy.float32)
|
|
1330
|
+
|
|
1331
|
+
|
|
1332
|
+
def _declick_join(previous, following, sample_rate, fade_ms=None):
|
|
1333
|
+
"""Butt-join two waveforms with a fade on each side of the seam.
|
|
1334
|
+
|
|
1335
|
+
The default is the few milliseconds that keep a butt-join from clicking.
|
|
1336
|
+
A longer fade is a deliberate edit - the graceful hard cut you want when
|
|
1337
|
+
neither a crossfade nor a bleed applies.
|
|
1338
|
+
"""
|
|
1339
|
+
ramp = int((DECLICK_MS if fade_ms is None else fade_ms) / 1000.0 * sample_rate)
|
|
1340
|
+
ramp = min(ramp, previous.shape[1], following.shape[1])
|
|
1341
|
+
if ramp > 0:
|
|
1342
|
+
fade_out, fade_in = _equal_power_ramps(ramp)
|
|
1343
|
+
previous = previous.copy()
|
|
1344
|
+
following = following.copy()
|
|
1345
|
+
previous[:, -ramp:] *= fade_out # cos: 1 down to ~0
|
|
1346
|
+
following[:, :ramp] *= fade_in # sin: ~0 up to 1
|
|
1347
|
+
return numpy.concatenate([previous, following], axis=1)
|
|
1348
|
+
|
|
1349
|
+
|
|
1350
|
+
def _matched_channels(*waveforms):
|
|
1351
|
+
"""Tile mono up so every waveform has the same channel count."""
|
|
1352
|
+
channels = max(waveform.shape[0] for waveform in waveforms)
|
|
1353
|
+
return tuple(
|
|
1354
|
+
(
|
|
1355
|
+
numpy.tile(waveform, (channels, 1))
|
|
1356
|
+
if waveform.shape[0] == 1 and channels > 1
|
|
1357
|
+
else waveform
|
|
1358
|
+
)
|
|
1359
|
+
for waveform in waveforms
|
|
1360
|
+
)
|
|
1361
|
+
|
|
1362
|
+
|
|
1363
|
+
def fade_audio(audio, fade_in_ms=0, fade_out_ms=0, sample_rate=None):
|
|
1364
|
+
"""Task command: fade a track in from silence and out to it.
|
|
1365
|
+
|
|
1366
|
+
A slice cut out of the middle of a piece ends on whatever was sounding at
|
|
1367
|
+
the cut; a short fade turns that into an ending. The curve is the
|
|
1368
|
+
equal-power cosine the seam joins use, so a fade sounds like a fade and
|
|
1369
|
+
not a volume knob.
|
|
1370
|
+
|
|
1371
|
+
Args:
|
|
1372
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
1373
|
+
soundtrack is taken), a video generated with a
|
|
1374
|
+
soundtrack, or a waveform (which needs sample_rate alongside it)
|
|
1375
|
+
fade_in_ms: Length of the fade in, from the head of the track
|
|
1376
|
+
fade_out_ms: Length of the fade out, to the tail of the track
|
|
1377
|
+
sample_rate: Sample rate of a waveform passed directly
|
|
1378
|
+
|
|
1379
|
+
Returns:
|
|
1380
|
+
An AudioTrack holding the faded waveform and its rate
|
|
1381
|
+
"""
|
|
1382
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "fade_audio")
|
|
1383
|
+
if fade_in_ms < 0 or fade_out_ms < 0:
|
|
1384
|
+
raise ValueError("fade_audio fade lengths cannot be negative")
|
|
1385
|
+
faded = waveform.copy()
|
|
1386
|
+
length = faded.shape[1]
|
|
1387
|
+
|
|
1388
|
+
fade_in = min(int(round(fade_in_ms / 1000 * sample_rate)), length)
|
|
1389
|
+
if fade_in:
|
|
1390
|
+
faded[:, :fade_in] *= _fade_curve(fade_in)[::-1]
|
|
1391
|
+
fade_out = min(int(round(fade_out_ms / 1000 * sample_rate)), length)
|
|
1392
|
+
if fade_out:
|
|
1393
|
+
faded[:, length - fade_out :] *= _fade_curve(fade_out)
|
|
1394
|
+
return _as_track(faded, sample_rate, "fade_audio")
|
|
1395
|
+
|
|
1396
|
+
|
|
1397
|
+
def normalize_audio(audio, peak_dbfs=-1.0, target_lufs=None, sample_rate=None):
|
|
1398
|
+
"""Task command: scale a track so its loudest sample sits at a level.
|
|
1399
|
+
|
|
1400
|
+
Generated music comes out at whatever level the model happened to land
|
|
1401
|
+
on - quiet takes need lifting before they sit under a picture, and a
|
|
1402
|
+
hot one needs headroom before the encoder. Peak normalization changes
|
|
1403
|
+
nothing but the gain, so the dynamics survive.
|
|
1404
|
+
|
|
1405
|
+
Args:
|
|
1406
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
1407
|
+
soundtrack is taken), a video generated with a
|
|
1408
|
+
soundtrack, or a waveform (which needs sample_rate alongside it)
|
|
1409
|
+
peak_dbfs: The level the loudest sample is moved to, in dB below full
|
|
1410
|
+
scale. 0 is full scale; -1 leaves a little headroom. Still
|
|
1411
|
+
applies as a ceiling when target_lufs is also given
|
|
1412
|
+
target_lufs: Integrated loudness (BS.1770) to gain the track to, in
|
|
1413
|
+
LUFS. Peak alone says nothing about how loud a track sounds - a
|
|
1414
|
+
sparse voice-over and a dense score can share a peak and still
|
|
1415
|
+
sit tens of dB apart to the ear (#361). When given, the gain
|
|
1416
|
+
targets this loudness first; peak_dbfs still holds as a ceiling,
|
|
1417
|
+
and if reaching target_lufs would cross it the gain stops at the
|
|
1418
|
+
ceiling and a warning names the shortfall in LU. None (the
|
|
1419
|
+
default) leaves behavior exactly as peak-only
|
|
1420
|
+
sample_rate: Sample rate of a waveform passed directly
|
|
1421
|
+
|
|
1422
|
+
Returns:
|
|
1423
|
+
An AudioTrack holding the scaled waveform and its rate; a silent
|
|
1424
|
+
track is returned unchanged
|
|
1425
|
+
"""
|
|
1426
|
+
check_arguments("normalize_audio", sample_rate=sample_rate, target_lufs=target_lufs)
|
|
1427
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "normalize_audio")
|
|
1428
|
+
if peak_dbfs > 0:
|
|
1429
|
+
raise ValueError("normalize_audio 'peak_dbfs' cannot be above full scale (0)")
|
|
1430
|
+
peak = float(numpy.abs(waveform).max()) if waveform.size else 0.0
|
|
1431
|
+
if peak == 0.0:
|
|
1432
|
+
logger.warning("normalize_audio: the track is silent - left unchanged")
|
|
1433
|
+
return _as_track(waveform, sample_rate, "normalize_audio")
|
|
1434
|
+
|
|
1435
|
+
peak_db = 20 * numpy.log10(peak)
|
|
1436
|
+
measured_lufs = None
|
|
1437
|
+
if target_lufs is None:
|
|
1438
|
+
constraint = "peak_dbfs"
|
|
1439
|
+
gain_db = peak_dbfs - peak_db
|
|
1440
|
+
else:
|
|
1441
|
+
ceiling_gain_db = peak_dbfs - peak_db
|
|
1442
|
+
current_lufs = integrated_lufs(waveform.T, sample_rate)
|
|
1443
|
+
measured_lufs = current_lufs
|
|
1444
|
+
if current_lufs is None:
|
|
1445
|
+
emit_warning(
|
|
1446
|
+
f"normalize_audio: target_lufs={target_lufs} was given, but the "
|
|
1447
|
+
"track's loudness could not be measured (shorter than the 400 ms "
|
|
1448
|
+
"gating block, or silent throughout) - falling back to peak_dbfs "
|
|
1449
|
+
"alone.",
|
|
1450
|
+
kind="target_lufs_unmeasurable",
|
|
1451
|
+
command="normalize_audio",
|
|
1452
|
+
target_lufs=target_lufs,
|
|
1453
|
+
)
|
|
1454
|
+
gain_db = ceiling_gain_db
|
|
1455
|
+
constraint = "peak_ceiling"
|
|
1456
|
+
else:
|
|
1457
|
+
target_gain_db = target_lufs - current_lufs
|
|
1458
|
+
gain_db = min(target_gain_db, ceiling_gain_db)
|
|
1459
|
+
constraint = "peak_ceiling" if gain_db < target_gain_db else "target_lufs"
|
|
1460
|
+
if gain_db < target_gain_db:
|
|
1461
|
+
emit_warning(
|
|
1462
|
+
f"normalize_audio: target_lufs={target_lufs} would need "
|
|
1463
|
+
f"{target_gain_db:+.1f} dB of gain, but peak_dbfs={peak_dbfs} "
|
|
1464
|
+
f"caps it at {gain_db:+.1f} dB - "
|
|
1465
|
+
f"{target_gain_db - gain_db:.1f} LU short of the target.",
|
|
1466
|
+
kind="target_lufs_capped",
|
|
1467
|
+
command="normalize_audio",
|
|
1468
|
+
target_lufs=target_lufs,
|
|
1469
|
+
peak_dbfs=peak_dbfs,
|
|
1470
|
+
shortfall_lu=target_gain_db - gain_db,
|
|
1471
|
+
)
|
|
1472
|
+
gain = 10 ** (gain_db / 20)
|
|
1473
|
+
emit_log(
|
|
1474
|
+
f"normalize_audio: measured {peak_db:.1f} dBFS peak"
|
|
1475
|
+
+ ("" if measured_lufs is None else f", {measured_lufs:.1f} LUFS")
|
|
1476
|
+
+ f" -> gain {gain_db:+.1f} dB, set by {constraint}",
|
|
1477
|
+
command="normalize_audio",
|
|
1478
|
+
measured_peak_dbfs=round(peak_db, 1),
|
|
1479
|
+
measured_lufs=round(measured_lufs, 1) if measured_lufs is not None else None,
|
|
1480
|
+
gain_db=round(gain_db, 1),
|
|
1481
|
+
constraint=constraint,
|
|
1482
|
+
)
|
|
1483
|
+
return _as_track(
|
|
1484
|
+
(waveform * gain).astype(numpy.float32), sample_rate, "normalize_audio"
|
|
1485
|
+
)
|
|
1486
|
+
|
|
1487
|
+
|
|
1488
|
+
def _fade_curve(window):
|
|
1489
|
+
"""A cosine fall from full level to exact silence, both ends included -
|
|
1490
|
+
unlike the seam ramps, which stop short of the endpoint so two of them
|
|
1491
|
+
tile a crossfade without a doubled sample."""
|
|
1492
|
+
theta = numpy.linspace(0.0, numpy.pi / 2.0, window, endpoint=True)
|
|
1493
|
+
return numpy.cos(theta, dtype=numpy.float32)
|
|
1494
|
+
|
|
1495
|
+
|
|
1496
|
+
def _waveform_and_rate(audio, sample_rate, command):
|
|
1497
|
+
"""A command's audio argument as a (channels, samples) array with its rate.
|
|
1498
|
+
|
|
1499
|
+
A path loads with the file's own rate; a video generated with a soundtrack
|
|
1500
|
+
(an AudioVideo, or anything carrying `.audio`) contributes that track and
|
|
1501
|
+
its rate; a bare waveform needs the rate given. A given rate always wins -
|
|
1502
|
+
correct for a raw waveform, which carries none of its own, but for a named
|
|
1503
|
+
source (a file or a video) a rate that disagrees with the one it actually
|
|
1504
|
+
carries relabels the samples rather than converting them, changing speed
|
|
1505
|
+
and pitch with nothing saying so (#180) - so that case warns.
|
|
1506
|
+
"""
|
|
1507
|
+
if isinstance(audio, str):
|
|
1508
|
+
waveform, file_rate = load_audio(audio)
|
|
1509
|
+
if (
|
|
1510
|
+
sample_rate is not None
|
|
1511
|
+
and file_rate is not None
|
|
1512
|
+
and sample_rate != file_rate
|
|
1513
|
+
):
|
|
1514
|
+
_warn_on_rate_override(command, file_rate, sample_rate)
|
|
1515
|
+
return waveform, sample_rate if sample_rate is not None else file_rate
|
|
1516
|
+
if hasattr(audio, "audio"):
|
|
1517
|
+
if audio.audio is None:
|
|
1518
|
+
raise ValueError(
|
|
1519
|
+
f"{command} needs an audio track - the video it was given carries none"
|
|
1520
|
+
)
|
|
1521
|
+
if (
|
|
1522
|
+
sample_rate is not None
|
|
1523
|
+
and audio.sample_rate is not None
|
|
1524
|
+
and sample_rate != audio.sample_rate
|
|
1525
|
+
):
|
|
1526
|
+
_warn_on_rate_override(command, audio.sample_rate, sample_rate)
|
|
1527
|
+
rate = sample_rate if sample_rate is not None else audio.sample_rate
|
|
1528
|
+
if rate is None:
|
|
1529
|
+
raise ValueError(
|
|
1530
|
+
f"{command} needs 'sample_rate' - the video it was given does not "
|
|
1531
|
+
"carry one of its own"
|
|
1532
|
+
)
|
|
1533
|
+
return as_channels_samples(audio.audio), rate
|
|
1534
|
+
if sample_rate is None:
|
|
1535
|
+
raise ValueError(f"{command} needs 'sample_rate' with a raw waveform")
|
|
1536
|
+
return as_channels_samples(audio), sample_rate
|
|
1537
|
+
|
|
1538
|
+
|
|
1539
|
+
def _warn_on_rate_override(command, actual_rate, given_rate):
|
|
1540
|
+
"""Say when a given sample_rate relabels a named source's real rate.
|
|
1541
|
+
|
|
1542
|
+
'sample_rate' always overrides the rate a file or video carries - that is
|
|
1543
|
+
what lets a raw waveform (which has none of its own) be handed in at all -
|
|
1544
|
+
but for a named source it is easy to mistake for a conversion: a workflow
|
|
1545
|
+
reused one variable as both 'the rate a mix runs at' and 'the rate this
|
|
1546
|
+
file is at', and the mismatch reached nobody until the deliverable played
|
|
1547
|
+
at the wrong speed with `warnings: []` (#180). emit_warning rather than
|
|
1548
|
+
logger.warning for the reason every other run-time audio warning here is
|
|
1549
|
+
(#82, #108): a caller reading the job over the API or MCP sees the
|
|
1550
|
+
warnings list and nothing else.
|
|
1551
|
+
"""
|
|
1552
|
+
emit_warning(
|
|
1553
|
+
f"{command}: sample_rate={given_rate} was given, but the source "
|
|
1554
|
+
f"actually carries {actual_rate} Hz. The samples are being relabeled "
|
|
1555
|
+
f"at {given_rate} Hz, not resampled - this changes speed and pitch. "
|
|
1556
|
+
f"If you meant to convert the rate, use 'resample_audio' "
|
|
1557
|
+
f"(target_sample_rate={given_rate}) instead.",
|
|
1558
|
+
kind="rate_override_mismatch",
|
|
1559
|
+
command=command,
|
|
1560
|
+
file_rate=actual_rate,
|
|
1561
|
+
given_rate=given_rate,
|
|
1562
|
+
)
|
|
1563
|
+
|
|
1564
|
+
|
|
1565
|
+
COMPRESS_MODES = ("compress", "limit", "gate")
|
|
1566
|
+
|
|
1567
|
+
# A floor below which an envelope is treated as digital silence, so its dBFS
|
|
1568
|
+
# reading is a large negative number rather than -inf
|
|
1569
|
+
_ENVELOPE_FLOOR_DBFS = -120.0
|
|
1570
|
+
_ENVELOPE_FLOOR_LINEAR = 10.0 ** (_ENVELOPE_FLOOR_DBFS / 20.0)
|
|
1571
|
+
|
|
1572
|
+
|
|
1573
|
+
def compress_audio(
|
|
1574
|
+
audio,
|
|
1575
|
+
threshold_dbfs,
|
|
1576
|
+
ratio=4.0,
|
|
1577
|
+
attack_ms=10.0,
|
|
1578
|
+
release_ms=100.0,
|
|
1579
|
+
mode="compress",
|
|
1580
|
+
sample_rate=None,
|
|
1581
|
+
):
|
|
1582
|
+
"""Task command: shape a track's dynamics with an envelope-follower.
|
|
1583
|
+
|
|
1584
|
+
A compressor, a limiter and a gate are the same envelope-follower
|
|
1585
|
+
algorithm with different knob settings: a limiter is a ratio pushed
|
|
1586
|
+
toward infinity with a fast attack, and a gate is downward expansion
|
|
1587
|
+
below the threshold rather than compression above it - so one command
|
|
1588
|
+
covers all three through 'mode' rather than three near-duplicate ones.
|
|
1589
|
+
|
|
1590
|
+
Args:
|
|
1591
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
1592
|
+
soundtrack is taken), a video generated with a
|
|
1593
|
+
soundtrack, or a waveform (which needs sample_rate alongside it)
|
|
1594
|
+
threshold_dbfs: The level, in dB below full scale, above which
|
|
1595
|
+
'compress'/'limit' reduce gain, or below which 'gate' does
|
|
1596
|
+
ratio: How strongly gain is reduced past the threshold. Unused by
|
|
1597
|
+
'limit', which reduces enough to hold the signal at the
|
|
1598
|
+
threshold regardless
|
|
1599
|
+
attack_ms: How fast the envelope follows a rise in level
|
|
1600
|
+
release_ms: How fast the envelope follows a fall in level
|
|
1601
|
+
mode: 'compress' (downward compression above threshold), 'limit'
|
|
1602
|
+
(holds the signal at threshold), or 'gate' (downward expansion
|
|
1603
|
+
below threshold)
|
|
1604
|
+
sample_rate: Sample rate of a waveform passed directly
|
|
1605
|
+
|
|
1606
|
+
Returns:
|
|
1607
|
+
An AudioTrack holding the processed waveform and its rate
|
|
1608
|
+
"""
|
|
1609
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "compress_audio")
|
|
1610
|
+
check_arguments(
|
|
1611
|
+
"compress_audio",
|
|
1612
|
+
ratio=ratio,
|
|
1613
|
+
attack_ms=attack_ms,
|
|
1614
|
+
release_ms=release_ms,
|
|
1615
|
+
sample_rate=sample_rate,
|
|
1616
|
+
)
|
|
1617
|
+
if mode not in COMPRESS_MODES:
|
|
1618
|
+
raise ValueError(
|
|
1619
|
+
f"compress_audio mode must be one of {COMPRESS_MODES}, got {mode!r}"
|
|
1620
|
+
)
|
|
1621
|
+
if threshold_dbfs > 0:
|
|
1622
|
+
raise ValueError(
|
|
1623
|
+
"compress_audio 'threshold_dbfs' cannot be above full scale (0)"
|
|
1624
|
+
)
|
|
1625
|
+
if waveform.size == 0:
|
|
1626
|
+
return _as_track(waveform, sample_rate, "compress_audio")
|
|
1627
|
+
|
|
1628
|
+
envelope = _follow_envelope(waveform, sample_rate, attack_ms, release_ms)
|
|
1629
|
+
envelope_dbfs = 20.0 * numpy.log10(numpy.maximum(envelope, _ENVELOPE_FLOOR_LINEAR))
|
|
1630
|
+
|
|
1631
|
+
if mode == "gate":
|
|
1632
|
+
past_threshold = numpy.maximum(0.0, threshold_dbfs - envelope_dbfs)
|
|
1633
|
+
else:
|
|
1634
|
+
past_threshold = numpy.maximum(0.0, envelope_dbfs - threshold_dbfs)
|
|
1635
|
+
|
|
1636
|
+
if mode == "limit":
|
|
1637
|
+
reduction_db = past_threshold
|
|
1638
|
+
else:
|
|
1639
|
+
reduction_db = past_threshold * (1.0 - 1.0 / ratio)
|
|
1640
|
+
|
|
1641
|
+
gain = (10.0 ** (-reduction_db / 20.0)).astype(numpy.float32)
|
|
1642
|
+
processed = (waveform * gain[numpy.newaxis, :]).astype(numpy.float32)
|
|
1643
|
+
return _as_track(processed, sample_rate, "compress_audio")
|
|
1644
|
+
|
|
1645
|
+
|
|
1646
|
+
def _follow_envelope(waveform, sample_rate, attack_ms, release_ms):
|
|
1647
|
+
"""A linked (all-channels) peak envelope, smoothed by separate attack and
|
|
1648
|
+
release time constants - the same detector a hardware compressor uses,
|
|
1649
|
+
tracking the loudest channel so a stereo image does not shift."""
|
|
1650
|
+
rectified = numpy.abs(waveform).max(axis=0)
|
|
1651
|
+
attack_coef = _time_constant_coef(attack_ms, sample_rate)
|
|
1652
|
+
release_coef = _time_constant_coef(release_ms, sample_rate)
|
|
1653
|
+
# The branch on the running level is what makes this a loop rather than a
|
|
1654
|
+
# filter, but the per-sample numpy indexing was the expensive half of it:
|
|
1655
|
+
# a 3-minute track is ~8M samples, and this runs on the single FIFO
|
|
1656
|
+
# worker. tolist() hands the loop plain Python floats, which is the same
|
|
1657
|
+
# arithmetic on the same values, several times faster
|
|
1658
|
+
samples = rectified.tolist()
|
|
1659
|
+
envelope = []
|
|
1660
|
+
level = 0.0
|
|
1661
|
+
for sample in samples:
|
|
1662
|
+
coef = attack_coef if sample > level else release_coef
|
|
1663
|
+
level = coef * level + (1.0 - coef) * sample
|
|
1664
|
+
envelope.append(level)
|
|
1665
|
+
return numpy.asarray(envelope, dtype=rectified.dtype)
|
|
1666
|
+
|
|
1667
|
+
|
|
1668
|
+
def _time_constant_coef(time_ms, sample_rate):
|
|
1669
|
+
"""The per-sample smoothing coefficient for an exponential time constant.
|
|
1670
|
+
0 ms means the envelope follows instantly, with no smoothing at all."""
|
|
1671
|
+
if time_ms <= 0:
|
|
1672
|
+
return 0.0
|
|
1673
|
+
return float(numpy.exp(-1.0 / (time_ms / 1000.0 * sample_rate)))
|
|
1674
|
+
|
|
1675
|
+
|
|
1676
|
+
FILTER_KINDS = ("lowpass", "highpass", "bandpass", "notch")
|
|
1677
|
+
|
|
1678
|
+
|
|
1679
|
+
def filter_audio(audio, cutoff_hz, kind="lowpass", q=0.707, sample_rate=None):
|
|
1680
|
+
"""Task command: run a track through a single biquad filter stage.
|
|
1681
|
+
|
|
1682
|
+
Args:
|
|
1683
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
1684
|
+
soundtrack is taken), a video generated with a
|
|
1685
|
+
soundtrack, or a waveform (which needs sample_rate alongside it)
|
|
1686
|
+
cutoff_hz: The filter's corner (lowpass/highpass) or center
|
|
1687
|
+
(bandpass/notch) frequency
|
|
1688
|
+
kind: 'lowpass', 'highpass', 'bandpass', or 'notch'
|
|
1689
|
+
q: Resonance/bandwidth of the filter. Higher narrows a bandpass or
|
|
1690
|
+
notch, and peaks the corner of a lowpass or highpass
|
|
1691
|
+
sample_rate: Sample rate of a waveform passed directly
|
|
1692
|
+
|
|
1693
|
+
Returns:
|
|
1694
|
+
An AudioTrack holding the filtered waveform and its rate
|
|
1695
|
+
"""
|
|
1696
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "filter_audio")
|
|
1697
|
+
check_arguments("filter_audio", cutoff_hz=cutoff_hz, sample_rate=sample_rate)
|
|
1698
|
+
if kind not in FILTER_KINDS:
|
|
1699
|
+
raise ValueError(
|
|
1700
|
+
f"filter_audio kind must be one of {FILTER_KINDS}, got {kind!r}"
|
|
1701
|
+
)
|
|
1702
|
+
if q <= 0:
|
|
1703
|
+
raise ValueError("filter_audio 'q' must be above zero")
|
|
1704
|
+
nyquist = sample_rate / 2.0
|
|
1705
|
+
if cutoff_hz >= nyquist:
|
|
1706
|
+
raise ValueError(
|
|
1707
|
+
f"filter_audio 'cutoff_hz' ({cutoff_hz}) must be below the "
|
|
1708
|
+
f"Nyquist frequency ({nyquist}) for sample_rate {sample_rate}"
|
|
1709
|
+
)
|
|
1710
|
+
if waveform.size == 0:
|
|
1711
|
+
return _as_track(waveform, sample_rate, "filter_audio")
|
|
1712
|
+
|
|
1713
|
+
b, a = _biquad_coefficients(kind, cutoff_hz, q, sample_rate)
|
|
1714
|
+
filtered = numpy.stack(
|
|
1715
|
+
[_apply_biquad(channel, b, a) for channel in waveform]
|
|
1716
|
+
).astype(numpy.float32)
|
|
1717
|
+
return _as_track(filtered, sample_rate, "filter_audio")
|
|
1718
|
+
|
|
1719
|
+
|
|
1720
|
+
def _biquad_coefficients(kind, cutoff_hz, q, sample_rate):
|
|
1721
|
+
"""RBJ Audio EQ Cookbook coefficients for a single biquad stage,
|
|
1722
|
+
normalized so a0 is 1."""
|
|
1723
|
+
w0 = 2.0 * numpy.pi * cutoff_hz / sample_rate
|
|
1724
|
+
cos_w0 = numpy.cos(w0)
|
|
1725
|
+
sin_w0 = numpy.sin(w0)
|
|
1726
|
+
alpha = sin_w0 / (2.0 * q)
|
|
1727
|
+
|
|
1728
|
+
if kind == "lowpass":
|
|
1729
|
+
b0 = (1.0 - cos_w0) / 2.0
|
|
1730
|
+
b1 = 1.0 - cos_w0
|
|
1731
|
+
b2 = (1.0 - cos_w0) / 2.0
|
|
1732
|
+
elif kind == "highpass":
|
|
1733
|
+
b0 = (1.0 + cos_w0) / 2.0
|
|
1734
|
+
b1 = -(1.0 + cos_w0)
|
|
1735
|
+
b2 = (1.0 + cos_w0) / 2.0
|
|
1736
|
+
elif kind == "bandpass":
|
|
1737
|
+
b0 = alpha
|
|
1738
|
+
b1 = 0.0
|
|
1739
|
+
b2 = -alpha
|
|
1740
|
+
else: # notch
|
|
1741
|
+
b0 = 1.0
|
|
1742
|
+
b1 = -2.0 * cos_w0
|
|
1743
|
+
b2 = 1.0
|
|
1744
|
+
a0 = 1.0 + alpha
|
|
1745
|
+
a1 = -2.0 * cos_w0
|
|
1746
|
+
a2 = 1.0 - alpha
|
|
1747
|
+
return (
|
|
1748
|
+
numpy.array([b0, b1, b2], dtype=numpy.float64) / a0,
|
|
1749
|
+
numpy.array([a1, a2], dtype=numpy.float64) / a0,
|
|
1750
|
+
)
|
|
1751
|
+
|
|
1752
|
+
|
|
1753
|
+
def _apply_biquad(channel, b, a):
|
|
1754
|
+
"""One second-order section, run over a channel.
|
|
1755
|
+
|
|
1756
|
+
The feedback cannot be vectorized away, but it does not have to be run in
|
|
1757
|
+
Python either: scipy's lfilter is this exact recursion in C, and scipy is
|
|
1758
|
+
already in every install (controlnet-aux brings it). The Python Direct
|
|
1759
|
+
Form I below is the fallback for an environment without it - same
|
|
1760
|
+
recursion, same zero initial conditions, ~50x slower on a full track.
|
|
1761
|
+
"""
|
|
1762
|
+
b0, b1, b2 = b
|
|
1763
|
+
a1, a2 = a
|
|
1764
|
+
try:
|
|
1765
|
+
from scipy.signal import lfilter
|
|
1766
|
+
except ImportError:
|
|
1767
|
+
pass
|
|
1768
|
+
else:
|
|
1769
|
+
return lfilter(
|
|
1770
|
+
numpy.array([b0, b1, b2], dtype=numpy.float64),
|
|
1771
|
+
numpy.array([1.0, a1, a2], dtype=numpy.float64),
|
|
1772
|
+
numpy.asarray(channel, dtype=numpy.float64),
|
|
1773
|
+
)
|
|
1774
|
+
|
|
1775
|
+
out = numpy.empty_like(channel, dtype=numpy.float64)
|
|
1776
|
+
x1 = x2 = y1 = y2 = 0.0
|
|
1777
|
+
for i in range(channel.shape[0]):
|
|
1778
|
+
x0 = float(channel[i])
|
|
1779
|
+
y0 = b0 * x0 + b1 * x1 + b2 * x2 - a1 * y1 - a2 * y2
|
|
1780
|
+
out[i] = y0
|
|
1781
|
+
x2, x1 = x1, x0
|
|
1782
|
+
y2, y1 = y1, y0
|
|
1783
|
+
return out
|
|
1784
|
+
|
|
1785
|
+
|
|
1786
|
+
_SPECTRAL_BANDS = {
|
|
1787
|
+
"low_dbfs": (20.0, 250.0),
|
|
1788
|
+
"mid_dbfs": (250.0, 4000.0),
|
|
1789
|
+
"high_dbfs": (4000.0, 20000.0),
|
|
1790
|
+
}
|
|
1791
|
+
|
|
1792
|
+
|
|
1793
|
+
def analyze_audio(audio, sample_rate=None):
|
|
1794
|
+
"""Task command: measure a track without changing it.
|
|
1795
|
+
|
|
1796
|
+
Read-only: the waveform passes through unmodified, and what comes back
|
|
1797
|
+
is diagnostics rather than an AudioTrack, since there is no processed
|
|
1798
|
+
track to hand a later step. Meant to feed a decision earlier in a
|
|
1799
|
+
workflow (whether 'compress_audio' or 'filter_audio' is needed, and
|
|
1800
|
+
with what settings) rather than to sit in the middle of a chain.
|
|
1801
|
+
|
|
1802
|
+
Args:
|
|
1803
|
+
audio: Path or URL of an audio file (or of a video file, whose
|
|
1804
|
+
soundtrack is taken), a video generated with a
|
|
1805
|
+
soundtrack, or a waveform (which needs sample_rate alongside it)
|
|
1806
|
+
sample_rate: Sample rate of a waveform passed directly
|
|
1807
|
+
|
|
1808
|
+
Returns:
|
|
1809
|
+
A dict: peak_dbfs, rms_dbfs, crest_factor_db (peak minus rms), and
|
|
1810
|
+
a rough low_dbfs/mid_dbfs/high_dbfs spectral-balance reading whose
|
|
1811
|
+
three bands are shares of the same power that gives rms_dbfs, so
|
|
1812
|
+
they sit on that scale rather than tens of dB under it. Any value
|
|
1813
|
+
is None where a silent track leaves it undefined.
|
|
1814
|
+
"""
|
|
1815
|
+
waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "analyze_audio")
|
|
1816
|
+
check_arguments("analyze_audio", sample_rate=sample_rate)
|
|
1817
|
+
|
|
1818
|
+
peak_dbfs = level_dbfs(waveform, measure="peak")
|
|
1819
|
+
rms_dbfs = level_dbfs(waveform, measure="rms")
|
|
1820
|
+
crest_factor_db = (
|
|
1821
|
+
peak_dbfs - rms_dbfs if peak_dbfs is not None and rms_dbfs is not None else None
|
|
1822
|
+
)
|
|
1823
|
+
bands = _spectral_balance(waveform, sample_rate)
|
|
1824
|
+
return {
|
|
1825
|
+
"peak_dbfs": peak_dbfs,
|
|
1826
|
+
"rms_dbfs": rms_dbfs,
|
|
1827
|
+
"crest_factor_db": crest_factor_db,
|
|
1828
|
+
**bands,
|
|
1829
|
+
}
|
|
1830
|
+
|
|
1831
|
+
|
|
1832
|
+
def _spectral_balance(waveform, sample_rate):
|
|
1833
|
+
"""A rough low/mid/high energy reading in dBFS, from one FFT of the
|
|
1834
|
+
channel-averaged track - not a spectrogram, just enough to say whether
|
|
1835
|
+
a track leans bright or boomy.
|
|
1836
|
+
|
|
1837
|
+
Each band's power is a share of the same Parseval sum that gives
|
|
1838
|
+
rms_dbfs (mean(x**2)): a one-sided rfft bin's power is doubled to
|
|
1839
|
+
account for its mirrored negative-frequency twin, except the DC and
|
|
1840
|
+
(for even n) Nyquist bins, which have no twin. Summed over the full
|
|
1841
|
+
spectrum this equals mean(x**2) exactly, so a *_dbfs band sits on the
|
|
1842
|
+
same scale as rms_dbfs rather than ~40 dB under it (#211)."""
|
|
1843
|
+
if waveform.size == 0:
|
|
1844
|
+
return {name: None for name in _SPECTRAL_BANDS}
|
|
1845
|
+
mono = waveform.mean(axis=0)
|
|
1846
|
+
n = mono.shape[0]
|
|
1847
|
+
spectrum = numpy.fft.rfft(mono)
|
|
1848
|
+
power = numpy.square(numpy.abs(spectrum), dtype=numpy.float64) / (n * n)
|
|
1849
|
+
if n % 2 == 0:
|
|
1850
|
+
power[1:-1] *= 2.0
|
|
1851
|
+
else:
|
|
1852
|
+
power[1:] *= 2.0
|
|
1853
|
+
freqs = numpy.fft.rfftfreq(n, d=1.0 / sample_rate)
|
|
1854
|
+
result = {}
|
|
1855
|
+
for name, (low, high) in _SPECTRAL_BANDS.items():
|
|
1856
|
+
band = power[(freqs >= low) & (freqs < min(high, sample_rate / 2.0))]
|
|
1857
|
+
if band.size == 0:
|
|
1858
|
+
result[name] = None
|
|
1859
|
+
continue
|
|
1860
|
+
energy = float(numpy.sum(band))
|
|
1861
|
+
result[name] = 10.0 * numpy.log10(energy) if energy > 0.0 else None
|
|
1862
|
+
return result
|