diffusers-workflow 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
  2. diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
  3. diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
  4. diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
  5. diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
  6. diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
  7. dw/__init__.py +440 -0
  8. dw/adapter_compatibility.py +226 -0
  9. dw/arguments.py +1231 -0
  10. dw/assessment_rules.py +159 -0
  11. dw/assets.py +130 -0
  12. dw/cache_blocks.json +16 -0
  13. dw/cache_blocks.py +146 -0
  14. dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
  15. dw/content_types.py +150 -0
  16. dw/dissolve_frame_errors.py +121 -0
  17. dw/docs/ACCELERATION.md +352 -0
  18. dw/docs/AGENT_LOOP.md +95 -0
  19. dw/docs/DEPENDENCIES.md +91 -0
  20. dw/docs/IP_ADAPTER.md +109 -0
  21. dw/docs/LORAS.md +131 -0
  22. dw/docs/MCP.md +517 -0
  23. dw/docs/PROMPT_WEIGHTING.md +78 -0
  24. dw/docs/QUANTIZATION.md +230 -0
  25. dw/docs/RECIPES_24GB.md +201 -0
  26. dw/docs/RELEASING.md +195 -0
  27. dw/docs/REMOTE.md +140 -0
  28. dw/docs/REPL_COMMANDS.md +121 -0
  29. dw/docs/REPL_WORKER_GUIDE.md +51 -0
  30. dw/docs/SECURITY.md +272 -0
  31. dw/docs/SECURITY_QUICKREF.md +112 -0
  32. dw/docs/SERVER.md +679 -0
  33. dw/docs/TASKS.md +1741 -0
  34. dw/docs/TESTING.md +71 -0
  35. dw/docs/WORKFLOW_GUIDE.md +2038 -0
  36. dw/docs/WORKSPACES.md +316 -0
  37. dw/download_watch.py +335 -0
  38. dw/elision.py +306 -0
  39. dw/events.py +275 -0
  40. dw/for_each.py +409 -0
  41. dw/host_memory.py +258 -0
  42. dw/host_memory_projection.py +230 -0
  43. dw/hub_cache.py +432 -0
  44. dw/introspection.py +1228 -0
  45. dw/kernel_availability.py +208 -0
  46. dw/locations.py +599 -0
  47. dw/log_setup.py +45 -0
  48. dw/loudness.py +82 -0
  49. dw/media_audio.py +217 -0
  50. dw/media_frames.py +367 -0
  51. dw/media_info.py +297 -0
  52. dw/pipeline_processors/chain.py +821 -0
  53. dw/pipeline_processors/config_objects.py +237 -0
  54. dw/pipeline_processors/pipeline.py +2297 -0
  55. dw/pipeline_processors/remote.py +46 -0
  56. dw/plan.py +920 -0
  57. dw/previous_results.py +411 -0
  58. dw/probe_paths.py +59 -0
  59. dw/prompt_schema.json +48 -0
  60. dw/prompt_weighting.py +378 -0
  61. dw/prompts.py +159 -0
  62. dw/realize.py +250 -0
  63. dw/reference_limits.py +215 -0
  64. dw/reference_names.py +125 -0
  65. dw/repl.py +338 -0
  66. dw/repl_commands.py +836 -0
  67. dw/repl_worker.py +159 -0
  68. dw/result.py +1720 -0
  69. dw/result_fps.py +82 -0
  70. dw/run.py +162 -0
  71. dw/runs.py +768 -0
  72. dw/scalar_result_validation.py +97 -0
  73. dw/schema.py +283 -0
  74. dw/security.py +1038 -0
  75. dw/select_validation.py +115 -0
  76. dw/serve.py +277 -0
  77. dw/server/__init__.py +2 -0
  78. dw/server/app.py +4586 -0
  79. dw/server/assess.py +132 -0
  80. dw/server/catalog_shape.py +487 -0
  81. dw/server/enhancers.py +129 -0
  82. dw/server/exports.py +480 -0
  83. dw/server/guides.py +257 -0
  84. dw/server/jobs.py +1561 -0
  85. dw/server/mcp_mount.py +95 -0
  86. dw/server/netinfo.py +124 -0
  87. dw/server/observed_cost.py +379 -0
  88. dw/server/sysinfo.py +71 -0
  89. dw/server/ui/assets/abap-08VXUWAP.js +1 -0
  90. dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
  91. dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
  92. dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
  93. dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
  94. dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
  95. dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
  96. dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
  97. dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
  98. dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
  99. dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
  100. dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
  101. dw/server/ui/assets/css-DIMkf-bt.js +3 -0
  102. dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
  103. dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
  104. dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
  105. dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
  106. dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
  107. dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
  108. dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
  109. dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
  110. dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
  111. dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
  112. dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
  113. dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
  114. dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
  115. dw/server/ui/assets/go-C-y9NEjX.js +1 -0
  116. dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
  117. dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
  118. dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
  119. dw/server/ui/assets/html-PhsdjHSr.js +1 -0
  120. dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
  121. dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
  122. dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
  123. dw/server/ui/assets/index-DgrYhQd9.js +43 -0
  124. dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
  125. dw/server/ui/assets/java-BEtHBSE6.js +1 -0
  126. dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
  127. dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
  128. dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
  129. dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
  130. dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
  131. dw/server/ui/assets/less-B9JPFI3C.js +2 -0
  132. dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
  133. dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
  134. dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
  135. dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
  136. dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
  137. dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
  138. dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
  139. dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
  140. dw/server/ui/assets/monaco--ixms01u.css +1 -0
  141. dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
  142. dw/server/ui/assets/msdax-DauUninz.js +1 -0
  143. dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
  144. dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
  145. dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
  146. dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
  147. dw/server/ui/assets/perl-oz_6vUea.js +1 -0
  148. dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
  149. dw/server/ui/assets/php-nr791fC2.js +1 -0
  150. dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
  151. dw/server/ui/assets/postiats-43DmfD33.js +1 -0
  152. dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
  153. dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
  154. dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
  155. dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
  156. dw/server/ui/assets/python-Bcn70HdC.js +1 -0
  157. dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
  158. dw/server/ui/assets/r-BwWrilGY.js +1 -0
  159. dw/server/ui/assets/razor-D1HmNnby.js +1 -0
  160. dw/server/ui/assets/redis-ClamHrr6.js +1 -0
  161. dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
  162. dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
  163. dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
  164. dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
  165. dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
  166. dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
  167. dw/server/ui/assets/scheme-BeGwcela.js +1 -0
  168. dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
  169. dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
  170. dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
  171. dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
  172. dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
  173. dw/server/ui/assets/sql-NEE52Syq.js +1 -0
  174. dw/server/ui/assets/st-DbInun42.js +1 -0
  175. dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
  176. dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
  177. dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
  178. dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
  179. dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
  180. dw/server/ui/assets/twig-De2hgUGE.js +1 -0
  181. dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
  182. dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
  183. dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
  184. dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
  185. dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
  186. dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
  187. dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
  188. dw/server/ui/index.html +17 -0
  189. dw/server/updater.py +192 -0
  190. dw/settings.py +98 -0
  191. dw/shot_span_preflight.py +116 -0
  192. dw/shots.py +359 -0
  193. dw/slice_preflight.py +148 -0
  194. dw/step.py +187 -0
  195. dw/step_cache.py +442 -0
  196. dw/subfolders.py +107 -0
  197. dw/task_domains.py +307 -0
  198. dw/tasks/assess.py +826 -0
  199. dw/tasks/audio_transcription.py +88 -0
  200. dw/tasks/audio_utils.py +1862 -0
  201. dw/tasks/background_remover.py +43 -0
  202. dw/tasks/borders.py +113 -0
  203. dw/tasks/compose_text.py +74 -0
  204. dw/tasks/concat_videos.py +300 -0
  205. dw/tasks/depth_estimator.py +54 -0
  206. dw/tasks/diffusion_upscale.py +109 -0
  207. dw/tasks/dissolve_videos.py +342 -0
  208. dw/tasks/format_messages.py +24 -0
  209. dw/tasks/gather.py +173 -0
  210. dw/tasks/grade.py +97 -0
  211. dw/tasks/image_to_text.py +43 -0
  212. dw/tasks/image_utils.py +764 -0
  213. dw/tasks/interpolate_frames.py +252 -0
  214. dw/tasks/judge.py +68 -0
  215. dw/tasks/model_cache.py +55 -0
  216. dw/tasks/pair_audio.py +268 -0
  217. dw/tasks/qr_code.py +19 -0
  218. dw/tasks/restore_faces.py +175 -0
  219. dw/tasks/rife_model.py +192 -0
  220. dw/tasks/segment.py +121 -0
  221. dw/tasks/select.py +111 -0
  222. dw/tasks/speech_generation.py +228 -0
  223. dw/tasks/stabilize.py +129 -0
  224. dw/tasks/task.py +920 -0
  225. dw/tasks/tensor_image.py +57 -0
  226. dw/tasks/text_generation.py +169 -0
  227. dw/tasks/text_sections.py +80 -0
  228. dw/tasks/upscale.py +203 -0
  229. dw/tasks/video_utils.py +624 -0
  230. dw/tasks/zoe_depth.py +71 -0
  231. dw/teacache.py +381 -0
  232. dw/teacache_models.json +99 -0
  233. dw/test.py +29 -0
  234. dw/type_helpers.py +231 -0
  235. dw/validate.py +68 -0
  236. dw/variable_constraints.py +444 -0
  237. dw/variables.py +443 -0
  238. dw/video_extensions.py +141 -0
  239. dw/vram_estimate.py +116 -0
  240. dw/worker.py +764 -0
  241. dw/workflow.py +2007 -0
  242. dw/workflow_schema.json +1346 -0
  243. dw/workflow_sources.py +383 -0
  244. dw/workflows/h3_context_ir.json +57 -0
  245. dw/workflows/test.json +31 -0
  246. dw/workspace.py +730 -0
  247. dw_mcp/__init__.py +6 -0
  248. dw_mcp/__main__.py +133 -0
  249. dw_mcp/assets.py +336 -0
  250. dw_mcp/authoring.py +114 -0
  251. dw_mcp/catalog.py +360 -0
  252. dw_mcp/client.py +486 -0
  253. dw_mcp/diagnose.py +371 -0
  254. dw_mcp/exports.py +84 -0
  255. dw_mcp/guides.py +35 -0
  256. dw_mcp/media.py +638 -0
  257. dw_mcp/models.py +97 -0
  258. dw_mcp/prompts.py +104 -0
  259. dw_mcp/server.py +1343 -0
  260. dw_mcp/workspaces.py +212 -0
@@ -0,0 +1,1862 @@
1
+ """Waveform utilities for audio tasks and segment-chained video generation.
2
+
3
+ Waveforms are handled as (channels, samples) float32 numpy arrays throughout -
4
+ as_channels_samples normalizes the shapes pipelines and files actually produce
5
+ into that layout.
6
+ """
7
+
8
+ import io
9
+ import os
10
+ import logging
11
+ from fractions import Fraction
12
+
13
+ import numpy
14
+ import soundfile
15
+ import torch
16
+
17
+ from ..events import emit_log, emit_warning
18
+ from ..loudness import integrated_lufs
19
+ from ..task_domains import as_number, check_arguments
20
+ from ..security import (
21
+ validate_file_extension,
22
+ ALLOWED_AUDIO_EXTENSIONS,
23
+ )
24
+
25
+ logger = logging.getLogger("dw")
26
+
27
+ # A few milliseconds of fade applied on each side of a butt-joined seam so the
28
+ # discontinuity does not click
29
+ DECLICK_MS = 3.0
30
+
31
+ # Padding shorter than this at the end of a slice is the rounding that
32
+ # frame-aligned slicing produces, not a slice that overran its source
33
+ SLICE_PAD_WARN_MS = 10.0
34
+
35
+ # A dropped tail is only the "almost reached the end" signature this warning
36
+ # exists for when it is both short next to the slice and short in absolute
37
+ # terms - a deliberate excerpt out of a long recording drops most of the
38
+ # source and should not warn
39
+ SLICE_TRIM_WARN_SECONDS = 10.0
40
+ SLICE_TRIM_WARN_FRACTION = 0.05
41
+
42
+
43
+ def as_channels_samples(audio):
44
+ """Normalize a waveform to a (channels, samples) float32 numpy array.
45
+
46
+ Accepts torch tensors or numpy arrays shaped (samples,), (channels, samples),
47
+ (samples, channels), or a one-item batch (1, channels, samples). Channel
48
+ position is decided the way normalize_audio in result.py decides it: there
49
+ are always more samples than channels.
50
+ """
51
+ if torch.is_tensor(audio):
52
+ audio = audio.detach().cpu().float().numpy()
53
+ audio = numpy.asarray(audio, dtype=numpy.float32)
54
+
55
+ if audio.ndim == 1:
56
+ return audio[numpy.newaxis, :]
57
+
58
+ if audio.ndim == 3:
59
+ if audio.shape[0] != 1:
60
+ raise ValueError(f"Cannot normalize a waveform batch of {audio.shape[0]}")
61
+ audio = audio[0]
62
+
63
+ if audio.ndim != 2:
64
+ raise ValueError(f"A waveform must have 1-3 dimensions, got {audio.ndim}")
65
+
66
+ if audio.shape[0] > audio.shape[1]: # (samples, channels) -> transpose
67
+ audio = audio.T
68
+
69
+ return numpy.ascontiguousarray(audio)
70
+
71
+
72
+ def frames_to_samples(frames, fps, sample_rate):
73
+ """The number of audio samples spanning a run of video frames."""
74
+ return int(round(frames / fps * sample_rate))
75
+
76
+
77
+ def fit_audio_to_frames(audio, sample_rate, total_frames, fps, command):
78
+ """Pad a joined track that falls short of its frame grid, and warn when
79
+ the gap is a frame or more.
80
+
81
+ concat_videos and dissolve_videos each build their joined track by
82
+ measuring and concatenating/crossfading the actual input waveforms, with
83
+ nothing reconciling a shortfall against total_frames - so an input that
84
+ is itself short of its own frame grid (#435 traced this to an
85
+ ltx2/keyframes clip short of its 121-frame bucket) propagates its
86
+ shortfall into the join, and the shortfall compounds across further
87
+ joins that each take the previous join's output as an input. The same
88
+ remedy #428 gave pair_audio's 'fit: video' for a shortfall, applied here
89
+ at the one place every join's audio passes through before its shot map
90
+ is measured.
91
+
92
+ A track *longer* than its frame grid is left alone: concat_videos has
93
+ measured such an overrun deliberately since #378 (its own shot keeps the
94
+ samples it actually took, not a count derived from frame/fps
95
+ arithmetic), and trimming it here would silently reverse that contract
96
+ for the whole joined track.
97
+ """
98
+ if audio is None or not total_frames or not fps or not sample_rate:
99
+ return audio
100
+
101
+ wanted = frames_to_samples(total_frames, fps, sample_rate)
102
+ have = audio.shape[1]
103
+ if have >= wanted:
104
+ return audio
105
+
106
+ audio_seconds = have / float(sample_rate)
107
+ video_seconds = total_frames / float(fps)
108
+ pad_samples = wanted - have
109
+ audio = numpy.pad(audio, ((0, 0), (0, pad_samples)))
110
+ if pad_samples < sample_rate / fps:
111
+ # Less than one frame is rounding between the rate and the frame
112
+ # grid, the gap pair_audio's own unfitted check leaves unwarned
113
+ # (LENGTH_WARN_MS): padded, and logged, but not a warning on every
114
+ # stock join (#454)
115
+ emit_log(
116
+ f"{command}: padded the joined track with {pad_samples} sample"
117
+ f"{'s' if pad_samples != 1 else ''} of silence to the frame grid",
118
+ command=command,
119
+ pad_samples=pad_samples,
120
+ )
121
+ return audio
122
+ emit_warning(
123
+ f"{command}: the joined track is {audio_seconds:.3f} s and the "
124
+ f"joined video is {video_seconds:.3f} s ({total_frames} frames at "
125
+ f"{fps:g} fps) - padded the track with {pad_samples} sample"
126
+ f"{'s' if pad_samples != 1 else ''} of silence to reach the frame "
127
+ "grid, so the shortfall does not carry into a later join.",
128
+ kind="joined_audio_padded_to_frames",
129
+ command=command,
130
+ audio_seconds=audio_seconds,
131
+ video_seconds=video_seconds,
132
+ pad_samples=pad_samples,
133
+ )
134
+ return audio
135
+
136
+
137
+ def slice_samples(waveform, start, length):
138
+ """Cut length samples out of a (channels, samples) waveform from start.
139
+
140
+ A slice reaching past the end of the waveform is zero-padded to the
141
+ requested length, so frame-aligned slicing near the end of a track always
142
+ yields full-size chunks.
143
+ """
144
+ channels, total = waveform.shape
145
+ piece = waveform[:, start : start + length]
146
+ if piece.shape[1] < length:
147
+ padding = numpy.zeros((channels, length - piece.shape[1]), dtype=waveform.dtype)
148
+ piece = numpy.concatenate([piece, padding], axis=1)
149
+ return piece
150
+
151
+
152
+ def equal_power_crossfade_join(
153
+ previous, head, following, sample_rate, crossfade_ms, seam_fade_ms=None
154
+ ):
155
+ """Join two segments' audio at a seam without changing the total duration.
156
+
157
+ previous ends at the seam. head is the audio trimmed off the next segment's
158
+ start - it covers the same stretch of time as the tail of previous, so the
159
+ two are blended with an equal-power crossfade over the last
160
+ min(crossfade_ms, len(head)) of that stretch. following is the next
161
+ segment's on-timeline audio and is appended unchanged.
162
+
163
+ With no head material (nothing was trimmed), the seam gets a fade-out and
164
+ fade-in in place instead, of seam_fade_ms - a few milliseconds by default,
165
+ just enough not to click.
166
+ """
167
+ previous, head, following = _matched_channels(previous, head, following)
168
+
169
+ window = min(
170
+ int(crossfade_ms / 1000.0 * sample_rate),
171
+ head.shape[1],
172
+ previous.shape[1],
173
+ )
174
+
175
+ if window == 0:
176
+ return _declick_join(previous, following, sample_rate, seam_fade_ms)
177
+
178
+ fade_out, fade_in = _equal_power_ramps(window)
179
+ blended = previous[:, -window:] * fade_out + head[:, -window:] * fade_in
180
+ return numpy.concatenate([previous[:, :-window], blended, following], axis=1)
181
+
182
+
183
+ # Below this, the reversed tail's spectral energy is concentrated in a few
184
+ # bins rather than spread across the band - speech or a pitched/tonal element
185
+ # rather than room tone or crowd noise, the direction-agnostic material a
186
+ # bleed is meant for
187
+ TONAL_FLATNESS_THRESHOLD = 0.3
188
+
189
+ # Above this, the tail's waveform repeats closely enough within a plausible
190
+ # pitch period to be voiced speech or a pitched note rather than noise -
191
+ # a bandwidth-insensitive companion to flatness, since a resample's own
192
+ # band limiting does not touch how periodic the waveform is (#198)
193
+ HARMONICITY_THRESHOLD = 0.45
194
+
195
+ # Typical fundamental range for a human voice or a pitched instrument note;
196
+ # the periodicity search only looks at lags in this range so a slow room-tone
197
+ # swell or hum near DC cannot register as a pitch
198
+ _PERIODICITY_MIN_HZ = 60.0
199
+ _PERIODICITY_MAX_HZ = 500.0
200
+
201
+
202
+ def _spectral_flatness(waveform, sample_rate=None, native_sample_rate=None):
203
+ """Geometric-mean-over-arithmetic-mean of the magnitude spectrum, averaged
204
+ across channels - near 0 for tonal/speech material, near 1 for noise-like
205
+ material (see TONAL_FLATNESS_THRESHOLD).
206
+
207
+ When the material was upsampled, band-limited interpolation leaves near
208
+ zero energy above the original Nyquist - a large near-silent band that
209
+ depresses the geometric mean relative to the arithmetic one regardless of
210
+ what the material actually is, reading as spuriously tonal (#198). Given
211
+ both rates, the spectrum is limited to bins below the native Nyquist so an
212
+ upsampled tail is measured the same as it would be at its own rate.
213
+ """
214
+ spectrum = numpy.abs(numpy.fft.rfft(waveform, axis=1))
215
+ if sample_rate and native_sample_rate and native_sample_rate < sample_rate:
216
+ native_bins = max(
217
+ 2,
218
+ int(spectrum.shape[1] * native_sample_rate / sample_rate),
219
+ )
220
+ spectrum = spectrum[:, :native_bins]
221
+ spectrum = numpy.maximum(spectrum, 1e-10)
222
+ geometric_mean = numpy.exp(numpy.mean(numpy.log(spectrum), axis=1))
223
+ arithmetic_mean = numpy.mean(spectrum, axis=1)
224
+ return float(numpy.mean(geometric_mean / arithmetic_mean))
225
+
226
+
227
+ def _harmonicity(waveform, sample_rate):
228
+ """Normalized autocorrelation peak within a plausible pitch range,
229
+ averaged across channels - near 1 for a strongly periodic signal (voiced
230
+ speech, a pitched note), near 0 for noise (see HARMONICITY_THRESHOLD).
231
+
232
+ Spectral flatness alone missed real speech (#198): a vowel's formants
233
+ spread its energy broadly enough across the band that flatness reads
234
+ similar to noise, even though the waveform itself repeats every pitch
235
+ period. Autocorrelation measures that repetition directly and is
236
+ insensitive to how the spectrum happens to be shaped, so it catches what
237
+ flatness cannot.
238
+ """
239
+ min_lag = max(int(sample_rate / _PERIODICITY_MAX_HZ), 1)
240
+ max_lag = min(int(sample_rate / _PERIODICITY_MIN_HZ), waveform.shape[1] - 1)
241
+ if max_lag <= min_lag:
242
+ return 0.0
243
+
244
+ scores = []
245
+ for channel in waveform:
246
+ centered = channel - channel.mean()
247
+ energy = float(numpy.dot(centered, centered))
248
+ if energy <= 1e-12:
249
+ continue
250
+ correlation = numpy.correlate(centered, centered, mode="full")
251
+ zero_lag = correlation.shape[0] // 2
252
+ window = correlation[zero_lag + min_lag : zero_lag + max_lag + 1]
253
+ if window.size == 0:
254
+ continue
255
+ scores.append(float(numpy.max(window) / energy))
256
+ return max(scores) if scores else 0.0
257
+
258
+
259
+ def bleed_join(
260
+ previous,
261
+ following,
262
+ sample_rate,
263
+ bleed_ms,
264
+ seam_fade_ms=None,
265
+ gain_db=0.0,
266
+ native_sample_rate=None,
267
+ ):
268
+ """Butt-join two waveforms, ringing the outgoing tail on across the seam.
269
+
270
+ Cut-based workflows generate every shot independently, so nothing overlaps at
271
+ a seam and there is no trimmed material to crossfade. Generated shots also
272
+ tend to open on near-silence and end mid-sound - a laugh track still rolling,
273
+ a room still ringing - so a plain butt-join drops a wall of sound into a hole.
274
+
275
+ This lays a decaying copy of the outgoing tail over the head of the incoming
276
+ waveform, the way an audience carries across a picture cut. The copy is
277
+ time-reversed so it starts on the outgoing waveform's own last sample and the
278
+ seam stays continuous without a declick fade; crowd noise and room tone are
279
+ direction-agnostic, so the reversal itself is not audible - speech or a
280
+ tonal/musical tail is not, which is what gets a warning below rather than a
281
+ refusal, since a caller who has already listened to the material may still
282
+ want the bleed.
283
+
284
+ The tail is added to whatever the incoming waveform already carries, and
285
+ neither side is shortened, so frames and samples stay in step.
286
+
287
+ Args:
288
+ previous: Waveform ending at the seam
289
+ following: Waveform starting at the seam
290
+ sample_rate: Sample rate of both waveforms
291
+ bleed_ms: How long the tail rings on, clamped to the material available
292
+ seam_fade_ms: Fade applied on each side of the seam when there is no
293
+ material to bleed at all
294
+ gain_db: Gain applied to the bled copy before it is added, in dB -
295
+ negative ducks a tail that would otherwise push the seam over
296
+ 0 dBFS; 0 (the default) is unchanged, full-scale, the prior
297
+ behavior
298
+ native_sample_rate: The rate the outgoing tail was actually recorded
299
+ or generated at, when that differs from sample_rate because the
300
+ caller upsampled it to join. Band-limits the flatness check to
301
+ below the tail's own Nyquist, so upsampling's near-silent high
302
+ band cannot itself read as tonal (#198). Omit when the tail is
303
+ already at its native rate
304
+
305
+ Returns:
306
+ The two waveforms joined, of their full combined length
307
+ """
308
+ previous, following = _matched_channels(previous, following)
309
+
310
+ window = min(
311
+ int(bleed_ms / 1000.0 * sample_rate),
312
+ previous.shape[1],
313
+ following.shape[1],
314
+ )
315
+ if window <= 0:
316
+ return _declick_join(previous, following, sample_rate, seam_fade_ms)
317
+
318
+ tail = previous[:, ::-1][:, :window]
319
+
320
+ tail_source = previous[:, -window:]
321
+ flatness = _spectral_flatness(tail_source, sample_rate, native_sample_rate)
322
+ # sample_rate, not native_sample_rate: _harmonicity turns a rate into lag
323
+ # bounds in samples of the waveform it is handed, and that waveform is at
324
+ # sample_rate however it got there. Passing the native rate of an upsampled
325
+ # tail searched the wrong lag range (16k against a 48k track: 180-1500 Hz
326
+ # rather than 60-500) and could miss the voiced speech #198 added it for.
327
+ # Only _spectral_flatness wants the native rate, to band-limit its window.
328
+ harmonicity = _harmonicity(tail_source, sample_rate)
329
+ if flatness < TONAL_FLATNESS_THRESHOLD or harmonicity > HARMONICITY_THRESHOLD:
330
+ emit_warning(
331
+ f"bleed_join: the tail being reversed onto the seam looks tonal or "
332
+ f"speech-like (spectral flatness {flatness:.2f}, harmonicity "
333
+ f"{harmonicity:.2f}) rather than the room tone or crowd noise a "
334
+ f"bleed is meant for - the reversal is likely to be audible as a "
335
+ f"stutter or a note running backwards. Pass 'audio_bleed_ms': 0 "
336
+ f"for a hard cut on this material instead - seam_fade_ms has no "
337
+ f"effect while audio_bleed_ms is non-zero.",
338
+ kind="bleed_tonal_material",
339
+ command="bleed_join",
340
+ flatness=round(flatness, 3),
341
+ harmonicity=round(harmonicity, 3),
342
+ )
343
+
344
+ decay, _ = _equal_power_ramps(window) # cos: 1 down to ~0
345
+ gain = 10.0 ** (gain_db / 20.0) if gain_db else 1.0
346
+ following = following.copy()
347
+ following[:, :window] += tail * decay * gain
348
+
349
+ peak = numpy.abs(following[:, :window]).max()
350
+ if peak > 1.0:
351
+ logger.warning(
352
+ f"Audio bleed pushed the seam to {peak:.2f} - it is added to the "
353
+ f"incoming track, which was not silent enough to absorb it. Pass "
354
+ f"a negative gain_db to duck the bled copy."
355
+ )
356
+ return numpy.concatenate([previous, following], axis=1)
357
+
358
+
359
+ def crossfade_concat(waveforms, sample_rate, crossfade_ms, starts=None):
360
+ """Concatenate waveforms, overlapping each seam by an equal-power crossfade.
361
+
362
+ The classic crossfade: each seam overlaps the two waveforms by the fade
363
+ window, so the result is shorter than the plain sum by one window per seam.
364
+
365
+ `starts`, when given a list, is filled with the sample each waveform
366
+ begins at in the result - where its crossfade opens - measured as the
367
+ result grows rather than worked out from the lengths (#378).
368
+ """
369
+ waveforms = [as_channels_samples(waveform) for waveform in waveforms]
370
+ if not waveforms:
371
+ raise ValueError("No waveforms to concatenate")
372
+
373
+ result = waveforms[0]
374
+ if starts is not None:
375
+ starts.append(0)
376
+ for following in waveforms[1:]:
377
+ result, following = _matched_channels(result, following)
378
+ window = min(
379
+ int(round(crossfade_ms / 1000.0 * sample_rate)),
380
+ result.shape[1],
381
+ following.shape[1],
382
+ )
383
+ if starts is not None:
384
+ starts.append(result.shape[1] - window)
385
+ if window == 0:
386
+ result = _declick_join(result, following, sample_rate)
387
+ continue
388
+
389
+ fade_out, fade_in = _equal_power_ramps(window)
390
+ blended = result[:, -window:] * fade_out + following[:, :window] * fade_in
391
+ result = numpy.concatenate(
392
+ [result[:, :-window], blended, following[:, window:]], axis=1
393
+ )
394
+
395
+ return result
396
+
397
+
398
+ def load_audio(location, base_dir=None):
399
+ """Load an audio file from a local path or http(s) URL.
400
+
401
+ A video file loads too, and contributes the track muxed into it: the cut
402
+ an earlier run wrote is exactly what a scoring pass wants to mix under,
403
+ and refusing its extension made an agent extract the audio by hand
404
+ (2026-09-08). A video without an audio stream is an error, not silence.
405
+
406
+ Returns:
407
+ Tuple of a (channels, samples) float32 waveform and its sample rate
408
+ """
409
+ from ..security import ALLOWED_VIDEO_EXTENSIONS
410
+
411
+ extension = os.path.splitext(location.split("?", 1)[0])[1].lower()
412
+ if extension in ALLOWED_VIDEO_EXTENSIONS:
413
+ from .video_utils import load_audio_video
414
+
415
+ video = load_audio_video(location, base_dir=base_dir)
416
+ if video.audio is None:
417
+ raise ValueError(
418
+ f"{location} carries no audio track - a video's soundtrack is "
419
+ "what an audio task takes from it"
420
+ )
421
+ return as_channels_samples(video.audio), video.sample_rate
422
+
423
+ if location.startswith(("http://", "https://")):
424
+ from ..locations import safe_get
425
+
426
+ logger.debug(f"Downloading audio from {location}")
427
+ response = safe_get(location, "an audio argument", timeout=60)
428
+ data, sample_rate = soundfile.read(
429
+ io.BytesIO(response.content), dtype="float32"
430
+ )
431
+ else:
432
+ from ..locations import validate_media_path
433
+
434
+ validated_path = validate_media_path(location, base_dir, "an audio argument")
435
+ validate_file_extension(validated_path, ALLOWED_AUDIO_EXTENSIONS)
436
+ logger.debug(f"Reading audio from {validated_path}")
437
+ data, sample_rate = soundfile.read(validated_path, dtype="float32")
438
+
439
+ # soundfile returns (samples,) or (samples, channels)
440
+ return as_channels_samples(data), sample_rate
441
+
442
+
443
+ def _as_track(waveform, sample_rate, command="an audio task", source_mean_dbfs=None):
444
+ """An audio task's return value: the waveform with the rate it is at.
445
+
446
+ Every one of these commands already knows the rate - it was given, or it
447
+ came off the file or the video the track was taken from - and dropping it
448
+ on the way out made the next command in the chain ask for it again. A
449
+ resample fed straight from a slice failed for want of a number the slice
450
+ had read and thrown away (2026-09-11). An AudioTrack carries it, and
451
+ everything downstream of audio reads '.audio'/'.sample_rate' already; a
452
+ 'sample_rate' the workflow declares on the result still wins at save.
453
+
454
+ A rate that is not a rate stops here. Saving falls back to 44100 Hz for a
455
+ track that carries none (DEFAULT_AUDIO_SAMPLE_RATE in result.py), so a
456
+ zero handed through would have been written as a 44100 Hz header over
457
+ samples at some other rate - the same audio at the wrong speed and pitch,
458
+ reported as a success (#140). There is no waveform whose rate is zero, so
459
+ the only thing to do with one is refuse it.
460
+ """
461
+ from ..result import AudioTrack
462
+
463
+ rate = int(sample_rate) if sample_rate is not None else 0
464
+ if rate <= 0:
465
+ raise ValueError(
466
+ f"{command} ended up with a sample rate of {sample_rate!r}, which "
467
+ f"is not a rate. Labelling a waveform with a rate it is not at "
468
+ f"changes its speed and pitch, so it is refused rather than "
469
+ f"written"
470
+ )
471
+ return AudioTrack(
472
+ numpy.ascontiguousarray(waveform), rate, source_mean_dbfs=source_mean_dbfs
473
+ )
474
+
475
+
476
+ def _as_number(value, kind, name, command="slice_audio"):
477
+ """Coerce a numeric task argument given as a string, leaving None alone."""
478
+ if not isinstance(value, str):
479
+ return value
480
+ try:
481
+ return kind(value)
482
+ except ValueError as e:
483
+ raise ValueError(f"{command} needs a number for '{name}', got {value!r}") from e
484
+
485
+
486
+ def slice_audio(
487
+ audio,
488
+ start_seconds=None,
489
+ duration_seconds=None,
490
+ start_frame=None,
491
+ num_frames=None,
492
+ fps=None,
493
+ sample_rate=None,
494
+ ):
495
+ """Task command: cut a slice out of an audio track.
496
+
497
+ The slice is addressed either in seconds (start_seconds + duration_seconds)
498
+ or in video frames (start_frame + num_frames + fps).
499
+
500
+ A slice reaching past the end of the track is zero-padded to the length
501
+ asked for - it does not fail and it is not shortened - and the padding is
502
+ digital silence, so asking for more than the source holds returns a track
503
+ that is partly empty. Anything past a few milliseconds of that is
504
+ reported as a 'slice_past_end' warning on the job. To fill a cut longer
505
+ than the recording, make a bed with the 'loop_audio' task first
506
+ ('target_frames' + 'fps' matches one exactly) and slice that.
507
+
508
+ Either half of a pair may be left out: with no start the slice begins at the
509
+ head of the track, and with no duration it runs to the end of it. A workflow
510
+ that trims only when it is told a length therefore still produces the track
511
+ rather than failing.
512
+
513
+ Args:
514
+ audio: Path or URL of an audio file (or of a video file, whose
515
+ soundtrack is taken), a video generated with a
516
+ soundtrack (which brings its sample rate along), or a waveform
517
+ (which needs sample_rate alongside it)
518
+ sample_rate: Sample rate of a waveform passed directly; given for a
519
+ file or a video it overrides the rate they carry
520
+
521
+ Returns:
522
+ An AudioTrack holding the slice and the rate it is at, so the next
523
+ audio command in the chain does not have to be told the rate again
524
+ """
525
+ # A variable a workflow declares null carries no type, so a value given for
526
+ # it on the command line arrives as a string - the same coercion the upscale
527
+ # and interpolation tasks do on their numeric arguments
528
+ start_seconds = _as_number(start_seconds, float, "start_seconds")
529
+ duration_seconds = _as_number(duration_seconds, float, "duration_seconds")
530
+ start_frame = _as_number(start_frame, int, "start_frame")
531
+ num_frames = _as_number(num_frames, int, "num_frames")
532
+ fps = _as_number(fps, Fraction, "fps")
533
+
534
+ # A count or an offset outside its domain is refused rather than handed to
535
+ # Python's slice semantics, which answered a negative 'num_frames' with
536
+ # the track minus its last N frames and called it a success (#139).
537
+ # validate_workflow refuses a literal one for free; this is the same
538
+ # refusal for a value that arrived from a variable or an earlier step
539
+ check_arguments(
540
+ "slice_audio",
541
+ start_seconds=start_seconds,
542
+ duration_seconds=duration_seconds,
543
+ start_frame=start_frame,
544
+ num_frames=num_frames,
545
+ fps=fps,
546
+ sample_rate=sample_rate,
547
+ )
548
+
549
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "slice_audio")
550
+ total = waveform.shape[1]
551
+
552
+ if start_seconds is not None or duration_seconds is not None:
553
+ start = int(round((start_seconds or 0) * sample_rate))
554
+ length = (
555
+ max(total - start, 0)
556
+ if duration_seconds is None
557
+ else int(round(duration_seconds * sample_rate))
558
+ )
559
+ elif start_frame is not None or num_frames is not None:
560
+ if fps is None:
561
+ raise ValueError("slice_audio needs 'fps' to address a slice in frames")
562
+ start = frames_to_samples(start_frame or 0, fps, sample_rate)
563
+ length = (
564
+ max(total - start, 0)
565
+ if num_frames is None
566
+ else frames_to_samples(num_frames, fps, sample_rate)
567
+ )
568
+ else:
569
+ raise ValueError(
570
+ "slice_audio needs either 'start_seconds'/'duration_seconds' or "
571
+ "'start_frame'/'num_frames'/'fps'"
572
+ )
573
+
574
+ _warn_on_slice_past_end(total, start, length, sample_rate)
575
+ _warn_on_slice_trims_tail(waveform, total, start, length, sample_rate)
576
+ # #309: a cut out of a source that was already near-silent (room tone,
577
+ # a deliberate quiet bed) is not a defect the slice introduced - measure
578
+ # the source before cutting it down, so save can tell the two apart from
579
+ # a track that arrived at a normal level and something upstream lost
580
+ source_mean_dbfs = level_dbfs(waveform, "rms")
581
+ return _as_track(
582
+ slice_samples(waveform, start, length),
583
+ sample_rate,
584
+ "slice_audio",
585
+ source_mean_dbfs=source_mean_dbfs,
586
+ )
587
+
588
+
589
+ def gain_audio(
590
+ audio,
591
+ gain_db,
592
+ start_seconds=None,
593
+ duration_seconds=None,
594
+ start_frame=None,
595
+ num_frames=None,
596
+ fps=None,
597
+ sample_rate=None,
598
+ ):
599
+ """Task command: apply a gain to a region of an audio track.
600
+
601
+ The region is addressed the same way slice_audio's is - either in
602
+ seconds (start_seconds + duration_seconds) or in video frames
603
+ (start_frame + num_frames + fps). Everything outside the region is
604
+ passed through unchanged, so ducking a scene under another is one step
605
+ rather than the slice/gain/mix/rejoin/pair_audio chain that was
606
+ previously the only way to apply a gain to part of a track rather than
607
+ all of it (#187). With no region given at all, the gain applies to the
608
+ whole track - the same "no region means everything" reading mix_audio's
609
+ gains use, and the obvious meaning of "duck this clip by 8 dB" (#395).
610
+ To gain everything from some point on, give just start_seconds=0 (or
611
+ start_frame=0 + fps) and leave duration_seconds/num_frames unset, which
612
+ runs to the end of the track without the caller needing to already know
613
+ how long that is.
614
+
615
+ Unlike slice_audio, a region reaching past the end of the track is
616
+ clipped to it rather than zero-padded: there is no silence there to
617
+ gain, only the end of the real material.
618
+
619
+ A file's or video's own sample rate is read automatically; sample_rate
620
+ is for a waveform passed directly, or to override what a file carries -
621
+ which relabels the waveform at that rate rather than resampling it, the
622
+ same caveat slice_audio's sample_rate carries (#180).
623
+
624
+ Args:
625
+ audio: Path or URL of an audio file (or of a video file, whose
626
+ soundtrack is taken), a generated video carrying its own
627
+ soundtrack, or a waveform (which needs sample_rate alongside it)
628
+ gain_db: Gain to apply within the region, in decibels - negative
629
+ ducks it, positive boosts it
630
+ start_seconds: Start of the region, in seconds. Omitted along with
631
+ every other region argument, the gain applies to the whole track
632
+ duration_seconds: Length of the region, in seconds
633
+ start_frame: Start of the region, in video frames
634
+ num_frames: Length of the region, in video frames
635
+ fps: Frame rate used to convert start_frame/num_frames to samples
636
+ sample_rate: Sample rate of a waveform passed directly; given for a
637
+ file or a video it overrides the rate they carry
638
+
639
+ Returns:
640
+ An AudioTrack holding the whole track with the region's gain
641
+ applied, and the rate it is at
642
+ """
643
+ start_seconds = _as_number(
644
+ start_seconds, float, "start_seconds", command="gain_audio"
645
+ )
646
+ duration_seconds = _as_number(
647
+ duration_seconds, float, "duration_seconds", command="gain_audio"
648
+ )
649
+ start_frame = _as_number(start_frame, int, "start_frame", command="gain_audio")
650
+ num_frames = _as_number(num_frames, int, "num_frames", command="gain_audio")
651
+ fps = _as_number(fps, Fraction, "fps", command="gain_audio")
652
+ gain_db = _as_number(gain_db, float, "gain_db", command="gain_audio")
653
+
654
+ check_arguments(
655
+ "gain_audio",
656
+ start_seconds=start_seconds,
657
+ duration_seconds=duration_seconds,
658
+ start_frame=start_frame,
659
+ num_frames=num_frames,
660
+ fps=fps,
661
+ sample_rate=sample_rate,
662
+ )
663
+
664
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "gain_audio")
665
+ total = waveform.shape[1]
666
+
667
+ if start_seconds is not None or duration_seconds is not None:
668
+ start = int(round((start_seconds or 0) * sample_rate))
669
+ length = (
670
+ max(total - start, 0)
671
+ if duration_seconds is None
672
+ else int(round(duration_seconds * sample_rate))
673
+ )
674
+ elif start_frame is not None or num_frames is not None:
675
+ if fps is None:
676
+ raise ValueError("gain_audio needs 'fps' to address a region in frames")
677
+ start = frames_to_samples(start_frame or 0, fps, sample_rate)
678
+ length = (
679
+ max(total - start, 0)
680
+ if num_frames is None
681
+ else frames_to_samples(num_frames, fps, sample_rate)
682
+ )
683
+ else:
684
+ start = 0
685
+ length = total
686
+
687
+ region_start = max(0, min(start, total))
688
+ region_end = max(region_start, min(start + max(length, 0), total))
689
+
690
+ gained = waveform.copy()
691
+ if region_end > region_start:
692
+ gain = 10 ** (gain_db / 20)
693
+ gained[:, region_start:region_end] = (
694
+ gained[:, region_start:region_end] * gain
695
+ ).astype(waveform.dtype)
696
+
697
+ region_start_seconds = region_start / float(sample_rate)
698
+ region_end_seconds = region_end / float(sample_rate)
699
+ emit_log(
700
+ f"gain_audio: {gain_db:.1f} dB over "
701
+ f"{region_start_seconds:.3f}-{region_end_seconds:.3f} s "
702
+ f"(samples {region_start}-{region_end} @ {sample_rate} Hz)",
703
+ command="gain_audio",
704
+ gain_db=gain_db,
705
+ start_seconds=region_start_seconds,
706
+ duration_seconds=region_end_seconds - region_start_seconds,
707
+ start_sample=region_start,
708
+ end_sample=region_end,
709
+ sample_rate=sample_rate,
710
+ )
711
+
712
+ return _as_track(gained, sample_rate, "gain_audio")
713
+
714
+
715
+ def _warn_on_slice_past_end(total, start, length, sample_rate):
716
+ """Say when a slice asked for more material than its source holds.
717
+
718
+ slice_samples zero-pads the shortfall, which is what makes frame-aligned
719
+ chunking near the end of a track work at all - but the same padding is
720
+ how a score shorter than the film it is laid under leaves the film
721
+ unscored for the rest of its length, with nothing anywhere saying so
722
+ (#126). emit_warning rather than logger.warning for the reason the
723
+ concat_videos resample warning is emitted: silently substituting silence
724
+ for four fifths of a track is an audio decision made on the caller's
725
+ behalf, and a caller reading the job over the API or MCP sees the
726
+ warnings list and nothing else (#82, #108).
727
+ """
728
+ available = max(0, min(total - start, length))
729
+ padded = length - available
730
+ if padded <= 0 or not sample_rate:
731
+ return
732
+ padded_seconds = padded / float(sample_rate)
733
+ if padded_seconds * 1000.0 < SLICE_PAD_WARN_MS:
734
+ # Frame-aligned slicing lands a sample or two past the end routinely;
735
+ # that is rounding, not a decision anyone can act on
736
+ return
737
+ emit_warning(
738
+ f"slice_audio: the requested slice runs "
739
+ f"{padded_seconds:.2f} s past the end of a "
740
+ f"{total / float(sample_rate):.2f} s source, so that much of the "
741
+ f"{length / float(sample_rate):.2f} s returned is digital silence. "
742
+ f"If you meant to fill a cut of this length, make a bed with the "
743
+ f"'loop_audio' task ('target_frames' + 'fps' matches one exactly) "
744
+ f"and slice that; if you meant the tail pad, nothing is wrong.",
745
+ kind="slice_past_end",
746
+ command="slice_audio",
747
+ source_seconds=round(total / float(sample_rate), 3),
748
+ requested_seconds=round(length / float(sample_rate), 3),
749
+ padded_seconds=round(padded_seconds, 3),
750
+ sample_rate=sample_rate,
751
+ )
752
+
753
+
754
+ def _warn_on_slice_trims_tail(waveform, total, start, length, sample_rate):
755
+ """Say when a slice left material behind that the caller likely wanted.
756
+
757
+ slice_audio is a slice, so most unused remainders are deliberate excerpts
758
+ and warning on every one would be noise. What #342 found is a narrower
759
+ signature: a cut landing a few seconds short of a source's natural end
760
+ (a frame-lattice total that cannot land exactly on the score's length)
761
+ silently drops the source's tail, including whatever is loudest there.
762
+ Only fires when the dropped remainder is both short in absolute terms
763
+ and small next to the slice itself, and only when that remainder is not
764
+ already silence - a track that legitimately ends in a fade should not
765
+ warn just because its last seconds are quiet.
766
+ """
767
+ if not sample_rate or length <= 0:
768
+ return
769
+ slice_end = start + length
770
+ remainder = total - slice_end
771
+ if remainder <= 0:
772
+ return
773
+ remainder_seconds = remainder / float(sample_rate)
774
+ if remainder_seconds >= SLICE_TRIM_WARN_SECONDS:
775
+ return
776
+ if remainder_seconds / (length / float(sample_rate)) >= SLICE_TRIM_WARN_FRACTION:
777
+ return
778
+ dropped = waveform[:, slice_end:total]
779
+ peak_dbfs = level_dbfs(dropped, "peak")
780
+ if peak_dbfs is None:
781
+ # No level at all is silence - nothing was lost
782
+ return
783
+ emit_warning(
784
+ f"slice_audio: the slice ends {remainder_seconds:.2f} s before the "
785
+ f"{total / float(sample_rate):.2f} s source does, dropping its tail "
786
+ f"(peak {peak_dbfs:.1f} dBFS in the dropped {remainder_seconds:.2f} s) "
787
+ f"- if the slice was meant to reach the source's end, adjust "
788
+ f"start/length to land there, or fade the source's own tail first",
789
+ kind="slice_trimmed_tail",
790
+ command="slice_audio",
791
+ source_seconds=round(total / float(sample_rate), 3),
792
+ dropped_seconds=round(remainder_seconds, 3),
793
+ dropped_peak_dbfs=round(peak_dbfs, 1),
794
+ sample_rate=sample_rate,
795
+ )
796
+
797
+
798
+ def resample_waveform(waveform, sample_rate, target_sample_rate):
799
+ """A waveform at a different rate, as a plain (channels, samples) array.
800
+
801
+ The conversion resample_audio performs, without the task's argument
802
+ handling or its AudioTrack return, so a task that has waveforms in hand
803
+ already can reach the rate conversion directly.
804
+ """
805
+ # PyAV's resampler accepts a zero rate and answers with the samples
806
+ # unchanged, which is indistinguishable from a conversion that happened
807
+ # (#140) - so neither rate is allowed to be one that cannot be a rate
808
+ for name, rate in (("sample_rate", sample_rate), ("target", target_sample_rate)):
809
+ if as_number(rate) is None or as_number(rate) <= 0:
810
+ raise ValueError(
811
+ f"resample_waveform needs a {name} above zero, got {rate!r}"
812
+ )
813
+ if sample_rate == target_sample_rate:
814
+ return waveform
815
+
816
+ import av
817
+ from av.audio.resampler import AudioResampler
818
+
819
+ channels = waveform.shape[0]
820
+ layout = {1: "mono", 2: "stereo"}.get(channels, f"{channels}c")
821
+ frame = av.AudioFrame.from_ndarray(
822
+ numpy.ascontiguousarray(waveform, dtype=numpy.float32),
823
+ format="fltp",
824
+ layout=layout,
825
+ )
826
+ frame.sample_rate = sample_rate
827
+ frame.pts = 0
828
+ frame.time_base = Fraction(1, sample_rate)
829
+
830
+ resampler = AudioResampler(format="fltp", layout=layout, rate=target_sample_rate)
831
+ converted = [f.to_ndarray() for f in resampler.resample(frame)]
832
+ converted += [f.to_ndarray() for f in resampler.resample(None)]
833
+ logger.debug(
834
+ f"Resampled {waveform.shape[1]} samples at {sample_rate}Hz "
835
+ f"to {target_sample_rate}Hz"
836
+ )
837
+ return numpy.concatenate(converted, axis=1).astype(numpy.float32)
838
+
839
+
840
+ def resample_audio(audio, target_sample_rate, sample_rate=None):
841
+ """Task command: resample an audio track to a different sample rate.
842
+
843
+ MiniMax H3 conditions on audio at its audio VAE's own rate and resamples
844
+ anything else with torchaudio, which dw does not depend on. Resampling a
845
+ supplied recording once, up front, feeds the pipeline what it already wants
846
+ and keeps the dependency out - PyAV, which dw needs for video anyway, does
847
+ the conversion.
848
+
849
+ Args:
850
+ audio: Path or URL of an audio file (or of a video file, whose
851
+ soundtrack is taken), a video generated with a
852
+ soundtrack (which brings its sample rate along), or a waveform
853
+ (which needs sample_rate alongside it)
854
+ target_sample_rate: Rate to convert to
855
+ sample_rate: Sample rate of a waveform passed directly; given for a
856
+ file or a video it overrides the rate they carry
857
+
858
+ Returns:
859
+ An AudioTrack holding the resampled waveform and its new rate
860
+ """
861
+ target_sample_rate = _as_number(
862
+ target_sample_rate, int, "target_sample_rate", "resample_audio"
863
+ )
864
+ # A zero or negative rate is not a rate. It used to reach PyAV's resampler,
865
+ # which left the samples alone, and then the save, which fell back to
866
+ # 44100 Hz - the original audio under a header 38% off, reported as a
867
+ # success (#140)
868
+ check_arguments(
869
+ "resample_audio",
870
+ target_sample_rate=target_sample_rate,
871
+ sample_rate=sample_rate,
872
+ )
873
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "resample_audio")
874
+ resampled = resample_waveform(waveform, sample_rate, target_sample_rate)
875
+ emit_log(
876
+ f"resample_audio: {sample_rate} → {target_sample_rate} Hz, "
877
+ f"{resampled.shape[-1] / target_sample_rate:.2f} s",
878
+ command="resample_audio",
879
+ source_sample_rate=sample_rate,
880
+ target_sample_rate=target_sample_rate,
881
+ seconds=round(resampled.shape[-1] / target_sample_rate, 2),
882
+ )
883
+ return _as_track(
884
+ resampled,
885
+ target_sample_rate,
886
+ "resample_audio",
887
+ )
888
+
889
+
890
+ def _track_names(audios):
891
+ """A name per audio track, for a warning that has to say which one.
892
+
893
+ Mirrors concat_videos' video_names: a caller passes a path, or a
894
+ previous step's result; only the path says anything by itself, so the
895
+ rest are named by position.
896
+ """
897
+ return [
898
+ original if isinstance(original, str) else f"track {index + 1}"
899
+ for index, original in enumerate(audios)
900
+ ]
901
+
902
+
903
+ def _load_tracks_matching_rate(audios, sample_rate, command):
904
+ """Load a list of audio tracks, resampling any that disagree on rate.
905
+
906
+ A mismatch among tracks that bring their own rate has no editorial
907
+ meaning - the same reasoning concat_videos (#108) and dissolve_videos
908
+ (#287) apply to shots - so when the caller has not pinned a rate the
909
+ highest one found is chosen as the target and the rest are converted,
910
+ with a warning naming each track's rate. When the caller *does* pin
911
+ `sample_rate`, _waveform_and_rate's own per-track relabel warning
912
+ (#180) still applies and this function changes nothing about it.
913
+ """
914
+ names = _track_names(audios)
915
+ waveforms, native_rates, bare = [], [], False
916
+ for audio in audios:
917
+ if isinstance(audio, str) or hasattr(audio, "audio"):
918
+ waveform, rate = _waveform_and_rate(audio, sample_rate, command)
919
+ native_rates.append(rate)
920
+ else:
921
+ waveform, bare = as_channels_samples(audio), True
922
+ native_rates.append(None)
923
+ waveforms.append(waveform)
924
+
925
+ if sample_rate is not None:
926
+ return waveforms, sample_rate
927
+
928
+ resolved = [rate for rate in native_rates if rate is not None]
929
+ if bare or not resolved:
930
+ raise ValueError(f"{command} needs 'sample_rate' with a raw waveform")
931
+ target_rate = max(resolved)
932
+ if len(set(resolved)) > 1:
933
+ per_track = {
934
+ name: rate for name, rate in zip(names, native_rates) if rate is not None
935
+ }
936
+ emit_warning(
937
+ f"{command}: tracks carry audio at different sample rates ("
938
+ + ", ".join(f"{name}: {rate} Hz" for name, rate in per_track.items())
939
+ + f") - resampling them all to {target_rate} Hz. Pass "
940
+ "'sample_rate' to pin a different target, or resample ahead of "
941
+ "this step with the 'resample_audio' task.",
942
+ kind="sample_rate_mismatch",
943
+ command=command,
944
+ sample_rate=target_rate,
945
+ sample_rates=per_track,
946
+ )
947
+ waveforms = [
948
+ (
949
+ waveform
950
+ if rate is None or rate == target_rate
951
+ else resample_waveform(waveform, rate, target_rate)
952
+ )
953
+ for waveform, rate in zip(waveforms, native_rates)
954
+ ]
955
+ return waveforms, target_rate
956
+
957
+
958
+ def crossfade_audio(audios, crossfade_ms=75, sample_rate=None):
959
+ """Task command: join audio tracks with an equal-power crossfade.
960
+
961
+ Each seam overlaps the two tracks by the fade window, so the result is
962
+ shorter than the plain sum by one window per seam.
963
+
964
+ Args:
965
+ audios: The tracks to join, in order - waveforms, audio or video file
966
+ paths, or videos generated with a soundtrack
967
+ crossfade_ms: Length of each crossfade
968
+ sample_rate: Sample rate of the joined track. Required unless every
969
+ track brings its own. Left unset, tracks at different rates are
970
+ not a constraint - the highest rate found is used and the rest
971
+ are resampled up to it, with a warning naming which (#108, #287,
972
+ #293). Given here instead, it *relabels* rather than resamples
973
+ any track whose real rate disagrees - changing its speed and
974
+ pitch, not just its rate - which warns separately (#180); use
975
+ resample_audio ahead of this step if conversion is what is
976
+ wanted at a pinned rate
977
+
978
+ Returns:
979
+ An AudioTrack holding the joined waveform and its rate
980
+ """
981
+ if not isinstance(audios, list) or not audios:
982
+ raise ValueError("crossfade_audio needs a non-empty list of audio tracks")
983
+ waveforms, sample_rate = _load_tracks_matching_rate(
984
+ audios, sample_rate, "crossfade_audio"
985
+ )
986
+ return _as_track(
987
+ crossfade_concat(waveforms, sample_rate, crossfade_ms), sample_rate
988
+ )
989
+
990
+
991
+ # #306: templates/assemble-and-score and templates/dissolve-between-shots
992
+ # both ship a stock world_gain of 1.8 - a deliberate multiplier, not a dB
993
+ # figure typed into the wrong unit - so the not-dB heuristic below has to sit
994
+ # above it
995
+ GAIN_LOOKS_LIKE_DB_ABOVE = 3.0
996
+
997
+
998
+ def mix_audio(audios, gains=None, sample_rate=None):
999
+ """Task command: layer audio tracks on top of one another.
1000
+
1001
+ crossfade_audio puts tracks one after another; this puts them on top of
1002
+ each other. It is what a score laid under a film's own sound needs: the
1003
+ music runs unbroken while the world underneath it is replaced at every cut.
1004
+
1005
+ Tracks of different lengths are padded with silence to the longest, so a
1006
+ score shorter than the picture leaves the tail dry rather than cutting the
1007
+ picture down to fit.
1008
+
1009
+ Summing can push peaks past full scale. This returns the plain weighted sum
1010
+ and does not rescale it, since quietening a mix is a decision about how it
1011
+ should sound - follow it with normalize_audio to bring the peak back down.
1012
+
1013
+ Args:
1014
+ audios: The tracks to layer - waveforms, audio or video file paths, or videos
1015
+ generated with a soundtrack
1016
+ gains: One plain multiplier per track, in the same order - not decibels.
1017
+ Defaults to unity on every track
1018
+ sample_rate: Sample rate of the joined mix. Required unless every
1019
+ track brings its own. Left unset, tracks at different rates are
1020
+ not a constraint - the highest rate found is used and the rest
1021
+ are resampled up to it, with a warning naming which (#108, #287,
1022
+ #293). Given here instead, it *relabels* rather than resamples
1023
+ any track whose real rate disagrees - changing its speed and
1024
+ pitch, not just its rate - which warns separately (#180); use
1025
+ resample_audio ahead of this step if conversion is what is
1026
+ wanted at a pinned rate
1027
+
1028
+ Returns:
1029
+ An AudioTrack holding the mixed waveform and its rate
1030
+ """
1031
+ if not isinstance(audios, list) or not audios:
1032
+ raise ValueError("mix_audio needs a non-empty list of audio tracks")
1033
+ if gains is not None and len(gains) != len(audios):
1034
+ raise ValueError(
1035
+ f"mix_audio needs one gain per track - got {len(gains)} for "
1036
+ f"{len(audios)} tracks"
1037
+ )
1038
+ check_arguments("mix_audio", gains=gains, sample_rate=sample_rate)
1039
+ if gains is not None:
1040
+ # #306: a modest boost (a stock template's world_gain: 1.8 among them)
1041
+ # is a legitimate multiplier a caller chose on purpose, not a typo -
1042
+ # only a gain loud enough that a caller almost certainly meant it as
1043
+ # dB (12, 6, 20, ...) is worth flagging. GAIN_LOOKS_LIKE_DB_ABOVE sits
1044
+ # above any observed catalog default and below the smallest figure a
1045
+ # dB-as-multiplier typo would produce (a "6 dB" or "12 dB" boost)
1046
+ loud = [
1047
+ g
1048
+ for g in gains
1049
+ if as_number(g) is not None and as_number(g) > GAIN_LOOKS_LIKE_DB_ABOVE
1050
+ ]
1051
+ if loud:
1052
+ emit_warning(
1053
+ f"mix_audio: gain(s) {loud} are a multiplier, not decibels - "
1054
+ f"a value like 12, 6 or -3 is almost always a dB figure typed "
1055
+ f"into the wrong unit. A multiplier above 1 boosts the track; "
1056
+ f"convert a dB figure with 10 ** (db / 20) if that was intended.",
1057
+ kind="mix_audio_gain_not_db",
1058
+ command="mix_audio",
1059
+ gains=gains,
1060
+ )
1061
+
1062
+ waveforms, sample_rate = _load_tracks_matching_rate(
1063
+ audios, sample_rate, "mix_audio"
1064
+ )
1065
+
1066
+ waveforms = _matched_channels(*waveforms)
1067
+ channels = waveforms[0].shape[0]
1068
+ length = max(waveform.shape[1] for waveform in waveforms)
1069
+
1070
+ mixed = numpy.zeros((channels, length), dtype=numpy.float32)
1071
+ applied = []
1072
+ for index, waveform in enumerate(waveforms):
1073
+ gain = 1.0 if gains is None else float(gains[index])
1074
+ applied.append(gain)
1075
+ mixed[:, : waveform.shape[1]] += waveform * gain
1076
+ emit_log(
1077
+ f"mix_audio: {len(waveforms)} tracks, gains {applied}",
1078
+ command="mix_audio",
1079
+ gains=applied,
1080
+ )
1081
+ return _as_track(mixed, sample_rate, "mix_audio")
1082
+
1083
+
1084
+ def loop_audio(
1085
+ audio,
1086
+ duration_seconds=None,
1087
+ target_frames=None,
1088
+ fps=None,
1089
+ crossfade_ms=250,
1090
+ sample_rate=None,
1091
+ ):
1092
+ """Task command: make a bed of a given length out of a short recording.
1093
+
1094
+ A cut between two independently generated shots has a hole in it: each
1095
+ shot carries its own room, and nothing runs underneath the seam. A
1096
+ continuous bed laid under the whole cut is what fills it - the way a
1097
+ location's room tone is laid under a dialogue scene so the edits stop
1098
+ being audible - and a bed is made by looping a few seconds of tone to
1099
+ the length of the picture.
1100
+
1101
+ Laps are joined with an equal-power crossfade rather than butted
1102
+ together, so the loop point itself is not a click. That only smooths the
1103
+ seam, though: a transient in the source (a hit, a swell) still recurs
1104
+ once per lap at full strength, so the loop still reads as a level pulse
1105
+ at the lap rate - measured at 9.3 dB on a source with one such transient.
1106
+ Picking a source with even internal level avoids the pulse; the
1107
+ crossfade does not. The source is used whole
1108
+ every lap; only the last one is trimmed, to land exactly on the
1109
+ requested length. A source longer than the request is trimmed to it.
1110
+
1111
+ Args:
1112
+ audio: Path or URL of an audio file (or of a video file, whose
1113
+ soundtrack is taken), a video generated with a soundtrack, or a
1114
+ waveform (which needs sample_rate alongside it)
1115
+ duration_seconds: How long the bed should be, in seconds
1116
+ target_frames: How long the bed should be, in video frames - needs
1117
+ 'fps', and is how a bed is matched to a cut exactly
1118
+ fps: Frame rate 'target_frames' is counted at
1119
+ crossfade_ms: Length of the crossfade at each loop point, clamped to
1120
+ the material available
1121
+ sample_rate: Sample rate of a waveform passed directly; given for a
1122
+ file or a video it overrides the rate they carry
1123
+
1124
+ Returns:
1125
+ An AudioTrack holding the bed and the rate it is at
1126
+ """
1127
+ duration_seconds = _as_number(duration_seconds, float, "duration_seconds")
1128
+ target_frames = _as_number(target_frames, int, "target_frames")
1129
+ fps = _as_number(fps, Fraction, "fps")
1130
+ crossfade_ms = _as_number(crossfade_ms, float, "crossfade_ms")
1131
+
1132
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "loop_audio")
1133
+ if waveform.size == 0:
1134
+ raise ValueError("loop_audio needs a source with samples in it")
1135
+
1136
+ if duration_seconds is not None:
1137
+ length = int(round(duration_seconds * sample_rate))
1138
+ elif target_frames is not None:
1139
+ if fps is None:
1140
+ raise ValueError("loop_audio needs 'fps' to count a length in frames")
1141
+ length = frames_to_samples(target_frames, fps, sample_rate)
1142
+ else:
1143
+ raise ValueError(
1144
+ "loop_audio needs either 'duration_seconds' or 'target_frames'/'fps' "
1145
+ "to know how long a bed to make"
1146
+ )
1147
+ if length <= 0:
1148
+ raise ValueError(f"loop_audio needs a length above zero, got {length} samples")
1149
+ if crossfade_ms < 0:
1150
+ raise ValueError("loop_audio 'crossfade_ms' cannot be negative")
1151
+
1152
+ window = min(int(crossfade_ms / 1000.0 * sample_rate), waveform.shape[1] // 2)
1153
+ bed = waveform
1154
+ laps = 1
1155
+ # Each lap after the first overlaps the one before it by the crossfade, so
1156
+ # a lap adds (source - window) samples rather than a whole source
1157
+ while bed.shape[1] < length:
1158
+ bed = crossfade_concat(
1159
+ [bed, waveform], sample_rate, window / sample_rate * 1000.0
1160
+ )
1161
+ laps += 1
1162
+ emit_log(
1163
+ f"loop_audio: {waveform.shape[1]} samples at {sample_rate}Hz looped "
1164
+ f"{laps}x to {length} samples ({length / sample_rate:.2f} s)",
1165
+ command="loop_audio",
1166
+ laps=laps,
1167
+ output_samples=length,
1168
+ output_seconds=round(length / sample_rate, 2),
1169
+ )
1170
+ return _as_track(bed[:, :length], sample_rate, "loop_audio")
1171
+
1172
+
1173
+ # Shots generated independently land at whatever level the model chose, and
1174
+ # joining two of them butts one loudness against another - the one seam
1175
+ # artifact no fade can hide, because it is not at the seam, it is either side
1176
+ # of it. These are the levels a matched join targets, and the spread at which
1177
+ # an unmatched one is worth warning about
1178
+ MATCH_MEASURES = ("peak", "rms")
1179
+ DEFAULT_MATCH_DBFS = {"peak": -1.0, "rms": -20.0}
1180
+ # Matching to an rms target can ask for a gain that would clip; the peak is
1181
+ # held here instead, which keeps a loud shot's relative level honest rather
1182
+ # than squaring off its transients
1183
+ MATCH_CEILING_DBFS = -0.5
1184
+ LEVEL_SPREAD_WARN_DB = 6.0
1185
+ # Mirrors result.py's NEAR_SILENT_WARN_DBFS: the same mean/rms level a job's
1186
+ # own near-silent check treats as having no real content. Gaining an input
1187
+ # already this quiet up to the target raises a noise floor rather than
1188
+ # leveling a performance, and #434 found a +29.9 dB case that only reached
1189
+ # the log, never job.warnings
1190
+ MATCH_NEAR_SILENT_DBFS = -40.0
1191
+ MATCH_LARGE_GAIN_WARN_DB = 20.0
1192
+
1193
+
1194
+ def level_dbfs(waveform, measure="peak"):
1195
+ """A waveform's level in dBFS, measured as `peak` or `rms`.
1196
+
1197
+ `rms` is the same measurement `get_gallery_metadata` reports as
1198
+ `mean_dbfs`, so a matched join can be checked against what the gallery
1199
+ said about the shots going into it. A silent track has no level: None.
1200
+ """
1201
+ if measure not in MATCH_MEASURES:
1202
+ raise ValueError(
1203
+ f"level measure must be one of {MATCH_MEASURES}, got '{measure}'"
1204
+ )
1205
+ if waveform is None or waveform.size == 0:
1206
+ return None
1207
+ if measure == "peak":
1208
+ value = float(numpy.abs(waveform).max())
1209
+ else:
1210
+ value = float(
1211
+ numpy.sqrt(numpy.mean(numpy.square(waveform, dtype=numpy.float64)))
1212
+ )
1213
+ if value <= 0.0:
1214
+ return None
1215
+ return 20.0 * numpy.log10(value)
1216
+
1217
+
1218
+ def match_levels(waveforms, measure, target_dbfs=None, command="concat_videos"):
1219
+ """Scale each waveform so its level sits at one shared target.
1220
+
1221
+ Returns a new list in the same order and shape; a None entry (a video
1222
+ with no soundtrack) and a silent track pass through untouched, since
1223
+ neither has a level to move. A gain that would push the peak past
1224
+ MATCH_CEILING_DBFS is held there and said so in the log - the shot is
1225
+ then quieter than the target rather than clipped.
1226
+ """
1227
+ if measure not in MATCH_MEASURES:
1228
+ raise ValueError(
1229
+ f"{command} 'match_levels' must be one of {MATCH_MEASURES}, got '{measure}'"
1230
+ )
1231
+ if target_dbfs is None:
1232
+ target_dbfs = DEFAULT_MATCH_DBFS[measure]
1233
+ if target_dbfs > 0:
1234
+ raise ValueError(
1235
+ f"{command} 'match_levels_dbfs' cannot be above full scale (0)"
1236
+ )
1237
+
1238
+ matched = []
1239
+ for index, waveform in enumerate(waveforms):
1240
+ level = level_dbfs(waveform, measure)
1241
+ if level is None:
1242
+ matched.append(waveform)
1243
+ continue
1244
+ target_gain_db = target_dbfs - level
1245
+ gain_db = target_gain_db
1246
+ peak = level_dbfs(waveform, "peak")
1247
+ held = False
1248
+ if peak is not None and peak + gain_db > MATCH_CEILING_DBFS:
1249
+ gain_db = MATCH_CEILING_DBFS - peak
1250
+ held = True
1251
+ shortfall_db = target_gain_db - gain_db
1252
+ # emit_warning rather than logger.warning: a clip-held shot stays
1253
+ # off the target and the residual spread is exactly the level
1254
+ # jump match_levels exists to remove (#214) - a caller reading
1255
+ # the job's warnings list is the one who can act on it (#82)
1256
+ emit_warning(
1257
+ f"{command}: video {index + 1} would clip at the {measure} target "
1258
+ f"({peak + target_gain_db:+.1f} dBFS peak) - held to "
1259
+ f"{MATCH_CEILING_DBFS} dBFS, {shortfall_db:.1f} dB short of target",
1260
+ kind="match_levels_held",
1261
+ command=command,
1262
+ index=index,
1263
+ measure_dbfs=round(level, 1),
1264
+ target_dbfs=target_dbfs,
1265
+ gain_db=round(gain_db, 1),
1266
+ shortfall_db=round(shortfall_db, 1),
1267
+ ceiling_dbfs=MATCH_CEILING_DBFS,
1268
+ )
1269
+ elif level <= MATCH_NEAR_SILENT_DBFS or gain_db >= MATCH_LARGE_GAIN_WARN_DB:
1270
+ # The other end of the range `held` covers (#434): an input this
1271
+ # quiet is noise floor, not a performance at a lower level, and
1272
+ # matching it up to the target passes that noise off as content -
1273
+ # a consumer reading job.warnings sees nothing was wrong
1274
+ emit_warning(
1275
+ f"{command}: video {index + 1} {measure} {level:.1f} dBFS is "
1276
+ f"near-silent - matched up to the target with a {gain_db:+.1f} dB "
1277
+ "gain, raising its noise floor rather than leveling content",
1278
+ kind="match_levels_near_silent",
1279
+ command=command,
1280
+ index=index,
1281
+ measure_dbfs=round(level, 1),
1282
+ target_dbfs=target_dbfs,
1283
+ gain_db=round(gain_db, 1),
1284
+ )
1285
+ emit_log(
1286
+ f"{command}: video {index + 1} {measure} {level:.1f} dBFS, "
1287
+ f"gain {gain_db:+.1f} dB{' (held)' if held else ''}",
1288
+ index=index,
1289
+ measure_dbfs=round(level, 1),
1290
+ gain_db=round(gain_db, 1),
1291
+ held=held,
1292
+ )
1293
+ matched.append((waveform * (10 ** (gain_db / 20.0))).astype(numpy.float32))
1294
+ return matched
1295
+
1296
+
1297
+ def warn_on_level_spread(waveforms, command="concat_videos", measure="rms"):
1298
+ """Say something when shots about to be joined are levels apart.
1299
+
1300
+ Independently generated shots drift by 10 dB and more, and each one reads
1301
+ as fine on its own - it is only wrong relative to what it is cut against,
1302
+ and nothing else compares them.
1303
+ """
1304
+ levels = [level for level in (level_dbfs(w, measure) for w in waveforms) if level]
1305
+ if len(levels) < 2:
1306
+ return None
1307
+ spread = max(levels) - min(levels)
1308
+ if spread >= LEVEL_SPREAD_WARN_DB:
1309
+ # emit_warning rather than logger.warning: this is a property of the
1310
+ # file the run is about to write, and the caller reading the job is
1311
+ # the one who can act on it (#82)
1312
+ emit_warning(
1313
+ f"{command}: the tracks being joined span {spread:.1f} dB "
1314
+ f"({measure} {min(levels):.1f} to {max(levels):.1f} dBFS) - "
1315
+ "audible as a level jump unless the difference is intended (a "
1316
+ "shot written silent against the score). If it is not, pass "
1317
+ "match_levels to even them out",
1318
+ kind="level_spread",
1319
+ command=command,
1320
+ spread_db=round(spread, 1),
1321
+ measure=measure,
1322
+ )
1323
+ return spread
1324
+
1325
+
1326
+ def _equal_power_ramps(window):
1327
+ """Cosine/sine fade curves that sum to constant power across the window."""
1328
+ theta = numpy.linspace(0.0, numpy.pi / 2.0, window, endpoint=False)
1329
+ return numpy.cos(theta, dtype=numpy.float32), numpy.sin(theta, dtype=numpy.float32)
1330
+
1331
+
1332
+ def _declick_join(previous, following, sample_rate, fade_ms=None):
1333
+ """Butt-join two waveforms with a fade on each side of the seam.
1334
+
1335
+ The default is the few milliseconds that keep a butt-join from clicking.
1336
+ A longer fade is a deliberate edit - the graceful hard cut you want when
1337
+ neither a crossfade nor a bleed applies.
1338
+ """
1339
+ ramp = int((DECLICK_MS if fade_ms is None else fade_ms) / 1000.0 * sample_rate)
1340
+ ramp = min(ramp, previous.shape[1], following.shape[1])
1341
+ if ramp > 0:
1342
+ fade_out, fade_in = _equal_power_ramps(ramp)
1343
+ previous = previous.copy()
1344
+ following = following.copy()
1345
+ previous[:, -ramp:] *= fade_out # cos: 1 down to ~0
1346
+ following[:, :ramp] *= fade_in # sin: ~0 up to 1
1347
+ return numpy.concatenate([previous, following], axis=1)
1348
+
1349
+
1350
+ def _matched_channels(*waveforms):
1351
+ """Tile mono up so every waveform has the same channel count."""
1352
+ channels = max(waveform.shape[0] for waveform in waveforms)
1353
+ return tuple(
1354
+ (
1355
+ numpy.tile(waveform, (channels, 1))
1356
+ if waveform.shape[0] == 1 and channels > 1
1357
+ else waveform
1358
+ )
1359
+ for waveform in waveforms
1360
+ )
1361
+
1362
+
1363
+ def fade_audio(audio, fade_in_ms=0, fade_out_ms=0, sample_rate=None):
1364
+ """Task command: fade a track in from silence and out to it.
1365
+
1366
+ A slice cut out of the middle of a piece ends on whatever was sounding at
1367
+ the cut; a short fade turns that into an ending. The curve is the
1368
+ equal-power cosine the seam joins use, so a fade sounds like a fade and
1369
+ not a volume knob.
1370
+
1371
+ Args:
1372
+ audio: Path or URL of an audio file (or of a video file, whose
1373
+ soundtrack is taken), a video generated with a
1374
+ soundtrack, or a waveform (which needs sample_rate alongside it)
1375
+ fade_in_ms: Length of the fade in, from the head of the track
1376
+ fade_out_ms: Length of the fade out, to the tail of the track
1377
+ sample_rate: Sample rate of a waveform passed directly
1378
+
1379
+ Returns:
1380
+ An AudioTrack holding the faded waveform and its rate
1381
+ """
1382
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "fade_audio")
1383
+ if fade_in_ms < 0 or fade_out_ms < 0:
1384
+ raise ValueError("fade_audio fade lengths cannot be negative")
1385
+ faded = waveform.copy()
1386
+ length = faded.shape[1]
1387
+
1388
+ fade_in = min(int(round(fade_in_ms / 1000 * sample_rate)), length)
1389
+ if fade_in:
1390
+ faded[:, :fade_in] *= _fade_curve(fade_in)[::-1]
1391
+ fade_out = min(int(round(fade_out_ms / 1000 * sample_rate)), length)
1392
+ if fade_out:
1393
+ faded[:, length - fade_out :] *= _fade_curve(fade_out)
1394
+ return _as_track(faded, sample_rate, "fade_audio")
1395
+
1396
+
1397
+ def normalize_audio(audio, peak_dbfs=-1.0, target_lufs=None, sample_rate=None):
1398
+ """Task command: scale a track so its loudest sample sits at a level.
1399
+
1400
+ Generated music comes out at whatever level the model happened to land
1401
+ on - quiet takes need lifting before they sit under a picture, and a
1402
+ hot one needs headroom before the encoder. Peak normalization changes
1403
+ nothing but the gain, so the dynamics survive.
1404
+
1405
+ Args:
1406
+ audio: Path or URL of an audio file (or of a video file, whose
1407
+ soundtrack is taken), a video generated with a
1408
+ soundtrack, or a waveform (which needs sample_rate alongside it)
1409
+ peak_dbfs: The level the loudest sample is moved to, in dB below full
1410
+ scale. 0 is full scale; -1 leaves a little headroom. Still
1411
+ applies as a ceiling when target_lufs is also given
1412
+ target_lufs: Integrated loudness (BS.1770) to gain the track to, in
1413
+ LUFS. Peak alone says nothing about how loud a track sounds - a
1414
+ sparse voice-over and a dense score can share a peak and still
1415
+ sit tens of dB apart to the ear (#361). When given, the gain
1416
+ targets this loudness first; peak_dbfs still holds as a ceiling,
1417
+ and if reaching target_lufs would cross it the gain stops at the
1418
+ ceiling and a warning names the shortfall in LU. None (the
1419
+ default) leaves behavior exactly as peak-only
1420
+ sample_rate: Sample rate of a waveform passed directly
1421
+
1422
+ Returns:
1423
+ An AudioTrack holding the scaled waveform and its rate; a silent
1424
+ track is returned unchanged
1425
+ """
1426
+ check_arguments("normalize_audio", sample_rate=sample_rate, target_lufs=target_lufs)
1427
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "normalize_audio")
1428
+ if peak_dbfs > 0:
1429
+ raise ValueError("normalize_audio 'peak_dbfs' cannot be above full scale (0)")
1430
+ peak = float(numpy.abs(waveform).max()) if waveform.size else 0.0
1431
+ if peak == 0.0:
1432
+ logger.warning("normalize_audio: the track is silent - left unchanged")
1433
+ return _as_track(waveform, sample_rate, "normalize_audio")
1434
+
1435
+ peak_db = 20 * numpy.log10(peak)
1436
+ measured_lufs = None
1437
+ if target_lufs is None:
1438
+ constraint = "peak_dbfs"
1439
+ gain_db = peak_dbfs - peak_db
1440
+ else:
1441
+ ceiling_gain_db = peak_dbfs - peak_db
1442
+ current_lufs = integrated_lufs(waveform.T, sample_rate)
1443
+ measured_lufs = current_lufs
1444
+ if current_lufs is None:
1445
+ emit_warning(
1446
+ f"normalize_audio: target_lufs={target_lufs} was given, but the "
1447
+ "track's loudness could not be measured (shorter than the 400 ms "
1448
+ "gating block, or silent throughout) - falling back to peak_dbfs "
1449
+ "alone.",
1450
+ kind="target_lufs_unmeasurable",
1451
+ command="normalize_audio",
1452
+ target_lufs=target_lufs,
1453
+ )
1454
+ gain_db = ceiling_gain_db
1455
+ constraint = "peak_ceiling"
1456
+ else:
1457
+ target_gain_db = target_lufs - current_lufs
1458
+ gain_db = min(target_gain_db, ceiling_gain_db)
1459
+ constraint = "peak_ceiling" if gain_db < target_gain_db else "target_lufs"
1460
+ if gain_db < target_gain_db:
1461
+ emit_warning(
1462
+ f"normalize_audio: target_lufs={target_lufs} would need "
1463
+ f"{target_gain_db:+.1f} dB of gain, but peak_dbfs={peak_dbfs} "
1464
+ f"caps it at {gain_db:+.1f} dB - "
1465
+ f"{target_gain_db - gain_db:.1f} LU short of the target.",
1466
+ kind="target_lufs_capped",
1467
+ command="normalize_audio",
1468
+ target_lufs=target_lufs,
1469
+ peak_dbfs=peak_dbfs,
1470
+ shortfall_lu=target_gain_db - gain_db,
1471
+ )
1472
+ gain = 10 ** (gain_db / 20)
1473
+ emit_log(
1474
+ f"normalize_audio: measured {peak_db:.1f} dBFS peak"
1475
+ + ("" if measured_lufs is None else f", {measured_lufs:.1f} LUFS")
1476
+ + f" -> gain {gain_db:+.1f} dB, set by {constraint}",
1477
+ command="normalize_audio",
1478
+ measured_peak_dbfs=round(peak_db, 1),
1479
+ measured_lufs=round(measured_lufs, 1) if measured_lufs is not None else None,
1480
+ gain_db=round(gain_db, 1),
1481
+ constraint=constraint,
1482
+ )
1483
+ return _as_track(
1484
+ (waveform * gain).astype(numpy.float32), sample_rate, "normalize_audio"
1485
+ )
1486
+
1487
+
1488
+ def _fade_curve(window):
1489
+ """A cosine fall from full level to exact silence, both ends included -
1490
+ unlike the seam ramps, which stop short of the endpoint so two of them
1491
+ tile a crossfade without a doubled sample."""
1492
+ theta = numpy.linspace(0.0, numpy.pi / 2.0, window, endpoint=True)
1493
+ return numpy.cos(theta, dtype=numpy.float32)
1494
+
1495
+
1496
+ def _waveform_and_rate(audio, sample_rate, command):
1497
+ """A command's audio argument as a (channels, samples) array with its rate.
1498
+
1499
+ A path loads with the file's own rate; a video generated with a soundtrack
1500
+ (an AudioVideo, or anything carrying `.audio`) contributes that track and
1501
+ its rate; a bare waveform needs the rate given. A given rate always wins -
1502
+ correct for a raw waveform, which carries none of its own, but for a named
1503
+ source (a file or a video) a rate that disagrees with the one it actually
1504
+ carries relabels the samples rather than converting them, changing speed
1505
+ and pitch with nothing saying so (#180) - so that case warns.
1506
+ """
1507
+ if isinstance(audio, str):
1508
+ waveform, file_rate = load_audio(audio)
1509
+ if (
1510
+ sample_rate is not None
1511
+ and file_rate is not None
1512
+ and sample_rate != file_rate
1513
+ ):
1514
+ _warn_on_rate_override(command, file_rate, sample_rate)
1515
+ return waveform, sample_rate if sample_rate is not None else file_rate
1516
+ if hasattr(audio, "audio"):
1517
+ if audio.audio is None:
1518
+ raise ValueError(
1519
+ f"{command} needs an audio track - the video it was given carries none"
1520
+ )
1521
+ if (
1522
+ sample_rate is not None
1523
+ and audio.sample_rate is not None
1524
+ and sample_rate != audio.sample_rate
1525
+ ):
1526
+ _warn_on_rate_override(command, audio.sample_rate, sample_rate)
1527
+ rate = sample_rate if sample_rate is not None else audio.sample_rate
1528
+ if rate is None:
1529
+ raise ValueError(
1530
+ f"{command} needs 'sample_rate' - the video it was given does not "
1531
+ "carry one of its own"
1532
+ )
1533
+ return as_channels_samples(audio.audio), rate
1534
+ if sample_rate is None:
1535
+ raise ValueError(f"{command} needs 'sample_rate' with a raw waveform")
1536
+ return as_channels_samples(audio), sample_rate
1537
+
1538
+
1539
+ def _warn_on_rate_override(command, actual_rate, given_rate):
1540
+ """Say when a given sample_rate relabels a named source's real rate.
1541
+
1542
+ 'sample_rate' always overrides the rate a file or video carries - that is
1543
+ what lets a raw waveform (which has none of its own) be handed in at all -
1544
+ but for a named source it is easy to mistake for a conversion: a workflow
1545
+ reused one variable as both 'the rate a mix runs at' and 'the rate this
1546
+ file is at', and the mismatch reached nobody until the deliverable played
1547
+ at the wrong speed with `warnings: []` (#180). emit_warning rather than
1548
+ logger.warning for the reason every other run-time audio warning here is
1549
+ (#82, #108): a caller reading the job over the API or MCP sees the
1550
+ warnings list and nothing else.
1551
+ """
1552
+ emit_warning(
1553
+ f"{command}: sample_rate={given_rate} was given, but the source "
1554
+ f"actually carries {actual_rate} Hz. The samples are being relabeled "
1555
+ f"at {given_rate} Hz, not resampled - this changes speed and pitch. "
1556
+ f"If you meant to convert the rate, use 'resample_audio' "
1557
+ f"(target_sample_rate={given_rate}) instead.",
1558
+ kind="rate_override_mismatch",
1559
+ command=command,
1560
+ file_rate=actual_rate,
1561
+ given_rate=given_rate,
1562
+ )
1563
+
1564
+
1565
+ COMPRESS_MODES = ("compress", "limit", "gate")
1566
+
1567
+ # A floor below which an envelope is treated as digital silence, so its dBFS
1568
+ # reading is a large negative number rather than -inf
1569
+ _ENVELOPE_FLOOR_DBFS = -120.0
1570
+ _ENVELOPE_FLOOR_LINEAR = 10.0 ** (_ENVELOPE_FLOOR_DBFS / 20.0)
1571
+
1572
+
1573
+ def compress_audio(
1574
+ audio,
1575
+ threshold_dbfs,
1576
+ ratio=4.0,
1577
+ attack_ms=10.0,
1578
+ release_ms=100.0,
1579
+ mode="compress",
1580
+ sample_rate=None,
1581
+ ):
1582
+ """Task command: shape a track's dynamics with an envelope-follower.
1583
+
1584
+ A compressor, a limiter and a gate are the same envelope-follower
1585
+ algorithm with different knob settings: a limiter is a ratio pushed
1586
+ toward infinity with a fast attack, and a gate is downward expansion
1587
+ below the threshold rather than compression above it - so one command
1588
+ covers all three through 'mode' rather than three near-duplicate ones.
1589
+
1590
+ Args:
1591
+ audio: Path or URL of an audio file (or of a video file, whose
1592
+ soundtrack is taken), a video generated with a
1593
+ soundtrack, or a waveform (which needs sample_rate alongside it)
1594
+ threshold_dbfs: The level, in dB below full scale, above which
1595
+ 'compress'/'limit' reduce gain, or below which 'gate' does
1596
+ ratio: How strongly gain is reduced past the threshold. Unused by
1597
+ 'limit', which reduces enough to hold the signal at the
1598
+ threshold regardless
1599
+ attack_ms: How fast the envelope follows a rise in level
1600
+ release_ms: How fast the envelope follows a fall in level
1601
+ mode: 'compress' (downward compression above threshold), 'limit'
1602
+ (holds the signal at threshold), or 'gate' (downward expansion
1603
+ below threshold)
1604
+ sample_rate: Sample rate of a waveform passed directly
1605
+
1606
+ Returns:
1607
+ An AudioTrack holding the processed waveform and its rate
1608
+ """
1609
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "compress_audio")
1610
+ check_arguments(
1611
+ "compress_audio",
1612
+ ratio=ratio,
1613
+ attack_ms=attack_ms,
1614
+ release_ms=release_ms,
1615
+ sample_rate=sample_rate,
1616
+ )
1617
+ if mode not in COMPRESS_MODES:
1618
+ raise ValueError(
1619
+ f"compress_audio mode must be one of {COMPRESS_MODES}, got {mode!r}"
1620
+ )
1621
+ if threshold_dbfs > 0:
1622
+ raise ValueError(
1623
+ "compress_audio 'threshold_dbfs' cannot be above full scale (0)"
1624
+ )
1625
+ if waveform.size == 0:
1626
+ return _as_track(waveform, sample_rate, "compress_audio")
1627
+
1628
+ envelope = _follow_envelope(waveform, sample_rate, attack_ms, release_ms)
1629
+ envelope_dbfs = 20.0 * numpy.log10(numpy.maximum(envelope, _ENVELOPE_FLOOR_LINEAR))
1630
+
1631
+ if mode == "gate":
1632
+ past_threshold = numpy.maximum(0.0, threshold_dbfs - envelope_dbfs)
1633
+ else:
1634
+ past_threshold = numpy.maximum(0.0, envelope_dbfs - threshold_dbfs)
1635
+
1636
+ if mode == "limit":
1637
+ reduction_db = past_threshold
1638
+ else:
1639
+ reduction_db = past_threshold * (1.0 - 1.0 / ratio)
1640
+
1641
+ gain = (10.0 ** (-reduction_db / 20.0)).astype(numpy.float32)
1642
+ processed = (waveform * gain[numpy.newaxis, :]).astype(numpy.float32)
1643
+ return _as_track(processed, sample_rate, "compress_audio")
1644
+
1645
+
1646
+ def _follow_envelope(waveform, sample_rate, attack_ms, release_ms):
1647
+ """A linked (all-channels) peak envelope, smoothed by separate attack and
1648
+ release time constants - the same detector a hardware compressor uses,
1649
+ tracking the loudest channel so a stereo image does not shift."""
1650
+ rectified = numpy.abs(waveform).max(axis=0)
1651
+ attack_coef = _time_constant_coef(attack_ms, sample_rate)
1652
+ release_coef = _time_constant_coef(release_ms, sample_rate)
1653
+ # The branch on the running level is what makes this a loop rather than a
1654
+ # filter, but the per-sample numpy indexing was the expensive half of it:
1655
+ # a 3-minute track is ~8M samples, and this runs on the single FIFO
1656
+ # worker. tolist() hands the loop plain Python floats, which is the same
1657
+ # arithmetic on the same values, several times faster
1658
+ samples = rectified.tolist()
1659
+ envelope = []
1660
+ level = 0.0
1661
+ for sample in samples:
1662
+ coef = attack_coef if sample > level else release_coef
1663
+ level = coef * level + (1.0 - coef) * sample
1664
+ envelope.append(level)
1665
+ return numpy.asarray(envelope, dtype=rectified.dtype)
1666
+
1667
+
1668
+ def _time_constant_coef(time_ms, sample_rate):
1669
+ """The per-sample smoothing coefficient for an exponential time constant.
1670
+ 0 ms means the envelope follows instantly, with no smoothing at all."""
1671
+ if time_ms <= 0:
1672
+ return 0.0
1673
+ return float(numpy.exp(-1.0 / (time_ms / 1000.0 * sample_rate)))
1674
+
1675
+
1676
+ FILTER_KINDS = ("lowpass", "highpass", "bandpass", "notch")
1677
+
1678
+
1679
+ def filter_audio(audio, cutoff_hz, kind="lowpass", q=0.707, sample_rate=None):
1680
+ """Task command: run a track through a single biquad filter stage.
1681
+
1682
+ Args:
1683
+ audio: Path or URL of an audio file (or of a video file, whose
1684
+ soundtrack is taken), a video generated with a
1685
+ soundtrack, or a waveform (which needs sample_rate alongside it)
1686
+ cutoff_hz: The filter's corner (lowpass/highpass) or center
1687
+ (bandpass/notch) frequency
1688
+ kind: 'lowpass', 'highpass', 'bandpass', or 'notch'
1689
+ q: Resonance/bandwidth of the filter. Higher narrows a bandpass or
1690
+ notch, and peaks the corner of a lowpass or highpass
1691
+ sample_rate: Sample rate of a waveform passed directly
1692
+
1693
+ Returns:
1694
+ An AudioTrack holding the filtered waveform and its rate
1695
+ """
1696
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "filter_audio")
1697
+ check_arguments("filter_audio", cutoff_hz=cutoff_hz, sample_rate=sample_rate)
1698
+ if kind not in FILTER_KINDS:
1699
+ raise ValueError(
1700
+ f"filter_audio kind must be one of {FILTER_KINDS}, got {kind!r}"
1701
+ )
1702
+ if q <= 0:
1703
+ raise ValueError("filter_audio 'q' must be above zero")
1704
+ nyquist = sample_rate / 2.0
1705
+ if cutoff_hz >= nyquist:
1706
+ raise ValueError(
1707
+ f"filter_audio 'cutoff_hz' ({cutoff_hz}) must be below the "
1708
+ f"Nyquist frequency ({nyquist}) for sample_rate {sample_rate}"
1709
+ )
1710
+ if waveform.size == 0:
1711
+ return _as_track(waveform, sample_rate, "filter_audio")
1712
+
1713
+ b, a = _biquad_coefficients(kind, cutoff_hz, q, sample_rate)
1714
+ filtered = numpy.stack(
1715
+ [_apply_biquad(channel, b, a) for channel in waveform]
1716
+ ).astype(numpy.float32)
1717
+ return _as_track(filtered, sample_rate, "filter_audio")
1718
+
1719
+
1720
+ def _biquad_coefficients(kind, cutoff_hz, q, sample_rate):
1721
+ """RBJ Audio EQ Cookbook coefficients for a single biquad stage,
1722
+ normalized so a0 is 1."""
1723
+ w0 = 2.0 * numpy.pi * cutoff_hz / sample_rate
1724
+ cos_w0 = numpy.cos(w0)
1725
+ sin_w0 = numpy.sin(w0)
1726
+ alpha = sin_w0 / (2.0 * q)
1727
+
1728
+ if kind == "lowpass":
1729
+ b0 = (1.0 - cos_w0) / 2.0
1730
+ b1 = 1.0 - cos_w0
1731
+ b2 = (1.0 - cos_w0) / 2.0
1732
+ elif kind == "highpass":
1733
+ b0 = (1.0 + cos_w0) / 2.0
1734
+ b1 = -(1.0 + cos_w0)
1735
+ b2 = (1.0 + cos_w0) / 2.0
1736
+ elif kind == "bandpass":
1737
+ b0 = alpha
1738
+ b1 = 0.0
1739
+ b2 = -alpha
1740
+ else: # notch
1741
+ b0 = 1.0
1742
+ b1 = -2.0 * cos_w0
1743
+ b2 = 1.0
1744
+ a0 = 1.0 + alpha
1745
+ a1 = -2.0 * cos_w0
1746
+ a2 = 1.0 - alpha
1747
+ return (
1748
+ numpy.array([b0, b1, b2], dtype=numpy.float64) / a0,
1749
+ numpy.array([a1, a2], dtype=numpy.float64) / a0,
1750
+ )
1751
+
1752
+
1753
+ def _apply_biquad(channel, b, a):
1754
+ """One second-order section, run over a channel.
1755
+
1756
+ The feedback cannot be vectorized away, but it does not have to be run in
1757
+ Python either: scipy's lfilter is this exact recursion in C, and scipy is
1758
+ already in every install (controlnet-aux brings it). The Python Direct
1759
+ Form I below is the fallback for an environment without it - same
1760
+ recursion, same zero initial conditions, ~50x slower on a full track.
1761
+ """
1762
+ b0, b1, b2 = b
1763
+ a1, a2 = a
1764
+ try:
1765
+ from scipy.signal import lfilter
1766
+ except ImportError:
1767
+ pass
1768
+ else:
1769
+ return lfilter(
1770
+ numpy.array([b0, b1, b2], dtype=numpy.float64),
1771
+ numpy.array([1.0, a1, a2], dtype=numpy.float64),
1772
+ numpy.asarray(channel, dtype=numpy.float64),
1773
+ )
1774
+
1775
+ out = numpy.empty_like(channel, dtype=numpy.float64)
1776
+ x1 = x2 = y1 = y2 = 0.0
1777
+ for i in range(channel.shape[0]):
1778
+ x0 = float(channel[i])
1779
+ y0 = b0 * x0 + b1 * x1 + b2 * x2 - a1 * y1 - a2 * y2
1780
+ out[i] = y0
1781
+ x2, x1 = x1, x0
1782
+ y2, y1 = y1, y0
1783
+ return out
1784
+
1785
+
1786
+ _SPECTRAL_BANDS = {
1787
+ "low_dbfs": (20.0, 250.0),
1788
+ "mid_dbfs": (250.0, 4000.0),
1789
+ "high_dbfs": (4000.0, 20000.0),
1790
+ }
1791
+
1792
+
1793
+ def analyze_audio(audio, sample_rate=None):
1794
+ """Task command: measure a track without changing it.
1795
+
1796
+ Read-only: the waveform passes through unmodified, and what comes back
1797
+ is diagnostics rather than an AudioTrack, since there is no processed
1798
+ track to hand a later step. Meant to feed a decision earlier in a
1799
+ workflow (whether 'compress_audio' or 'filter_audio' is needed, and
1800
+ with what settings) rather than to sit in the middle of a chain.
1801
+
1802
+ Args:
1803
+ audio: Path or URL of an audio file (or of a video file, whose
1804
+ soundtrack is taken), a video generated with a
1805
+ soundtrack, or a waveform (which needs sample_rate alongside it)
1806
+ sample_rate: Sample rate of a waveform passed directly
1807
+
1808
+ Returns:
1809
+ A dict: peak_dbfs, rms_dbfs, crest_factor_db (peak minus rms), and
1810
+ a rough low_dbfs/mid_dbfs/high_dbfs spectral-balance reading whose
1811
+ three bands are shares of the same power that gives rms_dbfs, so
1812
+ they sit on that scale rather than tens of dB under it. Any value
1813
+ is None where a silent track leaves it undefined.
1814
+ """
1815
+ waveform, sample_rate = _waveform_and_rate(audio, sample_rate, "analyze_audio")
1816
+ check_arguments("analyze_audio", sample_rate=sample_rate)
1817
+
1818
+ peak_dbfs = level_dbfs(waveform, measure="peak")
1819
+ rms_dbfs = level_dbfs(waveform, measure="rms")
1820
+ crest_factor_db = (
1821
+ peak_dbfs - rms_dbfs if peak_dbfs is not None and rms_dbfs is not None else None
1822
+ )
1823
+ bands = _spectral_balance(waveform, sample_rate)
1824
+ return {
1825
+ "peak_dbfs": peak_dbfs,
1826
+ "rms_dbfs": rms_dbfs,
1827
+ "crest_factor_db": crest_factor_db,
1828
+ **bands,
1829
+ }
1830
+
1831
+
1832
+ def _spectral_balance(waveform, sample_rate):
1833
+ """A rough low/mid/high energy reading in dBFS, from one FFT of the
1834
+ channel-averaged track - not a spectrogram, just enough to say whether
1835
+ a track leans bright or boomy.
1836
+
1837
+ Each band's power is a share of the same Parseval sum that gives
1838
+ rms_dbfs (mean(x**2)): a one-sided rfft bin's power is doubled to
1839
+ account for its mirrored negative-frequency twin, except the DC and
1840
+ (for even n) Nyquist bins, which have no twin. Summed over the full
1841
+ spectrum this equals mean(x**2) exactly, so a *_dbfs band sits on the
1842
+ same scale as rms_dbfs rather than ~40 dB under it (#211)."""
1843
+ if waveform.size == 0:
1844
+ return {name: None for name in _SPECTRAL_BANDS}
1845
+ mono = waveform.mean(axis=0)
1846
+ n = mono.shape[0]
1847
+ spectrum = numpy.fft.rfft(mono)
1848
+ power = numpy.square(numpy.abs(spectrum), dtype=numpy.float64) / (n * n)
1849
+ if n % 2 == 0:
1850
+ power[1:-1] *= 2.0
1851
+ else:
1852
+ power[1:] *= 2.0
1853
+ freqs = numpy.fft.rfftfreq(n, d=1.0 / sample_rate)
1854
+ result = {}
1855
+ for name, (low, high) in _SPECTRAL_BANDS.items():
1856
+ band = power[(freqs >= low) & (freqs < min(high, sample_rate / 2.0))]
1857
+ if band.size == 0:
1858
+ result[name] = None
1859
+ continue
1860
+ energy = float(numpy.sum(band))
1861
+ result[name] = 10.0 * numpy.log10(energy) if energy > 0.0 else None
1862
+ return result