diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/media_info.py
ADDED
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
"""What the server knows about a generated media file and would otherwise
|
|
2
|
+
not say. An agent cannot listen: duration against a ceiling, peak against
|
|
3
|
+
a normalization target and the level at a seam are the only checks it can
|
|
4
|
+
make on an audio deliverable, and every one of them was being made by
|
|
5
|
+
fetching the file and running ffprobe by hand.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
import math
|
|
10
|
+
|
|
11
|
+
import av
|
|
12
|
+
import numpy
|
|
13
|
+
|
|
14
|
+
from .loudness import _dbfs, integrated_lufs, true_peak_dbfs
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger("dw")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def probe_media(path, envelope=False):
|
|
20
|
+
"""Duration, format and level of an audio or video file, or None.
|
|
21
|
+
|
|
22
|
+
Video answers fps, frame_count, width and height, plus the soundtrack's
|
|
23
|
+
sample_rate, channels, peak_dbfs, mean_dbfs, integrated_lufs and
|
|
24
|
+
true_peak_dbfs when it carries one; audio answers the soundtrack fields.
|
|
25
|
+
Levels come from decoding the whole track, which is cheap next to
|
|
26
|
+
generating it.
|
|
27
|
+
|
|
28
|
+
peak_dbfs and mean_dbfs are a single sample's level; integrated_lufs is
|
|
29
|
+
the BS.1770 loudness of the whole track (#361) - a sparse voice-over and
|
|
30
|
+
a dense score can share a peak and still sit tens of dB apart in how
|
|
31
|
+
loud they sound. integrated_lufs is None for a track shorter than the
|
|
32
|
+
400 ms gating block or one that is silent throughout - "unmeasurable",
|
|
33
|
+
not zero. true_peak_dbfs is the inter-sample (oversampled) peak BS.1770
|
|
34
|
+
also defines, which can read higher than peak_dbfs when an encoder's
|
|
35
|
+
reconstruction filter rings a decoded peak up past what any single
|
|
36
|
+
sample showed.
|
|
37
|
+
|
|
38
|
+
When a frame count still needs counting and/or a soundtrack still needs
|
|
39
|
+
its levels measured, both are gathered from a single decode pass over
|
|
40
|
+
whichever streams are involved - `container.decode()` demuxes to EOF, so
|
|
41
|
+
two separate passes (count video, then decode audio) would leave the
|
|
42
|
+
second one nothing to read.
|
|
43
|
+
|
|
44
|
+
With `envelope=True` the same decode also reports the level second by
|
|
45
|
+
second, as `envelope: {"interval_seconds": 1.0, "rms_dbfs": [...],
|
|
46
|
+
"peak_dbfs": [...]}` - which is what tells an agent *where* in a track
|
|
47
|
+
something is, rather than only how loud the whole thing was: whether a
|
|
48
|
+
shot is still voiced at its last frame, where a score's quiet passage
|
|
49
|
+
sits, how deep the hole at a seam goes. Off by default, because a
|
|
50
|
+
ten-minute track is 600 numbers nobody asked for and the default
|
|
51
|
+
metadata call has to stay small.
|
|
52
|
+
|
|
53
|
+
Args:
|
|
54
|
+
path: The file to probe
|
|
55
|
+
envelope: Also report the per-second level of the soundtrack. A lossy
|
|
56
|
+
codec's decode can run a fraction of a second past the file's
|
|
57
|
+
reported duration - its own priming and padding - so samples past
|
|
58
|
+
that duration are dropped rather than filling a bin of their own
|
|
59
|
+
(#277). A track whose *real* length isn't a whole number of
|
|
60
|
+
seconds still gets a genuine final bin shorter than the rest -
|
|
61
|
+
`len(rms_dbfs)` is `ceil(duration_seconds)`, not `floor` (#278)
|
|
62
|
+
"""
|
|
63
|
+
try:
|
|
64
|
+
container = av.open(path)
|
|
65
|
+
except Exception as e:
|
|
66
|
+
logger.debug(f"Not probeable as media: {path}: {e}")
|
|
67
|
+
return None
|
|
68
|
+
with container:
|
|
69
|
+
video = container.streams.video[0] if container.streams.video else None
|
|
70
|
+
audio = container.streams.audio[0] if container.streams.audio else None
|
|
71
|
+
if video is None and audio is None:
|
|
72
|
+
return None
|
|
73
|
+
info = {}
|
|
74
|
+
if video is not None:
|
|
75
|
+
info["kind"] = "video"
|
|
76
|
+
info["fps"] = float(video.average_rate) if video.average_rate else None
|
|
77
|
+
info["width"] = int(video.width)
|
|
78
|
+
info["height"] = int(video.height)
|
|
79
|
+
else:
|
|
80
|
+
info["kind"] = "audio"
|
|
81
|
+
if container.duration is not None:
|
|
82
|
+
info["duration_seconds"] = float(container.duration / av.time_base)
|
|
83
|
+
if audio is not None:
|
|
84
|
+
info["sample_rate"] = int(audio.rate)
|
|
85
|
+
info["channels"] = int(audio.channels)
|
|
86
|
+
# The audio *stream's* own reported duration, not the
|
|
87
|
+
# container's - assess.py's read_media() trims the decoded
|
|
88
|
+
# track to this figure (`_stream_seconds`), and a lossy mux can
|
|
89
|
+
# report the two slightly differently (#426), so a caller that
|
|
90
|
+
# needs to agree with what a probe will actually measure reads
|
|
91
|
+
# this rather than duration_seconds.
|
|
92
|
+
if audio.duration is not None and audio.time_base is not None:
|
|
93
|
+
info["audio_stream_seconds"] = float(audio.duration * audio.time_base)
|
|
94
|
+
|
|
95
|
+
# Some muxers don't write a frame count up front (0 means "count
|
|
96
|
+
# them"); a soundtrack always needs decoding to measure its level.
|
|
97
|
+
# Do both together, since decoding is a one-way trip through the file.
|
|
98
|
+
need_frame_count = video is not None and not video.frames
|
|
99
|
+
if video is not None and not need_frame_count:
|
|
100
|
+
info["frame_count"] = int(video.frames)
|
|
101
|
+
|
|
102
|
+
if need_frame_count or audio is not None:
|
|
103
|
+
frame_count = 0
|
|
104
|
+
peak = 0.0
|
|
105
|
+
total = 0.0
|
|
106
|
+
count = 0
|
|
107
|
+
# One bin per second of the soundtrack, filled as frames decode:
|
|
108
|
+
# [sum of squares, sample count, peak] - the same numbers the
|
|
109
|
+
# whole-track level is made of, kept per second instead of once
|
|
110
|
+
bins = [] if envelope and audio is not None else None
|
|
111
|
+
elapsed = 0 # samples of the soundtrack seen so far
|
|
112
|
+
# Samples past the file's reported duration are a lossy codec's
|
|
113
|
+
# own priming/padding, not real track content - drop them rather
|
|
114
|
+
# than let them fill (or half-fill) a bin of their own (#277).
|
|
115
|
+
# When the duration itself is unknown there is nothing to clip
|
|
116
|
+
# against, so the trailing-fragment merge below is the fallback.
|
|
117
|
+
max_envelope_samples = (
|
|
118
|
+
int(round(info["duration_seconds"] * audio.rate))
|
|
119
|
+
if bins is not None and info.get("duration_seconds") is not None
|
|
120
|
+
else None
|
|
121
|
+
)
|
|
122
|
+
# The whole soundtrack, accumulated the same way regardless of
|
|
123
|
+
# envelope - integrated loudness and true peak are measured over
|
|
124
|
+
# the full track, not per frame, so they need it assembled
|
|
125
|
+
# rather than the running peak/sum above. Trimmed past the
|
|
126
|
+
# file's reported duration for the same reason as the envelope
|
|
127
|
+
# bins (#277): priming/padding is not real content to measure.
|
|
128
|
+
lufs_chunks = [] if audio is not None else None
|
|
129
|
+
lufs_seen = 0
|
|
130
|
+
max_lufs_samples = (
|
|
131
|
+
int(round(info["duration_seconds"] * audio.rate))
|
|
132
|
+
if lufs_chunks is not None and info.get("duration_seconds") is not None
|
|
133
|
+
else None
|
|
134
|
+
)
|
|
135
|
+
streams = [
|
|
136
|
+
s
|
|
137
|
+
for s in ((video if need_frame_count else None), audio)
|
|
138
|
+
if s is not None
|
|
139
|
+
]
|
|
140
|
+
try:
|
|
141
|
+
for frame in container.decode(*streams):
|
|
142
|
+
if isinstance(frame, av.VideoFrame):
|
|
143
|
+
frame_count += 1
|
|
144
|
+
elif isinstance(frame, av.AudioFrame):
|
|
145
|
+
samples = frame.to_ndarray()
|
|
146
|
+
if samples.dtype.kind == "u":
|
|
147
|
+
iinfo = numpy.iinfo(samples.dtype)
|
|
148
|
+
half = (iinfo.max + 1) / 2
|
|
149
|
+
samples = (samples.astype(numpy.float32) - half) / half
|
|
150
|
+
elif samples.dtype.kind == "i":
|
|
151
|
+
samples = (
|
|
152
|
+
samples.astype(numpy.float32)
|
|
153
|
+
/ numpy.iinfo(samples.dtype).max
|
|
154
|
+
)
|
|
155
|
+
samples = samples.astype(numpy.float32)
|
|
156
|
+
peak = max(peak, float(numpy.abs(samples).max(initial=0.0)))
|
|
157
|
+
total += float(numpy.square(samples).sum())
|
|
158
|
+
count += samples.size
|
|
159
|
+
if bins is not None:
|
|
160
|
+
elapsed = _fill_envelope(
|
|
161
|
+
bins,
|
|
162
|
+
samples,
|
|
163
|
+
elapsed,
|
|
164
|
+
audio.rate,
|
|
165
|
+
int(audio.channels),
|
|
166
|
+
max_envelope_samples,
|
|
167
|
+
)
|
|
168
|
+
if lufs_chunks is not None:
|
|
169
|
+
frame = _as_frame_samples(samples, int(audio.channels))
|
|
170
|
+
full_length = frame.shape[0]
|
|
171
|
+
length = (
|
|
172
|
+
full_length
|
|
173
|
+
if max_lufs_samples is None
|
|
174
|
+
else max(
|
|
175
|
+
0, min(full_length, max_lufs_samples - lufs_seen)
|
|
176
|
+
)
|
|
177
|
+
)
|
|
178
|
+
if length > 0:
|
|
179
|
+
lufs_chunks.append(frame[:length])
|
|
180
|
+
lufs_seen += full_length
|
|
181
|
+
if bins is not None and max_envelope_samples is None:
|
|
182
|
+
_merge_trailing_fragment(bins, audio.rate)
|
|
183
|
+
except Exception as e:
|
|
184
|
+
# A track that opens fine can still fail mid-decode (damage
|
|
185
|
+
# past the header); the fields already gathered - duration,
|
|
186
|
+
# format - are still true, so report those rather than
|
|
187
|
+
# failing the whole probe. Matches read_embedded_metadata's
|
|
188
|
+
# precedent of degrading rather than raising.
|
|
189
|
+
logger.debug(f"Decode failed partway through {path}: {e}")
|
|
190
|
+
return info
|
|
191
|
+
if need_frame_count:
|
|
192
|
+
info["frame_count"] = frame_count
|
|
193
|
+
if audio is not None:
|
|
194
|
+
rms = math.sqrt(total / count) if count else 0.0
|
|
195
|
+
info["peak_dbfs"] = _dbfs(peak)
|
|
196
|
+
info["mean_dbfs"] = _dbfs(rms)
|
|
197
|
+
full = numpy.concatenate(lufs_chunks, axis=0) if lufs_chunks else None
|
|
198
|
+
info["integrated_lufs"] = integrated_lufs(full, audio.rate)
|
|
199
|
+
info["true_peak_dbfs"] = true_peak_dbfs(full)
|
|
200
|
+
if bins is not None:
|
|
201
|
+
info["envelope"] = _as_envelope(bins)
|
|
202
|
+
return info
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _as_frame_samples(samples, channels):
|
|
206
|
+
"""One decoded audio frame as a (samples, channels) array.
|
|
207
|
+
|
|
208
|
+
A planar format decodes to (channels, samples); a packed one decodes to
|
|
209
|
+
(1, samples * channels) interleaved. Both have to become a run of
|
|
210
|
+
samples before they can be cut on a second boundary, or a stereo packed
|
|
211
|
+
frame would be counted as twice as much time as it holds.
|
|
212
|
+
"""
|
|
213
|
+
if samples.ndim == 1:
|
|
214
|
+
return samples[:, numpy.newaxis]
|
|
215
|
+
if samples.shape[0] == channels and channels > 1:
|
|
216
|
+
return samples.T
|
|
217
|
+
if samples.shape[0] == 1 and channels > 1:
|
|
218
|
+
return samples.reshape(-1, channels)
|
|
219
|
+
return samples.T if samples.shape[0] < samples.shape[1] else samples
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _fill_envelope(bins, samples, elapsed, rate, channels, max_samples=None):
|
|
223
|
+
"""Add a decoded audio frame's samples to the per-second bins.
|
|
224
|
+
|
|
225
|
+
A bin covers one second of the track regardless of how the decoder
|
|
226
|
+
happened to chop it, so a frame straddling a second boundary is split
|
|
227
|
+
across the two bins rather than counted in whichever one it started in.
|
|
228
|
+
`elapsed` is how many samples of the track came before this frame; the
|
|
229
|
+
new total is returned - the frame's *full* length, even when `max_samples`
|
|
230
|
+
clipped how much of it was actually binned, so a later frame's position
|
|
231
|
+
is still measured against the real track rather than the clipped one.
|
|
232
|
+
|
|
233
|
+
`max_samples` is the file's reported duration in samples: a frame (or
|
|
234
|
+
the tail of one) landing past it is a lossy codec's own priming/padding
|
|
235
|
+
rather than real content, and is dropped instead of filling a bin (#277).
|
|
236
|
+
"""
|
|
237
|
+
frame = _as_frame_samples(samples, channels)
|
|
238
|
+
full_length = frame.shape[0]
|
|
239
|
+
length = (
|
|
240
|
+
full_length
|
|
241
|
+
if max_samples is None
|
|
242
|
+
else max(0, min(full_length, max_samples - elapsed))
|
|
243
|
+
)
|
|
244
|
+
start = 0
|
|
245
|
+
while start < length:
|
|
246
|
+
second = (elapsed + start) // rate
|
|
247
|
+
while len(bins) <= second:
|
|
248
|
+
bins.append([0.0, 0, 0.0])
|
|
249
|
+
# How much of this frame still belongs to the second it is in
|
|
250
|
+
room = int((second + 1) * rate - (elapsed + start))
|
|
251
|
+
stop = min(length, start + max(room, 1))
|
|
252
|
+
piece = frame[start:stop]
|
|
253
|
+
entry = bins[second]
|
|
254
|
+
entry[0] += float(numpy.square(piece).sum())
|
|
255
|
+
entry[1] += int(piece.size)
|
|
256
|
+
entry[2] = max(entry[2], float(numpy.abs(piece).max(initial=0.0)))
|
|
257
|
+
start = stop
|
|
258
|
+
return elapsed + full_length
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _merge_trailing_fragment(bins, rate):
|
|
262
|
+
"""Fold a short trailing bin into the one before it, in place.
|
|
263
|
+
|
|
264
|
+
Fallback for when the file's duration is unknown, so `_fill_envelope`
|
|
265
|
+
had nothing to clip decoding against: a lossy codec's decode can still
|
|
266
|
+
run a fraction of a second past the real track (its own priming and
|
|
267
|
+
padding), which leaves the last bin holding only a handful of samples -
|
|
268
|
+
not a real last second. Read at face value that fragment looks like a
|
|
269
|
+
hole (near -inf, since so little energy lands in so few samples), when
|
|
270
|
+
the actual last second is whatever the bin before it says. Merging need
|
|
271
|
+
only ever touch the last bin: every earlier one was closed out by a full
|
|
272
|
+
second's worth of samples arriving after it. When the duration *is*
|
|
273
|
+
known, `_fill_envelope`'s `max_samples` drops the same padding before it
|
|
274
|
+
ever reaches a bin, which also lets a genuinely partial final second
|
|
275
|
+
(a real duration that isn't a whole number of seconds) stand on its own
|
|
276
|
+
instead of being folded away (#278).
|
|
277
|
+
"""
|
|
278
|
+
if len(bins) < 2:
|
|
279
|
+
return
|
|
280
|
+
total, count, peak = bins[-1]
|
|
281
|
+
if count >= rate:
|
|
282
|
+
return
|
|
283
|
+
prev_total, prev_count, prev_peak = bins[-2]
|
|
284
|
+
bins[-2] = [prev_total + total, prev_count + count, max(prev_peak, peak)]
|
|
285
|
+
bins.pop()
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _as_envelope(bins):
|
|
289
|
+
"""The per-second bins as the levels an agent reads."""
|
|
290
|
+
return {
|
|
291
|
+
"interval_seconds": 1.0,
|
|
292
|
+
"rms_dbfs": [
|
|
293
|
+
_dbfs(math.sqrt(total / count) if count else 0.0)
|
|
294
|
+
for total, count, _peak in bins
|
|
295
|
+
],
|
|
296
|
+
"peak_dbfs": [_dbfs(peak) for _total, _count, peak in bins],
|
|
297
|
+
}
|