diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/tasks/assess.py
ADDED
|
@@ -0,0 +1,826 @@
|
|
|
1
|
+
"""Assessment probes: measure a finished cut and say where to look (#387).
|
|
2
|
+
|
|
3
|
+
Three read-only task commands - `analyze_shots`, `analyze_seams` and
|
|
4
|
+
`analyze_sync_drift` - each take a video (a path, or the AudioVideo an
|
|
5
|
+
earlier step returned) and answer a JSON-safe dict: every measurement it
|
|
6
|
+
took, the `findings` its rules raised (`dw/assessment_rules.py`), the names
|
|
7
|
+
of the rules it read the measurements against and where the shot
|
|
8
|
+
boundaries came from. Nothing acts on a finding; they are places to look.
|
|
9
|
+
|
|
10
|
+
The reader streams. A cut is minutes of full-resolution picture, and every
|
|
11
|
+
one of these measurements needs only a 64x36 grey thumbnail of each frame
|
|
12
|
+
and the soundtrack, so `read_media` decodes the file once and keeps exactly
|
|
13
|
+
that - never a frame list, and never `load_audio`'s 0.25 s fit of the track
|
|
14
|
+
to the picture, which would move the very sample count `analyze_sync_drift`
|
|
15
|
+
is measuring. The soundtrack is trimmed to the audio stream's own duration,
|
|
16
|
+
which the container already reports net of the encoder's priming, so a
|
|
17
|
+
lossy codec's padding is not read as drift.
|
|
18
|
+
|
|
19
|
+
Shot boundaries come, in order, from the `shots` argument, the AudioVideo's
|
|
20
|
+
own `shots`, the run manifest beside a file (`dw.runs.shots_beside`), and
|
|
21
|
+
otherwise the whole file is one shot. `shots_source` says which. A shot
|
|
22
|
+
record may carry `hard_cut: true` - a cut meant as a cut - and the frame
|
|
23
|
+
jump rule does not fire at the seam that shot opens.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
import logging
|
|
27
|
+
import math
|
|
28
|
+
|
|
29
|
+
import numpy
|
|
30
|
+
|
|
31
|
+
from ..assessment_rules import HOLE_VOICED_DBFS, crosses, finding, rules_for
|
|
32
|
+
from ..events import emit_warning
|
|
33
|
+
|
|
34
|
+
logger = logging.getLogger("dw")
|
|
35
|
+
|
|
36
|
+
THUMB_WIDTH = 64
|
|
37
|
+
THUMB_HEIGHT = 36
|
|
38
|
+
|
|
39
|
+
# Audio windows around a seam, in seconds. The edge windows say whether the
|
|
40
|
+
# join itself is voiced (the hole guard) and what the balance does across it;
|
|
41
|
+
# they are not the level step, which is shot against shot (`_shot_rms`): a
|
|
42
|
+
# shot's own last and first quarter-second differ by whatever the take does
|
|
43
|
+
# there - 20 dB on a line that trails off and opens on a breath - and read
|
|
44
|
+
# as a step at every seam of a cut made of one clip (#387's bounce).
|
|
45
|
+
LEVEL_WINDOW = 0.25
|
|
46
|
+
FLOOR_WINDOW = 0.02
|
|
47
|
+
CLICK_WINDOW = 0.002
|
|
48
|
+
CLICK_NEIGHBOUR_WINDOW = 0.01
|
|
49
|
+
|
|
50
|
+
# The inter-frame difference a shot is expected to show, below which its
|
|
51
|
+
# own motion is not a baseline: a static shot's 90th percentile is ~0, and
|
|
52
|
+
# dividing a seam's change by it would call any change at all a jump.
|
|
53
|
+
# Grey levels on the 0-255 scale.
|
|
54
|
+
TYPICAL_DELTA_FLOOR = 2.0
|
|
55
|
+
TYPICAL_DELTA_PERCENTILE = 90
|
|
56
|
+
|
|
57
|
+
# A ratio against a zero neighbour has no size; report it capped
|
|
58
|
+
CLICK_CAP_DB = 120.0
|
|
59
|
+
|
|
60
|
+
_SILENCE = 1e-10
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class Media:
|
|
64
|
+
"""What the probes read from a video: thumbnails, soundtrack, timing.
|
|
65
|
+
|
|
66
|
+
thumbs: (frames, THUMB_HEIGHT, THUMB_WIDTH) uint8 grey, or None
|
|
67
|
+
audio: (channels, samples) float32, or None
|
|
68
|
+
video_seconds / audio_seconds: each stream's own duration, or None
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
def __init__(
|
|
72
|
+
self,
|
|
73
|
+
thumbs,
|
|
74
|
+
audio,
|
|
75
|
+
sample_rate,
|
|
76
|
+
fps,
|
|
77
|
+
video_seconds=None,
|
|
78
|
+
audio_seconds=None,
|
|
79
|
+
shots=None,
|
|
80
|
+
):
|
|
81
|
+
self.thumbs = thumbs
|
|
82
|
+
self.audio = audio
|
|
83
|
+
self.sample_rate = sample_rate
|
|
84
|
+
self.fps = fps
|
|
85
|
+
self.video_seconds = video_seconds
|
|
86
|
+
self.audio_seconds = audio_seconds
|
|
87
|
+
self.shots = shots
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def frame_count(self):
|
|
91
|
+
return 0 if self.thumbs is None else int(self.thumbs.shape[0])
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _stream_seconds(stream):
|
|
95
|
+
if stream is None or stream.duration is None or stream.time_base is None:
|
|
96
|
+
return None
|
|
97
|
+
return float(stream.duration * stream.time_base)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _as_float_samples(samples):
|
|
101
|
+
if samples.dtype.kind == "u":
|
|
102
|
+
iinfo = numpy.iinfo(samples.dtype)
|
|
103
|
+
half = (iinfo.max + 1) / 2
|
|
104
|
+
return (samples.astype(numpy.float32) - half) / half
|
|
105
|
+
if samples.dtype.kind == "i":
|
|
106
|
+
return samples.astype(numpy.float32) / numpy.iinfo(samples.dtype).max
|
|
107
|
+
return samples.astype(numpy.float32)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def read_media(path):
|
|
111
|
+
"""Stream a video file once into a Media.
|
|
112
|
+
|
|
113
|
+
Each decoded frame is reduced to a grey thumbnail as it arrives and then
|
|
114
|
+
dropped, so memory holds thumbnails, not pictures. The soundtrack is
|
|
115
|
+
kept whole (the probes cut windows out of it anywhere) and trimmed to
|
|
116
|
+
the audio stream's reported duration.
|
|
117
|
+
"""
|
|
118
|
+
import av
|
|
119
|
+
|
|
120
|
+
from ..media_info import _as_frame_samples
|
|
121
|
+
|
|
122
|
+
thumbs = []
|
|
123
|
+
chunks = []
|
|
124
|
+
with av.open(path) as container:
|
|
125
|
+
video = container.streams.video[0] if container.streams.video else None
|
|
126
|
+
audio = container.streams.audio[0] if container.streams.audio else None
|
|
127
|
+
if video is None and audio is None:
|
|
128
|
+
raise ValueError(f"{path} has neither a video nor an audio stream")
|
|
129
|
+
fps = float(video.average_rate) if video and video.average_rate else None
|
|
130
|
+
video_seconds = _stream_seconds(video)
|
|
131
|
+
audio_seconds = _stream_seconds(audio)
|
|
132
|
+
channels = int(audio.channels) if audio is not None else 0
|
|
133
|
+
streams = [s for s in (video, audio) if s is not None]
|
|
134
|
+
for frame in container.decode(*streams):
|
|
135
|
+
if isinstance(frame, av.VideoFrame):
|
|
136
|
+
thumbs.append(
|
|
137
|
+
frame.reformat(
|
|
138
|
+
width=THUMB_WIDTH, height=THUMB_HEIGHT, format="gray"
|
|
139
|
+
).to_ndarray()[:THUMB_HEIGHT, :THUMB_WIDTH]
|
|
140
|
+
)
|
|
141
|
+
elif isinstance(frame, av.AudioFrame):
|
|
142
|
+
samples = _as_float_samples(frame.to_ndarray())
|
|
143
|
+
chunks.append(_as_frame_samples(samples, channels))
|
|
144
|
+
sample_rate = int(audio.rate) if audio is not None else None
|
|
145
|
+
|
|
146
|
+
waveform = None
|
|
147
|
+
if chunks:
|
|
148
|
+
waveform = numpy.ascontiguousarray(numpy.concatenate(chunks, axis=0).T)
|
|
149
|
+
if audio_seconds is not None:
|
|
150
|
+
waveform = waveform[:, : int(round(audio_seconds * sample_rate))]
|
|
151
|
+
if video_seconds is None and fps and thumbs:
|
|
152
|
+
video_seconds = len(thumbs) / fps
|
|
153
|
+
return Media(
|
|
154
|
+
numpy.stack(thumbs) if thumbs else None,
|
|
155
|
+
waveform,
|
|
156
|
+
sample_rate,
|
|
157
|
+
fps,
|
|
158
|
+
video_seconds,
|
|
159
|
+
audio_seconds,
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _thumb(frame):
|
|
164
|
+
"""One in-memory frame (PIL image, or an HxWxC array or tensor, uint8 or
|
|
165
|
+
float in [0, 1]) as a grey thumbnail."""
|
|
166
|
+
from PIL import Image
|
|
167
|
+
|
|
168
|
+
if not isinstance(frame, Image.Image):
|
|
169
|
+
array = frame
|
|
170
|
+
if hasattr(array, "detach"):
|
|
171
|
+
array = array.detach().cpu().numpy()
|
|
172
|
+
array = numpy.asarray(array)
|
|
173
|
+
if array.dtype.kind == "f":
|
|
174
|
+
array = numpy.clip(array * 255.0, 0, 255)
|
|
175
|
+
array = array.astype(numpy.uint8)
|
|
176
|
+
if array.ndim == 3 and array.shape[-1] == 1:
|
|
177
|
+
array = array[..., 0]
|
|
178
|
+
frame = Image.fromarray(array)
|
|
179
|
+
return numpy.asarray(
|
|
180
|
+
frame.convert("L").resize((THUMB_WIDTH, THUMB_HEIGHT), Image.BILINEAR),
|
|
181
|
+
dtype=numpy.uint8,
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _in_memory_thumbs(frames):
|
|
186
|
+
"""Thumbnails of an AudioVideo's frames, one frame at a time: a list, an
|
|
187
|
+
(N, H, W, C) array, or a SegmentedFrames replaying (N, H, W, 3) chunks."""
|
|
188
|
+
from ..pipeline_processors.chain import SegmentedFrames
|
|
189
|
+
|
|
190
|
+
thumbs = []
|
|
191
|
+
if isinstance(frames, SegmentedFrames):
|
|
192
|
+
for chunk in frames:
|
|
193
|
+
for frame in chunk:
|
|
194
|
+
thumbs.append(_thumb(frame))
|
|
195
|
+
else:
|
|
196
|
+
for frame in frames:
|
|
197
|
+
thumbs.append(_thumb(frame))
|
|
198
|
+
return numpy.stack(thumbs) if thumbs else None
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _file_path(video):
|
|
202
|
+
"""The validated file a probe's 'video' names, or None for an in-memory
|
|
203
|
+
clip: a path, or the VideoFileReference an asset:/output: reference or a
|
|
204
|
+
literal path is realized to (dw/arguments.py, #387)."""
|
|
205
|
+
from ..locations import validate_media_path
|
|
206
|
+
from .video_utils import VideoFileReference
|
|
207
|
+
|
|
208
|
+
if isinstance(video, VideoFileReference):
|
|
209
|
+
video = video.path
|
|
210
|
+
if isinstance(video, str):
|
|
211
|
+
return validate_media_path(video, None, "a video to assess")
|
|
212
|
+
return None
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def media_from(video):
|
|
216
|
+
"""A Media from a path, a VideoFileReference or an in-memory AudioVideo."""
|
|
217
|
+
path = _file_path(video)
|
|
218
|
+
if path is not None:
|
|
219
|
+
return read_media(path)
|
|
220
|
+
if hasattr(video, "frames") or hasattr(video, "audio"):
|
|
221
|
+
from .audio_utils import as_channels_samples
|
|
222
|
+
|
|
223
|
+
audio = getattr(video, "audio", None)
|
|
224
|
+
waveform = None if audio is None else as_channels_samples(audio)
|
|
225
|
+
thumbs = (
|
|
226
|
+
_in_memory_thumbs(video.frames)
|
|
227
|
+
if getattr(video, "frames", None) is not None
|
|
228
|
+
else None
|
|
229
|
+
)
|
|
230
|
+
sample_rate = getattr(video, "sample_rate", None)
|
|
231
|
+
fps = getattr(video, "fps", None)
|
|
232
|
+
media = Media(
|
|
233
|
+
thumbs,
|
|
234
|
+
waveform,
|
|
235
|
+
sample_rate,
|
|
236
|
+
fps,
|
|
237
|
+
video_seconds=(thumbs.shape[0] / fps)
|
|
238
|
+
if thumbs is not None and fps
|
|
239
|
+
else None,
|
|
240
|
+
audio_seconds=(
|
|
241
|
+
waveform.shape[1] / sample_rate
|
|
242
|
+
if waveform is not None and sample_rate
|
|
243
|
+
else None
|
|
244
|
+
),
|
|
245
|
+
shots=getattr(video, "shots", None),
|
|
246
|
+
)
|
|
247
|
+
return media
|
|
248
|
+
raise ValueError(
|
|
249
|
+
"a probe takes a stored video (asset:, output: or a path) or the "
|
|
250
|
+
f"video an earlier step returned, not {type(video).__name__} - a URL "
|
|
251
|
+
"downloads as bare frames with no soundtrack to measure"
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def resolve_shots(video, media, shots=None):
|
|
256
|
+
"""The shot records to measure against, and where they came from."""
|
|
257
|
+
if shots:
|
|
258
|
+
return [dict(shot) for shot in shots], "argument"
|
|
259
|
+
if media.shots:
|
|
260
|
+
return [dict(shot) for shot in media.shots], "artifact"
|
|
261
|
+
path = _file_path(video)
|
|
262
|
+
if path is not None:
|
|
263
|
+
from ..runs import shots_beside
|
|
264
|
+
|
|
265
|
+
recorded = shots_beside(path)
|
|
266
|
+
if recorded:
|
|
267
|
+
return [dict(shot) for shot in recorded], "manifest"
|
|
268
|
+
return None, "none"
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _whole_file_shot(media):
|
|
272
|
+
samples = media.audio.shape[1] if media.audio is not None else None
|
|
273
|
+
return {
|
|
274
|
+
"name": "whole",
|
|
275
|
+
"start_frame": 0,
|
|
276
|
+
"num_frames": media.frame_count,
|
|
277
|
+
"start_sample": 0 if samples is not None else None,
|
|
278
|
+
"num_samples": samples,
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _sample_span(shot, media):
|
|
283
|
+
"""A shot's (start, count, source) on the soundtrack: recorded, else
|
|
284
|
+
derived from its frames."""
|
|
285
|
+
start = shot.get("start_sample")
|
|
286
|
+
count = shot.get("num_samples")
|
|
287
|
+
if start is not None and count is not None:
|
|
288
|
+
return int(start), int(count), "recorded"
|
|
289
|
+
if media.fps and media.sample_rate:
|
|
290
|
+
scale = media.sample_rate / media.fps
|
|
291
|
+
return (
|
|
292
|
+
int(round(shot.get("start_frame", 0) * scale)),
|
|
293
|
+
int(round(shot.get("num_frames", 0) * scale)),
|
|
294
|
+
"derived",
|
|
295
|
+
)
|
|
296
|
+
return None, None, None
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _db(value):
|
|
300
|
+
return None if value is None or value <= _SILENCE else 20.0 * math.log10(value)
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _rms(window):
|
|
304
|
+
if window is None or window.size == 0:
|
|
305
|
+
return None
|
|
306
|
+
return math.sqrt(float(numpy.mean(numpy.square(window, dtype=numpy.float64))))
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _peak(window):
|
|
310
|
+
if window is None or window.size == 0:
|
|
311
|
+
return None
|
|
312
|
+
return float(numpy.max(numpy.abs(window)))
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def _clip(media, start, end):
|
|
316
|
+
total = media.audio.shape[1]
|
|
317
|
+
start = max(0, min(total, int(start)))
|
|
318
|
+
end = max(start, min(total, int(end)))
|
|
319
|
+
return media.audio[:, start:end]
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _round(value, places=2):
|
|
323
|
+
return None if value is None else round(float(value), places)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _findings(probe, record, at, skip=()):
|
|
327
|
+
found = []
|
|
328
|
+
for rule in rules_for(probe):
|
|
329
|
+
if rule["name"] in skip or rule["field"] not in record:
|
|
330
|
+
continue
|
|
331
|
+
if crosses(rule, record[rule["field"]]):
|
|
332
|
+
found.append(finding(rule, record[rule["field"]], at))
|
|
333
|
+
return found
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def _span_overrun(media, shot):
|
|
337
|
+
"""How far past the file `shot`'s frames or samples reach, or None when
|
|
338
|
+
it fits. `_clip` silently clamps an out-of-range window to the file
|
|
339
|
+
(#425): a shot record with `start_frame + num_frames` past the video's
|
|
340
|
+
real length, or explicit `start_sample + num_samples` past the
|
|
341
|
+
soundtrack's, is measured over a window shorter than the caller asked
|
|
342
|
+
for and nothing says so unless this is checked first."""
|
|
343
|
+
over_frames = None
|
|
344
|
+
if media.thumbs is not None:
|
|
345
|
+
start_frame = shot.get("start_frame", 0)
|
|
346
|
+
num_frames = shot.get("num_frames")
|
|
347
|
+
if isinstance(start_frame, (int, float)) and not isinstance(start_frame, bool):
|
|
348
|
+
if isinstance(num_frames, (int, float)) and not isinstance(
|
|
349
|
+
num_frames, bool
|
|
350
|
+
):
|
|
351
|
+
end = int(start_frame) + int(num_frames)
|
|
352
|
+
if end > media.frame_count:
|
|
353
|
+
over_frames = end - media.frame_count
|
|
354
|
+
over_samples = None
|
|
355
|
+
if media.audio is not None:
|
|
356
|
+
start_sample, num_samples = shot.get("start_sample"), shot.get("num_samples")
|
|
357
|
+
if isinstance(start_sample, (int, float)) and not isinstance(
|
|
358
|
+
start_sample, bool
|
|
359
|
+
):
|
|
360
|
+
if isinstance(num_samples, (int, float)) and not isinstance(
|
|
361
|
+
num_samples, bool
|
|
362
|
+
):
|
|
363
|
+
total = media.audio.shape[1]
|
|
364
|
+
end = int(start_sample) + int(num_samples)
|
|
365
|
+
if end > total:
|
|
366
|
+
over_samples = end - total
|
|
367
|
+
if over_frames is None and over_samples is None:
|
|
368
|
+
return None
|
|
369
|
+
return {"frames": over_frames, "samples": over_samples}
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _shot_span_findings(probe, records, media):
|
|
373
|
+
"""Findings (and a run warning) for every shot record whose declared
|
|
374
|
+
span reaches past the file - clipped silently otherwise (#425)."""
|
|
375
|
+
findings = []
|
|
376
|
+
overrun_names = []
|
|
377
|
+
for shot in records or []:
|
|
378
|
+
overrun = _span_overrun(media, shot)
|
|
379
|
+
if overrun is None:
|
|
380
|
+
continue
|
|
381
|
+
detail = (
|
|
382
|
+
f"{overrun['frames']} frame(s)"
|
|
383
|
+
if overrun["frames"] is not None
|
|
384
|
+
else f"{overrun['samples']} sample(s)"
|
|
385
|
+
)
|
|
386
|
+
findings.append(
|
|
387
|
+
{
|
|
388
|
+
"rule": "shot_span_overrun",
|
|
389
|
+
"severity": "warning",
|
|
390
|
+
"at": {"shot": shot.get("name")},
|
|
391
|
+
"value": overrun,
|
|
392
|
+
"threshold": 0,
|
|
393
|
+
"says": f"shot {shot.get('name')!r} reaches {detail} past the file's end",
|
|
394
|
+
}
|
|
395
|
+
)
|
|
396
|
+
overrun_names.append(shot.get("name"))
|
|
397
|
+
if overrun_names:
|
|
398
|
+
emit_warning(
|
|
399
|
+
f"{probe}: shot record(s) {', '.join(str(name) for name in overrun_names)} "
|
|
400
|
+
"reach past the file's end and were silently clipped to it",
|
|
401
|
+
kind="shot_span_overrun",
|
|
402
|
+
probe=probe,
|
|
403
|
+
shots=overrun_names,
|
|
404
|
+
)
|
|
405
|
+
return findings
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def _answer(probe, measurements, findings, shots_source, shot_dependent=()):
|
|
409
|
+
"""A probe's answer, with `rules_applied` cut down to the rules that
|
|
410
|
+
actually ran. `shot_dependent` names (or `True` for all of the probe's
|
|
411
|
+
rules) the ones that only mean anything measured shot against shot; with
|
|
412
|
+
no shot boundaries at all (`shots_source == "none"`) those are reported
|
|
413
|
+
as `rules_skipped` instead of `rules_applied`, and a run warning says why
|
|
414
|
+
- without this a shotless file (an asset kept before #393, an upload, a
|
|
415
|
+
cut joined outside dw) read as a clean pass with nothing measured (#394).
|
|
416
|
+
"""
|
|
417
|
+
names = [rule["name"] for rule in rules_for(probe)]
|
|
418
|
+
dependent = set(names) if shot_dependent is True else set(shot_dependent)
|
|
419
|
+
applied, skipped = names, []
|
|
420
|
+
if shots_source == "none" and dependent:
|
|
421
|
+
applied = [name for name in names if name not in dependent]
|
|
422
|
+
skipped = [
|
|
423
|
+
{"rule": name, "reason": "no shot boundaries"}
|
|
424
|
+
for name in names
|
|
425
|
+
if name in dependent
|
|
426
|
+
]
|
|
427
|
+
emit_warning(
|
|
428
|
+
f"{probe} found no shot boundaries for this file, so "
|
|
429
|
+
f"{', '.join(sorted(dependent))} could not be measured - pass "
|
|
430
|
+
"shots= to supply them",
|
|
431
|
+
kind="no_shot_boundaries",
|
|
432
|
+
probe=probe,
|
|
433
|
+
)
|
|
434
|
+
return {
|
|
435
|
+
**measurements,
|
|
436
|
+
"findings": findings,
|
|
437
|
+
"rules_applied": applied,
|
|
438
|
+
"rules_skipped": skipped,
|
|
439
|
+
"shots_source": shots_source,
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def analyze_shots(video, shots=None):
|
|
444
|
+
"""Task command: each shot's level and spectral balance, and how far
|
|
445
|
+
apart the shots sit.
|
|
446
|
+
|
|
447
|
+
Args:
|
|
448
|
+
video: A video file's path, or the video an earlier step returned
|
|
449
|
+
shots: Shot records to measure by, overriding any the video carries
|
|
450
|
+
|
|
451
|
+
Returns:
|
|
452
|
+
{shots: [{name, start_frame, num_frames, peak_dbfs, rms_dbfs, crest_db, low_dbfs, mid_dbfs,
|
|
453
|
+
high_dbfs, samples}], rms_range_db, findings, rules_applied,
|
|
454
|
+
rules_skipped, shots_source}
|
|
455
|
+
"""
|
|
456
|
+
media = media_from(video)
|
|
457
|
+
return shots_answer(media, *resolve_shots(video, media, shots))
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
def shots_answer(media, records, source):
|
|
461
|
+
"""`analyze_shots` over an already-read Media and resolved shots."""
|
|
462
|
+
from .audio_utils import _spectral_balance
|
|
463
|
+
|
|
464
|
+
if media.audio is None:
|
|
465
|
+
return _answer(
|
|
466
|
+
"analyze_shots",
|
|
467
|
+
{"shots": [], "rms_range_db": None, "has_audio": False},
|
|
468
|
+
_shot_span_findings("analyze_shots", records, media),
|
|
469
|
+
source,
|
|
470
|
+
shot_dependent={"shot_level_spread"},
|
|
471
|
+
)
|
|
472
|
+
records = records or [_whole_file_shot(media)]
|
|
473
|
+
overrun_findings = _shot_span_findings("analyze_shots", records, media)
|
|
474
|
+
|
|
475
|
+
measured = []
|
|
476
|
+
for shot in records:
|
|
477
|
+
start, count, samples_source = _sample_span(shot, media)
|
|
478
|
+
window = _clip(media, start, start + count) if start is not None else None
|
|
479
|
+
peak = _db(_peak(window))
|
|
480
|
+
rms = _db(_rms(window))
|
|
481
|
+
balance = (
|
|
482
|
+
_spectral_balance(window, media.sample_rate)
|
|
483
|
+
if window is not None and window.size
|
|
484
|
+
else {"low_dbfs": None, "mid_dbfs": None, "high_dbfs": None}
|
|
485
|
+
)
|
|
486
|
+
measured.append(
|
|
487
|
+
{
|
|
488
|
+
"name": shot.get("name"),
|
|
489
|
+
"start_frame": shot.get("start_frame"),
|
|
490
|
+
"num_frames": shot.get("num_frames"),
|
|
491
|
+
"peak_dbfs": _round(peak),
|
|
492
|
+
"rms_dbfs": _round(rms),
|
|
493
|
+
"crest_db": _round(None if peak is None or rms is None else peak - rms),
|
|
494
|
+
**{key: _round(value) for key, value in balance.items()},
|
|
495
|
+
"samples": samples_source,
|
|
496
|
+
}
|
|
497
|
+
)
|
|
498
|
+
|
|
499
|
+
voiced = [shot for shot in measured if shot["rms_dbfs"] is not None]
|
|
500
|
+
rms_range = None
|
|
501
|
+
at = None
|
|
502
|
+
if len(voiced) > 1:
|
|
503
|
+
loudest = max(voiced, key=lambda shot: shot["rms_dbfs"])
|
|
504
|
+
quietest = min(voiced, key=lambda shot: shot["rms_dbfs"])
|
|
505
|
+
rms_range = _round(loudest["rms_dbfs"] - quietest["rms_dbfs"])
|
|
506
|
+
at = {"between": [loudest["name"], quietest["name"]]}
|
|
507
|
+
elif voiced:
|
|
508
|
+
rms_range = 0.0
|
|
509
|
+
answer = {"shots": measured, "rms_range_db": rms_range, "has_audio": True}
|
|
510
|
+
return _answer(
|
|
511
|
+
"analyze_shots",
|
|
512
|
+
answer,
|
|
513
|
+
overrun_findings + (_findings("analyze_shots", answer, at) if at else []),
|
|
514
|
+
source,
|
|
515
|
+
shot_dependent={"shot_level_spread"},
|
|
516
|
+
)
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _band_shares(window, sample_rate):
|
|
520
|
+
from .audio_utils import _spectral_balance
|
|
521
|
+
|
|
522
|
+
if window is None or window.size == 0:
|
|
523
|
+
return None
|
|
524
|
+
bands = _spectral_balance(window, sample_rate)
|
|
525
|
+
energies = {
|
|
526
|
+
key: (10.0 ** (value / 10.0) if value is not None else 0.0)
|
|
527
|
+
for key, value in bands.items()
|
|
528
|
+
}
|
|
529
|
+
total = sum(energies.values())
|
|
530
|
+
if total <= 0.0:
|
|
531
|
+
return None
|
|
532
|
+
return {key: value / total for key, value in energies.items()}
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def _shot_rms(media, shot):
|
|
536
|
+
"""A shot's RMS level over its whole sample span, in dBFS, or None."""
|
|
537
|
+
start, count, _source = _sample_span(shot, media)
|
|
538
|
+
if start is None:
|
|
539
|
+
return None
|
|
540
|
+
return _db(_rms(_clip(media, start, start + count)))
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _seam_audio(media, before_end, after_start, previous_rms, next_rms):
|
|
544
|
+
"""Audio measurements at a seam: the edge windows end at `before_end`
|
|
545
|
+
and open at `after_start` (the same sample at a cut, either side of the
|
|
546
|
+
fade at a dissolve), and the join is what lies between them - or the
|
|
547
|
+
FLOOR_WINDOW centred on the cut. The level step is between the two
|
|
548
|
+
shots' own levels, `previous_rms` and `next_rms`."""
|
|
549
|
+
rate = media.sample_rate
|
|
550
|
+
level = int(round(LEVEL_WINDOW * rate))
|
|
551
|
+
before = _clip(media, before_end - level, before_end)
|
|
552
|
+
after = _clip(media, after_start, after_start + level)
|
|
553
|
+
before_rms = _db(_rms(before))
|
|
554
|
+
after_rms = _db(_rms(after))
|
|
555
|
+
|
|
556
|
+
centre = (before_end + after_start) // 2
|
|
557
|
+
half_floor = max(1, int(round(FLOOR_WINDOW * rate / 2)))
|
|
558
|
+
join = (
|
|
559
|
+
_clip(media, before_end, after_start)
|
|
560
|
+
if after_start - before_end > 2 * half_floor
|
|
561
|
+
else _clip(media, centre - half_floor, centre + half_floor)
|
|
562
|
+
)
|
|
563
|
+
floor = _db(_rms(join))
|
|
564
|
+
|
|
565
|
+
half_click = max(1, int(round(CLICK_WINDOW * rate / 2)))
|
|
566
|
+
neighbour = int(round(CLICK_NEIGHBOUR_WINDOW * rate))
|
|
567
|
+
click_peak = _peak(_clip(media, centre - half_click, centre + half_click))
|
|
568
|
+
neighbour_peak = max(
|
|
569
|
+
_peak(_clip(media, centre - half_click - neighbour, centre - half_click))
|
|
570
|
+
or 0.0,
|
|
571
|
+
_peak(_clip(media, centre + half_click, centre + half_click + neighbour))
|
|
572
|
+
or 0.0,
|
|
573
|
+
)
|
|
574
|
+
click = None
|
|
575
|
+
if click_peak is not None and click_peak > _SILENCE:
|
|
576
|
+
click = (
|
|
577
|
+
CLICK_CAP_DB
|
|
578
|
+
if neighbour_peak <= _SILENCE
|
|
579
|
+
else min(CLICK_CAP_DB, 20.0 * math.log10(click_peak / neighbour_peak))
|
|
580
|
+
)
|
|
581
|
+
|
|
582
|
+
shares_before = _band_shares(before, rate)
|
|
583
|
+
shares_after = _band_shares(after, rate)
|
|
584
|
+
spectral_shift = (
|
|
585
|
+
sum(abs(shares_after[key] - shares_before[key]) for key in shares_before) / 2.0
|
|
586
|
+
if shares_before and shares_after
|
|
587
|
+
else None
|
|
588
|
+
)
|
|
589
|
+
return {
|
|
590
|
+
"before_rms_dbfs": _round(before_rms),
|
|
591
|
+
"after_rms_dbfs": _round(after_rms),
|
|
592
|
+
"before_shot_rms_dbfs": _round(previous_rms),
|
|
593
|
+
"after_shot_rms_dbfs": _round(next_rms),
|
|
594
|
+
"level_step_db": _round(
|
|
595
|
+
None
|
|
596
|
+
if previous_rms is None or next_rms is None
|
|
597
|
+
else abs(next_rms - previous_rms)
|
|
598
|
+
),
|
|
599
|
+
"floor_dbfs": _round(floor),
|
|
600
|
+
"click_db": _round(click),
|
|
601
|
+
"spectral_shift": _round(spectral_shift, 3),
|
|
602
|
+
}
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
def _typical_delta(thumbs, start, end):
|
|
606
|
+
"""The TYPICAL_DELTA_PERCENTILE of frame-to-frame change inside
|
|
607
|
+
thumbs[start:end], or None for fewer than two frames."""
|
|
608
|
+
span = thumbs[max(0, start) : max(0, end)]
|
|
609
|
+
if span.shape[0] < 2:
|
|
610
|
+
return None
|
|
611
|
+
deltas = numpy.abs(numpy.diff(span.astype(numpy.int16), axis=0)).mean(axis=(1, 2))
|
|
612
|
+
return float(numpy.percentile(deltas, TYPICAL_DELTA_PERCENTILE))
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
def _seam_video(media, previous, shot, fade):
|
|
616
|
+
"""Picture measurements at the seam `shot` opens: the largest single-frame
|
|
617
|
+
change across it (at a dissolve, across the whole fade - each step of a
|
|
618
|
+
fade is small, which is what a dissolve is), against the larger of the
|
|
619
|
+
two shots' own typical change, floored."""
|
|
620
|
+
thumbs = media.thumbs
|
|
621
|
+
start = int(shot.get("start_frame", 0))
|
|
622
|
+
first = max(0, start - 1)
|
|
623
|
+
last = min(thumbs.shape[0] - 1, start + max(0, fade - 1) if fade else start)
|
|
624
|
+
if last <= first:
|
|
625
|
+
return {"frame_delta": None, "typical_delta": None, "jump_ratio": None}
|
|
626
|
+
across = numpy.abs(
|
|
627
|
+
numpy.diff(thumbs[first : last + 1].astype(numpy.int16), axis=0)
|
|
628
|
+
).mean(axis=(1, 2))
|
|
629
|
+
frame_delta = float(across.max())
|
|
630
|
+
typical = [
|
|
631
|
+
_typical_delta(
|
|
632
|
+
thumbs,
|
|
633
|
+
int(previous.get("start_frame", 0))
|
|
634
|
+
+ int(previous.get("overlap_frames") or 0),
|
|
635
|
+
start,
|
|
636
|
+
),
|
|
637
|
+
_typical_delta(thumbs, start + fade, start + int(shot.get("num_frames", 0))),
|
|
638
|
+
]
|
|
639
|
+
typical = max(
|
|
640
|
+
[TYPICAL_DELTA_FLOOR] + [value for value in typical if value is not None]
|
|
641
|
+
)
|
|
642
|
+
return {
|
|
643
|
+
"frame_delta": _round(frame_delta),
|
|
644
|
+
"typical_delta": _round(typical),
|
|
645
|
+
"jump_ratio": _round(frame_delta / typical),
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def analyze_seams(video, shots=None):
|
|
650
|
+
"""Task command: measure every seam between shots, audio and picture.
|
|
651
|
+
|
|
652
|
+
Args:
|
|
653
|
+
video: A video file's path, or the video an earlier step returned
|
|
654
|
+
shots: Shot records to measure by, overriding any the video carries.
|
|
655
|
+
A shot marked `hard_cut: true` opens a seam meant as a cut
|
|
656
|
+
|
|
657
|
+
Returns:
|
|
658
|
+
{seams: [{seam, between, seconds, kind, level_step_db,
|
|
659
|
+
before_shot_rms_dbfs, after_shot_rms_dbfs, floor_dbfs, click_db,
|
|
660
|
+
spectral_shift, before_rms_dbfs, after_rms_dbfs,
|
|
661
|
+
frame_delta, typical_delta, jump_ratio}], findings, rules_applied,
|
|
662
|
+
rules_skipped, shots_source}
|
|
663
|
+
"""
|
|
664
|
+
media = media_from(video)
|
|
665
|
+
return seams_answer(media, *resolve_shots(video, media, shots))
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
def seams_answer(media, records, source):
|
|
669
|
+
"""`analyze_seams` over an already-read Media and resolved shots."""
|
|
670
|
+
if not records or len(records) < 2:
|
|
671
|
+
return _answer(
|
|
672
|
+
"analyze_seams",
|
|
673
|
+
{"seams": []},
|
|
674
|
+
_shot_span_findings("analyze_seams", records, media),
|
|
675
|
+
source,
|
|
676
|
+
shot_dependent=True,
|
|
677
|
+
)
|
|
678
|
+
|
|
679
|
+
seams = []
|
|
680
|
+
findings = _shot_span_findings("analyze_seams", records, media)
|
|
681
|
+
for index in range(1, len(records)):
|
|
682
|
+
previous, shot = records[index - 1], records[index]
|
|
683
|
+
fade = int(shot.get("overlap_frames") or 0)
|
|
684
|
+
start_frame = int(shot.get("start_frame", 0))
|
|
685
|
+
seam_frame = start_frame + fade / 2.0
|
|
686
|
+
seconds = seam_frame / media.fps if media.fps else None
|
|
687
|
+
record = {
|
|
688
|
+
"seam": index,
|
|
689
|
+
"between": [previous.get("name"), shot.get("name")],
|
|
690
|
+
"seconds": _round(seconds, 3),
|
|
691
|
+
"kind": "dissolve" if fade else "cut",
|
|
692
|
+
"hard_cut": bool(shot.get("hard_cut")),
|
|
693
|
+
}
|
|
694
|
+
skip = set()
|
|
695
|
+
if media.audio is not None and media.sample_rate:
|
|
696
|
+
start, _count, _source = _sample_span(shot, media)
|
|
697
|
+
if start is not None:
|
|
698
|
+
fade_samples = (
|
|
699
|
+
int(round(fade / media.fps * media.sample_rate))
|
|
700
|
+
if fade and media.fps
|
|
701
|
+
else 0
|
|
702
|
+
)
|
|
703
|
+
record.update(
|
|
704
|
+
_seam_audio(
|
|
705
|
+
media,
|
|
706
|
+
start,
|
|
707
|
+
start + fade_samples,
|
|
708
|
+
_shot_rms(media, previous),
|
|
709
|
+
_shot_rms(media, shot),
|
|
710
|
+
)
|
|
711
|
+
)
|
|
712
|
+
if (
|
|
713
|
+
record["before_rms_dbfs"] is None
|
|
714
|
+
or record["after_rms_dbfs"] is None
|
|
715
|
+
or record["before_rms_dbfs"] <= HOLE_VOICED_DBFS
|
|
716
|
+
or record["after_rms_dbfs"] <= HOLE_VOICED_DBFS
|
|
717
|
+
):
|
|
718
|
+
skip.add("seam_hole")
|
|
719
|
+
else:
|
|
720
|
+
skip.add("seam_hole")
|
|
721
|
+
if media.thumbs is not None:
|
|
722
|
+
record.update(_seam_video(media, previous, shot, fade))
|
|
723
|
+
if record["hard_cut"]:
|
|
724
|
+
skip.add("seam_frame_jump")
|
|
725
|
+
seams.append(record)
|
|
726
|
+
findings.extend(
|
|
727
|
+
_findings(
|
|
728
|
+
"analyze_seams",
|
|
729
|
+
record,
|
|
730
|
+
{
|
|
731
|
+
"seam": index,
|
|
732
|
+
"between": record["between"],
|
|
733
|
+
"seconds": record["seconds"],
|
|
734
|
+
},
|
|
735
|
+
skip,
|
|
736
|
+
)
|
|
737
|
+
)
|
|
738
|
+
return _answer(
|
|
739
|
+
"analyze_seams", {"seams": seams}, findings, source, shot_dependent=True
|
|
740
|
+
)
|
|
741
|
+
|
|
742
|
+
|
|
743
|
+
def analyze_sync_drift(video, shots=None):
|
|
744
|
+
"""Task command: how far the soundtrack sits from the picture, shot by
|
|
745
|
+
shot and over the whole file.
|
|
746
|
+
|
|
747
|
+
Args:
|
|
748
|
+
video: A video file's path, or the video an earlier step returned
|
|
749
|
+
shots: Shot records to measure by, overriding any the video carries
|
|
750
|
+
|
|
751
|
+
Returns:
|
|
752
|
+
{shots: [{name, start_offset_ms, end_offset_ms}], max_offset_ms,
|
|
753
|
+
video_seconds, audio_seconds, length_delta_ms, findings,
|
|
754
|
+
rules_applied, rules_skipped, shots_source}
|
|
755
|
+
"""
|
|
756
|
+
media = media_from(video)
|
|
757
|
+
return sync_drift_answer(media, *resolve_shots(video, media, shots))
|
|
758
|
+
|
|
759
|
+
|
|
760
|
+
def sync_drift_answer(media, records, source):
|
|
761
|
+
"""`analyze_sync_drift` over an already-read Media and resolved shots."""
|
|
762
|
+
measured = []
|
|
763
|
+
findings = _shot_span_findings("analyze_sync_drift", records, media)
|
|
764
|
+
rate = media.sample_rate
|
|
765
|
+
fps = media.fps
|
|
766
|
+
for shot in records or []:
|
|
767
|
+
start, count = shot.get("start_sample"), shot.get("num_samples")
|
|
768
|
+
if start is None or count is None or not rate or not fps:
|
|
769
|
+
continue
|
|
770
|
+
start_frame = int(shot.get("start_frame", 0))
|
|
771
|
+
end_frame = start_frame + int(shot.get("num_frames", 0))
|
|
772
|
+
record = {
|
|
773
|
+
"name": shot.get("name"),
|
|
774
|
+
"start_offset_ms": _round((int(start) / rate - start_frame / fps) * 1000.0),
|
|
775
|
+
"end_offset_ms": _round(
|
|
776
|
+
((int(start) + int(count)) / rate - end_frame / fps) * 1000.0
|
|
777
|
+
),
|
|
778
|
+
}
|
|
779
|
+
measured.append(record)
|
|
780
|
+
findings.extend(
|
|
781
|
+
_findings(
|
|
782
|
+
"analyze_sync_drift",
|
|
783
|
+
record,
|
|
784
|
+
{
|
|
785
|
+
"shot": record["name"],
|
|
786
|
+
"seconds": _round(end_frame / fps, 3),
|
|
787
|
+
},
|
|
788
|
+
)
|
|
789
|
+
)
|
|
790
|
+
|
|
791
|
+
length_delta = None
|
|
792
|
+
if media.audio_seconds is not None and media.video_seconds is not None:
|
|
793
|
+
length_delta = _round((media.audio_seconds - media.video_seconds) * 1000.0)
|
|
794
|
+
answer = {
|
|
795
|
+
"shots": measured,
|
|
796
|
+
"max_offset_ms": (
|
|
797
|
+
max((record["end_offset_ms"] for record in measured), key=abs)
|
|
798
|
+
if measured
|
|
799
|
+
else None
|
|
800
|
+
),
|
|
801
|
+
"video_seconds": _round(media.video_seconds, 4),
|
|
802
|
+
"audio_seconds": _round(media.audio_seconds, 4),
|
|
803
|
+
"length_delta_ms": length_delta,
|
|
804
|
+
}
|
|
805
|
+
findings.extend(
|
|
806
|
+
_findings(
|
|
807
|
+
"analyze_sync_drift",
|
|
808
|
+
{"length_delta_ms": length_delta},
|
|
809
|
+
{"file": True},
|
|
810
|
+
)
|
|
811
|
+
)
|
|
812
|
+
return _answer("analyze_sync_drift", answer, findings, source)
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
__all__ = [
|
|
816
|
+
"Media",
|
|
817
|
+
"analyze_seams",
|
|
818
|
+
"analyze_shots",
|
|
819
|
+
"analyze_sync_drift",
|
|
820
|
+
"media_from",
|
|
821
|
+
"read_media",
|
|
822
|
+
"resolve_shots",
|
|
823
|
+
"seams_answer",
|
|
824
|
+
"shots_answer",
|
|
825
|
+
"sync_drift_answer",
|
|
826
|
+
]
|