diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/loudness.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""BS.1770 loudness measurement, shared by the media probe and the audio
|
|
2
|
+
tasks so both read a level the same way.
|
|
3
|
+
|
|
4
|
+
Peak and loudness are different quantities - a sparse voice-over and a dense
|
|
5
|
+
score can share a peak and still sit tens of dB apart in how loud they sound,
|
|
6
|
+
because a peak is one sample and loudness is measured over the whole track
|
|
7
|
+
(#361). `integrated_lufs` is the BS.1770 integrated measure pyloudnorm
|
|
8
|
+
implements; `true_peak_dbfs` is the inter-sample peak BS.1770 defines
|
|
9
|
+
alongside it - a sample-peak reading can miss a peak that only appears
|
|
10
|
+
between samples, which is what an encoder's reconstruction filter can ring
|
|
11
|
+
up past 0 dBFS even when every decoded sample was under it.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
import math
|
|
16
|
+
|
|
17
|
+
import numpy
|
|
18
|
+
import pyloudnorm
|
|
19
|
+
import scipy.signal
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger("dw")
|
|
22
|
+
|
|
23
|
+
# The floor a level is reported at rather than -inf, which JSON cannot carry
|
|
24
|
+
SILENCE_DBFS = -120.0
|
|
25
|
+
|
|
26
|
+
# BS.1770's gating block is 400 ms; pyloudnorm refuses anything shorter
|
|
27
|
+
MIN_LUFS_SECONDS = 0.4
|
|
28
|
+
|
|
29
|
+
# Minimum oversampling BS.1770 defines for a true-peak measurement
|
|
30
|
+
TRUE_PEAK_OVERSAMPLE = 4
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _dbfs(value):
|
|
34
|
+
if value <= 0:
|
|
35
|
+
return SILENCE_DBFS
|
|
36
|
+
return max(SILENCE_DBFS, 20.0 * math.log10(float(value)))
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def integrated_lufs(samples, rate):
|
|
40
|
+
"""Integrated loudness in LUFS, or None when it cannot be measured.
|
|
41
|
+
|
|
42
|
+
`samples` is (frames, channels) or (frames,). None for a clip shorter
|
|
43
|
+
than the 400 ms gating block, an all-silent clip (pyloudnorm's own
|
|
44
|
+
-inf, which JSON cannot carry either), or anything pyloudnorm refuses -
|
|
45
|
+
a measurement that failed reads as "unknown" rather than as a crashed
|
|
46
|
+
task or probe.
|
|
47
|
+
"""
|
|
48
|
+
if samples is None or samples.size == 0:
|
|
49
|
+
return None
|
|
50
|
+
frames = samples.shape[0]
|
|
51
|
+
if frames < int(round(MIN_LUFS_SECONDS * rate)):
|
|
52
|
+
return None
|
|
53
|
+
try:
|
|
54
|
+
meter = pyloudnorm.Meter(rate)
|
|
55
|
+
value = float(meter.integrated_loudness(samples))
|
|
56
|
+
except Exception as e:
|
|
57
|
+
logger.debug(f"integrated_lufs: could not measure: {e}")
|
|
58
|
+
return None
|
|
59
|
+
if not math.isfinite(value):
|
|
60
|
+
return None
|
|
61
|
+
return value
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def true_peak_dbfs(samples, oversample=TRUE_PEAK_OVERSAMPLE):
|
|
65
|
+
"""The inter-sample (true) peak of a track, in dBFS.
|
|
66
|
+
|
|
67
|
+
`samples` is (frames, channels) or (frames,). Oversamples with a
|
|
68
|
+
polyphase FIR (BS.1770's own reconstruction) and reads the peak of the
|
|
69
|
+
interpolated signal, which is what a lossy encoder's own reconstruction
|
|
70
|
+
filter can ring up past a sample-peak reading that stayed under 0 dBFS.
|
|
71
|
+
SILENCE_DBFS for an empty track, never None - unlike integrated
|
|
72
|
+
loudness, a true peak is defined for any track, including a silent one.
|
|
73
|
+
"""
|
|
74
|
+
if samples is None or samples.size == 0:
|
|
75
|
+
return SILENCE_DBFS
|
|
76
|
+
try:
|
|
77
|
+
oversampled = scipy.signal.resample_poly(samples, oversample, 1, axis=0)
|
|
78
|
+
except Exception as e:
|
|
79
|
+
logger.debug(f"true_peak_dbfs: could not oversample: {e}")
|
|
80
|
+
oversampled = samples
|
|
81
|
+
peak = float(numpy.abs(oversampled).max(initial=0.0))
|
|
82
|
+
return _dbfs(peak)
|
dw/media_audio.py
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
"""A soundtrack, or a named slice of one, as WAV bytes - the form
|
|
2
|
+
get_output_audio hands an agent for a muxed video (whose container it
|
|
3
|
+
cannot play) or for a track too long to send whole (#193).
|
|
4
|
+
|
|
5
|
+
PyAV only, like media_info: this runs in the server process, where a
|
|
6
|
+
request must not pull in torch or materialise frames.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import io
|
|
10
|
+
import logging
|
|
11
|
+
import math
|
|
12
|
+
import wave
|
|
13
|
+
|
|
14
|
+
import av
|
|
15
|
+
import numpy
|
|
16
|
+
from av.audio.resampler import AudioResampler
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger("dw")
|
|
19
|
+
|
|
20
|
+
# The most a soundtrack may be as base64 before the gallery route refuses to
|
|
21
|
+
# extract it whole - the twin of dw_mcp/media.py's MAX_RETURNED_BYTES (the
|
|
22
|
+
# MCP package's cap on any inline payload). Two constants because the two
|
|
23
|
+
# packages do not import each other; a whole track over this is cut off at
|
|
24
|
+
# the header, before a frame is decoded, rather than decoded, shipped and
|
|
25
|
+
# then refused by the client.
|
|
26
|
+
MAX_INLINE_AUDIO_BYTES = 4 * 1024 * 1024
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class NoSoundtrack(ValueError):
|
|
30
|
+
"""The file has no audio stream to extract."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _container_duration(container):
|
|
34
|
+
"""The container's own duration (seconds) from its header alone - no
|
|
35
|
+
stream is decoded to produce this number."""
|
|
36
|
+
return (
|
|
37
|
+
float(container.duration / av.time_base)
|
|
38
|
+
if container.duration is not None
|
|
39
|
+
else None
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def media_duration(path):
|
|
44
|
+
"""The container's duration (seconds), read from its header alone - no
|
|
45
|
+
stream is decoded. `extract_audio` needs the same figure as `total`;
|
|
46
|
+
this is that computation, exposed for a caller that wants only the
|
|
47
|
+
length, since decoding to get one number defeats the "nothing to
|
|
48
|
+
extract" case (serving an audio file whole - #193)."""
|
|
49
|
+
with av.open(path) as container:
|
|
50
|
+
return _container_duration(container)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def audio_shape(path):
|
|
54
|
+
"""The soundtrack's duration (seconds), sample rate and channel count
|
|
55
|
+
from the container's headers alone - what projecting the size of a
|
|
56
|
+
whole-track WAV needs, and nothing decoded to get it. `None` when the
|
|
57
|
+
file has no audio stream."""
|
|
58
|
+
with av.open(path) as container:
|
|
59
|
+
if not container.streams.audio:
|
|
60
|
+
return None
|
|
61
|
+
stream = container.streams.audio[0]
|
|
62
|
+
return {
|
|
63
|
+
"duration_seconds": _container_duration(container),
|
|
64
|
+
"sample_rate": int(stream.rate),
|
|
65
|
+
"channels": int(stream.channels),
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def projected_wav_base64_size(shape):
|
|
70
|
+
"""How many bytes the whole track would be as base64 16-bit PCM WAV -
|
|
71
|
+
`extract_audio`'s output for the same file, sized from `audio_shape`
|
|
72
|
+
without producing it. `None` when the container states no duration."""
|
|
73
|
+
if shape is None or shape["duration_seconds"] is None:
|
|
74
|
+
return None
|
|
75
|
+
pcm = int(shape["duration_seconds"] * shape["sample_rate"] * shape["channels"] * 2)
|
|
76
|
+
return 4 * math.ceil(pcm / 3)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def extract_audio(path, start=None, duration=None):
|
|
80
|
+
"""The soundtrack of `path` as 16-bit PCM WAV bytes, plus what was cut.
|
|
81
|
+
|
|
82
|
+
With `start` and `duration` (seconds) only that slice is decoded and
|
|
83
|
+
returned; the info dict says so (`excerpt: True`) and carries the whole
|
|
84
|
+
track's length as `of_seconds`, so a slice always names itself. A
|
|
85
|
+
slice that runs past the end is clipped to it; a `start` past the end
|
|
86
|
+
is refused, since an empty answer would read as a silent track.
|
|
87
|
+
|
|
88
|
+
Returns:
|
|
89
|
+
(wav_bytes, info) with info = {sample_rate, channels,
|
|
90
|
+
duration_seconds, of_seconds, start, excerpt}
|
|
91
|
+
"""
|
|
92
|
+
excerpt = start is not None or duration is not None
|
|
93
|
+
start = float(start or 0.0)
|
|
94
|
+
if excerpt and (duration is None or float(duration) <= 0):
|
|
95
|
+
raise ValueError("An excerpt needs a duration above zero")
|
|
96
|
+
if start < 0:
|
|
97
|
+
raise ValueError("An excerpt cannot start before zero")
|
|
98
|
+
|
|
99
|
+
with av.open(path) as container:
|
|
100
|
+
if not container.streams.audio:
|
|
101
|
+
raise NoSoundtrack(f"{path} has no soundtrack")
|
|
102
|
+
stream = container.streams.audio[0]
|
|
103
|
+
total = _container_duration(container)
|
|
104
|
+
if total is not None and start >= total:
|
|
105
|
+
raise ValueError(
|
|
106
|
+
f"start {start:.2f}s is past the end of a {total:.2f}s track"
|
|
107
|
+
)
|
|
108
|
+
stop = start + float(duration) if excerpt else None
|
|
109
|
+
if stop is not None and total is not None:
|
|
110
|
+
stop = min(stop, total)
|
|
111
|
+
|
|
112
|
+
rate = int(stream.rate)
|
|
113
|
+
channels = int(stream.channels)
|
|
114
|
+
layout = (
|
|
115
|
+
"stereo"
|
|
116
|
+
if channels == 2
|
|
117
|
+
else ("mono" if channels == 1 else stream.layout.name)
|
|
118
|
+
)
|
|
119
|
+
resampler = AudioResampler(format="s16", layout=layout, rate=rate)
|
|
120
|
+
|
|
121
|
+
# pts is a timestamp on the container's clock, not an offset from
|
|
122
|
+
# this stream's first sample: a stream with an edit list or a
|
|
123
|
+
# non-zero start (which real muxers write) carries a `start_time`,
|
|
124
|
+
# and both the seek target and each frame's clock have to be
|
|
125
|
+
# zeroed against it before they mean seconds into the track -
|
|
126
|
+
# the same anchor `_read_frames` in media_frames uses. Unanchored,
|
|
127
|
+
# `start=2.5` on a track shifted by one second came back from
|
|
128
|
+
# about 1.5 s: the wrong audio, silently.
|
|
129
|
+
anchor = stream.start_time if stream.start_time is not None else 0
|
|
130
|
+
if start > 0:
|
|
131
|
+
# Seek to the keyframe at or before `start`, less one packet of
|
|
132
|
+
# pre-roll: the first packet a decoder sees after a seek is its
|
|
133
|
+
# warm-up, and for AAC and mp3 the samples it yields come out
|
|
134
|
+
# attenuated or silent - so an excerpt at 2.5 s opened with a
|
|
135
|
+
# gap that is not in the file. Landing a packet early hands that
|
|
136
|
+
# warm-up to samples the pts-based trim below drops anyway.
|
|
137
|
+
frame_size = int(stream.codec_context.frame_size or 0)
|
|
138
|
+
preroll = frame_size / rate if frame_size else 0.1
|
|
139
|
+
container.seek(
|
|
140
|
+
int(max(0.0, start - preroll) / stream.time_base) + anchor,
|
|
141
|
+
stream=stream,
|
|
142
|
+
backward=True,
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
pieces = []
|
|
146
|
+
seen = 0 # samples decoded so far - the clock for a frame with no pts
|
|
147
|
+
started = False
|
|
148
|
+
done = False # Break outer loop when stop time is reached
|
|
149
|
+
sought = start > 0
|
|
150
|
+
restarted = False
|
|
151
|
+
for frame in container.decode(stream):
|
|
152
|
+
if done:
|
|
153
|
+
break
|
|
154
|
+
if frame.pts is None and sought and not restarted:
|
|
155
|
+
# No pts means the seek's landing cannot be known, so the
|
|
156
|
+
# sample count is the only clock there is - and it has to
|
|
157
|
+
# count from the top. Start over once and read to `start`.
|
|
158
|
+
container.seek(0, stream=stream, backward=True)
|
|
159
|
+
resampler = AudioResampler(format="s16", layout=layout, rate=rate)
|
|
160
|
+
restarted = True
|
|
161
|
+
seen = 0
|
|
162
|
+
continue
|
|
163
|
+
frame_start = (
|
|
164
|
+
float((frame.pts - anchor) * stream.time_base)
|
|
165
|
+
if frame.pts is not None
|
|
166
|
+
else seen / rate
|
|
167
|
+
)
|
|
168
|
+
for chunk in resampler.resample(frame):
|
|
169
|
+
samples = chunk.to_ndarray() # (1, samples * channels) packed s16
|
|
170
|
+
samples = samples.reshape(-1, channels)
|
|
171
|
+
chunk_start = frame_start
|
|
172
|
+
chunk_end = chunk_start + samples.shape[0] / rate
|
|
173
|
+
seen += samples.shape[0]
|
|
174
|
+
frame_start = chunk_end # a frame yielding two chunks: the second follows the first
|
|
175
|
+
if chunk_end <= start:
|
|
176
|
+
continue
|
|
177
|
+
if not started and chunk_start < start:
|
|
178
|
+
# round, not floor: float pts arithmetic lands a hair
|
|
179
|
+
# under the exact sample and floor then keeps one too many
|
|
180
|
+
samples = samples[int(round((start - chunk_start) * rate)) :]
|
|
181
|
+
chunk_start = start
|
|
182
|
+
started = True
|
|
183
|
+
if stop is not None and chunk_end > stop:
|
|
184
|
+
samples = samples[: max(0, int(round((stop - chunk_start) * rate)))]
|
|
185
|
+
pieces.append(samples)
|
|
186
|
+
if stop is not None and chunk_end >= stop:
|
|
187
|
+
done = True
|
|
188
|
+
break
|
|
189
|
+
# Flush the resampler whatever ended the loop: it may still hold
|
|
190
|
+
# samples that belong before `stop`. Anything past `stop` is cut.
|
|
191
|
+
held = sum(p.shape[0] for p in pieces)
|
|
192
|
+
for chunk in resampler.resample(None):
|
|
193
|
+
samples = chunk.to_ndarray().reshape(-1, channels)
|
|
194
|
+
if stop is not None:
|
|
195
|
+
room = max(0, int(round((stop - start) * rate)) - held)
|
|
196
|
+
samples = samples[:room]
|
|
197
|
+
if samples.shape[0]:
|
|
198
|
+
pieces.append(samples)
|
|
199
|
+
held += samples.shape[0]
|
|
200
|
+
|
|
201
|
+
pcm = numpy.concatenate(pieces) if pieces else numpy.zeros((0, channels), "<i2")
|
|
202
|
+
buffer = io.BytesIO()
|
|
203
|
+
with wave.open(buffer, "w") as handle:
|
|
204
|
+
handle.setnchannels(channels)
|
|
205
|
+
handle.setsampwidth(2)
|
|
206
|
+
handle.setframerate(rate)
|
|
207
|
+
handle.writeframes(pcm.astype("<i2").tobytes())
|
|
208
|
+
|
|
209
|
+
returned = pcm.shape[0] / rate
|
|
210
|
+
return buffer.getvalue(), {
|
|
211
|
+
"sample_rate": rate,
|
|
212
|
+
"channels": channels,
|
|
213
|
+
"duration_seconds": returned,
|
|
214
|
+
"of_seconds": total if total is not None else returned,
|
|
215
|
+
"start": start,
|
|
216
|
+
"excerpt": excerpt,
|
|
217
|
+
}
|
dw/media_frames.py
ADDED
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
"""Frames out of a video file by seeking to them - a moment, an evenly
|
|
2
|
+
spaced contact sheet, or the frame pair either side of a seam - without
|
|
3
|
+
decoding the clip whole. The server process runs this per request; a
|
|
4
|
+
four-minute 1080p clip materialised as PIL frames is tens of gigabytes,
|
|
5
|
+
so nothing here ever holds more than the frames it returns, each already
|
|
6
|
+
fitted to its tile where a caller asked for many (#193).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import logging
|
|
10
|
+
import math
|
|
11
|
+
|
|
12
|
+
import av
|
|
13
|
+
import numpy
|
|
14
|
+
from PIL import Image
|
|
15
|
+
|
|
16
|
+
from .tasks.video_utils import (
|
|
17
|
+
_compose_grid,
|
|
18
|
+
_default_columns,
|
|
19
|
+
_evenly_spaced_indices,
|
|
20
|
+
_format_timestamp,
|
|
21
|
+
_grid_tile,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
logger = logging.getLogger("dw")
|
|
25
|
+
|
|
26
|
+
# Each cell of a contact sheet and each side of a seam pair is a seek, a
|
|
27
|
+
# decode and a resize in the server process; a sheet past 64 cells is
|
|
28
|
+
# unreadable anyway, and `at` has its own cap in the route
|
|
29
|
+
# (MAX_FRAME_MOMENTS). Both are ValueErrors, so the route answers 400.
|
|
30
|
+
MAX_CONTACT_SHEET_FRAMES = 64
|
|
31
|
+
MAX_SEAMS = 32
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def video_shape(path):
|
|
35
|
+
"""Frame count, fps and size, from the container's own headers where
|
|
36
|
+
they are written and by counting otherwise."""
|
|
37
|
+
with av.open(path) as container:
|
|
38
|
+
if not container.streams.video:
|
|
39
|
+
raise ValueError(f"{path} has no video stream")
|
|
40
|
+
stream = container.streams.video[0]
|
|
41
|
+
fps = float(stream.average_rate) if stream.average_rate else None
|
|
42
|
+
count = int(stream.frames) if stream.frames else None
|
|
43
|
+
if count is None:
|
|
44
|
+
count = sum(1 for _ in container.decode(stream))
|
|
45
|
+
return {
|
|
46
|
+
"frame_count": count,
|
|
47
|
+
"fps": fps,
|
|
48
|
+
"width": int(stream.width),
|
|
49
|
+
"height": int(stream.height),
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def resolve_crop_box(crop, width, height):
|
|
54
|
+
"""`[x, y, w, h]` as Pillow's `(left, upper, right, lower)`, clamped to
|
|
55
|
+
a frame. Every frame of one video shares the same dimensions, so a
|
|
56
|
+
caller resolves this once against `video_shape(path)` and reuses it
|
|
57
|
+
across every tile - the same `[x, y, width, height]` convention
|
|
58
|
+
`get_output_image`'s crop uses. Refused when it is not four
|
|
59
|
+
non-negative integers, starts outside the frame, or has nothing in it."""
|
|
60
|
+
try:
|
|
61
|
+
x, y, w, h = (int(v) for v in crop)
|
|
62
|
+
except (TypeError, ValueError):
|
|
63
|
+
raise ValueError(f"crop must be [x, y, width, height] in pixels, got {crop!r}.")
|
|
64
|
+
if x < 0 or y < 0 or w <= 0 or h <= 0:
|
|
65
|
+
raise ValueError(
|
|
66
|
+
f"crop must have a non-negative origin and a positive size, got {crop!r}."
|
|
67
|
+
)
|
|
68
|
+
if x >= width or y >= height:
|
|
69
|
+
raise ValueError(
|
|
70
|
+
f"crop origin ({x}, {y}) lies outside the {width}x{height} frame."
|
|
71
|
+
)
|
|
72
|
+
return (x, y, min(x + w, width), min(y + h, height))
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def frames_at(path, moments, shape=None, crop_box=None):
|
|
76
|
+
"""One tile per moment - a float in seconds or "frame:N" - in the order
|
|
77
|
+
asked for. Each tile is {label, frame, seconds, image}.
|
|
78
|
+
|
|
79
|
+
`shape` reuses an already-computed `video_shape(path)` - a caller such
|
|
80
|
+
as the gallery route that also reports the clip's own frame_count/fps
|
|
81
|
+
would otherwise pay for `video_shape`'s container open (and, lacking a
|
|
82
|
+
header frame count, a full decode) a second time for the same answer.
|
|
83
|
+
|
|
84
|
+
`crop_box` (`resolve_crop_box`'s Pillow box) is cut from each frame
|
|
85
|
+
right after decode, before it is returned - full source resolution,
|
|
86
|
+
not whatever a caller's `max_dimension` later downscales it to."""
|
|
87
|
+
shape = shape if shape is not None else video_shape(path)
|
|
88
|
+
indexes = [_moment_to_index(moment, shape) for moment in moments]
|
|
89
|
+
fit = (lambda image, index: image.crop(crop_box)) if crop_box else None
|
|
90
|
+
images = _read_frames(path, indexes, fit=fit)
|
|
91
|
+
return [_tile(index, images[index], shape) for index in indexes]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def contact_sheet(path, count, tile_width=320, shape=None, crop_box=None):
|
|
95
|
+
"""N evenly spaced frames (first and last included) tiled into one
|
|
96
|
+
image - frame_grid without a workflow. `shape` reuses an
|
|
97
|
+
already-computed `video_shape(path)`; see `frames_at`.
|
|
98
|
+
|
|
99
|
+
`crop_box`, as in `frames_at`, is cut from each frame before it is
|
|
100
|
+
stamped and tiled into the sheet - the sheet is built from cropped
|
|
101
|
+
frames rather than cropped after assembly, so the box means the same
|
|
102
|
+
source-pixel region whatever the sheet's own `tile_width` ends up."""
|
|
103
|
+
if int(count) < 1:
|
|
104
|
+
raise ValueError("count must be at least 1")
|
|
105
|
+
shape = shape if shape is not None else video_shape(path)
|
|
106
|
+
# Clamp to the clip's own length before checking the cap: a count that
|
|
107
|
+
# would only ever produce a handful of cells (a short clip) should not
|
|
108
|
+
# be refused for the raw number the caller asked for.
|
|
109
|
+
count = min(int(count), shape["frame_count"])
|
|
110
|
+
if count > MAX_CONTACT_SHEET_FRAMES:
|
|
111
|
+
raise ValueError(
|
|
112
|
+
f"count {count} is more than a contact sheet holds "
|
|
113
|
+
f"({MAX_CONTACT_SHEET_FRAMES}); ask for a smaller one, or `at` for moments"
|
|
114
|
+
)
|
|
115
|
+
indexes = _evenly_spaced_indices(shape["frame_count"], count)
|
|
116
|
+
|
|
117
|
+
def fit(image, index):
|
|
118
|
+
if crop_box:
|
|
119
|
+
image = image.crop(crop_box)
|
|
120
|
+
return _stamped_tile(image, index, shape["fps"], tile_width)
|
|
121
|
+
|
|
122
|
+
images = _read_frames(path, indexes, fit=fit)
|
|
123
|
+
tiles = [images[index] for index in indexes]
|
|
124
|
+
grid = _compose_grid(tiles, _default_columns(len(tiles)))
|
|
125
|
+
return {
|
|
126
|
+
"label": f"contact sheet, {len(tiles)} frames",
|
|
127
|
+
"frame": indexes[0],
|
|
128
|
+
"seconds": _seconds(indexes[0], shape),
|
|
129
|
+
"image": grid,
|
|
130
|
+
"frames": list(indexes),
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def seam_tiles(
|
|
135
|
+
path, boundaries, names=None, tile_width=320, shape=None, wanted=None, crop_box=None
|
|
136
|
+
):
|
|
137
|
+
"""For each boundary (the frame index a shot *starts* at), the last
|
|
138
|
+
frame before it and the first frame at it, side by side - the seam and
|
|
139
|
+
continuity evidence in one image. Seam i sits between shot i and shot
|
|
140
|
+
i+1; `names` names the shots, "shot 1".. by default.
|
|
141
|
+
|
|
142
|
+
`boundaries` and `names` are validated in full regardless of `wanted` -
|
|
143
|
+
they describe the whole cut, and a seam's label names the shots either
|
|
144
|
+
side of it by position in that full list. `wanted` (a set of 1-based
|
|
145
|
+
seam numbers, or None for all) then limits which seams are actually
|
|
146
|
+
decoded and composed: a caller after seam 2 of twelve should not pay to
|
|
147
|
+
decode and tile the other eleven. `shape` reuses an already-computed
|
|
148
|
+
`video_shape(path)`; see `frames_at`.
|
|
149
|
+
|
|
150
|
+
`crop_box`, as in `frames_at`, is cut from each source frame before it
|
|
151
|
+
is fit to `tile_width` and paired into a seam - the pair is built from
|
|
152
|
+
cropped frames rather than cropped after pairing."""
|
|
153
|
+
shape = shape if shape is not None else video_shape(path)
|
|
154
|
+
total = shape["frame_count"]
|
|
155
|
+
for boundary in boundaries:
|
|
156
|
+
if not 1 <= int(boundary) <= total - 1:
|
|
157
|
+
raise ValueError(
|
|
158
|
+
f"boundary {boundary} is not inside the clip (1..{total - 1})"
|
|
159
|
+
)
|
|
160
|
+
boundaries = [int(b) for b in boundaries]
|
|
161
|
+
names = list(names or [f"shot {n + 1}" for n in range(len(boundaries) + 1)])
|
|
162
|
+
if len(names) != len(boundaries) + 1:
|
|
163
|
+
raise ValueError(
|
|
164
|
+
f"{len(boundaries)} boundaries make {len(boundaries) + 1} shots, "
|
|
165
|
+
f"but {len(names)} names were given"
|
|
166
|
+
)
|
|
167
|
+
if wanted is not None:
|
|
168
|
+
off = sorted(int(s) for s in wanted if not 1 <= int(s) <= len(boundaries))
|
|
169
|
+
if off:
|
|
170
|
+
raise ValueError(
|
|
171
|
+
f"seam {', '.join(str(s) for s in off)} is not in this cut - "
|
|
172
|
+
f"{len(boundaries)} boundaries make seams 1..{len(boundaries)}"
|
|
173
|
+
)
|
|
174
|
+
chosen = [
|
|
175
|
+
(seam, boundary)
|
|
176
|
+
for seam, boundary in enumerate(boundaries, start=1)
|
|
177
|
+
if wanted is None or seam in wanted
|
|
178
|
+
]
|
|
179
|
+
if len(chosen) > MAX_SEAMS:
|
|
180
|
+
raise ValueError(
|
|
181
|
+
f"{len(chosen)} seams is more than one call serves ({MAX_SEAMS}); "
|
|
182
|
+
"name the seams wanted (`seams=1,2,...`)"
|
|
183
|
+
)
|
|
184
|
+
frame_indexes = sorted({b - 1 for _, b in chosen} | {b for _, b in chosen})
|
|
185
|
+
|
|
186
|
+
def fit(image, _index):
|
|
187
|
+
if crop_box:
|
|
188
|
+
image = image.crop(crop_box)
|
|
189
|
+
return _fit_width(image, tile_width)
|
|
190
|
+
|
|
191
|
+
images = _read_frames(path, frame_indexes, fit=fit)
|
|
192
|
+
tiles = []
|
|
193
|
+
for seam, boundary in chosen:
|
|
194
|
+
before = images[boundary - 1]
|
|
195
|
+
after = images[boundary]
|
|
196
|
+
pair = _compose_grid([before, after], 2)
|
|
197
|
+
difference = float(
|
|
198
|
+
numpy.abs(
|
|
199
|
+
numpy.asarray(before, dtype=numpy.int16)
|
|
200
|
+
- numpy.asarray(after, dtype=numpy.int16)
|
|
201
|
+
).mean()
|
|
202
|
+
)
|
|
203
|
+
tiles.append(
|
|
204
|
+
{
|
|
205
|
+
"label": f"seam {seam}: {names[seam - 1]} | {names[seam]}",
|
|
206
|
+
"frame": boundary,
|
|
207
|
+
"seconds": _seconds(boundary, shape),
|
|
208
|
+
"difference": round(difference, 2),
|
|
209
|
+
"image": pair,
|
|
210
|
+
}
|
|
211
|
+
)
|
|
212
|
+
return tiles
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _moment_to_index(moment, shape):
|
|
216
|
+
total = shape["frame_count"]
|
|
217
|
+
if isinstance(moment, str) and moment.startswith("frame:"):
|
|
218
|
+
raw = moment[len("frame:") :]
|
|
219
|
+
try:
|
|
220
|
+
index = int(raw)
|
|
221
|
+
except ValueError:
|
|
222
|
+
raise ValueError(f'"{moment}" - frame index must be a whole number')
|
|
223
|
+
if index < 0:
|
|
224
|
+
index += total
|
|
225
|
+
if not 0 <= index < total:
|
|
226
|
+
raise ValueError(
|
|
227
|
+
f"frame {index} is past the end of a {total}-frame clip "
|
|
228
|
+
f"(frames 0-{total - 1})"
|
|
229
|
+
)
|
|
230
|
+
return index
|
|
231
|
+
fps = shape["fps"]
|
|
232
|
+
if fps is None:
|
|
233
|
+
raise ValueError("this clip has no frame rate, so name a frame: 'frame:N'")
|
|
234
|
+
seconds = float(moment)
|
|
235
|
+
if not math.isfinite(seconds):
|
|
236
|
+
raise ValueError(f"{moment!r} is not a moment in seconds")
|
|
237
|
+
index = int(round(seconds * fps))
|
|
238
|
+
if index < 0:
|
|
239
|
+
index += total
|
|
240
|
+
if not 0 <= index < total:
|
|
241
|
+
duration = total / fps
|
|
242
|
+
message = (
|
|
243
|
+
f"{moment!r} s is past the end of a {duration:.2f} s "
|
|
244
|
+
f"({total}-frame) clip - bare numbers in 'at' are seconds"
|
|
245
|
+
)
|
|
246
|
+
# A fractional value is a seconds overshoot, not a frame index in
|
|
247
|
+
# disguise - the frame: hint only makes sense for a whole number
|
|
248
|
+
# that would itself be a valid frame index.
|
|
249
|
+
if seconds.is_integer() and 0 <= int(seconds) < total:
|
|
250
|
+
message += f'; use "frame:{_format_moment(moment)}" for a frame index'
|
|
251
|
+
raise ValueError(message)
|
|
252
|
+
return index
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _format_moment(moment):
|
|
256
|
+
if isinstance(moment, float) and moment.is_integer():
|
|
257
|
+
return str(int(moment))
|
|
258
|
+
return str(moment)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _seconds(index, shape):
|
|
262
|
+
return index / shape["fps"] if shape["fps"] else float(index)
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _tile(index, image, shape):
|
|
266
|
+
fps = shape["fps"]
|
|
267
|
+
stamp = _format_timestamp(index, fps) if fps else f"#{index}"
|
|
268
|
+
return {
|
|
269
|
+
"label": f"{stamp} (frame {index})",
|
|
270
|
+
"frame": index,
|
|
271
|
+
"seconds": _seconds(index, shape),
|
|
272
|
+
"image": image,
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _stamped_tile(image, index, fps, tile_width):
|
|
277
|
+
"""A contact-sheet cell: fitted like `_fit_width` (never upscaled) and
|
|
278
|
+
stamped with its timestamp by `frame_grid`'s own tile maker, so the
|
|
279
|
+
sheet says which cell is which without the text part."""
|
|
280
|
+
return _grid_tile(image, index, fps, min(int(tile_width), image.width), label=True)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _fit_width(image, tile_width):
|
|
284
|
+
# never upscaled: a tile wider than its source is a blurred enlargement
|
|
285
|
+
# that costs bytes and shows nothing the source holds
|
|
286
|
+
tile_width = min(int(tile_width), image.width)
|
|
287
|
+
height = max(1, round(image.height * tile_width / image.width))
|
|
288
|
+
return image.resize((tile_width, height), Image.LANCZOS).convert("RGB")
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _read_frames(path, indexes, fit=None):
|
|
292
|
+
"""The frames at these indexes, as {index: PIL image}, in one forward
|
|
293
|
+
pass that seeks to the keyframe before each wanted frame rather than
|
|
294
|
+
decoding from the top. Decodes are dropped as soon as they are past.
|
|
295
|
+
|
|
296
|
+
`fit(image, index)`, when given, is applied to each frame as it is
|
|
297
|
+
decoded, so a caller tiling many frames never holds one at source size.
|
|
298
|
+
|
|
299
|
+
This assumes a constant frame rate, which every file this engine writes
|
|
300
|
+
has (`encode_video` / `export_to_video` write a fixed `fps`): a
|
|
301
|
+
`backward=True` seek lands on the keyframe at or before the target
|
|
302
|
+
timestamp, and `frame.pts * stream.time_base` converts that keyframe's
|
|
303
|
+
own presentation time back to an exact frame index (`round(seconds *
|
|
304
|
+
fps)`) - verified against PyAV 18.1's actual seek landings (a
|
|
305
|
+
single-keyframe short clip, where every seek lands on frame 0; a `g=10`
|
|
306
|
+
multi-keyframe clip, where a seek to frame 95 lands exactly on frame 90;
|
|
307
|
+
and a clip whose packets carry a 5-frame pts offset - an edit list or a
|
|
308
|
+
non-zero start, which real muxers write - where the raw pts arithmetic
|
|
309
|
+
landed 5 frames off until it was anchored on `stream.start_time`).
|
|
310
|
+
`start_pts` is that anchor: pts is a timestamp against the *container's*
|
|
311
|
+
clock, not a frame count from this stream's first frame, so it has to be
|
|
312
|
+
zeroed against wherever this stream actually starts before it means a
|
|
313
|
+
frame index. `position` is then a plain frame counter from that
|
|
314
|
+
landing, decoding forward to the target and dropping what is skipped
|
|
315
|
+
past.
|
|
316
|
+
"""
|
|
317
|
+
wanted = sorted(set(int(i) for i in indexes))
|
|
318
|
+
found = {}
|
|
319
|
+
with av.open(path) as container:
|
|
320
|
+
stream = container.streams.video[0]
|
|
321
|
+
fps = float(stream.average_rate) if stream.average_rate else None
|
|
322
|
+
start_pts = stream.start_time if stream.start_time is not None else 0
|
|
323
|
+
position = 0 # index of the next frame decode() will yield
|
|
324
|
+
for target in wanted:
|
|
325
|
+
if target < position or target - position > 2 * (int(fps) if fps else 24):
|
|
326
|
+
# seek back or a long way forward: land on the keyframe at
|
|
327
|
+
# or before the target, then read up to it. The seek target
|
|
328
|
+
# is a container timestamp too, so it needs the same anchor.
|
|
329
|
+
seconds = target / fps if fps else 0.0
|
|
330
|
+
container.seek(
|
|
331
|
+
int(seconds / stream.time_base) + start_pts,
|
|
332
|
+
stream=stream,
|
|
333
|
+
backward=True,
|
|
334
|
+
)
|
|
335
|
+
position = None
|
|
336
|
+
recovered = False
|
|
337
|
+
for frame in container.decode(stream):
|
|
338
|
+
if position is None:
|
|
339
|
+
# first frame after a seek says where we landed
|
|
340
|
+
position = (
|
|
341
|
+
int(
|
|
342
|
+
round(
|
|
343
|
+
float((frame.pts - start_pts) * stream.time_base) * fps
|
|
344
|
+
)
|
|
345
|
+
)
|
|
346
|
+
if fps and frame.pts is not None
|
|
347
|
+
else 0
|
|
348
|
+
)
|
|
349
|
+
if position > target and not recovered:
|
|
350
|
+
# Landed past the target: the keyframe estimate was
|
|
351
|
+
# wrong for this file (off-rate or VFR). Reading on
|
|
352
|
+
# would scan to EOF and blame the caller; read from
|
|
353
|
+
# the top once instead, which is always correct.
|
|
354
|
+
container.seek(start_pts, stream=stream, backward=True)
|
|
355
|
+
position = None
|
|
356
|
+
recovered = True
|
|
357
|
+
continue
|
|
358
|
+
if position == target:
|
|
359
|
+
image = frame.to_image()
|
|
360
|
+
found[target] = fit(image, target) if fit is not None else image
|
|
361
|
+
position += 1
|
|
362
|
+
break
|
|
363
|
+
position += 1
|
|
364
|
+
missing = [i for i in wanted if i not in found]
|
|
365
|
+
if missing:
|
|
366
|
+
raise ValueError(f"could not decode frame(s) {missing} of {path}")
|
|
367
|
+
return found
|