diffusers-workflow 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
  2. diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
  3. diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
  4. diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
  5. diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
  6. diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
  7. dw/__init__.py +440 -0
  8. dw/adapter_compatibility.py +226 -0
  9. dw/arguments.py +1231 -0
  10. dw/assessment_rules.py +159 -0
  11. dw/assets.py +130 -0
  12. dw/cache_blocks.json +16 -0
  13. dw/cache_blocks.py +146 -0
  14. dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
  15. dw/content_types.py +150 -0
  16. dw/dissolve_frame_errors.py +121 -0
  17. dw/docs/ACCELERATION.md +352 -0
  18. dw/docs/AGENT_LOOP.md +95 -0
  19. dw/docs/DEPENDENCIES.md +91 -0
  20. dw/docs/IP_ADAPTER.md +109 -0
  21. dw/docs/LORAS.md +131 -0
  22. dw/docs/MCP.md +517 -0
  23. dw/docs/PROMPT_WEIGHTING.md +78 -0
  24. dw/docs/QUANTIZATION.md +230 -0
  25. dw/docs/RECIPES_24GB.md +201 -0
  26. dw/docs/RELEASING.md +195 -0
  27. dw/docs/REMOTE.md +140 -0
  28. dw/docs/REPL_COMMANDS.md +121 -0
  29. dw/docs/REPL_WORKER_GUIDE.md +51 -0
  30. dw/docs/SECURITY.md +272 -0
  31. dw/docs/SECURITY_QUICKREF.md +112 -0
  32. dw/docs/SERVER.md +679 -0
  33. dw/docs/TASKS.md +1741 -0
  34. dw/docs/TESTING.md +71 -0
  35. dw/docs/WORKFLOW_GUIDE.md +2038 -0
  36. dw/docs/WORKSPACES.md +316 -0
  37. dw/download_watch.py +335 -0
  38. dw/elision.py +306 -0
  39. dw/events.py +275 -0
  40. dw/for_each.py +409 -0
  41. dw/host_memory.py +258 -0
  42. dw/host_memory_projection.py +230 -0
  43. dw/hub_cache.py +432 -0
  44. dw/introspection.py +1228 -0
  45. dw/kernel_availability.py +208 -0
  46. dw/locations.py +599 -0
  47. dw/log_setup.py +45 -0
  48. dw/loudness.py +82 -0
  49. dw/media_audio.py +217 -0
  50. dw/media_frames.py +367 -0
  51. dw/media_info.py +297 -0
  52. dw/pipeline_processors/chain.py +821 -0
  53. dw/pipeline_processors/config_objects.py +237 -0
  54. dw/pipeline_processors/pipeline.py +2297 -0
  55. dw/pipeline_processors/remote.py +46 -0
  56. dw/plan.py +920 -0
  57. dw/previous_results.py +411 -0
  58. dw/probe_paths.py +59 -0
  59. dw/prompt_schema.json +48 -0
  60. dw/prompt_weighting.py +378 -0
  61. dw/prompts.py +159 -0
  62. dw/realize.py +250 -0
  63. dw/reference_limits.py +215 -0
  64. dw/reference_names.py +125 -0
  65. dw/repl.py +338 -0
  66. dw/repl_commands.py +836 -0
  67. dw/repl_worker.py +159 -0
  68. dw/result.py +1720 -0
  69. dw/result_fps.py +82 -0
  70. dw/run.py +162 -0
  71. dw/runs.py +768 -0
  72. dw/scalar_result_validation.py +97 -0
  73. dw/schema.py +283 -0
  74. dw/security.py +1038 -0
  75. dw/select_validation.py +115 -0
  76. dw/serve.py +277 -0
  77. dw/server/__init__.py +2 -0
  78. dw/server/app.py +4586 -0
  79. dw/server/assess.py +132 -0
  80. dw/server/catalog_shape.py +487 -0
  81. dw/server/enhancers.py +129 -0
  82. dw/server/exports.py +480 -0
  83. dw/server/guides.py +257 -0
  84. dw/server/jobs.py +1561 -0
  85. dw/server/mcp_mount.py +95 -0
  86. dw/server/netinfo.py +124 -0
  87. dw/server/observed_cost.py +379 -0
  88. dw/server/sysinfo.py +71 -0
  89. dw/server/ui/assets/abap-08VXUWAP.js +1 -0
  90. dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
  91. dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
  92. dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
  93. dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
  94. dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
  95. dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
  96. dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
  97. dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
  98. dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
  99. dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
  100. dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
  101. dw/server/ui/assets/css-DIMkf-bt.js +3 -0
  102. dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
  103. dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
  104. dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
  105. dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
  106. dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
  107. dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
  108. dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
  109. dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
  110. dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
  111. dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
  112. dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
  113. dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
  114. dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
  115. dw/server/ui/assets/go-C-y9NEjX.js +1 -0
  116. dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
  117. dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
  118. dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
  119. dw/server/ui/assets/html-PhsdjHSr.js +1 -0
  120. dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
  121. dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
  122. dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
  123. dw/server/ui/assets/index-DgrYhQd9.js +43 -0
  124. dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
  125. dw/server/ui/assets/java-BEtHBSE6.js +1 -0
  126. dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
  127. dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
  128. dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
  129. dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
  130. dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
  131. dw/server/ui/assets/less-B9JPFI3C.js +2 -0
  132. dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
  133. dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
  134. dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
  135. dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
  136. dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
  137. dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
  138. dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
  139. dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
  140. dw/server/ui/assets/monaco--ixms01u.css +1 -0
  141. dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
  142. dw/server/ui/assets/msdax-DauUninz.js +1 -0
  143. dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
  144. dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
  145. dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
  146. dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
  147. dw/server/ui/assets/perl-oz_6vUea.js +1 -0
  148. dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
  149. dw/server/ui/assets/php-nr791fC2.js +1 -0
  150. dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
  151. dw/server/ui/assets/postiats-43DmfD33.js +1 -0
  152. dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
  153. dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
  154. dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
  155. dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
  156. dw/server/ui/assets/python-Bcn70HdC.js +1 -0
  157. dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
  158. dw/server/ui/assets/r-BwWrilGY.js +1 -0
  159. dw/server/ui/assets/razor-D1HmNnby.js +1 -0
  160. dw/server/ui/assets/redis-ClamHrr6.js +1 -0
  161. dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
  162. dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
  163. dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
  164. dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
  165. dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
  166. dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
  167. dw/server/ui/assets/scheme-BeGwcela.js +1 -0
  168. dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
  169. dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
  170. dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
  171. dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
  172. dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
  173. dw/server/ui/assets/sql-NEE52Syq.js +1 -0
  174. dw/server/ui/assets/st-DbInun42.js +1 -0
  175. dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
  176. dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
  177. dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
  178. dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
  179. dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
  180. dw/server/ui/assets/twig-De2hgUGE.js +1 -0
  181. dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
  182. dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
  183. dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
  184. dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
  185. dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
  186. dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
  187. dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
  188. dw/server/ui/index.html +17 -0
  189. dw/server/updater.py +192 -0
  190. dw/settings.py +98 -0
  191. dw/shot_span_preflight.py +116 -0
  192. dw/shots.py +359 -0
  193. dw/slice_preflight.py +148 -0
  194. dw/step.py +187 -0
  195. dw/step_cache.py +442 -0
  196. dw/subfolders.py +107 -0
  197. dw/task_domains.py +307 -0
  198. dw/tasks/assess.py +826 -0
  199. dw/tasks/audio_transcription.py +88 -0
  200. dw/tasks/audio_utils.py +1862 -0
  201. dw/tasks/background_remover.py +43 -0
  202. dw/tasks/borders.py +113 -0
  203. dw/tasks/compose_text.py +74 -0
  204. dw/tasks/concat_videos.py +300 -0
  205. dw/tasks/depth_estimator.py +54 -0
  206. dw/tasks/diffusion_upscale.py +109 -0
  207. dw/tasks/dissolve_videos.py +342 -0
  208. dw/tasks/format_messages.py +24 -0
  209. dw/tasks/gather.py +173 -0
  210. dw/tasks/grade.py +97 -0
  211. dw/tasks/image_to_text.py +43 -0
  212. dw/tasks/image_utils.py +764 -0
  213. dw/tasks/interpolate_frames.py +252 -0
  214. dw/tasks/judge.py +68 -0
  215. dw/tasks/model_cache.py +55 -0
  216. dw/tasks/pair_audio.py +268 -0
  217. dw/tasks/qr_code.py +19 -0
  218. dw/tasks/restore_faces.py +175 -0
  219. dw/tasks/rife_model.py +192 -0
  220. dw/tasks/segment.py +121 -0
  221. dw/tasks/select.py +111 -0
  222. dw/tasks/speech_generation.py +228 -0
  223. dw/tasks/stabilize.py +129 -0
  224. dw/tasks/task.py +920 -0
  225. dw/tasks/tensor_image.py +57 -0
  226. dw/tasks/text_generation.py +169 -0
  227. dw/tasks/text_sections.py +80 -0
  228. dw/tasks/upscale.py +203 -0
  229. dw/tasks/video_utils.py +624 -0
  230. dw/tasks/zoe_depth.py +71 -0
  231. dw/teacache.py +381 -0
  232. dw/teacache_models.json +99 -0
  233. dw/test.py +29 -0
  234. dw/type_helpers.py +231 -0
  235. dw/validate.py +68 -0
  236. dw/variable_constraints.py +444 -0
  237. dw/variables.py +443 -0
  238. dw/video_extensions.py +141 -0
  239. dw/vram_estimate.py +116 -0
  240. dw/worker.py +764 -0
  241. dw/workflow.py +2007 -0
  242. dw/workflow_schema.json +1346 -0
  243. dw/workflow_sources.py +383 -0
  244. dw/workflows/h3_context_ir.json +57 -0
  245. dw/workflows/test.json +31 -0
  246. dw/workspace.py +730 -0
  247. dw_mcp/__init__.py +6 -0
  248. dw_mcp/__main__.py +133 -0
  249. dw_mcp/assets.py +336 -0
  250. dw_mcp/authoring.py +114 -0
  251. dw_mcp/catalog.py +360 -0
  252. dw_mcp/client.py +486 -0
  253. dw_mcp/diagnose.py +371 -0
  254. dw_mcp/exports.py +84 -0
  255. dw_mcp/guides.py +35 -0
  256. dw_mcp/media.py +638 -0
  257. dw_mcp/models.py +97 -0
  258. dw_mcp/prompts.py +104 -0
  259. dw_mcp/server.py +1343 -0
  260. dw_mcp/workspaces.py +212 -0
dw/loudness.py ADDED
@@ -0,0 +1,82 @@
1
+ """BS.1770 loudness measurement, shared by the media probe and the audio
2
+ tasks so both read a level the same way.
3
+
4
+ Peak and loudness are different quantities - a sparse voice-over and a dense
5
+ score can share a peak and still sit tens of dB apart in how loud they sound,
6
+ because a peak is one sample and loudness is measured over the whole track
7
+ (#361). `integrated_lufs` is the BS.1770 integrated measure pyloudnorm
8
+ implements; `true_peak_dbfs` is the inter-sample peak BS.1770 defines
9
+ alongside it - a sample-peak reading can miss a peak that only appears
10
+ between samples, which is what an encoder's reconstruction filter can ring
11
+ up past 0 dBFS even when every decoded sample was under it.
12
+ """
13
+
14
+ import logging
15
+ import math
16
+
17
+ import numpy
18
+ import pyloudnorm
19
+ import scipy.signal
20
+
21
+ logger = logging.getLogger("dw")
22
+
23
+ # The floor a level is reported at rather than -inf, which JSON cannot carry
24
+ SILENCE_DBFS = -120.0
25
+
26
+ # BS.1770's gating block is 400 ms; pyloudnorm refuses anything shorter
27
+ MIN_LUFS_SECONDS = 0.4
28
+
29
+ # Minimum oversampling BS.1770 defines for a true-peak measurement
30
+ TRUE_PEAK_OVERSAMPLE = 4
31
+
32
+
33
+ def _dbfs(value):
34
+ if value <= 0:
35
+ return SILENCE_DBFS
36
+ return max(SILENCE_DBFS, 20.0 * math.log10(float(value)))
37
+
38
+
39
+ def integrated_lufs(samples, rate):
40
+ """Integrated loudness in LUFS, or None when it cannot be measured.
41
+
42
+ `samples` is (frames, channels) or (frames,). None for a clip shorter
43
+ than the 400 ms gating block, an all-silent clip (pyloudnorm's own
44
+ -inf, which JSON cannot carry either), or anything pyloudnorm refuses -
45
+ a measurement that failed reads as "unknown" rather than as a crashed
46
+ task or probe.
47
+ """
48
+ if samples is None or samples.size == 0:
49
+ return None
50
+ frames = samples.shape[0]
51
+ if frames < int(round(MIN_LUFS_SECONDS * rate)):
52
+ return None
53
+ try:
54
+ meter = pyloudnorm.Meter(rate)
55
+ value = float(meter.integrated_loudness(samples))
56
+ except Exception as e:
57
+ logger.debug(f"integrated_lufs: could not measure: {e}")
58
+ return None
59
+ if not math.isfinite(value):
60
+ return None
61
+ return value
62
+
63
+
64
+ def true_peak_dbfs(samples, oversample=TRUE_PEAK_OVERSAMPLE):
65
+ """The inter-sample (true) peak of a track, in dBFS.
66
+
67
+ `samples` is (frames, channels) or (frames,). Oversamples with a
68
+ polyphase FIR (BS.1770's own reconstruction) and reads the peak of the
69
+ interpolated signal, which is what a lossy encoder's own reconstruction
70
+ filter can ring up past a sample-peak reading that stayed under 0 dBFS.
71
+ SILENCE_DBFS for an empty track, never None - unlike integrated
72
+ loudness, a true peak is defined for any track, including a silent one.
73
+ """
74
+ if samples is None or samples.size == 0:
75
+ return SILENCE_DBFS
76
+ try:
77
+ oversampled = scipy.signal.resample_poly(samples, oversample, 1, axis=0)
78
+ except Exception as e:
79
+ logger.debug(f"true_peak_dbfs: could not oversample: {e}")
80
+ oversampled = samples
81
+ peak = float(numpy.abs(oversampled).max(initial=0.0))
82
+ return _dbfs(peak)
dw/media_audio.py ADDED
@@ -0,0 +1,217 @@
1
+ """A soundtrack, or a named slice of one, as WAV bytes - the form
2
+ get_output_audio hands an agent for a muxed video (whose container it
3
+ cannot play) or for a track too long to send whole (#193).
4
+
5
+ PyAV only, like media_info: this runs in the server process, where a
6
+ request must not pull in torch or materialise frames.
7
+ """
8
+
9
+ import io
10
+ import logging
11
+ import math
12
+ import wave
13
+
14
+ import av
15
+ import numpy
16
+ from av.audio.resampler import AudioResampler
17
+
18
+ logger = logging.getLogger("dw")
19
+
20
+ # The most a soundtrack may be as base64 before the gallery route refuses to
21
+ # extract it whole - the twin of dw_mcp/media.py's MAX_RETURNED_BYTES (the
22
+ # MCP package's cap on any inline payload). Two constants because the two
23
+ # packages do not import each other; a whole track over this is cut off at
24
+ # the header, before a frame is decoded, rather than decoded, shipped and
25
+ # then refused by the client.
26
+ MAX_INLINE_AUDIO_BYTES = 4 * 1024 * 1024
27
+
28
+
29
+ class NoSoundtrack(ValueError):
30
+ """The file has no audio stream to extract."""
31
+
32
+
33
+ def _container_duration(container):
34
+ """The container's own duration (seconds) from its header alone - no
35
+ stream is decoded to produce this number."""
36
+ return (
37
+ float(container.duration / av.time_base)
38
+ if container.duration is not None
39
+ else None
40
+ )
41
+
42
+
43
+ def media_duration(path):
44
+ """The container's duration (seconds), read from its header alone - no
45
+ stream is decoded. `extract_audio` needs the same figure as `total`;
46
+ this is that computation, exposed for a caller that wants only the
47
+ length, since decoding to get one number defeats the "nothing to
48
+ extract" case (serving an audio file whole - #193)."""
49
+ with av.open(path) as container:
50
+ return _container_duration(container)
51
+
52
+
53
+ def audio_shape(path):
54
+ """The soundtrack's duration (seconds), sample rate and channel count
55
+ from the container's headers alone - what projecting the size of a
56
+ whole-track WAV needs, and nothing decoded to get it. `None` when the
57
+ file has no audio stream."""
58
+ with av.open(path) as container:
59
+ if not container.streams.audio:
60
+ return None
61
+ stream = container.streams.audio[0]
62
+ return {
63
+ "duration_seconds": _container_duration(container),
64
+ "sample_rate": int(stream.rate),
65
+ "channels": int(stream.channels),
66
+ }
67
+
68
+
69
+ def projected_wav_base64_size(shape):
70
+ """How many bytes the whole track would be as base64 16-bit PCM WAV -
71
+ `extract_audio`'s output for the same file, sized from `audio_shape`
72
+ without producing it. `None` when the container states no duration."""
73
+ if shape is None or shape["duration_seconds"] is None:
74
+ return None
75
+ pcm = int(shape["duration_seconds"] * shape["sample_rate"] * shape["channels"] * 2)
76
+ return 4 * math.ceil(pcm / 3)
77
+
78
+
79
+ def extract_audio(path, start=None, duration=None):
80
+ """The soundtrack of `path` as 16-bit PCM WAV bytes, plus what was cut.
81
+
82
+ With `start` and `duration` (seconds) only that slice is decoded and
83
+ returned; the info dict says so (`excerpt: True`) and carries the whole
84
+ track's length as `of_seconds`, so a slice always names itself. A
85
+ slice that runs past the end is clipped to it; a `start` past the end
86
+ is refused, since an empty answer would read as a silent track.
87
+
88
+ Returns:
89
+ (wav_bytes, info) with info = {sample_rate, channels,
90
+ duration_seconds, of_seconds, start, excerpt}
91
+ """
92
+ excerpt = start is not None or duration is not None
93
+ start = float(start or 0.0)
94
+ if excerpt and (duration is None or float(duration) <= 0):
95
+ raise ValueError("An excerpt needs a duration above zero")
96
+ if start < 0:
97
+ raise ValueError("An excerpt cannot start before zero")
98
+
99
+ with av.open(path) as container:
100
+ if not container.streams.audio:
101
+ raise NoSoundtrack(f"{path} has no soundtrack")
102
+ stream = container.streams.audio[0]
103
+ total = _container_duration(container)
104
+ if total is not None and start >= total:
105
+ raise ValueError(
106
+ f"start {start:.2f}s is past the end of a {total:.2f}s track"
107
+ )
108
+ stop = start + float(duration) if excerpt else None
109
+ if stop is not None and total is not None:
110
+ stop = min(stop, total)
111
+
112
+ rate = int(stream.rate)
113
+ channels = int(stream.channels)
114
+ layout = (
115
+ "stereo"
116
+ if channels == 2
117
+ else ("mono" if channels == 1 else stream.layout.name)
118
+ )
119
+ resampler = AudioResampler(format="s16", layout=layout, rate=rate)
120
+
121
+ # pts is a timestamp on the container's clock, not an offset from
122
+ # this stream's first sample: a stream with an edit list or a
123
+ # non-zero start (which real muxers write) carries a `start_time`,
124
+ # and both the seek target and each frame's clock have to be
125
+ # zeroed against it before they mean seconds into the track -
126
+ # the same anchor `_read_frames` in media_frames uses. Unanchored,
127
+ # `start=2.5` on a track shifted by one second came back from
128
+ # about 1.5 s: the wrong audio, silently.
129
+ anchor = stream.start_time if stream.start_time is not None else 0
130
+ if start > 0:
131
+ # Seek to the keyframe at or before `start`, less one packet of
132
+ # pre-roll: the first packet a decoder sees after a seek is its
133
+ # warm-up, and for AAC and mp3 the samples it yields come out
134
+ # attenuated or silent - so an excerpt at 2.5 s opened with a
135
+ # gap that is not in the file. Landing a packet early hands that
136
+ # warm-up to samples the pts-based trim below drops anyway.
137
+ frame_size = int(stream.codec_context.frame_size or 0)
138
+ preroll = frame_size / rate if frame_size else 0.1
139
+ container.seek(
140
+ int(max(0.0, start - preroll) / stream.time_base) + anchor,
141
+ stream=stream,
142
+ backward=True,
143
+ )
144
+
145
+ pieces = []
146
+ seen = 0 # samples decoded so far - the clock for a frame with no pts
147
+ started = False
148
+ done = False # Break outer loop when stop time is reached
149
+ sought = start > 0
150
+ restarted = False
151
+ for frame in container.decode(stream):
152
+ if done:
153
+ break
154
+ if frame.pts is None and sought and not restarted:
155
+ # No pts means the seek's landing cannot be known, so the
156
+ # sample count is the only clock there is - and it has to
157
+ # count from the top. Start over once and read to `start`.
158
+ container.seek(0, stream=stream, backward=True)
159
+ resampler = AudioResampler(format="s16", layout=layout, rate=rate)
160
+ restarted = True
161
+ seen = 0
162
+ continue
163
+ frame_start = (
164
+ float((frame.pts - anchor) * stream.time_base)
165
+ if frame.pts is not None
166
+ else seen / rate
167
+ )
168
+ for chunk in resampler.resample(frame):
169
+ samples = chunk.to_ndarray() # (1, samples * channels) packed s16
170
+ samples = samples.reshape(-1, channels)
171
+ chunk_start = frame_start
172
+ chunk_end = chunk_start + samples.shape[0] / rate
173
+ seen += samples.shape[0]
174
+ frame_start = chunk_end # a frame yielding two chunks: the second follows the first
175
+ if chunk_end <= start:
176
+ continue
177
+ if not started and chunk_start < start:
178
+ # round, not floor: float pts arithmetic lands a hair
179
+ # under the exact sample and floor then keeps one too many
180
+ samples = samples[int(round((start - chunk_start) * rate)) :]
181
+ chunk_start = start
182
+ started = True
183
+ if stop is not None and chunk_end > stop:
184
+ samples = samples[: max(0, int(round((stop - chunk_start) * rate)))]
185
+ pieces.append(samples)
186
+ if stop is not None and chunk_end >= stop:
187
+ done = True
188
+ break
189
+ # Flush the resampler whatever ended the loop: it may still hold
190
+ # samples that belong before `stop`. Anything past `stop` is cut.
191
+ held = sum(p.shape[0] for p in pieces)
192
+ for chunk in resampler.resample(None):
193
+ samples = chunk.to_ndarray().reshape(-1, channels)
194
+ if stop is not None:
195
+ room = max(0, int(round((stop - start) * rate)) - held)
196
+ samples = samples[:room]
197
+ if samples.shape[0]:
198
+ pieces.append(samples)
199
+ held += samples.shape[0]
200
+
201
+ pcm = numpy.concatenate(pieces) if pieces else numpy.zeros((0, channels), "<i2")
202
+ buffer = io.BytesIO()
203
+ with wave.open(buffer, "w") as handle:
204
+ handle.setnchannels(channels)
205
+ handle.setsampwidth(2)
206
+ handle.setframerate(rate)
207
+ handle.writeframes(pcm.astype("<i2").tobytes())
208
+
209
+ returned = pcm.shape[0] / rate
210
+ return buffer.getvalue(), {
211
+ "sample_rate": rate,
212
+ "channels": channels,
213
+ "duration_seconds": returned,
214
+ "of_seconds": total if total is not None else returned,
215
+ "start": start,
216
+ "excerpt": excerpt,
217
+ }
dw/media_frames.py ADDED
@@ -0,0 +1,367 @@
1
+ """Frames out of a video file by seeking to them - a moment, an evenly
2
+ spaced contact sheet, or the frame pair either side of a seam - without
3
+ decoding the clip whole. The server process runs this per request; a
4
+ four-minute 1080p clip materialised as PIL frames is tens of gigabytes,
5
+ so nothing here ever holds more than the frames it returns, each already
6
+ fitted to its tile where a caller asked for many (#193).
7
+ """
8
+
9
+ import logging
10
+ import math
11
+
12
+ import av
13
+ import numpy
14
+ from PIL import Image
15
+
16
+ from .tasks.video_utils import (
17
+ _compose_grid,
18
+ _default_columns,
19
+ _evenly_spaced_indices,
20
+ _format_timestamp,
21
+ _grid_tile,
22
+ )
23
+
24
+ logger = logging.getLogger("dw")
25
+
26
+ # Each cell of a contact sheet and each side of a seam pair is a seek, a
27
+ # decode and a resize in the server process; a sheet past 64 cells is
28
+ # unreadable anyway, and `at` has its own cap in the route
29
+ # (MAX_FRAME_MOMENTS). Both are ValueErrors, so the route answers 400.
30
+ MAX_CONTACT_SHEET_FRAMES = 64
31
+ MAX_SEAMS = 32
32
+
33
+
34
+ def video_shape(path):
35
+ """Frame count, fps and size, from the container's own headers where
36
+ they are written and by counting otherwise."""
37
+ with av.open(path) as container:
38
+ if not container.streams.video:
39
+ raise ValueError(f"{path} has no video stream")
40
+ stream = container.streams.video[0]
41
+ fps = float(stream.average_rate) if stream.average_rate else None
42
+ count = int(stream.frames) if stream.frames else None
43
+ if count is None:
44
+ count = sum(1 for _ in container.decode(stream))
45
+ return {
46
+ "frame_count": count,
47
+ "fps": fps,
48
+ "width": int(stream.width),
49
+ "height": int(stream.height),
50
+ }
51
+
52
+
53
+ def resolve_crop_box(crop, width, height):
54
+ """`[x, y, w, h]` as Pillow's `(left, upper, right, lower)`, clamped to
55
+ a frame. Every frame of one video shares the same dimensions, so a
56
+ caller resolves this once against `video_shape(path)` and reuses it
57
+ across every tile - the same `[x, y, width, height]` convention
58
+ `get_output_image`'s crop uses. Refused when it is not four
59
+ non-negative integers, starts outside the frame, or has nothing in it."""
60
+ try:
61
+ x, y, w, h = (int(v) for v in crop)
62
+ except (TypeError, ValueError):
63
+ raise ValueError(f"crop must be [x, y, width, height] in pixels, got {crop!r}.")
64
+ if x < 0 or y < 0 or w <= 0 or h <= 0:
65
+ raise ValueError(
66
+ f"crop must have a non-negative origin and a positive size, got {crop!r}."
67
+ )
68
+ if x >= width or y >= height:
69
+ raise ValueError(
70
+ f"crop origin ({x}, {y}) lies outside the {width}x{height} frame."
71
+ )
72
+ return (x, y, min(x + w, width), min(y + h, height))
73
+
74
+
75
+ def frames_at(path, moments, shape=None, crop_box=None):
76
+ """One tile per moment - a float in seconds or "frame:N" - in the order
77
+ asked for. Each tile is {label, frame, seconds, image}.
78
+
79
+ `shape` reuses an already-computed `video_shape(path)` - a caller such
80
+ as the gallery route that also reports the clip's own frame_count/fps
81
+ would otherwise pay for `video_shape`'s container open (and, lacking a
82
+ header frame count, a full decode) a second time for the same answer.
83
+
84
+ `crop_box` (`resolve_crop_box`'s Pillow box) is cut from each frame
85
+ right after decode, before it is returned - full source resolution,
86
+ not whatever a caller's `max_dimension` later downscales it to."""
87
+ shape = shape if shape is not None else video_shape(path)
88
+ indexes = [_moment_to_index(moment, shape) for moment in moments]
89
+ fit = (lambda image, index: image.crop(crop_box)) if crop_box else None
90
+ images = _read_frames(path, indexes, fit=fit)
91
+ return [_tile(index, images[index], shape) for index in indexes]
92
+
93
+
94
+ def contact_sheet(path, count, tile_width=320, shape=None, crop_box=None):
95
+ """N evenly spaced frames (first and last included) tiled into one
96
+ image - frame_grid without a workflow. `shape` reuses an
97
+ already-computed `video_shape(path)`; see `frames_at`.
98
+
99
+ `crop_box`, as in `frames_at`, is cut from each frame before it is
100
+ stamped and tiled into the sheet - the sheet is built from cropped
101
+ frames rather than cropped after assembly, so the box means the same
102
+ source-pixel region whatever the sheet's own `tile_width` ends up."""
103
+ if int(count) < 1:
104
+ raise ValueError("count must be at least 1")
105
+ shape = shape if shape is not None else video_shape(path)
106
+ # Clamp to the clip's own length before checking the cap: a count that
107
+ # would only ever produce a handful of cells (a short clip) should not
108
+ # be refused for the raw number the caller asked for.
109
+ count = min(int(count), shape["frame_count"])
110
+ if count > MAX_CONTACT_SHEET_FRAMES:
111
+ raise ValueError(
112
+ f"count {count} is more than a contact sheet holds "
113
+ f"({MAX_CONTACT_SHEET_FRAMES}); ask for a smaller one, or `at` for moments"
114
+ )
115
+ indexes = _evenly_spaced_indices(shape["frame_count"], count)
116
+
117
+ def fit(image, index):
118
+ if crop_box:
119
+ image = image.crop(crop_box)
120
+ return _stamped_tile(image, index, shape["fps"], tile_width)
121
+
122
+ images = _read_frames(path, indexes, fit=fit)
123
+ tiles = [images[index] for index in indexes]
124
+ grid = _compose_grid(tiles, _default_columns(len(tiles)))
125
+ return {
126
+ "label": f"contact sheet, {len(tiles)} frames",
127
+ "frame": indexes[0],
128
+ "seconds": _seconds(indexes[0], shape),
129
+ "image": grid,
130
+ "frames": list(indexes),
131
+ }
132
+
133
+
134
+ def seam_tiles(
135
+ path, boundaries, names=None, tile_width=320, shape=None, wanted=None, crop_box=None
136
+ ):
137
+ """For each boundary (the frame index a shot *starts* at), the last
138
+ frame before it and the first frame at it, side by side - the seam and
139
+ continuity evidence in one image. Seam i sits between shot i and shot
140
+ i+1; `names` names the shots, "shot 1".. by default.
141
+
142
+ `boundaries` and `names` are validated in full regardless of `wanted` -
143
+ they describe the whole cut, and a seam's label names the shots either
144
+ side of it by position in that full list. `wanted` (a set of 1-based
145
+ seam numbers, or None for all) then limits which seams are actually
146
+ decoded and composed: a caller after seam 2 of twelve should not pay to
147
+ decode and tile the other eleven. `shape` reuses an already-computed
148
+ `video_shape(path)`; see `frames_at`.
149
+
150
+ `crop_box`, as in `frames_at`, is cut from each source frame before it
151
+ is fit to `tile_width` and paired into a seam - the pair is built from
152
+ cropped frames rather than cropped after pairing."""
153
+ shape = shape if shape is not None else video_shape(path)
154
+ total = shape["frame_count"]
155
+ for boundary in boundaries:
156
+ if not 1 <= int(boundary) <= total - 1:
157
+ raise ValueError(
158
+ f"boundary {boundary} is not inside the clip (1..{total - 1})"
159
+ )
160
+ boundaries = [int(b) for b in boundaries]
161
+ names = list(names or [f"shot {n + 1}" for n in range(len(boundaries) + 1)])
162
+ if len(names) != len(boundaries) + 1:
163
+ raise ValueError(
164
+ f"{len(boundaries)} boundaries make {len(boundaries) + 1} shots, "
165
+ f"but {len(names)} names were given"
166
+ )
167
+ if wanted is not None:
168
+ off = sorted(int(s) for s in wanted if not 1 <= int(s) <= len(boundaries))
169
+ if off:
170
+ raise ValueError(
171
+ f"seam {', '.join(str(s) for s in off)} is not in this cut - "
172
+ f"{len(boundaries)} boundaries make seams 1..{len(boundaries)}"
173
+ )
174
+ chosen = [
175
+ (seam, boundary)
176
+ for seam, boundary in enumerate(boundaries, start=1)
177
+ if wanted is None or seam in wanted
178
+ ]
179
+ if len(chosen) > MAX_SEAMS:
180
+ raise ValueError(
181
+ f"{len(chosen)} seams is more than one call serves ({MAX_SEAMS}); "
182
+ "name the seams wanted (`seams=1,2,...`)"
183
+ )
184
+ frame_indexes = sorted({b - 1 for _, b in chosen} | {b for _, b in chosen})
185
+
186
+ def fit(image, _index):
187
+ if crop_box:
188
+ image = image.crop(crop_box)
189
+ return _fit_width(image, tile_width)
190
+
191
+ images = _read_frames(path, frame_indexes, fit=fit)
192
+ tiles = []
193
+ for seam, boundary in chosen:
194
+ before = images[boundary - 1]
195
+ after = images[boundary]
196
+ pair = _compose_grid([before, after], 2)
197
+ difference = float(
198
+ numpy.abs(
199
+ numpy.asarray(before, dtype=numpy.int16)
200
+ - numpy.asarray(after, dtype=numpy.int16)
201
+ ).mean()
202
+ )
203
+ tiles.append(
204
+ {
205
+ "label": f"seam {seam}: {names[seam - 1]} | {names[seam]}",
206
+ "frame": boundary,
207
+ "seconds": _seconds(boundary, shape),
208
+ "difference": round(difference, 2),
209
+ "image": pair,
210
+ }
211
+ )
212
+ return tiles
213
+
214
+
215
+ def _moment_to_index(moment, shape):
216
+ total = shape["frame_count"]
217
+ if isinstance(moment, str) and moment.startswith("frame:"):
218
+ raw = moment[len("frame:") :]
219
+ try:
220
+ index = int(raw)
221
+ except ValueError:
222
+ raise ValueError(f'"{moment}" - frame index must be a whole number')
223
+ if index < 0:
224
+ index += total
225
+ if not 0 <= index < total:
226
+ raise ValueError(
227
+ f"frame {index} is past the end of a {total}-frame clip "
228
+ f"(frames 0-{total - 1})"
229
+ )
230
+ return index
231
+ fps = shape["fps"]
232
+ if fps is None:
233
+ raise ValueError("this clip has no frame rate, so name a frame: 'frame:N'")
234
+ seconds = float(moment)
235
+ if not math.isfinite(seconds):
236
+ raise ValueError(f"{moment!r} is not a moment in seconds")
237
+ index = int(round(seconds * fps))
238
+ if index < 0:
239
+ index += total
240
+ if not 0 <= index < total:
241
+ duration = total / fps
242
+ message = (
243
+ f"{moment!r} s is past the end of a {duration:.2f} s "
244
+ f"({total}-frame) clip - bare numbers in 'at' are seconds"
245
+ )
246
+ # A fractional value is a seconds overshoot, not a frame index in
247
+ # disguise - the frame: hint only makes sense for a whole number
248
+ # that would itself be a valid frame index.
249
+ if seconds.is_integer() and 0 <= int(seconds) < total:
250
+ message += f'; use "frame:{_format_moment(moment)}" for a frame index'
251
+ raise ValueError(message)
252
+ return index
253
+
254
+
255
+ def _format_moment(moment):
256
+ if isinstance(moment, float) and moment.is_integer():
257
+ return str(int(moment))
258
+ return str(moment)
259
+
260
+
261
+ def _seconds(index, shape):
262
+ return index / shape["fps"] if shape["fps"] else float(index)
263
+
264
+
265
+ def _tile(index, image, shape):
266
+ fps = shape["fps"]
267
+ stamp = _format_timestamp(index, fps) if fps else f"#{index}"
268
+ return {
269
+ "label": f"{stamp} (frame {index})",
270
+ "frame": index,
271
+ "seconds": _seconds(index, shape),
272
+ "image": image,
273
+ }
274
+
275
+
276
+ def _stamped_tile(image, index, fps, tile_width):
277
+ """A contact-sheet cell: fitted like `_fit_width` (never upscaled) and
278
+ stamped with its timestamp by `frame_grid`'s own tile maker, so the
279
+ sheet says which cell is which without the text part."""
280
+ return _grid_tile(image, index, fps, min(int(tile_width), image.width), label=True)
281
+
282
+
283
+ def _fit_width(image, tile_width):
284
+ # never upscaled: a tile wider than its source is a blurred enlargement
285
+ # that costs bytes and shows nothing the source holds
286
+ tile_width = min(int(tile_width), image.width)
287
+ height = max(1, round(image.height * tile_width / image.width))
288
+ return image.resize((tile_width, height), Image.LANCZOS).convert("RGB")
289
+
290
+
291
+ def _read_frames(path, indexes, fit=None):
292
+ """The frames at these indexes, as {index: PIL image}, in one forward
293
+ pass that seeks to the keyframe before each wanted frame rather than
294
+ decoding from the top. Decodes are dropped as soon as they are past.
295
+
296
+ `fit(image, index)`, when given, is applied to each frame as it is
297
+ decoded, so a caller tiling many frames never holds one at source size.
298
+
299
+ This assumes a constant frame rate, which every file this engine writes
300
+ has (`encode_video` / `export_to_video` write a fixed `fps`): a
301
+ `backward=True` seek lands on the keyframe at or before the target
302
+ timestamp, and `frame.pts * stream.time_base` converts that keyframe's
303
+ own presentation time back to an exact frame index (`round(seconds *
304
+ fps)`) - verified against PyAV 18.1's actual seek landings (a
305
+ single-keyframe short clip, where every seek lands on frame 0; a `g=10`
306
+ multi-keyframe clip, where a seek to frame 95 lands exactly on frame 90;
307
+ and a clip whose packets carry a 5-frame pts offset - an edit list or a
308
+ non-zero start, which real muxers write - where the raw pts arithmetic
309
+ landed 5 frames off until it was anchored on `stream.start_time`).
310
+ `start_pts` is that anchor: pts is a timestamp against the *container's*
311
+ clock, not a frame count from this stream's first frame, so it has to be
312
+ zeroed against wherever this stream actually starts before it means a
313
+ frame index. `position` is then a plain frame counter from that
314
+ landing, decoding forward to the target and dropping what is skipped
315
+ past.
316
+ """
317
+ wanted = sorted(set(int(i) for i in indexes))
318
+ found = {}
319
+ with av.open(path) as container:
320
+ stream = container.streams.video[0]
321
+ fps = float(stream.average_rate) if stream.average_rate else None
322
+ start_pts = stream.start_time if stream.start_time is not None else 0
323
+ position = 0 # index of the next frame decode() will yield
324
+ for target in wanted:
325
+ if target < position or target - position > 2 * (int(fps) if fps else 24):
326
+ # seek back or a long way forward: land on the keyframe at
327
+ # or before the target, then read up to it. The seek target
328
+ # is a container timestamp too, so it needs the same anchor.
329
+ seconds = target / fps if fps else 0.0
330
+ container.seek(
331
+ int(seconds / stream.time_base) + start_pts,
332
+ stream=stream,
333
+ backward=True,
334
+ )
335
+ position = None
336
+ recovered = False
337
+ for frame in container.decode(stream):
338
+ if position is None:
339
+ # first frame after a seek says where we landed
340
+ position = (
341
+ int(
342
+ round(
343
+ float((frame.pts - start_pts) * stream.time_base) * fps
344
+ )
345
+ )
346
+ if fps and frame.pts is not None
347
+ else 0
348
+ )
349
+ if position > target and not recovered:
350
+ # Landed past the target: the keyframe estimate was
351
+ # wrong for this file (off-rate or VFR). Reading on
352
+ # would scan to EOF and blame the caller; read from
353
+ # the top once instead, which is always correct.
354
+ container.seek(start_pts, stream=stream, backward=True)
355
+ position = None
356
+ recovered = True
357
+ continue
358
+ if position == target:
359
+ image = frame.to_image()
360
+ found[target] = fit(image, target) if fit is not None else image
361
+ position += 1
362
+ break
363
+ position += 1
364
+ missing = [i for i in wanted if i not in found]
365
+ if missing:
366
+ raise ValueError(f"could not decode frame(s) {missing} of {path}")
367
+ return found