shotdrift 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- shotdrift/__init__.py +25 -0
- shotdrift/cli.py +95 -0
- shotdrift/core.py +236 -0
- shotdrift/expect.py +132 -0
- shotdrift/frames.py +135 -0
- shotdrift/ingest.py +109 -0
- shotdrift/motion.py +220 -0
- shotdrift/path.py +312 -0
- shotdrift/verdict.py +218 -0
- shotdrift-0.2.0.dist-info/METADATA +283 -0
- shotdrift-0.2.0.dist-info/RECORD +15 -0
- shotdrift-0.2.0.dist-info/WHEEL +5 -0
- shotdrift-0.2.0.dist-info/entry_points.txt +2 -0
- shotdrift-0.2.0.dist-info/licenses/LICENSE +21 -0
- shotdrift-0.2.0.dist-info/top_level.txt +1 -0
shotdrift/__init__.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""shotdrift - measure whether a video holds the camera move it was given.
|
|
2
|
+
|
|
3
|
+
AI video is asked for a camera move and is not obliged to deliver one. The usual
|
|
4
|
+
check is a person watching sixty clips. This measures the camera path out of the
|
|
5
|
+
pixels - pan, zoom, roll, frame by frame - and reports whether one physical camera
|
|
6
|
+
could have produced it.
|
|
7
|
+
|
|
8
|
+
from shotdrift import measure
|
|
9
|
+
r = measure("take_07.mp4", expect="push-in")
|
|
10
|
+
print(r.verdict, [f.code for f in r.findings])
|
|
11
|
+
|
|
12
|
+
Frames already in memory - inside a generation graph, say, where the clip was
|
|
13
|
+
never written to disk - skip ffmpeg entirely:
|
|
14
|
+
|
|
15
|
+
from shotdrift import measure_frames, report
|
|
16
|
+
r = measure_frames(image_batch, expect="push-in") # (n, h, w, c), 0..1
|
|
17
|
+
print(report(r))
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from .core import Result, Shot, measure, measure_frames, report
|
|
23
|
+
|
|
24
|
+
__version__ = "0.2.0"
|
|
25
|
+
__all__ = ["measure", "measure_frames", "report", "Result", "Shot", "__version__"]
|
shotdrift/cli.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""shotdrift - measure whether a video holds the camera move it was given."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from . import __version__
|
|
10
|
+
from .core import measure, report
|
|
11
|
+
from .expect import known as known_moves
|
|
12
|
+
from .frames import DecodeError, FFmpegMissing
|
|
13
|
+
from .verdict import (BROKEN, CLEAN, SOFT, UNKNOWN, at_or_above)
|
|
14
|
+
|
|
15
|
+
EXIT_OK, EXIT_FINDINGS, EXIT_ERROR, EXIT_UNMEASURABLE = 0, 1, 2, 3
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def main(argv=None) -> int:
|
|
19
|
+
ap = argparse.ArgumentParser(
|
|
20
|
+
prog="shotdrift",
|
|
21
|
+
description="Measure the camera move in a video, and whether it is one a real "
|
|
22
|
+
"camera could have made.",
|
|
23
|
+
epilog="Exit 0 clean, 1 findings at or above --min-severity, "
|
|
24
|
+
"2 could not run, 3 clip could not be measured.",
|
|
25
|
+
)
|
|
26
|
+
ap.add_argument("clips", nargs="*", metavar="CLIP")
|
|
27
|
+
ap.add_argument("--expect", metavar="MOVE",
|
|
28
|
+
help="the move you asked for: " + ", ".join(known_moves()))
|
|
29
|
+
ap.add_argument("--list-moves", action="store_true")
|
|
30
|
+
ap.add_argument("--min-severity", choices=[CLEAN, SOFT, BROKEN], default=BROKEN,
|
|
31
|
+
help="which findings make this exit non-zero (default: broken; "
|
|
32
|
+
"soft findings are always printed either way)")
|
|
33
|
+
ap.add_argument("--max-side", type=int, default=512,
|
|
34
|
+
help="analysis resolution, long edge in px (default 512)")
|
|
35
|
+
ap.add_argument("--grid", type=int, default=4, help="tiles per axis (default 4)")
|
|
36
|
+
ap.add_argument("--start", type=float, default=0.0, metavar="SEC")
|
|
37
|
+
ap.add_argument("--duration", type=float, default=None, metavar="SEC")
|
|
38
|
+
ap.add_argument("--sample-fps", type=float, default=None,
|
|
39
|
+
help="resample before measuring; changes WHAT is measured, "
|
|
40
|
+
"not just the cost")
|
|
41
|
+
ap.add_argument("--max-frames", type=int, default=600)
|
|
42
|
+
ap.add_argument("--anchors", type=int, default=4,
|
|
43
|
+
help="independent first-to-frame-k checks (0 disables closure)")
|
|
44
|
+
ap.add_argument("--no-segment", action="store_true",
|
|
45
|
+
help="measure the file as one take even if it contains cuts")
|
|
46
|
+
ap.add_argument("--json", action="store_true", help="machine-readable output")
|
|
47
|
+
ap.add_argument("--quiet", action="store_true", help="findings without the advice")
|
|
48
|
+
ap.add_argument("--version", action="version", version=f"shotdrift {__version__}")
|
|
49
|
+
args = ap.parse_args(argv)
|
|
50
|
+
|
|
51
|
+
if args.list_moves:
|
|
52
|
+
for m in known_moves():
|
|
53
|
+
print(m)
|
|
54
|
+
return EXIT_OK
|
|
55
|
+
if not args.clips:
|
|
56
|
+
ap.error("give at least one CLIP, or --list-moves")
|
|
57
|
+
|
|
58
|
+
if args.expect and args.expect not in known_moves():
|
|
59
|
+
ap.error(f"unknown move {args.expect!r}. --list-moves shows them all.")
|
|
60
|
+
|
|
61
|
+
results, rc = [], EXIT_OK
|
|
62
|
+
for name in args.clips:
|
|
63
|
+
try:
|
|
64
|
+
r = measure(name, expect=args.expect, max_side=args.max_side,
|
|
65
|
+
grid=args.grid, start=args.start, duration=args.duration,
|
|
66
|
+
sample_fps=args.sample_fps, anchors=args.anchors,
|
|
67
|
+
max_frames=args.max_frames,
|
|
68
|
+
segment=not args.no_segment)
|
|
69
|
+
except FFmpegMissing as e:
|
|
70
|
+
print(f"shotdrift: {e}", file=sys.stderr)
|
|
71
|
+
return EXIT_ERROR
|
|
72
|
+
except (DecodeError, ValueError) as e:
|
|
73
|
+
print(f"shotdrift: {name}: {e}", file=sys.stderr)
|
|
74
|
+
rc = max(rc, EXIT_ERROR)
|
|
75
|
+
continue
|
|
76
|
+
|
|
77
|
+
for s in r.shots:
|
|
78
|
+
if s.verdict == UNKNOWN:
|
|
79
|
+
rc = max(rc, EXIT_UNMEASURABLE)
|
|
80
|
+
elif at_or_above(s.findings, args.min_severity) or \
|
|
81
|
+
(s.expect is not None and not s.expect.ok):
|
|
82
|
+
rc = max(rc, EXIT_FINDINGS)
|
|
83
|
+
|
|
84
|
+
if args.json:
|
|
85
|
+
results.append(r.as_dict())
|
|
86
|
+
else:
|
|
87
|
+
sys.stdout.write(report(r, quiet=args.quiet))
|
|
88
|
+
|
|
89
|
+
if args.json:
|
|
90
|
+
print(json.dumps({"shotdrift": __version__, "clips": results}, indent=2))
|
|
91
|
+
return rc
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
if __name__ == "__main__":
|
|
95
|
+
sys.exit(main())
|
shotdrift/core.py
ADDED
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""One measurement, however the frames arrived, and one way of reporting it.
|
|
2
|
+
|
|
3
|
+
Both halves of this file are corrections.
|
|
4
|
+
|
|
5
|
+
`measure()` used to call `analyse` where the command line calls `analyse_clip`,
|
|
6
|
+
so the Python API did not segment. On an edited reel that is not a small
|
|
7
|
+
difference: measured as one take, a six-shot sequence reports a broken closure
|
|
8
|
+
and a pile of reversals, all true and all useless, because no single camera move
|
|
9
|
+
was ever there to hold. The command line was fixed for exactly that and the
|
|
10
|
+
library was left behind - so `import shotdrift` gave a worse answer than
|
|
11
|
+
`shotdrift` did, which nothing warned anyone about.
|
|
12
|
+
|
|
13
|
+
And the text report lived inside the command line, writing straight to stdout.
|
|
14
|
+
Anything else wanting the same words - a node in a generation graph, a CI
|
|
15
|
+
comment - had to reimplement them, which is how two surfaces of one tool come to
|
|
16
|
+
disagree about what a clip measured. It renders to a string here, and the command
|
|
17
|
+
line prints what it returns.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
|
|
24
|
+
import numpy as np
|
|
25
|
+
|
|
26
|
+
from .verdict import BROKEN, CLEAN, SOFT, UNKNOWN, at_or_above
|
|
27
|
+
|
|
28
|
+
__all__ = ["Shot", "Result", "measure", "measure_frames", "report"]
|
|
29
|
+
|
|
30
|
+
_MARK = {CLEAN: "ok", SOFT: "soft", BROKEN: "BROKEN", UNKNOWN: "unmeasurable"}
|
|
31
|
+
_RANK = {CLEAN: 0, SOFT: 1, BROKEN: 2, UNKNOWN: 0}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class Shot:
|
|
36
|
+
"""One continuous take, measured on its own."""
|
|
37
|
+
start: int
|
|
38
|
+
end: int
|
|
39
|
+
path: object = field(repr=False)
|
|
40
|
+
findings: list = field(default_factory=list)
|
|
41
|
+
verdict: str = CLEAN
|
|
42
|
+
expect: object | None = None
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def frames(self) -> int:
|
|
46
|
+
return self.end - self.start
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def ok(self) -> bool:
|
|
50
|
+
return self.verdict == CLEAN and (self.expect is None or self.expect.ok)
|
|
51
|
+
|
|
52
|
+
def as_dict(self) -> dict:
|
|
53
|
+
return {
|
|
54
|
+
"frames": [self.start, self.end],
|
|
55
|
+
"path": self.path.summary(),
|
|
56
|
+
"verdict": self.verdict,
|
|
57
|
+
"expect": self.expect.as_dict() if self.expect else None,
|
|
58
|
+
"findings": [f.as_dict() for f in self.findings],
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class Result:
|
|
64
|
+
clip: str
|
|
65
|
+
shots: list = field(default_factory=list)
|
|
66
|
+
cuts: list = field(default_factory=list)
|
|
67
|
+
src_width: int = 0
|
|
68
|
+
src_height: int = 0
|
|
69
|
+
width: int = 0
|
|
70
|
+
height: int = 0
|
|
71
|
+
frame_count: int = 0
|
|
72
|
+
fps: float = 0.0
|
|
73
|
+
truncated: bool = False
|
|
74
|
+
|
|
75
|
+
# --- clip-level views over the shots -------------------------------------
|
|
76
|
+
# A clip with one shot answers exactly as it did before segmentation existed.
|
|
77
|
+
# With several, these summarise, and `shots` is there for anyone who needs the
|
|
78
|
+
# detail - a clip-level verdict on an edited sequence is a summary, not a
|
|
79
|
+
# measurement, and the distinction is kept rather than blurred.
|
|
80
|
+
|
|
81
|
+
@property
|
|
82
|
+
def path(self):
|
|
83
|
+
"""The dominant shot's path: the longest one, not the first."""
|
|
84
|
+
if not self.shots:
|
|
85
|
+
return None
|
|
86
|
+
return max(self.shots, key=lambda s: s.frames).path
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def findings(self) -> list:
|
|
90
|
+
return [f for s in self.shots for f in s.findings]
|
|
91
|
+
|
|
92
|
+
@property
|
|
93
|
+
def verdict(self) -> str:
|
|
94
|
+
if not self.shots:
|
|
95
|
+
return UNKNOWN
|
|
96
|
+
return max((s.verdict for s in self.shots), key=lambda v: _RANK[v])
|
|
97
|
+
|
|
98
|
+
@property
|
|
99
|
+
def expect(self):
|
|
100
|
+
"""The failing expectation if any shot failed it, else the dominant one."""
|
|
101
|
+
failed = [s.expect for s in self.shots if s.expect is not None and not s.expect.ok]
|
|
102
|
+
if failed:
|
|
103
|
+
return failed[0]
|
|
104
|
+
with_exp = [s for s in self.shots if s.expect is not None]
|
|
105
|
+
return max(with_exp, key=lambda s: s.frames).expect if with_exp else None
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def ok(self) -> bool:
|
|
109
|
+
"""Every shot clean, and - if a move was declared - every shot held it."""
|
|
110
|
+
return bool(self.shots) and all(s.ok for s in self.shots)
|
|
111
|
+
|
|
112
|
+
def gated(self, min_severity: str = BROKEN) -> list:
|
|
113
|
+
return at_or_above(self.findings, min_severity)
|
|
114
|
+
|
|
115
|
+
def as_dict(self) -> dict:
|
|
116
|
+
return {
|
|
117
|
+
"clip": self.clip,
|
|
118
|
+
"source": {"width": self.src_width, "height": self.src_height,
|
|
119
|
+
"fps": self.fps},
|
|
120
|
+
"measured_at": {"width": self.width, "height": self.height,
|
|
121
|
+
"frames": self.frame_count,
|
|
122
|
+
"truncated": self.truncated},
|
|
123
|
+
"cuts": self.cuts,
|
|
124
|
+
"shots": [s.as_dict() for s in self.shots],
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _assemble(name: str, frames, grid: int, anchors: int, segment: bool,
|
|
129
|
+
expect: str | None) -> tuple[list, list]:
|
|
130
|
+
from .expect import check as check_expect
|
|
131
|
+
from .path import analyse_clip
|
|
132
|
+
from .verdict import judge, worst
|
|
133
|
+
|
|
134
|
+
shots_in, cuts = analyse_clip(frames, grid=grid, anchors=anchors,
|
|
135
|
+
segment=segment)
|
|
136
|
+
shots = []
|
|
137
|
+
for (a, b), p in shots_in:
|
|
138
|
+
f = judge(p)
|
|
139
|
+
shots.append(Shot(start=a, end=b, path=p, findings=f, verdict=worst(f),
|
|
140
|
+
expect=check_expect(p, expect) if expect else None))
|
|
141
|
+
return shots, cuts
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def measure(clip: str, expect: str | None = None, *, max_side: int = 512,
|
|
145
|
+
grid: int = 4, start: float = 0.0, duration: float | None = None,
|
|
146
|
+
sample_fps: float | None = None, anchors: int = 4,
|
|
147
|
+
max_frames: int = 600, segment: bool = True) -> Result:
|
|
148
|
+
"""Measure a video file. Cuts are detected and each shot measured on its own."""
|
|
149
|
+
from .frames import load
|
|
150
|
+
|
|
151
|
+
c = load(clip, max_side=max_side, sample_fps=sample_fps, start=start,
|
|
152
|
+
duration=duration, max_frames=max_frames)
|
|
153
|
+
shots, cuts = _assemble(clip, c.frames, grid, anchors, segment, expect)
|
|
154
|
+
return Result(clip=clip, shots=shots, cuts=cuts,
|
|
155
|
+
src_width=c.src_width, src_height=c.src_height,
|
|
156
|
+
width=c.width, height=c.height, frame_count=len(c),
|
|
157
|
+
fps=c.fps, truncated=c.truncated)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def measure_frames(frames, expect: str | None = None, *, name: str = "<frames>",
|
|
161
|
+
max_side: int = 512, grid: int = 4, anchors: int = 4,
|
|
162
|
+
segment: bool = True, fps: float = 0.0) -> Result:
|
|
163
|
+
"""Measure frames that are already in memory - no file, no ffmpeg.
|
|
164
|
+
|
|
165
|
+
Takes (n, h, w) or (n, h, w, c) in uint8 0..255 or float 0..1, as a numpy
|
|
166
|
+
array or a torch tensor. This is the entry point a generation graph uses: the
|
|
167
|
+
frames exist, they were never written to disk, and the question is whether
|
|
168
|
+
the move that was asked for is in them.
|
|
169
|
+
"""
|
|
170
|
+
from .ingest import to_gray_stack
|
|
171
|
+
|
|
172
|
+
stack, sw, sh = to_gray_stack(frames, max_side=max_side)
|
|
173
|
+
shots, cuts = _assemble(name, stack, grid, anchors, segment, expect)
|
|
174
|
+
n, h, w = stack.shape
|
|
175
|
+
return Result(clip=name, shots=shots, cuts=cuts, src_width=sw, src_height=sh,
|
|
176
|
+
width=w, height=h, frame_count=n, fps=fps)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
# ------------------------------------------------------------------ report ----
|
|
180
|
+
def _fmt(v, nd=4):
|
|
181
|
+
if v is None:
|
|
182
|
+
return "-"
|
|
183
|
+
if isinstance(v, float) and not np.isfinite(v):
|
|
184
|
+
return "not measurable"
|
|
185
|
+
return f"{v:.{nd}f}" if isinstance(v, float) else str(v)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def report(r: Result, *, quiet: bool = False, header: bool = True) -> str:
|
|
189
|
+
"""The human-readable report, as a string. The CLI prints this verbatim."""
|
|
190
|
+
out = []
|
|
191
|
+
w = out.append
|
|
192
|
+
|
|
193
|
+
if header:
|
|
194
|
+
w(f"\n{r.clip}")
|
|
195
|
+
rate = f" @ {r.fps:g} fps" if r.fps else ""
|
|
196
|
+
w(f" {r.src_width}x{r.src_height} -> measured at {r.width}x{r.height}, "
|
|
197
|
+
f"{r.frame_count} frames{rate}")
|
|
198
|
+
if r.truncated:
|
|
199
|
+
w(f" only the first {r.frame_count} frames were measured "
|
|
200
|
+
f"(--max-frames); the rest of the clip was not looked at.")
|
|
201
|
+
if r.cuts:
|
|
202
|
+
w(f" {len(r.cuts)} cut(s) detected: this is {len(r.shots)} shot(s), "
|
|
203
|
+
f"not one take. Each is measured on its own.")
|
|
204
|
+
|
|
205
|
+
multi = len(r.shots) > 1
|
|
206
|
+
for s in r.shots:
|
|
207
|
+
p = s.path
|
|
208
|
+
if multi:
|
|
209
|
+
w(f"\n shot {s.start}-{s.end} ({s.frames} frames)")
|
|
210
|
+
w("\n camera path")
|
|
211
|
+
w(f" dominant move {p.dominant}")
|
|
212
|
+
w(f" pan {_fmt(p.net_pan)} of frame width "
|
|
213
|
+
f"(x {_fmt(float(p.tx[-1]))}, y {_fmt(float(p.ty[-1]))})")
|
|
214
|
+
w(f" zoom {_fmt(p.zoom, 3)}x")
|
|
215
|
+
w(f" roll {_fmt(float(np.degrees(p.net_roll)), 2)} deg")
|
|
216
|
+
w(f" reversals {p.reversals} (measured, not judged)")
|
|
217
|
+
w(f" jerk {_fmt(p.jerk, 2)}")
|
|
218
|
+
w(f" incoherence {_fmt(p.incoherence)}")
|
|
219
|
+
w(f" morph {_fmt(p.morph, 3)} (used to find cuts, not judged)")
|
|
220
|
+
w(f" closure {_fmt(p.closure)}")
|
|
221
|
+
w(f" confidence {_fmt(p.confidence, 2)}")
|
|
222
|
+
|
|
223
|
+
if s.expect is not None:
|
|
224
|
+
w(f"\n asked for: {s.expect.move}")
|
|
225
|
+
w(f" {'HELD' if s.expect.ok else 'NOT HELD'} - {s.expect.detail}")
|
|
226
|
+
|
|
227
|
+
if not s.findings:
|
|
228
|
+
w("\n no findings: the motion is consistent with one physical camera.")
|
|
229
|
+
else:
|
|
230
|
+
w(f"\n {len(s.findings)} finding(s)")
|
|
231
|
+
for f in s.findings:
|
|
232
|
+
w(f" [{_MARK[f.severity]}] {f.summary}")
|
|
233
|
+
w(f" evidence: {f.evidence}")
|
|
234
|
+
if not quiet:
|
|
235
|
+
w(f" {f.advice}")
|
|
236
|
+
return "\n".join(out) + "\n"
|
shotdrift/expect.py
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""Check a measured path against the move that was ASKED for.
|
|
2
|
+
|
|
3
|
+
This is the part that does not exist elsewhere. Everything else here describes
|
|
4
|
+
what a clip did; this asks whether it did what it was told, which is the actual
|
|
5
|
+
question when you type "slow push in" into a generator and get back something
|
|
6
|
+
that drifts left.
|
|
7
|
+
|
|
8
|
+
A declared move is checked three ways, because each failure is different and the
|
|
9
|
+
distinction is what makes the output useful:
|
|
10
|
+
happened - is there any motion on that channel at all?
|
|
11
|
+
direction - is it the way round you asked?
|
|
12
|
+
held - was it monotonic, or did it wobble there?
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
|
|
21
|
+
# name -> (channel, sign, human); sign 0 means "must not move"
|
|
22
|
+
# THE SIGNS ARE CAMERA-RELATIVE, NOT CONTENT-RELATIVE, and that is the one thing
|
|
23
|
+
# in this file worth reading twice. A camera panning RIGHT makes the picture move
|
|
24
|
+
# LEFT, so "pan-right" expects a NEGATIVE x. Writing these the intuitive way round
|
|
25
|
+
# made a correctly-measured clean pan report "x went the other way" - the
|
|
26
|
+
# measurement was right and the vocabulary was wrong. Verified by test.
|
|
27
|
+
MOVES: dict[str, tuple[str, int, str]] = {
|
|
28
|
+
"static": ("none", 0, "locked off"),
|
|
29
|
+
"locked": ("none", 0, "locked off"),
|
|
30
|
+
"push-in": ("zoom", +1, "push in / dolly in"), # subject grows
|
|
31
|
+
"dolly-in": ("zoom", +1, "push in / dolly in"),
|
|
32
|
+
"zoom-in": ("zoom", +1, "zoom in"),
|
|
33
|
+
"pull-out": ("zoom", -1, "pull out / dolly out"),
|
|
34
|
+
"dolly-out": ("zoom", -1, "pull out / dolly out"),
|
|
35
|
+
"zoom-out": ("zoom", -1, "zoom out"),
|
|
36
|
+
"pan-left": ("x", +1, "pan left"), # camera left -> picture right
|
|
37
|
+
"pan-right": ("x", -1, "pan right"), # camera right -> picture left
|
|
38
|
+
"tilt-up": ("y", +1, "tilt up"), # camera up -> picture down
|
|
39
|
+
"tilt-down": ("y", -1, "tilt down"),
|
|
40
|
+
"roll-cw": ("roll", -1, "roll clockwise"), # camera cw -> picture ccw
|
|
41
|
+
"roll-ccw": ("roll", +1, "roll counter-clockwise"),
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
# A move has to clear this to count as having happened at all, in frame widths
|
|
45
|
+
# (or radians / log-scale). Below it, the honest answer is "nothing happened".
|
|
46
|
+
FLOOR = {"x": 0.02, "y": 0.02, "zoom": 0.02, "roll": 0.02}
|
|
47
|
+
# Fraction of the travel allowed to run backwards before the move is not "held".
|
|
48
|
+
BACKTRACK_OK = 0.25
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass
|
|
52
|
+
class Expectation:
|
|
53
|
+
move: str
|
|
54
|
+
ok: bool
|
|
55
|
+
happened: bool
|
|
56
|
+
direction_ok: bool
|
|
57
|
+
held: bool
|
|
58
|
+
measured: float
|
|
59
|
+
backtrack: float
|
|
60
|
+
detail: str
|
|
61
|
+
|
|
62
|
+
def as_dict(self) -> dict:
|
|
63
|
+
return dict(move=self.move, ok=self.ok, happened=self.happened,
|
|
64
|
+
direction_ok=self.direction_ok, held=self.held,
|
|
65
|
+
measured=round(self.measured, 5),
|
|
66
|
+
backtrack=round(self.backtrack, 4), detail=self.detail)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def known() -> list[str]:
|
|
70
|
+
return sorted(MOVES)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _series(p, chan: str) -> np.ndarray:
|
|
74
|
+
if chan == "x":
|
|
75
|
+
return np.asarray(p.tx, dtype=float)
|
|
76
|
+
if chan == "y":
|
|
77
|
+
return np.asarray(p.ty, dtype=float)
|
|
78
|
+
if chan == "zoom":
|
|
79
|
+
return np.asarray(p.logscale, dtype=float)
|
|
80
|
+
if chan == "roll":
|
|
81
|
+
return np.asarray(p.roll, dtype=float)
|
|
82
|
+
raise KeyError(chan)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def check(p, move: str) -> Expectation:
|
|
86
|
+
key = move.strip().lower().replace("_", "-")
|
|
87
|
+
if key not in MOVES:
|
|
88
|
+
raise KeyError(f"unknown move {move!r}; known: {', '.join(known())}")
|
|
89
|
+
chan, sign, human = MOVES[key]
|
|
90
|
+
|
|
91
|
+
if chan == "none":
|
|
92
|
+
# "Locked off" is a claim about every channel at once, so it is the one
|
|
93
|
+
# case that cannot be judged on a single series.
|
|
94
|
+
worst_name, worst_val, worst_floor = "pan", p.net_pan, FLOOR["x"]
|
|
95
|
+
for nm, val, fl in (("pan", p.net_pan, FLOOR["x"]),
|
|
96
|
+
("zoom", abs(float(np.log(max(p.zoom, 1e-6)))), FLOOR["zoom"]),
|
|
97
|
+
("roll", abs(p.net_roll), FLOOR["roll"])):
|
|
98
|
+
if val / fl > worst_val / worst_floor:
|
|
99
|
+
worst_name, worst_val, worst_floor = nm, val, fl
|
|
100
|
+
still = worst_val < worst_floor
|
|
101
|
+
return Expectation(
|
|
102
|
+
move=key, ok=still, happened=not still, direction_ok=still, held=still,
|
|
103
|
+
measured=worst_val, backtrack=0.0,
|
|
104
|
+
detail=(f"locked off as asked; largest movement was {worst_name} "
|
|
105
|
+
f"{worst_val:.4f} (floor {worst_floor})") if still else
|
|
106
|
+
(f"asked for {human}, but {worst_name} moved {worst_val:.4f}, "
|
|
107
|
+
f"over the {worst_floor} floor"),
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
s = _series(p, chan)
|
|
111
|
+
net = float(s[-1] - s[0])
|
|
112
|
+
step = np.diff(s)
|
|
113
|
+
travel = float(np.sum(np.abs(step))) or 1e-9
|
|
114
|
+
against = float(np.sum(np.abs(step[np.sign(step) == -sign])))
|
|
115
|
+
backtrack = against / travel
|
|
116
|
+
|
|
117
|
+
happened = abs(net) >= FLOOR[chan]
|
|
118
|
+
direction_ok = bool(np.sign(net) == sign) if happened else False
|
|
119
|
+
held = backtrack <= BACKTRACK_OK
|
|
120
|
+
ok = happened and direction_ok and held
|
|
121
|
+
|
|
122
|
+
if not happened:
|
|
123
|
+
detail = (f"asked for {human}; {chan} moved {net:+.4f}, under the "
|
|
124
|
+
f"{FLOOR[chan]} floor - that move did not happen")
|
|
125
|
+
elif not direction_ok:
|
|
126
|
+
detail = (f"asked for {human}; {chan} went the other way, {net:+.4f}")
|
|
127
|
+
elif not held:
|
|
128
|
+
detail = (f"{human} happened ({net:+.4f}) but {backtrack * 100:.0f}% of the "
|
|
129
|
+
f"travel ran backwards - the move is not held")
|
|
130
|
+
else:
|
|
131
|
+
detail = f"{human} as asked: {chan} {net:+.4f}, {backtrack * 100:.0f}% backtrack"
|
|
132
|
+
return Expectation(key, ok, happened, direction_ok, held, net, backtrack, detail)
|
shotdrift/frames.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Decode a video to small grayscale frames with ffmpeg, and nothing else.
|
|
2
|
+
|
|
3
|
+
Deliberately no OpenCV and no torch. ffmpeg is already on every machine that
|
|
4
|
+
edits video, and a tool that needs a 2 GB wheel to measure a camera move will
|
|
5
|
+
not get run.
|
|
6
|
+
|
|
7
|
+
Everything downstream works in NORMALISED units - fractions of the frame width -
|
|
8
|
+
so a threshold means the same thing on a 720p proxy and a 4K master. Reporting
|
|
9
|
+
pixels at the analysis resolution would make every bound silently resolution
|
|
10
|
+
dependent, which is how a measurement tool ends up with magic numbers.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import shutil
|
|
17
|
+
import subprocess
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
|
|
22
|
+
from .ingest import analysis_size
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class FFmpegMissing(RuntimeError):
|
|
26
|
+
pass
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class DecodeError(RuntimeError):
|
|
30
|
+
pass
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class Clip:
|
|
35
|
+
frames: np.ndarray # (n, h, w) float32 in 0..1
|
|
36
|
+
width: int # analysis width (px)
|
|
37
|
+
height: int # analysis height (px)
|
|
38
|
+
src_width: int # original width (px)
|
|
39
|
+
src_height: int # original height (px)
|
|
40
|
+
fps: float
|
|
41
|
+
duration: float
|
|
42
|
+
truncated: bool = False # hit the frame budget with picture still to come
|
|
43
|
+
|
|
44
|
+
def __len__(self) -> int:
|
|
45
|
+
return int(self.frames.shape[0])
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _exe(name: str) -> str:
|
|
49
|
+
p = shutil.which(name)
|
|
50
|
+
if not p:
|
|
51
|
+
raise FFmpegMissing(
|
|
52
|
+
f"{name} not found on PATH. shotdrift measures video, so it needs "
|
|
53
|
+
"ffmpeg: `brew install ffmpeg` or `apt install ffmpeg`."
|
|
54
|
+
)
|
|
55
|
+
return p
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def probe(path: str) -> dict:
|
|
59
|
+
out = subprocess.run(
|
|
60
|
+
[_exe("ffprobe"), "-v", "error", "-select_streams", "v:0",
|
|
61
|
+
"-show_entries", "stream=width,height,avg_frame_rate,nb_read_packets",
|
|
62
|
+
"-show_entries", "format=duration", "-of", "json", path],
|
|
63
|
+
capture_output=True, text=True,
|
|
64
|
+
)
|
|
65
|
+
if out.returncode != 0:
|
|
66
|
+
raise DecodeError(f"ffprobe could not read {path}: {out.stderr.strip()[:300]}")
|
|
67
|
+
d = json.loads(out.stdout or "{}")
|
|
68
|
+
streams = d.get("streams") or []
|
|
69
|
+
if not streams:
|
|
70
|
+
raise DecodeError(f"{path} has no video stream.")
|
|
71
|
+
s = streams[0]
|
|
72
|
+
num, _, den = (s.get("avg_frame_rate") or "0/1").partition("/")
|
|
73
|
+
try:
|
|
74
|
+
fps = float(num) / float(den) if float(den) else 0.0
|
|
75
|
+
except (ValueError, ZeroDivisionError):
|
|
76
|
+
fps = 0.0
|
|
77
|
+
try:
|
|
78
|
+
dur = float((d.get("format") or {}).get("duration") or 0.0)
|
|
79
|
+
except ValueError:
|
|
80
|
+
dur = 0.0
|
|
81
|
+
return {"width": int(s["width"]), "height": int(s["height"]), "fps": fps, "duration": dur}
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def load(path: str, max_side: int = 512, sample_fps: float | None = None,
|
|
85
|
+
start: float = 0.0, duration: float | None = None,
|
|
86
|
+
max_frames: int = 600) -> Clip:
|
|
87
|
+
"""Decode to grayscale. `sample_fps=None` keeps the clip's own rate.
|
|
88
|
+
|
|
89
|
+
Camera motion is measured BETWEEN ADJACENT FRAMES, so dropping the rate
|
|
90
|
+
changes what is being measured, not just the cost of measuring it. The
|
|
91
|
+
default therefore keeps every frame; `--sample-fps` exists for long clips
|
|
92
|
+
and says so in the report.
|
|
93
|
+
"""
|
|
94
|
+
meta = probe(path)
|
|
95
|
+
sw, sh = meta["width"], meta["height"]
|
|
96
|
+
if sw <= 0 or sh <= 0:
|
|
97
|
+
raise DecodeError(f"{path} reports a {sw}x{sh} frame size.")
|
|
98
|
+
|
|
99
|
+
w, h = analysis_size(sw, sh, max_side)
|
|
100
|
+
|
|
101
|
+
cmd = [_exe("ffmpeg"), "-v", "error"]
|
|
102
|
+
if start > 0:
|
|
103
|
+
cmd += ["-ss", f"{start:.3f}"]
|
|
104
|
+
cmd += ["-i", path]
|
|
105
|
+
if duration is not None:
|
|
106
|
+
cmd += ["-t", f"{duration:.3f}"]
|
|
107
|
+
vf = [f"scale={w}:{h}:flags=bilinear"]
|
|
108
|
+
if sample_fps:
|
|
109
|
+
vf.insert(0, f"fps={sample_fps:g}")
|
|
110
|
+
cmd += ["-vf", ",".join(vf), "-frames:v", str(max_frames),
|
|
111
|
+
"-pix_fmt", "gray", "-f", "rawvideo", "-"]
|
|
112
|
+
|
|
113
|
+
out = subprocess.run(cmd, capture_output=True)
|
|
114
|
+
if out.returncode != 0:
|
|
115
|
+
raise DecodeError(f"ffmpeg failed on {path}: {out.stderr.decode(errors='replace').strip()[:300]}")
|
|
116
|
+
|
|
117
|
+
buf = np.frombuffer(out.stdout, dtype=np.uint8)
|
|
118
|
+
stride = w * h
|
|
119
|
+
n = buf.size // stride
|
|
120
|
+
if n < 2:
|
|
121
|
+
raise DecodeError(
|
|
122
|
+
f"{path} decoded to {n} frame(s) at {w}x{h}. Camera motion needs at "
|
|
123
|
+
"least two frames; check the file is not truncated and that any "
|
|
124
|
+
"--start/--duration window actually contains picture."
|
|
125
|
+
)
|
|
126
|
+
frames = buf[: n * stride].reshape(n, h, w).astype(np.float32) / 255.0
|
|
127
|
+
eff_fps = float(sample_fps) if sample_fps else meta["fps"]
|
|
128
|
+
# A frame budget that silently drops the rest of the clip is how a tool comes
|
|
129
|
+
# to report confidently on the first twelve seconds of a two-minute take. The
|
|
130
|
+
# window is reported, not assumed.
|
|
131
|
+
window = duration if duration is not None else max(0.0, meta["duration"] - start)
|
|
132
|
+
expected = window * eff_fps if (window and eff_fps) else 0.0
|
|
133
|
+
truncated = bool(n >= max_frames and expected > n + 1)
|
|
134
|
+
return Clip(frames=frames, width=w, height=h, src_width=sw, src_height=sh,
|
|
135
|
+
fps=eff_fps, duration=meta["duration"], truncated=truncated)
|