shotdrift 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
shotdrift/__init__.py ADDED
@@ -0,0 +1,25 @@
1
+ """shotdrift - measure whether a video holds the camera move it was given.
2
+
3
+ AI video is asked for a camera move and is not obliged to deliver one. The usual
4
+ check is a person watching sixty clips. This measures the camera path out of the
5
+ pixels - pan, zoom, roll, frame by frame - and reports whether one physical camera
6
+ could have produced it.
7
+
8
+ from shotdrift import measure
9
+ r = measure("take_07.mp4", expect="push-in")
10
+ print(r.verdict, [f.code for f in r.findings])
11
+
12
+ Frames already in memory - inside a generation graph, say, where the clip was
13
+ never written to disk - skip ffmpeg entirely:
14
+
15
+ from shotdrift import measure_frames, report
16
+ r = measure_frames(image_batch, expect="push-in") # (n, h, w, c), 0..1
17
+ print(report(r))
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from .core import Result, Shot, measure, measure_frames, report
23
+
24
+ __version__ = "0.2.0"
25
+ __all__ = ["measure", "measure_frames", "report", "Result", "Shot", "__version__"]
shotdrift/cli.py ADDED
@@ -0,0 +1,95 @@
1
+ """shotdrift - measure whether a video holds the camera move it was given."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+
9
+ from . import __version__
10
+ from .core import measure, report
11
+ from .expect import known as known_moves
12
+ from .frames import DecodeError, FFmpegMissing
13
+ from .verdict import (BROKEN, CLEAN, SOFT, UNKNOWN, at_or_above)
14
+
15
+ EXIT_OK, EXIT_FINDINGS, EXIT_ERROR, EXIT_UNMEASURABLE = 0, 1, 2, 3
16
+
17
+
18
+ def main(argv=None) -> int:
19
+ ap = argparse.ArgumentParser(
20
+ prog="shotdrift",
21
+ description="Measure the camera move in a video, and whether it is one a real "
22
+ "camera could have made.",
23
+ epilog="Exit 0 clean, 1 findings at or above --min-severity, "
24
+ "2 could not run, 3 clip could not be measured.",
25
+ )
26
+ ap.add_argument("clips", nargs="*", metavar="CLIP")
27
+ ap.add_argument("--expect", metavar="MOVE",
28
+ help="the move you asked for: " + ", ".join(known_moves()))
29
+ ap.add_argument("--list-moves", action="store_true")
30
+ ap.add_argument("--min-severity", choices=[CLEAN, SOFT, BROKEN], default=BROKEN,
31
+ help="which findings make this exit non-zero (default: broken; "
32
+ "soft findings are always printed either way)")
33
+ ap.add_argument("--max-side", type=int, default=512,
34
+ help="analysis resolution, long edge in px (default 512)")
35
+ ap.add_argument("--grid", type=int, default=4, help="tiles per axis (default 4)")
36
+ ap.add_argument("--start", type=float, default=0.0, metavar="SEC")
37
+ ap.add_argument("--duration", type=float, default=None, metavar="SEC")
38
+ ap.add_argument("--sample-fps", type=float, default=None,
39
+ help="resample before measuring; changes WHAT is measured, "
40
+ "not just the cost")
41
+ ap.add_argument("--max-frames", type=int, default=600)
42
+ ap.add_argument("--anchors", type=int, default=4,
43
+ help="independent first-to-frame-k checks (0 disables closure)")
44
+ ap.add_argument("--no-segment", action="store_true",
45
+ help="measure the file as one take even if it contains cuts")
46
+ ap.add_argument("--json", action="store_true", help="machine-readable output")
47
+ ap.add_argument("--quiet", action="store_true", help="findings without the advice")
48
+ ap.add_argument("--version", action="version", version=f"shotdrift {__version__}")
49
+ args = ap.parse_args(argv)
50
+
51
+ if args.list_moves:
52
+ for m in known_moves():
53
+ print(m)
54
+ return EXIT_OK
55
+ if not args.clips:
56
+ ap.error("give at least one CLIP, or --list-moves")
57
+
58
+ if args.expect and args.expect not in known_moves():
59
+ ap.error(f"unknown move {args.expect!r}. --list-moves shows them all.")
60
+
61
+ results, rc = [], EXIT_OK
62
+ for name in args.clips:
63
+ try:
64
+ r = measure(name, expect=args.expect, max_side=args.max_side,
65
+ grid=args.grid, start=args.start, duration=args.duration,
66
+ sample_fps=args.sample_fps, anchors=args.anchors,
67
+ max_frames=args.max_frames,
68
+ segment=not args.no_segment)
69
+ except FFmpegMissing as e:
70
+ print(f"shotdrift: {e}", file=sys.stderr)
71
+ return EXIT_ERROR
72
+ except (DecodeError, ValueError) as e:
73
+ print(f"shotdrift: {name}: {e}", file=sys.stderr)
74
+ rc = max(rc, EXIT_ERROR)
75
+ continue
76
+
77
+ for s in r.shots:
78
+ if s.verdict == UNKNOWN:
79
+ rc = max(rc, EXIT_UNMEASURABLE)
80
+ elif at_or_above(s.findings, args.min_severity) or \
81
+ (s.expect is not None and not s.expect.ok):
82
+ rc = max(rc, EXIT_FINDINGS)
83
+
84
+ if args.json:
85
+ results.append(r.as_dict())
86
+ else:
87
+ sys.stdout.write(report(r, quiet=args.quiet))
88
+
89
+ if args.json:
90
+ print(json.dumps({"shotdrift": __version__, "clips": results}, indent=2))
91
+ return rc
92
+
93
+
94
+ if __name__ == "__main__":
95
+ sys.exit(main())
shotdrift/core.py ADDED
@@ -0,0 +1,236 @@
1
+ """One measurement, however the frames arrived, and one way of reporting it.
2
+
3
+ Both halves of this file are corrections.
4
+
5
+ `measure()` used to call `analyse` where the command line calls `analyse_clip`,
6
+ so the Python API did not segment. On an edited reel that is not a small
7
+ difference: measured as one take, a six-shot sequence reports a broken closure
8
+ and a pile of reversals, all true and all useless, because no single camera move
9
+ was ever there to hold. The command line was fixed for exactly that and the
10
+ library was left behind - so `import shotdrift` gave a worse answer than
11
+ `shotdrift` did, which nothing warned anyone about.
12
+
13
+ And the text report lived inside the command line, writing straight to stdout.
14
+ Anything else wanting the same words - a node in a generation graph, a CI
15
+ comment - had to reimplement them, which is how two surfaces of one tool come to
16
+ disagree about what a clip measured. It renders to a string here, and the command
17
+ line prints what it returns.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from dataclasses import dataclass, field
23
+
24
+ import numpy as np
25
+
26
+ from .verdict import BROKEN, CLEAN, SOFT, UNKNOWN, at_or_above
27
+
28
+ __all__ = ["Shot", "Result", "measure", "measure_frames", "report"]
29
+
30
+ _MARK = {CLEAN: "ok", SOFT: "soft", BROKEN: "BROKEN", UNKNOWN: "unmeasurable"}
31
+ _RANK = {CLEAN: 0, SOFT: 1, BROKEN: 2, UNKNOWN: 0}
32
+
33
+
34
+ @dataclass
35
+ class Shot:
36
+ """One continuous take, measured on its own."""
37
+ start: int
38
+ end: int
39
+ path: object = field(repr=False)
40
+ findings: list = field(default_factory=list)
41
+ verdict: str = CLEAN
42
+ expect: object | None = None
43
+
44
+ @property
45
+ def frames(self) -> int:
46
+ return self.end - self.start
47
+
48
+ @property
49
+ def ok(self) -> bool:
50
+ return self.verdict == CLEAN and (self.expect is None or self.expect.ok)
51
+
52
+ def as_dict(self) -> dict:
53
+ return {
54
+ "frames": [self.start, self.end],
55
+ "path": self.path.summary(),
56
+ "verdict": self.verdict,
57
+ "expect": self.expect.as_dict() if self.expect else None,
58
+ "findings": [f.as_dict() for f in self.findings],
59
+ }
60
+
61
+
62
+ @dataclass
63
+ class Result:
64
+ clip: str
65
+ shots: list = field(default_factory=list)
66
+ cuts: list = field(default_factory=list)
67
+ src_width: int = 0
68
+ src_height: int = 0
69
+ width: int = 0
70
+ height: int = 0
71
+ frame_count: int = 0
72
+ fps: float = 0.0
73
+ truncated: bool = False
74
+
75
+ # --- clip-level views over the shots -------------------------------------
76
+ # A clip with one shot answers exactly as it did before segmentation existed.
77
+ # With several, these summarise, and `shots` is there for anyone who needs the
78
+ # detail - a clip-level verdict on an edited sequence is a summary, not a
79
+ # measurement, and the distinction is kept rather than blurred.
80
+
81
+ @property
82
+ def path(self):
83
+ """The dominant shot's path: the longest one, not the first."""
84
+ if not self.shots:
85
+ return None
86
+ return max(self.shots, key=lambda s: s.frames).path
87
+
88
+ @property
89
+ def findings(self) -> list:
90
+ return [f for s in self.shots for f in s.findings]
91
+
92
+ @property
93
+ def verdict(self) -> str:
94
+ if not self.shots:
95
+ return UNKNOWN
96
+ return max((s.verdict for s in self.shots), key=lambda v: _RANK[v])
97
+
98
+ @property
99
+ def expect(self):
100
+ """The failing expectation if any shot failed it, else the dominant one."""
101
+ failed = [s.expect for s in self.shots if s.expect is not None and not s.expect.ok]
102
+ if failed:
103
+ return failed[0]
104
+ with_exp = [s for s in self.shots if s.expect is not None]
105
+ return max(with_exp, key=lambda s: s.frames).expect if with_exp else None
106
+
107
+ @property
108
+ def ok(self) -> bool:
109
+ """Every shot clean, and - if a move was declared - every shot held it."""
110
+ return bool(self.shots) and all(s.ok for s in self.shots)
111
+
112
+ def gated(self, min_severity: str = BROKEN) -> list:
113
+ return at_or_above(self.findings, min_severity)
114
+
115
+ def as_dict(self) -> dict:
116
+ return {
117
+ "clip": self.clip,
118
+ "source": {"width": self.src_width, "height": self.src_height,
119
+ "fps": self.fps},
120
+ "measured_at": {"width": self.width, "height": self.height,
121
+ "frames": self.frame_count,
122
+ "truncated": self.truncated},
123
+ "cuts": self.cuts,
124
+ "shots": [s.as_dict() for s in self.shots],
125
+ }
126
+
127
+
128
+ def _assemble(name: str, frames, grid: int, anchors: int, segment: bool,
129
+ expect: str | None) -> tuple[list, list]:
130
+ from .expect import check as check_expect
131
+ from .path import analyse_clip
132
+ from .verdict import judge, worst
133
+
134
+ shots_in, cuts = analyse_clip(frames, grid=grid, anchors=anchors,
135
+ segment=segment)
136
+ shots = []
137
+ for (a, b), p in shots_in:
138
+ f = judge(p)
139
+ shots.append(Shot(start=a, end=b, path=p, findings=f, verdict=worst(f),
140
+ expect=check_expect(p, expect) if expect else None))
141
+ return shots, cuts
142
+
143
+
144
+ def measure(clip: str, expect: str | None = None, *, max_side: int = 512,
145
+ grid: int = 4, start: float = 0.0, duration: float | None = None,
146
+ sample_fps: float | None = None, anchors: int = 4,
147
+ max_frames: int = 600, segment: bool = True) -> Result:
148
+ """Measure a video file. Cuts are detected and each shot measured on its own."""
149
+ from .frames import load
150
+
151
+ c = load(clip, max_side=max_side, sample_fps=sample_fps, start=start,
152
+ duration=duration, max_frames=max_frames)
153
+ shots, cuts = _assemble(clip, c.frames, grid, anchors, segment, expect)
154
+ return Result(clip=clip, shots=shots, cuts=cuts,
155
+ src_width=c.src_width, src_height=c.src_height,
156
+ width=c.width, height=c.height, frame_count=len(c),
157
+ fps=c.fps, truncated=c.truncated)
158
+
159
+
160
+ def measure_frames(frames, expect: str | None = None, *, name: str = "<frames>",
161
+ max_side: int = 512, grid: int = 4, anchors: int = 4,
162
+ segment: bool = True, fps: float = 0.0) -> Result:
163
+ """Measure frames that are already in memory - no file, no ffmpeg.
164
+
165
+ Takes (n, h, w) or (n, h, w, c) in uint8 0..255 or float 0..1, as a numpy
166
+ array or a torch tensor. This is the entry point a generation graph uses: the
167
+ frames exist, they were never written to disk, and the question is whether
168
+ the move that was asked for is in them.
169
+ """
170
+ from .ingest import to_gray_stack
171
+
172
+ stack, sw, sh = to_gray_stack(frames, max_side=max_side)
173
+ shots, cuts = _assemble(name, stack, grid, anchors, segment, expect)
174
+ n, h, w = stack.shape
175
+ return Result(clip=name, shots=shots, cuts=cuts, src_width=sw, src_height=sh,
176
+ width=w, height=h, frame_count=n, fps=fps)
177
+
178
+
179
+ # ------------------------------------------------------------------ report ----
180
+ def _fmt(v, nd=4):
181
+ if v is None:
182
+ return "-"
183
+ if isinstance(v, float) and not np.isfinite(v):
184
+ return "not measurable"
185
+ return f"{v:.{nd}f}" if isinstance(v, float) else str(v)
186
+
187
+
188
+ def report(r: Result, *, quiet: bool = False, header: bool = True) -> str:
189
+ """The human-readable report, as a string. The CLI prints this verbatim."""
190
+ out = []
191
+ w = out.append
192
+
193
+ if header:
194
+ w(f"\n{r.clip}")
195
+ rate = f" @ {r.fps:g} fps" if r.fps else ""
196
+ w(f" {r.src_width}x{r.src_height} -> measured at {r.width}x{r.height}, "
197
+ f"{r.frame_count} frames{rate}")
198
+ if r.truncated:
199
+ w(f" only the first {r.frame_count} frames were measured "
200
+ f"(--max-frames); the rest of the clip was not looked at.")
201
+ if r.cuts:
202
+ w(f" {len(r.cuts)} cut(s) detected: this is {len(r.shots)} shot(s), "
203
+ f"not one take. Each is measured on its own.")
204
+
205
+ multi = len(r.shots) > 1
206
+ for s in r.shots:
207
+ p = s.path
208
+ if multi:
209
+ w(f"\n shot {s.start}-{s.end} ({s.frames} frames)")
210
+ w("\n camera path")
211
+ w(f" dominant move {p.dominant}")
212
+ w(f" pan {_fmt(p.net_pan)} of frame width "
213
+ f"(x {_fmt(float(p.tx[-1]))}, y {_fmt(float(p.ty[-1]))})")
214
+ w(f" zoom {_fmt(p.zoom, 3)}x")
215
+ w(f" roll {_fmt(float(np.degrees(p.net_roll)), 2)} deg")
216
+ w(f" reversals {p.reversals} (measured, not judged)")
217
+ w(f" jerk {_fmt(p.jerk, 2)}")
218
+ w(f" incoherence {_fmt(p.incoherence)}")
219
+ w(f" morph {_fmt(p.morph, 3)} (used to find cuts, not judged)")
220
+ w(f" closure {_fmt(p.closure)}")
221
+ w(f" confidence {_fmt(p.confidence, 2)}")
222
+
223
+ if s.expect is not None:
224
+ w(f"\n asked for: {s.expect.move}")
225
+ w(f" {'HELD' if s.expect.ok else 'NOT HELD'} - {s.expect.detail}")
226
+
227
+ if not s.findings:
228
+ w("\n no findings: the motion is consistent with one physical camera.")
229
+ else:
230
+ w(f"\n {len(s.findings)} finding(s)")
231
+ for f in s.findings:
232
+ w(f" [{_MARK[f.severity]}] {f.summary}")
233
+ w(f" evidence: {f.evidence}")
234
+ if not quiet:
235
+ w(f" {f.advice}")
236
+ return "\n".join(out) + "\n"
shotdrift/expect.py ADDED
@@ -0,0 +1,132 @@
1
+ """Check a measured path against the move that was ASKED for.
2
+
3
+ This is the part that does not exist elsewhere. Everything else here describes
4
+ what a clip did; this asks whether it did what it was told, which is the actual
5
+ question when you type "slow push in" into a generator and get back something
6
+ that drifts left.
7
+
8
+ A declared move is checked three ways, because each failure is different and the
9
+ distinction is what makes the output useful:
10
+ happened - is there any motion on that channel at all?
11
+ direction - is it the way round you asked?
12
+ held - was it monotonic, or did it wobble there?
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass
18
+
19
+ import numpy as np
20
+
21
+ # name -> (channel, sign, human); sign 0 means "must not move"
22
+ # THE SIGNS ARE CAMERA-RELATIVE, NOT CONTENT-RELATIVE, and that is the one thing
23
+ # in this file worth reading twice. A camera panning RIGHT makes the picture move
24
+ # LEFT, so "pan-right" expects a NEGATIVE x. Writing these the intuitive way round
25
+ # made a correctly-measured clean pan report "x went the other way" - the
26
+ # measurement was right and the vocabulary was wrong. Verified by test.
27
+ MOVES: dict[str, tuple[str, int, str]] = {
28
+ "static": ("none", 0, "locked off"),
29
+ "locked": ("none", 0, "locked off"),
30
+ "push-in": ("zoom", +1, "push in / dolly in"), # subject grows
31
+ "dolly-in": ("zoom", +1, "push in / dolly in"),
32
+ "zoom-in": ("zoom", +1, "zoom in"),
33
+ "pull-out": ("zoom", -1, "pull out / dolly out"),
34
+ "dolly-out": ("zoom", -1, "pull out / dolly out"),
35
+ "zoom-out": ("zoom", -1, "zoom out"),
36
+ "pan-left": ("x", +1, "pan left"), # camera left -> picture right
37
+ "pan-right": ("x", -1, "pan right"), # camera right -> picture left
38
+ "tilt-up": ("y", +1, "tilt up"), # camera up -> picture down
39
+ "tilt-down": ("y", -1, "tilt down"),
40
+ "roll-cw": ("roll", -1, "roll clockwise"), # camera cw -> picture ccw
41
+ "roll-ccw": ("roll", +1, "roll counter-clockwise"),
42
+ }
43
+
44
+ # A move has to clear this to count as having happened at all, in frame widths
45
+ # (or radians / log-scale). Below it, the honest answer is "nothing happened".
46
+ FLOOR = {"x": 0.02, "y": 0.02, "zoom": 0.02, "roll": 0.02}
47
+ # Fraction of the travel allowed to run backwards before the move is not "held".
48
+ BACKTRACK_OK = 0.25
49
+
50
+
51
+ @dataclass
52
+ class Expectation:
53
+ move: str
54
+ ok: bool
55
+ happened: bool
56
+ direction_ok: bool
57
+ held: bool
58
+ measured: float
59
+ backtrack: float
60
+ detail: str
61
+
62
+ def as_dict(self) -> dict:
63
+ return dict(move=self.move, ok=self.ok, happened=self.happened,
64
+ direction_ok=self.direction_ok, held=self.held,
65
+ measured=round(self.measured, 5),
66
+ backtrack=round(self.backtrack, 4), detail=self.detail)
67
+
68
+
69
+ def known() -> list[str]:
70
+ return sorted(MOVES)
71
+
72
+
73
+ def _series(p, chan: str) -> np.ndarray:
74
+ if chan == "x":
75
+ return np.asarray(p.tx, dtype=float)
76
+ if chan == "y":
77
+ return np.asarray(p.ty, dtype=float)
78
+ if chan == "zoom":
79
+ return np.asarray(p.logscale, dtype=float)
80
+ if chan == "roll":
81
+ return np.asarray(p.roll, dtype=float)
82
+ raise KeyError(chan)
83
+
84
+
85
+ def check(p, move: str) -> Expectation:
86
+ key = move.strip().lower().replace("_", "-")
87
+ if key not in MOVES:
88
+ raise KeyError(f"unknown move {move!r}; known: {', '.join(known())}")
89
+ chan, sign, human = MOVES[key]
90
+
91
+ if chan == "none":
92
+ # "Locked off" is a claim about every channel at once, so it is the one
93
+ # case that cannot be judged on a single series.
94
+ worst_name, worst_val, worst_floor = "pan", p.net_pan, FLOOR["x"]
95
+ for nm, val, fl in (("pan", p.net_pan, FLOOR["x"]),
96
+ ("zoom", abs(float(np.log(max(p.zoom, 1e-6)))), FLOOR["zoom"]),
97
+ ("roll", abs(p.net_roll), FLOOR["roll"])):
98
+ if val / fl > worst_val / worst_floor:
99
+ worst_name, worst_val, worst_floor = nm, val, fl
100
+ still = worst_val < worst_floor
101
+ return Expectation(
102
+ move=key, ok=still, happened=not still, direction_ok=still, held=still,
103
+ measured=worst_val, backtrack=0.0,
104
+ detail=(f"locked off as asked; largest movement was {worst_name} "
105
+ f"{worst_val:.4f} (floor {worst_floor})") if still else
106
+ (f"asked for {human}, but {worst_name} moved {worst_val:.4f}, "
107
+ f"over the {worst_floor} floor"),
108
+ )
109
+
110
+ s = _series(p, chan)
111
+ net = float(s[-1] - s[0])
112
+ step = np.diff(s)
113
+ travel = float(np.sum(np.abs(step))) or 1e-9
114
+ against = float(np.sum(np.abs(step[np.sign(step) == -sign])))
115
+ backtrack = against / travel
116
+
117
+ happened = abs(net) >= FLOOR[chan]
118
+ direction_ok = bool(np.sign(net) == sign) if happened else False
119
+ held = backtrack <= BACKTRACK_OK
120
+ ok = happened and direction_ok and held
121
+
122
+ if not happened:
123
+ detail = (f"asked for {human}; {chan} moved {net:+.4f}, under the "
124
+ f"{FLOOR[chan]} floor - that move did not happen")
125
+ elif not direction_ok:
126
+ detail = (f"asked for {human}; {chan} went the other way, {net:+.4f}")
127
+ elif not held:
128
+ detail = (f"{human} happened ({net:+.4f}) but {backtrack * 100:.0f}% of the "
129
+ f"travel ran backwards - the move is not held")
130
+ else:
131
+ detail = f"{human} as asked: {chan} {net:+.4f}, {backtrack * 100:.0f}% backtrack"
132
+ return Expectation(key, ok, happened, direction_ok, held, net, backtrack, detail)
shotdrift/frames.py ADDED
@@ -0,0 +1,135 @@
1
+ """Decode a video to small grayscale frames with ffmpeg, and nothing else.
2
+
3
+ Deliberately no OpenCV and no torch. ffmpeg is already on every machine that
4
+ edits video, and a tool that needs a 2 GB wheel to measure a camera move will
5
+ not get run.
6
+
7
+ Everything downstream works in NORMALISED units - fractions of the frame width -
8
+ so a threshold means the same thing on a 720p proxy and a 4K master. Reporting
9
+ pixels at the analysis resolution would make every bound silently resolution
10
+ dependent, which is how a measurement tool ends up with magic numbers.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import shutil
17
+ import subprocess
18
+ from dataclasses import dataclass
19
+
20
+ import numpy as np
21
+
22
+ from .ingest import analysis_size
23
+
24
+
25
+ class FFmpegMissing(RuntimeError):
26
+ pass
27
+
28
+
29
+ class DecodeError(RuntimeError):
30
+ pass
31
+
32
+
33
+ @dataclass
34
+ class Clip:
35
+ frames: np.ndarray # (n, h, w) float32 in 0..1
36
+ width: int # analysis width (px)
37
+ height: int # analysis height (px)
38
+ src_width: int # original width (px)
39
+ src_height: int # original height (px)
40
+ fps: float
41
+ duration: float
42
+ truncated: bool = False # hit the frame budget with picture still to come
43
+
44
+ def __len__(self) -> int:
45
+ return int(self.frames.shape[0])
46
+
47
+
48
+ def _exe(name: str) -> str:
49
+ p = shutil.which(name)
50
+ if not p:
51
+ raise FFmpegMissing(
52
+ f"{name} not found on PATH. shotdrift measures video, so it needs "
53
+ "ffmpeg: `brew install ffmpeg` or `apt install ffmpeg`."
54
+ )
55
+ return p
56
+
57
+
58
+ def probe(path: str) -> dict:
59
+ out = subprocess.run(
60
+ [_exe("ffprobe"), "-v", "error", "-select_streams", "v:0",
61
+ "-show_entries", "stream=width,height,avg_frame_rate,nb_read_packets",
62
+ "-show_entries", "format=duration", "-of", "json", path],
63
+ capture_output=True, text=True,
64
+ )
65
+ if out.returncode != 0:
66
+ raise DecodeError(f"ffprobe could not read {path}: {out.stderr.strip()[:300]}")
67
+ d = json.loads(out.stdout or "{}")
68
+ streams = d.get("streams") or []
69
+ if not streams:
70
+ raise DecodeError(f"{path} has no video stream.")
71
+ s = streams[0]
72
+ num, _, den = (s.get("avg_frame_rate") or "0/1").partition("/")
73
+ try:
74
+ fps = float(num) / float(den) if float(den) else 0.0
75
+ except (ValueError, ZeroDivisionError):
76
+ fps = 0.0
77
+ try:
78
+ dur = float((d.get("format") or {}).get("duration") or 0.0)
79
+ except ValueError:
80
+ dur = 0.0
81
+ return {"width": int(s["width"]), "height": int(s["height"]), "fps": fps, "duration": dur}
82
+
83
+
84
+ def load(path: str, max_side: int = 512, sample_fps: float | None = None,
85
+ start: float = 0.0, duration: float | None = None,
86
+ max_frames: int = 600) -> Clip:
87
+ """Decode to grayscale. `sample_fps=None` keeps the clip's own rate.
88
+
89
+ Camera motion is measured BETWEEN ADJACENT FRAMES, so dropping the rate
90
+ changes what is being measured, not just the cost of measuring it. The
91
+ default therefore keeps every frame; `--sample-fps` exists for long clips
92
+ and says so in the report.
93
+ """
94
+ meta = probe(path)
95
+ sw, sh = meta["width"], meta["height"]
96
+ if sw <= 0 or sh <= 0:
97
+ raise DecodeError(f"{path} reports a {sw}x{sh} frame size.")
98
+
99
+ w, h = analysis_size(sw, sh, max_side)
100
+
101
+ cmd = [_exe("ffmpeg"), "-v", "error"]
102
+ if start > 0:
103
+ cmd += ["-ss", f"{start:.3f}"]
104
+ cmd += ["-i", path]
105
+ if duration is not None:
106
+ cmd += ["-t", f"{duration:.3f}"]
107
+ vf = [f"scale={w}:{h}:flags=bilinear"]
108
+ if sample_fps:
109
+ vf.insert(0, f"fps={sample_fps:g}")
110
+ cmd += ["-vf", ",".join(vf), "-frames:v", str(max_frames),
111
+ "-pix_fmt", "gray", "-f", "rawvideo", "-"]
112
+
113
+ out = subprocess.run(cmd, capture_output=True)
114
+ if out.returncode != 0:
115
+ raise DecodeError(f"ffmpeg failed on {path}: {out.stderr.decode(errors='replace').strip()[:300]}")
116
+
117
+ buf = np.frombuffer(out.stdout, dtype=np.uint8)
118
+ stride = w * h
119
+ n = buf.size // stride
120
+ if n < 2:
121
+ raise DecodeError(
122
+ f"{path} decoded to {n} frame(s) at {w}x{h}. Camera motion needs at "
123
+ "least two frames; check the file is not truncated and that any "
124
+ "--start/--duration window actually contains picture."
125
+ )
126
+ frames = buf[: n * stride].reshape(n, h, w).astype(np.float32) / 255.0
127
+ eff_fps = float(sample_fps) if sample_fps else meta["fps"]
128
+ # A frame budget that silently drops the rest of the clip is how a tool comes
129
+ # to report confidently on the first twelve seconds of a two-minute take. The
130
+ # window is reported, not assumed.
131
+ window = duration if duration is not None else max(0.0, meta["duration"] - start)
132
+ expected = window * eff_fps if (window and eff_fps) else 0.0
133
+ truncated = bool(n >= max_frames and expected > n + 1)
134
+ return Clip(frames=frames, width=w, height=h, src_width=sw, src_height=sh,
135
+ fps=eff_fps, duration=meta["duration"], truncated=truncated)