spritegen-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,172 @@
1
+ """The frame an animation starts from, and the one it ends on.
2
+
3
+ The anchor is neutral on purpose — facing south, weight even, hands empty. That is the
4
+ rest state a pose-driven endpoint deforms from, and an anchor caught mid-swing puts that
5
+ swing into every frame of the walk.
6
+
7
+ Which leaves a gap. A sword attack begins with the sword already in hand and already
8
+ drawn back, in a pose the anchor does not have. Asking a video model to introduce the
9
+ weapon *and* the movement in one call is asking for two things, and identity is the one
10
+ that gets dropped — the character comes back holding a different sword, or a different
11
+ character comes back holding the right one.
12
+
13
+ So the first frame is generated as an image: cheap, reviewable, and quoting the anchor for
14
+ identity, which separates the two questions. **The last frame closes the cycle**, and by
15
+ default it *is* the first — copied, not generated, because a cycle that ends where it
16
+ began is the ordinary case and paying twice for one image would be absurd. `seedance` and
17
+ `kling` take that as `end_image_url` and `tail_image_url`; `video --loop` was already
18
+ sending the anchor to both ends for want of anything better.
19
+
20
+ One pose per animation, under its own name: `pose/attack/start.png`. That is the same
21
+ shape `sheet/<direction>/` has, and for the same reason — one sprite has several, and
22
+ none of them is *the* pose.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import argparse
28
+ from pathlib import Path
29
+
30
+ from .. import fal, imaging, prompts, workspace
31
+
32
+ MODEL = "openai/gpt-image-2/edit"
33
+
34
+ #: What each file is called inside the pose's directory. `video` looks for these names.
35
+ START = "start.png"
36
+ END = "end.png"
37
+
38
+
39
+ def build_payload(args: argparse.Namespace, prompt: str, refs: list[str]) -> dict:
40
+ """The payload as the endpoint takes it.
41
+
42
+ Always the edit endpoint, never the text-to-image one: a pose that does not quote the
43
+ anchor is a different character in the first frame, which is the whole failure this
44
+ stage exists to avoid.
45
+ """
46
+ from .anchor import build_prompt, parse_size
47
+
48
+ return {
49
+ "prompt": build_prompt(args, prompt),
50
+ "image_urls": refs,
51
+ "image_size": parse_size(args.size),
52
+ "quality": args.quality,
53
+ "num_images": 1,
54
+ "output_format": "png",
55
+ }
56
+
57
+
58
+ def read_prompts(args: argparse.Namespace) -> tuple[str, str | None]:
59
+ """The start prompt, and the end prompt if one was given."""
60
+ start = prompts.resolve(args.prompt, args.prompt_file, "the pose")
61
+ if args.no_end:
62
+ if args.end_prompt or args.end_prompt_file:
63
+ raise ValueError("--no-end and an end prompt ask for opposite things")
64
+ return start, None
65
+ if not (args.end_prompt or args.end_prompt_file):
66
+ return start, None
67
+ return start, prompts.resolve(args.end_prompt, args.end_prompt_file, "the pose's end")
68
+
69
+
70
+ def check_name(name: str) -> str:
71
+ """The animation's name, if it can be a directory.
72
+
73
+ Checked here rather than left to the filesystem, and by the same rule an asset name
74
+ is: it becomes a path segment under `pose/`, so a separator in it would write outside
75
+ the sprite. `workspace.check_name` is that rule, and there is no second one.
76
+ """
77
+ if not name:
78
+ raise ValueError(
79
+ "a pose is for an animation; name it with --animation, e.g. --animation attack"
80
+ )
81
+ return workspace.check_name(name)
82
+
83
+
84
+ def run(args: argparse.Namespace) -> int:
85
+ check_name(args.animation)
86
+ context = workspace.context_from_args(args)
87
+ start_prompt, end_prompt = read_prompts(args)
88
+
89
+ anchor = context.source_files("*.png")
90
+ anchor = [path for path in anchor if not path.stem.endswith(".raw")]
91
+ if not anchor:
92
+ raise workspace.StageRefused(f"{context.asset}: no anchor in {context.source}")
93
+
94
+ refs = [anchor[0], *(Path(one) for one in (args.ref or []))]
95
+ for one in refs:
96
+ if not one.is_file():
97
+ raise FileNotFoundError(f"no reference image at {one}")
98
+
99
+ calls = 1 if end_prompt is None else 2
100
+
101
+ if context.dry_run:
102
+ fal.show(MODEL, build_payload(args, start_prompt, [str(one) for one in refs]))
103
+ if end_prompt is not None:
104
+ fal.show(MODEL, build_payload(args, end_prompt, [str(one) for one in refs]))
105
+ closing = "copied from the start frame" if not args.no_end else "not written"
106
+ print(f"would spend {calls} call(s); the end frame would be {closing}")
107
+ return 0
108
+
109
+ fal.require_key()
110
+ context.keep_prompt(start_prompt)
111
+ uploaded = [fal.upload(one) for one in refs]
112
+ out = context.ensure_out()
113
+
114
+ written = [_one(context, args, start_prompt, uploaded, out / START)]
115
+ report = {"start": _finish(args, out / START)}
116
+
117
+ if args.no_end:
118
+ pass
119
+ elif end_prompt is None:
120
+ # Copied, not generated. A cycle that ends where it began is the ordinary case,
121
+ # and a second paid call for the same image would buy nothing but drift.
122
+ (out / END).write_bytes((out / START).read_bytes())
123
+ written.append(out / END)
124
+ report["end"] = "copied from the start frame"
125
+ else:
126
+ # Before its call, like the start's: a second paid call whose text nobody kept is
127
+ # an image nobody can regenerate or amend, which is the failure `prompts` exists
128
+ # to prevent — and it is worse here, because the artifact would then cite the
129
+ # start's text for both.
130
+ context.keep_prompt(end_prompt, part="end")
131
+ written.append(_one(context, args, end_prompt, uploaded, out / END))
132
+ report["end"] = _finish(args, out / END)
133
+
134
+ context.record(
135
+ endpoint=MODEL,
136
+ animation=args.animation,
137
+ images=[path.name for path in written],
138
+ reports=report,
139
+ )
140
+ for path in written:
141
+ print(path)
142
+ return 0
143
+
144
+
145
+ def _one(context, args: argparse.Namespace, prompt: str, refs: list[str], dest: Path) -> Path:
146
+ """One paid call, downloaded to `dest` and recorded against the asset."""
147
+ payload = build_payload(args, prompt, refs)
148
+ result = fal.call(MODEL, payload)
149
+
150
+ urls = fal.urls_in(result)
151
+ if not urls:
152
+ raise ValueError(f"{MODEL} returned no image; nothing to write")
153
+
154
+ written = fal.download(urls[0], dest)
155
+ context.paid_call(endpoint=MODEL, payload=payload, urls=urls[:1], files=[written])
156
+ return written
157
+
158
+
159
+ def _finish(args: argparse.Namespace, path: Path) -> dict:
160
+ """The unpaid transforms, the same ones and in the same order the anchor applies."""
161
+ report: dict = {}
162
+ if args.transparent:
163
+ report["chroma"] = imaging.cut_chroma(
164
+ path,
165
+ imaging.parse_chroma(args.chroma),
166
+ args.tol,
167
+ args.feather,
168
+ args.despill,
169
+ args.despill_radius,
170
+ args.chroma_mode,
171
+ )
172
+ return report
@@ -0,0 +1,196 @@
1
+ """Animate the anchor directly and pull a board of frames out of the clip.
2
+
3
+ Absorbed from the `gen_video.py` this grew out of. Four endpoints in two shapes:
4
+
5
+ - **grok**, reference-to-video, takes a **list** of references and the prompt addresses
6
+ each by `<IMAGE_0>`, `<IMAGE_1>` in the order given. The only one here that takes more
7
+ than one image, so the only one where the character and a pose sheet go up together.
8
+ - **grok-i2v**, **seedance**, **kling**, image-to-video, take one first frame. The last
9
+ two also take a last frame, and sending the anchor as both is what closes the cycle:
10
+ the final frame meets the first, which is the condition for a walk that loops. Without
11
+ it the clip ends wherever it likes and the seam jumps.
12
+
13
+ The anchor is not an argument. It is the asset's own anchor, which is what stops a clip
14
+ from being generated against an image nobody can find again.
15
+
16
+ **Video is billed per second of output.** `--duration` and `--resolution` are the two
17
+ options that move the bill, and this is the most expensive call in the tool.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import argparse
23
+ from pathlib import Path
24
+
25
+ from .. import clip, fal, prompts, workspace
26
+
27
+ MODELS = {
28
+ "grok": "xai/grok-imagine-video/v1.5/reference-to-video",
29
+ "grok-i2v": "xai/grok-imagine-video/image-to-video",
30
+ "seedance": "bytedance/seedance-2.5/image-to-video",
31
+ "kling": "fal-ai/kling-video/v2.5-turbo/pro/image-to-video",
32
+ }
33
+
34
+ #: The one endpoint that takes a list of references rather than a single first frame.
35
+ REFERENCE_LIST = {"grok"}
36
+
37
+ #: What each endpoint calls its last frame. None means it does not take one.
38
+ TAIL_FIELD = {
39
+ "seedance": "end_image_url",
40
+ "kling": "tail_image_url",
41
+ "grok-i2v": None,
42
+ "grok": None,
43
+ }
44
+
45
+
46
+ def read_prompt(args: argparse.Namespace) -> str:
47
+ return prompts.resolve(args.prompt, args.prompt_file, "the video stage")
48
+
49
+
50
+ def first_and_last(context, args: argparse.Namespace) -> tuple[Path, Path | None]:
51
+ """What the clip starts from, and what it ends on.
52
+
53
+ Without `--pose` this is the anchor, and `--loop` sends it to both ends — which is
54
+ what there was before a pose existed, and the reason `--loop` reads oddly: the clip
55
+ was told to return to a rest state it never left.
56
+
57
+ With `--pose` the clip starts from that animation's first frame and ends on its last,
58
+ which is the first frame again unless somebody generated a different one.
59
+
60
+ `--loop` on top of a pose usually has nothing left to say, since the pose already
61
+ closes. It has one thing to say on a pose made with `--no-end` — a death, a movement
62
+ that does not return — where asking for a loop and silently getting no last frame
63
+ would be a trap. There, `--loop` means what it always meant: end where you began.
64
+ """
65
+ if not args.pose:
66
+ # `.raw` is the copy `upscale` keeps of what the paid call returned, and it is
67
+ # not the anchor: animating it would throw away the enlargement. Sorting hid
68
+ # this — `anchor.raw.png` follows `anchor.png` for the same stem — but a
69
+ # directory holding only the backup would have animated it.
70
+ anchors = [
71
+ path for path in context.source_files("*.png") if not path.stem.endswith(".raw")
72
+ ]
73
+ if not anchors:
74
+ raise workspace.StageRefused(f"{context.asset}: no anchor in {context.source}")
75
+ # `--loop` is this stage's option; `motion` shares the lookup and not the flag.
76
+ return anchors[0], (anchors[0] if getattr(args, "loop", False) else None)
77
+
78
+ from . import pose as pose_stage
79
+
80
+ key = f"pose:{pose_stage.check_name(args.pose)}"
81
+ entry = context.state.artifacts.get(key)
82
+ if entry is None:
83
+ made = ", ".join(sorted(k for k in context.state.artifacts if k.startswith("pose:")))
84
+ raise workspace.StageRefused(
85
+ f"{context.asset}: no pose {args.pose!r}; it has: {made or 'none'}"
86
+ )
87
+
88
+ directory = workspace.inside(context.asset, entry["dir"])
89
+ start = directory / pose_stage.START
90
+ if not start.is_file():
91
+ raise workspace.StageRefused(f"{context.asset}: {start} is not there")
92
+ end = directory / pose_stage.END
93
+ if end.is_file():
94
+ return start, end
95
+ return start, (start if getattr(args, "loop", False) else None)
96
+
97
+
98
+ def build_payload(args: argparse.Namespace, prompt: str, urls: dict) -> dict:
99
+ """The payload each endpoint takes, from the URLs already uploaded."""
100
+ if args.model in REFERENCE_LIST:
101
+ return {
102
+ "prompt": prompt,
103
+ "reference_image_urls": urls["refs"],
104
+ "duration": args.duration,
105
+ "resolution": args.resolution,
106
+ "aspect_ratio": args.aspect,
107
+ }
108
+
109
+ payload: dict = {"prompt": prompt, "image_url": urls["image"]}
110
+
111
+ if "end" in urls:
112
+ tail = TAIL_FIELD[args.model]
113
+ if tail is None:
114
+ asked = "--pose" if args.pose else "--loop"
115
+ raise ValueError(f"{args.model} does not take a last frame, so {asked} cannot use one")
116
+ payload[tail] = urls["end"]
117
+
118
+ if args.model == "seedance":
119
+ payload["duration"] = str(args.duration)
120
+ payload["resolution"] = args.resolution
121
+ payload["generate_audio"] = False
122
+ elif args.model == "kling":
123
+ # Kling only takes 5 or 10 seconds.
124
+ payload["duration"] = "10" if args.duration > 5 else "5"
125
+ payload["negative_prompt"] = args.negative
126
+ else:
127
+ payload["duration"] = args.duration
128
+ payload["resolution"] = args.resolution
129
+ return payload
130
+
131
+
132
+ def run(args: argparse.Namespace) -> int:
133
+ context = workspace.context_from_args(args)
134
+ prompt = read_prompt(args)
135
+ pick = clip.parse_pick(args.pick)
136
+
137
+ first, last = first_and_last(context, args)
138
+
139
+ extra = [Path(one) for one in (args.ref or [])]
140
+ for one in extra:
141
+ if not one.is_file():
142
+ raise FileNotFoundError(f"no reference image at {one}")
143
+ if extra and args.model not in REFERENCE_LIST:
144
+ raise ValueError(f"{args.model} takes one first frame, not a reference list")
145
+
146
+ endpoint = MODELS[args.model]
147
+
148
+ if context.dry_run:
149
+ placeholder = (
150
+ {"refs": [str(path) for path in [first, *extra]]}
151
+ if args.model in REFERENCE_LIST
152
+ else {"image": str(first), **({"end": str(last)} if last else {})}
153
+ )
154
+ fal.show(endpoint, build_payload(args, prompt, placeholder))
155
+ return 0
156
+
157
+ fal.require_key()
158
+ context.keep_prompt(prompt)
159
+ out = context.ensure_out()
160
+
161
+ if args.model in REFERENCE_LIST:
162
+ urls = {"refs": [fal.upload(path) for path in [first, *extra]]}
163
+ else:
164
+ uploaded = fal.upload(first)
165
+ # The same file uploads once: `--loop` on an anchor sends one URL to both ends,
166
+ # and a pose whose end frame is a copy of its start does the same.
167
+ tail = uploaded if last == first else fal.upload(last) if last else None
168
+ urls = {"image": uploaded, **({"end": tail} if tail else {})}
169
+
170
+ payload = build_payload(args, prompt, urls)
171
+ result = fal.call(endpoint, payload)
172
+
173
+ found = fal.urls_in(result)
174
+ if not found:
175
+ raise ValueError(f"{endpoint} returned no clip")
176
+
177
+ video = fal.download(found[0], out / "clip.mp4")
178
+ frames = clip.decode_frames(video)
179
+ chosen, indices = clip.choose_frames(frames, args.frames, args.start, args.end, pick)
180
+ board = out / "board.png"
181
+ cols, rows = clip.pack_board(chosen, args.cols, board)
182
+
183
+ context.paid_call(endpoint=endpoint, payload=payload, urls=found, files=[video, board])
184
+ context.record(
185
+ endpoint=endpoint,
186
+ model=args.model,
187
+ clip="clip.mp4",
188
+ board="board.png",
189
+ decoded=len(frames),
190
+ indices=indices,
191
+ grid=f"{cols}x{rows}",
192
+ )
193
+
194
+ print(f"{video} {len(frames)} frames decoded")
195
+ print(f"{board} {len(chosen)} frames at {indices}, packed {cols}x{rows}")
196
+ return 0