spritegen-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spritegen/__init__.py +3 -0
- spritegen/atlas.py +99 -0
- spritegen/cli.py +178 -0
- spritegen/clip.py +81 -0
- spritegen/drive.py +163 -0
- spritegen/endpoints.py +113 -0
- spritegen/fal.py +228 -0
- spritegen/imaging.py +219 -0
- spritegen/ledger.py +143 -0
- spritegen/matting.py +136 -0
- spritegen/migrate.py +162 -0
- spritegen/prompts.py +96 -0
- spritegen/rrdb.py +91 -0
- spritegen/settings.py +127 -0
- spritegen/sheet.py +370 -0
- spritegen/skill/__init__.py +303 -0
- spritegen/skill/files/SKILL.md +553 -0
- spritegen/stages/__init__.py +490 -0
- spritegen/stages/anchor.py +215 -0
- spritegen/stages/board.py +68 -0
- spritegen/stages/matte.py +209 -0
- spritegen/stages/motion.py +164 -0
- spritegen/stages/pose.py +172 -0
- spritegen/stages/video.py +196 -0
- spritegen/upscale.py +444 -0
- spritegen/workspace.py +852 -0
- spritegen_cli-0.1.0.dist-info/METADATA +16 -0
- spritegen_cli-0.1.0.dist-info/RECORD +30 -0
- spritegen_cli-0.1.0.dist-info/WHEEL +4 -0
- spritegen_cli-0.1.0.dist-info/entry_points.txt +2 -0
spritegen/stages/pose.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""The frame an animation starts from, and the one it ends on.
|
|
2
|
+
|
|
3
|
+
The anchor is neutral on purpose — facing south, weight even, hands empty. That is the
|
|
4
|
+
rest state a pose-driven endpoint deforms from, and an anchor caught mid-swing puts that
|
|
5
|
+
swing into every frame of the walk.
|
|
6
|
+
|
|
7
|
+
Which leaves a gap. A sword attack begins with the sword already in hand and already
|
|
8
|
+
drawn back, in a pose the anchor does not have. Asking a video model to introduce the
|
|
9
|
+
weapon *and* the movement in one call is asking for two things, and identity is the one
|
|
10
|
+
that gets dropped — the character comes back holding a different sword, or a different
|
|
11
|
+
character comes back holding the right one.
|
|
12
|
+
|
|
13
|
+
So the first frame is generated as an image: cheap, reviewable, and quoting the anchor for
|
|
14
|
+
identity, which separates the two questions. **The last frame closes the cycle**, and by
|
|
15
|
+
default it *is* the first — copied, not generated, because a cycle that ends where it
|
|
16
|
+
began is the ordinary case and paying twice for one image would be absurd. `seedance` and
|
|
17
|
+
`kling` take that as `end_image_url` and `tail_image_url`; `video --loop` was already
|
|
18
|
+
sending the anchor to both ends for want of anything better.
|
|
19
|
+
|
|
20
|
+
One pose per animation, under its own name: `pose/attack/start.png`. That is the same
|
|
21
|
+
shape `sheet/<direction>/` has, and for the same reason — one sprite has several, and
|
|
22
|
+
none of them is *the* pose.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import argparse
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
from .. import fal, imaging, prompts, workspace
|
|
31
|
+
|
|
32
|
+
MODEL = "openai/gpt-image-2/edit"
|
|
33
|
+
|
|
34
|
+
#: What each file is called inside the pose's directory. `video` looks for these names.
|
|
35
|
+
START = "start.png"
|
|
36
|
+
END = "end.png"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def build_payload(args: argparse.Namespace, prompt: str, refs: list[str]) -> dict:
|
|
40
|
+
"""The payload as the endpoint takes it.
|
|
41
|
+
|
|
42
|
+
Always the edit endpoint, never the text-to-image one: a pose that does not quote the
|
|
43
|
+
anchor is a different character in the first frame, which is the whole failure this
|
|
44
|
+
stage exists to avoid.
|
|
45
|
+
"""
|
|
46
|
+
from .anchor import build_prompt, parse_size
|
|
47
|
+
|
|
48
|
+
return {
|
|
49
|
+
"prompt": build_prompt(args, prompt),
|
|
50
|
+
"image_urls": refs,
|
|
51
|
+
"image_size": parse_size(args.size),
|
|
52
|
+
"quality": args.quality,
|
|
53
|
+
"num_images": 1,
|
|
54
|
+
"output_format": "png",
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def read_prompts(args: argparse.Namespace) -> tuple[str, str | None]:
|
|
59
|
+
"""The start prompt, and the end prompt if one was given."""
|
|
60
|
+
start = prompts.resolve(args.prompt, args.prompt_file, "the pose")
|
|
61
|
+
if args.no_end:
|
|
62
|
+
if args.end_prompt or args.end_prompt_file:
|
|
63
|
+
raise ValueError("--no-end and an end prompt ask for opposite things")
|
|
64
|
+
return start, None
|
|
65
|
+
if not (args.end_prompt or args.end_prompt_file):
|
|
66
|
+
return start, None
|
|
67
|
+
return start, prompts.resolve(args.end_prompt, args.end_prompt_file, "the pose's end")
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def check_name(name: str) -> str:
|
|
71
|
+
"""The animation's name, if it can be a directory.
|
|
72
|
+
|
|
73
|
+
Checked here rather than left to the filesystem, and by the same rule an asset name
|
|
74
|
+
is: it becomes a path segment under `pose/`, so a separator in it would write outside
|
|
75
|
+
the sprite. `workspace.check_name` is that rule, and there is no second one.
|
|
76
|
+
"""
|
|
77
|
+
if not name:
|
|
78
|
+
raise ValueError(
|
|
79
|
+
"a pose is for an animation; name it with --animation, e.g. --animation attack"
|
|
80
|
+
)
|
|
81
|
+
return workspace.check_name(name)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def run(args: argparse.Namespace) -> int:
|
|
85
|
+
check_name(args.animation)
|
|
86
|
+
context = workspace.context_from_args(args)
|
|
87
|
+
start_prompt, end_prompt = read_prompts(args)
|
|
88
|
+
|
|
89
|
+
anchor = context.source_files("*.png")
|
|
90
|
+
anchor = [path for path in anchor if not path.stem.endswith(".raw")]
|
|
91
|
+
if not anchor:
|
|
92
|
+
raise workspace.StageRefused(f"{context.asset}: no anchor in {context.source}")
|
|
93
|
+
|
|
94
|
+
refs = [anchor[0], *(Path(one) for one in (args.ref or []))]
|
|
95
|
+
for one in refs:
|
|
96
|
+
if not one.is_file():
|
|
97
|
+
raise FileNotFoundError(f"no reference image at {one}")
|
|
98
|
+
|
|
99
|
+
calls = 1 if end_prompt is None else 2
|
|
100
|
+
|
|
101
|
+
if context.dry_run:
|
|
102
|
+
fal.show(MODEL, build_payload(args, start_prompt, [str(one) for one in refs]))
|
|
103
|
+
if end_prompt is not None:
|
|
104
|
+
fal.show(MODEL, build_payload(args, end_prompt, [str(one) for one in refs]))
|
|
105
|
+
closing = "copied from the start frame" if not args.no_end else "not written"
|
|
106
|
+
print(f"would spend {calls} call(s); the end frame would be {closing}")
|
|
107
|
+
return 0
|
|
108
|
+
|
|
109
|
+
fal.require_key()
|
|
110
|
+
context.keep_prompt(start_prompt)
|
|
111
|
+
uploaded = [fal.upload(one) for one in refs]
|
|
112
|
+
out = context.ensure_out()
|
|
113
|
+
|
|
114
|
+
written = [_one(context, args, start_prompt, uploaded, out / START)]
|
|
115
|
+
report = {"start": _finish(args, out / START)}
|
|
116
|
+
|
|
117
|
+
if args.no_end:
|
|
118
|
+
pass
|
|
119
|
+
elif end_prompt is None:
|
|
120
|
+
# Copied, not generated. A cycle that ends where it began is the ordinary case,
|
|
121
|
+
# and a second paid call for the same image would buy nothing but drift.
|
|
122
|
+
(out / END).write_bytes((out / START).read_bytes())
|
|
123
|
+
written.append(out / END)
|
|
124
|
+
report["end"] = "copied from the start frame"
|
|
125
|
+
else:
|
|
126
|
+
# Before its call, like the start's: a second paid call whose text nobody kept is
|
|
127
|
+
# an image nobody can regenerate or amend, which is the failure `prompts` exists
|
|
128
|
+
# to prevent — and it is worse here, because the artifact would then cite the
|
|
129
|
+
# start's text for both.
|
|
130
|
+
context.keep_prompt(end_prompt, part="end")
|
|
131
|
+
written.append(_one(context, args, end_prompt, uploaded, out / END))
|
|
132
|
+
report["end"] = _finish(args, out / END)
|
|
133
|
+
|
|
134
|
+
context.record(
|
|
135
|
+
endpoint=MODEL,
|
|
136
|
+
animation=args.animation,
|
|
137
|
+
images=[path.name for path in written],
|
|
138
|
+
reports=report,
|
|
139
|
+
)
|
|
140
|
+
for path in written:
|
|
141
|
+
print(path)
|
|
142
|
+
return 0
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _one(context, args: argparse.Namespace, prompt: str, refs: list[str], dest: Path) -> Path:
|
|
146
|
+
"""One paid call, downloaded to `dest` and recorded against the asset."""
|
|
147
|
+
payload = build_payload(args, prompt, refs)
|
|
148
|
+
result = fal.call(MODEL, payload)
|
|
149
|
+
|
|
150
|
+
urls = fal.urls_in(result)
|
|
151
|
+
if not urls:
|
|
152
|
+
raise ValueError(f"{MODEL} returned no image; nothing to write")
|
|
153
|
+
|
|
154
|
+
written = fal.download(urls[0], dest)
|
|
155
|
+
context.paid_call(endpoint=MODEL, payload=payload, urls=urls[:1], files=[written])
|
|
156
|
+
return written
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _finish(args: argparse.Namespace, path: Path) -> dict:
|
|
160
|
+
"""The unpaid transforms, the same ones and in the same order the anchor applies."""
|
|
161
|
+
report: dict = {}
|
|
162
|
+
if args.transparent:
|
|
163
|
+
report["chroma"] = imaging.cut_chroma(
|
|
164
|
+
path,
|
|
165
|
+
imaging.parse_chroma(args.chroma),
|
|
166
|
+
args.tol,
|
|
167
|
+
args.feather,
|
|
168
|
+
args.despill,
|
|
169
|
+
args.despill_radius,
|
|
170
|
+
args.chroma_mode,
|
|
171
|
+
)
|
|
172
|
+
return report
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""Animate the anchor directly and pull a board of frames out of the clip.
|
|
2
|
+
|
|
3
|
+
Absorbed from the `gen_video.py` this grew out of. Four endpoints in two shapes:
|
|
4
|
+
|
|
5
|
+
- **grok**, reference-to-video, takes a **list** of references and the prompt addresses
|
|
6
|
+
each by `<IMAGE_0>`, `<IMAGE_1>` in the order given. The only one here that takes more
|
|
7
|
+
than one image, so the only one where the character and a pose sheet go up together.
|
|
8
|
+
- **grok-i2v**, **seedance**, **kling**, image-to-video, take one first frame. The last
|
|
9
|
+
two also take a last frame, and sending the anchor as both is what closes the cycle:
|
|
10
|
+
the final frame meets the first, which is the condition for a walk that loops. Without
|
|
11
|
+
it the clip ends wherever it likes and the seam jumps.
|
|
12
|
+
|
|
13
|
+
The anchor is not an argument. It is the asset's own anchor, which is what stops a clip
|
|
14
|
+
from being generated against an image nobody can find again.
|
|
15
|
+
|
|
16
|
+
**Video is billed per second of output.** `--duration` and `--resolution` are the two
|
|
17
|
+
options that move the bill, and this is the most expensive call in the tool.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import argparse
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
from .. import clip, fal, prompts, workspace
|
|
26
|
+
|
|
27
|
+
MODELS = {
|
|
28
|
+
"grok": "xai/grok-imagine-video/v1.5/reference-to-video",
|
|
29
|
+
"grok-i2v": "xai/grok-imagine-video/image-to-video",
|
|
30
|
+
"seedance": "bytedance/seedance-2.5/image-to-video",
|
|
31
|
+
"kling": "fal-ai/kling-video/v2.5-turbo/pro/image-to-video",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
#: The one endpoint that takes a list of references rather than a single first frame.
|
|
35
|
+
REFERENCE_LIST = {"grok"}
|
|
36
|
+
|
|
37
|
+
#: What each endpoint calls its last frame. None means it does not take one.
|
|
38
|
+
TAIL_FIELD = {
|
|
39
|
+
"seedance": "end_image_url",
|
|
40
|
+
"kling": "tail_image_url",
|
|
41
|
+
"grok-i2v": None,
|
|
42
|
+
"grok": None,
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def read_prompt(args: argparse.Namespace) -> str:
|
|
47
|
+
return prompts.resolve(args.prompt, args.prompt_file, "the video stage")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def first_and_last(context, args: argparse.Namespace) -> tuple[Path, Path | None]:
|
|
51
|
+
"""What the clip starts from, and what it ends on.
|
|
52
|
+
|
|
53
|
+
Without `--pose` this is the anchor, and `--loop` sends it to both ends — which is
|
|
54
|
+
what there was before a pose existed, and the reason `--loop` reads oddly: the clip
|
|
55
|
+
was told to return to a rest state it never left.
|
|
56
|
+
|
|
57
|
+
With `--pose` the clip starts from that animation's first frame and ends on its last,
|
|
58
|
+
which is the first frame again unless somebody generated a different one.
|
|
59
|
+
|
|
60
|
+
`--loop` on top of a pose usually has nothing left to say, since the pose already
|
|
61
|
+
closes. It has one thing to say on a pose made with `--no-end` — a death, a movement
|
|
62
|
+
that does not return — where asking for a loop and silently getting no last frame
|
|
63
|
+
would be a trap. There, `--loop` means what it always meant: end where you began.
|
|
64
|
+
"""
|
|
65
|
+
if not args.pose:
|
|
66
|
+
# `.raw` is the copy `upscale` keeps of what the paid call returned, and it is
|
|
67
|
+
# not the anchor: animating it would throw away the enlargement. Sorting hid
|
|
68
|
+
# this — `anchor.raw.png` follows `anchor.png` for the same stem — but a
|
|
69
|
+
# directory holding only the backup would have animated it.
|
|
70
|
+
anchors = [
|
|
71
|
+
path for path in context.source_files("*.png") if not path.stem.endswith(".raw")
|
|
72
|
+
]
|
|
73
|
+
if not anchors:
|
|
74
|
+
raise workspace.StageRefused(f"{context.asset}: no anchor in {context.source}")
|
|
75
|
+
# `--loop` is this stage's option; `motion` shares the lookup and not the flag.
|
|
76
|
+
return anchors[0], (anchors[0] if getattr(args, "loop", False) else None)
|
|
77
|
+
|
|
78
|
+
from . import pose as pose_stage
|
|
79
|
+
|
|
80
|
+
key = f"pose:{pose_stage.check_name(args.pose)}"
|
|
81
|
+
entry = context.state.artifacts.get(key)
|
|
82
|
+
if entry is None:
|
|
83
|
+
made = ", ".join(sorted(k for k in context.state.artifacts if k.startswith("pose:")))
|
|
84
|
+
raise workspace.StageRefused(
|
|
85
|
+
f"{context.asset}: no pose {args.pose!r}; it has: {made or 'none'}"
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
directory = workspace.inside(context.asset, entry["dir"])
|
|
89
|
+
start = directory / pose_stage.START
|
|
90
|
+
if not start.is_file():
|
|
91
|
+
raise workspace.StageRefused(f"{context.asset}: {start} is not there")
|
|
92
|
+
end = directory / pose_stage.END
|
|
93
|
+
if end.is_file():
|
|
94
|
+
return start, end
|
|
95
|
+
return start, (start if getattr(args, "loop", False) else None)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def build_payload(args: argparse.Namespace, prompt: str, urls: dict) -> dict:
|
|
99
|
+
"""The payload each endpoint takes, from the URLs already uploaded."""
|
|
100
|
+
if args.model in REFERENCE_LIST:
|
|
101
|
+
return {
|
|
102
|
+
"prompt": prompt,
|
|
103
|
+
"reference_image_urls": urls["refs"],
|
|
104
|
+
"duration": args.duration,
|
|
105
|
+
"resolution": args.resolution,
|
|
106
|
+
"aspect_ratio": args.aspect,
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
payload: dict = {"prompt": prompt, "image_url": urls["image"]}
|
|
110
|
+
|
|
111
|
+
if "end" in urls:
|
|
112
|
+
tail = TAIL_FIELD[args.model]
|
|
113
|
+
if tail is None:
|
|
114
|
+
asked = "--pose" if args.pose else "--loop"
|
|
115
|
+
raise ValueError(f"{args.model} does not take a last frame, so {asked} cannot use one")
|
|
116
|
+
payload[tail] = urls["end"]
|
|
117
|
+
|
|
118
|
+
if args.model == "seedance":
|
|
119
|
+
payload["duration"] = str(args.duration)
|
|
120
|
+
payload["resolution"] = args.resolution
|
|
121
|
+
payload["generate_audio"] = False
|
|
122
|
+
elif args.model == "kling":
|
|
123
|
+
# Kling only takes 5 or 10 seconds.
|
|
124
|
+
payload["duration"] = "10" if args.duration > 5 else "5"
|
|
125
|
+
payload["negative_prompt"] = args.negative
|
|
126
|
+
else:
|
|
127
|
+
payload["duration"] = args.duration
|
|
128
|
+
payload["resolution"] = args.resolution
|
|
129
|
+
return payload
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def run(args: argparse.Namespace) -> int:
|
|
133
|
+
context = workspace.context_from_args(args)
|
|
134
|
+
prompt = read_prompt(args)
|
|
135
|
+
pick = clip.parse_pick(args.pick)
|
|
136
|
+
|
|
137
|
+
first, last = first_and_last(context, args)
|
|
138
|
+
|
|
139
|
+
extra = [Path(one) for one in (args.ref or [])]
|
|
140
|
+
for one in extra:
|
|
141
|
+
if not one.is_file():
|
|
142
|
+
raise FileNotFoundError(f"no reference image at {one}")
|
|
143
|
+
if extra and args.model not in REFERENCE_LIST:
|
|
144
|
+
raise ValueError(f"{args.model} takes one first frame, not a reference list")
|
|
145
|
+
|
|
146
|
+
endpoint = MODELS[args.model]
|
|
147
|
+
|
|
148
|
+
if context.dry_run:
|
|
149
|
+
placeholder = (
|
|
150
|
+
{"refs": [str(path) for path in [first, *extra]]}
|
|
151
|
+
if args.model in REFERENCE_LIST
|
|
152
|
+
else {"image": str(first), **({"end": str(last)} if last else {})}
|
|
153
|
+
)
|
|
154
|
+
fal.show(endpoint, build_payload(args, prompt, placeholder))
|
|
155
|
+
return 0
|
|
156
|
+
|
|
157
|
+
fal.require_key()
|
|
158
|
+
context.keep_prompt(prompt)
|
|
159
|
+
out = context.ensure_out()
|
|
160
|
+
|
|
161
|
+
if args.model in REFERENCE_LIST:
|
|
162
|
+
urls = {"refs": [fal.upload(path) for path in [first, *extra]]}
|
|
163
|
+
else:
|
|
164
|
+
uploaded = fal.upload(first)
|
|
165
|
+
# The same file uploads once: `--loop` on an anchor sends one URL to both ends,
|
|
166
|
+
# and a pose whose end frame is a copy of its start does the same.
|
|
167
|
+
tail = uploaded if last == first else fal.upload(last) if last else None
|
|
168
|
+
urls = {"image": uploaded, **({"end": tail} if tail else {})}
|
|
169
|
+
|
|
170
|
+
payload = build_payload(args, prompt, urls)
|
|
171
|
+
result = fal.call(endpoint, payload)
|
|
172
|
+
|
|
173
|
+
found = fal.urls_in(result)
|
|
174
|
+
if not found:
|
|
175
|
+
raise ValueError(f"{endpoint} returned no clip")
|
|
176
|
+
|
|
177
|
+
video = fal.download(found[0], out / "clip.mp4")
|
|
178
|
+
frames = clip.decode_frames(video)
|
|
179
|
+
chosen, indices = clip.choose_frames(frames, args.frames, args.start, args.end, pick)
|
|
180
|
+
board = out / "board.png"
|
|
181
|
+
cols, rows = clip.pack_board(chosen, args.cols, board)
|
|
182
|
+
|
|
183
|
+
context.paid_call(endpoint=endpoint, payload=payload, urls=found, files=[video, board])
|
|
184
|
+
context.record(
|
|
185
|
+
endpoint=endpoint,
|
|
186
|
+
model=args.model,
|
|
187
|
+
clip="clip.mp4",
|
|
188
|
+
board="board.png",
|
|
189
|
+
decoded=len(frames),
|
|
190
|
+
indices=indices,
|
|
191
|
+
grid=f"{cols}x{rows}",
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
print(f"{video} {len(frames)} frames decoded")
|
|
195
|
+
print(f"{board} {len(chosen)} frames at {indices}, packed {cols}x{rows}")
|
|
196
|
+
return 0
|