spritegen-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spritegen/__init__.py +3 -0
- spritegen/atlas.py +99 -0
- spritegen/cli.py +178 -0
- spritegen/clip.py +81 -0
- spritegen/drive.py +163 -0
- spritegen/endpoints.py +113 -0
- spritegen/fal.py +228 -0
- spritegen/imaging.py +219 -0
- spritegen/ledger.py +143 -0
- spritegen/matting.py +136 -0
- spritegen/migrate.py +162 -0
- spritegen/prompts.py +96 -0
- spritegen/rrdb.py +91 -0
- spritegen/settings.py +127 -0
- spritegen/sheet.py +370 -0
- spritegen/skill/__init__.py +303 -0
- spritegen/skill/files/SKILL.md +553 -0
- spritegen/stages/__init__.py +490 -0
- spritegen/stages/anchor.py +215 -0
- spritegen/stages/board.py +68 -0
- spritegen/stages/matte.py +209 -0
- spritegen/stages/motion.py +164 -0
- spritegen/stages/pose.py +172 -0
- spritegen/stages/video.py +196 -0
- spritegen/upscale.py +444 -0
- spritegen/workspace.py +852 -0
- spritegen_cli-0.1.0.dist-info/METADATA +16 -0
- spritegen_cli-0.1.0.dist-info/RECORD +30 -0
- spritegen_cli-0.1.0.dist-info/WHEEL +4 -0
- spritegen_cli-0.1.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""One generated image: an anchor, box art, or an icon.
|
|
2
|
+
|
|
3
|
+
`--kind` says which, and that is the directory it goes in — all three come out of this
|
|
4
|
+
stage and none of them is the others. The **anchor** is the neutral sprite every later
|
|
5
|
+
stage animates, and it is the one call that decides everything after it: a wrong anchor
|
|
6
|
+
wastes every paid call downstream, which is why `--dry-run` prints the payload and why
|
|
7
|
+
the report names what it found. **Box art** settles the look before a sprite exists and
|
|
8
|
+
then becomes the anchor's identity reference. An **icon** is one object and stops here.
|
|
9
|
+
|
|
10
|
+
Absorbed from the `gen_image.py` this grew out of. What changed is where things live, not what
|
|
11
|
+
they do — the transforms are in `imaging` and pinned by `tests/parity`, and the paths
|
|
12
|
+
come from the asset directory instead of the command line.
|
|
13
|
+
|
|
14
|
+
Two endpoints, chosen by whether references were given. `openai/gpt-image-2` is
|
|
15
|
+
text-to-image and takes no image at all; `openai/gpt-image-2/edit` takes a list, and
|
|
16
|
+
**the order matters** — the prompt has to say what each reference is for.
|
|
17
|
+
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import argparse
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
from .. import fal, imaging, prompts, workspace
|
|
26
|
+
|
|
27
|
+
MODEL = "openai/gpt-image-2"
|
|
28
|
+
MODEL_EDIT = "openai/gpt-image-2/edit"
|
|
29
|
+
|
|
30
|
+
PRESETS = (
|
|
31
|
+
"square_hd",
|
|
32
|
+
"square",
|
|
33
|
+
"portrait_4_3",
|
|
34
|
+
"portrait_16_9",
|
|
35
|
+
"landscape_4_3",
|
|
36
|
+
"landscape_16_9",
|
|
37
|
+
"auto",
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
#: The endpoint refuses an area below this and rounds each side to a multiple of 16.
|
|
41
|
+
MIN_PIXELS = 655_360
|
|
42
|
+
EDGE_MULTIPLE = 16
|
|
43
|
+
MAX_EDGE = 3840
|
|
44
|
+
|
|
45
|
+
#: Chroma green. Far from skin, steel, fire and stone, which is what this game draws.
|
|
46
|
+
DEFAULT_CHROMA = "00b140"
|
|
47
|
+
|
|
48
|
+
CHROMA_INSTRUCTION = (
|
|
49
|
+
"The background is a single solid flat #{hex} chroma-green field, edge to edge, "
|
|
50
|
+
"with nothing else in it: no scenery, no ground plane, no shadow cast onto it, "
|
|
51
|
+
"no gradient, no vignette, no border, no text and no watermark. The subject does "
|
|
52
|
+
"not touch or overlap the edges of the image, and nothing in the subject itself "
|
|
53
|
+
"uses that chroma green."
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def parse_size(text: str) -> str | dict:
|
|
58
|
+
"""A fal preset, or WIDTHxHEIGHT in pixels."""
|
|
59
|
+
if text in PRESETS:
|
|
60
|
+
return text
|
|
61
|
+
if "x" not in text.lower():
|
|
62
|
+
raise ValueError(f"a size is WIDTHxHEIGHT or one of: {', '.join(PRESETS)}")
|
|
63
|
+
width_text, _, height_text = text.lower().partition("x")
|
|
64
|
+
try:
|
|
65
|
+
width, height = int(width_text), int(height_text)
|
|
66
|
+
except ValueError:
|
|
67
|
+
raise ValueError(f"size {text!r} is not WIDTHxHEIGHT") from None
|
|
68
|
+
if width <= 0 or height <= 0:
|
|
69
|
+
raise ValueError("a size needs positive sides")
|
|
70
|
+
if max(width, height) > MAX_EDGE:
|
|
71
|
+
raise ValueError(f"size {text!r} has a side longer than {MAX_EDGE}")
|
|
72
|
+
return {"width": width, "height": height}
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def check_size(size: str | dict) -> str | None:
|
|
76
|
+
"""The warning this size deserves, or None.
|
|
77
|
+
|
|
78
|
+
A warning rather than a refusal: the endpoint rounds a side up and takes it, so the
|
|
79
|
+
run still works. What it must not do is round silently — the anchor decides
|
|
80
|
+
everything after it, and a size that came back different from the one asked for is
|
|
81
|
+
exactly the kind of surprise worth seeing before the next call is built on it.
|
|
82
|
+
"""
|
|
83
|
+
if not isinstance(size, dict):
|
|
84
|
+
return None
|
|
85
|
+
width, height = size["width"], size["height"]
|
|
86
|
+
problems = []
|
|
87
|
+
if width * height < MIN_PIXELS:
|
|
88
|
+
problems.append(
|
|
89
|
+
f"{width}x{height} = {width * height:,} px, under the {MIN_PIXELS:,} minimum "
|
|
90
|
+
f"the endpoint accepts"
|
|
91
|
+
)
|
|
92
|
+
off = [str(value) for value in (width, height) if value % EDGE_MULTIPLE]
|
|
93
|
+
if off:
|
|
94
|
+
problems.append(
|
|
95
|
+
f"side {' and '.join(off)} is not a multiple of {EDGE_MULTIPLE}; fal rounds it"
|
|
96
|
+
)
|
|
97
|
+
return " ; ".join(problems) if problems else None
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def build_prompt(args: argparse.Namespace, prompt: str) -> str:
|
|
101
|
+
"""The prompt as sent.
|
|
102
|
+
|
|
103
|
+
With `--transparent`, the chroma instruction is appended. Without it the flag would
|
|
104
|
+
cut a background nobody asked the model to paint: the cut only works because the
|
|
105
|
+
prompt demanded a flat key field in the first place.
|
|
106
|
+
"""
|
|
107
|
+
if not args.transparent:
|
|
108
|
+
return prompt
|
|
109
|
+
red, green, blue = imaging.parse_chroma(args.chroma)
|
|
110
|
+
key = f"{red:02x}{green:02x}{blue:02x}"
|
|
111
|
+
return f"{prompt.rstrip()}\n\n{CHROMA_INSTRUCTION.format(hex=key)}"
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def read_prompt(args: argparse.Namespace) -> str:
|
|
115
|
+
"""The prompt, from `--prompt` or `--prompt-file`, and never from both."""
|
|
116
|
+
return prompts.resolve(args.prompt, args.prompt_file, "the anchor")
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def build_payload(args: argparse.Namespace, prompt: str, refs: list[str]) -> dict:
|
|
120
|
+
"""The payload as the endpoint takes it.
|
|
121
|
+
|
|
122
|
+
Built before anything is uploaded or spent, so `--dry-run` shows the real thing.
|
|
123
|
+
"""
|
|
124
|
+
payload: dict = {
|
|
125
|
+
"prompt": build_prompt(args, prompt),
|
|
126
|
+
"image_size": parse_size(args.size),
|
|
127
|
+
"quality": args.quality,
|
|
128
|
+
"num_images": args.count,
|
|
129
|
+
"output_format": "png",
|
|
130
|
+
}
|
|
131
|
+
if refs:
|
|
132
|
+
payload["image_urls"] = refs
|
|
133
|
+
return payload
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def post_process(path: Path, args: argparse.Namespace) -> dict:
|
|
137
|
+
"""The unpaid transforms, in the order that makes each one cheaper than the next.
|
|
138
|
+
|
|
139
|
+
Chroma first, because the grid detector should see alpha where the background was
|
|
140
|
+
rather than a flat colour competing for a palette slot.
|
|
141
|
+
"""
|
|
142
|
+
report: dict = {}
|
|
143
|
+
if args.transparent:
|
|
144
|
+
report["chroma"] = imaging.cut_chroma(
|
|
145
|
+
path,
|
|
146
|
+
imaging.parse_chroma(args.chroma),
|
|
147
|
+
args.tol,
|
|
148
|
+
args.feather,
|
|
149
|
+
args.despill,
|
|
150
|
+
args.despill_radius,
|
|
151
|
+
args.chroma_mode,
|
|
152
|
+
)
|
|
153
|
+
if args.pixelart:
|
|
154
|
+
report["pixelart"] = imaging.to_pixelart(path, args.colors, args.scale)
|
|
155
|
+
return report
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def run(args: argparse.Namespace) -> int:
|
|
159
|
+
context = workspace.context_from_args(args)
|
|
160
|
+
prompt = read_prompt(args)
|
|
161
|
+
|
|
162
|
+
refs = [Path(one) for one in (args.ref or [])]
|
|
163
|
+
for one in refs:
|
|
164
|
+
if not one.is_file():
|
|
165
|
+
raise FileNotFoundError(f"no reference image at {one}")
|
|
166
|
+
|
|
167
|
+
endpoint = MODEL_EDIT if refs else MODEL
|
|
168
|
+
|
|
169
|
+
warning = check_size(parse_size(args.size))
|
|
170
|
+
if warning:
|
|
171
|
+
print(f"warning: {warning}")
|
|
172
|
+
|
|
173
|
+
if context.dry_run:
|
|
174
|
+
# Uploading a reference costs nothing but is still a request; a dry run names
|
|
175
|
+
# the files it would send instead of sending them.
|
|
176
|
+
payload = build_payload(args, prompt, [str(one) for one in refs])
|
|
177
|
+
fal.show(endpoint, payload)
|
|
178
|
+
return 0
|
|
179
|
+
|
|
180
|
+
fal.require_key()
|
|
181
|
+
# Before the upload and before the call: the text has to survive a run that fails
|
|
182
|
+
# halfway, and a --prompt-file can be edited the moment this returns — R2.4.
|
|
183
|
+
context.keep_prompt(prompt)
|
|
184
|
+
uploaded = [fal.upload(one) for one in refs]
|
|
185
|
+
payload = build_payload(args, prompt, uploaded)
|
|
186
|
+
result = fal.call(endpoint, payload)
|
|
187
|
+
|
|
188
|
+
urls = fal.urls_in(result)
|
|
189
|
+
if not urls:
|
|
190
|
+
raise ValueError(f"{endpoint} returned no image; nothing to write")
|
|
191
|
+
|
|
192
|
+
out = context.ensure_out()
|
|
193
|
+
written = [
|
|
194
|
+
fal.download(url, out / _name(context.kind or "anchor", index, len(urls)))
|
|
195
|
+
for index, url in enumerate(urls)
|
|
196
|
+
]
|
|
197
|
+
context.paid_call(endpoint=endpoint, payload=payload, urls=urls, files=written)
|
|
198
|
+
|
|
199
|
+
reports = [post_process(path, args) for path in written]
|
|
200
|
+
context.record(endpoint=endpoint, images=[path.name for path in written], reports=reports)
|
|
201
|
+
|
|
202
|
+
for path, report in zip(written, reports, strict=True):
|
|
203
|
+
grid = report.get("pixelart")
|
|
204
|
+
found = f" grid {grid['cols']}x{grid['rows']} {grid['colours']} colours" if grid else ""
|
|
205
|
+
print(f"{path}{found}")
|
|
206
|
+
return 0
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _name(kind: str, index: int, total: int) -> str:
|
|
210
|
+
"""The file, named after what it is rather than after the stage that made it.
|
|
211
|
+
|
|
212
|
+
`--count 4` gives four candidates for one prompt, and they are numbered from one
|
|
213
|
+
rather than from zero: the number is what somebody says out loud when picking.
|
|
214
|
+
"""
|
|
215
|
+
return f"{kind}.png" if total == 1 else f"{kind}-{index + 1}.png"
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Close every matted board into a row of a sprite sheet.
|
|
2
|
+
|
|
3
|
+
The only free stage. Everything it does is local, which is why it is worth re-running:
|
|
4
|
+
a row that came out wrong costs nothing to rebuild from a board that was paid for once.
|
|
5
|
+
|
|
6
|
+
The work itself is in `sheet`, pinned byte for byte by `tests/parity`. This is the part
|
|
7
|
+
that knows about assets — which boards to read, where the rows go, and what to record.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
|
|
14
|
+
from .. import sheet, workspace
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def run(args: argparse.Namespace) -> int:
|
|
18
|
+
context = workspace.context_from_args(args)
|
|
19
|
+
sources = context.source_files("*.png")
|
|
20
|
+
# A mask sits beside its cut and is not a board; matting one would double every row.
|
|
21
|
+
sources = [path for path in sources if not path.name.endswith(".mask.png")]
|
|
22
|
+
if not sources:
|
|
23
|
+
raise workspace.StageRefused(f"{context.asset}: no board to close in {context.source}")
|
|
24
|
+
|
|
25
|
+
grid = sheet.parse_grid(args.grid)
|
|
26
|
+
out = context.ensure_out()
|
|
27
|
+
reports: dict[str, dict] = {}
|
|
28
|
+
|
|
29
|
+
for source in sources:
|
|
30
|
+
row = out / f"{source.stem}.row.png"
|
|
31
|
+
gif = None if args.no_gif else out / f"{source.stem}.row.gif"
|
|
32
|
+
# Keyed by what was written, not by what was read. The two live in different
|
|
33
|
+
# directories, and reporting the source's name under the output's directory
|
|
34
|
+
# names a path that does not exist.
|
|
35
|
+
reports[row.name] = sheet.board_to_row(
|
|
36
|
+
source,
|
|
37
|
+
row,
|
|
38
|
+
grid=grid,
|
|
39
|
+
frames=args.frames or None,
|
|
40
|
+
cell=args.cell,
|
|
41
|
+
scale=args.scale or None,
|
|
42
|
+
fit=args.fit,
|
|
43
|
+
art=args.art,
|
|
44
|
+
colors=args.colors,
|
|
45
|
+
alpha_floor=args.alpha_floor,
|
|
46
|
+
alpha_ceil=args.alpha_ceil,
|
|
47
|
+
chroma=args.chroma,
|
|
48
|
+
tol=args.tol,
|
|
49
|
+
feather=args.feather,
|
|
50
|
+
chroma_mode=args.chroma_mode,
|
|
51
|
+
gif=gif,
|
|
52
|
+
fps=args.fps,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
context.record(grid=args.grid, cell=args.cell, art=args.art, rows=reports)
|
|
56
|
+
|
|
57
|
+
for name, report in reports.items():
|
|
58
|
+
native = f" native {report['native']}" if report["native"] else ""
|
|
59
|
+
print(
|
|
60
|
+
f"{out / name}{native} union {report['union']}, scale {report['scale']}x, "
|
|
61
|
+
f"placed {report['placed']} at {report['offset']}"
|
|
62
|
+
)
|
|
63
|
+
if report["overflow"]:
|
|
64
|
+
print(
|
|
65
|
+
f"warning: {name} does not fit the cell; give a smaller --scale, "
|
|
66
|
+
f"or --fit to bring it down"
|
|
67
|
+
)
|
|
68
|
+
return 0
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""Cut the board's background by segmentation rather than by colour.
|
|
2
|
+
|
|
3
|
+
Absorbed from the `bg_remove.py` this grew out of. It sits next to `anchor`'s `--transparent`
|
|
4
|
+
because the two fail differently: a chroma key decides by distance to one colour, so
|
|
5
|
+
wherever the subject is near that colour it eats art, and wherever the key painted over
|
|
6
|
+
a thin contour it leaves halo. Hair is the worst of it — a strand is thinner than the
|
|
7
|
+
tolerance band.
|
|
8
|
+
|
|
9
|
+
Segmentation decides by asking what the subject is, so it has neither problem. Two
|
|
10
|
+
options carry the result: `Matting` returns continuous alpha instead of a hard cut, and
|
|
11
|
+
`refine_foreground` pulls the background back out of the RGB of edge pixels. The second
|
|
12
|
+
is what kills the halo, not the alpha.
|
|
13
|
+
|
|
14
|
+
**Two backends, and `local` is the default.** `--backend local` runs BiRefNet through
|
|
15
|
+
`rembg` on this machine — the same family of model, no invoice, and on a GPU no slower.
|
|
16
|
+
`--backend fal` is the paid endpoint, for a machine with neither the extra installed nor
|
|
17
|
+
a GPU worth using. Everything downstream is identical: the same files land in the same
|
|
18
|
+
directory with the same names, and `board` cannot tell which one ran.
|
|
19
|
+
|
|
20
|
+
**Several images go up as one, on the paid backend.** Sent separately they are several
|
|
21
|
+
charges and several independent decisions about where the edge is, so the contour alpha
|
|
22
|
+
varies between frames and the silhouette flickers — the same failure `sheet` avoids by
|
|
23
|
+
recovering one grid and one palette across a whole board. Joined, the endpoint answers
|
|
24
|
+
once, for all of them. What it costs is resolution: the endpoint operates at a fixed
|
|
25
|
+
size, so four 1024 images join into exactly the default 2048 and lose nothing, while
|
|
26
|
+
four 1400 boards do not fit and the stage says so before spending. `--no-join` goes back
|
|
27
|
+
to one call each.
|
|
28
|
+
|
|
29
|
+
The local backend does not join, and does not need to. Joining bought two things: one
|
|
30
|
+
charge instead of six, and one decision instead of six. Locally the first is worth
|
|
31
|
+
nothing, and the second comes free from reusing one session across every frame — while
|
|
32
|
+
joining would still cost the resolution. So each frame is cut at its own full size.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
import argparse
|
|
38
|
+
from pathlib import Path
|
|
39
|
+
|
|
40
|
+
from .. import atlas, fal, imaging, matting, workspace
|
|
41
|
+
|
|
42
|
+
MODEL = "fal-ai/birefnet/v2"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def build_payload(args: argparse.Namespace) -> dict:
|
|
46
|
+
"""The payload, without the image, which is added once the upload has a URL."""
|
|
47
|
+
return {
|
|
48
|
+
"model": args.variant,
|
|
49
|
+
"operating_resolution": args.resolution,
|
|
50
|
+
"refine_foreground": not args.no_refine,
|
|
51
|
+
"output_format": "png",
|
|
52
|
+
"output_mask": bool(args.mask),
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _cut_one(context, payload: dict, source: Path, dest: Path, want_mask: bool) -> list[Path]:
|
|
57
|
+
"""One paid call: upload, cut, download, record. Returns what it wrote."""
|
|
58
|
+
sent = {**payload, "image_url": fal.upload(source)}
|
|
59
|
+
result = fal.call(MODEL, sent)
|
|
60
|
+
|
|
61
|
+
url = (result.get("image") or {}).get("url")
|
|
62
|
+
if not url:
|
|
63
|
+
raise ValueError(f"{MODEL} returned no image for {source.name}")
|
|
64
|
+
|
|
65
|
+
written = [fal.download(url, dest)]
|
|
66
|
+
fetched = [url]
|
|
67
|
+
if want_mask:
|
|
68
|
+
mask_url = (result.get("mask_image") or {}).get("url")
|
|
69
|
+
if mask_url:
|
|
70
|
+
written.append(fal.download(mask_url, dest.with_name(f"{dest.stem}.mask.png")))
|
|
71
|
+
fetched.append(mask_url)
|
|
72
|
+
else:
|
|
73
|
+
print(f"warning: no mask came back for {source.name}")
|
|
74
|
+
|
|
75
|
+
# The uploaded URL is kept out of the record by `ledger.redact`, for every stage
|
|
76
|
+
# at once rather than by each one remembering to strip it. Only the URLs actually
|
|
77
|
+
# fetched are recorded: `urls_in` finds every `url` in the response, and a mask
|
|
78
|
+
# nobody asked for has no local copy behind it — recording it would persist a live
|
|
79
|
+
# link to something this asset does not hold.
|
|
80
|
+
context.paid_call(endpoint=MODEL, payload=sent, urls=fetched, files=written)
|
|
81
|
+
return written
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def run_local(args: argparse.Namespace, context, sources: list[Path]) -> int:
|
|
85
|
+
"""Cut every frame here, with one session, and spend nothing.
|
|
86
|
+
|
|
87
|
+
No `paid_call`, because there was none. `--dry-run` still exists and still prints
|
|
88
|
+
what would happen: the point of it on this backend is not the money but seeing what
|
|
89
|
+
would be written before it is.
|
|
90
|
+
"""
|
|
91
|
+
if context.dry_run:
|
|
92
|
+
where = "GPU" if matting.on_gpu() else "CPU"
|
|
93
|
+
model = args.model or "the configured model"
|
|
94
|
+
print(f"dry run: {model} locally on the {where}, one session for all of them")
|
|
95
|
+
for source in sources:
|
|
96
|
+
print(f" {source}")
|
|
97
|
+
return 0
|
|
98
|
+
|
|
99
|
+
matting.require_rembg()
|
|
100
|
+
if not matting.on_gpu():
|
|
101
|
+
print("warning: onnxruntime reports no GPU; this runs on the CPU and is slow")
|
|
102
|
+
|
|
103
|
+
out = context.ensure_out()
|
|
104
|
+
current = matting.session(args.model)
|
|
105
|
+
alpha = {}
|
|
106
|
+
for source in sources:
|
|
107
|
+
alpha[source.name] = matting.cut_file(
|
|
108
|
+
source, out / source.name, current=current, refine=not args.no_refine
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
context.record(
|
|
112
|
+
endpoint=None,
|
|
113
|
+
backend="local",
|
|
114
|
+
model=args.model or matting.DEFAULT_MODEL,
|
|
115
|
+
images=list(alpha),
|
|
116
|
+
alpha=alpha,
|
|
117
|
+
joined=None,
|
|
118
|
+
)
|
|
119
|
+
for name, stats in alpha.items():
|
|
120
|
+
print(
|
|
121
|
+
f"{out / name} {stats['size']} "
|
|
122
|
+
f"{stats['transparent_px']:,} transparent, {stats['partial_px']:,} at the edge"
|
|
123
|
+
)
|
|
124
|
+
return 0
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def run(args: argparse.Namespace) -> int:
|
|
128
|
+
context = workspace.context_from_args(args)
|
|
129
|
+
sources = context.source_files("*.png")
|
|
130
|
+
if not sources:
|
|
131
|
+
raise workspace.StageRefused(f"{context.asset}: nothing to matte in {context.source}")
|
|
132
|
+
|
|
133
|
+
from .. import settings
|
|
134
|
+
|
|
135
|
+
backend = args.backend or settings.load().matte_backend
|
|
136
|
+
if backend == "local":
|
|
137
|
+
return run_local(args, context, sources)
|
|
138
|
+
|
|
139
|
+
payload = build_payload(args)
|
|
140
|
+
joining = len(sources) > 1 and not args.no_join
|
|
141
|
+
|
|
142
|
+
if context.dry_run:
|
|
143
|
+
fal.show(MODEL, {**payload, "image_url": "<upload>"})
|
|
144
|
+
if joining:
|
|
145
|
+
print(f"would join {len(sources)} images and cut them in 1 paid call")
|
|
146
|
+
else:
|
|
147
|
+
count = len(sources)
|
|
148
|
+
print(f"would cut {count} image{'' if count == 1 else 's'}, one call each")
|
|
149
|
+
return 0
|
|
150
|
+
|
|
151
|
+
fal.require_key()
|
|
152
|
+
out = context.ensure_out()
|
|
153
|
+
written: list[Path] = []
|
|
154
|
+
layout = None
|
|
155
|
+
|
|
156
|
+
if joining:
|
|
157
|
+
# Under a subdirectory: the joined image and its cut are what was paid for and
|
|
158
|
+
# worth keeping, but `board` globs *.png here and must not mistake them for
|
|
159
|
+
# boards of their own.
|
|
160
|
+
scratch = out / "_join"
|
|
161
|
+
layout = atlas.pack(sources, scratch / "joined.png")
|
|
162
|
+
if not atlas.fits(layout, args.resolution):
|
|
163
|
+
# Said before spending, not after: the loss is in the upload, and by the
|
|
164
|
+
# time the cut comes back there is nothing to decide.
|
|
165
|
+
print(
|
|
166
|
+
f"warning: the joined image is {layout['size']} against an operating "
|
|
167
|
+
f"resolution of {args.resolution}; detail is lost on the way in. "
|
|
168
|
+
f"--no-join sends them one at a time at full size."
|
|
169
|
+
)
|
|
170
|
+
cut_atlas = scratch / "joined.cut.png"
|
|
171
|
+
_cut_one(context, payload, scratch / "joined.png", cut_atlas, args.mask)
|
|
172
|
+
written = atlas.unpack(cut_atlas, layout, out)
|
|
173
|
+
if args.mask:
|
|
174
|
+
mask = cut_atlas.with_name(f"{cut_atlas.stem}.mask.png")
|
|
175
|
+
if mask.is_file():
|
|
176
|
+
written += atlas.unpack(mask, _mask_layout(layout), out)
|
|
177
|
+
else:
|
|
178
|
+
for source in sources:
|
|
179
|
+
written += _cut_one(context, payload, source, out / source.name, args.mask)
|
|
180
|
+
|
|
181
|
+
alpha = {
|
|
182
|
+
path.name: imaging.measure_alpha(path)
|
|
183
|
+
for path in written
|
|
184
|
+
if not path.name.endswith(".mask.png")
|
|
185
|
+
}
|
|
186
|
+
context.record(
|
|
187
|
+
endpoint=MODEL,
|
|
188
|
+
images=[path.name for path in written],
|
|
189
|
+
alpha=alpha,
|
|
190
|
+
joined=layout,
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
for name, stats in alpha.items():
|
|
194
|
+
print(
|
|
195
|
+
f"{out / name} {stats['size']} "
|
|
196
|
+
f"{stats['transparent_px']:,} transparent, {stats['partial_px']:,} at the edge"
|
|
197
|
+
)
|
|
198
|
+
return 0
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _mask_layout(layout: dict) -> dict:
|
|
202
|
+
"""The same boxes, named for the masks they cut out of the joined mask."""
|
|
203
|
+
return {
|
|
204
|
+
**layout,
|
|
205
|
+
"placed": [
|
|
206
|
+
{**entry, "name": f"{Path(entry['name']).stem}.mask.png"}
|
|
207
|
+
for entry in layout["placed"]
|
|
208
|
+
],
|
|
209
|
+
}
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Transfer movement from a driving clip onto the anchor, or onto a pose.
|
|
2
|
+
|
|
3
|
+
The alternative to `video`, and the difference is where the movement comes from. `video`
|
|
4
|
+
asks a model to invent a walk and hopes the result is a cycle. This hands the model a
|
|
5
|
+
clip built from a reference sheet the game already ships, and the model's job is only to
|
|
6
|
+
put the new character through that movement — identity from the anchor, motion from the
|
|
7
|
+
clip.
|
|
8
|
+
|
|
9
|
+
That is the answer to what `plans/character-forge.md` measured: generating frame by
|
|
10
|
+
frame moved the head-and-torso region 0.25 between steps against the reference's 0.11,
|
|
11
|
+
because the model redrew the character in every cell instead of moving it. A pose-driven
|
|
12
|
+
endpoint cannot redraw it, because the anchor is what pins the appearance.
|
|
13
|
+
|
|
14
|
+
The reference sheet is named on the command line, and that is deliberate rather than an
|
|
15
|
+
exception to R3.2: it is a shared game asset, like a prompt file, not something the asset
|
|
16
|
+
directory holds. Everything written still goes to the asset.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from .. import clip, drive, endpoints, fal, workspace
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def parse_set(pairs: list[str] | None, endpoint: endpoints.Endpoint) -> dict:
|
|
28
|
+
"""`NAME=VALUE` pairs, checked against the endpoint before anything is spent — R4.4.
|
|
29
|
+
|
|
30
|
+
Values are read as JSON when they parse as JSON, so `true`, `7` and `"regular"` all
|
|
31
|
+
arrive as the type the endpoint's schema expects rather than as strings.
|
|
32
|
+
"""
|
|
33
|
+
import json
|
|
34
|
+
|
|
35
|
+
settings: dict = {}
|
|
36
|
+
for pair in pairs or []:
|
|
37
|
+
name, sep, raw = pair.partition("=")
|
|
38
|
+
if not sep:
|
|
39
|
+
raise ValueError(f"an endpoint option is NAME=VALUE; got {pair!r}")
|
|
40
|
+
try:
|
|
41
|
+
value = json.loads(raw)
|
|
42
|
+
except json.JSONDecodeError:
|
|
43
|
+
value = raw
|
|
44
|
+
endpoint.check(name, value)
|
|
45
|
+
settings[name] = value
|
|
46
|
+
return settings
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def build_payload(
|
|
50
|
+
endpoint: endpoints.Endpoint,
|
|
51
|
+
image_url: str,
|
|
52
|
+
video_url: str,
|
|
53
|
+
prompt: str,
|
|
54
|
+
settings: dict,
|
|
55
|
+
) -> dict:
|
|
56
|
+
"""The payload as this endpoint takes it."""
|
|
57
|
+
return {
|
|
58
|
+
"image_url": image_url,
|
|
59
|
+
"video_url": video_url,
|
|
60
|
+
# Defaults first, so a prompt given here wins over an endpoint that carries an
|
|
61
|
+
# empty one only to satisfy its schema. `run` has already refused a prompt for
|
|
62
|
+
# an endpoint that does not declare one.
|
|
63
|
+
**endpoint.defaults,
|
|
64
|
+
**({"prompt": prompt} if prompt else {}),
|
|
65
|
+
**settings,
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def driving_clip(args: argparse.Namespace, out: Path) -> tuple[Path, dict]:
|
|
70
|
+
"""The clip that carries the movement, built or handed over.
|
|
71
|
+
|
|
72
|
+
`--clip` exists for a clip that already is what it should be — a previous run's, or
|
|
73
|
+
footage. Everything else comes from a sheet, because that is the movement this
|
|
74
|
+
project has already measured.
|
|
75
|
+
"""
|
|
76
|
+
if args.clip:
|
|
77
|
+
given = Path(args.clip)
|
|
78
|
+
if not given.is_file():
|
|
79
|
+
raise FileNotFoundError(f"no driving clip at {given}")
|
|
80
|
+
return given, {"source": str(given)}
|
|
81
|
+
|
|
82
|
+
if not args.sheet:
|
|
83
|
+
raise ValueError("the motion stage needs --sheet, or --clip")
|
|
84
|
+
sheet = Path(args.sheet)
|
|
85
|
+
if not sheet.is_file():
|
|
86
|
+
raise FileNotFoundError(f"no reference sheet at {sheet}")
|
|
87
|
+
|
|
88
|
+
built = out / "driving.mp4"
|
|
89
|
+
report = drive.build(
|
|
90
|
+
sheet,
|
|
91
|
+
built,
|
|
92
|
+
row=args.row,
|
|
93
|
+
frames=args.sheet_frames or None,
|
|
94
|
+
target=args.target,
|
|
95
|
+
fps=args.drive_fps,
|
|
96
|
+
seconds=args.seconds,
|
|
97
|
+
)
|
|
98
|
+
return built, {"source": str(sheet), **report}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def run(args: argparse.Namespace) -> int:
|
|
102
|
+
context = workspace.context_from_args(args)
|
|
103
|
+
endpoint = endpoints.get(args.endpoint)
|
|
104
|
+
settings = parse_set(args.set, endpoint)
|
|
105
|
+
if args.prompt:
|
|
106
|
+
# `--prompt` is one of this stage's own flags, but it is still an endpoint
|
|
107
|
+
# option: only one of the three declares it, and the other two would be paid to
|
|
108
|
+
# ignore it.
|
|
109
|
+
endpoint.check("prompt", args.prompt)
|
|
110
|
+
pick = clip.parse_pick(args.pick)
|
|
111
|
+
|
|
112
|
+
# A pose-driven endpoint takes one image and deforms it, so there is no last frame
|
|
113
|
+
# here — only which image the movement is applied to.
|
|
114
|
+
from .video import first_and_last
|
|
115
|
+
|
|
116
|
+
anchor, _ = first_and_last(context, args)
|
|
117
|
+
|
|
118
|
+
if context.dry_run:
|
|
119
|
+
# The driving clip is built even for a dry run: it costs nothing, it is the part
|
|
120
|
+
# most likely to be wrong, and a payload naming a clip nobody looked at is not
|
|
121
|
+
# what a dry run is for. It goes beside the stage's output rather than into it,
|
|
122
|
+
# because a dry run that filled `board/` would make the real run look already
|
|
123
|
+
# done and be refused.
|
|
124
|
+
built, report = driving_clip(args, context.directory / "_dry")
|
|
125
|
+
fal.show(
|
|
126
|
+
endpoint.id,
|
|
127
|
+
build_payload(endpoint, str(anchor), str(built), args.prompt, settings),
|
|
128
|
+
)
|
|
129
|
+
print(f"driving clip: {built} {report.get('frames', '?')} frames")
|
|
130
|
+
return 0
|
|
131
|
+
|
|
132
|
+
fal.require_key()
|
|
133
|
+
out = context.ensure_out()
|
|
134
|
+
built, report = driving_clip(args, out)
|
|
135
|
+
|
|
136
|
+
payload = build_payload(endpoint, fal.upload(anchor), fal.upload(built), args.prompt, settings)
|
|
137
|
+
result = fal.call(endpoint.id, payload)
|
|
138
|
+
|
|
139
|
+
found = fal.urls_in(result)
|
|
140
|
+
if not found:
|
|
141
|
+
raise ValueError(f"{endpoint.id} returned no clip")
|
|
142
|
+
|
|
143
|
+
animated = fal.download(found[0], out / "clip.mp4")
|
|
144
|
+
frames = clip.decode_frames(animated)
|
|
145
|
+
chosen, indices = clip.choose_frames(frames, args.frames, args.start, args.end, pick)
|
|
146
|
+
board = out / "board.png"
|
|
147
|
+
cols, rows = clip.pack_board(chosen, args.cols, board)
|
|
148
|
+
|
|
149
|
+
context.paid_call(endpoint=endpoint.id, payload=payload, urls=found, files=[animated, board])
|
|
150
|
+
context.record(
|
|
151
|
+
endpoint=endpoint.id,
|
|
152
|
+
driving=report,
|
|
153
|
+
clip="clip.mp4",
|
|
154
|
+
board="board.png",
|
|
155
|
+
decoded=len(frames),
|
|
156
|
+
indices=indices,
|
|
157
|
+
grid=f"{cols}x{rows}",
|
|
158
|
+
settings=settings,
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
print(f"{built} driving, {report.get('frames', '?')} frames")
|
|
162
|
+
print(f"{animated} {len(frames)} frames decoded")
|
|
163
|
+
print(f"{board} {len(chosen)} frames at {indices}, packed {cols}x{rows}")
|
|
164
|
+
return 0
|