agency-lang 0.26.0 → 0.26.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/lib/cli/diffusersImageRules.py +189 -75
- package/dist/lib/cli/diffusersImageServer.py +40 -15
- package/dist/lib/cli/localFlag.js +4 -1
- package/dist/lib/stdlib/image.js +22 -10
- package/dist/lib/stdlib/localImageInputs.d.ts +21 -13
- package/dist/lib/stdlib/localImageInputs.js +58 -21
- package/dist/lib/stdlib/localModels.d.ts +3 -1
- package/dist/lib/stdlib/localModels.js +10 -3
- package/dist/lib/stdlib/mlxImage.js +1 -0
- package/dist/lib/stdlib/modelCatalog.js +6 -4
- package/package.json +1 -1
- package/stdlib/docs/stdlib/image.md +31 -5
- package/stdlib/image.agency +26 -1
- package/stdlib/image.js +24 -7
|
@@ -64,6 +64,7 @@ FIELDS = [
|
|
|
64
64
|
"control_invert",
|
|
65
65
|
"images",
|
|
66
66
|
"start_image",
|
|
67
|
+
"mask_image",
|
|
67
68
|
"strength",
|
|
68
69
|
]
|
|
69
70
|
|
|
@@ -105,8 +106,10 @@ MAX_REFERENCE_ASPECT = 8
|
|
|
105
106
|
|
|
106
107
|
# The images a request can carry, one row per request field. The field a
|
|
107
108
|
# request carries decides its mode; a request with no image is in "plain"
|
|
108
|
-
# mode. `
|
|
109
|
-
#
|
|
109
|
+
# mode. A field with `goes_with` comes only with that other field, and the
|
|
110
|
+
# pair decides the mode: a start image with a mask is "inpaint".
|
|
111
|
+
# `LOCAL_IMAGE_FIELDS` in lib/stdlib/localImageInputs.ts is the same table
|
|
112
|
+
# for the stdlib, and a test compares the two.
|
|
110
113
|
#
|
|
111
114
|
# mode the mode the field puts a request in. A family takes a mode
|
|
112
115
|
# when its row in FAMILIES has a pipeline for it.
|
|
@@ -121,8 +124,9 @@ MAX_REFERENCE_ASPECT = 8
|
|
|
121
124
|
# many and leaves a smaller one as it is, and "cover" scales
|
|
122
125
|
# it to cover the output and crops the overflow evenly from
|
|
123
126
|
# both sides
|
|
124
|
-
#
|
|
125
|
-
# made RGB.
|
|
127
|
+
# background the color a transparent image is pasted onto before it is
|
|
128
|
+
# made RGB: "white" or "black". None: its color channels are
|
|
129
|
+
# kept as stored, and the alpha band is dropped.
|
|
126
130
|
# refusal the message for a family with no pipeline for the mode, with
|
|
127
131
|
# {label} for the family and {families} for those that have one
|
|
128
132
|
# prepare optional: the name of a step the server runs on a decoded
|
|
@@ -130,6 +134,12 @@ MAX_REFERENCE_ASPECT = 8
|
|
|
130
134
|
# when the request asks.
|
|
131
135
|
# check optional: the name of a check in CHECKS the server runs on a
|
|
132
136
|
# decoded image's size. It returns a refusal or None.
|
|
137
|
+
# arg the pipeline argument the field's images are passed as
|
|
138
|
+
# goes_with optional: the field this one comes with. A request with this
|
|
139
|
+
# field and not the other is refused.
|
|
140
|
+
# same_size_as
|
|
141
|
+
# optional: the field whose image this one's must match in
|
|
142
|
+
# width and height, so both are cropped the same way
|
|
133
143
|
INPUT_IMAGES = {
|
|
134
144
|
"control_image": {
|
|
135
145
|
"mode": "control",
|
|
@@ -138,9 +148,10 @@ INPUT_IMAGES = {
|
|
|
138
148
|
# A control image is a drawing to follow, not a picture to keep.
|
|
139
149
|
"sets_size": False,
|
|
140
150
|
"fit": "letterbox",
|
|
141
|
-
"
|
|
151
|
+
"background": None,
|
|
142
152
|
"refusal": "{label} does not take a ControlNet. Leave controlnet empty.",
|
|
143
153
|
"prepare": "invert",
|
|
154
|
+
"arg": "image",
|
|
144
155
|
},
|
|
145
156
|
"images": {
|
|
146
157
|
"mode": "reference",
|
|
@@ -149,9 +160,10 @@ INPUT_IMAGES = {
|
|
|
149
160
|
"sets_size": True,
|
|
150
161
|
# The model only looks at a reference, so it can stay any shape.
|
|
151
162
|
"fit": "shrink",
|
|
152
|
-
"
|
|
163
|
+
"background": "white",
|
|
153
164
|
"refusal": "{label} does not take reference images. Only {families} takes them.",
|
|
154
165
|
"check": "reference_problem",
|
|
166
|
+
"arg": "image",
|
|
155
167
|
},
|
|
156
168
|
"start_image": {
|
|
157
169
|
"mode": "img2img",
|
|
@@ -161,9 +173,28 @@ INPUT_IMAGES = {
|
|
|
161
173
|
# Cropped, not letterboxed: black bands would be part of the
|
|
162
174
|
# picture, and the model would redraw them as black bars.
|
|
163
175
|
"fit": "cover",
|
|
164
|
-
"
|
|
176
|
+
"background": "white",
|
|
165
177
|
"refusal": "{label} does not redraw a start image. {families} do.",
|
|
166
178
|
"check": "start_image_problem",
|
|
179
|
+
"arg": "image",
|
|
180
|
+
},
|
|
181
|
+
# White marks the part of the start image to redraw, and black the part
|
|
182
|
+
# to keep. Grey redraws partly.
|
|
183
|
+
"mask_image": {
|
|
184
|
+
"mode": "inpaint",
|
|
185
|
+
"max_count": 1,
|
|
186
|
+
"max_bytes": MAX_INPUT_IMAGE_BYTES,
|
|
187
|
+
# The start image sets the size.
|
|
188
|
+
"sets_size": False,
|
|
189
|
+
# Cropped exactly as the start image is, since the two are the same
|
|
190
|
+
# size.
|
|
191
|
+
"fit": "cover",
|
|
192
|
+
# A transparent part of a mask is black, and so kept.
|
|
193
|
+
"background": "black",
|
|
194
|
+
"refusal": "{label} does not redraw part of a picture. {families} do.",
|
|
195
|
+
"arg": "mask_image",
|
|
196
|
+
"goes_with": "start_image",
|
|
197
|
+
"same_size_as": "start_image",
|
|
167
198
|
},
|
|
168
199
|
}
|
|
169
200
|
|
|
@@ -173,18 +204,35 @@ MODE_FIELDS = {
|
|
|
173
204
|
"control": ["controlnet", "control_image", "control_scale", "control_invert"],
|
|
174
205
|
"reference": ["images"],
|
|
175
206
|
"img2img": ["start_image", "strength"],
|
|
207
|
+
"inpaint": ["start_image", "mask_image", "strength"],
|
|
176
208
|
}
|
|
177
209
|
|
|
210
|
+
# The modes that start from a start image and take a strength.
|
|
211
|
+
REDRAW_MODES = ["img2img", "inpaint"]
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def image_fields_of(mode):
|
|
215
|
+
"""The image fields a request in `mode` carries, the one that sets the
|
|
216
|
+
size first: [] for plain. A field that goes with another brings it
|
|
217
|
+
along, so inpaint is [start_image, mask_image]."""
|
|
218
|
+
fields = [field for field, row in INPUT_IMAGES.items() if row["mode"] == mode]
|
|
219
|
+
partners = [INPUT_IMAGES[field]["goes_with"] for field in fields if "goes_with" in INPUT_IMAGES[field]]
|
|
220
|
+
return partners + fields
|
|
221
|
+
|
|
178
222
|
# Room in a request body for everything but its images: the prompt and the
|
|
179
223
|
# settings.
|
|
180
224
|
REQUEST_SETTINGS_BYTES = 64 * 1024
|
|
181
225
|
|
|
182
226
|
# The largest request body: the settings, plus the base64 of the most image
|
|
183
|
-
# bytes one
|
|
184
|
-
#
|
|
185
|
-
#
|
|
227
|
+
# bytes one mode's fields may carry. `localBodyBytes()` in
|
|
228
|
+
# lib/stdlib/localImageInputs.ts is the same number, and the front door
|
|
229
|
+
# holds image requests to it.
|
|
186
230
|
MAX_BODY_BYTES = REQUEST_SETTINGS_BYTES + max(
|
|
187
|
-
|
|
231
|
+
sum(
|
|
232
|
+
base64_length(INPUT_IMAGES[field]["max_count"] * INPUT_IMAGES[field]["max_bytes"])
|
|
233
|
+
for field in image_fields_of(mode)
|
|
234
|
+
)
|
|
235
|
+
for mode in MODE_FIELDS
|
|
188
236
|
)
|
|
189
237
|
|
|
190
238
|
# The pixel budget of a size taken from a picture: the default size's.
|
|
@@ -231,18 +279,29 @@ MAX_LOADED_ADAPTERS = 2
|
|
|
231
279
|
# False: it has no such arguments and draws at the start
|
|
232
280
|
# image's size. The server fits the start image to the
|
|
233
281
|
# output size first, so the result is the same.
|
|
234
|
-
# img2img_steps how the img2img
|
|
235
|
-
# key of STEP_FORMULAS
|
|
282
|
+
# img2img_steps how the img2img and inpaint pipelines round the steps
|
|
283
|
+
# they run: a key of STEP_FORMULAS. A family's two
|
|
284
|
+
# pipelines round the same way.
|
|
285
|
+
# inpaint_default_strength
|
|
286
|
+
# how much of the masked part of a start image to redraw
|
|
287
|
+
# when the request gives no strength: the inpaint
|
|
288
|
+
# pipeline's own default. None for a family with no
|
|
289
|
+
# inpaint pipeline.
|
|
236
290
|
# components every component model_index.json must name, as
|
|
237
291
|
# [library, class]. [None, None] is a slot the file
|
|
238
292
|
# lists and leaves empty.
|
|
293
|
+
|
|
239
294
|
# settings the other values model_index.json may carry, which
|
|
240
295
|
# from_pretrained passes to the pipeline, each with the
|
|
241
296
|
# one value allowed
|
|
242
297
|
FAMILIES = {
|
|
243
298
|
"ZImagePipeline": {
|
|
244
299
|
"label": "Z-Image Turbo",
|
|
245
|
-
"pipelines": {
|
|
300
|
+
"pipelines": {
|
|
301
|
+
"plain": "ZImagePipeline",
|
|
302
|
+
"img2img": "ZImageImg2ImgPipeline",
|
|
303
|
+
"inpaint": "ZImageInpaintPipeline",
|
|
304
|
+
},
|
|
246
305
|
"default_steps": 9,
|
|
247
306
|
"max_steps": 50,
|
|
248
307
|
"default_guidance": 0.0,
|
|
@@ -253,18 +312,23 @@ FAMILIES = {
|
|
|
253
312
|
"default_strength": 0.6,
|
|
254
313
|
"img2img_takes_size": True,
|
|
255
314
|
"img2img_steps": "up",
|
|
315
|
+
"inpaint_default_strength": 1.0,
|
|
256
316
|
"components": {
|
|
257
|
-
"scheduler": ["diffusers", "FlowMatchEulerDiscreteScheduler"],
|
|
258
|
-
"text_encoder": ["transformers", "Qwen3Model"],
|
|
259
|
-
"tokenizer": ["transformers", "Qwen2Tokenizer"],
|
|
260
|
-
"transformer": ["diffusers", "ZImageTransformer2DModel"],
|
|
261
|
-
"vae": ["diffusers", "AutoencoderKL"],
|
|
317
|
+
"scheduler": [["diffusers", "FlowMatchEulerDiscreteScheduler"]],
|
|
318
|
+
"text_encoder": [["transformers", "Qwen3Model"]],
|
|
319
|
+
"tokenizer": [["transformers", "Qwen2Tokenizer"]],
|
|
320
|
+
"transformer": [["diffusers", "ZImageTransformer2DModel"]],
|
|
321
|
+
"vae": [["diffusers", "AutoencoderKL"]],
|
|
262
322
|
},
|
|
263
323
|
"settings": {},
|
|
264
324
|
},
|
|
265
325
|
"ChromaPipeline": {
|
|
266
326
|
"label": "Chroma",
|
|
267
|
-
"pipelines": {
|
|
327
|
+
"pipelines": {
|
|
328
|
+
"plain": "ChromaPipeline",
|
|
329
|
+
"img2img": "ChromaImg2ImgPipeline",
|
|
330
|
+
"inpaint": "ChromaInpaintPipeline",
|
|
331
|
+
},
|
|
268
332
|
"default_steps": 40,
|
|
269
333
|
"max_steps": 80,
|
|
270
334
|
"default_guidance": 3.0,
|
|
@@ -275,20 +339,25 @@ FAMILIES = {
|
|
|
275
339
|
"default_strength": 0.9,
|
|
276
340
|
"img2img_takes_size": True,
|
|
277
341
|
"img2img_steps": "up",
|
|
342
|
+
"inpaint_default_strength": 0.6,
|
|
278
343
|
"components": {
|
|
279
|
-
"feature_extractor": [None, None],
|
|
280
|
-
"image_encoder": [None, None],
|
|
281
|
-
"scheduler": ["diffusers", "FlowMatchEulerDiscreteScheduler"],
|
|
282
|
-
"text_encoder": ["transformers", "T5EncoderModel"],
|
|
283
|
-
"tokenizer": ["transformers", "T5Tokenizer"],
|
|
284
|
-
"transformer": ["diffusers", "ChromaTransformer2DModel"],
|
|
285
|
-
"vae": ["diffusers", "AutoencoderKL"],
|
|
344
|
+
"feature_extractor": [[None, None]],
|
|
345
|
+
"image_encoder": [[None, None]],
|
|
346
|
+
"scheduler": [["diffusers", "FlowMatchEulerDiscreteScheduler"]],
|
|
347
|
+
"text_encoder": [["transformers", "T5EncoderModel"]],
|
|
348
|
+
"tokenizer": [["transformers", "T5Tokenizer"]],
|
|
349
|
+
"transformer": [["diffusers", "ChromaTransformer2DModel"]],
|
|
350
|
+
"vae": [["diffusers", "AutoencoderKL"]],
|
|
286
351
|
},
|
|
287
352
|
"settings": {},
|
|
288
353
|
},
|
|
289
354
|
"QwenImagePipeline": {
|
|
290
355
|
"label": "Qwen-Image",
|
|
291
|
-
"pipelines": {
|
|
356
|
+
"pipelines": {
|
|
357
|
+
"plain": "QwenImagePipeline",
|
|
358
|
+
"img2img": "QwenImageImg2ImgPipeline",
|
|
359
|
+
"inpaint": "QwenImageInpaintPipeline",
|
|
360
|
+
},
|
|
292
361
|
"default_steps": 50,
|
|
293
362
|
"max_steps": 80,
|
|
294
363
|
"default_guidance": 4.0,
|
|
@@ -301,12 +370,13 @@ FAMILIES = {
|
|
|
301
370
|
"default_strength": 0.6,
|
|
302
371
|
"img2img_takes_size": True,
|
|
303
372
|
"img2img_steps": "up",
|
|
373
|
+
"inpaint_default_strength": 0.6,
|
|
304
374
|
"components": {
|
|
305
|
-
"scheduler": ["diffusers", "FlowMatchEulerDiscreteScheduler"],
|
|
306
|
-
"text_encoder": ["transformers", "Qwen2_5_VLForConditionalGeneration"],
|
|
307
|
-
"tokenizer": ["transformers", "Qwen2Tokenizer"],
|
|
308
|
-
"transformer": ["diffusers", "QwenImageTransformer2DModel"],
|
|
309
|
-
"vae": ["diffusers", "AutoencoderKLQwenImage"],
|
|
375
|
+
"scheduler": [["diffusers", "FlowMatchEulerDiscreteScheduler"]],
|
|
376
|
+
"text_encoder": [["transformers", "Qwen2_5_VLForConditionalGeneration"]],
|
|
377
|
+
"tokenizer": [["transformers", "Qwen2Tokenizer"]],
|
|
378
|
+
"transformer": [["diffusers", "QwenImageTransformer2DModel"]],
|
|
379
|
+
"vae": [["diffusers", "AutoencoderKLQwenImage"]],
|
|
310
380
|
},
|
|
311
381
|
"settings": {},
|
|
312
382
|
},
|
|
@@ -323,16 +393,18 @@ FAMILIES = {
|
|
|
323
393
|
"guidance_arg": "guidance_scale",
|
|
324
394
|
"default_negative_prompt": "",
|
|
325
395
|
"takes_lora": False,
|
|
326
|
-
# klein edits from references instead
|
|
396
|
+
# klein edits from references instead. diffusers has an inpaint
|
|
397
|
+
# pipeline for it, but no img2img one, and neither is served yet.
|
|
327
398
|
"default_strength": None,
|
|
328
399
|
"img2img_takes_size": None,
|
|
329
400
|
"img2img_steps": None,
|
|
401
|
+
"inpaint_default_strength": None,
|
|
330
402
|
"components": {
|
|
331
|
-
"scheduler": ["diffusers", "FlowMatchEulerDiscreteScheduler"],
|
|
332
|
-
"text_encoder": ["transformers", "Qwen3ForCausalLM"],
|
|
333
|
-
"tokenizer": ["transformers", "Qwen2TokenizerFast"],
|
|
334
|
-
"transformer": ["diffusers", "Flux2Transformer2DModel"],
|
|
335
|
-
"vae": ["diffusers", "AutoencoderKLFlux2"],
|
|
403
|
+
"scheduler": [["diffusers", "FlowMatchEulerDiscreteScheduler"]],
|
|
404
|
+
"text_encoder": [["transformers", "Qwen3ForCausalLM"]],
|
|
405
|
+
"tokenizer": [["transformers", "Qwen2TokenizerFast"]],
|
|
406
|
+
"transformer": [["diffusers", "Flux2Transformer2DModel"]],
|
|
407
|
+
"vae": [["diffusers", "AutoencoderKLFlux2"]],
|
|
336
408
|
},
|
|
337
409
|
# Only the step-distilled checkpoint is served. The base model
|
|
338
410
|
# needs guidance and about 50 steps, which this row does not allow.
|
|
@@ -348,6 +420,7 @@ FAMILIES = {
|
|
|
348
420
|
"plain": "StableDiffusionXLPipeline",
|
|
349
421
|
"control": "StableDiffusionXLControlNetPipeline",
|
|
350
422
|
"img2img": "StableDiffusionXLImg2ImgPipeline",
|
|
423
|
+
"inpaint": "StableDiffusionXLInpaintPipeline",
|
|
351
424
|
},
|
|
352
425
|
"default_steps": 28,
|
|
353
426
|
"max_steps": 80,
|
|
@@ -361,16 +434,24 @@ FAMILIES = {
|
|
|
361
434
|
"default_strength": 0.6,
|
|
362
435
|
"img2img_takes_size": False,
|
|
363
436
|
"img2img_steps": "down",
|
|
437
|
+
# Just under 1, so the masked part keeps a trace of the start image.
|
|
438
|
+
"inpaint_default_strength": 0.9999,
|
|
364
439
|
"components": {
|
|
365
|
-
"feature_extractor": [None, None],
|
|
366
|
-
"image_encoder": [None, None],
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
"
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
440
|
+
"feature_extractor": [[None, None]],
|
|
441
|
+
"image_encoder": [[None, None]],
|
|
442
|
+
# Many community SDXL finetunes ship the ancestral Euler
|
|
443
|
+
# sampler. Both classes read the same
|
|
444
|
+
# scheduler_config.json.
|
|
445
|
+
"scheduler": [
|
|
446
|
+
["diffusers", "EulerDiscreteScheduler"],
|
|
447
|
+
["diffusers", "EulerAncestralDiscreteScheduler"],
|
|
448
|
+
],
|
|
449
|
+
"text_encoder": [["transformers", "CLIPTextModel"]],
|
|
450
|
+
"text_encoder_2": [["transformers", "CLIPTextModelWithProjection"]],
|
|
451
|
+
"tokenizer": [["transformers", "CLIPTokenizer"]],
|
|
452
|
+
"tokenizer_2": [["transformers", "CLIPTokenizer"]],
|
|
453
|
+
"unet": [["diffusers", "UNet2DConditionModel"]],
|
|
454
|
+
"vae": [["diffusers", "AutoencoderKL"]],
|
|
374
455
|
},
|
|
375
456
|
# Every SDXL checkpoint sets this; it makes an empty negative prompt
|
|
376
457
|
# encode as zeros, as the model was trained.
|
|
@@ -421,15 +502,16 @@ def family_of(model_index):
|
|
|
421
502
|
)
|
|
422
503
|
named = {key: value for key, value in named.items() if key not in rules["settings"]}
|
|
423
504
|
for component, value in named.items():
|
|
424
|
-
|
|
425
|
-
if
|
|
505
|
+
allowed = rules["components"].get(component)
|
|
506
|
+
if allowed is None:
|
|
426
507
|
raise ValueError(
|
|
427
508
|
f'{rules["label"]} has no component "{component}", '
|
|
428
509
|
"and this model_index.json names one."
|
|
429
510
|
)
|
|
430
|
-
if value
|
|
511
|
+
if value not in allowed:
|
|
431
512
|
raise ValueError(
|
|
432
|
-
f'{rules["label"]}\'s "{component}" must be
|
|
513
|
+
f'{rules["label"]}\'s "{component}" must be '
|
|
514
|
+
f"{join_names([str(pair) for pair in allowed], 'or')}. "
|
|
433
515
|
f"This model_index.json says {value}."
|
|
434
516
|
)
|
|
435
517
|
missing = [c for c in rules["components"] if c not in named]
|
|
@@ -526,17 +608,18 @@ def _negative_prompt_of(rules, body):
|
|
|
526
608
|
|
|
527
609
|
def _strength_of(rules, body, mode):
|
|
528
610
|
"""How much of the start image to redraw: the request's strength, or
|
|
529
|
-
the family's default. None outside
|
|
530
|
-
already refused a strength."""
|
|
531
|
-
if mode
|
|
611
|
+
the family's default for the mode. None outside REDRAW_MODES, where
|
|
612
|
+
mode_of has already refused a strength."""
|
|
613
|
+
if mode not in REDRAW_MODES:
|
|
532
614
|
return None
|
|
615
|
+
default = rules["default_strength"] if mode == "img2img" else rules["inpaint_default_strength"]
|
|
533
616
|
strength = body.get("strength")
|
|
534
617
|
if strength is None:
|
|
535
|
-
return
|
|
618
|
+
return default
|
|
536
619
|
if not _is_number(strength) or strength <= 0 or strength > 1:
|
|
537
620
|
raise RequestError(
|
|
538
621
|
"strength must be a number above 0 and at most 1. Low keeps the start image close; "
|
|
539
|
-
f"{rules['label']} uses {
|
|
622
|
+
f"{rules['label']} uses {default} when strength is left out."
|
|
540
623
|
)
|
|
541
624
|
return float(strength)
|
|
542
625
|
|
|
@@ -804,16 +887,28 @@ def _check_unknown_fields(body):
|
|
|
804
887
|
|
|
805
888
|
def mode_of(rules, body):
|
|
806
889
|
"""The request's mode: the mode of the image field it carries, or
|
|
807
|
-
"plain" when it carries none.
|
|
808
|
-
|
|
809
|
-
|
|
890
|
+
"plain" when it carries none. A field that goes with another decides
|
|
891
|
+
the mode of the pair: a start image with a mask is inpaint. Refuses
|
|
892
|
+
such a field without its partner, image fields of two modes, a field
|
|
893
|
+
of a mode the request is not in, and a mode the family has no pipeline
|
|
894
|
+
for."""
|
|
810
895
|
present = [field for field in INPUT_IMAGES if body.get(field) is not None]
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
896
|
+
for field in present:
|
|
897
|
+
partner = INPUT_IMAGES[field].get("goes_with")
|
|
898
|
+
if partner is not None and partner not in present:
|
|
899
|
+
raise RequestError(f"{field} goes with {partner}, and this request has none.")
|
|
900
|
+
leads = [field for field in present if "goes_with" not in INPUT_IMAGES[field]]
|
|
901
|
+
if len(leads) > 1:
|
|
902
|
+
choices = [field for field in INPUT_IMAGES if "goes_with" not in INPUT_IMAGES[field]]
|
|
903
|
+
raise RequestError(f"a request takes one of {join_names(choices, 'or')}.")
|
|
904
|
+
deciding = [field for field in present if "goes_with" in INPUT_IMAGES[field]] or leads
|
|
905
|
+
field = deciding[0] if deciding else None
|
|
814
906
|
mode = "plain" if field is None else INPUT_IMAGES[field]["mode"]
|
|
907
|
+
allowed = MODE_FIELDS.get(mode, [])
|
|
908
|
+
# A field of several modes, such as strength, is named with the first
|
|
909
|
+
# mode that has it.
|
|
815
910
|
for other, fields in MODE_FIELDS.items():
|
|
816
|
-
stray = [name for name in fields if
|
|
911
|
+
stray = [name for name in fields if name not in allowed and body.get(name) is not None]
|
|
817
912
|
if stray:
|
|
818
913
|
verb = "goes" if len(stray) == 1 else "go"
|
|
819
914
|
raise RequestError(
|
|
@@ -835,6 +930,12 @@ def image_field_of(mode):
|
|
|
835
930
|
return None
|
|
836
931
|
|
|
837
932
|
|
|
933
|
+
def input_images_of(body, fields):
|
|
934
|
+
"""The bytes of each image the request carries, by field, for the image
|
|
935
|
+
fields of its mode."""
|
|
936
|
+
return {field: input_bytes(body, field) for field in fields}
|
|
937
|
+
|
|
938
|
+
|
|
838
939
|
def input_bytes(body, field):
|
|
839
940
|
"""The bytes of each image in the field `field`, from the request's
|
|
840
941
|
base64: a list, empty when `field` is None. Holds the field to its
|
|
@@ -1046,6 +1147,18 @@ def start_image_problem(width, height):
|
|
|
1046
1147
|
return _shape_problem(width, height, "A start image")
|
|
1047
1148
|
|
|
1048
1149
|
|
|
1150
|
+
def size_match_problem(field, size, other, other_size):
|
|
1151
|
+
"""Why an image of `size` in `field` cannot go with one of `other_size`
|
|
1152
|
+
in the field `other`, which its row says it must match, or None. Sizes
|
|
1153
|
+
are (width, height)."""
|
|
1154
|
+
if size == other_size:
|
|
1155
|
+
return None
|
|
1156
|
+
return (
|
|
1157
|
+
f"{field} is {size[0]}x{size[1]} and {other} is {other_size[0]}x{other_size[1]}. "
|
|
1158
|
+
"They must be the same size."
|
|
1159
|
+
)
|
|
1160
|
+
|
|
1161
|
+
|
|
1049
1162
|
# The checks a row of INPUT_IMAGES names in `check`.
|
|
1050
1163
|
CHECKS = {"reference_problem": reference_problem, "start_image_problem": start_image_problem}
|
|
1051
1164
|
|
|
@@ -1081,8 +1194,8 @@ STEP_FORMULAS = {"down": _steps_down, "up": _steps_up}
|
|
|
1081
1194
|
|
|
1082
1195
|
def steps_run(rules, mode, steps, strength):
|
|
1083
1196
|
"""How many steps the pipeline runs for a request in `mode` that asks
|
|
1084
|
-
for `steps` at `strength`: all of them, except in
|
|
1085
|
-
if mode
|
|
1197
|
+
for `steps` at `strength`: all of them, except in REDRAW_MODES."""
|
|
1198
|
+
if mode not in REDRAW_MODES:
|
|
1086
1199
|
return steps
|
|
1087
1200
|
return STEP_FORMULAS[rules["img2img_steps"]](steps, strength)
|
|
1088
1201
|
|
|
@@ -1095,13 +1208,13 @@ def check_request(rules, body, adapters_dir=None, controlnets_dir=None):
|
|
|
1095
1208
|
model takes instead.
|
|
1096
1209
|
|
|
1097
1210
|
`steps_run` is how many of `steps` the pipeline runs, which is fewer
|
|
1098
|
-
in
|
|
1211
|
+
in REDRAW_MODES. A request that would run none is refused.
|
|
1099
1212
|
|
|
1100
1213
|
`size` is the (width, height) the request gave, or None: with none,
|
|
1101
1214
|
the size depends on the first input image, which only the server can
|
|
1102
|
-
open, so output_size decides it there. `
|
|
1103
|
-
the request carries,
|
|
1104
|
-
bytes of each image in
|
|
1215
|
+
open, so output_size decides it there. `image_fields` lists the image
|
|
1216
|
+
fields the request carries, the one that sets the size first, and
|
|
1217
|
+
`input_images` holds the bytes of each image in each of them."""
|
|
1105
1218
|
if not isinstance(body, dict):
|
|
1106
1219
|
raise RequestError("The request body must be a JSON object.")
|
|
1107
1220
|
_check_unknown_fields(body)
|
|
@@ -1111,8 +1224,8 @@ def check_request(rules, body, adapters_dir=None, controlnets_dir=None):
|
|
|
1111
1224
|
_check_control_pairing(body)
|
|
1112
1225
|
mode = mode_of(rules, body)
|
|
1113
1226
|
control = _controlnet_of(body, controlnets_dir)
|
|
1114
|
-
|
|
1115
|
-
input_images =
|
|
1227
|
+
image_fields = image_fields_of(mode)
|
|
1228
|
+
input_images = input_images_of(body, image_fields)
|
|
1116
1229
|
steps = _steps_of(rules, body)
|
|
1117
1230
|
strength = _strength_of(rules, body, mode)
|
|
1118
1231
|
run = steps_run(rules, mode, steps, strength)
|
|
@@ -1122,7 +1235,7 @@ def check_request(rules, body, adapters_dir=None, controlnets_dir=None):
|
|
|
1122
1235
|
"prompt": _prompt_of(body),
|
|
1123
1236
|
"size": size,
|
|
1124
1237
|
"mode": mode,
|
|
1125
|
-
"
|
|
1238
|
+
"image_fields": image_fields,
|
|
1126
1239
|
"input_images": input_images,
|
|
1127
1240
|
"steps": steps,
|
|
1128
1241
|
"strength": strength,
|
|
@@ -1153,9 +1266,10 @@ def pipeline_args(rules, request, width, height):
|
|
|
1153
1266
|
args["negative_prompt"] = negative
|
|
1154
1267
|
if request["mode"] == "control":
|
|
1155
1268
|
args["controlnet_conditioning_scale"] = request["control_scale"]
|
|
1156
|
-
if request["mode"]
|
|
1269
|
+
if request["mode"] in REDRAW_MODES:
|
|
1157
1270
|
args["strength"] = request["strength"]
|
|
1158
|
-
|
|
1271
|
+
# Every inpaint pipeline takes a size.
|
|
1272
|
+
if request["mode"] == "img2img" and not rules["img2img_takes_size"]:
|
|
1159
1273
|
# The pipeline draws at the start image's size, which the
|
|
1160
1274
|
# server has already fitted to width x height.
|
|
1161
1275
|
del args["width"], args["height"]
|
|
@@ -54,6 +54,7 @@ from diffusersImageRules import ( # noqa: E402
|
|
|
54
54
|
join_names,
|
|
55
55
|
output_size,
|
|
56
56
|
pipeline_args,
|
|
57
|
+
size_match_problem,
|
|
57
58
|
warm_up_request,
|
|
58
59
|
)
|
|
59
60
|
|
|
@@ -80,10 +81,11 @@ def decode_image(data, field):
|
|
|
80
81
|
MAX_INPUT_IMAGE_PIXELS is refused before its pixels are decoded. A file
|
|
81
82
|
that is cut short or damaged is refused too. The EXIF orientation is
|
|
82
83
|
applied, so a portrait photo from a phone stays upright. When the
|
|
83
|
-
field's row
|
|
84
|
-
first
|
|
85
|
-
|
|
86
|
-
|
|
84
|
+
field's row names a background, a transparent image is pasted onto that
|
|
85
|
+
color first. Converting it directly would drop the alpha band and keep
|
|
86
|
+
whatever color is stored under a transparent pixel, which editors often
|
|
87
|
+
save as white. Then the check the field's row names, if any, runs on the
|
|
88
|
+
upright image's size."""
|
|
87
89
|
from PIL import Image, ImageOps
|
|
88
90
|
|
|
89
91
|
unreadable = RequestError(
|
|
@@ -110,10 +112,10 @@ def decode_image(data, field):
|
|
|
110
112
|
# A palette or RGB image marks its transparent color in `info`, with no
|
|
111
113
|
# alpha band.
|
|
112
114
|
transparent = image.mode in ("RGBA", "LA", "PA") or "transparency" in image.info
|
|
113
|
-
|
|
115
|
+
background = INPUT_IMAGES[field]["background"]
|
|
116
|
+
if background is not None and transparent:
|
|
114
117
|
image = image.convert("RGBA")
|
|
115
|
-
|
|
116
|
-
image = Image.alpha_composite(white, image)
|
|
118
|
+
image = Image.alpha_composite(Image.new("RGBA", image.size, background), image)
|
|
117
119
|
problem = image_problem(field, image.width, image.height)
|
|
118
120
|
if problem is not None:
|
|
119
121
|
raise RequestError(f"{field}: {problem}")
|
|
@@ -155,6 +157,35 @@ def fitted(image, field, width, height):
|
|
|
155
157
|
canvas.paste(image.resize(size, Image.LANCZOS, box=source_box), position)
|
|
156
158
|
return canvas
|
|
157
159
|
|
|
160
|
+
def input_images(request):
|
|
161
|
+
"""(width, height, images) for a checked request: the output size, and
|
|
162
|
+
the request's input images decoded, prepared, and fitted to that size,
|
|
163
|
+
by the pipeline argument each field's row names. One image when the
|
|
164
|
+
field takes one, a list otherwise. The first image field sets the size,
|
|
165
|
+
and a field whose row names `same_size_as` must match that field's
|
|
166
|
+
picture before either is fitted."""
|
|
167
|
+
fields = request["image_fields"]
|
|
168
|
+
decoded = {
|
|
169
|
+
field: [prepared(decode_image(data, field), field, request) for data in request["input_images"][field]]
|
|
170
|
+
for field in fields
|
|
171
|
+
}
|
|
172
|
+
for field in fields:
|
|
173
|
+
other = INPUT_IMAGES[field].get("same_size_as")
|
|
174
|
+
if other is not None:
|
|
175
|
+
problem = size_match_problem(field, decoded[field][0].size, other, decoded[other][0].size)
|
|
176
|
+
if problem is not None:
|
|
177
|
+
raise RequestError(problem)
|
|
178
|
+
first_size = decoded[fields[0]][0].size if fields else None
|
|
179
|
+
lead = fields[0] if fields else None
|
|
180
|
+
width, height = output_size(request["size"], lead, first_size)
|
|
181
|
+
images = {}
|
|
182
|
+
for field in fields:
|
|
183
|
+
row = INPUT_IMAGES[field]
|
|
184
|
+
fitted_images = [fitted(image, field, width, height) for image in decoded[field]]
|
|
185
|
+
images[row["arg"]] = fitted_images[0] if row["max_count"] == 1 else fitted_images
|
|
186
|
+
return width, height, images
|
|
187
|
+
|
|
188
|
+
|
|
158
189
|
PIL_FORMATS = {"png": "PNG", "jpeg": "JPEG", "webp": "WEBP"}
|
|
159
190
|
|
|
160
191
|
|
|
@@ -364,19 +395,13 @@ class Generator:
|
|
|
364
395
|
|
|
365
396
|
# Everything up to the lock runs first, so a bad image never waits
|
|
366
397
|
# for the GPU or loads a ControlNet.
|
|
367
|
-
|
|
368
|
-
images = [prepared(decode_image(data, field), field, request) for data in request["input_images"]]
|
|
369
|
-
first_size = images[0].size if images else None
|
|
370
|
-
width, height = output_size(request["size"], field, first_size)
|
|
371
|
-
fitted_images = [fitted(image, field, width, height) for image in images]
|
|
398
|
+
width, height, images = input_images(request)
|
|
372
399
|
kwargs = {
|
|
373
400
|
**pipeline_args(self.rules, request, width, height),
|
|
401
|
+
**images,
|
|
374
402
|
"generator": torch.Generator("cpu").manual_seed(request["seed"]),
|
|
375
403
|
"callback_on_step_end": on_step_end,
|
|
376
404
|
}
|
|
377
|
-
if fitted_images:
|
|
378
|
-
one = INPUT_IMAGES[field]["max_count"] == 1
|
|
379
|
-
kwargs["image"] = fitted_images[0] if one else fitted_images
|
|
380
405
|
with self.lock:
|
|
381
406
|
# A client that hung up while it waited for the lock gets
|
|
382
407
|
# nothing started at all.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import * as path from "node:path";
|
|
2
|
-
import { _registerLocalModel, _resolveModel, _mlxServedName } from "../stdlib/localModels.js";
|
|
2
|
+
import { _registerLocalModel, _resolveModel, _mlxServedName, _catalogKind, } from "../stdlib/localModels.js";
|
|
3
3
|
/** Turn `agency run --local <value>` into the shared model-flag shape:
|
|
4
4
|
* resolve the name, and for a GGUF model download and verify if needed
|
|
5
5
|
* (progress prints here, in the parent, before the program starts) and pin
|
|
@@ -13,6 +13,9 @@ import { _registerLocalModel, _resolveModel, _mlxServedName } from "../stdlib/lo
|
|
|
13
13
|
* absolute path also keeps the child process independent of cwd drift. */
|
|
14
14
|
export async function resolveLocalRunFlag(value, draft) {
|
|
15
15
|
const resolved = _resolveModel(value);
|
|
16
|
+
if (_catalogKind(value) === "controlnet") {
|
|
17
|
+
throw new Error(`${value} is a ControlNet, which an SDXL image server loads for a request. Pass it to generateImageLocal as the controlnet argument.`);
|
|
18
|
+
}
|
|
16
19
|
if (resolved.backend === "diffusers") {
|
|
17
20
|
throw new Error(`${value} is an image model. Serve it with agency local serve --image ${value} and call generateImageLocal.`);
|
|
18
21
|
}
|
package/dist/lib/stdlib/image.js
CHANGED
|
@@ -5,7 +5,7 @@ import { getRuntimeContext } from "../runtime/asyncContext.js";
|
|
|
5
5
|
import { success, failure } from "../runtime/result.js";
|
|
6
6
|
import { recordUsage, meteredDispatch } from "../runtime/recordPaidUsage.js";
|
|
7
7
|
import { classifySource } from "./thread.js";
|
|
8
|
-
import { _resolveModel, _mlxServedName } from "./localModels.js";
|
|
8
|
+
import { _resolveModel, _mlxServedName, _catalogKind } from "./localModels.js";
|
|
9
9
|
import { mlxBaseUrl, isNoServerError } from "./mlxServerModels.js";
|
|
10
10
|
import { LOCAL_IMAGE_FORMATS } from "./mlxImage.js";
|
|
11
11
|
import { PROMPT_PREVIEW_MAX } from "../statelogClient.js";
|
|
@@ -173,6 +173,12 @@ function checkLocalImageArgs(prompt, model, format) {
|
|
|
173
173
|
catch (err) {
|
|
174
174
|
return { error: err.message };
|
|
175
175
|
}
|
|
176
|
+
if (_catalogKind(model) === "controlnet") {
|
|
177
|
+
return {
|
|
178
|
+
error: `"${model}" is a ControlNet, not an image model. Pass it as the controlnet ` +
|
|
179
|
+
"argument, with a controlImage, and name an SDXL image model as the model.",
|
|
180
|
+
};
|
|
181
|
+
}
|
|
176
182
|
if (resolved.backend !== "diffusers") {
|
|
177
183
|
const what = resolved.backend === "mlx" ? "an MLX model" : "a GGUF model";
|
|
178
184
|
return {
|
|
@@ -182,22 +188,28 @@ function checkLocalImageArgs(prompt, model, format) {
|
|
|
182
188
|
}
|
|
183
189
|
return { servedName: _mlxServedName(resolved) };
|
|
184
190
|
}
|
|
185
|
-
/** The request
|
|
191
|
+
/** The request fields that carry a call's input images, with each file's
|
|
186
192
|
* bytes as base64: one string when the field takes one image, a list
|
|
187
193
|
* otherwise. The files are read here, after the Agency side raised
|
|
188
194
|
* std::readImage for each, so the server never opens a path a request
|
|
189
195
|
* wrote. `approvedFileBytes` refuses a symlink that appeared while the
|
|
190
196
|
* prompt was pending, and a file over the field's size cap. */
|
|
191
197
|
function imageFields(inputs) {
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
+
const encoded = {};
|
|
199
|
+
for (const file of inputs.files) {
|
|
200
|
+
const row = LOCAL_IMAGE_FIELDS[file.field];
|
|
201
|
+
if (row === undefined) {
|
|
202
|
+
throw new Error(`${file.field} is not an image field of the image server.`);
|
|
203
|
+
}
|
|
204
|
+
encoded[file.field] = [
|
|
205
|
+
...(encoded[file.field] ?? []),
|
|
206
|
+
approvedFileBytes(file.path, row.maxBytes).toString("base64"),
|
|
207
|
+
];
|
|
198
208
|
}
|
|
199
|
-
|
|
200
|
-
|
|
209
|
+
return Object.fromEntries(Object.entries(encoded).map(([field, images]) => [
|
|
210
|
+
field,
|
|
211
|
+
LOCAL_IMAGE_FIELDS[field].maxCount === 1 ? images[0] : images,
|
|
212
|
+
]));
|
|
201
213
|
}
|
|
202
214
|
/** The settings a call gives, as the request fields the server takes. A
|
|
203
215
|
* null setting, or an empty negative prompt, is left out so the server
|
|
@@ -3,7 +3,8 @@
|
|
|
3
3
|
* table for the image server, and a test compares the two.
|
|
4
4
|
*
|
|
5
5
|
* mode which kind of request the field makes: a ControlNet request,
|
|
6
|
-
* an edit from reference pictures,
|
|
6
|
+
* an edit from reference pictures, a redraw of a start image, or
|
|
7
|
+
* a redraw of the part of a start image a mask marks
|
|
7
8
|
* maxCount how many images the field takes. One is sent as a base64
|
|
8
9
|
* string, more as a list of them
|
|
9
10
|
* maxBytes the largest file each image may be
|
|
@@ -11,14 +12,17 @@
|
|
|
11
12
|
* question what the std::readImage interrupt asks before the file is read
|
|
12
13
|
* readEachStep true: the model reads each image at every step, as much
|
|
13
14
|
* work as one more megapixel of output, and the provider's
|
|
14
|
-
* timeout budgets for it
|
|
15
|
+
* timeout budgets for it
|
|
16
|
+
* goesWith optional: the field this one comes only with. A mask goes
|
|
17
|
+
* with a start image */
|
|
15
18
|
export type LocalImageField = {
|
|
16
|
-
mode: "control" | "reference" | "img2img";
|
|
19
|
+
mode: "control" | "reference" | "img2img" | "inpaint";
|
|
17
20
|
maxCount: number;
|
|
18
21
|
maxBytes: number;
|
|
19
22
|
parameter: string;
|
|
20
23
|
question: string;
|
|
21
24
|
readEachStep: boolean;
|
|
25
|
+
goesWith?: string;
|
|
22
26
|
};
|
|
23
27
|
/** A control image is read up to the size every local server takes. */
|
|
24
28
|
export declare const MAX_CONTROL_IMAGE_BYTES = 50000000;
|
|
@@ -31,12 +35,16 @@ export declare const MAX_REFERENCE_IMAGES = 4;
|
|
|
31
35
|
* the settings. */
|
|
32
36
|
export declare const REQUEST_SETTINGS_BYTES: number;
|
|
33
37
|
export declare const LOCAL_IMAGE_FIELDS: Record<string, LocalImageField>;
|
|
38
|
+
/** The image fields a request in `mode` carries, the one that sets the size
|
|
39
|
+
* first. A field that goes with another brings it along, so inpaint is
|
|
40
|
+
* start_image and mask_image. `image_fields_of` in diffusersImageRules.py
|
|
41
|
+
* is the same. */
|
|
42
|
+
export declare function imageFieldsOf(mode: LocalImageField["mode"]): string[];
|
|
34
43
|
/** How many characters base64 turns `bytes` bytes into. */
|
|
35
44
|
export declare function base64Length(bytes: number): number;
|
|
36
45
|
/** The largest request body the image server takes: the settings, plus the
|
|
37
|
-
* base64 of the most image bytes
|
|
38
|
-
*
|
|
39
|
-
* diffusersImageRules.py is computed the same way. */
|
|
46
|
+
* base64 of the most image bytes one mode's fields may carry.
|
|
47
|
+
* `MAX_BODY_BYTES` in diffusersImageRules.py is computed the same way. */
|
|
40
48
|
export declare function localBodyBytes(): number;
|
|
41
49
|
/** The image types a local input may be, by extension. */
|
|
42
50
|
export declare const IMAGE_MIME_TYPES: Record<string, string>;
|
|
@@ -48,19 +56,19 @@ export declare function isRemoteSource(source: string): boolean;
|
|
|
48
56
|
export declare function checkedImageFile(spelling: string, maxBytes: number, caller: string): string;
|
|
49
57
|
/** One file a `generateImageLocal` call reads, after the std::readImage
|
|
50
58
|
* interrupt that shows its folder and name asks `question`. `path` is its
|
|
51
|
-
* real spelling, the one the interrupt shows
|
|
59
|
+
* real spelling, the one the interrupt shows, and `field` is the request
|
|
60
|
+
* field it goes in. */
|
|
52
61
|
export type LocalImageFile = {
|
|
62
|
+
field: string;
|
|
53
63
|
path: string;
|
|
54
64
|
dir: string;
|
|
55
65
|
filename: string;
|
|
56
66
|
question: string;
|
|
57
67
|
};
|
|
58
|
-
/** The input images of one `generateImageLocal` call
|
|
59
|
-
*
|
|
60
|
-
*
|
|
61
|
-
* `controlnet` and `control_scale`. */
|
|
68
|
+
/** The input images of one `generateImageLocal` call, none for a call
|
|
69
|
+
* with none. `settings` holds the other request fields of the call's
|
|
70
|
+
* mode, such as `controlnet` and `control_scale`. */
|
|
62
71
|
export type LocalImageInputs = {
|
|
63
|
-
field: string | null;
|
|
64
72
|
files: LocalImageFile[];
|
|
65
73
|
settings: Record<string, unknown>;
|
|
66
74
|
};
|
|
@@ -73,4 +81,4 @@ export declare function referenceCount(inputs: LocalImageInputs): number;
|
|
|
73
81
|
* its field's size cap. Returns the files to raise std::readImage for.
|
|
74
82
|
* Throws with the message to fail with. The messages match the image
|
|
75
83
|
* server's, with the parameters' names for the request fields'. */
|
|
76
|
-
export declare function _localImageInputs(controlnet: string, controlImage: string, controlScale: number | null, invertControlImage: boolean, images: string[], startImage: string, strength: number | null): LocalImageInputs;
|
|
84
|
+
export declare function _localImageInputs(controlnet: string, controlImage: string, controlScale: number | null, invertControlImage: boolean, images: string[], startImage: string, strength: number | null, mask: string): LocalImageInputs;
|
|
@@ -38,18 +38,42 @@ export const LOCAL_IMAGE_FIELDS = {
|
|
|
38
38
|
// The model starts from it once, instead of from noise.
|
|
39
39
|
readEachStep: false,
|
|
40
40
|
},
|
|
41
|
+
mask_image: {
|
|
42
|
+
mode: "inpaint",
|
|
43
|
+
maxCount: 1,
|
|
44
|
+
maxBytes: MAX_INPUT_IMAGE_BYTES,
|
|
45
|
+
parameter: "mask",
|
|
46
|
+
question: "Read this mask to choose which part of the picture to redraw?",
|
|
47
|
+
readEachStep: false,
|
|
48
|
+
goesWith: "start_image",
|
|
49
|
+
},
|
|
41
50
|
};
|
|
51
|
+
/** The image fields a request in `mode` carries, the one that sets the size
|
|
52
|
+
* first. A field that goes with another brings it along, so inpaint is
|
|
53
|
+
* start_image and mask_image. `image_fields_of` in diffusersImageRules.py
|
|
54
|
+
* is the same. */
|
|
55
|
+
export function imageFieldsOf(mode) {
|
|
56
|
+
const fields = Object.keys(LOCAL_IMAGE_FIELDS).filter((field) => LOCAL_IMAGE_FIELDS[field].mode === mode);
|
|
57
|
+
const partners = fields.flatMap((field) => {
|
|
58
|
+
const partner = LOCAL_IMAGE_FIELDS[field].goesWith;
|
|
59
|
+
return partner === undefined ? [] : [partner];
|
|
60
|
+
});
|
|
61
|
+
return [...partners, ...fields];
|
|
62
|
+
}
|
|
42
63
|
/** How many characters base64 turns `bytes` bytes into. */
|
|
43
64
|
export function base64Length(bytes) {
|
|
44
65
|
return 4 * Math.ceil(bytes / 3);
|
|
45
66
|
}
|
|
46
67
|
/** The largest request body the image server takes: the settings, plus the
|
|
47
|
-
* base64 of the most image bytes
|
|
48
|
-
*
|
|
49
|
-
* diffusersImageRules.py is computed the same way. */
|
|
68
|
+
* base64 of the most image bytes one mode's fields may carry.
|
|
69
|
+
* `MAX_BODY_BYTES` in diffusersImageRules.py is computed the same way. */
|
|
50
70
|
export function localBodyBytes() {
|
|
51
|
-
const
|
|
52
|
-
|
|
71
|
+
const modes = Object.values(LOCAL_IMAGE_FIELDS).map((row) => row.mode);
|
|
72
|
+
const most = Math.max(...modes.map((mode) => imageFieldsOf(mode).reduce((total, field) => {
|
|
73
|
+
const row = LOCAL_IMAGE_FIELDS[field];
|
|
74
|
+
return total + base64Length(row.maxCount * row.maxBytes);
|
|
75
|
+
}, 0)));
|
|
76
|
+
return REQUEST_SETTINGS_BYTES + most;
|
|
53
77
|
}
|
|
54
78
|
/** The image types a local input may be, by extension. */
|
|
55
79
|
export const IMAGE_MIME_TYPES = Object.fromEntries(Object.entries(MIME_TYPES).filter(([, mime]) => mime.startsWith("image/")));
|
|
@@ -109,7 +133,8 @@ function refusal(message) {
|
|
|
109
133
|
/** A local path checked for the row's byte cap, with what its interrupt
|
|
110
134
|
* shows. A URL or a data URI is refused: the image server never fetches
|
|
111
135
|
* anything, so every input is a file on this machine. */
|
|
112
|
-
function localImageFile(spelling,
|
|
136
|
+
function localImageFile(spelling, field) {
|
|
137
|
+
const row = LOCAL_IMAGE_FIELDS[field];
|
|
113
138
|
if (isRemoteSource(spelling)) {
|
|
114
139
|
throw new Error(`${CALLER} reads files on this machine only.`);
|
|
115
140
|
}
|
|
@@ -121,6 +146,7 @@ function localImageFile(spelling, row) {
|
|
|
121
146
|
throw refusal(err.message);
|
|
122
147
|
}
|
|
123
148
|
return {
|
|
149
|
+
field,
|
|
124
150
|
path: real,
|
|
125
151
|
dir: path.dirname(real),
|
|
126
152
|
filename: path.basename(real),
|
|
@@ -130,10 +156,7 @@ function localImageFile(spelling, row) {
|
|
|
130
156
|
/** How many images of `inputs` the model reads at every step, which the
|
|
131
157
|
* provider's timeout budgets for. 0 for a call with none. */
|
|
132
158
|
export function referenceCount(inputs) {
|
|
133
|
-
|
|
134
|
-
return 0;
|
|
135
|
-
}
|
|
136
|
-
return inputs.files.length;
|
|
159
|
+
return inputs.files.filter((file) => LOCAL_IMAGE_FIELDS[file.field].readEachStep).length;
|
|
137
160
|
}
|
|
138
161
|
/** Backs the checks `generateImageLocal` makes before it asks anything:
|
|
139
162
|
* which input images the call has, whether they go together, whether
|
|
@@ -141,10 +164,15 @@ export function referenceCount(inputs) {
|
|
|
141
164
|
* its field's size cap. Returns the files to raise std::readImage for.
|
|
142
165
|
* Throws with the message to fail with. The messages match the image
|
|
143
166
|
* server's, with the parameters' names for the request fields'. */
|
|
144
|
-
export function _localImageInputs(controlnet, controlImage, controlScale, invertControlImage, images, startImage, strength) {
|
|
167
|
+
export function _localImageInputs(controlnet, controlImage, controlScale, invertControlImage, images, startImage, strength, mask) {
|
|
145
168
|
if ((controlnet === "") !== (controlImage === "")) {
|
|
146
169
|
throw refusal("controlnet and controlImage go together: the ControlNet's name, and the image it conditions the generation on.");
|
|
147
170
|
}
|
|
171
|
+
// The mask is checked first, as the server does, so a call with a mask
|
|
172
|
+
// and a strength but no start image hears about the mask.
|
|
173
|
+
if (mask !== "" && startImage === "") {
|
|
174
|
+
throw refusal("mask goes with startImage, and this call has none.");
|
|
175
|
+
}
|
|
148
176
|
if (strength !== null && startImage === "") {
|
|
149
177
|
throw refusal("strength goes with startImage, and this call has none.");
|
|
150
178
|
}
|
|
@@ -157,29 +185,38 @@ export function _localImageInputs(controlnet, controlImage, controlScale, invert
|
|
|
157
185
|
control_image: controlImage === "" ? [] : [controlImage],
|
|
158
186
|
images,
|
|
159
187
|
start_image: startImage === "" ? [] : [startImage],
|
|
188
|
+
mask_image: mask === "" ? [] : [mask],
|
|
160
189
|
};
|
|
161
190
|
// The other request fields of each image field's mode.
|
|
162
191
|
const settingsOf = {
|
|
163
192
|
control_image: controlSettings(controlnet, controlScale, invertControlImage),
|
|
164
193
|
images: {},
|
|
165
194
|
start_image: strengthSettings(strength),
|
|
195
|
+
mask_image: {},
|
|
166
196
|
};
|
|
167
|
-
|
|
197
|
+
// A field that goes with another, such as the mask, is not a choice of
|
|
198
|
+
// its own: it comes along with its partner.
|
|
199
|
+
const leads = Object.keys(paths).filter((field) => LOCAL_IMAGE_FIELDS[field].goesWith === undefined);
|
|
200
|
+
const given = leads.filter((field) => paths[field].length > 0);
|
|
168
201
|
if (given.length === 0) {
|
|
169
|
-
return {
|
|
202
|
+
return { files: [], settings: {} };
|
|
170
203
|
}
|
|
171
204
|
if (given.length > 1) {
|
|
172
|
-
const names =
|
|
205
|
+
const names = leads.map((field) => LOCAL_IMAGE_FIELDS[field].parameter);
|
|
173
206
|
throw refusal(`a call takes one of ${orList(names)}.`);
|
|
174
207
|
}
|
|
175
|
-
const
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
208
|
+
const fields = [
|
|
209
|
+
given[0],
|
|
210
|
+
...Object.keys(paths).filter((field) => LOCAL_IMAGE_FIELDS[field].goesWith === given[0] && paths[field].length > 0),
|
|
211
|
+
];
|
|
212
|
+
for (const field of fields) {
|
|
213
|
+
const row = LOCAL_IMAGE_FIELDS[field];
|
|
214
|
+
if (paths[field].length > row.maxCount) {
|
|
215
|
+
throw refusal(`${row.parameter} takes at most ${row.maxCount} images. This call has ${paths[field].length}.`);
|
|
216
|
+
}
|
|
179
217
|
}
|
|
180
218
|
return {
|
|
181
|
-
field,
|
|
182
|
-
|
|
183
|
-
settings: settingsOf[field],
|
|
219
|
+
files: fields.flatMap((field) => paths[field].map((spelling) => localImageFile(spelling, field))),
|
|
220
|
+
settings: Object.assign({}, ...fields.map((field) => settingsOf[field])),
|
|
184
221
|
};
|
|
185
222
|
}
|
|
@@ -142,7 +142,9 @@ export declare function _removeModel(name: string, cacheDir?: string): boolean;
|
|
|
142
142
|
export declare function _removeServedModel(backend: ServedBackend, repo: string, cacheDir?: string): boolean;
|
|
143
143
|
/** Where a resolved model's files are on disk, or null when nothing is
|
|
144
144
|
* there. A GGUF model is found through the download manifest; an MLX model
|
|
145
|
-
* through its record under the cache, or the directory it points at.
|
|
145
|
+
* through its record under the cache, or the directory it points at.
|
|
146
|
+
* `insideCache` is false for a ControlNet, whose files are listed with the
|
|
147
|
+
* models but live in `client.controlnetsDir`, where remove cannot reach. */
|
|
146
148
|
export declare function _modelFilesOnDisk(resolved: ResolvedModel, cacheDir?: string): {
|
|
147
149
|
path: string;
|
|
148
150
|
sizeBytes: number;
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { root, wholePath, stat, list, remove, readText, writeText, } from "./contained.js";
|
|
1
|
+
import { root, wholePath, stat, list, remove, readText, writeText, isContained, } from "./contained.js";
|
|
2
2
|
import * as os from "node:os";
|
|
3
3
|
import * as path from "node:path";
|
|
4
4
|
import { fileURLToPath } from "node:url";
|
|
@@ -543,7 +543,9 @@ export function _removeServedModel(backend, repo, cacheDir = "") {
|
|
|
543
543
|
}
|
|
544
544
|
/** Where a resolved model's files are on disk, or null when nothing is
|
|
545
545
|
* there. A GGUF model is found through the download manifest; an MLX model
|
|
546
|
-
* through its record under the cache, or the directory it points at.
|
|
546
|
+
* through its record under the cache, or the directory it points at.
|
|
547
|
+
* `insideCache` is false for a ControlNet, whose files are listed with the
|
|
548
|
+
* models but live in `client.controlnetsDir`, where remove cannot reach. */
|
|
547
549
|
export function _modelFilesOnDisk(resolved, cacheDir = "") {
|
|
548
550
|
const dir = resolveCacheDir(cacheDir);
|
|
549
551
|
const onDisk = _listDownloadedModels(dir);
|
|
@@ -551,7 +553,12 @@ export function _modelFilesOnDisk(resolved, cacheDir = "") {
|
|
|
551
553
|
const f = onDisk.find(match);
|
|
552
554
|
return f === undefined
|
|
553
555
|
? null
|
|
554
|
-
: {
|
|
556
|
+
: {
|
|
557
|
+
path: f.path,
|
|
558
|
+
sizeBytes: f.sizeBytes,
|
|
559
|
+
insideCache: isContained(f.path, dir),
|
|
560
|
+
layout: f.layout,
|
|
561
|
+
};
|
|
555
562
|
};
|
|
556
563
|
if (resolved.backend === "llama-cpp") {
|
|
557
564
|
if (isGgufPath(resolved.target)) {
|
|
@@ -651,9 +651,11 @@ export const CURATED_LOCAL_MODELS = {
|
|
|
651
651
|
// Not served: `agency local download` puts one in client.controlnetsDir,
|
|
652
652
|
// and an SDXL image server loads it when a request names it with a
|
|
653
653
|
// control image. Sizes are the config and the one weights file kept.
|
|
654
|
+
// The backend is diffusers because the download is recorded that way;
|
|
655
|
+
// `agency local list` only ticks a row whose backend matches its files.
|
|
654
656
|
"controlnet-scribble-sdxl": {
|
|
655
|
-
backend: "
|
|
656
|
-
uri: "
|
|
657
|
+
backend: "diffusers",
|
|
658
|
+
uri: "diffusers:xinsir/controlnet-scribble-sdxl-1.0",
|
|
657
659
|
params: "1.3B",
|
|
658
660
|
sizeBytes: 2502140339,
|
|
659
661
|
kind: "controlnet",
|
|
@@ -663,8 +665,8 @@ export const CURATED_LOCAL_MODELS = {
|
|
|
663
665
|
description: "Constrains an SDXL image to a rough line drawing: a stick figure becomes the pose. Give it a scribble as controlImage.",
|
|
664
666
|
},
|
|
665
667
|
"controlnet-openpose-sdxl": {
|
|
666
|
-
backend: "
|
|
667
|
-
uri: "
|
|
668
|
+
backend: "diffusers",
|
|
669
|
+
uri: "diffusers:xinsir/controlnet-openpose-sdxl-1.0",
|
|
668
670
|
params: "1.3B",
|
|
669
671
|
sizeBytes: 2502140339,
|
|
670
672
|
kind: "controlnet",
|
package/package.json
CHANGED
|
@@ -209,6 +209,7 @@ generateImageLocal(
|
|
|
209
209
|
images: string[] = [],
|
|
210
210
|
startImage: string = "",
|
|
211
211
|
strength: number | null = null,
|
|
212
|
+
mask: string = "",
|
|
212
213
|
): Result<LocalImage> raises <std::readImage>
|
|
213
214
|
```
|
|
214
215
|
|
|
@@ -249,6 +250,9 @@ Generate an image on this machine with a local image model. The model
|
|
|
249
250
|
scaled to cover `size` and the overflow is cropped from both sides, so a
|
|
250
251
|
4:3 photo redrawn as a square loses a strip at its left and right.
|
|
251
252
|
|
|
253
|
+
To redraw only part of the picture, also pass a `mask`: white is
|
|
254
|
+
redrawn and black is kept.
|
|
255
|
+
|
|
252
256
|
@param prompt - What to draw
|
|
253
257
|
@param model - The image model: a catalog name such as "z-image-turbo", a diffusers: URI, or a model directory
|
|
254
258
|
@param size - Width and height joined by "x", each a multiple of 16, such as "1024x1024" or "1344x768". Empty is 1024x1024. Leave it empty when editing or redrawing a picture, and the result keeps the picture's shape
|
|
@@ -265,7 +269,28 @@ Generate an image on this machine with a local image model. The model
|
|
|
265
269
|
@param invertControlImage - Swap black and white in the drawing before using it. Set it for dark lines on white with the scribble ControlNet, which reads white lines on black
|
|
266
270
|
@param images - Pictures to edit, as paths to files on this machine. The prompt says what to change: 'add a hat to the character'. Only FLUX.2 [klein] takes them, and at most 4
|
|
267
271
|
@param startImage - A picture to redraw, as a path to a file on this machine. The layout stays and the style changes. Goes with strength. Every model but FLUX.2 [klein] takes one
|
|
268
|
-
@param strength - How much of startImage to redraw, above 0 and up to 1. Low keeps it close, high changes more. Null uses the model's default
|
|
272
|
+
@param strength - How much of startImage to redraw, above 0 and up to 1. Low keeps it close, high changes more. Null uses the model's default, which can change when a mask is added: for Z-Image Turbo it goes from 0.6 to 1.0, which redraws the white part from scratch
|
|
273
|
+
@param mask - A black-and-white picture the same size as startImage, as a path to a file on this machine. White is redrawn and black is kept. Goes with startImage. Every model but FLUX.2 [klein] takes one
|
|
274
|
+
|
|
275
|
+
Redrawing part of a picture: pass a `mask` with `startImage`. The mask
|
|
276
|
+
is a picture the same size as the start image. White marks the part to
|
|
277
|
+
redraw and black the part to keep, so a white shape over a vase changes
|
|
278
|
+
only the vase:
|
|
279
|
+
|
|
280
|
+
```ts
|
|
281
|
+
generateImageLocal("a vase of sunflowers", "z-image-turbo",
|
|
282
|
+
startImage: "photo.png", mask: "vase-mask.png")
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
Grey redraws partly, so a mask whose edge fades from white to black
|
|
286
|
+
blends the new part in, where a hard edge can leave a faint seam. A
|
|
287
|
+
transparent part of a mask counts as black. The mask is read from this
|
|
288
|
+
machine under std::readImage, like the start image.
|
|
289
|
+
|
|
290
|
+
Each model has its own default strength with a mask, which can differ
|
|
291
|
+
from its default without one. Z-Image Turbo's goes from 0.6 to 1.0, and
|
|
292
|
+
1.0 draws the white part from scratch, keeping nothing of what was
|
|
293
|
+
there. Pass a lower strength to keep some of it.
|
|
269
294
|
|
|
270
295
|
**Parameters:**
|
|
271
296
|
|
|
@@ -288,12 +313,13 @@ Generate an image on this machine with a local image model. The model
|
|
|
288
313
|
| images | `string[]` | [] |
|
|
289
314
|
| startImage | `string` | "" |
|
|
290
315
|
| strength | `number \| null` | null |
|
|
316
|
+
| mask | `string` | "" |
|
|
291
317
|
|
|
292
318
|
**Returns:** `Result<LocalImage>`
|
|
293
319
|
|
|
294
320
|
**Throws:** `std::readImage`
|
|
295
321
|
|
|
296
|
-
([source](https://github.com/egonSchiele/agency-lang/tree/main/packages/agency-lang/stdlib/image.agency#
|
|
322
|
+
([source](https://github.com/egonSchiele/agency-lang/tree/main/packages/agency-lang/stdlib/image.agency#L159))
|
|
297
323
|
|
|
298
324
|
### cropImage
|
|
299
325
|
|
|
@@ -332,7 +358,7 @@ Cut a box out of an image and write it as a new image. Returns the path
|
|
|
332
358
|
|
|
333
359
|
**Throws:** `std::cropImage`
|
|
334
360
|
|
|
335
|
-
([source](https://github.com/egonSchiele/agency-lang/tree/main/packages/agency-lang/stdlib/image.agency#
|
|
361
|
+
([source](https://github.com/egonSchiele/agency-lang/tree/main/packages/agency-lang/stdlib/image.agency#L272))
|
|
336
362
|
|
|
337
363
|
### imageSize
|
|
338
364
|
|
|
@@ -354,7 +380,7 @@ The width and height of an image in pixels.
|
|
|
354
380
|
|
|
355
381
|
**Throws:** `std::readImage`
|
|
356
382
|
|
|
357
|
-
([source](https://github.com/egonSchiele/agency-lang/tree/main/packages/agency-lang/stdlib/image.agency#
|
|
383
|
+
([source](https://github.com/egonSchiele/agency-lang/tree/main/packages/agency-lang/stdlib/image.agency#L308))
|
|
358
384
|
|
|
359
385
|
### pasteImages
|
|
360
386
|
|
|
@@ -387,4 +413,4 @@ Lay images out on one white canvas, in rows of `columns`, each at its own
|
|
|
387
413
|
|
|
388
414
|
**Throws:** `std::pasteImages`
|
|
389
415
|
|
|
390
|
-
([source](https://github.com/egonSchiele/agency-lang/tree/main/packages/agency-lang/stdlib/image.agency#
|
|
416
|
+
([source](https://github.com/egonSchiele/agency-lang/tree/main/packages/agency-lang/stdlib/image.agency#L325))
|
package/stdlib/image.agency
CHANGED
|
@@ -137,6 +137,25 @@ export type LocalImage = {
|
|
|
137
137
|
seed: number
|
|
138
138
|
}
|
|
139
139
|
|
|
140
|
+
/** Redrawing part of a picture: pass a `mask` with `startImage`. The mask
|
|
141
|
+
is a picture the same size as the start image. White marks the part to
|
|
142
|
+
redraw and black the part to keep, so a white shape over a vase changes
|
|
143
|
+
only the vase:
|
|
144
|
+
|
|
145
|
+
```ts
|
|
146
|
+
generateImageLocal("a vase of sunflowers", "z-image-turbo",
|
|
147
|
+
startImage: "photo.png", mask: "vase-mask.png")
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
Grey redraws partly, so a mask whose edge fades from white to black
|
|
151
|
+
blends the new part in, where a hard edge can leave a faint seam. A
|
|
152
|
+
transparent part of a mask counts as black. The mask is read from this
|
|
153
|
+
machine under std::readImage, like the start image.
|
|
154
|
+
|
|
155
|
+
Each model has its own default strength with a mask, which can differ
|
|
156
|
+
from its default without one. Z-Image Turbo's goes from 0.6 to 1.0, and
|
|
157
|
+
1.0 draws the white part from scratch, keeping nothing of what was
|
|
158
|
+
there. Pass a lower strength to keep some of it. */
|
|
140
159
|
export def generateImageLocal(
|
|
141
160
|
prompt: string,
|
|
142
161
|
model: string,
|
|
@@ -155,6 +174,7 @@ export def generateImageLocal(
|
|
|
155
174
|
images: string[] = [],
|
|
156
175
|
startImage: string = "",
|
|
157
176
|
strength: number | null = null,
|
|
177
|
+
mask: string = "",
|
|
158
178
|
): Result<LocalImage> raises <std::readImage> {
|
|
159
179
|
"""
|
|
160
180
|
Generate an image on this machine with a local image model. The model
|
|
@@ -194,6 +214,9 @@ export def generateImageLocal(
|
|
|
194
214
|
scaled to cover `size` and the overflow is cropped from both sides, so a
|
|
195
215
|
4:3 photo redrawn as a square loses a strip at its left and right.
|
|
196
216
|
|
|
217
|
+
To redraw only part of the picture, also pass a `mask`: white is
|
|
218
|
+
redrawn and black is kept.
|
|
219
|
+
|
|
197
220
|
@param prompt - What to draw
|
|
198
221
|
@param model - The image model: a catalog name such as "z-image-turbo", a diffusers: URI, or a model directory
|
|
199
222
|
@param size - Width and height joined by "x", each a multiple of 16, such as "1024x1024" or "1344x768". Empty is 1024x1024. Leave it empty when editing or redrawing a picture, and the result keeps the picture's shape
|
|
@@ -210,7 +233,8 @@ export def generateImageLocal(
|
|
|
210
233
|
@param invertControlImage - Swap black and white in the drawing before using it. Set it for dark lines on white with the scribble ControlNet, which reads white lines on black
|
|
211
234
|
@param images - Pictures to edit, as paths to files on this machine. The prompt says what to change: 'add a hat to the character'. Only FLUX.2 [klein] takes them, and at most 4
|
|
212
235
|
@param startImage - A picture to redraw, as a path to a file on this machine. The layout stays and the style changes. Goes with strength. Every model but FLUX.2 [klein] takes one
|
|
213
|
-
@param strength - How much of startImage to redraw, above 0 and up to 1. Low keeps it close, high changes more. Null uses the model's default
|
|
236
|
+
@param strength - How much of startImage to redraw, above 0 and up to 1. Low keeps it close, high changes more. Null uses the model's default, which can change when a mask is added: for Z-Image Turbo it goes from 0.6 to 1.0, which redraws the white part from scratch
|
|
237
|
+
@param mask - A black-and-white picture the same size as startImage, as a path to a file on this machine. White is redrawn and black is kept. Goes with startImage. Every model but FLUX.2 [klein] takes one
|
|
214
238
|
"""
|
|
215
239
|
// Every input path is checked before the first interrupt, so a call that
|
|
216
240
|
// is going to fail asks for nothing.
|
|
@@ -222,6 +246,7 @@ export def generateImageLocal(
|
|
|
222
246
|
images,
|
|
223
247
|
startImage,
|
|
224
248
|
strength,
|
|
249
|
+
mask,
|
|
225
250
|
)
|
|
226
251
|
if (isFailure(inputs)) {
|
|
227
252
|
return inputs
|
package/stdlib/image.js
CHANGED
|
@@ -155,7 +155,7 @@ function registerTools(tools) {
|
|
|
155
155
|
}
|
|
156
156
|
}
|
|
157
157
|
}
|
|
158
|
-
__registerModuleFingerprint("stdlib/image.agency", "
|
|
158
|
+
__registerModuleFingerprint("stdlib/image.agency", "0ec6efc3a40477986a951b0956cb9478a563e76095f8e704b5f7aaf12df68210", import.meta.url);
|
|
159
159
|
__registerTool(print);
|
|
160
160
|
__registerTool(printJSON);
|
|
161
161
|
__registerTool(input);
|
|
@@ -559,7 +559,7 @@ const generateImage = __AgencyFunction.create({
|
|
|
559
559
|
exported: true
|
|
560
560
|
}, __toolRegistry);
|
|
561
561
|
const LocalImage = z.object({ "base64": z.string(), "mimeType": z.string(), "seed": z.number() });
|
|
562
|
-
async function __generateImageLocal_impl(prompt, model, size = __UNSET, steps = __UNSET, guidance = __UNSET, seed = __UNSET, negativePrompt = __UNSET, format = __UNSET, lora = __UNSET, loraScale = __UNSET, controlnet = __UNSET, controlImage = __UNSET, controlScale = __UNSET, invertControlImage = __UNSET, images = __UNSET, startImage = __UNSET, strength = __UNSET) {
|
|
562
|
+
async function __generateImageLocal_impl(prompt, model, size = __UNSET, steps = __UNSET, guidance = __UNSET, seed = __UNSET, negativePrompt = __UNSET, format = __UNSET, lora = __UNSET, loraScale = __UNSET, controlnet = __UNSET, controlImage = __UNSET, controlScale = __UNSET, invertControlImage = __UNSET, images = __UNSET, startImage = __UNSET, strength = __UNSET, mask = __UNSET) {
|
|
563
563
|
const __setupData = setupFunction();
|
|
564
564
|
const __stack = __setupData.stack;
|
|
565
565
|
const __step = __setupData.step;
|
|
@@ -589,6 +589,7 @@ async function __generateImageLocal_impl(prompt, model, size = __UNSET, steps =
|
|
|
589
589
|
__stack.args["images"] = images === __UNSET ? [] : images;
|
|
590
590
|
__stack.args["startImage"] = startImage === __UNSET ? `` : startImage;
|
|
591
591
|
__stack.args["strength"] = strength === __UNSET ? null : strength;
|
|
592
|
+
__stack.args["mask"] = mask === __UNSET ? `` : mask;
|
|
592
593
|
__self.__destructiveRan = __self.__destructiveRan ?? false;
|
|
593
594
|
const runner = new Runner(__ctx, __stack, { state: __stack, moduleId: "stdlib/image.agency", scopeName: "generateImageLocal", threads: __setupData.threads });
|
|
594
595
|
let __resultCheckpointId = -1;
|
|
@@ -663,6 +664,10 @@ async function __generateImageLocal_impl(prompt, model, size = __UNSET, steps =
|
|
|
663
664
|
strength = __overrides["strength"];
|
|
664
665
|
__stack.args["strength"] = strength;
|
|
665
666
|
}
|
|
667
|
+
if ("mask" in __overrides) {
|
|
668
|
+
mask = __overrides["mask"];
|
|
669
|
+
__stack.args["mask"] = mask;
|
|
670
|
+
}
|
|
666
671
|
}
|
|
667
672
|
try {
|
|
668
673
|
await agencyStore.run({
|
|
@@ -693,7 +698,8 @@ async function __generateImageLocal_impl(prompt, model, size = __UNSET, steps =
|
|
|
693
698
|
invertControlImage,
|
|
694
699
|
images,
|
|
695
700
|
startImage,
|
|
696
|
-
strength
|
|
701
|
+
strength,
|
|
702
|
+
mask
|
|
697
703
|
},
|
|
698
704
|
moduleId: "stdlib/image.agency"
|
|
699
705
|
}
|
|
@@ -704,7 +710,7 @@ async function __generateImageLocal_impl(prompt, model, size = __UNSET, steps =
|
|
|
704
710
|
await runner.step(2, async (runner2) => {
|
|
705
711
|
__stack.locals.inputs = await __tryCall(async () => await __call(_localImageInputs, {
|
|
706
712
|
type: "positional",
|
|
707
|
-
args: [__stack.args.controlnet, __stack.args.controlImage, __stack.args.controlScale, __stack.args.invertControlImage, __stack.args.images, __stack.args.startImage, __stack.args.strength]
|
|
713
|
+
args: [__stack.args.controlnet, __stack.args.controlImage, __stack.args.controlScale, __stack.args.invertControlImage, __stack.args.images, __stack.args.startImage, __stack.args.strength, __stack.args.mask]
|
|
708
714
|
}), {
|
|
709
715
|
checkpoint: getRuntimeContext().ctx.getResultCheckpoint(),
|
|
710
716
|
functionName: "generateImageLocal",
|
|
@@ -949,6 +955,13 @@ const generateImageLocal = __AgencyFunction.create({
|
|
|
949
955
|
variadic: false,
|
|
950
956
|
isFunctionTyped: false,
|
|
951
957
|
acceptsResult: false
|
|
958
|
+
}, {
|
|
959
|
+
name: "mask",
|
|
960
|
+
hasDefault: true,
|
|
961
|
+
defaultValue: void 0,
|
|
962
|
+
variadic: false,
|
|
963
|
+
isFunctionTyped: false,
|
|
964
|
+
acceptsResult: false
|
|
952
965
|
}],
|
|
953
966
|
toolDefinition: {
|
|
954
967
|
name: "generateImageLocal",
|
|
@@ -989,6 +1002,9 @@ const generateImageLocal = __AgencyFunction.create({
|
|
|
989
1002
|
scaled to cover \`size\` and the overflow is cropped from both sides, so a
|
|
990
1003
|
4:3 photo redrawn as a square loses a strip at its left and right.
|
|
991
1004
|
|
|
1005
|
+
To redraw only part of the picture, also pass a \`mask\`: white is
|
|
1006
|
+
redrawn and black is kept.
|
|
1007
|
+
|
|
992
1008
|
@param prompt - What to draw
|
|
993
1009
|
@param model - The image model: a catalog name such as "z-image-turbo", a diffusers: URI, or a model directory
|
|
994
1010
|
@param size - Width and height joined by "x", each a multiple of 16, such as "1024x1024" or "1344x768". Empty is 1024x1024. Leave it empty when editing or redrawing a picture, and the result keeps the picture's shape
|
|
@@ -1005,8 +1021,9 @@ const generateImageLocal = __AgencyFunction.create({
|
|
|
1005
1021
|
@param invertControlImage - Swap black and white in the drawing before using it. Set it for dark lines on white with the scribble ControlNet, which reads white lines on black
|
|
1006
1022
|
@param images - Pictures to edit, as paths to files on this machine. The prompt says what to change: 'add a hat to the character'. Only FLUX.2 [klein] takes them, and at most 4
|
|
1007
1023
|
@param startImage - A picture to redraw, as a path to a file on this machine. The layout stays and the style changes. Goes with strength. Every model but FLUX.2 [klein] takes one
|
|
1008
|
-
@param strength - How much of startImage to redraw, above 0 and up to 1. Low keeps it close, high changes more. Null uses the model's default
|
|
1009
|
-
|
|
1024
|
+
@param strength - How much of startImage to redraw, above 0 and up to 1. Low keeps it close, high changes more. Null uses the model's default, which can change when a mask is added: for Z-Image Turbo it goes from 0.6 to 1.0, which redraws the white part from scratch
|
|
1025
|
+
@param mask - A black-and-white picture the same size as startImage, as a path to a file on this machine. White is redrawn and black is kept. Goes with startImage. Every model but FLUX.2 [klein] takes one`,
|
|
1026
|
+
schema: z.object({ "prompt": z.string(), "model": z.string(), "size": z.string().nullable().describe("Default: "), "steps": z.union([z.number(), z.null()]).describe("Default: null"), "guidance": z.union([z.number(), z.null()]).describe("Default: null"), "seed": z.union([z.number(), z.null()]).describe("Default: null"), "negativePrompt": z.string().nullable().describe("Default: "), "format": z.string().nullable().describe("Default: png"), "lora": z.string().nullable().describe("Default: "), "loraScale": z.union([z.number(), z.null()]).describe("Default: null"), "controlnet": z.string().nullable().describe("Default: "), "controlImage": z.string().nullable().describe("Default: "), "controlScale": z.union([z.number(), z.null()]).describe("Default: null"), "invertControlImage": z.boolean().nullable().describe("Default: false"), "images": z.array(z.string()).nullable().describe("Default: []"), "startImage": z.string().nullable().describe("Default: "), "strength": z.union([z.number(), z.null()]).describe("Default: null"), "mask": z.string().nullable().describe("Default: ") })
|
|
1010
1027
|
},
|
|
1011
1028
|
exported: true
|
|
1012
1029
|
}, __toolRegistry);
|
|
@@ -1902,7 +1919,7 @@ const pasteImages = __AgencyFunction.create({
|
|
|
1902
1919
|
exported: true
|
|
1903
1920
|
}, __toolRegistry);
|
|
1904
1921
|
var stdin_default = graph;
|
|
1905
|
-
const __sourceMap = { "stdlib/image.agency:generateImage": { "2": { "line": 105, "col": 2 }, "3": { "line": 106, "col": 6 }, "4": { "line": 106, "col": 2 }, "5": { "line": 109, "col": 2 }, "6": { "line": 110, "col": 2 }, "7": { "line": 121, "col": 2 }, "4.0": { "line": 107, "col": 4 }, "6.0.0": { "line": 113, "col": 13 }, "6.0.1": { "line": 114, "col": 18 }, "6.0.2": { "line": 112, "col": 6 }, "6.0": { "line": 111, "col": 4 } }, "stdlib/image.agency:generateImageLocal": { "2": { "line":
|
|
1922
|
+
const __sourceMap = { "stdlib/image.agency:generateImage": { "2": { "line": 105, "col": 2 }, "3": { "line": 106, "col": 6 }, "4": { "line": 106, "col": 2 }, "5": { "line": 109, "col": 2 }, "6": { "line": 110, "col": 2 }, "7": { "line": 121, "col": 2 }, "4.0": { "line": 107, "col": 4 }, "6.0.0": { "line": 113, "col": 13 }, "6.0.1": { "line": 114, "col": 18 }, "6.0.2": { "line": 112, "col": 6 }, "6.0": { "line": 111, "col": 4 } }, "stdlib/image.agency:generateImageLocal": { "2": { "line": 240, "col": 2 }, "3": { "line": 250, "col": 6 }, "4": { "line": 250, "col": 2 }, "5": { "line": 253, "col": 2 }, "6": { "line": 256, "col": 2 }, "4.0": { "line": 251, "col": 4 }, "5.0": { "line": 254, "col": 4 } }, "stdlib/image.agency:cropImage": { "1": { "line": 290, "col": 2 }, "2": { "line": 291, "col": 6 }, "3": { "line": 291, "col": 2 }, "4": { "line": 294, "col": 2 }, "5": { "line": 295, "col": 6 }, "6": { "line": 295, "col": 2 }, "7": { "line": 299, "col": 9 }, "8": { "line": 300, "col": 14 }, "9": { "line": 301, "col": 12 }, "10": { "line": 302, "col": 17 }, "11": { "line": 298, "col": 2 }, "12": { "line": 304, "col": 2 }, "3.0": { "line": 292, "col": 4 }, "6.0": { "line": 296, "col": 4 } }, "stdlib/image.agency:imageSize": { "1": { "line": 313, "col": 2 }, "2": { "line": 314, "col": 6 }, "3": { "line": 314, "col": 2 }, "4": { "line": 318, "col": 9 }, "5": { "line": 319, "col": 14 }, "6": { "line": 317, "col": 2 }, "7": { "line": 321, "col": 2 }, "3.0": { "line": 315, "col": 4 } }, "stdlib/image.agency:pasteImages": { "2": { "line": 341, "col": 2 }, "3": { "line": 342, "col": 2 }, "4": { "line": 349, "col": 2 }, "5": { "line": 350, "col": 6 }, "6": { "line": 350, "col": 2 }, "7": { "line": 355, "col": 12 }, "8": { "line": 356, "col": 17 }, "9": { "line": 353, "col": 2 }, "10": { "line": 358, "col": 2 }, "3.0": { "line": 343, "col": 4 }, "3.1": { "line": 344, "col": 8 }, "3.2.0": { "line": 345, "col": 6 }, "3.2": { "line": 344, "col": 4 }, "3.3": { "line": 347, "col": 4 }, "6.0": { "line": 351, "col": 4 } } };
|
|
1906
1923
|
export {
|
|
1907
1924
|
GeneratedImage,
|
|
1908
1925
|
ImageBox,
|