@goodandready/dsh-image-gen 0.8.7 → 0.8.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -16,6 +16,7 @@ import z from '@deepseek-ai/schemastery'
16
16
  import { defineTool } from '@deepseek-ai/dsh-tools'
17
17
  import { credentialRef } from '@deepseek-ai/dsh-credentials'
18
18
  import { mkdir, readFile, writeFile } from 'node:fs/promises'
19
+ import os from 'node:os'
19
20
  import path from 'node:path'
20
21
  import {
21
22
  IMAGE_SIZES,
@@ -128,6 +129,22 @@ export const Config = z.object({
128
129
  .number()
129
130
  .description('provider=local: CFG scale.')
130
131
  .default(7),
132
+ seedreamKeyEnv: z
133
+ .string()
134
+ .description('provider=seedream: credential reference / env var holding the API key.')
135
+ .default('SEEDREAM_API_KEY'),
136
+ seedreamModel: z
137
+ .string()
138
+ .description('provider=seedream: model id, e.g. seedream-4.0.')
139
+ .default('seedream-4.0'),
140
+ geminiKeyEnv: z
141
+ .string()
142
+ .description('provider=gemini: credential reference / env var holding the Google API key.')
143
+ .default('GEMINI_API_KEY'),
144
+ geminiModel: z
145
+ .string()
146
+ .description('provider=gemini: model id, e.g. gemini-2.0-flash-exp-image-generation.')
147
+ .default('gemini-2.0-flash-exp-image-generation'),
131
148
  outputDir: z
132
149
  .string()
133
150
  .description('Where generated images are saved. A relative path resolves against the session working directory; an absolute path is used as given.')
@@ -148,6 +165,14 @@ export const Config = z.object({
148
165
  .number()
149
166
  .description('Only enhance prompts shorter than this many characters.')
150
167
  .default(200),
168
+ stylePreset: z
169
+ .string()
170
+ .description('Optional style suffix appended to the prompt before generation. Empty (default) means no style is applied and the prompt is used as-is.')
171
+ .default(''),
172
+ cacheBySeed: z
173
+ .boolean()
174
+ .description('If the same seed+prompt was already generated (present in history), return the cached result instead of generating again. Off by default.')
175
+ .default(false),
151
176
  })
152
177
 
153
178
  /** Keep a file stem safe for the filesystem. */
@@ -214,8 +239,52 @@ function migrateLegacySettings(sctx, scope) {
214
239
  }
215
240
  }
216
241
 
217
- /** In-memory история генераций (пополняется в execute, где доступен cwd). */
218
- const history = []
242
+ /** Каталог истории: общий для всех устройств, переживает рестарт. */
243
+ export function historyFile() {
244
+ return path.join(process.env.DSH_HOME || path.join(os.homedir(), '.dsh'), 'dsh-image-gen', 'history.json')
245
+ }
246
+
247
+ /** Прочитать историю из файла (пусто, если файла нет). */
248
+ /** Найти запись истории по seed+prompt, если файл ещё существует. */
249
+ export async function findCached(entries, seed, prompt) {
250
+ if (seed === undefined) return undefined
251
+ return entries.find((e) => e.seed === seed && e.prompt === prompt)
252
+ }
253
+
254
+ /** Вернуть запись из кэша, если файл существует; иначе undefined. */
255
+ export async function cachedResult(entry) {
256
+ if (!entry || !entry.path) return undefined
257
+ try { require('node:fs').accessSync(entry.path) } catch (e) { return undefined }
258
+ return {
259
+ path: entry.path,
260
+ url: entry.attachmentId ? `/dsh-image-gen/image?id=${encodeURIComponent(entry.attachmentId)}` : '',
261
+ width: 0,
262
+ height: 0,
263
+ seed: entry.seed,
264
+ prompt: entry.prompt,
265
+ format: 'png',
266
+ cached: true,
267
+ attachment: entry.attachmentId ? { attachmentId: entry.attachmentId, mediaType: 'image/png', bytes: 0, width: 0, height: 0, name: '' } : undefined,
268
+ }
269
+ }
270
+
271
+ export async function readHistory() {
272
+ try {
273
+ const raw = await readFile(historyFile(), 'utf-8')
274
+ const parsed = JSON.parse(raw)
275
+ return Array.isArray(parsed) ? parsed : []
276
+ } catch (e) {
277
+ return []
278
+ }
279
+ }
280
+
281
+ /** Записать историю в файл (перезапись целиком). */
282
+ export async function writeHistory(entries) {
283
+ try {
284
+ await mkdir(path.dirname(historyFile()), { recursive: true })
285
+ await writeFile(historyFile(), JSON.stringify(entries, null, 2))
286
+ } catch (e) { /* история не критична */ }
287
+ }
219
288
 
220
289
  /** Отфильтровать записи, чьи файлы ещё существуют; новые первыми. */
221
290
  export function filterHistory(entries, exists) {
@@ -349,8 +418,13 @@ export function apply(ctx, config) {
349
418
  return
350
419
  }
351
420
  const exists = (p) => { try { return require('node:fs').existsSync(p) } catch { return false } }
421
+ const entries = await readHistory()
422
+ const withThumbs = filterHistory(entries, exists).map((e) => ({
423
+ ...e,
424
+ thumbnailUrl: e.attachmentId ? `/dsh-image-gen/image?id=${encodeURIComponent(e.attachmentId)}` : '',
425
+ }))
352
426
  res.writeHead(200, { 'Content-Type': 'application/json' })
353
- res.end(JSON.stringify(filterHistory(history, exists)))
427
+ res.end(JSON.stringify(withThumbs))
354
428
  },
355
429
  }), 'dsh-image-gen: history route')
356
430
 
@@ -390,6 +464,11 @@ export function apply(ctx, config) {
390
464
  type: 'integer',
391
465
  description: 'Number of variations to generate in one call, 1-4 (default 1). Cost scales with count.',
392
466
  },
467
+ prompts: {
468
+ type: 'array',
469
+ items: { type: 'string' },
470
+ description: 'Optional list of prompts; one image is generated per prompt. When set, overrides count and the single prompt.',
471
+ },
393
472
  negative_prompt: {
394
473
  type: 'string',
395
474
  description: 'Optional text describing what NOT to draw; sent to providers that support it (ignored otherwise).',
@@ -398,6 +477,11 @@ export function apply(ctx, config) {
398
477
  type: 'number',
399
478
  description: 'Optional prompt adherence (e.g. 1-20); sent to providers that support it (ignored otherwise).',
400
479
  },
480
+ quality: {
481
+ type: 'string',
482
+ enum: ['low', 'medium', 'high'],
483
+ description: 'Optional quality for providers that support it (seedream, gemini). Ignored otherwise.',
484
+ },
401
485
  source_image: {
402
486
  type: 'string',
403
487
  description: 'Optional path to an image file or an attachment id (sha256:...) to edit instead of drawing from scratch. Only providers that support image editing accept it; others refuse with a clear reason.',
@@ -488,22 +572,23 @@ export function apply(ctx, config) {
488
572
  const source = args.source_image ? await resolveSource(ctx, exec, args.source_image) : undefined
489
573
  const mask = args.mask ? await resolveSource(ctx, exec, args.mask) : undefined
490
574
  const enhanced = await enhancePrompt(ctx, cfg, args.prompt, exec.signal)
491
- const effectivePrompt = enhanced.prompt
575
+ const styleSuffix = cfg.stylePreset ? `, ${cfg.stylePreset}` : ''
576
+ const effectivePrompt = enhanced.prompt + styleSuffix
492
577
  const providers = makeProviders(
493
578
  { fetchImpl: fetch, resolveKey: (ref) => resolveApiKey(ctx, ref), cfg, subscriptionImages },
494
- { prompt: effectivePrompt, size, format, seed: args.seed, signal: exec.signal, negativePrompt: args.negative_prompt, guidanceScale: args.guidance_scale, source, mask, strength: args.strength },
579
+ { prompt: effectivePrompt, size, format, seed: args.seed, signal: exec.signal, negativePrompt: args.negative_prompt, guidanceScale: args.guidance_scale, source, mask, strength: args.strength, quality: args.quality },
495
580
  )
496
581
  // Подписочные провайдеры (codex/grok) отдают {ok:false, reason} вместо исключения:
497
582
  // отказ должен дойти до модели текстом. Без проверки execute шёл дальше с пустыми
498
583
  // байтами, и пользователь получал битую карточку вместо внятного отказа.
499
584
  const guard = (generated) => { if (generated && generated.ok === false) throw new Error(generated.reason) }
500
- const one = async (jobSeed, providerKey) => {
501
- const gen = await providers[providerKey](jobSeed)
585
+ const one = async (jobSeed, providerKey, promptArg = effectivePrompt) => {
586
+ const gen = await providers[providerKey](jobSeed, promptArg)
502
587
  guard(gen)
503
588
  const bytes = gen.bytes
504
589
  const mediaType = gen.mediaType
505
590
  const extension = mediaType === 'image/jpeg' ? 'jpg' : mediaType === 'image/webp' ? 'webp' : 'png'
506
- const stem = `${slugify(args.output_name || effectivePrompt)}-${Date.now().toString(36)}-${jobSeed}`
591
+ const stem = `${slugify(args.output_name || promptArg)}-${Date.now().toString(36)}-${jobSeed}`
507
592
  const name = `${stem}.${extension}`
508
593
 
509
594
  const attachment = await ctx.attachments.saveImage({
@@ -529,7 +614,7 @@ export function apply(ctx, config) {
529
614
  await writeFile(
530
615
  path.join(outDir, `${stem}.json`),
531
616
  JSON.stringify(buildSidecar({
532
- prompt: effectivePrompt,
617
+ prompt: promptArg,
533
618
  size,
534
619
  format,
535
620
  seed: gen.seed,
@@ -544,17 +629,20 @@ export function apply(ctx, config) {
544
629
  }), null, 2),
545
630
  )
546
631
 
547
- history.unshift({
632
+ const entry = {
548
633
  path: filePath,
549
- prompt: effectivePrompt,
634
+ prompt: promptArg,
550
635
  provider,
551
636
  size,
552
637
  format,
553
638
  seed: gen.seed,
554
639
  createdAt: new Date().toISOString(),
555
640
  attachmentId: attachment.attachmentId,
556
- })
557
- if (history.length > (cfg.historyLimit || 50)) history.length = cfg.historyLimit || 50
641
+ }
642
+ const current = await readHistory()
643
+ current.unshift(entry)
644
+ if (current.length > (cfg.historyLimit || 50)) current.length = cfg.historyLimit || 50
645
+ await writeHistory(current)
558
646
 
559
647
  return {
560
648
  path: filePath,
@@ -562,7 +650,7 @@ export function apply(ctx, config) {
562
650
  width: gen.width || attachment.width,
563
651
  height: gen.height || attachment.height,
564
652
  seed: gen.seed,
565
- prompt: effectivePrompt,
653
+ prompt: promptArg,
566
654
  originalPrompt: enhanced.enhanced ? args.prompt : undefined,
567
655
  cost: gen.cost,
568
656
  format: mediaType.replace('image/', ''),
@@ -577,15 +665,25 @@ export function apply(ctx, config) {
577
665
  }
578
666
  }
579
667
 
580
- const count = normalizeCount(args.count)
581
668
  const seedBase = args.seed ?? Math.floor(Math.random() * 100000)
582
669
  // Fallback-цепочка: пробуем текущий провайдер, при отказе — следующий
583
670
  // по порядку (fal → custom → codex → grok), собирая причины отказов.
584
671
  const order = fallbackOrder(provider)
585
- const generators = Object.fromEntries(PROVIDER_KEYS.map((k) => [k, (s) => one(s, k)]))
672
+ const generators = Object.fromEntries(PROVIDER_KEYS.map((k) => [k, (s, p) => one(s, k, p)]))
586
673
  const images = []
587
- for (let i = 0; i < count; i += 1) {
588
- images.push(await tryGenerate(generators, order, seedBase + i))
674
+ const historyEntries = cfg.cacheBySeed ? await readHistory() : []
675
+ const batchPrompts = Array.isArray(args.prompts) && args.prompts.length ? args.prompts : null
676
+ if (batchPrompts) {
677
+ for (let i = 0; i < batchPrompts.length; i += 1) {
678
+ const cached = cfg.cacheBySeed ? await cachedResult(findCached(historyEntries, seedBase + i, batchPrompts[i])) : undefined
679
+ images.push(cached || await tryGenerate(generators, order, seedBase + i, batchPrompts[i]))
680
+ }
681
+ } else {
682
+ const count = normalizeCount(args.count)
683
+ for (let i = 0; i < count; i += 1) {
684
+ const cached = cfg.cacheBySeed ? await cachedResult(findCached(historyEntries, seedBase + i, effectivePrompt)) : undefined
685
+ images.push(cached || await tryGenerate(generators, order, seedBase + i))
686
+ }
589
687
  }
590
688
  const first = images[0]
591
689
  return { ...first, images }
package/lib/providers.js CHANGED
@@ -7,7 +7,7 @@
7
7
  // Сеть приходит параметром (fetchImpl), ключ — через resolveKey, поэтому оба
8
8
  // провайдера проверяются юнит-тестами без единого реального запроса.
9
9
 
10
- export const PROVIDER_KEYS = ['fal', 'custom', 'codex', 'grok', 'local']
10
+ export const PROVIDER_KEYS = ['fal', 'custom', 'codex', 'grok', 'local', 'seedream', 'gemini']
11
11
 
12
12
  /** Привести count из аргумента инструмента к диапазону 1..4. */
13
13
  /** Порядок провайдеров для fallback: основной первым, остальные по PROVIDER_KEYS. */
@@ -21,11 +21,11 @@ export function fallbackOrder(primary) {
21
21
  * @param generators - массив функций (key, seed) => Promise<generated>.
22
22
  * @param order - порядок ключей, длина = generators.length.
23
23
  */
24
- export async function tryGenerate(generators, order, seed) {
24
+ export async function tryGenerate(generators, order, seed, promptArg) {
25
25
  const refusals = []
26
26
  for (const key of order) {
27
27
  try {
28
- const produced = await generators[key](seed)
28
+ const produced = await generators[key](seed, promptArg)
29
29
  return produced
30
30
  } catch (e) {
31
31
  refusals.push(`${key}: ${e.message || String(e)}`)
@@ -179,11 +179,11 @@ export function buildEditForm({ source, mask, prompt, size, strength }) {
179
179
 
180
180
  export function makeProviders(deps, job) {
181
181
  const { fetchImpl, resolveKey, cfg } = deps
182
- const { prompt, size, format, seed, signal, negativePrompt, guidanceScale, source, mask, strength } = job
182
+ const { prompt, size, format, seed, signal, negativePrompt, guidanceScale, source, mask, strength, quality } = job
183
183
 
184
- async function fal(seedArg = seed) {
184
+ async function fal(seedArg = seed, promptArg = prompt) {
185
185
  const key = await resolveKey(cfg.apiKeyEnv)
186
- const body = { prompt, image_size: size, num_images: 1 }
186
+ const body = { prompt: promptArg, image_size: size, num_images: 1 }
187
187
  if (seedArg !== undefined) body.seed = seedArg
188
188
  if (negativePrompt !== undefined) body.negative_prompt = negativePrompt
189
189
  if (guidanceScale !== undefined) body.guidance_scale = guidanceScale
@@ -218,7 +218,7 @@ export function makeProviders(deps, job) {
218
218
  }
219
219
 
220
220
  // Любой OpenAI-совместимый API картинок. Один запрос вместо очереди FAL.
221
- async function custom(seedArg = seed) {
221
+ async function custom(seedArg = seed, promptArg = prompt) {
222
222
  const base = String(cfg.customBaseURL || '').replace(/\/+$/, '')
223
223
  if (!base) throw new Error('Custom image provider: base URL is not configured (Settings → Image generation)')
224
224
  if (!cfg.customModel) throw new Error('Custom image provider: model is not configured')
@@ -236,10 +236,10 @@ export function makeProviders(deps, job) {
236
236
  method: 'POST',
237
237
  headers,
238
238
  body: source
239
- ? buildEditForm({ source, mask, prompt, size: cfg.customSize || SIZE_PIXELS[size] || size, strength })
239
+ ? buildEditForm({ source, mask, prompt: promptArg, size: cfg.customSize || SIZE_PIXELS[size] || size, strength })
240
240
  : JSON.stringify({
241
241
  model: cfg.customModel,
242
- prompt,
242
+ prompt: promptArg,
243
243
  n: 1,
244
244
  size: cfg.customSize || SIZE_PIXELS[size] || size,
245
245
  ...(negativePrompt !== undefined ? { negative_prompt: negativePrompt } : {}),
@@ -291,7 +291,7 @@ export function makeProviders(deps, job) {
291
291
  // только запрос и разбор ответа. Токен сюда не попадает вовсе — служба
292
292
  // отдаёт готовую картинку, а не ключ доступа.
293
293
  function subscription(provider) {
294
- return async function generate(seedArg = seed) {
294
+ return async function generate(seedArg = seed, promptArg = prompt) {
295
295
  if (source) {
296
296
  return { ok: false, provider, reason: `${provider}: не умеет править изображения — используйте fal, custom или local` }
297
297
  }
@@ -307,7 +307,7 @@ export function makeProviders(deps, job) {
307
307
  try {
308
308
  produced = await images.generate({
309
309
  provider,
310
- prompt,
310
+ prompt: promptArg,
311
311
  size: SUBSCRIPTION_SIZES[size] || '1024x1024',
312
312
  quality: cfg.subscriptionQuality || undefined,
313
313
  signal,
@@ -333,7 +333,7 @@ export function makeProviders(deps, job) {
333
333
  }
334
334
 
335
335
  // Локальная генерация: ComfyUI (очередь + опрос) или Automatic1111 (txt2img).
336
- async function local(seedArg = seed) {
336
+ async function local(seedArg = seed, promptArg = prompt) {
337
337
  const base = String(cfg.localBaseURL || '').replace(/\/+$/, '')
338
338
  if (!base) throw new Error('Local image provider: server address is not configured (Settings → Image generation)')
339
339
  const [width, height] = sizeToPixels(size)
@@ -341,7 +341,7 @@ export function makeProviders(deps, job) {
341
341
 
342
342
  if (kind === 'a1111') {
343
343
  const body = {
344
- prompt,
344
+ prompt: promptArg,
345
345
  negative_prompt: negativePrompt,
346
346
  width,
347
347
  height,
@@ -379,7 +379,7 @@ export function makeProviders(deps, job) {
379
379
  '3': { class_type: 'KSampler', inputs: { seed: seedArg ?? 0, steps: cfg.localSteps ?? 20, cfg: cfg.localCfg ?? 7, sampler_name: 'euler', scheduler: 'normal', denoise: 1, model: ['4', 0], positive: ['6', 0], negative: ['7', 0], latent_image: ['5', 0] } },
380
380
  '4': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: cfg.localModel || 'v1-5-pruned-emaonly.safetensors' } },
381
381
  '5': { class_type: 'EmptyLatentImage', inputs: { width, height, batch_size: 1 } },
382
- '6': { class_type: 'CLIPTextEncode', inputs: { text: prompt, clip: ['4', 1] } },
382
+ '6': { class_type: 'CLIPTextEncode', inputs: { text: promptArg, clip: ['4', 1] } },
383
383
  '7': { class_type: 'CLIPTextEncode', inputs: { text: negativePrompt || '', clip: ['4', 1] } },
384
384
  '8': { class_type: 'VAEDecode', inputs: { samples: ['3', 0], vae: ['4', 2] } },
385
385
  '9': { class_type: 'SaveImage', inputs: { filename_prefix: 'dsh', images: ['8', 0] } },
@@ -429,5 +429,61 @@ export function makeProviders(deps, job) {
429
429
  }
430
430
  }
431
431
 
432
- return { fal, custom, codex: subscription('codex'), grok: subscription('grok'), local }
432
+ // Seedream (ByteDance): OpenAI-совместимый API картинок.
433
+ async function seedream(seedArg = seed, promptArg = prompt) {
434
+ const key = await resolveKey(cfg.seedreamKeyEnv)
435
+ const base = 'https://api.bytedanceapi.com/v1'
436
+ const res = await fetchImpl(`${base}/images/generations`, {
437
+ method: 'POST',
438
+ headers: { 'Content-Type': 'application/json', Authorization: `Bearer ${key}` },
439
+ body: JSON.stringify({
440
+ model: cfg.seedreamModel || 'seedream-4.0',
441
+ prompt: promptArg,
442
+ n: 1,
443
+ size: SIZE_PIXELS[size] || size,
444
+ ...(negativePrompt !== undefined ? { negative_prompt: negativePrompt } : {}),
445
+ ...(quality !== undefined ? { quality } : {}),
446
+ }),
447
+ signal,
448
+ })
449
+ const data = await res.json().catch(() => ({}))
450
+ if (!res.ok) {
451
+ const detail = data?.error?.message || JSON.stringify(data).slice(0, 600)
452
+ throw new Error(`Seedream failed (HTTP ${res.status}): ${detail}`)
453
+ }
454
+ const item = data?.data?.[0]
455
+ if (!item) throw new Error('Seedream returned no images')
456
+ if (item.b64_json) {
457
+ return { bytes: Buffer.from(item.b64_json, 'base64'), mediaType: normalizeMediaType('image/png', format), width: 0, height: 0, seed: seedArg ?? 0, sourceUrl: '' }
458
+ }
459
+ if (!item.url) throw new Error('Seedream returned neither b64_json nor url')
460
+ const dl = await fetchImpl(item.url, { signal })
461
+ if (!dl.ok) throw new Error(`Seedream download failed (HTTP ${dl.status})`)
462
+ return { bytes: Buffer.from(await dl.arrayBuffer()), mediaType: normalizeMediaType('image/png', format), width: 0, height: 0, seed: seedArg ?? 0, sourceUrl: item.url }
463
+ }
464
+
465
+ // Gemini (Google): images.generate через GenAI API.
466
+ async function gemini(seedArg = seed, promptArg = prompt) {
467
+ const key = await resolveKey(cfg.geminiKeyEnv)
468
+ const model = cfg.geminiModel || 'gemini-2.0-flash-exp-image-generation'
469
+ const res = await fetchImpl(`https://generativelanguage.googleapis.com/v1beta/models/${model}:generateContent?key=${encodeURIComponent(key)}`, {
470
+ method: 'POST',
471
+ headers: { 'Content-Type': 'application/json' },
472
+ body: JSON.stringify({
473
+ contents: [{ parts: [{ text: promptArg }] }],
474
+ generationConfig: { responseModalities: ['IMAGE'], ...(quality !== undefined ? { imageConfig: { imageQuality: quality } } : {}) },
475
+ }),
476
+ signal,
477
+ })
478
+ const data = await res.json().catch(() => ({}))
479
+ if (!res.ok) {
480
+ const detail = data?.error?.message || JSON.stringify(data).slice(0, 600)
481
+ throw new Error(`Gemini failed (HTTP ${res.status}): ${detail}`)
482
+ }
483
+ const part = data?.candidates?.[0]?.content?.parts?.find((p) => p.inlineData?.data)
484
+ if (!part) throw new Error('Gemini returned no image')
485
+ return { bytes: Buffer.from(part.inlineData.data, 'base64'), mediaType: normalizeMediaType(part.inlineData.mimeType || 'image/png', format), width: 0, height: 0, seed: seedArg ?? 0, sourceUrl: '' }
486
+ }
487
+
488
+ return { fal, custom, codex: subscription('codex'), grok: subscription('grok'), local, seedream, gemini }
433
489
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@goodandready/dsh-image-gen",
3
- "version": "0.8.7",
3
+ "version": "0.8.8",
4
4
  "description": "Image generation for DeepSeek Harness: a generate_image tool with pluggable providers — the FAL queue, any OpenAI-compatible images API, or a ChatGPT/Grok subscription with no API key at all. The picture is shown inline in the conversation; the model receives either a link (works with any chat model) or the image itself (needs dsh-vision-bridge or a vision-capable model).",
5
5
  "keywords": [
6
6
  "deepseek-harness",