mohdel 1.3.2 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,42 @@
1
+ /**
2
+ * Coalesce the delta events of a `call()` stream, with the facade's
3
+ * `bufferOpts` rules (`createRealtimeDeltaBuffer`): deltas of one kind
4
+ * accumulate until `maxChars` is reached or `maxMs` has passed since the
5
+ * last flush, checked as each delta arrives — there is no timer. A change
6
+ * of kind, any other event, and the end of the stream flush first, so
7
+ * order is kept and the terminal event is never held back.
8
+ *
9
+ * @module client/coalesce
10
+ */
11
+
12
+ /**
13
+ * @param {AsyncIterable<import('#core/events.js').Event>} events
14
+ * @param {{maxChars?: number, maxMs?: number}} [bufferOpts]
15
+ * @returns {AsyncGenerator<import('#core/events.js').Event>}
16
+ */
17
+ export async function * coalesce (events, { maxChars = 250, maxMs = 10_000 } = {}) {
18
+ let buffer = ''
19
+ let kind = 'message'
20
+ let lastFlush = Date.now()
21
+
22
+ const flush = () => {
23
+ const ev = { type: 'delta', delta: { type: kind, delta: buffer } }
24
+ buffer = ''
25
+ lastFlush = Date.now()
26
+ return ev
27
+ }
28
+
29
+ for await (const ev of events) {
30
+ if (ev.type !== 'delta') {
31
+ if (buffer) yield flush()
32
+ yield ev
33
+ continue
34
+ }
35
+ if (!ev.delta.delta) continue
36
+ if (buffer && ev.delta.type !== kind) yield flush()
37
+ kind = ev.delta.type
38
+ buffer += ev.delta.delta
39
+ if (buffer.length >= maxChars || Date.now() - lastFlush >= maxMs) yield flush()
40
+ }
41
+ if (buffer) yield flush()
42
+ }
@@ -14,6 +14,7 @@
14
14
  */
15
15
 
16
16
  export { call } from './call.js'
17
+ export { coalesce } from './coalesce.js'
17
18
  export { callImage } from './call_image.js'
18
19
  export { callTranscription } from './call_transcription.js'
19
20
  export { callEmbedding } from './call_embedding.js'
@@ -119,8 +119,14 @@
119
119
 
120
120
  /**
121
121
  * @typedef {object} MessagePart
122
- * @property {('text'|'reasoning')} type
123
- * @property {string} text
122
+ * @property {('text'|'reasoning'|'image')} type
123
+ * @property {string} [text]
124
+ * Required on `text` and `reasoning` parts.
125
+ * @property {string} [fileUri]
126
+ * Required on `image` parts. Same schemes and local-read rules as
127
+ * an envelope `images` ref. Accepted on `user` and `tool` messages.
128
+ * @property {string} [mimeType]
129
+ * Required on `image` parts.
124
130
  * @property {('5m'|'1h')} [cache]
125
131
  * Prompt-cache marker. On system parts: a breakpoint at this block.
126
132
  * On non-system parts: opts the whole conversation into prefix
@@ -19,6 +19,8 @@
19
19
  import { getSpec } from './_catalog.js'
20
20
  import { capOutput } from './_output_cap.js'
21
21
  import { classifyProviderError } from './_errors.js'
22
+ import { hasImagePart, loadImage, loadImageParts } from './_images.js'
23
+ import { isTrustedMedia, mediaError, mediaScheme } from './_media.js'
22
24
  import { costFor } from './_pricing.js'
23
25
  import { cancelledDone } from './_cancelled.js'
24
26
  import { catalogKey, bareOf } from '#core/model-id.js'
@@ -75,6 +77,9 @@ const DSML_PARAM_RE = /<\uFF5CDSML\uFF5Cparameter\s+name="([^"]+)"(?:\s+string="
75
77
  * @property {(envelope: any, args: any) => void} [mutateArgs]
76
78
  * Last-mile hook to splice provider-specific fields into the
77
79
  * request (e.g. OpenRouter routing prefs).
80
+ * @property {boolean} [inlineImagesOnly]
81
+ * The provider takes images as data URIs only; an `https://` image
82
+ * is refused before dispatch.
78
83
  */
79
84
 
80
85
  /**
@@ -88,7 +93,17 @@ export async function * runChatCompletions (envelope, client, config, deps = {})
88
93
  const spec = getSpec(catalogKey(envelope.model)) || {}
89
94
  const start = String(process.hrtime.bigint())
90
95
 
91
- const args = buildRequest(envelope, spec, config)
96
+ let images
97
+ try {
98
+ images = await loadChatImages(envelope, config)
99
+ } catch (e) {
100
+ deps.log?.warn({ err: e }, `[mohdel:${config.provider}] image load failed`)
101
+ const typed = /** @type {any} */(e).typed
102
+ yield { type: 'error', error: typed || classifyProviderError(e, envelope.auth?.key, { provider: config.provider }) }
103
+ return
104
+ }
105
+
106
+ const args = buildRequest(envelope, spec, config, images)
92
107
  if (config.mutateArgs) config.mutateArgs(envelope, args)
93
108
 
94
109
  if (config.stream) {
@@ -323,18 +338,65 @@ function finalize ({ envelope, content, toolCalls, usage, finishReason, start, f
323
338
  return done
324
339
  }
325
340
 
341
+ /**
342
+ * @typedef {object} ChatImages
343
+ * @property {Map<object, any>} parts `image` part → rendered `image_url` part
344
+ * @property {any[]} envelope Rendered envelope `images` refs
345
+ */
346
+
347
+ /**
348
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
349
+ * @param {ChatCompletionsConfig} config
350
+ * @returns {Promise<ChatImages>}
351
+ */
352
+ async function loadChatImages (envelope, config) {
353
+ const trusted = isTrustedMedia(envelope)
354
+ const parts = new Map()
355
+ for (const [part, loaded] of await loadImageParts(envelope.prompt, { trusted })) {
356
+ parts.set(part, toChatImagePart(part, loaded, config))
357
+ }
358
+ const rendered = []
359
+ for (const ref of envelope.images ?? []) {
360
+ rendered.push(toChatImagePart(ref, await loadImage(ref, { trusted }), config))
361
+ }
362
+ return { parts, envelope: rendered }
363
+ }
364
+
365
+ /**
366
+ * A `data:` URI goes out exactly as the caller sent it.
367
+ *
368
+ * @param {{fileUri: string}} ref
369
+ * @param {import('./_images.js').LoadedImage} loaded
370
+ * @param {ChatCompletionsConfig} config
371
+ */
372
+ function toChatImagePart (ref, loaded, config) {
373
+ if (loaded.url && config.inlineImagesOnly) {
374
+ throw mediaError(
375
+ `${config.provider} does not accept image URLs`,
376
+ 'SESSION_INVALID_IMAGE',
377
+ 'send the image as a data: URI or a file:// path'
378
+ )
379
+ }
380
+ const url = mediaScheme(ref.fileUri) === 'file'
381
+ ? `data:${loaded.mimeType};base64,${loaded.base64}`
382
+ : ref.fileUri
383
+ return { type: 'image_url', image_url: { url, detail: 'high' } }
384
+ }
385
+
326
386
  /**
327
387
  * @param {import('#core/envelope.js').CallEnvelope} envelope
328
388
  * @param {any} spec
329
389
  * @param {ChatCompletionsConfig} config
390
+ * @param {ChatImages} images
330
391
  */
331
- function buildRequest (envelope, spec, config) {
392
+ function buildRequest (envelope, spec, config, images) {
332
393
  /** @type {Record<string, any>} */
333
394
  const args = {
334
395
  model: spec?.model ?? bareOf(envelope.model),
335
396
  temperature: 0,
336
397
  messages: toChatMessages(envelope.prompt, {
337
- reasoningPad: typeof spec?.reasoningContentPlaceholder === 'string' ? spec.reasoningContentPlaceholder : null
398
+ reasoningPad: typeof spec?.reasoningContentPlaceholder === 'string' ? spec.reasoningContentPlaceholder : null,
399
+ imageParts: images.parts
338
400
  })
339
401
  }
340
402
 
@@ -342,8 +404,8 @@ function buildRequest (envelope, spec, config) {
342
404
  args.max_tokens = envelope.outputBudget
343
405
  }
344
406
 
345
- if (envelope.images?.length) {
346
- injectImages(args, envelope.images)
407
+ if (images.envelope.length) {
408
+ injectImages(args, images.envelope)
347
409
  }
348
410
 
349
411
  if (envelope.tools?.length) {
@@ -393,8 +455,11 @@ function buildRequest (envelope, spec, config) {
393
455
  }
394
456
 
395
457
  /**
458
+ * A `tool` message carries text only, so a tool result's images go
459
+ * into one user message after the run of tool results.
460
+ *
396
461
  * @param {string | import('#core/envelope.js').Message[]} prompt
397
- * @param {{ reasoningPad?: string | null }} [opts]
462
+ * @param {{ reasoningPad?: string | null, imageParts: Map<object, any> }} opts
398
463
  * `reasoningPad`: when a string (including `''`), assistant messages without
399
464
  * extractable reasoning get `reasoning_content: <pad>` so providers that
400
465
  * require the field on every assistant turn (e.g. deepseek-v4-pro) accept
@@ -406,42 +471,74 @@ function buildRequest (envelope, spec, config) {
406
471
  * own adapters and have their own roundtrip rules.
407
472
  * @returns {Array<any>}
408
473
  */
409
- function toChatMessages (prompt, opts = {}) {
474
+ function toChatMessages (prompt, opts) {
410
475
  const pad = typeof opts.reasoningPad === 'string' ? opts.reasoningPad : null
411
476
  if (typeof prompt === 'string') return [{ role: 'user', content: prompt }]
412
- return prompt.map(m => {
477
+ const out = []
478
+ let hoisted = []
479
+ for (const m of prompt) {
413
480
  if (m.role === 'tool') {
414
- return {
415
- role: 'tool',
416
- tool_call_id: m.toolCallId,
417
- content: flattenText(m.content)
481
+ if (Array.isArray(m.content)) {
482
+ hoisted.push(...m.content.filter(p => p.type === 'image').map(p => opts.imageParts.get(p)))
418
483
  }
484
+ } else if (hoisted.length) {
485
+ out.push({ role: 'user', content: hoisted })
486
+ hoisted = []
419
487
  }
420
- const reasoning = m.role === 'assistant' ? extractReasoning(m.content) : null
421
- const reasoningField = reasoning ?? (pad !== null && m.role === 'assistant' ? pad : null)
422
- if (m.role === 'assistant' && m.toolCalls?.length) {
423
- // Chat Completions assistant turn: optional `content` + the
424
- // `tool_calls` array. `arguments` must be a JSON string on
425
- // the wire.
426
- const msg = {
427
- role: 'assistant',
428
- content: flattenText(m.content) || '',
429
- tool_calls: m.toolCalls.map(tc => ({
430
- id: tc.id,
431
- type: 'function',
432
- function: {
433
- name: tc.name,
434
- arguments: stringifyToolArgs(tc.arguments)
435
- }
436
- }))
437
- }
438
- if (reasoningField !== null) msg.reasoning_content = reasoningField
439
- return msg
488
+ out.push(toChatMessage(m, pad, opts.imageParts))
489
+ }
490
+ if (hoisted.length) out.push({ role: 'user', content: hoisted })
491
+ return out
492
+ }
493
+
494
+ /**
495
+ * @param {import('#core/envelope.js').Message} m
496
+ * @param {string | null} pad
497
+ * @param {Map<object, any>} imageParts
498
+ */
499
+ function toChatMessage (m, pad, imageParts) {
500
+ if (m.role === 'tool') {
501
+ return {
502
+ role: 'tool',
503
+ tool_call_id: m.toolCallId,
504
+ content: flattenText(m.content)
505
+ }
506
+ }
507
+ const reasoning = m.role === 'assistant' ? extractReasoning(m.content) : null
508
+ const reasoningField = reasoning ?? (pad !== null && m.role === 'assistant' ? pad : null)
509
+ if (m.role === 'assistant' && m.toolCalls?.length) {
510
+ // Chat Completions assistant turn: optional `content` + the
511
+ // `tool_calls` array. `arguments` must be a JSON string on
512
+ // the wire.
513
+ const msg = {
514
+ role: 'assistant',
515
+ content: flattenText(m.content) || '',
516
+ tool_calls: m.toolCalls.map(tc => ({
517
+ id: tc.id,
518
+ type: 'function',
519
+ function: {
520
+ name: tc.name,
521
+ arguments: stringifyToolArgs(tc.arguments)
522
+ }
523
+ }))
440
524
  }
441
- const msg = { role: m.role, content: flattenText(m.content) }
442
525
  if (reasoningField !== null) msg.reasoning_content = reasoningField
443
526
  return msg
444
- })
527
+ }
528
+ const content = hasImagePart(m.content) ? toChatContent(m.content, imageParts) : flattenText(m.content)
529
+ const msg = { role: m.role, content }
530
+ if (reasoningField !== null) msg.reasoning_content = reasoningField
531
+ return msg
532
+ }
533
+
534
+ /**
535
+ * @param {import('#core/envelope.js').MessagePart[]} content
536
+ * @param {Map<object, any>} imageParts
537
+ */
538
+ function toChatContent (content, imageParts) {
539
+ return content
540
+ .filter(p => p.type === 'image' || (p.type === 'text' && p.text))
541
+ .map(p => p.type === 'image' ? imageParts.get(p) : { type: 'text', text: p.text })
445
542
  }
446
543
 
447
544
  /** @param {unknown} args */
@@ -466,17 +563,9 @@ function extractReasoning (content) {
466
563
 
467
564
  /**
468
565
  * @param {any} args
469
- * @param {import('#core/envelope.js').MediaRef[]} images
566
+ * @param {any[]} blocks Rendered `image_url` parts
470
567
  */
471
- function injectImages (args, images) {
472
- const blocks = images
473
- .filter(i => i?.fileUri && i?.mimeType)
474
- .map(i => ({
475
- type: 'image_url',
476
- image_url: { url: i.fileUri, detail: 'high' }
477
- }))
478
- if (!blocks.length) return
479
-
568
+ function injectImages (args, blocks) {
480
569
  // Append to the last user message; fallback to creating one.
481
570
  for (let i = args.messages.length - 1; i >= 0; i--) {
482
571
  const m = args.messages[i]
@@ -37,13 +37,43 @@ const IMAGE_ERROR = 'SESSION_INVALID_IMAGE'
37
37
  export async function loadImages (images, opts = {}) {
38
38
  if (!images || !Array.isArray(images)) return []
39
39
  const out = []
40
- for (const img of images) {
41
- if (!img?.fileUri || !img?.mimeType) continue
42
- out.push(await loadImage(img, opts))
40
+ for (const img of images) out.push(await loadImage(img, opts))
41
+ return out
42
+ }
43
+
44
+ /**
45
+ * Loads every `image` part in a message prompt, keyed by the part
46
+ * object so a message builder can render each one where it sits.
47
+ *
48
+ * @param {string | import('#core/envelope.js').Message[]} prompt
49
+ * @param {{trusted?: boolean}} [opts]
50
+ * @returns {Promise<Map<object, LoadedImage>>}
51
+ */
52
+ export async function loadImageParts (prompt, opts = {}) {
53
+ const out = new Map()
54
+ if (typeof prompt === 'string') return out
55
+ for (const m of prompt) {
56
+ if (!Array.isArray(m.content)) continue
57
+ for (const p of m.content) {
58
+ if (p.type !== 'image') continue
59
+ if (m.role !== 'user' && m.role !== 'tool') {
60
+ throw mediaError(
61
+ `image parts are not accepted on ${m.role} messages`,
62
+ IMAGE_ERROR,
63
+ 'put images on user or tool messages'
64
+ )
65
+ }
66
+ out.set(p, await loadImage(p, opts))
67
+ }
43
68
  }
44
69
  return out
45
70
  }
46
71
 
72
+ /** @param {string | import('#core/envelope.js').MessagePart[]} content */
73
+ export function hasImagePart (content) {
74
+ return Array.isArray(content) && content.some(p => p.type === 'image')
75
+ }
76
+
47
77
  /**
48
78
  * @param {{fileUri: string, mimeType: string}} image
49
79
  * @param {{trusted?: boolean}} [opts]
@@ -51,6 +81,9 @@ export async function loadImages (images, opts = {}) {
51
81
  */
52
82
  export async function loadImage (image, opts = {}) {
53
83
  const { fileUri, mimeType } = image
84
+ if (!fileUri || !mimeType) {
85
+ throw mediaError('image is missing fileUri or mimeType', IMAGE_ERROR)
86
+ }
54
87
  switch (mediaScheme(fileUri)) {
55
88
  case 'file': {
56
89
  const { bytes } = await readLocalMedia(fileUri, { type: IMAGE_ERROR, trusted: opts.trusted })
@@ -27,7 +27,7 @@ import {
27
27
  import { cancelledDone } from './_cancelled.js'
28
28
  import { getSpec } from './_catalog.js'
29
29
  import { classifyProviderError } from './_errors.js'
30
- import { loadImages } from './_images.js'
30
+ import { hasImagePart, loadImageParts, loadImages } from './_images.js'
31
31
  import { isTrustedMedia } from './_media.js'
32
32
  import { costFor } from './_pricing.js'
33
33
  import { catalogKey, bareOf } from '#core/model-id.js'
@@ -92,22 +92,24 @@ export async function * anthropic (envelope, deps = {}) {
92
92
  const start = String(process.hrtime.bigint())
93
93
  let first = null
94
94
 
95
- const { system, conversation, conversationCacheTtl } = splitPrompt(envelope.prompt)
96
-
97
- // Attach images to the last user message before building the request.
98
- if (envelope.images?.length) {
99
- try {
100
- const loaded = await loadImages(envelope.images, { trusted: isTrustedMedia(envelope) })
101
- const blocks = loaded.map(toAnthropicImageBlock).filter(Boolean)
102
- if (blocks.length) injectImageBlocks(conversation, blocks)
103
- } catch (e) {
104
- log?.warn({ err: e }, '[mohdel:anthropic] image load failed')
105
- const typed = /** @type {any} */(e).typed
106
- yield { type: 'error', error: typed || classifyProviderError(e, envelope.auth?.key, { provider: 'anthropic' }) }
107
- return
95
+ let imageParts
96
+ let imageBlocks = []
97
+ try {
98
+ const trusted = isTrustedMedia(envelope)
99
+ imageParts = await loadImageParts(envelope.prompt, { trusted })
100
+ if (envelope.images?.length) {
101
+ imageBlocks = (await loadImages(envelope.images, { trusted })).map(toAnthropicImageBlock)
108
102
  }
103
+ } catch (e) {
104
+ log?.warn({ err: e }, '[mohdel:anthropic] image load failed')
105
+ const typed = /** @type {any} */(e).typed
106
+ yield { type: 'error', error: typed || classifyProviderError(e, envelope.auth?.key, { provider: 'anthropic' }) }
107
+ return
109
108
  }
110
109
 
110
+ const { system, conversation, conversationCacheTtl } = splitPrompt(envelope.prompt, imageParts)
111
+ if (imageBlocks.length) injectImageBlocks(conversation, imageBlocks)
112
+
111
113
  const request = buildRequest(envelope, conversation, system, conversationCacheTtl)
112
114
 
113
115
  // Array + join, not `+=`: per-delta cons-strings are the cost on a long stream.
@@ -435,7 +437,7 @@ function placeConversationBreakpoints (messages, ttl) {
435
437
  }
436
438
 
437
439
  /** @param {string | import('#core/envelope.js').Message[]} prompt */
438
- function splitPrompt (prompt) {
440
+ function splitPrompt (prompt, imageParts) {
439
441
  if (typeof prompt === 'string') {
440
442
  return { system: '', conversation: [{ role: 'user', content: prompt }], conversationCacheTtl: null }
441
443
  }
@@ -488,7 +490,7 @@ function splitPrompt (prompt) {
488
490
  content: [{
489
491
  type: 'tool_result',
490
492
  tool_use_id: m.toolCallId ?? '',
491
- content: flattenText(m.content)
493
+ content: hasImagePart(m.content) ? toAnthropicContent(m.content, imageParts) : flattenText(m.content)
492
494
  }]
493
495
  })
494
496
  } else if (m.role === 'assistant' && m.toolCalls?.length) {
@@ -509,7 +511,7 @@ function splitPrompt (prompt) {
509
511
  } else {
510
512
  conversation.push({
511
513
  role: m.role,
512
- content: toAnthropicContent(m.content)
514
+ content: toAnthropicContent(m.content, imageParts)
513
515
  })
514
516
  }
515
517
  }
@@ -529,11 +531,15 @@ function flattenText (content) {
529
531
  return content.filter(p => p.type === 'text' && p.text).map(p => p.text).join('\n')
530
532
  }
531
533
 
532
- /** @param {string | import('#core/envelope.js').MessagePart[]} content */
533
- function toAnthropicContent (content) {
534
+ /**
535
+ * @param {string | import('#core/envelope.js').MessagePart[]} content
536
+ * @param {Map<object, import('./_images.js').LoadedImage>} imageParts
537
+ */
538
+ function toAnthropicContent (content, imageParts) {
534
539
  if (typeof content === 'string') return content
535
- return content.map(p => {
540
+ return content.filter(p => p.type !== 'reasoning').map(p => {
536
541
  if (p.type === 'text') return { type: 'text', text: p.text ?? '' }
542
+ if (p.type === 'image') return toAnthropicImageBlock(imageParts.get(p))
537
543
  throw new Error(`unsupported content part type: ${p.type}`)
538
544
  })
539
545
  }
@@ -20,7 +20,8 @@ export async function * cerebras (envelope, deps = {}) {
20
20
  stream: true,
21
21
  provider: 'cerebras',
22
22
  toolChoiceFlavor: 'cerebras',
23
- reasoningField: 'cerebras_zai'
23
+ reasoningField: 'cerebras_zai',
24
+ inlineImagesOnly: true
24
25
  }, {
25
26
  signal: deps.signal,
26
27
  log: deps.log,
@@ -28,7 +28,7 @@ import {
28
28
  import { cancelledDone } from './_cancelled.js'
29
29
  import { getSpec } from './_catalog.js'
30
30
  import { classifyProviderError } from './_errors.js'
31
- import { loadImages } from './_images.js'
31
+ import { loadImageParts, loadImages } from './_images.js'
32
32
  import { isTrustedMedia } from './_media.js'
33
33
  import { loadVideos } from './_videos.js'
34
34
  import { costFor } from './_pricing.js'
@@ -51,21 +51,24 @@ export async function * gemini (envelope, deps = {}) {
51
51
  const start = String(process.hrtime.bigint())
52
52
  let first = null
53
53
 
54
- const { systemInstruction, contents } = buildContents(envelope.prompt)
55
-
56
- if (envelope.images?.length) {
57
- try {
58
- const loaded = await loadImages(envelope.images, { trusted: isTrustedMedia(envelope) })
59
- const parts = loaded.map(toGeminiImagePart).filter(Boolean)
60
- if (parts.length) injectParts(contents, parts)
61
- } catch (e) {
62
- log?.warn({ err: e }, '[mohdel:gemini] image load failed')
63
- const typed = /** @type {any} */(e).typed
64
- yield { type: 'error', error: typed || classifyProviderError(e, envelope.auth?.key, { provider: 'gemini' }) }
65
- return
54
+ let imageParts
55
+ let imageInputs = []
56
+ try {
57
+ const trusted = isTrustedMedia(envelope)
58
+ imageParts = await loadImageParts(envelope.prompt, { trusted })
59
+ if (envelope.images?.length) {
60
+ imageInputs = (await loadImages(envelope.images, { trusted })).map(toGeminiImagePart)
66
61
  }
62
+ } catch (e) {
63
+ log?.warn({ err: e }, '[mohdel:gemini] image load failed')
64
+ const typed = /** @type {any} */(e).typed
65
+ yield { type: 'error', error: typed || classifyProviderError(e, envelope.auth?.key, { provider: 'gemini' }) }
66
+ return
67
67
  }
68
68
 
69
+ const { systemInstruction, contents } = buildContents(envelope.prompt, imageParts)
70
+ if (imageInputs.length) injectParts(contents, imageInputs)
71
+
69
72
  if (envelope.videos?.length) {
70
73
  try {
71
74
  const parts = await loadVideos(envelope.videos, {
@@ -278,7 +281,15 @@ function buildRequest (envelope, contents, systemInstruction) {
278
281
  }
279
282
 
280
283
  /** @param {string | import('#core/envelope.js').Message[]} prompt */
281
- function buildContents (prompt) {
284
+ /**
285
+ * A tool result's images go into one user content after the run of
286
+ * tool results: only Gemini 3 accepts media inside a
287
+ * `functionResponse`.
288
+ *
289
+ * @param {string | import('#core/envelope.js').Message[]} prompt
290
+ * @param {Map<object, import('./_images.js').LoadedImage>} imageParts
291
+ */
292
+ function buildContents (prompt, imageParts) {
282
293
  if (typeof prompt === 'string') {
283
294
  return {
284
295
  systemInstruction: '',
@@ -289,10 +300,20 @@ function buildContents (prompt) {
289
300
  const systemParts = []
290
301
  /** @type {Array<{role: string, parts: any[]}>} */
291
302
  const contents = []
303
+ /** @type {Array<any>} */
304
+ let hoisted = []
305
+ const flushHoisted = () => {
306
+ if (hoisted.length) contents.push({ role: 'user', parts: hoisted })
307
+ hoisted = []
308
+ }
292
309
  for (const m of prompt) {
310
+ if (m.role !== 'tool') flushHoisted()
293
311
  if (m.role === 'system') {
294
312
  systemParts.push(flattenText(m.content))
295
313
  } else if (m.role === 'tool') {
314
+ if (Array.isArray(m.content)) {
315
+ hoisted.push(...m.content.filter(p => p.type === 'image').map(p => toGeminiImagePart(imageParts.get(p))))
316
+ }
296
317
  contents.push({
297
318
  role: 'user',
298
319
  parts: [{
@@ -322,10 +343,11 @@ function buildContents (prompt) {
322
343
  } else {
323
344
  contents.push({
324
345
  role: mapRole(m.role),
325
- parts: toGeminiParts(m.content)
346
+ parts: toGeminiParts(m.content, imageParts)
326
347
  })
327
348
  }
328
349
  }
350
+ flushHoisted()
329
351
  return {
330
352
  systemInstruction: systemParts.filter(Boolean).join('\n\n'),
331
353
  contents
@@ -355,11 +377,15 @@ function flattenText (content) {
355
377
  return content.filter(p => p.type === 'text' && p.text).map(p => p.text).join('\n')
356
378
  }
357
379
 
358
- /** @param {string | import('#core/envelope.js').MessagePart[]} content */
359
- function toGeminiParts (content) {
380
+ /**
381
+ * @param {string | import('#core/envelope.js').MessagePart[]} content
382
+ * @param {Map<object, import('./_images.js').LoadedImage>} imageParts
383
+ */
384
+ function toGeminiParts (content, imageParts) {
360
385
  if (typeof content === 'string') return [{ text: content }]
361
- return content.map(p => {
386
+ return content.filter(p => p.type !== 'reasoning').map(p => {
362
387
  if (p.type === 'text') return { text: p.text ?? '' }
388
+ if (p.type === 'image') return toGeminiImagePart(imageParts.get(p))
363
389
  throw new Error(`unsupported content part type: ${p.type}`)
364
390
  })
365
391
  }
@@ -28,7 +28,7 @@ import {
28
28
  import { cancelledDone } from './_cancelled.js'
29
29
  import { getSpec } from './_catalog.js'
30
30
  import { classifyProviderError } from './_errors.js'
31
- import { loadImages } from './_images.js'
31
+ import { hasImagePart, loadImageParts, loadImages } from './_images.js'
32
32
  import { isTrustedMedia } from './_media.js'
33
33
  import { costFor } from './_pricing.js'
34
34
  import { catalogKey, providerOf, bareOf } from '#core/model-id.js'
@@ -54,21 +54,26 @@ export async function * openai (envelope, deps = {}) {
54
54
  const start = String(process.hrtime.bigint())
55
55
  let first = null
56
56
 
57
- const { instructions, input } = splitPrompt(envelope.prompt)
58
-
59
- if (envelope.images?.length) {
60
- try {
61
- const loaded = await loadImages(envelope.images, { trusted: isTrustedMedia(envelope) })
62
- const parts = loaded.map(toOpenAIImagePart).filter(Boolean)
63
- if (parts.length) injectImageParts(input, parts)
64
- } catch (e) {
65
- log?.warn({ err: e }, '[mohdel:openai] image load failed')
66
- const typed = /** @type {any} */(e).typed
67
- yield { type: 'error', error: typed || classifyProviderError(e, envelope.auth?.key, { provider: 'openai' }) }
68
- return
57
+ let imageParts
58
+ let imageInputs = []
59
+ try {
60
+ const trusted = isTrustedMedia(envelope)
61
+ imageParts = await loadImageParts(envelope.prompt, { trusted })
62
+ if (envelope.images?.length) {
63
+ imageInputs = (await loadImages(envelope.images, { trusted })).map(toOpenAIImagePart)
69
64
  }
65
+ } catch (e) {
66
+ log?.warn({ err: e }, '[mohdel:openai] image load failed')
67
+ const typed = /** @type {any} */(e).typed
68
+ yield { type: 'error', error: typed || classifyProviderError(e, envelope.auth?.key, { provider: 'openai' }) }
69
+ return
70
70
  }
71
71
 
72
+ const { instructions, input } = splitPrompt(envelope.prompt, imageParts, {
73
+ toolResultImages: providerOf(envelope.model) === 'openai'
74
+ })
75
+ if (imageInputs.length) injectImageParts(input, imageInputs)
76
+
72
77
  const request = buildRequest(envelope, input, instructions)
73
78
 
74
79
  // Array + join, not `+=`: per-delta cons-strings are the cost on a long stream.
@@ -358,8 +363,16 @@ function buildRequest (envelope, input, instructions) {
358
363
  return request
359
364
  }
360
365
 
361
- /** @param {string | import('#core/envelope.js').Message[]} prompt */
362
- function splitPrompt (prompt) {
366
+ /**
367
+ * `toolResultImages`: whether a `function_call_output` may carry
368
+ * images. Where it may not, a tool result's images go into one user
369
+ * message after the run of tool results.
370
+ *
371
+ * @param {string | import('#core/envelope.js').Message[]} prompt
372
+ * @param {Map<object, import('./_images.js').LoadedImage>} imageParts
373
+ * @param {{toolResultImages: boolean}} opts
374
+ */
375
+ function splitPrompt (prompt, imageParts, opts) {
363
376
  if (typeof prompt === 'string') {
364
377
  return { instructions: '', input: [{ role: 'user', content: prompt }] }
365
378
  }
@@ -367,14 +380,30 @@ function splitPrompt (prompt) {
367
380
  const systemParts = []
368
381
  /** @type {Array<any>} */
369
382
  const input = []
383
+ /** @type {Array<any>} */
384
+ let hoisted = []
385
+ const flushHoisted = () => {
386
+ if (hoisted.length) input.push({ role: 'user', content: hoisted })
387
+ hoisted = []
388
+ }
370
389
  for (const m of prompt) {
390
+ if (m.role !== 'tool') flushHoisted()
371
391
  if (m.role === 'system') {
372
392
  systemParts.push(flattenText(m.content))
373
393
  } else if (m.role === 'tool') {
394
+ let output = flattenText(m.content)
395
+ if (hasImagePart(m.content)) {
396
+ const content = toInputContent('user', m.content, imageParts)
397
+ if (opts.toolResultImages) {
398
+ output = content
399
+ } else {
400
+ hoisted.push(...content.filter(p => p.type === 'input_image'))
401
+ }
402
+ }
374
403
  input.push({
375
404
  type: 'function_call_output',
376
405
  call_id: m.toolCallId ?? '',
377
- output: flattenText(m.content)
406
+ output
378
407
  })
379
408
  } else if (m.role === 'assistant' && m.toolCalls?.length) {
380
409
  // Responses API wants a message item (if any text) followed
@@ -398,10 +427,11 @@ function splitPrompt (prompt) {
398
427
  } else {
399
428
  input.push({
400
429
  role: m.role,
401
- content: toInputContent(m.role, m.content)
430
+ content: toInputContent(m.role, m.content, imageParts)
402
431
  })
403
432
  }
404
433
  }
434
+ flushHoisted()
405
435
  return { instructions: systemParts.filter(Boolean).join('\n\n'), input }
406
436
  }
407
437
 
@@ -430,12 +460,14 @@ function stringifyToolArgs (args) {
430
460
  /**
431
461
  * @param {string} role
432
462
  * @param {string | import('#core/envelope.js').MessagePart[]} content
463
+ * @param {Map<object, import('./_images.js').LoadedImage>} imageParts
433
464
  */
434
- function toInputContent (role, content) {
465
+ function toInputContent (role, content, imageParts) {
435
466
  if (typeof content === 'string') return content
436
467
  const partType = role === 'assistant' ? 'output_text' : 'input_text'
437
- return content.map(p => {
468
+ return content.filter(p => p.type !== 'reasoning').map(p => {
438
469
  if (p.type === 'text') return { type: partType, text: p.text ?? '' }
470
+ if (p.type === 'image') return toOpenAIImagePart(imageParts.get(p))
439
471
  throw new Error(`unsupported content part type: ${p.type}`)
440
472
  })
441
473
  }
@@ -19,6 +19,7 @@ import { run } from './run.js'
19
19
  import { runImage } from './run_image.js'
20
20
  import { runTranscription } from './run_transcription.js'
21
21
  import { runEmbedding } from './run_embedding.js'
22
+ import { runInfo } from './run_info.js'
22
23
  import { setCatalog } from './adapters/_catalog.js'
23
24
 
24
25
  // Bounded memory for pre-dequeue cancels. Hostile/buggy supervisors
@@ -232,6 +233,13 @@ export async function drive (stdin, stdout) {
232
233
  } else {
233
234
  await writeLine({ type: 'error', error: out.error })
234
235
  }
236
+ } else if (envelope.op === 'info') {
237
+ const out = await runInfo(envelope)
238
+ if (out.ok) {
239
+ await writeLine({ type: 'info_done', result: out.result })
240
+ } else {
241
+ await writeLine({ type: 'error', error: out.error })
242
+ }
235
243
  } else {
236
244
  for await (const ev of run(envelope, { signal: controller.signal })) {
237
245
  await writeLine(ev)
package/js/session/run.js CHANGED
@@ -61,7 +61,7 @@ import { STATUS_INCOMPLETE, WARNING_CANCELLED } from '#core/status.js'
61
61
  // name is checked against the known list before it reaches the import path —
62
62
  // it comes off the envelope, and a computed specifier must never take an
63
63
  // arbitrary string.
64
- const loadAdapter = async (provider) => {
64
+ export const loadAdapter = async (provider) => {
65
65
  if (!ADAPTER_NAMES.includes(provider)) throw new Error(`unknown provider: ${provider}`)
66
66
  const module = await import(`./adapters/${provider}.js`)
67
67
  return module[provider]
@@ -312,7 +312,7 @@ export async function * run (envelope, {
312
312
  * error?: import('#core/events.js').ErrorEvent
313
313
  * }}
314
314
  */
315
- function normalizeModelId (envelope, resolveSpec) {
315
+ export function normalizeModelId (envelope, resolveSpec) {
316
316
  // A bare id that itself contains `:` or `@` is a catalog key in its
317
317
  // own right; resolving the whole string first stops it being split
318
318
  // into a base plus a suffix that was never meant as one.
@@ -374,7 +374,7 @@ function normalizeModelId (envelope, resolveSpec) {
374
374
  * @param {any} adapter
375
375
  * @returns {import('#core/events.js').ErrorEvent | undefined}
376
376
  */
377
- function speedError (key, speed, spec, provider, adapter) {
377
+ export function speedError (key, speed, spec, provider, adapter) {
378
378
  if (!hasSpeed(spec, speed)) {
379
379
  const available = speedNames(spec)
380
380
  const detail = available.length
@@ -0,0 +1,58 @@
1
+ /**
2
+ * Catalog entry lookup for a model id, answered by the session that
3
+ * holds the catalog. An embedder of the gate supplies or loads the
4
+ * catalog but never reads it back; this is how it asks.
5
+ *
6
+ * The answer is the entry a call with that `model` would run on: the
7
+ * same lookup and the same effort and speed-lane checks as `run.js`,
8
+ * so an id this accepts is one a call accepts. Shape follows the
9
+ * facade's `info()`: the entry, plus `speed` when a lane is named.
10
+ *
11
+ * @module session/run_info
12
+ */
13
+
14
+ import { getSpec } from './adapters/_catalog.js'
15
+ import { providerOf } from '#core/model-id.js'
16
+ import { loadAdapter, normalizeModelId, speedError } from './run.js'
17
+
18
+ /**
19
+ * @param {{model: string}} request
20
+ * @param {{
21
+ * resolveSpec?: (key: string) => any,
22
+ * resolveAdapter?: (provider: string) => Promise<any>
23
+ * }} [options]
24
+ * @returns {Promise<
25
+ * | {ok: true, result: Record<string, any> | null}
26
+ * | {ok: false, error: import('#core/errors.js').TypedError}
27
+ * >}
28
+ */
29
+ export async function runInfo ({ model }, {
30
+ resolveSpec = getSpec,
31
+ resolveAdapter = loadAdapter
32
+ } = {}) {
33
+ const norm = normalizeModelId(/** @type {any} */({ model }), resolveSpec)
34
+ if (norm.error) return { ok: false, error: norm.error.error }
35
+ if (!norm.spec) return { ok: true, result: null }
36
+
37
+ const speed = norm.envelope.speed
38
+ if (!speed) return { ok: true, result: { ...norm.spec } }
39
+
40
+ const provider = providerOf(norm.key)
41
+ let adapter
42
+ try {
43
+ adapter = await resolveAdapter(provider)
44
+ } catch (e) {
45
+ return {
46
+ ok: false,
47
+ error: {
48
+ message: e instanceof Error ? e.message : String(e),
49
+ severity: 'error',
50
+ retryable: false,
51
+ type: 'SESSION_UNKNOWN_PROVIDER'
52
+ }
53
+ }
54
+ }
55
+ const denied = speedError(norm.key, speed, norm.spec, provider, adapter)
56
+ if (denied) return { ok: false, error: denied.error }
57
+ return { ok: true, result: { ...norm.spec, speed } }
58
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "1.3.2",
3
+ "version": "1.5.0",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -109,6 +109,7 @@
109
109
  "test:provider": "vitest run test/integration/provider.test.js",
110
110
  "test:multiturn": "vitest run test/integration/multiturn.test.js",
111
111
  "test:vision": "vitest run test/integration/vision.test.js",
112
+ "test:vision:catalog": "vitest run test/integration/vision-catalog.test.js",
112
113
  "test:live": "vitest run test/live"
113
114
  },
114
115
  "release-it": {
@@ -135,17 +136,17 @@
135
136
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
136
137
  "@opentelemetry/sdk-node": "^0.222.0",
137
138
  "chalk": "^6.0.0",
138
- "mohdel-thin-gate-linux-x64-gnu": "1.3.2"
139
+ "mohdel-thin-gate-linux-x64-gnu": "1.5.0"
139
140
  },
140
141
  "dependencies": {
141
- "@anthropic-ai/sdk": "^0.125.0",
142
+ "@anthropic-ai/sdk": "^0.128.0",
142
143
  "@cerebras/cerebras_cloud_sdk": "^1.91.0",
143
144
  "@clack/prompts": "^1.8.1",
144
- "@google/genai": "^2.22.0",
145
+ "@google/genai": "^2.24.0",
145
146
  "@opentelemetry/api": "^1.9.1",
146
147
  "env-paths": "^4.0.0",
147
148
  "groq-sdk": "^1.6.0",
148
- "openai": "^7.15.0",
149
+ "openai": "^7.23.0",
149
150
  "undici": "^7.29.0"
150
151
  },
151
152
  "lint-staged": {
package/src/lib/utils.js CHANGED
@@ -64,7 +64,9 @@ export const createRealtimeDeltaBuffer = (handler, opts = {}) => {
64
64
 
65
65
  const push = (type, delta) => {
66
66
  if (!handler || !delta) return
67
- lastType = type || lastType || 'message'
67
+ const kind = type || lastType
68
+ if (buffer && kind !== lastType) flushInternal(true)
69
+ lastType = kind
68
70
  buffer += delta
69
71
  flushInternal(false)
70
72
  }