@animalabs/membrane 0.5.83 → 0.5.85

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/dist/formatters/openai-responses.js +1 -1
  2. package/dist/formatters/openai-responses.js.map +1 -1
  3. package/dist/membrane.d.ts +3 -1
  4. package/dist/membrane.d.ts.map +1 -1
  5. package/dist/membrane.js +6 -2
  6. package/dist/membrane.js.map +1 -1
  7. package/dist/providers/anthropic.d.ts +11 -1
  8. package/dist/providers/anthropic.d.ts.map +1 -1
  9. package/dist/providers/anthropic.js +67 -10
  10. package/dist/providers/anthropic.js.map +1 -1
  11. package/dist/providers/credentials.d.ts +20 -0
  12. package/dist/providers/credentials.d.ts.map +1 -0
  13. package/dist/providers/credentials.js +53 -0
  14. package/dist/providers/credentials.js.map +1 -0
  15. package/dist/providers/index.d.ts +1 -0
  16. package/dist/providers/index.d.ts.map +1 -1
  17. package/dist/providers/openai-responses-api.d.ts +22 -4
  18. package/dist/providers/openai-responses-api.d.ts.map +1 -1
  19. package/dist/providers/openai-responses-api.js +107 -42
  20. package/dist/providers/openai-responses-api.js.map +1 -1
  21. package/dist/providers/openai-responses.d.ts +61 -5
  22. package/dist/providers/openai-responses.d.ts.map +1 -1
  23. package/dist/providers/openai-responses.js +133 -22
  24. package/dist/providers/openai-responses.js.map +1 -1
  25. package/dist/providers/responses-input.d.ts +11 -0
  26. package/dist/providers/responses-input.d.ts.map +1 -0
  27. package/dist/providers/responses-input.js +128 -0
  28. package/dist/providers/responses-input.js.map +1 -0
  29. package/dist/providers/utils.d.ts +7 -3
  30. package/dist/providers/utils.d.ts.map +1 -1
  31. package/dist/providers/utils.js +28 -3
  32. package/dist/providers/utils.js.map +1 -1
  33. package/dist/types/provider.d.ts +4 -0
  34. package/dist/types/provider.d.ts.map +1 -1
  35. package/package.json +1 -1
  36. package/src/formatters/openai-responses.ts +1 -1
  37. package/src/membrane.ts +6 -2
  38. package/src/providers/anthropic.ts +79 -13
  39. package/src/providers/credentials.ts +77 -0
  40. package/src/providers/index.ts +2 -0
  41. package/src/providers/openai-responses-api.ts +118 -43
  42. package/src/providers/openai-responses.ts +174 -25
  43. package/src/providers/responses-input.ts +131 -0
  44. package/src/providers/utils.ts +24 -3
  45. package/src/types/provider.ts +5 -0
@@ -12,7 +12,8 @@
12
12
  * - If text-only → uses /generations
13
13
  *
14
14
  * Both endpoints:
15
- * - Take a single `prompt` string (not conversation messages)
15
+ * - Take a single `prompt` string (not conversation messages) — see
16
+ * `promptFormat` for how the conversation is flattened into it
16
17
  * - Return base64-encoded images in `data[].b64_json`
17
18
  * - No streaming support (returns complete image)
18
19
  * - Support `size`, `quality`, `n`, `background`, `output_format`
@@ -68,6 +69,13 @@ interface ImagesEditRequest {
68
69
 
69
70
  type ImagesRequest = ImagesGenerateRequest | ImagesEditRequest;
70
71
 
72
+ /** A base64 image found in the conversation, with its position for transcript references. */
73
+ interface ImageRef {
74
+ dataUrl: string;
75
+ msgIndex: number;
76
+ blockIndex: number;
77
+ }
78
+
71
79
  interface ImagesResponseData {
72
80
  b64_json?: string;
73
81
  url?: string;
@@ -115,8 +123,49 @@ export interface OpenAIResponsesAdapterConfig {
115
123
  * When false, always uses /v1/images/generations (text-only).
116
124
  */
117
125
  allowImageEditing?: boolean;
126
+
127
+ /**
128
+ * How the conversation is flattened into the single Images API `prompt`.
129
+ *
130
+ * - `'legacy'` (default): system prompt followed by `User:` / `Assistant:`
131
+ * text lines. Images are attached in conversation order but never
132
+ * mentioned in the prompt, and an image-only assistant message (the
133
+ * typical shape of a previous generation) leaves no trace in the text.
134
+ * - `'transcript'`: role-delimited turns (`### User` / `### Assistant`).
135
+ * Every attached image — user-provided or generated by the assistant in
136
+ * an earlier turn — is numbered in attachment order and referenced
137
+ * inline as `[Image N]` inside the turn it belongs to, so previous
138
+ * generations read as the assistant's own prior turns. The prompt ends
139
+ * with an open `### Assistant` header for the turn being generated.
140
+ */
141
+ promptFormat?: 'legacy' | 'transcript';
142
+
143
+ /**
144
+ * Maximum images attached to /v1/images/edits (API cap: 16). `'legacy'`
145
+ * keeps the first N in conversation order; `'transcript'` keeps the most
146
+ * recent N and marks the rest `[Image omitted]`.
147
+ */
148
+ maxInputImages?: number;
149
+
150
+ /**
151
+ * Text placed between the system prompt and the transcript in
152
+ * `'transcript'` mode, explaining the turn delimiters and image numbering
153
+ * to the model. Defaults to {@link DEFAULT_TRANSCRIPT_PREAMBLE}.
154
+ */
155
+ transcriptPreamble?: string;
118
156
  }
119
157
 
158
+ /** API cap on images per /v1/images/edits request. */
159
+ export const IMAGES_EDIT_MAX_INPUT_IMAGES = 16;
160
+
161
+ export const DEFAULT_TRANSCRIPT_PREAMBLE =
162
+ 'Below is a multi-turn conversation between users and you, the Assistant. ' +
163
+ 'Turns are delimited by "### User" and "### Assistant" headers. ' +
164
+ 'Attached images are numbered in attachment order; "[Image N]" marks where image N appeared. ' +
165
+ 'Images inside Assistant turns are images you generated earlier in this conversation; ' +
166
+ 'images inside User turns were provided by users. ' +
167
+ 'Generate the image for the next Assistant turn.';
168
+
120
169
  // ============================================================================
121
170
  // OpenAI Images Adapter
122
171
  // ============================================================================
@@ -130,12 +179,21 @@ export class OpenAIResponsesAdapter implements ProviderAdapter {
130
179
  private baseURL: string;
131
180
  private organization?: string;
132
181
  private allowImageEditing: boolean;
182
+ private promptFormat: 'legacy' | 'transcript';
183
+ private maxInputImages: number;
184
+ private transcriptPreamble: string;
133
185
 
134
186
  constructor(config: OpenAIResponsesAdapterConfig = {}) {
135
187
  this.apiKey = config.apiKey ?? process.env.OPENAI_API_KEY ?? '';
136
188
  this.baseURL = (config.baseURL ?? 'https://api.openai.com/v1').replace(/\/$/, '');
137
189
  this.organization = config.organization;
138
190
  this.allowImageEditing = config.allowImageEditing ?? true;
191
+ this.promptFormat = config.promptFormat ?? 'legacy';
192
+ this.maxInputImages = Math.max(
193
+ 1,
194
+ Math.min(config.maxInputImages ?? IMAGES_EDIT_MAX_INPUT_IMAGES, IMAGES_EDIT_MAX_INPUT_IMAGES)
195
+ );
196
+ this.transcriptPreamble = config.transcriptPreamble ?? DEFAULT_TRANSCRIPT_PREAMBLE;
139
197
 
140
198
  if (!this.apiKey) {
141
199
  throw new Error('OpenAI API key not provided');
@@ -150,10 +208,12 @@ export class OpenAIResponsesAdapter implements ProviderAdapter {
150
208
  request: ProviderRequest,
151
209
  options?: ProviderRequestOptions
152
210
  ): Promise<ProviderResponse> {
153
- const inputImages = this.allowImageEditing ? this.extractImages(request) : [];
211
+ const allImages = this.allowImageEditing ? this.collectImages(request) : [];
212
+ const selected = this.selectImages(allImages);
213
+ const inputImages = selected.map(ref => ref.dataUrl);
154
214
  const isEdit = inputImages.length > 0;
155
215
  const endpoint = isEdit ? 'images/edits' : 'images/generations';
156
- const imagesRequest = this.buildRequest(request, inputImages);
216
+ const imagesRequest = this.buildRequest(request, selected);
157
217
  options?.onRequest?.(imagesRequest);
158
218
 
159
219
  const { signal: combinedSignal, cleanup } = createCombinedSignal(options?.signal, options?.timeoutMs);
@@ -245,33 +305,47 @@ export class OpenAIResponsesAdapter implements ProviderAdapter {
245
305
  }
246
306
 
247
307
  /**
248
- * Extract base64 images from conversation messages as data URLs.
249
- * Used to determine whether to use /edits (with images) or /generations.
250
- * Returns up to 16 images (OpenAI limit for /v1/images/edits).
308
+ * Collect every base64 image in the conversation, in order, remembering
309
+ * which message/block it came from so the transcript can reference it.
310
+ * Both `image` blocks and `generated_image` blocks (this adapter's own
311
+ * earlier outputs, when a consumer carries them forward as-is) count.
251
312
  */
252
- private extractImages(request: ProviderRequest): string[] {
253
- const dataUrls: string[] = [];
254
- const MAX_IMAGES = 16;
255
-
256
- if (!request.messages) return dataUrls;
257
-
258
- for (const msg of request.messages as any[]) {
259
- if (!Array.isArray(msg.content)) continue;
260
-
261
- for (const block of msg.content) {
262
- if (dataUrls.length >= MAX_IMAGES) break;
263
-
264
- if (block.type === 'image') {
313
+ private collectImages(request: ProviderRequest): ImageRef[] {
314
+ const refs: ImageRef[] = [];
315
+ if (!request.messages) return refs;
316
+
317
+ (request.messages as any[]).forEach((msg, msgIndex) => {
318
+ if (!Array.isArray(msg.content)) return;
319
+ msg.content.forEach((block: any, blockIndex: number) => {
320
+ if (block?.type === 'image') {
265
321
  const source = block.source;
266
322
  if (source?.type === 'base64' && source.data) {
267
323
  const mimeType = source.media_type ?? source.mediaType ?? 'image/png';
268
- dataUrls.push(`data:${mimeType};base64,${source.data}`);
324
+ refs.push({ dataUrl: `data:${mimeType};base64,${source.data}`, msgIndex, blockIndex });
269
325
  }
326
+ } else if (block?.type === 'generated_image' && typeof block.data === 'string' && block.data) {
327
+ // A previous output of this adapter carried forward verbatim in
328
+ // history (consumers that keep ProviderResponse content rather
329
+ // than rebuilding from a channel). It is an image like any other.
330
+ const mimeType = block.mimeType ?? 'image/png';
331
+ refs.push({ dataUrl: `data:${mimeType};base64,${block.data}`, msgIndex, blockIndex });
270
332
  }
271
- }
272
- }
333
+ });
334
+ });
273
335
 
274
- return dataUrls;
336
+ return refs;
337
+ }
338
+
339
+ /**
340
+ * Apply the input-image cap. Legacy keeps the FIRST N (its historical
341
+ * behaviour); transcript keeps the MOST RECENT N, since the latest turns
342
+ * are what an edit request is about.
343
+ */
344
+ private selectImages(refs: ImageRef[]): ImageRef[] {
345
+ if (refs.length <= this.maxInputImages) return refs;
346
+ return this.promptFormat === 'transcript'
347
+ ? refs.slice(refs.length - this.maxInputImages)
348
+ : refs.slice(0, this.maxInputImages);
275
349
  }
276
350
 
277
351
  /**
@@ -288,8 +362,11 @@ export class OpenAIResponsesAdapter implements ProviderAdapter {
288
362
  };
289
363
  }
290
364
 
291
- private buildRequest(request: ProviderRequest, inputImages: string[]): ImagesRequest {
292
- const prompt = this.flattenToPrompt(request);
365
+ private buildRequest(request: ProviderRequest, selectedImages: ImageRef[]): ImagesRequest {
366
+ const inputImages = selectedImages.map(ref => ref.dataUrl);
367
+ const prompt = this.promptFormat === 'transcript'
368
+ ? this.flattenToTranscript(request, selectedImages)
369
+ : this.flattenToPrompt(request);
293
370
 
294
371
  const imagesRequest: ImagesRequest = {
295
372
  model: request.model,
@@ -362,6 +439,78 @@ export class OpenAIResponsesAdapter implements ProviderAdapter {
362
439
  return parts.join('\n\n');
363
440
  }
364
441
 
442
+ private systemText(request: ProviderRequest): string {
443
+ if (!request.system) return '';
444
+ if (typeof request.system === 'string') return request.system;
445
+ return (request.system as any[])
446
+ .filter((b: any) => b.type === 'text')
447
+ .map((b: any) => b.text)
448
+ .join('\n');
449
+ }
450
+
451
+ /**
452
+ * Flatten the conversation into a role-delimited transcript.
453
+ *
454
+ * Each message becomes (part of) a `### User` / `### Assistant` turn;
455
+ * consecutive same-role messages share one header. Text blocks are kept
456
+ * verbatim. Image blocks become `[Image N]` (N = 1-based position in the
457
+ * attached `image[]` list) when the image is attached, or
458
+ * `[Image omitted]` when it fell outside the input-image cap. An
459
+ * assistant message that only carried an image therefore still appears
460
+ * as an assistant turn — that is what lets a previous generation read as
461
+ * the assistant's own prior output. The prompt ends with an open
462
+ * `### Assistant` header for the turn being generated.
463
+ */
464
+ private flattenToTranscript(request: ProviderRequest, selectedImages: ImageRef[]): string {
465
+ const attachedIndex = new Map<string, number>();
466
+ selectedImages.forEach((ref, i) => attachedIndex.set(`${ref.msgIndex}:${ref.blockIndex}`, i + 1));
467
+
468
+ const sections: string[] = [];
469
+ const systemText = this.systemText(request);
470
+ if (systemText) sections.push(systemText);
471
+ if (this.transcriptPreamble) sections.push(this.transcriptPreamble);
472
+
473
+ const turns: string[] = [];
474
+ let currentRole: 'User' | 'Assistant' | undefined;
475
+ let currentLines: string[] = [];
476
+ const flushTurn = () => {
477
+ if (!currentRole) return;
478
+ const body = currentLines.join('\n').trim();
479
+ if (body) turns.push(`### ${currentRole}\n${body}`);
480
+ currentLines = [];
481
+ };
482
+
483
+ (request.messages as any[] | undefined)?.forEach((msg, msgIndex) => {
484
+ const role: 'User' | 'Assistant' = msg.role === 'assistant' ? 'Assistant' : 'User';
485
+ const lines: string[] = [];
486
+
487
+ if (typeof msg.content === 'string') {
488
+ if (msg.content) lines.push(msg.content);
489
+ } else if (Array.isArray(msg.content)) {
490
+ msg.content.forEach((block: any, blockIndex: number) => {
491
+ if (block?.type === 'text' && block.text) {
492
+ lines.push(block.text);
493
+ } else if (block?.type === 'image' || block?.type === 'generated_image') {
494
+ const n = attachedIndex.get(`${msgIndex}:${blockIndex}`);
495
+ lines.push(n !== undefined ? `[Image ${n}]` : '[Image omitted]');
496
+ }
497
+ });
498
+ }
499
+ if (lines.length === 0) return;
500
+
501
+ if (role !== currentRole) {
502
+ flushTurn();
503
+ currentRole = role;
504
+ }
505
+ currentLines.push(...lines);
506
+ });
507
+ flushTurn();
508
+
509
+ sections.push(turns.join('\n\n'));
510
+ sections.push('### Assistant');
511
+ return sections.filter(Boolean).join('\n\n');
512
+ }
513
+
365
514
  // --------------------------------------------------------------------------
366
515
  // Response Parsing
367
516
  // --------------------------------------------------------------------------
@@ -0,0 +1,131 @@
1
+ import type { ProviderRequest } from '../types/index.js';
2
+ import type { OpenAIResponsesInputItem } from './openai-responses-api.js';
3
+
4
+ type JsonObject = Record<string, unknown>;
5
+
6
+ /**
7
+ * Most agent turns arrive already formatted as provider-native Responses
8
+ * items. Internal maintenance calls, however, can bypass that formatter and
9
+ * carry Membrane's normalized `text`/`image`/tool blocks. Normalize at the
10
+ * final transport boundary so every call shape accepted by ProviderAdapter is
11
+ * valid on the Codex Responses endpoint.
12
+ */
13
+ export function normalizeResponsesInput(messages: ProviderRequest['messages']): OpenAIResponsesInputItem[] {
14
+ const output: unknown[] = [];
15
+
16
+ for (const rawMessage of messages as unknown[]) {
17
+ if (!isObject(rawMessage)) {
18
+ output.push(rawMessage);
19
+ continue;
20
+ }
21
+ if (rawMessage.type !== 'message' && rawMessage.role === undefined) {
22
+ output.push(normalizeStandaloneItem(rawMessage));
23
+ continue;
24
+ }
25
+
26
+ // Native messages (including phase, status and developer/system roles)
27
+ // must survive replay verbatim. Only translate normalized content blocks.
28
+ if (Array.isArray(rawMessage.content) && !rawMessage.content.some((block) =>
29
+ isObject(block) && ['text', 'image', 'tool_use', 'tool_result', 'redacted_thinking'].includes(asString(block.type))
30
+ )) {
31
+ output.push(rawMessage);
32
+ continue;
33
+ }
34
+ const role = typeof rawMessage.role === 'string' ? rawMessage.role : 'user';
35
+ const blocks = Array.isArray(rawMessage.content)
36
+ ? rawMessage.content
37
+ : typeof rawMessage.content === 'string'
38
+ ? [{ type: 'text', text: rawMessage.content }]
39
+ : [];
40
+ let parts: unknown[] = [];
41
+ const flush = () => {
42
+ if (parts.length === 0) return;
43
+ output.push({
44
+ type: 'message',
45
+ ...rawMessage,
46
+ role,
47
+ content: parts,
48
+ });
49
+ parts = [];
50
+ };
51
+
52
+ for (const rawBlock of blocks) {
53
+ if (!isObject(rawBlock)) continue;
54
+ if (rawBlock.type === 'text') {
55
+ parts.push({ type: role === 'assistant' ? 'output_text' : 'input_text', text: asString(rawBlock.text) });
56
+ } else if (rawBlock.type === 'image') {
57
+ const imageUrl = responsesImageUrl(rawBlock);
58
+ if (imageUrl && role !== 'assistant') parts.push({ type: 'input_image', image_url: imageUrl });
59
+ } else if (rawBlock.type === 'tool_use') {
60
+ flush();
61
+ output.push(normalizeStandaloneItem(rawBlock));
62
+ } else if (rawBlock.type === 'tool_result') {
63
+ flush();
64
+ output.push(normalizeStandaloneItem(rawBlock));
65
+ } else if (rawBlock.type === 'redacted_thinking') {
66
+ flush();
67
+ output.push(reasoningInputItem(rawBlock));
68
+ } else {
69
+ // Already-native input_text/output_text/input_image/refusal parts.
70
+ parts.push(rawBlock);
71
+ }
72
+ }
73
+ flush();
74
+ }
75
+
76
+ return output as OpenAIResponsesInputItem[];
77
+ }
78
+
79
+ function normalizeStandaloneItem(item: JsonObject): unknown {
80
+ if (item.type === 'tool_use') {
81
+ return {
82
+ type: 'function_call',
83
+ call_id: asString(item.id),
84
+ name: asString(item.name),
85
+ arguments: JSON.stringify(isObject(item.input) ? item.input : {}),
86
+ };
87
+ }
88
+ if (item.type === 'tool_result') {
89
+ const content = item.content;
90
+ return {
91
+ type: 'function_call_output',
92
+ call_id: asString(item.toolUseId) || asString(item.tool_use_id),
93
+ output: typeof content === 'string' ? content : JSON.stringify(content ?? null),
94
+ };
95
+ }
96
+ if (item.type === 'redacted_thinking') {
97
+ return reasoningInputItem(item);
98
+ }
99
+ return item;
100
+ }
101
+
102
+ /** Replay a captured reasoning carrier as a Responses input item.
103
+ *
104
+ * Prefer the provider-native item verbatim when the block still carries it
105
+ * (`rawItem` from response parsing). Otherwise reconstruct the minimum the
106
+ * Responses API accepts: `summary` is a REQUIRED field on reasoning input
107
+ * items (empty array = "no summaries") — omitting it 400s with
108
+ * "Missing required parameter: 'input[N].summary'". */
109
+ function reasoningInputItem(block: JsonObject): unknown {
110
+ const raw = block.rawItem;
111
+ if (isObject(raw) && raw.type === 'reasoning') return raw;
112
+ return { type: 'reasoning', summary: [], encrypted_content: asString(block.data) };
113
+ }
114
+
115
+ function responsesImageUrl(block: JsonObject): string | undefined {
116
+ const source = isObject(block.source) ? block.source : undefined;
117
+ if (!source) return typeof block.image_url === 'string' ? block.image_url : undefined;
118
+ if (source.type === 'url') return asString(source.url) || undefined;
119
+ if (source.type !== 'base64') return undefined;
120
+ const mediaType = asString(source.mediaType) || asString(source.media_type) || 'image/png';
121
+ const data = asString(source.data);
122
+ return data ? `data:${mediaType};base64,${data}` : undefined;
123
+ }
124
+
125
+ function asString(value: unknown): string {
126
+ return typeof value === 'string' ? value : '';
127
+ }
128
+
129
+ function isObject(value: unknown): value is JsonObject {
130
+ return Boolean(value) && typeof value === 'object' && !Array.isArray(value);
131
+ }
@@ -291,6 +291,9 @@ export function createCombinedSignal(
291
291
  */
292
292
  export class SSELineParser {
293
293
  private buffer: string = '';
294
+ private data: string[] = [];
295
+
296
+ constructor(private readonly options: { multiline?: boolean } = {}) {}
294
297
 
295
298
  /**
296
299
  * Feed a raw chunk from the stream reader and get back complete SSE data lines.
@@ -305,6 +308,10 @@ export class SSELineParser {
305
308
  this.buffer = lines.pop() || '';
306
309
 
307
310
  for (const line of lines) {
311
+ if (this.options.multiline) {
312
+ this.processEventLine(line.replace(/\r$/, ''), results);
313
+ continue;
314
+ }
308
315
  const trimmed = line.trim();
309
316
  if (trimmed.startsWith('data: ')) {
310
317
  results.push(trimmed.slice(6));
@@ -315,10 +322,24 @@ export class SSELineParser {
315
322
  return results;
316
323
  }
317
324
 
318
- /**
319
- * Flush any remaining buffered content (call when stream ends).
320
- */
325
+ private processEventLine(line: string, results: string[]): void {
326
+ if (line === '') {
327
+ if (this.data.length) results.push(this.data.join('\n'));
328
+ this.data = [];
329
+ } else if (line.startsWith('data:')) {
330
+ this.data.push(line.slice(5).replace(/^ /, ''));
331
+ }
332
+ }
333
+
334
+ /** Flush any remaining event/tail when the stream ends. */
321
335
  flush(): string[] {
336
+ if (this.options.multiline) {
337
+ const results: string[] = [];
338
+ if (this.buffer) this.processEventLine(this.buffer.replace(/\r$/, ''), results);
339
+ this.buffer = '';
340
+ this.processEventLine('', results);
341
+ return results;
342
+ }
322
343
  if (!this.buffer.trim()) return [];
323
344
  const trimmed = this.buffer.trim();
324
345
  this.buffer = '';
@@ -207,6 +207,11 @@ export interface ProviderAdapter {
207
207
  */
208
208
  usageCacheConvention?: UsageCacheConvention;
209
209
 
210
+ /** Whether this transport requires the configured Responses formatter.
211
+ * False permits generic per-request formatter overrides (e.g. named
212
+ * maintenance messages). Wrappers must forward this capability. */
213
+ readonly requiresNativeResponsesInput?: boolean;
214
+
210
215
  /** Check if this adapter handles a model */
211
216
  supportsModel(modelId: string): boolean;
212
217