@tanstack/ai 0.31.0 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/generateImage/adapter.d.ts +8 -4
- package/dist/esm/activities/generateImage/adapter.js.map +1 -1
- package/dist/esm/activities/generateImage/index.d.ts +19 -3
- package/dist/esm/activities/generateImage/index.js +12 -1
- package/dist/esm/activities/generateImage/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/adapter.d.ts +65 -6
- package/dist/esm/activities/generateVideo/adapter.js +14 -0
- package/dist/esm/activities/generateVideo/adapter.js.map +1 -1
- package/dist/esm/activities/generateVideo/index.d.ts +31 -5
- package/dist/esm/activities/generateVideo/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/snap.d.ts +14 -0
- package/dist/esm/activities/generateVideo/snap.js +54 -0
- package/dist/esm/activities/generateVideo/snap.js.map +1 -0
- package/dist/esm/activities/index.d.ts +3 -2
- package/dist/esm/activities/index.js +2 -0
- package/dist/esm/activities/index.js.map +1 -1
- package/dist/esm/client.d.ts +1 -1
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +2 -0
- package/dist/esm/index.js +2 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/types.d.ts +96 -7
- package/dist/esm/utilities/media-prompt.d.ts +35 -0
- package/dist/esm/utilities/media-prompt.js +43 -0
- package/dist/esm/utilities/media-prompt.js.map +1 -0
- package/package.json +2 -2
- package/skills/ai-core/media-generation/SKILL.md +173 -3
- package/src/activities/generateImage/adapter.ts +16 -3
- package/src/activities/generateImage/index.ts +48 -4
- package/src/activities/generateVideo/adapter.ts +80 -4
- package/src/activities/generateVideo/index.ts +53 -4
- package/src/activities/generateVideo/snap.ts +100 -0
- package/src/activities/index.ts +4 -0
- package/src/client.ts +4 -0
- package/src/index.ts +4 -0
- package/src/types.ts +119 -6
- package/src/utilities/media-prompt.ts +86 -0
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
function entryToSeconds(entry) {
|
|
2
|
+
if (typeof entry === "number") {
|
|
3
|
+
return Number.isFinite(entry) ? entry : null;
|
|
4
|
+
}
|
|
5
|
+
const stripped = entry.endsWith("s") ? entry.slice(0, -1) : entry;
|
|
6
|
+
const parsed = Number(stripped);
|
|
7
|
+
return Number.isFinite(parsed) ? parsed : null;
|
|
8
|
+
}
|
|
9
|
+
function snapToDurationOption(seconds, options) {
|
|
10
|
+
switch (options.kind) {
|
|
11
|
+
case "none":
|
|
12
|
+
return void 0;
|
|
13
|
+
case "discrete": {
|
|
14
|
+
return pickClosestDiscrete(seconds, options.values);
|
|
15
|
+
}
|
|
16
|
+
case "range": {
|
|
17
|
+
const step = options.step ?? 1;
|
|
18
|
+
const clamped = Math.min(options.max, Math.max(options.min, seconds));
|
|
19
|
+
const snapped = Math.round((clamped - options.min) / step) * step + options.min;
|
|
20
|
+
return Math.min(options.max, Math.max(options.min, snapped));
|
|
21
|
+
}
|
|
22
|
+
case "mixed": {
|
|
23
|
+
const discreteCandidate = pickClosestDiscrete(seconds, options.values);
|
|
24
|
+
if (!options.range) return discreteCandidate;
|
|
25
|
+
const { min, max, step = 1 } = options.range;
|
|
26
|
+
const clamped = Math.min(max, Math.max(min, seconds));
|
|
27
|
+
const rangeValue = Math.min(
|
|
28
|
+
max,
|
|
29
|
+
Math.max(min, Math.round((clamped - min) / step) * step + min)
|
|
30
|
+
);
|
|
31
|
+
const discreteSeconds = typeof discreteCandidate === "number" ? discreteCandidate : discreteCandidate !== void 0 ? entryToSeconds(discreteCandidate) ?? Infinity : Infinity;
|
|
32
|
+
return Math.abs(discreteSeconds - seconds) <= Math.abs(rangeValue - seconds) ? discreteCandidate : rangeValue;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
function pickClosestDiscrete(seconds, values) {
|
|
37
|
+
if (values.length === 0) return void 0;
|
|
38
|
+
let best;
|
|
39
|
+
let bestDistance = Infinity;
|
|
40
|
+
for (const value of values) {
|
|
41
|
+
const v = entryToSeconds(value);
|
|
42
|
+
if (v === null) continue;
|
|
43
|
+
const distance = Math.abs(v - seconds);
|
|
44
|
+
if (distance < bestDistance) {
|
|
45
|
+
bestDistance = distance;
|
|
46
|
+
best = value;
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
return best ?? values[0];
|
|
50
|
+
}
|
|
51
|
+
export {
|
|
52
|
+
snapToDurationOption
|
|
53
|
+
};
|
|
54
|
+
//# sourceMappingURL=snap.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"snap.js","sources":["../../../../src/activities/generateVideo/snap.ts"],"sourcesContent":["import type { DurationOptions } from './adapter'\n\n/**\n * Extract a numeric seconds value from a `DurationOptions` entry. Returns\n * `null` for entries that don't parse as a number — e.g. `'auto'`.\n *\n * Handles the keyword-with-unit form FAL uses for Luma/Veo (`'8s'`, `'9s'`)\n * by stripping a trailing `s`. Pure-numeric strings (`'5'`, `'10'`) parse via\n * Number(). Numbers pass through.\n */\nfunction entryToSeconds(entry: string | number): number | null {\n if (typeof entry === 'number') {\n return Number.isFinite(entry) ? entry : null\n }\n const stripped = entry.endsWith('s') ? entry.slice(0, -1) : entry\n const parsed = Number(stripped)\n return Number.isFinite(parsed) ? parsed : null\n}\n\n/**\n * Snap a raw seconds value to the closest valid duration for a model's\n * `DurationOptions`.\n *\n * - `none` → `undefined`\n * - `discrete` → closest numeric-parseable entry; if none parse,\n * returns `values[0]` (keyword-only models like 'auto')\n * - `range` → clamped to [min, max] and rounded to `step` (default 1)\n * - `mixed` → closest of (discrete numerics ∪ range values)\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function snapToDurationOption<T extends string | number | undefined>(\n seconds: number,\n options: DurationOptions<T>,\n): T | undefined {\n switch (options.kind) {\n case 'none':\n return undefined\n\n case 'discrete': {\n return pickClosestDiscrete(seconds, options.values)\n }\n\n case 'range': {\n const step = options.step ?? 1\n const clamped = Math.min(options.max, Math.max(options.min, seconds))\n const snapped =\n Math.round((clamped - options.min) / step) * step + options.min\n return Math.min(options.max, Math.max(options.min, snapped)) as T\n }\n\n case 'mixed': {\n const discreteCandidate = pickClosestDiscrete(seconds, options.values)\n if (!options.range) return discreteCandidate\n\n const { min, max, step = 1 } = options.range\n const clamped = Math.min(max, Math.max(min, seconds))\n const rangeValue = Math.min(\n max,\n Math.max(min, Math.round((clamped - min) / step) * step + min),\n )\n\n // Compare distance; range value is numeric, discrete may have non-numeric\n // first-entry fallback (return distance Infinity for non-numerics).\n const discreteSeconds =\n typeof discreteCandidate === 'number'\n ? discreteCandidate\n : discreteCandidate !== undefined\n ? (entryToSeconds(discreteCandidate) ?? Infinity)\n : Infinity\n\n return Math.abs(discreteSeconds - seconds) <=\n Math.abs(rangeValue - seconds)\n ? discreteCandidate\n : (rangeValue as T)\n }\n }\n}\n\nfunction pickClosestDiscrete<T extends string | number>(\n seconds: number,\n values: ReadonlyArray<T>,\n): T | undefined {\n if (values.length === 0) return undefined\n\n let best: T | undefined\n let bestDistance = Infinity\n for (const value of values) {\n const v = entryToSeconds(value)\n if (v === null) continue\n const distance = Math.abs(v - seconds)\n if (distance < bestDistance) {\n bestDistance = distance\n best = value\n }\n }\n\n // Keyword-only set (no numeric-parseable entries) — fall back to first entry.\n return best ?? values[0]\n}\n"],"names":[],"mappings":"AAUA,SAAS,eAAe,OAAuC;AAC7D,MAAI,OAAO,UAAU,UAAU;AAC7B,WAAO,OAAO,SAAS,KAAK,IAAI,QAAQ;AAAA,EAC1C;AACA,QAAM,WAAW,MAAM,SAAS,GAAG,IAAI,MAAM,MAAM,GAAG,EAAE,IAAI;AAC5D,QAAM,SAAS,OAAO,QAAQ;AAC9B,SAAO,OAAO,SAAS,MAAM,IAAI,SAAS;AAC5C;AAcO,SAAS,qBACd,SACA,SACe;AACf,UAAQ,QAAQ,MAAA;AAAA,IACd,KAAK;AACH,aAAO;AAAA,IAET,KAAK,YAAY;AACf,aAAO,oBAAoB,SAAS,QAAQ,MAAM;AAAA,IACpD;AAAA,IAEA,KAAK,SAAS;AACZ,YAAM,OAAO,QAAQ,QAAQ;AAC7B,YAAM,UAAU,KAAK,IAAI,QAAQ,KAAK,KAAK,IAAI,QAAQ,KAAK,OAAO,CAAC;AACpE,YAAM,UACJ,KAAK,OAAO,UAAU,QAAQ,OAAO,IAAI,IAAI,OAAO,QAAQ;AAC9D,aAAO,KAAK,IAAI,QAAQ,KAAK,KAAK,IAAI,QAAQ,KAAK,OAAO,CAAC;AAAA,IAC7D;AAAA,IAEA,KAAK,SAAS;AACZ,YAAM,oBAAoB,oBAAoB,SAAS,QAAQ,MAAM;AACrE,UAAI,CAAC,QAAQ,MAAO,QAAO;AAE3B,YAAM,EAAE,KAAK,KAAK,OAAO,EAAA,IAAM,QAAQ;AACvC,YAAM,UAAU,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,OAAO,CAAC;AACpD,YAAM,aAAa,KAAK;AAAA,QACtB;AAAA,QACA,KAAK,IAAI,KAAK,KAAK,OAAO,UAAU,OAAO,IAAI,IAAI,OAAO,GAAG;AAAA,MAAA;AAK/D,YAAM,kBACJ,OAAO,sBAAsB,WACzB,oBACA,sBAAsB,SACnB,eAAe,iBAAiB,KAAK,WACtC;AAER,aAAO,KAAK,IAAI,kBAAkB,OAAO,KACvC,KAAK,IAAI,aAAa,OAAO,IAC3B,oBACC;AAAA,IACP;AAAA,EAAA;AAEJ;AAEA,SAAS,oBACP,SACA,QACe;AACf,MAAI,OAAO,WAAW,EAAG,QAAO;AAEhC,MAAI;AACJ,MAAI,eAAe;AACnB,aAAW,SAAS,QAAQ;AAC1B,UAAM,IAAI,eAAe,KAAK;AAC9B,QAAI,MAAM,KAAM;AAChB,UAAM,WAAW,KAAK,IAAI,IAAI,OAAO;AACrC,QAAI,WAAW,cAAc;AAC3B,qBAAe;AACf,aAAO;AAAA,IACT;AAAA,EACF;AAGA,SAAO,QAAQ,OAAO,CAAC;AACzB;"}
|
|
@@ -14,8 +14,9 @@ export { kind as imageKind, generateImage, type ImageActivityOptions, type Image
|
|
|
14
14
|
export { BaseImageAdapter, type ImageAdapter, type ImageAdapterConfig, type AnyImageAdapter, } from './generateImage/adapter.js';
|
|
15
15
|
export { kind as audioKind, generateAudio, type AudioActivityOptions, type AudioActivityResult, type AudioProviderOptions, } from './generateAudio/index.js';
|
|
16
16
|
export { BaseAudioAdapter, type AudioAdapter, type AudioAdapterConfig, type AnyAudioAdapter, } from './generateAudio/adapter.js';
|
|
17
|
-
export { kind as videoKind, generateVideo, getVideoJobStatus, type VideoActivityOptions, type VideoActivityResult, type VideoProviderOptions, type VideoCreateOptions, type VideoStatusOptions, type VideoUrlOptions, } from './generateVideo/index.js';
|
|
18
|
-
export { BaseVideoAdapter, type VideoAdapter, type VideoAdapterConfig, type AnyVideoAdapter, } from './generateVideo/adapter.js';
|
|
17
|
+
export { kind as videoKind, generateVideo, getVideoJobStatus, type VideoActivityOptions, type VideoActivityResult, type VideoProviderOptions, type VideoCreateOptions, type VideoStatusOptions, type VideoUrlOptions, type VideoDurationForAdapter, } from './generateVideo/index.js';
|
|
18
|
+
export { BaseVideoAdapter, type VideoAdapter, type VideoAdapterConfig, type AnyVideoAdapter, type DurationOptions, } from './generateVideo/adapter.js';
|
|
19
|
+
export { snapToDurationOption } from './generateVideo/snap.js';
|
|
19
20
|
export { kind as ttsKind, generateSpeech, type TTSActivityOptions, type TTSActivityResult, type TTSProviderOptions, } from './generateSpeech/index.js';
|
|
20
21
|
export { BaseTTSAdapter, type TTSAdapter, type TTSAdapterConfig, type AnyTTSAdapter, } from './generateSpeech/adapter.js';
|
|
21
22
|
export { kind as transcriptionKind, generateTranscription, type TranscriptionActivityOptions, type TranscriptionActivityResult, type TranscriptionProviderOptions, } from './generateTranscription/index.js';
|
|
@@ -9,6 +9,7 @@ import { kind as kind4, generateAudio } from "./generateAudio/index.js";
|
|
|
9
9
|
import { BaseAudioAdapter } from "./generateAudio/adapter.js";
|
|
10
10
|
import { generateVideo, getVideoJobStatus, kind as kind5 } from "./generateVideo/index.js";
|
|
11
11
|
import { BaseVideoAdapter } from "./generateVideo/adapter.js";
|
|
12
|
+
import { snapToDurationOption } from "./generateVideo/snap.js";
|
|
12
13
|
import { generateSpeech, kind as kind6 } from "./generateSpeech/index.js";
|
|
13
14
|
import { BaseTTSAdapter } from "./generateSpeech/adapter.js";
|
|
14
15
|
import { generateTranscription, kind as kind7 } from "./generateTranscription/index.js";
|
|
@@ -31,6 +32,7 @@ export {
|
|
|
31
32
|
generateVideo,
|
|
32
33
|
getVideoJobStatus,
|
|
33
34
|
kind3 as imageKind,
|
|
35
|
+
snapToDurationOption,
|
|
34
36
|
summarize,
|
|
35
37
|
kind2 as summarizeKind,
|
|
36
38
|
kind as textKind,
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;"}
|
package/dist/esm/client.d.ts
CHANGED
|
@@ -40,5 +40,5 @@ export { BatchStrategy, CompositeStrategy, defaultJSONParser, ImmediateStrategy,
|
|
|
40
40
|
export type { ChunkRecording, ChunkStrategy, InternalToolCallState, JSONParser, ProcessorResult, ProcessorState, StreamProcessorEvents, StreamProcessorOptions, ToolCallState, ToolResultState, } from './activities/chat/stream/index.js';
|
|
41
41
|
export { uiMessagesToWire } from './utilities/ag-ui-wire.js';
|
|
42
42
|
export type { WireMessage } from './utilities/ag-ui-wire.js';
|
|
43
|
-
export type { AudioPart, ContentPart, ContentPartDataSource, ContentPartSource, ContentPartUrlSource, CustomEvent, DocumentPart, ImagePart, MessagePart, ModelMessage, RunErrorEvent, RunFinishedEvent, SchemaInput, StreamChunk, StructuredOutputPart, TextPart, ThinkingPart, ToolCall, ToolCallPart, ToolResultPart, UIMessage, VideoPart, InferSchemaType, } from './types.js';
|
|
43
|
+
export type { AudioPart, ContentPart, ContentPartDataSource, ContentPartSource, ContentPartUrlSource, CustomEvent, DocumentPart, ImagePart, MediaInputMetadata, MediaInputRole, MediaPrompt, MediaPromptPart, MessagePart, ModelMessage, RunErrorEvent, RunFinishedEvent, SchemaInput, StreamChunk, StructuredOutputPart, TextPart, ThinkingPart, ToolCall, ToolCallPart, ToolResultPart, UIMessage, VideoPart, InferSchemaType, } from './types.js';
|
|
44
44
|
export type { AudioVisualization, RealtimeError, RealtimeErrorCode, RealtimeEvent, RealtimeEventHandler, RealtimeEventPayloads, RealtimeMessage, RealtimeMessagePart, RealtimeMode, RealtimeSessionConfig, RealtimeStatus, RealtimeToken, RealtimeAudioPart, RealtimeImagePart, RealtimeTextPart, RealtimeToolCallPart, RealtimeToolResultPart, VADConfig, } from './realtime/types.js';
|
package/dist/esm/client.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"client.js","sources":["../../src/client.ts"],"sourcesContent":["export enum EventType {\n TEXT_MESSAGE_START = 'TEXT_MESSAGE_START',\n TEXT_MESSAGE_CONTENT = 'TEXT_MESSAGE_CONTENT',\n TEXT_MESSAGE_END = 'TEXT_MESSAGE_END',\n TEXT_MESSAGE_CHUNK = 'TEXT_MESSAGE_CHUNK',\n TOOL_CALL_START = 'TOOL_CALL_START',\n TOOL_CALL_ARGS = 'TOOL_CALL_ARGS',\n TOOL_CALL_END = 'TOOL_CALL_END',\n TOOL_CALL_CHUNK = 'TOOL_CALL_CHUNK',\n TOOL_CALL_RESULT = 'TOOL_CALL_RESULT',\n THINKING_START = 'THINKING_START',\n THINKING_END = 'THINKING_END',\n THINKING_TEXT_MESSAGE_START = 'THINKING_TEXT_MESSAGE_START',\n THINKING_TEXT_MESSAGE_CONTENT = 'THINKING_TEXT_MESSAGE_CONTENT',\n THINKING_TEXT_MESSAGE_END = 'THINKING_TEXT_MESSAGE_END',\n STATE_SNAPSHOT = 'STATE_SNAPSHOT',\n STATE_DELTA = 'STATE_DELTA',\n MESSAGES_SNAPSHOT = 'MESSAGES_SNAPSHOT',\n ACTIVITY_SNAPSHOT = 'ACTIVITY_SNAPSHOT',\n ACTIVITY_DELTA = 'ACTIVITY_DELTA',\n RAW = 'RAW',\n CUSTOM = 'CUSTOM',\n RUN_STARTED = 'RUN_STARTED',\n RUN_FINISHED = 'RUN_FINISHED',\n RUN_ERROR = 'RUN_ERROR',\n STEP_STARTED = 'STEP_STARTED',\n STEP_FINISHED = 'STEP_FINISHED',\n REASONING_START = 'REASONING_START',\n REASONING_MESSAGE_START = 'REASONING_MESSAGE_START',\n REASONING_MESSAGE_CONTENT = 'REASONING_MESSAGE_CONTENT',\n REASONING_MESSAGE_END = 'REASONING_MESSAGE_END',\n REASONING_MESSAGE_CHUNK = 'REASONING_MESSAGE_CHUNK',\n REASONING_END = 'REASONING_END',\n REASONING_ENCRYPTED_VALUE = 'REASONING_ENCRYPTED_VALUE',\n}\n\nexport {\n toolDefinition,\n type AnyClientTool,\n type ClientTool,\n type InferToolInput,\n type InferToolName,\n type InferToolOutput,\n type ToolDefinition,\n type ToolDefinitionConfig,\n type ToolDefinitionInstance,\n} from './activities/chat/tools/tool-definition'\n\nexport {\n convertSchemaToJsonSchema,\n isStandardSchema,\n parseWithStandardSchema,\n} from './activities/chat/tools/schema-converter'\n\nexport {\n convertMessagesToModelMessages,\n generateMessageId,\n modelMessageToUIMessage,\n modelMessagesToUIMessages,\n normalizeToUIMessage,\n uiMessageToModelMessages,\n} from './activities/chat/messages'\n\nexport {\n BatchStrategy,\n CompositeStrategy,\n defaultJSONParser,\n ImmediateStrategy,\n parsePartialJSON,\n PartialJSONParser,\n PunctuationStrategy,\n StreamProcessor,\n WordBoundaryStrategy,\n} from './activities/chat/stream/index'\nexport type {\n ChunkRecording,\n ChunkStrategy,\n InternalToolCallState,\n JSONParser,\n ProcessorResult,\n ProcessorState,\n StreamProcessorEvents,\n StreamProcessorOptions,\n ToolCallState,\n ToolResultState,\n} from './activities/chat/stream/index'\n\nexport { uiMessagesToWire } from './utilities/ag-ui-wire'\nexport type { WireMessage } from './utilities/ag-ui-wire'\n\nexport type {\n AudioPart,\n ContentPart,\n ContentPartDataSource,\n ContentPartSource,\n ContentPartUrlSource,\n CustomEvent,\n DocumentPart,\n ImagePart,\n MessagePart,\n ModelMessage,\n RunErrorEvent,\n RunFinishedEvent,\n SchemaInput,\n StreamChunk,\n StructuredOutputPart,\n TextPart,\n ThinkingPart,\n ToolCall,\n ToolCallPart,\n ToolResultPart,\n UIMessage,\n VideoPart,\n InferSchemaType,\n} from './types'\n\nexport type {\n AudioVisualization,\n RealtimeError,\n RealtimeErrorCode,\n RealtimeEvent,\n RealtimeEventHandler,\n RealtimeEventPayloads,\n RealtimeMessage,\n RealtimeMessagePart,\n RealtimeMode,\n RealtimeSessionConfig,\n RealtimeStatus,\n RealtimeToken,\n RealtimeAudioPart,\n RealtimeImagePart,\n RealtimeTextPart,\n RealtimeToolCallPart,\n RealtimeToolResultPart,\n VADConfig,\n} from './realtime/types'\n"],"names":["EventType"],"mappings":";;;;;;;AAAO,IAAK,8BAAAA,eAAL;AACLA,aAAA,oBAAA,IAAqB;AACrBA,aAAA,sBAAA,IAAuB;AACvBA,aAAA,kBAAA,IAAmB;AACnBA,aAAA,oBAAA,IAAqB;AACrBA,aAAA,iBAAA,IAAkB;AAClBA,aAAA,gBAAA,IAAiB;AACjBA,aAAA,eAAA,IAAgB;AAChBA,aAAA,iBAAA,IAAkB;AAClBA,aAAA,kBAAA,IAAmB;AACnBA,aAAA,gBAAA,IAAiB;AACjBA,aAAA,cAAA,IAAe;AACfA,aAAA,6BAAA,IAA8B;AAC9BA,aAAA,+BAAA,IAAgC;AAChCA,aAAA,2BAAA,IAA4B;AAC5BA,aAAA,gBAAA,IAAiB;AACjBA,aAAA,aAAA,IAAc;AACdA,aAAA,mBAAA,IAAoB;AACpBA,aAAA,mBAAA,IAAoB;AACpBA,aAAA,gBAAA,IAAiB;AACjBA,aAAA,KAAA,IAAM;AACNA,aAAA,QAAA,IAAS;AACTA,aAAA,aAAA,IAAc;AACdA,aAAA,cAAA,IAAe;AACfA,aAAA,WAAA,IAAY;AACZA,aAAA,cAAA,IAAe;AACfA,aAAA,eAAA,IAAgB;AAChBA,aAAA,iBAAA,IAAkB;AAClBA,aAAA,yBAAA,IAA0B;AAC1BA,aAAA,2BAAA,IAA4B;AAC5BA,aAAA,uBAAA,IAAwB;AACxBA,aAAA,yBAAA,IAA0B;AAC1BA,aAAA,eAAA,IAAgB;AAChBA,aAAA,2BAAA,IAA4B;AAjClB,SAAAA;AAAA,GAAA,aAAA,CAAA,CAAA;"}
|
|
1
|
+
{"version":3,"file":"client.js","sources":["../../src/client.ts"],"sourcesContent":["export enum EventType {\n TEXT_MESSAGE_START = 'TEXT_MESSAGE_START',\n TEXT_MESSAGE_CONTENT = 'TEXT_MESSAGE_CONTENT',\n TEXT_MESSAGE_END = 'TEXT_MESSAGE_END',\n TEXT_MESSAGE_CHUNK = 'TEXT_MESSAGE_CHUNK',\n TOOL_CALL_START = 'TOOL_CALL_START',\n TOOL_CALL_ARGS = 'TOOL_CALL_ARGS',\n TOOL_CALL_END = 'TOOL_CALL_END',\n TOOL_CALL_CHUNK = 'TOOL_CALL_CHUNK',\n TOOL_CALL_RESULT = 'TOOL_CALL_RESULT',\n THINKING_START = 'THINKING_START',\n THINKING_END = 'THINKING_END',\n THINKING_TEXT_MESSAGE_START = 'THINKING_TEXT_MESSAGE_START',\n THINKING_TEXT_MESSAGE_CONTENT = 'THINKING_TEXT_MESSAGE_CONTENT',\n THINKING_TEXT_MESSAGE_END = 'THINKING_TEXT_MESSAGE_END',\n STATE_SNAPSHOT = 'STATE_SNAPSHOT',\n STATE_DELTA = 'STATE_DELTA',\n MESSAGES_SNAPSHOT = 'MESSAGES_SNAPSHOT',\n ACTIVITY_SNAPSHOT = 'ACTIVITY_SNAPSHOT',\n ACTIVITY_DELTA = 'ACTIVITY_DELTA',\n RAW = 'RAW',\n CUSTOM = 'CUSTOM',\n RUN_STARTED = 'RUN_STARTED',\n RUN_FINISHED = 'RUN_FINISHED',\n RUN_ERROR = 'RUN_ERROR',\n STEP_STARTED = 'STEP_STARTED',\n STEP_FINISHED = 'STEP_FINISHED',\n REASONING_START = 'REASONING_START',\n REASONING_MESSAGE_START = 'REASONING_MESSAGE_START',\n REASONING_MESSAGE_CONTENT = 'REASONING_MESSAGE_CONTENT',\n REASONING_MESSAGE_END = 'REASONING_MESSAGE_END',\n REASONING_MESSAGE_CHUNK = 'REASONING_MESSAGE_CHUNK',\n REASONING_END = 'REASONING_END',\n REASONING_ENCRYPTED_VALUE = 'REASONING_ENCRYPTED_VALUE',\n}\n\nexport {\n toolDefinition,\n type AnyClientTool,\n type ClientTool,\n type InferToolInput,\n type InferToolName,\n type InferToolOutput,\n type ToolDefinition,\n type ToolDefinitionConfig,\n type ToolDefinitionInstance,\n} from './activities/chat/tools/tool-definition'\n\nexport {\n convertSchemaToJsonSchema,\n isStandardSchema,\n parseWithStandardSchema,\n} from './activities/chat/tools/schema-converter'\n\nexport {\n convertMessagesToModelMessages,\n generateMessageId,\n modelMessageToUIMessage,\n modelMessagesToUIMessages,\n normalizeToUIMessage,\n uiMessageToModelMessages,\n} from './activities/chat/messages'\n\nexport {\n BatchStrategy,\n CompositeStrategy,\n defaultJSONParser,\n ImmediateStrategy,\n parsePartialJSON,\n PartialJSONParser,\n PunctuationStrategy,\n StreamProcessor,\n WordBoundaryStrategy,\n} from './activities/chat/stream/index'\nexport type {\n ChunkRecording,\n ChunkStrategy,\n InternalToolCallState,\n JSONParser,\n ProcessorResult,\n ProcessorState,\n StreamProcessorEvents,\n StreamProcessorOptions,\n ToolCallState,\n ToolResultState,\n} from './activities/chat/stream/index'\n\nexport { uiMessagesToWire } from './utilities/ag-ui-wire'\nexport type { WireMessage } from './utilities/ag-ui-wire'\n\nexport type {\n AudioPart,\n ContentPart,\n ContentPartDataSource,\n ContentPartSource,\n ContentPartUrlSource,\n CustomEvent,\n DocumentPart,\n ImagePart,\n MediaInputMetadata,\n MediaInputRole,\n MediaPrompt,\n MediaPromptPart,\n MessagePart,\n ModelMessage,\n RunErrorEvent,\n RunFinishedEvent,\n SchemaInput,\n StreamChunk,\n StructuredOutputPart,\n TextPart,\n ThinkingPart,\n ToolCall,\n ToolCallPart,\n ToolResultPart,\n UIMessage,\n VideoPart,\n InferSchemaType,\n} from './types'\n\nexport type {\n AudioVisualization,\n RealtimeError,\n RealtimeErrorCode,\n RealtimeEvent,\n RealtimeEventHandler,\n RealtimeEventPayloads,\n RealtimeMessage,\n RealtimeMessagePart,\n RealtimeMode,\n RealtimeSessionConfig,\n RealtimeStatus,\n RealtimeToken,\n RealtimeAudioPart,\n RealtimeImagePart,\n RealtimeTextPart,\n RealtimeToolCallPart,\n RealtimeToolResultPart,\n VADConfig,\n} from './realtime/types'\n"],"names":["EventType"],"mappings":";;;;;;;AAAO,IAAK,8BAAAA,eAAL;AACLA,aAAA,oBAAA,IAAqB;AACrBA,aAAA,sBAAA,IAAuB;AACvBA,aAAA,kBAAA,IAAmB;AACnBA,aAAA,oBAAA,IAAqB;AACrBA,aAAA,iBAAA,IAAkB;AAClBA,aAAA,gBAAA,IAAiB;AACjBA,aAAA,eAAA,IAAgB;AAChBA,aAAA,iBAAA,IAAkB;AAClBA,aAAA,kBAAA,IAAmB;AACnBA,aAAA,gBAAA,IAAiB;AACjBA,aAAA,cAAA,IAAe;AACfA,aAAA,6BAAA,IAA8B;AAC9BA,aAAA,+BAAA,IAAgC;AAChCA,aAAA,2BAAA,IAA4B;AAC5BA,aAAA,gBAAA,IAAiB;AACjBA,aAAA,aAAA,IAAc;AACdA,aAAA,mBAAA,IAAoB;AACpBA,aAAA,mBAAA,IAAoB;AACpBA,aAAA,gBAAA,IAAiB;AACjBA,aAAA,KAAA,IAAM;AACNA,aAAA,QAAA,IAAS;AACTA,aAAA,aAAA,IAAc;AACdA,aAAA,cAAA,IAAe;AACfA,aAAA,WAAA,IAAY;AACZA,aAAA,cAAA,IAAe;AACfA,aAAA,eAAA,IAAgB;AAChBA,aAAA,iBAAA,IAAkB;AAClBA,aAAA,yBAAA,IAA0B;AAC1BA,aAAA,2BAAA,IAA4B;AAC5BA,aAAA,uBAAA,IAAwB;AACxBA,aAAA,yBAAA,IAA0B;AAC1BA,aAAA,eAAA,IAAgB;AAChBA,aAAA,2BAAA,IAA4B;AAjClB,SAAAA;AAAA,GAAA,aAAA,CAAA,CAAA;"}
|
package/dist/esm/index.d.ts
CHANGED
|
@@ -22,6 +22,8 @@ export { createCapability, defineChatMiddleware, createChatMiddleware, } from '.
|
|
|
22
22
|
export type { Capability, CapabilityHandle, CapabilityContext, CapabilityGetter, CapabilityProvider, } from './activities/chat/middleware/index.js';
|
|
23
23
|
export * from './types.js';
|
|
24
24
|
export { buildBaseUsage, type BaseUsageInput } from './utilities/usage.js';
|
|
25
|
+
export { resolveMediaPrompt } from './utilities/media-prompt.js';
|
|
26
|
+
export type { ResolvedMediaPrompt } from './utilities/media-prompt.js';
|
|
25
27
|
export type { SystemPrompt, NormalizedSystemPrompt } from './system-prompts.js';
|
|
26
28
|
export { normalizeSystemPrompts } from './system-prompts.js';
|
|
27
29
|
export { detectImageMimeType } from './utils.js';
|
package/dist/esm/index.js
CHANGED
|
@@ -14,6 +14,7 @@ import { brandProviderTool } from "./tools/provider-tool.js";
|
|
|
14
14
|
import { combineStrategies, maxIterations, untilFinishReason } from "./activities/chat/agent-loop-strategies.js";
|
|
15
15
|
import { createFrozenRegistry, createToolRegistry } from "./tool-registry.js";
|
|
16
16
|
import { buildBaseUsage } from "./utilities/usage.js";
|
|
17
|
+
import { resolveMediaPrompt } from "./utilities/media-prompt.js";
|
|
17
18
|
import { normalizeSystemPrompts } from "./system-prompts.js";
|
|
18
19
|
import { detectImageMimeType } from "./utils.js";
|
|
19
20
|
import { realtimeToken } from "./realtime/index.js";
|
|
@@ -88,6 +89,7 @@ export {
|
|
|
88
89
|
parsePartialJSON,
|
|
89
90
|
parseWithStandardSchema,
|
|
90
91
|
realtimeToken,
|
|
92
|
+
resolveMediaPrompt,
|
|
91
93
|
streamToText,
|
|
92
94
|
summarize,
|
|
93
95
|
toHttpResponse,
|
package/dist/esm/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;"}
|
package/dist/esm/types.d.ts
CHANGED
|
@@ -1189,6 +1189,76 @@ export interface SummarizationResult {
|
|
|
1189
1189
|
summary: string;
|
|
1190
1190
|
usage: TokenUsage;
|
|
1191
1191
|
}
|
|
1192
|
+
/**
|
|
1193
|
+
* Optional role hint on a media input part (image / video / audio). Adapters
|
|
1194
|
+
* read `metadata.role` to route the part to the provider-specific request
|
|
1195
|
+
* field — e.g. `'mask'` → OpenAI `mask` / fal `mask_url`, `'end_frame'` → fal
|
|
1196
|
+
* `end_image_url`, `'reference'` → fal `reference_image_urls`. When omitted
|
|
1197
|
+
* the adapter falls back to positional routing.
|
|
1198
|
+
*/
|
|
1199
|
+
export type MediaInputRole = 'reference' | 'mask' | 'control' | 'start_frame' | 'end_frame' | 'character';
|
|
1200
|
+
/**
|
|
1201
|
+
* Metadata convention for image / video / audio inputs to media generation.
|
|
1202
|
+
* Carried on `ImagePart.metadata` / `VideoPart.metadata` / `AudioPart.metadata`
|
|
1203
|
+
* when used as conditioning inputs to `generateImage()` or `generateVideo()`.
|
|
1204
|
+
*/
|
|
1205
|
+
export interface MediaInputMetadata {
|
|
1206
|
+
/** Optional role hint disambiguating the part's intent for the adapter */
|
|
1207
|
+
role?: MediaInputRole;
|
|
1208
|
+
/**
|
|
1209
|
+
* Optional user-defined label for this input (e.g. `'woman-in-red-dress'`).
|
|
1210
|
+
* **Informational only** — adapters never read it and the SDK never
|
|
1211
|
+
* rewrites prompt text based on it. Use it to correlate parts with the
|
|
1212
|
+
* references you write in your prompt using the provider's own syntax
|
|
1213
|
+
* (fal's `@Image1`, OpenAI's "image 1", etc.), or for your own
|
|
1214
|
+
* bookkeeping/logging.
|
|
1215
|
+
*/
|
|
1216
|
+
tag?: string;
|
|
1217
|
+
}
|
|
1218
|
+
/**
|
|
1219
|
+
* A single part of a multimodal media-generation prompt. Reuses the chat
|
|
1220
|
+
* content-part shapes: text parts carry the instruction, image / video /
|
|
1221
|
+
* audio parts carry conditioning inputs (with an optional
|
|
1222
|
+
* `metadata.role` hint — see {@link MediaInputRole}).
|
|
1223
|
+
*/
|
|
1224
|
+
export type MediaPromptPart = TextPart | ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata> | AudioPart<MediaInputMetadata>;
|
|
1225
|
+
/**
|
|
1226
|
+
* Prompt accepted by `generateImage()` / `generateVideo()`: a plain string,
|
|
1227
|
+
* or an ordered array of content parts for image-conditioned generation
|
|
1228
|
+
* ("not like this *(image)*, more like this *(image)*"). Part order is
|
|
1229
|
+
* meaningful — adapters with native multimodal prompts (Gemini, OpenRouter)
|
|
1230
|
+
* preserve the interleaving; named-field providers (fal, OpenAI, xAI)
|
|
1231
|
+
* extract the media parts and flatten the text. Text is always sent
|
|
1232
|
+
* verbatim: to reference inputs from the prompt, write the provider's own
|
|
1233
|
+
* syntax yourself (e.g. fal's `@Image1`, OpenAI's "image 1"). An array may
|
|
1234
|
+
* be media-only (e.g. upscalers or pure img2img endpoints that take no
|
|
1235
|
+
* instruction text).
|
|
1236
|
+
*/
|
|
1237
|
+
export type MediaPrompt = string | Array<MediaPromptPart>;
|
|
1238
|
+
/**
|
|
1239
|
+
* Non-text modalities a media-generation model can accept in its prompt.
|
|
1240
|
+
*/
|
|
1241
|
+
export type MediaPromptModality = 'image' | 'video' | 'audio';
|
|
1242
|
+
/** Maps a prompt modality to its content-part type. @internal */
|
|
1243
|
+
interface MediaPartByModality {
|
|
1244
|
+
image: ImagePart<MediaInputMetadata>;
|
|
1245
|
+
video: VideoPart<MediaInputMetadata>;
|
|
1246
|
+
audio: AudioPart<MediaInputMetadata>;
|
|
1247
|
+
}
|
|
1248
|
+
/**
|
|
1249
|
+
* Prompt type narrowed to the modalities a specific model supports.
|
|
1250
|
+
* `MediaPromptFor<never>` (a text-only model) is `string | Array<TextPart>`;
|
|
1251
|
+
* `MediaPromptFor<'image'>` additionally admits image parts, etc. Used by
|
|
1252
|
+
* the activity option types together with the adapter's per-model input
|
|
1253
|
+
* modality map so unsupported parts fail at compile time.
|
|
1254
|
+
*/
|
|
1255
|
+
export type MediaPromptFor<TModalities extends MediaPromptModality = never> = string | Array<TextPart | MediaPartByModality[TModalities]>;
|
|
1256
|
+
/**
|
|
1257
|
+
* Per-model map from model name to the prompt modalities it accepts, used as
|
|
1258
|
+
* an adapter type parameter (`TModelInputModalitiesByName`). Models absent
|
|
1259
|
+
* from the map fall back to the unconstrained {@link MediaPrompt}.
|
|
1260
|
+
*/
|
|
1261
|
+
export type ModelInputModalitiesByName = Record<string, ReadonlyArray<MediaPromptModality>>;
|
|
1192
1262
|
/**
|
|
1193
1263
|
* Options for image generation.
|
|
1194
1264
|
* These are the common options supported across providers.
|
|
@@ -1196,8 +1266,16 @@ export interface SummarizationResult {
|
|
|
1196
1266
|
export interface ImageGenerationOptions<TProviderOptions extends object = object, TSize extends string | undefined = string> {
|
|
1197
1267
|
/** The model to use for image generation */
|
|
1198
1268
|
model: string;
|
|
1199
|
-
/**
|
|
1200
|
-
|
|
1269
|
+
/**
|
|
1270
|
+
* Description of the desired image(s): a plain string, or an ordered array
|
|
1271
|
+
* of content parts for image-conditioned generation (image-to-image,
|
|
1272
|
+
* reference-guided, edit, multi-reference). Media parts may carry
|
|
1273
|
+
* `metadata.role` to disambiguate intent (mask, control, reference, …).
|
|
1274
|
+
* Adapters map parts onto the provider-native request — e.g. Gemini
|
|
1275
|
+
* multimodal `contents`, OpenAI `images.edit()`, fal `image_url` /
|
|
1276
|
+
* `mask_url` — and throw a clear runtime error for unsupported modalities.
|
|
1277
|
+
*/
|
|
1278
|
+
prompt: MediaPrompt;
|
|
1201
1279
|
/** Number of images to generate (default: 1) */
|
|
1202
1280
|
numberOfImages?: number;
|
|
1203
1281
|
/** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
|
|
@@ -1293,15 +1371,26 @@ export interface AudioGenerationResult {
|
|
|
1293
1371
|
*
|
|
1294
1372
|
* @experimental Video generation is an experimental feature and may change.
|
|
1295
1373
|
*/
|
|
1296
|
-
export interface VideoGenerationOptions<TProviderOptions extends object = object, TSize extends string | undefined = string> {
|
|
1374
|
+
export interface VideoGenerationOptions<TProviderOptions extends object = object, TSize extends string | undefined = string, TDuration extends string | number | undefined = number> {
|
|
1297
1375
|
/** The model to use for video generation */
|
|
1298
1376
|
model: string;
|
|
1299
|
-
/**
|
|
1300
|
-
|
|
1377
|
+
/**
|
|
1378
|
+
* Description of the desired video: a plain string, or an ordered array of
|
|
1379
|
+
* content parts for image-conditioned generation. Image parts may carry
|
|
1380
|
+
* `metadata.role` (`'start_frame' | 'end_frame' | 'reference' |
|
|
1381
|
+
* 'character'`) to disambiguate intent; adapters route them onto the
|
|
1382
|
+
* provider-native request (e.g. OpenAI Sora `input_reference`, fal
|
|
1383
|
+
* `image_url` / `end_image_url`) and throw at runtime if unsupported.
|
|
1384
|
+
*/
|
|
1385
|
+
prompt: MediaPrompt;
|
|
1301
1386
|
/** Video size — format depends on the provider (e.g., "16:9", "1280x720") */
|
|
1302
1387
|
size?: TSize;
|
|
1303
|
-
/**
|
|
1304
|
-
|
|
1388
|
+
/**
|
|
1389
|
+
* Video duration in seconds. Adapters that declare a per-model duration
|
|
1390
|
+
* map narrow this to the model's valid union; use
|
|
1391
|
+
* `adapter.snapDuration(seconds)` to coerce raw seconds to a valid value.
|
|
1392
|
+
*/
|
|
1393
|
+
duration?: TDuration;
|
|
1305
1394
|
/** Model-specific options for video generation */
|
|
1306
1395
|
modelOptions?: TProviderOptions;
|
|
1307
1396
|
/**
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { AudioPart, ImagePart, MediaInputMetadata, MediaPrompt, MediaPromptPart, VideoPart } from '../types.js';
|
|
2
|
+
/**
|
|
3
|
+
* A {@link MediaPrompt} decomposed into the views adapters consume.
|
|
4
|
+
*
|
|
5
|
+
* Adapters with native multimodal prompts (Gemini `contents`, OpenRouter
|
|
6
|
+
* chat content parts) consume `parts` to preserve interleaving; named-field
|
|
7
|
+
* providers (fal, OpenAI) consume `text` plus the typed media buckets.
|
|
8
|
+
*
|
|
9
|
+
* Prompt text is **never rewritten**: text parts are concatenated verbatim.
|
|
10
|
+
* Providers that support referencing inputs from the prompt (e.g. fal's
|
|
11
|
+
* `@Image1`, OpenAI's "image 1" prose) expect the user to write that syntax
|
|
12
|
+
* themselves — the SDK does not inject or substitute markers.
|
|
13
|
+
*/
|
|
14
|
+
export interface ResolvedMediaPrompt {
|
|
15
|
+
/**
|
|
16
|
+
* Text parts concatenated verbatim (paragraph-separated). Empty string
|
|
17
|
+
* for media-only prompts.
|
|
18
|
+
*/
|
|
19
|
+
text: string;
|
|
20
|
+
/** The prompt as ordered parts; a string prompt becomes one text part. */
|
|
21
|
+
parts: Array<MediaPromptPart>;
|
|
22
|
+
/** Image parts in prompt order. */
|
|
23
|
+
images: Array<ImagePart<MediaInputMetadata>>;
|
|
24
|
+
/** Video parts in prompt order. */
|
|
25
|
+
videos: Array<VideoPart<MediaInputMetadata>>;
|
|
26
|
+
/** Audio parts in prompt order. */
|
|
27
|
+
audios: Array<AudioPart<MediaInputMetadata>>;
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Decompose a {@link MediaPrompt} into flattened text and per-modality part
|
|
31
|
+
* buckets, preserving prompt order everywhere. This is the single downrev
|
|
32
|
+
* point from the canonical interleaved prompt shape to the named-field
|
|
33
|
+
* request shapes most providers expose.
|
|
34
|
+
*/
|
|
35
|
+
export declare function resolveMediaPrompt(prompt: MediaPrompt): ResolvedMediaPrompt;
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
function resolveMediaPrompt(prompt) {
|
|
2
|
+
if (typeof prompt === "string") {
|
|
3
|
+
const textPart = { type: "text", content: prompt };
|
|
4
|
+
return {
|
|
5
|
+
text: prompt,
|
|
6
|
+
parts: [textPart],
|
|
7
|
+
images: [],
|
|
8
|
+
videos: [],
|
|
9
|
+
audios: []
|
|
10
|
+
};
|
|
11
|
+
}
|
|
12
|
+
const images = [];
|
|
13
|
+
const videos = [];
|
|
14
|
+
const audios = [];
|
|
15
|
+
const textSegments = [];
|
|
16
|
+
for (const part of prompt) {
|
|
17
|
+
switch (part.type) {
|
|
18
|
+
case "text":
|
|
19
|
+
if (part.content) textSegments.push(part.content);
|
|
20
|
+
break;
|
|
21
|
+
case "image":
|
|
22
|
+
images.push(part);
|
|
23
|
+
break;
|
|
24
|
+
case "video":
|
|
25
|
+
videos.push(part);
|
|
26
|
+
break;
|
|
27
|
+
case "audio":
|
|
28
|
+
audios.push(part);
|
|
29
|
+
break;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
return {
|
|
33
|
+
text: textSegments.join("\n\n"),
|
|
34
|
+
parts: prompt,
|
|
35
|
+
images,
|
|
36
|
+
videos,
|
|
37
|
+
audios
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
export {
|
|
41
|
+
resolveMediaPrompt
|
|
42
|
+
};
|
|
43
|
+
//# sourceMappingURL=media-prompt.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"media-prompt.js","sources":["../../../src/utilities/media-prompt.ts"],"sourcesContent":["import type {\n AudioPart,\n ImagePart,\n MediaInputMetadata,\n MediaPrompt,\n MediaPromptPart,\n TextPart,\n VideoPart,\n} from '../types'\n\n/**\n * A {@link MediaPrompt} decomposed into the views adapters consume.\n *\n * Adapters with native multimodal prompts (Gemini `contents`, OpenRouter\n * chat content parts) consume `parts` to preserve interleaving; named-field\n * providers (fal, OpenAI) consume `text` plus the typed media buckets.\n *\n * Prompt text is **never rewritten**: text parts are concatenated verbatim.\n * Providers that support referencing inputs from the prompt (e.g. fal's\n * `@Image1`, OpenAI's \"image 1\" prose) expect the user to write that syntax\n * themselves — the SDK does not inject or substitute markers.\n */\nexport interface ResolvedMediaPrompt {\n /**\n * Text parts concatenated verbatim (paragraph-separated). Empty string\n * for media-only prompts.\n */\n text: string\n /** The prompt as ordered parts; a string prompt becomes one text part. */\n parts: Array<MediaPromptPart>\n /** Image parts in prompt order. */\n images: Array<ImagePart<MediaInputMetadata>>\n /** Video parts in prompt order. */\n videos: Array<VideoPart<MediaInputMetadata>>\n /** Audio parts in prompt order. */\n audios: Array<AudioPart<MediaInputMetadata>>\n}\n\n/**\n * Decompose a {@link MediaPrompt} into flattened text and per-modality part\n * buckets, preserving prompt order everywhere. This is the single downrev\n * point from the canonical interleaved prompt shape to the named-field\n * request shapes most providers expose.\n */\nexport function resolveMediaPrompt(prompt: MediaPrompt): ResolvedMediaPrompt {\n if (typeof prompt === 'string') {\n const textPart: TextPart = { type: 'text', content: prompt }\n return {\n text: prompt,\n parts: [textPart],\n images: [],\n videos: [],\n audios: [],\n }\n }\n\n const images: Array<ImagePart<MediaInputMetadata>> = []\n const videos: Array<VideoPart<MediaInputMetadata>> = []\n const audios: Array<AudioPart<MediaInputMetadata>> = []\n const textSegments: Array<string> = []\n\n for (const part of prompt) {\n switch (part.type) {\n case 'text':\n if (part.content) textSegments.push(part.content)\n break\n case 'image':\n images.push(part)\n break\n case 'video':\n videos.push(part)\n break\n case 'audio':\n audios.push(part)\n break\n }\n }\n\n return {\n text: textSegments.join('\\n\\n'),\n parts: prompt,\n images,\n videos,\n audios,\n }\n}\n"],"names":[],"mappings":"AA4CO,SAAS,mBAAmB,QAA0C;AAC3E,MAAI,OAAO,WAAW,UAAU;AAC9B,UAAM,WAAqB,EAAE,MAAM,QAAQ,SAAS,OAAA;AACpD,WAAO;AAAA,MACL,MAAM;AAAA,MACN,OAAO,CAAC,QAAQ;AAAA,MAChB,QAAQ,CAAA;AAAA,MACR,QAAQ,CAAA;AAAA,MACR,QAAQ,CAAA;AAAA,IAAC;AAAA,EAEb;AAEA,QAAM,SAA+C,CAAA;AACrD,QAAM,SAA+C,CAAA;AACrD,QAAM,SAA+C,CAAA;AACrD,QAAM,eAA8B,CAAA;AAEpC,aAAW,QAAQ,QAAQ;AACzB,YAAQ,KAAK,MAAA;AAAA,MACX,KAAK;AACH,YAAI,KAAK,QAAS,cAAa,KAAK,KAAK,OAAO;AAChD;AAAA,MACF,KAAK;AACH,eAAO,KAAK,IAAI;AAChB;AAAA,MACF,KAAK;AACH,eAAO,KAAK,IAAI;AAChB;AAAA,MACF,KAAK;AACH,eAAO,KAAK,IAAI;AAChB;AAAA,IAAA;AAAA,EAEN;AAEA,SAAO;AAAA,IACL,MAAM,aAAa,KAAK,MAAM;AAAA,IAC9B,OAAO;AAAA,IACP;AAAA,IACA;AAAA,IACA;AAAA,EAAA;AAEJ;"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.32.0",
|
|
4
4
|
"description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -76,7 +76,7 @@
|
|
|
76
76
|
"@ag-ui/core": "^0.0.52",
|
|
77
77
|
"@standard-schema/spec": "^1.1.0",
|
|
78
78
|
"partial-json": "^0.1.7",
|
|
79
|
-
"@tanstack/ai-event-client": "0.6.
|
|
79
|
+
"@tanstack/ai-event-client": "0.6.3"
|
|
80
80
|
},
|
|
81
81
|
"peerDependencies": {
|
|
82
82
|
"@opentelemetry/api": ">=1.9.0"
|
|
@@ -3,8 +3,9 @@ name: ai-core/media-generation
|
|
|
3
3
|
description: >
|
|
4
4
|
Image, audio, video, speech (TTS), and transcription generation using
|
|
5
5
|
activity-specific adapters: generateImage() with openaiImage/geminiImage,
|
|
6
|
-
generateAudio() with geminiAudio/falAudio, generateVideo() with
|
|
7
|
-
polling,
|
|
6
|
+
generateAudio() with geminiAudio/falAudio, generateVideo() with
|
|
7
|
+
openaiVideo/geminiVideo (async polling, per-model typed durations),
|
|
8
|
+
generateSpeech() with openaiSpeech, generateTranscription() with
|
|
8
9
|
openaiTranscription. React hooks: useGenerateImage, useGenerateAudio,
|
|
9
10
|
useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
10
11
|
TanStack Start server function integration with toServerSentEventsResponse.
|
|
@@ -189,6 +190,103 @@ Result shape: `ImageGenerationResult` with `images` array where each entry
|
|
|
189
190
|
has `b64Json?`, `url?`, and `revisedPrompt?`. OpenAI image URLs expire
|
|
190
191
|
after 1 hour -- download or display immediately.
|
|
191
192
|
|
|
193
|
+
#### Image-conditioned generation: multimodal `prompt` parts
|
|
194
|
+
|
|
195
|
+
Both `generateImage()` and `generateVideo()` accept the `prompt` either as
|
|
196
|
+
a plain string or as an ordered array of content parts (`TextPart` /
|
|
197
|
+
`ImagePart` / `VideoPart` / `AudioPart` — the same shapes used elsewhere in
|
|
198
|
+
TanStack AI). Part order is meaningful: natively multimodal providers
|
|
199
|
+
(Gemini, OpenRouter) receive parts in order; named-field providers (OpenAI,
|
|
200
|
+
fal, xAI) extract media parts and flatten the text. Prompt text is always
|
|
201
|
+
sent verbatim — to reference inputs from the prompt, write the provider's
|
|
202
|
+
own syntax (fal `@Image1`, OpenAI "image 1" prose); the SDK never injects
|
|
203
|
+
or rewrites markers. Each media part may carry an optional
|
|
204
|
+
`metadata.role` hint that adapters use to route the part to the
|
|
205
|
+
provider-specific field. The accepted part types are narrowed per model at
|
|
206
|
+
compile time via the adapter's input-modality map.
|
|
207
|
+
|
|
208
|
+
```typescript
|
|
209
|
+
import { generateImage } from '@tanstack/ai'
|
|
210
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
211
|
+
|
|
212
|
+
// Image-to-image (OpenAI gpt-image-2 / gpt-image-1, dall-e-2)
|
|
213
|
+
await generateImage({
|
|
214
|
+
adapter: openaiImage('gpt-image-2'),
|
|
215
|
+
prompt: [
|
|
216
|
+
{ type: 'text', content: 'Turn this into a cinematic product photo' },
|
|
217
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/product.png' } },
|
|
218
|
+
],
|
|
219
|
+
})
|
|
220
|
+
|
|
221
|
+
// Multi-reference (up to 16 for gpt-image models; up to ~14 for Gemini native
|
|
222
|
+
// — a provider limit, not enforced by the SDK)
|
|
223
|
+
await generateImage({
|
|
224
|
+
adapter: openaiImage('gpt-image-2'),
|
|
225
|
+
prompt: [
|
|
226
|
+
{ type: 'text', content: 'Apply the second image as style to the first' },
|
|
227
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/product.png' } },
|
|
228
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/style.png' } },
|
|
229
|
+
],
|
|
230
|
+
})
|
|
231
|
+
|
|
232
|
+
// Inpaint via metadata.role === 'mask' (OpenAI gpt-image models, dall-e-2; fal mask_url)
|
|
233
|
+
await generateImage({
|
|
234
|
+
adapter: openaiImage('gpt-image-2'),
|
|
235
|
+
prompt: [
|
|
236
|
+
{ type: 'text', content: 'Replace the masked region with a tree' },
|
|
237
|
+
{ type: 'image', source: { type: 'url', value: photoUrl } },
|
|
238
|
+
{
|
|
239
|
+
type: 'image',
|
|
240
|
+
source: { type: 'url', value: maskUrl },
|
|
241
|
+
metadata: { role: 'mask' },
|
|
242
|
+
},
|
|
243
|
+
],
|
|
244
|
+
})
|
|
245
|
+
|
|
246
|
+
// Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional end_image_url)
|
|
247
|
+
import { generateVideo } from '@tanstack/ai'
|
|
248
|
+
import { falVideo } from '@tanstack/ai-fal'
|
|
249
|
+
|
|
250
|
+
await generateVideo({
|
|
251
|
+
adapter: falVideo('fal-ai/kling-video/v3/pro/image-to-video'),
|
|
252
|
+
prompt: [
|
|
253
|
+
{ type: 'image', source: { type: 'url', value: firstFrameUrl } },
|
|
254
|
+
{ type: 'text', content: 'Slow cinematic push-in' },
|
|
255
|
+
{
|
|
256
|
+
type: 'image',
|
|
257
|
+
source: { type: 'url', value: lastFrameUrl },
|
|
258
|
+
metadata: { role: 'end_frame' },
|
|
259
|
+
},
|
|
260
|
+
],
|
|
261
|
+
})
|
|
262
|
+
```
|
|
263
|
+
|
|
264
|
+
**Role hints** (`metadata.role`):
|
|
265
|
+
|
|
266
|
+
| Role | Maps to |
|
|
267
|
+
| --------------- | ----------------------------------------------------------------------------------------------------- |
|
|
268
|
+
| `'reference'` | fal `reference_image_urls`; Gemini multimodal part; positional otherwise |
|
|
269
|
+
| `'character'` | Same as `'reference'`; Veo `referenceImages` slot (planned — no Veo adapter yet) |
|
|
270
|
+
| `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
|
|
271
|
+
| `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
|
|
272
|
+
| `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); Veo `image` (planned) |
|
|
273
|
+
| `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); Veo `lastFrame` (planned) |
|
|
274
|
+
|
|
275
|
+
**Provider support matrix:**
|
|
276
|
+
|
|
277
|
+
| Provider | `generateImage` image parts | `generateVideo` image parts |
|
|
278
|
+
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
279
|
+
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
|
|
280
|
+
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | No native Veo adapter yet — deferred to a follow-up. |
|
|
281
|
+
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
|
|
282
|
+
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
|
|
283
|
+
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | n/a |
|
|
284
|
+
| Anthropic | n/a (no image generation API). | n/a |
|
|
285
|
+
|
|
286
|
+
Video and audio prompt parts follow the same `metadata.role` convention
|
|
287
|
+
for video-to-video and lipsync flows on fal; other providers throw when
|
|
288
|
+
they're passed.
|
|
289
|
+
|
|
192
290
|
### 2. Audio Generation (Music, Sound Effects)
|
|
193
291
|
|
|
194
292
|
Distinct from TTS — `generateAudio()` produces non-speech audio content.
|
|
@@ -331,6 +429,31 @@ const stream = generateVideo({
|
|
|
331
429
|
return toServerSentEventsResponse(stream)
|
|
332
430
|
```
|
|
333
431
|
|
|
432
|
+
Google Veo (`@tanstack/ai-gemini`) uses the same jobs/polling flow. Its
|
|
433
|
+
`duration` option is typed per model (e.g. `4 | 6 | 8` for Veo 3.x,
|
|
434
|
+
`5 | 6 | 8` for Veo 2); use `adapter.snapDuration(seconds)` to coerce raw
|
|
435
|
+
seconds and `adapter.availableDurations()` to enumerate the valid set.
|
|
436
|
+
Image prompt parts route by `metadata.role`: first un-roled /
|
|
437
|
+
`'start_frame'` image → input image, `'end_frame'` → `lastFrame`,
|
|
438
|
+
`'reference'` / `'character'` → `referenceImages`:
|
|
439
|
+
|
|
440
|
+
```typescript
|
|
441
|
+
import { geminiVideo } from '@tanstack/ai-gemini'
|
|
442
|
+
|
|
443
|
+
const adapter = geminiVideo('veo-3.1-generate-preview')
|
|
444
|
+
adapter.availableDurations() // { kind: 'discrete', values: [4, 6, 8] }
|
|
445
|
+
|
|
446
|
+
const { jobId } = await generateVideo({
|
|
447
|
+
adapter,
|
|
448
|
+
prompt: 'A golden retriever playing in sunflowers',
|
|
449
|
+
size: '16:9', // Veo sizes are aspect ratios: '16:9' | '9:16'
|
|
450
|
+
duration: adapter.snapDuration(7), // 6
|
|
451
|
+
modelOptions: { resolution: '1080p', generateAudio: true },
|
|
452
|
+
})
|
|
453
|
+
// Note: Veo result URLs require the Google API key to download
|
|
454
|
+
// (x-goog-api-key header or ?key= query parameter).
|
|
455
|
+
```
|
|
456
|
+
|
|
334
457
|
Client hook with job tracking:
|
|
335
458
|
|
|
336
459
|
```tsx
|
|
@@ -607,7 +730,54 @@ generateSpeech({
|
|
|
607
730
|
|
|
608
731
|
> Source: Gemini TTS adapter validation; CodeRabbit review of PR #463.
|
|
609
732
|
|
|
610
|
-
### h.
|
|
733
|
+
### h. HIGH: Passing image prompt parts to a model that doesn't support image-conditioned generation
|
|
734
|
+
|
|
735
|
+
Not every model accepts image-conditioned prompts. The `prompt` type is
|
|
736
|
+
narrowed per model, so passing an image part to a text-only model
|
|
737
|
+
(dall-e-3, Imagen, grok-2-image) is a **compile-time error**; adapters
|
|
738
|
+
also throw a clear runtime error as a backstop, so users learn at call
|
|
739
|
+
time rather than getting silently wrong output.
|
|
740
|
+
|
|
741
|
+
```typescript
|
|
742
|
+
// WRONG — dall-e-3 has no edit/inputs API; image parts are a type error
|
|
743
|
+
generateImage({
|
|
744
|
+
adapter: openaiImage('dall-e-3'),
|
|
745
|
+
prompt: [
|
|
746
|
+
{ type: 'text', content: 'Edit this' },
|
|
747
|
+
{ type: 'image', source: { type: 'url', value: url } }, // ❌ type error
|
|
748
|
+
],
|
|
749
|
+
})
|
|
750
|
+
|
|
751
|
+
// WRONG — Imagen is text-to-image only; same compile-time rejection
|
|
752
|
+
generateImage({
|
|
753
|
+
adapter: geminiImage('imagen-4.0-generate-001'),
|
|
754
|
+
prompt: [
|
|
755
|
+
{ type: 'text', content: 'Edit this' },
|
|
756
|
+
{ type: 'image', source: { type: 'url', value: url } }, // ❌ type error
|
|
757
|
+
],
|
|
758
|
+
})
|
|
759
|
+
|
|
760
|
+
// CORRECT — use a model that supports image-conditioned generation
|
|
761
|
+
generateImage({
|
|
762
|
+
adapter: openaiImage('gpt-image-2'), // edits up to 16 images
|
|
763
|
+
prompt: [
|
|
764
|
+
{ type: 'text', content: 'Edit this' },
|
|
765
|
+
{ type: 'image', source: { type: 'url', value: url } },
|
|
766
|
+
],
|
|
767
|
+
})
|
|
768
|
+
|
|
769
|
+
generateImage({
|
|
770
|
+
adapter: geminiImage('gemini-3.1-flash-image-preview'), // native multimodal
|
|
771
|
+
prompt: [
|
|
772
|
+
{ type: 'text', content: 'Edit this' },
|
|
773
|
+
{ type: 'image', source: { type: 'url', value: url } },
|
|
774
|
+
],
|
|
775
|
+
})
|
|
776
|
+
```
|
|
777
|
+
|
|
778
|
+
> Source: docs/media/image-generation.md, docs/media/video-generation.md.
|
|
779
|
+
|
|
780
|
+
### i. LOW: Writing a logging middleware to see media chunks flow through
|
|
611
781
|
|
|
612
782
|
Every media activity — `generateAudio`, `generateSpeech`,
|
|
613
783
|
`generateTranscription`, `generateImage`, `generateVideo` — accepts the
|