@tanstack/ai-groq 0.4.16 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -48,7 +48,23 @@ const GROQ_CHAT_MODELS = [
48
48
  KIMI_K2_INSTRUCT_0905.name,
49
49
  QWEN3_32B.name
50
50
  ];
51
+ const GROQ_TRANSCRIPTION_MODELS = [
52
+ "whisper-large-v3-turbo",
53
+ "whisper-large-v3"
54
+ ];
55
+ const ORPHEUS_V1_ENGLISH = {
56
+ name: "canopylabs/orpheus-v1-english"
57
+ };
58
+ const ORPHEUS_ARABIC_SAUDI = {
59
+ name: "canopylabs/orpheus-arabic-saudi"
60
+ };
61
+ const GROQ_TTS_MODELS = [
62
+ ORPHEUS_V1_ENGLISH.name,
63
+ ORPHEUS_ARABIC_SAUDI.name
64
+ ];
51
65
  export {
52
- GROQ_CHAT_MODELS
66
+ GROQ_CHAT_MODELS,
67
+ GROQ_TRANSCRIPTION_MODELS,
68
+ GROQ_TTS_MODELS
53
69
  };
54
70
  //# sourceMappingURL=model-meta.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"model-meta.js","sources":["../../src/model-meta.ts"],"sourcesContent":["import type { GroqTextProviderOptions } from './text/text-provider-options'\n\n/**\n * Internal metadata structure describing a Groq model's capabilities and pricing.\n */\ninterface ModelMeta<TProviderOptions = unknown> {\n name: string\n context_window?: number\n max_completion_tokens?: number\n pricing: {\n input?: { normal: number; cached?: number }\n output?: { normal: number }\n }\n supports: {\n input: Array<'text' | 'image' | 'audio'>\n output: Array<'text' | 'audio'>\n endpoints: Array<'chat' | 'tts' | 'transcription' | 'batch'>\n\n features: Array<\n | 'streaming'\n | 'tools'\n | 'json_object'\n | 'browser_search'\n | 'code_execution'\n | 'reasoning'\n | 'content_moderation'\n | 'json_schema'\n | 'vision'\n >\n tools?: ReadonlyArray<never>\n }\n /**\n * Type-level description of which provider options this model supports.\n */\n providerOptions?: TProviderOptions\n}\n\nconst LLAMA_3_3_70B_VERSATILE = {\n name: 'llama-3.3-70b-versatile',\n context_window: 131_072,\n max_completion_tokens: 32_768,\n pricing: {\n input: {\n normal: 0.59,\n },\n output: {\n normal: 0.79,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'tools', 'json_object'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_4_MAVERICK_17B_128E_INSTRUCT = {\n name: 'meta-llama/llama-4-maverick-17b-128e-instruct',\n context_window: 131_072,\n max_completion_tokens: 8_192,\n pricing: {\n input: {\n normal: 0.2,\n },\n output: {\n normal: 0.6,\n },\n },\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'tools', 'json_object', 'json_schema', 'vision'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_4_SCOUT_17B_16E_INSTRUCT = {\n name: 'meta-llama/llama-4-scout-17b-16e-instruct',\n context_window: 131_072,\n max_completion_tokens: 8_192,\n pricing: {\n input: {\n normal: 0.05,\n },\n output: {\n normal: 0.08,\n },\n },\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'tools', 'json_object'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_GUARD_4_12B = {\n name: 'meta-llama/llama-guard-4-12b',\n context_window: 131_072,\n max_completion_tokens: 1024,\n pricing: {\n input: {\n normal: 0.2,\n },\n output: {\n normal: 0.2,\n },\n },\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'json_object', 'content_moderation', 'vision'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_PROMPT_GUARD_2_86M = {\n name: 'meta-llama/llama-prompt-guard-2-86m',\n context_window: 512,\n max_completion_tokens: 512,\n pricing: {\n input: {\n normal: 0.04,\n },\n output: {\n normal: 0.04,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'content_moderation', 'json_object'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_3_1_8B_INSTANT = {\n name: 'llama-3.1-8b-instant',\n context_window: 131_072,\n max_completion_tokens: 131_072,\n pricing: {\n input: {\n normal: 0.05,\n },\n output: {\n normal: 0.08,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'json_object', 'tools'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_PROMPT_GUARD_2_22M = {\n name: 'meta-llama/llama-prompt-guard-2-22m',\n context_window: 512,\n max_completion_tokens: 512,\n pricing: {\n input: {\n normal: 0.03,\n },\n output: {\n normal: 0.03,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'content_moderation'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst GPT_OSS_120B = {\n name: 'openai/gpt-oss-120b',\n context_window: 131_072,\n max_completion_tokens: 65_536,\n pricing: {\n input: {\n normal: 0.15,\n cached: 0.075,\n },\n output: {\n normal: 0.6,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: [\n 'streaming',\n 'json_object',\n 'json_schema',\n 'tools',\n 'browser_search',\n 'code_execution',\n 'reasoning',\n ],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst GPT_OSS_SAFEGUARD_20B = {\n name: 'openai/gpt-oss-safeguard-20b',\n context_window: 131_072,\n max_completion_tokens: 65_536,\n pricing: {\n input: {\n normal: 0.075,\n cached: 0.037,\n },\n output: {\n normal: 0.3,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: [\n 'streaming',\n 'tools',\n 'browser_search',\n 'code_execution',\n 'json_object',\n 'json_schema',\n 'reasoning',\n 'content_moderation',\n ],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst GPT_OSS_20B = {\n name: 'openai/gpt-oss-20b',\n context_window: 131_072,\n max_completion_tokens: 65_536,\n pricing: {\n input: {\n normal: 0.075,\n cached: 0.037,\n },\n output: {\n normal: 0.3,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: [\n 'streaming',\n 'browser_search',\n 'code_execution',\n 'json_object',\n 'json_schema',\n 'reasoning',\n 'tools',\n ],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst KIMI_K2_INSTRUCT_0905 = {\n name: 'moonshotai/kimi-k2-instruct-0905',\n context_window: 262_144,\n max_completion_tokens: 16_384,\n pricing: {\n input: {\n normal: 1,\n cached: 0.5,\n },\n output: {\n normal: 3,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'tools', 'json_object', 'json_schema'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst QWEN3_32B = {\n name: 'qwen/qwen3-32b',\n context_window: 131_072,\n max_completion_tokens: 40_960,\n pricing: {\n input: {\n normal: 0.29,\n },\n output: {\n normal: 0.59,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'json_object', 'tools', 'reasoning'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\n/**\n * All supported Groq chat model identifiers.\n */\nexport const GROQ_CHAT_MODELS = [\n LLAMA_3_1_8B_INSTANT.name,\n LLAMA_3_3_70B_VERSATILE.name,\n LLAMA_4_MAVERICK_17B_128E_INSTRUCT.name,\n LLAMA_4_SCOUT_17B_16E_INSTRUCT.name,\n LLAMA_GUARD_4_12B.name,\n LLAMA_PROMPT_GUARD_2_86M.name,\n LLAMA_PROMPT_GUARD_2_22M.name,\n GPT_OSS_20B.name,\n GPT_OSS_120B.name,\n GPT_OSS_SAFEGUARD_20B.name,\n KIMI_K2_INSTRUCT_0905.name,\n QWEN3_32B.name,\n] as const\n\n/**\n * Union type of all supported Groq chat model names.\n */\nexport type GroqChatModels = (typeof GROQ_CHAT_MODELS)[number]\n\n/**\n * Type-only map from Groq chat model name to its supported input modalities.\n */\nexport type GroqModelInputModalitiesByName = {\n [LLAMA_3_1_8B_INSTANT.name]: typeof LLAMA_3_1_8B_INSTANT.supports.input\n [LLAMA_3_3_70B_VERSATILE.name]: typeof LLAMA_3_3_70B_VERSATILE.supports.input\n [LLAMA_4_MAVERICK_17B_128E_INSTRUCT.name]: typeof LLAMA_4_MAVERICK_17B_128E_INSTRUCT.supports.input\n [LLAMA_4_SCOUT_17B_16E_INSTRUCT.name]: typeof LLAMA_4_SCOUT_17B_16E_INSTRUCT.supports.input\n [LLAMA_GUARD_4_12B.name]: typeof LLAMA_GUARD_4_12B.supports.input\n [LLAMA_PROMPT_GUARD_2_86M.name]: typeof LLAMA_PROMPT_GUARD_2_86M.supports.input\n [LLAMA_PROMPT_GUARD_2_22M.name]: typeof LLAMA_PROMPT_GUARD_2_22M.supports.input\n [GPT_OSS_20B.name]: typeof GPT_OSS_20B.supports.input\n [GPT_OSS_120B.name]: typeof GPT_OSS_120B.supports.input\n [GPT_OSS_SAFEGUARD_20B.name]: typeof GPT_OSS_SAFEGUARD_20B.supports.input\n [KIMI_K2_INSTRUCT_0905.name]: typeof KIMI_K2_INSTRUCT_0905.supports.input\n [QWEN3_32B.name]: typeof QWEN3_32B.supports.input\n}\n\n/**\n * Type-only map from Groq chat model name to its provider options type.\n */\nexport type GroqChatModelProviderOptionsByName = {\n [K in (typeof GROQ_CHAT_MODELS)[number]]: GroqTextProviderOptions\n}\n\n/**\n * Type-only map from Groq chat model name to its supported provider tools.\n * Groq exposes no provider-specific tool factories, so every model gets an\n * empty tuple. This ensures that passing an Anthropic/OpenAI ProviderTool to\n * a Groq adapter produces a compile-time type error.\n */\nexport type GroqChatModelToolCapabilitiesByName = {\n [LLAMA_3_1_8B_INSTANT.name]: typeof LLAMA_3_1_8B_INSTANT.supports.tools\n [LLAMA_3_3_70B_VERSATILE.name]: typeof LLAMA_3_3_70B_VERSATILE.supports.tools\n [LLAMA_4_MAVERICK_17B_128E_INSTRUCT.name]: typeof LLAMA_4_MAVERICK_17B_128E_INSTRUCT.supports.tools\n [LLAMA_4_SCOUT_17B_16E_INSTRUCT.name]: typeof LLAMA_4_SCOUT_17B_16E_INSTRUCT.supports.tools\n [LLAMA_GUARD_4_12B.name]: typeof LLAMA_GUARD_4_12B.supports.tools\n [LLAMA_PROMPT_GUARD_2_86M.name]: typeof LLAMA_PROMPT_GUARD_2_86M.supports.tools\n [LLAMA_PROMPT_GUARD_2_22M.name]: typeof LLAMA_PROMPT_GUARD_2_22M.supports.tools\n [GPT_OSS_20B.name]: typeof GPT_OSS_20B.supports.tools\n [GPT_OSS_120B.name]: typeof GPT_OSS_120B.supports.tools\n [GPT_OSS_SAFEGUARD_20B.name]: typeof GPT_OSS_SAFEGUARD_20B.supports.tools\n [KIMI_K2_INSTRUCT_0905.name]: typeof KIMI_K2_INSTRUCT_0905.supports.tools\n [QWEN3_32B.name]: typeof QWEN3_32B.supports.tools\n}\n\n/**\n * Resolves the provider options type for a specific Groq model.\n * Falls back to generic GroqTextProviderOptions for unknown models.\n */\nexport type ResolveProviderOptions<TModel extends string> =\n TModel extends keyof GroqChatModelProviderOptionsByName\n ? GroqChatModelProviderOptionsByName[TModel]\n : GroqTextProviderOptions\n\n/**\n * Resolve input modalities for a specific model.\n * If the model has explicit modalities in the map, use those; otherwise use text only.\n */\nexport type ResolveInputModalities<TModel extends string> =\n TModel extends keyof GroqModelInputModalitiesByName\n ? GroqModelInputModalitiesByName[TModel]\n : readonly ['text']\n"],"names":[],"mappings":"AAqCA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAkBR;AAEA,MAAM,qCAAqC;AAAA,EACzC,MAAM;AAkBR;AAEA,MAAM,iCAAiC;AAAA,EACrC,MAAM;AAkBR;AAEA,MAAM,oBAAoB;AAAA,EACxB,MAAM;AAkBR;AAEA,MAAM,2BAA2B;AAAA,EAC/B,MAAM;AAkBR;AAEA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAkBR;AAEA,MAAM,2BAA2B;AAAA,EAC/B,MAAM;AAkBR;AAEA,MAAM,eAAe;AAAA,EACnB,MAAM;AA2BR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AA4BR;AAEA,MAAM,cAAc;AAAA,EAClB,MAAM;AA2BR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAmBR;AAEA,MAAM,YAAY;AAAA,EAChB,MAAM;AAkBR;AAKO,MAAM,mBAAmB;AAAA,EAC9B,qBAAqB;AAAA,EACrB,wBAAwB;AAAA,EACxB,mCAAmC;AAAA,EACnC,+BAA+B;AAAA,EAC/B,kBAAkB;AAAA,EAClB,yBAAyB;AAAA,EACzB,yBAAyB;AAAA,EACzB,YAAY;AAAA,EACZ,aAAa;AAAA,EACb,sBAAsB;AAAA,EACtB,sBAAsB;AAAA,EACtB,UAAU;AACZ;"}
1
+ {"version":3,"file":"model-meta.js","sources":["../../src/model-meta.ts"],"sourcesContent":["import type { GroqTextProviderOptions } from './text/text-provider-options'\nimport type { GroqTTSProviderOptions } from './audio/tts-provider-options'\n\n/**\n * Internal metadata structure describing a Groq model's capabilities and pricing.\n */\ninterface ModelMeta<TProviderOptions = unknown> {\n name: string\n context_window?: number\n max_completion_tokens?: number\n pricing: {\n input?: { normal: number; cached?: number }\n output?: { normal: number }\n }\n supports: {\n input: Array<'text' | 'image' | 'audio'>\n output: Array<'text' | 'audio'>\n endpoints: Array<'chat' | 'tts' | 'transcription' | 'batch'>\n\n features: Array<\n | 'streaming'\n | 'tools'\n | 'json_object'\n | 'browser_search'\n | 'code_execution'\n | 'reasoning'\n | 'content_moderation'\n | 'json_schema'\n | 'vision'\n >\n tools?: ReadonlyArray<never>\n }\n /**\n * Type-level description of which provider options this model supports.\n */\n providerOptions?: TProviderOptions\n}\n\nconst LLAMA_3_3_70B_VERSATILE = {\n name: 'llama-3.3-70b-versatile',\n context_window: 131_072,\n max_completion_tokens: 32_768,\n pricing: {\n input: {\n normal: 0.59,\n },\n output: {\n normal: 0.79,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'tools', 'json_object'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_4_MAVERICK_17B_128E_INSTRUCT = {\n name: 'meta-llama/llama-4-maverick-17b-128e-instruct',\n context_window: 131_072,\n max_completion_tokens: 8_192,\n pricing: {\n input: {\n normal: 0.2,\n },\n output: {\n normal: 0.6,\n },\n },\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'tools', 'json_object', 'json_schema', 'vision'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_4_SCOUT_17B_16E_INSTRUCT = {\n name: 'meta-llama/llama-4-scout-17b-16e-instruct',\n context_window: 131_072,\n max_completion_tokens: 8_192,\n pricing: {\n input: {\n normal: 0.05,\n },\n output: {\n normal: 0.08,\n },\n },\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'tools', 'json_object'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_GUARD_4_12B = {\n name: 'meta-llama/llama-guard-4-12b',\n context_window: 131_072,\n max_completion_tokens: 1024,\n pricing: {\n input: {\n normal: 0.2,\n },\n output: {\n normal: 0.2,\n },\n },\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'json_object', 'content_moderation', 'vision'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_PROMPT_GUARD_2_86M = {\n name: 'meta-llama/llama-prompt-guard-2-86m',\n context_window: 512,\n max_completion_tokens: 512,\n pricing: {\n input: {\n normal: 0.04,\n },\n output: {\n normal: 0.04,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'content_moderation', 'json_object'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_3_1_8B_INSTANT = {\n name: 'llama-3.1-8b-instant',\n context_window: 131_072,\n max_completion_tokens: 131_072,\n pricing: {\n input: {\n normal: 0.05,\n },\n output: {\n normal: 0.08,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'json_object', 'tools'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst LLAMA_PROMPT_GUARD_2_22M = {\n name: 'meta-llama/llama-prompt-guard-2-22m',\n context_window: 512,\n max_completion_tokens: 512,\n pricing: {\n input: {\n normal: 0.03,\n },\n output: {\n normal: 0.03,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'content_moderation'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst GPT_OSS_120B = {\n name: 'openai/gpt-oss-120b',\n context_window: 131_072,\n max_completion_tokens: 65_536,\n pricing: {\n input: {\n normal: 0.15,\n cached: 0.075,\n },\n output: {\n normal: 0.6,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: [\n 'streaming',\n 'json_object',\n 'json_schema',\n 'tools',\n 'browser_search',\n 'code_execution',\n 'reasoning',\n ],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst GPT_OSS_SAFEGUARD_20B = {\n name: 'openai/gpt-oss-safeguard-20b',\n context_window: 131_072,\n max_completion_tokens: 65_536,\n pricing: {\n input: {\n normal: 0.075,\n cached: 0.037,\n },\n output: {\n normal: 0.3,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: [\n 'streaming',\n 'tools',\n 'browser_search',\n 'code_execution',\n 'json_object',\n 'json_schema',\n 'reasoning',\n 'content_moderation',\n ],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst GPT_OSS_20B = {\n name: 'openai/gpt-oss-20b',\n context_window: 131_072,\n max_completion_tokens: 65_536,\n pricing: {\n input: {\n normal: 0.075,\n cached: 0.037,\n },\n output: {\n normal: 0.3,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: [\n 'streaming',\n 'browser_search',\n 'code_execution',\n 'json_object',\n 'json_schema',\n 'reasoning',\n 'tools',\n ],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst KIMI_K2_INSTRUCT_0905 = {\n name: 'moonshotai/kimi-k2-instruct-0905',\n context_window: 262_144,\n max_completion_tokens: 16_384,\n pricing: {\n input: {\n normal: 1,\n cached: 0.5,\n },\n output: {\n normal: 3,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'tools', 'json_object', 'json_schema'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\nconst QWEN3_32B = {\n name: 'qwen/qwen3-32b',\n context_window: 131_072,\n max_completion_tokens: 40_960,\n pricing: {\n input: {\n normal: 0.29,\n },\n output: {\n normal: 0.59,\n },\n },\n supports: {\n input: ['text'],\n output: ['text'],\n endpoints: ['chat'],\n features: ['streaming', 'json_object', 'tools', 'reasoning'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta<GroqTextProviderOptions>\n\n/**\n * All supported Groq chat model identifiers.\n */\nexport const GROQ_CHAT_MODELS = [\n LLAMA_3_1_8B_INSTANT.name,\n LLAMA_3_3_70B_VERSATILE.name,\n LLAMA_4_MAVERICK_17B_128E_INSTRUCT.name,\n LLAMA_4_SCOUT_17B_16E_INSTRUCT.name,\n LLAMA_GUARD_4_12B.name,\n LLAMA_PROMPT_GUARD_2_86M.name,\n LLAMA_PROMPT_GUARD_2_22M.name,\n GPT_OSS_20B.name,\n GPT_OSS_120B.name,\n GPT_OSS_SAFEGUARD_20B.name,\n KIMI_K2_INSTRUCT_0905.name,\n QWEN3_32B.name,\n] as const\n\n/**\n * Union type of all supported Groq chat model names.\n */\nexport type GroqChatModels = (typeof GROQ_CHAT_MODELS)[number]\n\n/**\n * Type-only map from Groq chat model name to its supported input modalities.\n */\nexport type GroqModelInputModalitiesByName = {\n [LLAMA_3_1_8B_INSTANT.name]: typeof LLAMA_3_1_8B_INSTANT.supports.input\n [LLAMA_3_3_70B_VERSATILE.name]: typeof LLAMA_3_3_70B_VERSATILE.supports.input\n [LLAMA_4_MAVERICK_17B_128E_INSTRUCT.name]: typeof LLAMA_4_MAVERICK_17B_128E_INSTRUCT.supports.input\n [LLAMA_4_SCOUT_17B_16E_INSTRUCT.name]: typeof LLAMA_4_SCOUT_17B_16E_INSTRUCT.supports.input\n [LLAMA_GUARD_4_12B.name]: typeof LLAMA_GUARD_4_12B.supports.input\n [LLAMA_PROMPT_GUARD_2_86M.name]: typeof LLAMA_PROMPT_GUARD_2_86M.supports.input\n [LLAMA_PROMPT_GUARD_2_22M.name]: typeof LLAMA_PROMPT_GUARD_2_22M.supports.input\n [GPT_OSS_20B.name]: typeof GPT_OSS_20B.supports.input\n [GPT_OSS_120B.name]: typeof GPT_OSS_120B.supports.input\n [GPT_OSS_SAFEGUARD_20B.name]: typeof GPT_OSS_SAFEGUARD_20B.supports.input\n [KIMI_K2_INSTRUCT_0905.name]: typeof KIMI_K2_INSTRUCT_0905.supports.input\n [QWEN3_32B.name]: typeof QWEN3_32B.supports.input\n}\n\n/**\n * Type-only map from Groq chat model name to its provider options type.\n */\nexport type GroqChatModelProviderOptionsByName = {\n [K in (typeof GROQ_CHAT_MODELS)[number]]: GroqTextProviderOptions\n}\n\n/**\n * Type-only map from Groq chat model name to its supported provider tools.\n * Groq exposes no provider-specific tool factories, so every model gets an\n * empty tuple. This ensures that passing an Anthropic/OpenAI ProviderTool to\n * a Groq adapter produces a compile-time type error.\n */\nexport type GroqChatModelToolCapabilitiesByName = {\n [LLAMA_3_1_8B_INSTANT.name]: typeof LLAMA_3_1_8B_INSTANT.supports.tools\n [LLAMA_3_3_70B_VERSATILE.name]: typeof LLAMA_3_3_70B_VERSATILE.supports.tools\n [LLAMA_4_MAVERICK_17B_128E_INSTRUCT.name]: typeof LLAMA_4_MAVERICK_17B_128E_INSTRUCT.supports.tools\n [LLAMA_4_SCOUT_17B_16E_INSTRUCT.name]: typeof LLAMA_4_SCOUT_17B_16E_INSTRUCT.supports.tools\n [LLAMA_GUARD_4_12B.name]: typeof LLAMA_GUARD_4_12B.supports.tools\n [LLAMA_PROMPT_GUARD_2_86M.name]: typeof LLAMA_PROMPT_GUARD_2_86M.supports.tools\n [LLAMA_PROMPT_GUARD_2_22M.name]: typeof LLAMA_PROMPT_GUARD_2_22M.supports.tools\n [GPT_OSS_20B.name]: typeof GPT_OSS_20B.supports.tools\n [GPT_OSS_120B.name]: typeof GPT_OSS_120B.supports.tools\n [GPT_OSS_SAFEGUARD_20B.name]: typeof GPT_OSS_SAFEGUARD_20B.supports.tools\n [KIMI_K2_INSTRUCT_0905.name]: typeof KIMI_K2_INSTRUCT_0905.supports.tools\n [QWEN3_32B.name]: typeof QWEN3_32B.supports.tools\n}\n\n/**\n * Type-only map from Groq TTS model name to its provider options type.\n */\nexport type GroqTTSModelProviderOptionsByName = {\n [K in GroqTTSModel]: GroqTTSProviderOptions\n}\n\n/**\n * Resolves the provider options type for a specific Groq model.\n * Checks TTS models first, then chat models, then falls back to generic options.\n */\nexport type ResolveProviderOptions<TModel extends string> =\n TModel extends GroqTTSModel\n ? GroqTTSProviderOptions\n : TModel extends keyof GroqChatModelProviderOptionsByName\n ? GroqChatModelProviderOptionsByName[TModel]\n : GroqTextProviderOptions\n\n/**\n * Resolve input modalities for a specific model.\n * If the model has explicit modalities in the map, use those; otherwise use text only.\n */\nexport type ResolveInputModalities<TModel extends string> =\n TModel extends keyof GroqModelInputModalitiesByName\n ? GroqModelInputModalitiesByName[TModel]\n : readonly ['text']\n\n/**\n * All supported Groq transcription model identifiers.\n */\nexport const GROQ_TRANSCRIPTION_MODELS = [\n 'whisper-large-v3-turbo',\n 'whisper-large-v3',\n] as const\n\n/**\n * Union type of all supported Groq transcription model names.\n */\nexport type GroqTranscriptionModel = (typeof GROQ_TRANSCRIPTION_MODELS)[number]\n\n// ============================================================================\n// TTS Models\n// ============================================================================\n\nconst ORPHEUS_V1_ENGLISH = {\n name: 'canopylabs/orpheus-v1-english',\n pricing: {\n input: {\n normal: 22,\n },\n },\n supports: {\n input: ['text'],\n output: ['audio'],\n endpoints: ['tts'],\n features: [],\n },\n} as const satisfies ModelMeta<GroqTTSProviderOptions>\n\nconst ORPHEUS_ARABIC_SAUDI = {\n name: 'canopylabs/orpheus-arabic-saudi',\n pricing: {\n input: {\n normal: 40,\n },\n },\n supports: {\n input: ['text'],\n output: ['audio'],\n endpoints: ['tts'],\n features: [],\n },\n} as const satisfies ModelMeta<GroqTTSProviderOptions>\n\n/**\n * All supported Groq TTS model identifiers.\n */\nexport const GROQ_TTS_MODELS = [\n ORPHEUS_V1_ENGLISH.name,\n ORPHEUS_ARABIC_SAUDI.name,\n] as const\n\n/**\n * Union type of all supported Groq TTS model names.\n */\nexport type GroqTTSModel = (typeof GROQ_TTS_MODELS)[number]\n"],"names":[],"mappings":"AAsCA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAkBR;AAEA,MAAM,qCAAqC;AAAA,EACzC,MAAM;AAkBR;AAEA,MAAM,iCAAiC;AAAA,EACrC,MAAM;AAkBR;AAEA,MAAM,oBAAoB;AAAA,EACxB,MAAM;AAkBR;AAEA,MAAM,2BAA2B;AAAA,EAC/B,MAAM;AAkBR;AAEA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAkBR;AAEA,MAAM,2BAA2B;AAAA,EAC/B,MAAM;AAkBR;AAEA,MAAM,eAAe;AAAA,EACnB,MAAM;AA2BR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AA4BR;AAEA,MAAM,cAAc;AAAA,EAClB,MAAM;AA2BR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAmBR;AAEA,MAAM,YAAY;AAAA,EAChB,MAAM;AAkBR;AAKO,MAAM,mBAAmB;AAAA,EAC9B,qBAAqB;AAAA,EACrB,wBAAwB;AAAA,EACxB,mCAAmC;AAAA,EACnC,+BAA+B;AAAA,EAC/B,kBAAkB;AAAA,EAClB,yBAAyB;AAAA,EACzB,yBAAyB;AAAA,EACzB,YAAY;AAAA,EACZ,aAAa;AAAA,EACb,sBAAsB;AAAA,EACtB,sBAAsB;AAAA,EACtB,UAAU;AACZ;AAmFO,MAAM,4BAA4B;AAAA,EACvC;AAAA,EACA;AACF;AAWA,MAAM,qBAAqB;AAAA,EACzB,MAAM;AAYR;AAEA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAYR;AAKO,MAAM,kBAAkB;AAAA,EAC7B,mBAAmB;AAAA,EACnB,qBAAqB;AACvB;"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai-groq",
3
- "version": "0.4.16",
3
+ "version": "0.5.1",
4
4
  "description": "Groq adapter for TanStack AI low-latency chat, tool calling, and structured outputs.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -53,12 +53,12 @@
53
53
  },
54
54
  "peerDependencies": {
55
55
  "zod": "^4.0.0",
56
- "@tanstack/ai": "^0.39.0"
56
+ "@tanstack/ai": "^0.40.0"
57
57
  },
58
58
  "dependencies": {
59
59
  "openai": "^6.41.0",
60
60
  "@tanstack/ai-utils": "0.3.1",
61
- "@tanstack/openai-base": "0.9.6"
61
+ "@tanstack/openai-base": "0.9.7"
62
62
  },
63
63
  "scripts": {
64
64
  "build": "vite build",
@@ -0,0 +1,334 @@
1
+ import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
2
+ import { base64ToArrayBuffer, generateId } from '@tanstack/ai-utils'
3
+ import { getGroqApiKeyFromEnv, withGroqDefaults } from '../utils/client'
4
+ import type {
5
+ TranscriptionOptions,
6
+ TranscriptionResult,
7
+ TranscriptionSegment,
8
+ } from '@tanstack/ai'
9
+ import type { GroqTranscriptionModel } from '../model-meta'
10
+ import type { GroqTranscriptionProviderOptions } from '../audio/transcription-provider-options'
11
+ import type { GroqClientConfig } from '../utils/client'
12
+
13
+ /**
14
+ * Configuration for the Groq Transcription adapter.
15
+ */
16
+ export interface GroqTranscriptionConfig extends GroqClientConfig {}
17
+
18
+ /**
19
+ * Flattens the `openai` SDK's `HeadersLike` config value into a plain record so
20
+ * it can be merged into the raw `fetch` request this adapter issues. Handles
21
+ * the shapes callers actually pass (`Headers`, an entries array, or a plain
22
+ * object); null/undefined values are dropped.
23
+ *
24
+ * ponytail: doesn't unwrap the SDK's internal `NullableHeaders` class; forward
25
+ * that shape here if the SDK ever hands it to adapter config.
26
+ */
27
+ function normalizeHeaders(
28
+ headers: GroqTranscriptionConfig['defaultHeaders'],
29
+ ): Record<string, string> {
30
+ const out: Record<string, string> = {}
31
+ if (!headers) return out
32
+ const assign = (key: string, value: unknown) => {
33
+ if (value != null) out[key] = String(value)
34
+ }
35
+ if (headers instanceof Headers) {
36
+ headers.forEach((value, key) => assign(key, value))
37
+ } else if (Array.isArray(headers)) {
38
+ for (const [key, value] of headers) assign(key, value)
39
+ } else {
40
+ for (const [key, value] of Object.entries(headers)) assign(key, value)
41
+ }
42
+ return out
43
+ }
44
+
45
+ // Shape of Groq's verbose_json transcription response
46
+ interface GroqVerboseTranscriptionResponse {
47
+ task?: string
48
+ language?: string
49
+ duration?: number
50
+ text: string
51
+ segments?: Array<{
52
+ id: number
53
+ seek?: number
54
+ start: number
55
+ end: number
56
+ text: string
57
+ tokens?: Array<number>
58
+ temperature?: number
59
+ avg_logprob: number
60
+ compression_ratio?: number
61
+ no_speech_prob?: number
62
+ }>
63
+ words?: Array<{ word: string; start: number; end: number }>
64
+ x_groq?: { id?: string }
65
+ }
66
+
67
+ // Shape of Groq's json transcription response
68
+ interface GroqJsonTranscriptionResponse {
69
+ text: string
70
+ x_groq?: { id?: string }
71
+ }
72
+
73
+ /**
74
+ * Groq Transcription (Speech-to-Text) Adapter
75
+ *
76
+ * Tree-shakeable adapter for Groq audio transcription. Supports
77
+ * whisper-large-v3 and whisper-large-v3-turbo.
78
+ *
79
+ * Features:
80
+ * - Audio file uploads (File, Blob, ArrayBuffer, base64/data URL)
81
+ * - Remote audio URLs passed directly via Groq's `url` field — no upload needed
82
+ * - Verbose JSON response with segment and word timestamps
83
+ * - Language detection or specification (ISO-639-1)
84
+ * - Confidence scores derived from segment avg_logprob
85
+ */
86
+ export class GroqTranscriptionAdapter<
87
+ TModel extends GroqTranscriptionModel,
88
+ > extends BaseTranscriptionAdapter<TModel, GroqTranscriptionProviderOptions> {
89
+ readonly name = 'groq' as const
90
+
91
+ private readonly apiKey: string
92
+ private readonly baseURL: string
93
+ private readonly defaultHeaders: Record<string, string>
94
+
95
+ constructor(config: GroqTranscriptionConfig, model: TModel) {
96
+ super(model, {})
97
+ const resolved = withGroqDefaults(config)
98
+ this.apiKey = resolved.apiKey
99
+ this.baseURL = resolved.baseURL ?? 'https://api.groq.com/openai/v1'
100
+ this.defaultHeaders = normalizeHeaders(resolved.defaultHeaders)
101
+ }
102
+
103
+ async transcribe(
104
+ options: TranscriptionOptions<GroqTranscriptionProviderOptions>,
105
+ ): Promise<TranscriptionResult> {
106
+ const { model, audio, language, prompt, responseFormat, modelOptions } =
107
+ options
108
+
109
+ // Groq's transcription endpoint only accepts 'json', 'text', and
110
+ // 'verbose_json'. Reject 'srt'/'vtt' up front so callers get a clear
111
+ // message instead of an opaque Groq HTTP error.
112
+ if (responseFormat === 'srt' || responseFormat === 'vtt') {
113
+ throw new Error(
114
+ `Groq transcription does not support responseFormat='${responseFormat}'. ` +
115
+ `Supported values: 'json', 'text', 'verbose_json'.`,
116
+ )
117
+ }
118
+
119
+ // Default to verbose_json so callers get language, duration, and timestamps
120
+ // without having to opt in explicitly. Both Groq whisper models support it.
121
+ const effectiveFormat = responseFormat ?? 'verbose_json'
122
+ const useVerbose = effectiveFormat === 'verbose_json'
123
+
124
+ const form = new FormData()
125
+ form.append('model', model)
126
+ form.append('response_format', effectiveFormat)
127
+ if (language !== undefined) form.append('language', language)
128
+ if (prompt !== undefined) form.append('prompt', prompt)
129
+ if (modelOptions?.temperature !== undefined) {
130
+ form.append('temperature', String(modelOptions.temperature))
131
+ }
132
+ if (modelOptions?.timestamp_granularities !== undefined) {
133
+ for (const g of modelOptions.timestamp_granularities) {
134
+ form.append('timestamp_granularities[]', g)
135
+ }
136
+ }
137
+
138
+ // HTTP/HTTPS URLs are forwarded directly via Groq's `url` field, which
139
+ // avoids a round-trip upload. All other inputs (File, Blob, ArrayBuffer,
140
+ // base64, data URL) are converted to a File and sent as `file`.
141
+ if (typeof audio === 'string' && /^https?:\/\//.test(audio)) {
142
+ form.append('url', audio)
143
+ } else {
144
+ form.append('file', this.prepareAudioFile(audio))
145
+ }
146
+
147
+ try {
148
+ options.logger.request(
149
+ `activity=transcription provider=${this.name} model=${model} verbose=${useVerbose}`,
150
+ { provider: this.name, model },
151
+ )
152
+
153
+ const response = await fetch(`${this.baseURL}/audio/transcriptions`, {
154
+ method: 'POST',
155
+ headers: {
156
+ ...this.defaultHeaders,
157
+ Authorization: `Bearer ${this.apiKey}`,
158
+ },
159
+ body: form,
160
+ })
161
+
162
+ if (!response.ok) {
163
+ const body = await response
164
+ .json()
165
+ .catch(() => null as Record<string, unknown> | null)
166
+ const message =
167
+ (body?.error as { message?: string } | undefined)?.message ??
168
+ `Groq API error ${response.status}`
169
+ throw new Error(message)
170
+ }
171
+
172
+ if (useVerbose) {
173
+ const data = (await response.json()) as GroqVerboseTranscriptionResponse
174
+ const requestId = data.x_groq?.id ?? generateId(this.name)
175
+
176
+ // `TranscriptionResult` declares optional fields without `| undefined`,
177
+ // so under exactOptionalPropertyTypes we must omit absent fields rather
178
+ // than assigning `undefined`.
179
+ const segments = data.segments?.map(
180
+ (seg): TranscriptionSegment => ({
181
+ id: seg.id,
182
+ start: seg.start,
183
+ end: seg.end,
184
+ text: seg.text,
185
+ confidence: Math.exp(seg.avg_logprob),
186
+ }),
187
+ )
188
+ const words = data.words?.map((w) => ({
189
+ word: w.word,
190
+ start: w.start,
191
+ end: w.end,
192
+ }))
193
+
194
+ return {
195
+ id: requestId,
196
+ model,
197
+ text: data.text,
198
+ ...(data.language !== undefined && { language: data.language }),
199
+ ...(data.duration !== undefined && { duration: data.duration }),
200
+ ...(segments !== undefined && { segments }),
201
+ ...(words !== undefined && { words }),
202
+ }
203
+ } else if (effectiveFormat === 'text') {
204
+ const text = await response.text()
205
+ return {
206
+ id: generateId(this.name),
207
+ model,
208
+ text,
209
+ ...(language !== undefined && { language }),
210
+ }
211
+ } else {
212
+ const data = (await response.json()) as GroqJsonTranscriptionResponse
213
+ return {
214
+ id: data.x_groq?.id ?? generateId(this.name),
215
+ model,
216
+ text: data.text,
217
+ ...(language !== undefined && { language }),
218
+ }
219
+ }
220
+ } catch (error: unknown) {
221
+ options.logger.errors(`${this.name}.transcribe fatal`, {
222
+ error,
223
+ source: `${this.name}.transcribe`,
224
+ })
225
+ throw error
226
+ }
227
+ }
228
+
229
+ private prepareAudioFile(audio: string | File | Blob | ArrayBuffer): File {
230
+ if (typeof File !== 'undefined' && audio instanceof File) {
231
+ return audio
232
+ }
233
+ if (typeof Blob !== 'undefined' && audio instanceof Blob) {
234
+ this.ensureFileSupport()
235
+ return new File([audio], 'audio.mp3', {
236
+ type: audio.type || 'audio/mpeg',
237
+ })
238
+ }
239
+ if (typeof ArrayBuffer !== 'undefined' && audio instanceof ArrayBuffer) {
240
+ this.ensureFileSupport()
241
+ return new File([audio], 'audio.mp3', { type: 'audio/mpeg' })
242
+ }
243
+ if (typeof audio === 'string') {
244
+ this.ensureFileSupport()
245
+
246
+ if (audio.startsWith('data:')) {
247
+ const parts = audio.split(',')
248
+ const header = parts[0]
249
+ const base64Data = parts[1] || ''
250
+ const mimeMatch = header?.match(/data:([^;]+)/)
251
+ const mimeType = mimeMatch?.[1] || 'audio/mpeg'
252
+ const bytes = base64ToArrayBuffer(base64Data)
253
+ const extension = mimeType.split('/')[1] || 'mp3'
254
+ return new File([bytes], `audio.${extension}`, { type: mimeType })
255
+ }
256
+
257
+ const bytes = base64ToArrayBuffer(audio)
258
+ return new File([bytes], 'audio.mp3', { type: 'audio/mpeg' })
259
+ }
260
+
261
+ throw new Error('Invalid audio input type')
262
+ }
263
+
264
+ // Throws on Node < 20 where the global `File` constructor is unavailable.
265
+ private ensureFileSupport(): void {
266
+ if (typeof File === 'undefined') {
267
+ throw new Error(
268
+ '`File` is not available in this environment. ' +
269
+ 'Use Node.js 20 or newer, or pass a File object directly.',
270
+ )
271
+ }
272
+ }
273
+ }
274
+
275
+ /**
276
+ * Creates a Groq transcription adapter with an explicit API key.
277
+ * Type resolution happens here at the call site.
278
+ *
279
+ * @param model - The model name (e.g., 'whisper-large-v3-turbo')
280
+ * @param apiKey - Your Groq API key
281
+ * @param config - Optional additional configuration
282
+ * @returns Configured Groq transcription adapter instance
283
+ *
284
+ * @example
285
+ * ```typescript
286
+ * const adapter = createGroqTranscription('whisper-large-v3-turbo', 'gsk_...');
287
+ *
288
+ * const result = await generateTranscription({
289
+ * adapter,
290
+ * audio: audioFile,
291
+ * language: 'en',
292
+ * });
293
+ * ```
294
+ */
295
+ export function createGroqTranscription<TModel extends GroqTranscriptionModel>(
296
+ model: TModel,
297
+ apiKey: string,
298
+ config?: Omit<GroqTranscriptionConfig, 'apiKey'>,
299
+ ): GroqTranscriptionAdapter<TModel> {
300
+ return new GroqTranscriptionAdapter({ apiKey, ...config }, model)
301
+ }
302
+
303
+ /**
304
+ * Creates a Groq transcription adapter using the `GROQ_API_KEY` environment
305
+ * variable. Type resolution happens here at the call site.
306
+ *
307
+ * Looks for `GROQ_API_KEY` in:
308
+ * - `process.env` (Node.js)
309
+ * - `window.env` (browser with injected env)
310
+ *
311
+ * @param model - The model name (e.g., 'whisper-large-v3-turbo')
312
+ * @param config - Optional configuration (excluding apiKey which is auto-detected)
313
+ * @returns Configured Groq transcription adapter instance
314
+ * @throws Error if GROQ_API_KEY is not found in environment
315
+ *
316
+ * @example
317
+ * ```typescript
318
+ * const adapter = groqTranscription('whisper-large-v3-turbo');
319
+ *
320
+ * const result = await generateTranscription({
321
+ * adapter,
322
+ * audio: 'https://example.com/audio.mp3',
323
+ * });
324
+ *
325
+ * console.log(result.text)
326
+ * ```
327
+ */
328
+ export function groqTranscription<TModel extends GroqTranscriptionModel>(
329
+ model: TModel,
330
+ config?: Omit<GroqTranscriptionConfig, 'apiKey'>,
331
+ ): GroqTranscriptionAdapter<TModel> {
332
+ const apiKey = getGroqApiKeyFromEnv()
333
+ return createGroqTranscription(model, apiKey, config)
334
+ }
@@ -0,0 +1,168 @@
1
+ import OpenAI from 'openai'
2
+ import { BaseTTSAdapter } from '@tanstack/ai/adapters'
3
+ import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
4
+ import { arrayBufferToBase64, generateId } from '@tanstack/ai-utils'
5
+ import { getGroqApiKeyFromEnv, withGroqDefaults } from '../utils/client'
6
+ import { validateAudioInput } from '../audio/audio-provider-options'
7
+ import type { TTSOptions, TTSResult } from '@tanstack/ai'
8
+ import type OpenAI_SDK from 'openai'
9
+ import type { GroqTTSModel } from '../model-meta'
10
+ import type { GroqTTSProviderOptions } from '../audio/tts-provider-options'
11
+ import type { GroqClientConfig } from '../utils'
12
+
13
+ /**
14
+ * Configuration for Groq TTS adapter
15
+ */
16
+ export interface GroqTTSConfig extends GroqClientConfig {}
17
+
18
+ /**
19
+ * Groq Text-to-Speech Adapter
20
+ *
21
+ * Tree-shakeable adapter for Groq TTS functionality. Groq exposes an
22
+ * OpenAI-compatible `/audio/speech` endpoint, so the adapter drives it with
23
+ * the OpenAI SDK via a `baseURL` override (the same pattern as the Groq text
24
+ * adapter).
25
+ *
26
+ * Supports `canopylabs/orpheus-v1-english` and
27
+ * `canopylabs/orpheus-arabic-saudi`.
28
+ *
29
+ * Features:
30
+ * - English voices: autumn(f), diana(f), hannah(f), austin(m), daniel(m), troy(m)
31
+ * - Arabic voices: fahad(m), sultan(m), lulwa(f), noura(f)
32
+ * - Output formats: flac, mp3, mulaw, ogg, wav (default wav)
33
+ * - Speed control
34
+ * - Configurable sample rate via `modelOptions`
35
+ */
36
+ export class GroqTTSAdapter<TModel extends GroqTTSModel> extends BaseTTSAdapter<
37
+ TModel,
38
+ GroqTTSProviderOptions
39
+ > {
40
+ readonly name = 'groq' as const
41
+
42
+ protected client: OpenAI
43
+
44
+ constructor(config: GroqTTSConfig, model: TModel) {
45
+ super(model, {})
46
+ this.client = new OpenAI(withGroqDefaults(config))
47
+ }
48
+
49
+ async generateSpeech(
50
+ options: TTSOptions<GroqTTSProviderOptions>,
51
+ ): Promise<TTSResult> {
52
+ const { model, text, voice, format, speed, modelOptions } = options
53
+
54
+ validateAudioInput({ input: text, model: this.model })
55
+
56
+ // Spreading optional inputs conditionally keeps the request compatible
57
+ // with the vendor SDK shape under exactOptionalPropertyTypes. `sample_rate`
58
+ // is a Groq-only body field carried via modelOptions.
59
+ const request: OpenAI_SDK.Audio.SpeechCreateParams = {
60
+ model,
61
+ input: text,
62
+ voice: voice ?? 'autumn',
63
+ response_format: format ?? 'wav',
64
+ ...(speed !== undefined && { speed }),
65
+ ...(modelOptions ?? {}),
66
+ }
67
+
68
+ try {
69
+ options.logger.request(
70
+ `activity=tts provider=${this.name} model=${model} format=${request.response_format ?? 'default'} voice=${request.voice}`,
71
+ { provider: this.name, model },
72
+ )
73
+ const response = await this.client.audio.speech.create(request)
74
+
75
+ const arrayBuffer = await response.arrayBuffer()
76
+ const base64 = arrayBufferToBase64(arrayBuffer)
77
+
78
+ const outputFormat = request.response_format ?? 'wav'
79
+ const contentType = this.getContentType(outputFormat)
80
+
81
+ return {
82
+ id: generateId(this.name),
83
+ model,
84
+ audio: base64,
85
+ format: outputFormat,
86
+ contentType,
87
+ }
88
+ } catch (error: unknown) {
89
+ // Narrow before logging: raw SDK errors can carry request metadata
90
+ // (including auth headers) which we must never surface to user loggers.
91
+ options.logger.errors(`${this.name}.generateSpeech fatal`, {
92
+ error: toRunErrorPayload(error, `${this.name}.generateSpeech failed`),
93
+ source: `${this.name}.generateSpeech`,
94
+ })
95
+ throw error
96
+ }
97
+ }
98
+
99
+ private getContentType(format: string): string {
100
+ const contentTypes: Record<string, string> = {
101
+ flac: 'audio/flac',
102
+ mp3: 'audio/mpeg',
103
+ mulaw: 'audio/basic',
104
+ ogg: 'audio/ogg',
105
+ wav: 'audio/wav',
106
+ }
107
+ return contentTypes[format] || 'audio/wav'
108
+ }
109
+ }
110
+
111
+ /**
112
+ * Creates a Groq speech adapter with explicit API key.
113
+ * Type resolution happens here at the call site.
114
+ *
115
+ * @param model - The model name (e.g., 'canopylabs/orpheus-v1-english')
116
+ * @param apiKey - Your Groq API key
117
+ * @param config - Optional additional configuration
118
+ * @returns Configured Groq speech adapter instance with resolved types
119
+ *
120
+ * @example
121
+ * ```typescript
122
+ * const adapter = createGroqSpeech('canopylabs/orpheus-v1-english', 'gsk_...')
123
+ *
124
+ * const result = await generateSpeech({
125
+ * adapter,
126
+ * text: 'Hello, world!',
127
+ * voice: 'autumn',
128
+ * })
129
+ * ```
130
+ */
131
+ export function createGroqSpeech<TModel extends GroqTTSModel>(
132
+ model: TModel,
133
+ apiKey: string,
134
+ config?: Omit<GroqTTSConfig, 'apiKey'>,
135
+ ): GroqTTSAdapter<TModel> {
136
+ return new GroqTTSAdapter({ apiKey, ...config }, model)
137
+ }
138
+
139
+ /**
140
+ * Creates a Groq speech adapter with automatic API key detection from
141
+ * environment variables.
142
+ *
143
+ * Looks for `GROQ_API_KEY` in the environment.
144
+ *
145
+ * @param model - The model name (e.g., 'canopylabs/orpheus-v1-english')
146
+ * @param config - Optional configuration (excluding apiKey which is auto-detected)
147
+ * @returns Configured Groq speech adapter instance with resolved types
148
+ * @throws Error if GROQ_API_KEY is not found in environment
149
+ *
150
+ * @example
151
+ * ```typescript
152
+ * const adapter = groqSpeech('canopylabs/orpheus-v1-english')
153
+ *
154
+ * const result = await generateSpeech({
155
+ * adapter,
156
+ * text: 'Welcome to TanStack AI!',
157
+ * voice: 'autumn',
158
+ * format: 'wav',
159
+ * })
160
+ * ```
161
+ */
162
+ export function groqSpeech<TModel extends GroqTTSModel>(
163
+ model: TModel,
164
+ config?: Omit<GroqTTSConfig, 'apiKey'>,
165
+ ): GroqTTSAdapter<TModel> {
166
+ const apiKey = getGroqApiKeyFromEnv()
167
+ return createGroqSpeech(model, apiKey, config)
168
+ }
@@ -0,0 +1,25 @@
1
+ /**
2
+ * Common audio provider options for Groq audio endpoints.
3
+ */
4
+ export interface AudioProviderOptions {
5
+ /**
6
+ * The text to generate audio for.
7
+ * Maximum length is 200 characters.
8
+ * Use [directions] for vocal control (English voices only).
9
+ */
10
+ input: string
11
+ /**
12
+ * The audio model to use for generation.
13
+ */
14
+ model: string
15
+ }
16
+
17
+ /**
18
+ * Validates that the audio input text does not exceed the maximum length.
19
+ * @throws Error if input text exceeds 200 characters
20
+ */
21
+ export const validateAudioInput = (options: AudioProviderOptions) => {
22
+ if (options.input.length > 200) {
23
+ throw new Error('Input text exceeds maximum length of 200 characters.')
24
+ }
25
+ }
@@ -0,0 +1,20 @@
1
+ /**
2
+ * Groq-specific options for audio transcription.
3
+ *
4
+ * These fields extend the shared `TranscriptionOptions` and are forwarded
5
+ * verbatim to the Groq transcription endpoint.
6
+ */
7
+ export interface GroqTranscriptionProviderOptions {
8
+ /**
9
+ * Sampling temperature between 0 and 1. Lower values produce more
10
+ * deterministic output. Groq recommends 0 (the default) for most use cases.
11
+ */
12
+ temperature?: number
13
+
14
+ /**
15
+ * Granularity levels to include when `response_format` is `verbose_json`.
16
+ * Pass `['word']`, `['segment']`, or both to control which timestamp arrays
17
+ * appear in the result.
18
+ */
19
+ timestamp_granularities?: Array<'word' | 'segment'>
20
+ }