beplus-mcp 0.32.0 → 0.33.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -84,7 +84,8 @@ Reinicie o cliente. Rode a tool **`whoami`** para confirmar o vínculo.
84
84
  | `generate_image` | Gera imagem (nano-banana / gpt-image-2). Bloqueante ~90s; retorna URL + imagem inline. Suporta refs, tamanho/qualidade, e extras do gpt-image-2 (background, moderation, output_format/compression). |
85
85
  | `generate_video` | Gera vídeo (Seedance / Kling). Aguarda ~120s; senão devolve o id pra `check_generation`. Suporta first/last frame, refs de imagem/vídeo/áudio, prompt negativo, áudio gerado, mode e motion-control. |
86
86
  | `upscale` | Melhora a resolução de uma imagem ou vídeo que já existe (Topaz). Detecta o tipo pelo arquivo. Imagem até 4x (~20s, a partir de 6💎); vídeo até 30s e saída até 1080p, custa por bloco de 10s e **descarta o áudio**. Mede o arquivo sozinho para o preço sair certo. ⚠️ Vídeo é restrito à equipe. |
87
- | `generate_audio` | Sintetiza fala (Gemini TTS, ~30 vozes, multi-locutor). Síncrono. |
87
+ | `generate_audio` | Sintetiza fala (Gemini TTS, multi-locutor). Padrão `gemini-3.8-flash-lite-tts`; os modelos 3.8 aceitam vozes da biblioteca, tags vocais (`<laugh>`, `<sigh>`...) e `style` por locutor; os legados só as 30 vozes clássicas. Síncrono. |
88
+ | `list_tts_voices` | Lista as vozes da biblioteca do TTS 3.8 (filtros: idioma, gênero, tom, contexto, busca) com o id para usar em `voice`. Somente leitura. |
88
89
  | `generate_music` | Gera música completa (Suno v5.5 / v4.5) — modo descrição ou letra própria, tags, instrumental, vocal_gender, controles criativos. Async. |
89
90
  | `check_generation` | Status de uma geração async por id. |
90
91
  | `cancel_generation` | Cancela uma geração em fila/processamento (estorna diamantes). |
package/dist/index.js CHANGED
@@ -21621,6 +21621,9 @@ var BeplusClient = class {
21621
21621
  tts(body2) {
21622
21622
  return this.request("POST", "/tts", { body: body2, timeoutMs: 9e4 });
21623
21623
  }
21624
+ ttsVoices(query) {
21625
+ return this.request("GET", "/tts/voices", { query, timeoutMs: 2e4 });
21626
+ }
21624
21627
  models() {
21625
21628
  return this.request("GET", "/models");
21626
21629
  }
@@ -22179,9 +22182,13 @@ var MUSIC_MODELS = [
22179
22182
  ];
22180
22183
  var VOCAL_GENDERS = ["f", "m"];
22181
22184
  var TTS_MODELS = [
22185
+ "gemini-3.8-flash-lite-tts",
22186
+ "gemini-3.8-flash-tts",
22182
22187
  "gemini-3.1-flash-tts-preview",
22183
22188
  "gemini-2.5-pro-preview-tts"
22184
22189
  ];
22190
+ var TTS_MODELS_38 = ["gemini-3.8-flash-lite-tts", "gemini-3.8-flash-tts"];
22191
+ var LIBRARY_VOICE_ID = /^[a-z]{2,3}-[a-z0-9]{2,4}-[a-z0-9-]{1,60}$/;
22185
22192
  var VOICES = [
22186
22193
  "Zephyr",
22187
22194
  "Puck",
@@ -22226,7 +22233,7 @@ var DEFAULTS = {
22226
22233
  videoAspect: "16:9",
22227
22234
  videoResolution: "720p",
22228
22235
  videoDuration: 5,
22229
- ttsModel: "gemini-3.1-flash-tts-preview",
22236
+ ttsModel: "gemini-3.8-flash-lite-tts",
22230
22237
  voice: "Kore",
22231
22238
  musicModel: "suno/v5-5"
22232
22239
  };
@@ -23463,26 +23470,50 @@ ${BUDGET_BLOCKED_GUIDANCE}`);
23463
23470
  }
23464
23471
 
23465
23472
  // src/tools/audio.ts
23473
+ var VOICE_HELP = "Uma das 30 vozes cl\xE1ssicas (Kore, Puck, Charon...) OU o id de uma voz da biblioteca (ex.: pt-br-advisor-10, s\xF3 nos modelos 3.8; descubra ids com list_tts_voices).";
23474
+ var voiceSchema = external_exports.string().refine((v) => VOICES.includes(v) || LIBRARY_VOICE_ID.test(v), {
23475
+ message: "Voz inv\xE1lida: use uma das 30 vozes cl\xE1ssicas (ex.: Kore) ou um id da biblioteca (ex.: pt-br-advisor-10)."
23476
+ });
23477
+ function isLibraryVoice(v) {
23478
+ return !VOICES.includes(v) && LIBRARY_VOICE_ID.test(v);
23479
+ }
23480
+ var speakerSchema = external_exports.object({
23481
+ name: external_exports.string().min(1).max(50),
23482
+ voice: voiceSchema.describe(VOICE_HELP),
23483
+ style: external_exports.string().max(500).optional().describe('Entrega deste locutor (s\xF3 modelos 3.8), ex.: "calmo, grave, pausado".')
23484
+ });
23466
23485
  function registerAudioTools(server, client, cfg) {
23467
23486
  server.registerTool(
23468
23487
  "generate_audio",
23469
23488
  {
23470
23489
  title: "Gerar \xE1udio (TTS)",
23471
- description: 'Sintetiza fala a partir de texto (s\xEDncrono \u2014 retorna a URL na hora) usando o IA Lab da BePlus. ~30 vozes. Opcional `style_instruction` (notas de dire\xE7\xE3o, ex.: "fale animado e pausado") e di\xE1logo com 2 locutores via `multi_speaker`. Consome diamantes por segundo de \xE1udio.' + TEAM_GATING_NOTE,
23490
+ description: 'Sintetiza fala a partir de texto (s\xEDncrono \u2014 retorna a URL na hora) usando o IA Lab da BePlus. Modelos Gemini 3.8 (`gemini-3.8-flash-lite-tts`, padr\xE3o, r\xE1pido e barato; `gemini-3.8-flash-tts`, qualidade de est\xFAdio, melhor para di\xE1logo e narra\xE7\xE3o longa) aceitam: vozes da biblioteca (milhares, por idioma; liste com `list_tts_voices` e passe o id em `voice`), tags vocais no texto como <laugh> <chuckle> <sigh> <breath> <gasp> <cough> <sneeze> <throat-clearing> <yawn> <groan> <scream> <whimper> <argh> <short pause> <long pause> (viram som de verdade), e estilo por locutor (`style` em cada speaker). No 3.8, `style_instruction` \xE9 dire\xE7\xE3o e N\xC3O \xE9 lida em voz alta. Os modelos legados (`gemini-3.1-flash-tts-preview`, `gemini-2.5-pro-preview-tts`) s\xF3 aceitam as 30 vozes cl\xE1ssicas (Kore, Puck, Charon...) e n\xE3o t\xEAm voz da biblioteca. Opcional `style_instruction` (ex.: "fale animado e pausado") e di\xE1logo com 2 locutores via `multi_speaker`. Consome diamantes por segundo de \xE1udio.' + TEAM_GATING_NOTE,
23472
23491
  inputSchema: {
23473
23492
  text: external_exports.string().min(1).max(1e4).describe("Texto a ser falado."),
23474
- voice: external_exports.enum(VOICES).default(DEFAULTS.voice).describe("Voz (ex.: Kore, Puck, Charon)."),
23493
+ voice: voiceSchema.default(DEFAULTS.voice).describe(VOICE_HELP),
23475
23494
  model: external_exports.enum(TTS_MODELS).default(DEFAULTS.ttsModel).describe("Modelo TTS."),
23476
23495
  style_instruction: external_exports.string().max(2e3).optional().describe("Instru\xE7\xE3o de estilo/tom aplicada \xE0 fala."),
23477
23496
  multi_speaker: external_exports.object({
23478
- speaker1: external_exports.object({ name: external_exports.string().min(1).max(50), voice: external_exports.enum(VOICES) }),
23479
- speaker2: external_exports.object({ name: external_exports.string().min(1).max(50), voice: external_exports.enum(VOICES) })
23497
+ speaker1: speakerSchema,
23498
+ speaker2: speakerSchema
23480
23499
  }).optional().describe('Di\xE1logo de 2 locutores. Marque as falas no texto como "Nome: ...".'),
23481
23500
  project: external_exports.string().max(64).optional().describe('Projeto (uuid OU code curto, ex.: "VRAO-26"). Omita para usar o projeto ativo da conta.')
23482
23501
  }
23483
23502
  },
23484
23503
  async (args) => {
23485
23504
  try {
23505
+ const ms = args.multi_speaker;
23506
+ const voices = [args.voice, ms?.speaker1.voice, ms?.speaker2.voice].filter((v) => !!v);
23507
+ if (!TTS_MODELS_38.includes(args.model)) {
23508
+ if (voices.some(isLibraryVoice)) {
23509
+ return errorResult(
23510
+ `Vozes da biblioteca s\xF3 funcionam nos modelos Gemini 3.8 (gemini-3.8-flash-lite-tts, gemini-3.8-flash-tts). O modelo ${args.model} s\xF3 aceita as 30 vozes cl\xE1ssicas (Kore, Puck, Charon...).`
23511
+ );
23512
+ }
23513
+ if (ms?.speaker1.style || ms?.speaker2.style) {
23514
+ return errorResult("O estilo por locutor (`style`) s\xF3 funciona nos modelos Gemini 3.8.");
23515
+ }
23516
+ }
23486
23517
  const body2 = compact({
23487
23518
  text: args.text,
23488
23519
  voice: args.voice,
@@ -23516,6 +23547,45 @@ ${BUDGET_BLOCKED_GUIDANCE}`);
23516
23547
  }
23517
23548
  }
23518
23549
  );
23550
+ server.registerTool(
23551
+ "list_tts_voices",
23552
+ {
23553
+ title: "Listar vozes da biblioteca TTS",
23554
+ description: 'Lista vozes da biblioteca do Gemini 3.8 TTS (n\xE3o inclui as 30 cl\xE1ssicas como Kore e Puck). Filtre por idioma (`language`, BCP-47, padr\xE3o pt-BR; "all" = todos), `gender`, `pitch`, `context` (ex.: Audiobook) e `search` (texto livre em nome, persona, descri\xE7\xE3o, sotaque). Passe o `id` da voz em `voice` do generate_audio com um modelo 3.8 (gemini-3.8-flash-lite-tts ou gemini-3.8-flash-tts). Somente leitura, sem custo.',
23555
+ inputSchema: {
23556
+ language: external_exports.string().max(12).optional().describe('Idioma BCP-47, ex.: pt-BR, en-US. Padr\xE3o pt-BR. "all" = sem filtro.'),
23557
+ gender: external_exports.enum(["female", "male", "neutral"]).optional(),
23558
+ pitch: external_exports.enum(["low", "medium", "high"]).optional(),
23559
+ context: external_exports.string().max(60).optional().describe("Uso da voz (trecho), ex.: Audiobook, Enterprise Agent."),
23560
+ search: external_exports.string().max(100).optional().describe("Texto livre: nome, persona, descri\xE7\xE3o, sotaque."),
23561
+ limit: external_exports.number().int().min(1).max(200).optional().describe("Padr\xE3o 60."),
23562
+ offset: external_exports.number().int().min(0).optional()
23563
+ }
23564
+ },
23565
+ async (args) => {
23566
+ try {
23567
+ const res = await client.ttsVoices(compact(args));
23568
+ const offset = args.offset ?? 0;
23569
+ if (!res.voices.length) {
23570
+ return textResult(`Nenhuma voz encontrada (total ${res.total}). Idiomas dispon\xEDveis: ${res.languages.join(", ")}.`);
23571
+ }
23572
+ const lines = res.voices.map((v) => {
23573
+ const desc = v.description ? ` | ${v.description.length > 110 ? `${v.description.slice(0, 107)}...` : v.description}` : "";
23574
+ return `${v.id} | ${v.name} | ${v.gender} | ${v.pitch}${desc}`;
23575
+ });
23576
+ const more = res.total > offset + res.voices.length ? ` Use offset=${offset + res.voices.length} para ver mais.` : "";
23577
+ return textResult(
23578
+ `Vozes ${offset + 1}-${offset + res.voices.length} de ${res.total} (id | nome | g\xEAnero | tom | descri\xE7\xE3o):
23579
+ ${lines.join("\n")}
23580
+
23581
+ Use o id em voice (modelos 3.8).${more}
23582
+ Contextos: ${res.contexts.join(", ")}.`
23583
+ );
23584
+ } catch (e) {
23585
+ return errorResult(apiErrorToUserMessage(e));
23586
+ }
23587
+ }
23588
+ );
23519
23589
  }
23520
23590
 
23521
23591
  // src/tools/music.ts
@@ -24321,7 +24391,7 @@ var KINDS = {
24321
24391
  tts: {
24322
24392
  title: "Locu\xE7\xE3o",
24323
24393
  size: { w: 320, h: 212 },
24324
- data: { model: "gemini-3.1-flash-tts-preview", voice: "Algieba", speed: "normal" },
24394
+ data: { model: "gemini-3.8-flash-lite-tts", voice: "Algieba", speed: "normal" },
24325
24395
  inputs: ["text"],
24326
24396
  outputs: ["audio", "approved"],
24327
24397
  dynamic: "style conforme o modelo"
@@ -26867,7 +26937,7 @@ N\xF3s: ${Object.entries(ids).map(([k, v]) => `${k}=${v}`).join(" \xB7 ")}` : ""
26867
26937
  import { createRequire } from "module";
26868
26938
  var require2 = createRequire(import.meta.url);
26869
26939
  function readVersion() {
26870
- if ("0.32.0") return "0.32.0";
26940
+ if ("0.33.0") return "0.33.0";
26871
26941
  try {
26872
26942
  const pkg = require2("../package.json");
26873
26943
  return pkg.version ?? "0.0.0-unknown";