@writepanda/mcp 1.116.1 → 1.122.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/bin/server.mjs +176 -4
  2. package/package.json +2 -2
package/bin/server.mjs CHANGED
@@ -391,6 +391,22 @@ const TOOLS = [
391
391
  },
392
392
  command: "workspace.set-brand",
393
393
  },
394
+ {
395
+ name: "workspace_capture_brand",
396
+ description:
397
+ "Auto-populate the active workspace's brand kit from a website URL. Runs HyperFrames' capture against the site, classifies its real colors (background/ink/primary/accent), fonts (display/body), and logo, and merges them into the brand kit (hand-set fields are preserved unless capture finds a better value). ASYNC — returns { jobId }; poll job_wait. Needs network; first run downloads the HyperFrames CLI, so allow a generous timeout. Use when the user says 'use my brand', 'pull my brand from <site>', or onboards a client by URL.",
398
+ inputSchema: {
399
+ type: "object",
400
+ properties: {
401
+ url: {
402
+ type: "string",
403
+ description: "The brand's website URL (e.g. https://acme.com). https:// is assumed if omitted.",
404
+ },
405
+ },
406
+ required: ["url"],
407
+ },
408
+ command: "workspace.capture-brand",
409
+ },
394
410
  {
395
411
  name: "workspace_get_project_defaults",
396
412
  description:
@@ -2623,7 +2639,7 @@ const TOOLS = [
2623
2639
  {
2624
2640
  name: "caption_set_template",
2625
2641
  description:
2626
- "Pick a caption template: classic, modern, minimal, bold, spotlight, boxed, neon, colored, or editorial. `editorial` is a magazine-emphasis style: the word being spoken right now renders large (and takes an accent color) while the rest of the line shrinks, so one big word sweeps across the line in time with the speech.",
2642
+ "Pick a caption template. Static styles: classic, modern, minimal, bold, spotlight, boxed, neon, colored, editorial (`editorial` = magazine emphasis: the spoken word renders large + accent-colored while the rest of the line shrinks). ANIMATED, transcript-driven styles (each word animates as it's spoken, in preview AND export): kineticSlam (words slam in), clipWipe (left-to-right wipe per word), gradientPop (gradient text, elastic pop), matrixDecode (character scramble that resolves), glitchRgb (RGB chromatic split), blendDifference (auto-inverts over any footage).",
2627
2643
  inputSchema: {
2628
2644
  type: "object",
2629
2645
  properties: {
@@ -2641,6 +2657,12 @@ const TOOLS = [
2641
2657
  "neon",
2642
2658
  "colored",
2643
2659
  "editorial",
2660
+ "kineticSlam",
2661
+ "clipWipe",
2662
+ "gradientPop",
2663
+ "matrixDecode",
2664
+ "glitchRgb",
2665
+ "blendDifference",
2644
2666
  ],
2645
2667
  },
2646
2668
  expectedRevision: { type: "number" },
@@ -2717,7 +2739,7 @@ const TOOLS = [
2717
2739
  {
2718
2740
  name: "media_generate_image",
2719
2741
  description:
2720
- 'Generate a single image via Replicate gpt-image-2 — for B-roll, concept stills, reference frames. Project-agnostic; writes to <userData>/generated-images/ and returns { imagePath, prompt, aspectRatio, predictionId }. The canonical B-roll workflow is: media_generate_image → motion_render_html (Ken-Burns + vignette wrap to a WebM) → project_add_motion_graphic — see SKILL.md "B-roll generation". Aspect ratios are gpt-image-2 native (1:1 / 3:2 / 2:3). For 16:9 video B-roll, generate 3:2 and crop in the wrap; for 9:16, generate 2:3. Requires the user\'s Replicate API key (Settings → Integrations).',
2742
+ 'Generate a single image via Replicate gpt-image-2 — for B-roll, concept stills, reference frames. Project-agnostic; writes to <userData>/generated-images/ and returns { imagePath, prompt, aspectRatio, predictionId }. For a FACELESS / image-driven video, pair with media_image_to_video (one-call Ken-Burns clip → project_add_clip). For B-roll layered over footage, the motion_render_html wrap (Ken-Burns + vignette → project_add_motion_graphic) still applies — see SKILL.md "Faceless videos" / "B-roll generation". Aspect ratios are gpt-image-2 native (1:1 / 3:2 / 2:3). For 16:9 generate 3:2; for 9:16 generate 2:3. Requires the user\'s Replicate API key (Settings → Integrations).',
2721
2743
  inputSchema: {
2722
2744
  type: "object",
2723
2745
  properties: {
@@ -2748,10 +2770,54 @@ const TOOLS = [
2748
2770
  },
2749
2771
  command: "media.generate-image",
2750
2772
  },
2773
+ {
2774
+ name: "media_image_to_video",
2775
+ description:
2776
+ "Turn a STILL image into a Ken-Burns video clip (slow pan/zoom) via FFmpeg — the native, one-call way to build faceless / image-driven videos. Takes an imagePath + durationMs, writes an H.264 MP4 to <userData>/ken-burns/, and returns { videoPath, durationMs, width, height }. Feed videoPath straight into project_add_clip. Fast, deterministic alternative to the motion_render_html Ken-Burns wrap (no headless browser). Canonical faceless-short loop: media_generate_image (2:3 for 9:16) → media_image_to_video (per beat, alternate zoom in/out) → project_add_clip, then media_generate_narration → project_add_audio at each clip's cumulative start.",
2777
+ inputSchema: {
2778
+ type: "object",
2779
+ properties: {
2780
+ imagePath: {
2781
+ type: "string",
2782
+ description: "Absolute path (or file:// URL) to the source still image.",
2783
+ },
2784
+ durationMs: {
2785
+ type: "number",
2786
+ description:
2787
+ "Clip length in ms. Match to the beat's narration length + a small tail (300-500ms) so the visual holds until the VO finishes.",
2788
+ },
2789
+ aspectRatio: {
2790
+ type: "string",
2791
+ description:
2792
+ "Output frame: '16:9' (default) | '9:16' (shorts) | '1:1' | '4:5' | '2:3' | '3:2'. The still is scaled to COVER these dims then panned.",
2793
+ },
2794
+ zoom: {
2795
+ type: "string",
2796
+ description:
2797
+ "Motion: 'in' (default, push in) | 'out' (pull out) | 'none' (static). Alternate in/out across beats for a lively cut.",
2798
+ },
2799
+ zoomAmount: {
2800
+ type: "number",
2801
+ description:
2802
+ "Peak zoom delta 0.05-0.4 (default 0.12 → ends at 1.12×). Keep ≤0.2 for a subtle documentary feel.",
2803
+ },
2804
+ fps: {
2805
+ type: "number",
2806
+ description: "Output frame rate (default 30).",
2807
+ },
2808
+ outputName: {
2809
+ type: "string",
2810
+ description: "Optional file-stem hint. Slugified; timestamp appended automatically.",
2811
+ },
2812
+ },
2813
+ required: ["imagePath", "durationMs"],
2814
+ },
2815
+ command: "media.image-to-video",
2816
+ },
2751
2817
  {
2752
2818
  name: "media_generate_narration",
2753
2819
  description:
2754
- "Generate a voiceover / narration audio clip. Project-agnostic; writes audio to <userData>/narration/ and returns { audioPath, durationMs, model, voice, predictionId }. Canonical workflow: media_generate_narration → project_add_audio (place at startMs, pass the returned durationMs so the overlay is sized to the speech). DEFAULT is the LOCAL, on-device Kokoro engine (English) — no API key, no cloud, runs on the user's machine; the model auto-downloads (~330 MB) on first use. Kokoro voices (pass in `voice`): af_heart (default), af_bella, am_michael, bf_emma, bm_george, etc. (a*=American, b*=British; f=female, m=male). To use CLOUD TTS instead (more voices/languages, needs the user's Replicate key), set `model` to a Replicate model: 'elevenlabs-v3' (most expressive — embed inline tags like [excited]/[whispers]) | 'gemini-flash-tts' (30 voices, multilingual, 'style' sets tone) | 'minimax-turbo' (fast, 'style' maps to an emotion, 'speed' 0.5-2). Cloud voices — gemini Kore/Puck/Charon; minimax Friendly_Person/Wise_Woman. Omit `model` to honour the workspace default (system_get_narration_engine).",
2820
+ "Generate a voiceover / narration audio clip. Project-agnostic; writes audio to <userData>/narration/ and returns { audioPath, durationMs, model, voice, predictionId }. Canonical workflow: media_generate_narration → project_add_audio (place at startMs, pass the returned durationMs so the overlay is sized to the speech). DEFAULT is the LOCAL, on-device Kokoro engine (English) — no API key, no cloud, runs on the user's machine; the model auto-downloads (~330 MB) on first use. Kokoro voices (pass in `voice`): af_heart (default), af_bella, am_michael, bf_emma, bm_george, etc. (a*=American, b*=British; f=female, m=male). To use CLOUD TTS instead (more voices/languages, needs the user's Replicate key), set `model` to a Replicate model: 'elevenlabs-v3' (most expressive — embed inline tags like [excited]/[whispers]; stock voices only) | 'elevenlabs-direct' (the user's OWN ElevenLabs account — CLONED voices; needs the ElevenLabs key from Settings → Integrations; pass the voice NAME or voice_id) | 'gemini-flash-tts' (30 voices, multilingual, 'style' sets tone) | 'minimax-turbo' (fast, 'style' maps to an emotion, 'speed' 0.5-2). Cloud voices — gemini Kore/Puck/Charon; minimax Friendly_Person/Wise_Woman. Omit `model` to honour the workspace default (system_get_narration_engine).",
2755
2821
  inputSchema: {
2756
2822
  type: "object",
2757
2823
  properties: {
@@ -2763,7 +2829,7 @@ const TOOLS = [
2763
2829
  model: {
2764
2830
  type: "string",
2765
2831
  description:
2766
- "Engine/model selector. Omit for the workspace default (local Kokoro). 'kokoro-local' forces on-device English. Cloud (needs Replicate key): 'elevenlabs-v3' | 'gemini-flash-tts' | 'minimax-turbo'.",
2832
+ "Engine/model selector. Omit for the workspace default (local Kokoro). 'kokoro-local' forces on-device English. Cloud (needs Replicate key): 'elevenlabs-v3' | 'gemini-flash-tts' | 'minimax-turbo'. Direct: 'elevenlabs-direct' (own ElevenLabs key; cloned voices).",
2767
2833
  },
2768
2834
  voice: {
2769
2835
  type: "string",
@@ -2814,6 +2880,112 @@ const TOOLS = [
2814
2880
  },
2815
2881
  command: "media.generate-music",
2816
2882
  },
2883
+ {
2884
+ name: "media_list_avatars",
2885
+ description:
2886
+ "List the HeyGen avatars on the user's account (studio avatars + talking photos) for talking-head video generation. Returns { avatars: [{ avatarId, name, kind, gender, previewImageUrl }], count }. Use an avatarId with media_generate_avatar_video (pass avatarKind='talking_photo' for a photo avatar). Requires the user's HeyGen API key (Settings → Integrations); the HeyGen API is a paid capability, not on every plan.",
2887
+ inputSchema: { type: "object", properties: {} },
2888
+ command: "media.list-avatars",
2889
+ },
2890
+ {
2891
+ name: "media_list_avatar_voices",
2892
+ description:
2893
+ "List the HeyGen voices on the user's account for talking-head video generation. Returns { voices: [{ voiceId, name, language, gender, previewAudioUrl }], count }. Use a voiceId with media_generate_avatar_video. Requires the user's HeyGen API key (Settings → Integrations).",
2894
+ inputSchema: { type: "object", properties: {} },
2895
+ command: "media.list-avatar-voices",
2896
+ },
2897
+ {
2898
+ name: "media_generate_avatar_video",
2899
+ description:
2900
+ "Generate a HeyGen talking-head video from the user's own avatar + a script. ASYNC: returns { jobId }; poll job_wait for { videoPath, durationSec, width, height }, then add it to the timeline with project_add_clip. Discover avatarId with media_list_avatars and voiceId with media_list_avatar_voices. Requires the user's HeyGen API key + credits (Settings → Integrations); renders server-side over minutes (poll with a generous timeout). Billed against the user's own HeyGen credits, not by PandaStudio.",
2901
+ inputSchema: {
2902
+ type: "object",
2903
+ properties: {
2904
+ avatarId: {
2905
+ type: "string",
2906
+ description: "HeyGen avatar id (from media_list_avatars).",
2907
+ },
2908
+ voiceId: {
2909
+ type: "string",
2910
+ description: "HeyGen voice id (from media_list_avatar_voices).",
2911
+ },
2912
+ script: {
2913
+ type: "string",
2914
+ description: "The text the avatar speaks. Plain text; HeyGen handles TTS.",
2915
+ },
2916
+ avatarKind: {
2917
+ type: "string",
2918
+ enum: ["avatar", "talking_photo"],
2919
+ description: "'avatar' (studio avatar, default) | 'talking_photo' (photo avatar).",
2920
+ },
2921
+ aspectRatio: {
2922
+ type: "string",
2923
+ enum: ["16:9", "9:16", "1:1"],
2924
+ description: "'16:9' (default, YouTube) | '9:16' (Shorts/Reels) | '1:1'.",
2925
+ },
2926
+ resolution: {
2927
+ type: "string",
2928
+ enum: ["720p", "1080p"],
2929
+ description: "'720p' (default, cheaper on credits) | '1080p'.",
2930
+ },
2931
+ speed: {
2932
+ type: "number",
2933
+ description: "Voice speed multiplier 0.5–1.5 (default 1).",
2934
+ },
2935
+ backgroundColor: {
2936
+ type: "string",
2937
+ description: "Optional solid background hex, e.g. '#000000'.",
2938
+ },
2939
+ outputName: {
2940
+ type: "string",
2941
+ description: "Optional file-stem hint; timestamp is appended. Defaults to the script head.",
2942
+ },
2943
+ },
2944
+ required: ["avatarId", "voiceId", "script"],
2945
+ },
2946
+ command: "media.generate-avatar-video",
2947
+ },
2948
+ {
2949
+ name: "memory_save",
2950
+ description:
2951
+ "Remember a durable fact or preference about this user/workspace, carried across ALL future chats (injected into every session). Use for STANDING context — brand, default caption/zoom styles, channel name + tone, recurring instructions ('always 9:16 for this client'). Do NOT use for one-off requests about the current edit. Idempotent on identical text. Returns { id, text }.",
2952
+ inputSchema: {
2953
+ type: "object",
2954
+ properties: {
2955
+ note: {
2956
+ type: "string",
2957
+ description:
2958
+ "One concise fact/preference, e.g. 'Channel is WritePanda; energetic tone, fast cuts' or 'Default caption template: editorial'.",
2959
+ },
2960
+ },
2961
+ required: ["note"],
2962
+ },
2963
+ command: "memory.save",
2964
+ },
2965
+ {
2966
+ name: "memory_forget",
2967
+ description:
2968
+ "Remove durable memory entries by id (from memory_list) or by a case-insensitive text substring. Use when a saved preference is outdated or wrong. Returns { removed, remaining }.",
2969
+ inputSchema: {
2970
+ type: "object",
2971
+ properties: {
2972
+ query: {
2973
+ type: "string",
2974
+ description:
2975
+ "An entry id (e.g. 'a1b2c3') for an exact remove, or a text fragment to remove every matching note.",
2976
+ },
2977
+ },
2978
+ required: ["query"],
2979
+ },
2980
+ command: "memory.forget",
2981
+ },
2982
+ {
2983
+ name: "memory_list",
2984
+ description:
2985
+ "List the durable memory for this workspace — the same notes injected into every chat. Returns { entries: [{ id, text }], count }. Rarely needed (memory is already in context); use to get ids for memory_forget or to audit what's stored.",
2986
+ inputSchema: { type: "object", properties: {} },
2987
+ command: "memory.list",
2988
+ },
2817
2989
  // `motion_generate` removed in v1.31.0 — see comment above the
2818
2990
  // motion-graphics section.
2819
2991
  {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@writepanda/mcp",
3
- "version": "1.116.1",
3
+ "version": "1.122.0",
4
4
  "description": "Model Context Protocol server for PandaStudio. Exposes the desktop video editor's automation surface to Cursor, Continue, Cline, Claude Desktop, and any MCP-compliant client.",
5
5
  "keywords": [
6
6
  "pandastudio",
@@ -36,7 +36,7 @@
36
36
  "LICENSE"
37
37
  ],
38
38
  "dependencies": {
39
- "@modelcontextprotocol/sdk": "^1.0.4"
39
+ "@modelcontextprotocol/sdk": "~1.29.0"
40
40
  },
41
41
  "publishConfig": {
42
42
  "access": "public"