@writepanda/mcp 1.90.0 → 1.98.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/bin/server.mjs +124 -16
  2. package/package.json +1 -1
package/bin/server.mjs CHANGED
@@ -155,6 +155,26 @@ const TOOLS = [
155
155
  inputSchema: { type: "object", properties: {} },
156
156
  command: "system.status",
157
157
  },
158
+ {
159
+ name: "agent_session_list",
160
+ description:
161
+ "List the embedded in-app agent's opencode sessions (id, title, created/updated timestamps). Observational only — never starts the agent server; returns running:false when it isn't up. Use before agent_session_stop to find the session to kill.",
162
+ inputSchema: { type: "object", properties: {} },
163
+ command: "agent.session-list",
164
+ },
165
+ {
166
+ name: "agent_session_stop",
167
+ description:
168
+ "Abort an in-app agent session by id (from agent_session_list), or every session with all:true. Stops that session's tool execution immediately; the transcript survives for replay. The kill switch for a runaway/ghost in-app agent session.",
169
+ inputSchema: {
170
+ type: "object",
171
+ properties: {
172
+ sessionId: { type: "string", description: "Session id to abort." },
173
+ all: { type: "boolean", description: "Abort every session instead of one." },
174
+ },
175
+ },
176
+ command: "agent.session-stop",
177
+ },
158
178
  {
159
179
  name: "system_list_commands",
160
180
  description:
@@ -165,7 +185,7 @@ const TOOLS = [
165
185
  {
166
186
  name: "system_get_transcription_language",
167
187
  description:
168
- "Read the active transcription engine selection for the current workspace. Returns { language } where language is one of: 'auto' (Parakeet TDT v3, default — auto-detects English + 25 European languages including Russian/Ukrainian), 'chinese' / 'japanese' / 'korean' / 'hindi' / 'arabic' / 'thai' (each routes to Whisper Large-v3-turbo). Use this when surfacing the active engine to the user or before recommending a switch.",
188
+ "Read the active transcription engine selection for the current workspace. Returns { language } where language is one of: 'auto' (Parakeet TDT v3, default — auto-detects English + 25 European languages including Russian/Ukrainian), or a Whisper Large-v3-turbo language: 'chinese' / 'japanese' / 'korean' / 'hindi' / 'arabic' / 'thai' / 'tamil' / 'telugu' / 'kannada' / 'malayalam' / 'bengali' / 'marathi' / 'gujarati' / 'punjabi'. Use this when surfacing the active engine to the user or before recommending a switch.",
169
189
  inputSchema: { type: "object", properties: {} },
170
190
  command: "system.getTranscriptionLanguage",
171
191
  },
@@ -178,7 +198,23 @@ const TOOLS = [
178
198
  properties: {
179
199
  language: {
180
200
  type: "string",
181
- enum: ["auto", "chinese", "japanese", "korean", "hindi", "arabic", "thai"],
201
+ enum: [
202
+ "auto",
203
+ "chinese",
204
+ "japanese",
205
+ "korean",
206
+ "hindi",
207
+ "arabic",
208
+ "thai",
209
+ "tamil",
210
+ "telugu",
211
+ "kannada",
212
+ "malayalam",
213
+ "bengali",
214
+ "marathi",
215
+ "gujarati",
216
+ "punjabi",
217
+ ],
182
218
  description: "Target language. 'auto' = Parakeet (default); the others route to Whisper.",
183
219
  },
184
220
  },
@@ -818,7 +854,7 @@ const TOOLS = [
818
854
  {
819
855
  name: "project_add_motion_graphic",
820
856
  description:
821
- "Drop a motion-graphic MP4/WebM onto the timeline. PREFERRED: pass `fromJob` (the jobId from motion_render_html / motion_generate / motion_concat) — the server resolves the outputPath internally, so you never handle the path string at all. Fallback: pass `file` (an absolute path) only for external/uploaded files. **Defaults to a `bundled:sound/mouse-click` SFX as the graphic appears** — pass `soundUrl: null` to make it silent, or any other `bundled:sound/<id>` to swap the sound.",
857
+ "Drop a motion-graphic MP4/WebM onto the timeline. PREFERRED: pass `fromJob` (the jobId from motion_render_html / motion_generate / motion_concat) — the server resolves the outputPath internally, so you never handle the path string at all. Fallback: pass `file` (an absolute path) only for external/uploaded files. Also accepts `file: 'bundled:transition/<id>'` (ids from asset_list_transitions) — the house-opener pattern: a bundled transition placed as an overlay auto-stamps transitionId so it COVER-fits any canvas (a 16:9 sweep fills a 9:16 frame) and carries its SFX via soundUrl. **Defaults to a `bundled:sound/mouse-click` SFX as the graphic appears** — pass `soundUrl: null` to make it silent, or any other `bundled:sound/<id>` to swap the sound.",
822
858
  inputSchema: {
823
859
  type: "object",
824
860
  properties: {
@@ -1265,6 +1301,60 @@ const TOOLS = [
1265
1301
  },
1266
1302
  command: "project.render-frame",
1267
1303
  },
1304
+ {
1305
+ name: "project_render_sheet",
1306
+ description:
1307
+ "Render N preview frames sampled across an edited-time range and tile them into ONE contact-sheet PNG (row-major grid). THE verification tool for pacing and motion: one call + one image read shows caption changes, zoom ramps, overlay mutations, and dead stretches — use this instead of N sequential project_render_frame calls. Returns { sheetPath, cols, rows, frames: [{index,row,col,atMs,timeMs,path}] }; per-cell timestamps come from the frames array (cell k = frames[k], left-to-right then top-to-bottom). Every frame's seek is verified — no stale frames. If tiling is unavailable, sheetPath is null and frames[].path are readable individually.",
1308
+ inputSchema: {
1309
+ type: "object",
1310
+ properties: {
1311
+ id: { type: "string" },
1312
+ path: { type: "string" },
1313
+ fromMs: { type: "number", description: "Range start, edited ms. Default 0." },
1314
+ toMs: { type: "number", description: "Range end, edited ms. Default: edited end." },
1315
+ count: { type: "number", description: "Frames sampled evenly. Default 12, max 24." },
1316
+ cols: { type: "number", description: "Grid columns. Default 4." },
1317
+ outPath: { type: "string", description: "Optional sheet PNG output path." },
1318
+ },
1319
+ },
1320
+ command: "project.render-sheet",
1321
+ },
1322
+ {
1323
+ name: "project_apply_edit_plan",
1324
+ description:
1325
+ 'Apply a whole editing plan in ONE call: a JSON array of timeline ops, applied atomically (single read, all ops, single save; any invalid op aborts with nothing written). Supported ops: {"op":"add-trim","startMs","endMs"} | {"op":"add-zoom","atMs","durationMs","depth?","focusX?","focusY?","soundUrl?","anchorSourceMs?"} | {"op":"add-speed","startMs","endMs","speed"}. ALWAYS prefer this over sequential project_add_trim/zoom/speed calls when executing a beat map — it replaces 20+ round-trips with one.',
1326
+ inputSchema: {
1327
+ type: "object",
1328
+ properties: {
1329
+ id: { type: "string" },
1330
+ path: { type: "string" },
1331
+ plan: {
1332
+ type: "string",
1333
+ description:
1334
+ 'JSON array of ops, e.g. [{"op":"add-trim","startMs":1000,"endMs":1400},{"op":"add-zoom","atMs":5000,"durationMs":2500,"depth":2}]',
1335
+ },
1336
+ expectedRevision: { type: "number" },
1337
+ },
1338
+ required: ["plan"],
1339
+ },
1340
+ command: "project.apply-edit-plan",
1341
+ },
1342
+ {
1343
+ name: "audio_probe",
1344
+ description:
1345
+ 'EARS without exporting: per-clip audio stats from the export-relevant track (cleanedAudioPath when present, else source). Returns hasAudio, mean/max volume (dB), detected silence spans (clip-SOURCE time), and the project audio overlays (music beds) — verify "is there dead air / is the level sane / did my music land" before export. Synchronous, a few seconds per clip.',
1346
+ inputSchema: {
1347
+ type: "object",
1348
+ properties: {
1349
+ id: { type: "string" },
1350
+ path: { type: "string" },
1351
+ clipId: { type: "string", description: "Optional — probe only this clip." },
1352
+ noiseDb: { type: "number", description: "Silence threshold dBFS. Default -30." },
1353
+ minSilenceSec: { type: "number", description: "Min silence to report, sec. Default 0.5." },
1354
+ },
1355
+ },
1356
+ command: "audio.probe",
1357
+ },
1268
1358
  {
1269
1359
  name: "project_set_region_sound",
1270
1360
  description:
@@ -1431,19 +1521,23 @@ const TOOLS = [
1431
1521
  {
1432
1522
  name: "project_save",
1433
1523
  description:
1434
- "Overwrite a project with the full JSON you supply. Use this when you've built or mutated a project object client-side and want to persist it. Always pass `expectedRevision` (the revision you read) — the server will reject the write if a concurrent edit happened between your read and save, returning { ok: false, details: { code: 'revision_conflict' } }.",
1524
+ "Persist a project. Two modes: (1) pass a full `project` JSON to OVERWRITE the file (use after building/mutating a project object client-side) — always pass `expectedRevision` so a concurrent edit returns { ok:false, details:{ code:'revision_conflict' } }; (2) OMIT `project` to simply confirm the current on-disk state. The add_*/set_* verbs already autosave, so a bodyless save is a no-op confirmation that returns the current revision — NOT an error. You usually don't need an explicit save at all.",
1435
1525
  inputSchema: {
1436
1526
  type: "object",
1437
1527
  properties: {
1438
1528
  id: { type: "string" },
1439
1529
  path: { type: "string" },
1440
- project: { type: "object", description: "Full project JSON to persist" },
1530
+ project: {
1531
+ type: "object",
1532
+ description:
1533
+ "Full project JSON to persist (overwrite mode). Omit to just confirm current on-disk state — edits via add_*/set_* verbs already autosave.",
1534
+ },
1441
1535
  expectedRevision: {
1442
1536
  type: "number",
1443
- description: "Revision from your last project_read. Strongly recommended.",
1537
+ description:
1538
+ "Revision from your last project_read. Strongly recommended in overwrite mode.",
1444
1539
  },
1445
1540
  },
1446
- required: ["project"],
1447
1541
  },
1448
1542
  command: "project.save",
1449
1543
  },
@@ -1570,7 +1664,7 @@ const TOOLS = [
1570
1664
  {
1571
1665
  name: "project_update_region",
1572
1666
  description:
1573
- "Patch fields on an existing region without replacing it. Accepts any subset of the region's own properties — only the supplied fields are changed, everything else is preserved. Works on zoom (depth, focusX, focusY, focusMode: 'static'|'auto' — set 'auto' for the cursor-follow camera), trim (startMs, endMs), speed (startMs, endMs, speed), fx (atMs, durationMs, opacity, blendMode, speed), annotation, overlay, clip-transform (startMs, endMs, preset, transitionMs), and audio-overlay (startMs, endMs, sourceStartMs, volume — lets you drag-trim an audio overlay without removing it). Time shifts on a link-grouped region (e.g. one half of a designed segment) propagate to every peer in the group so the pair stays coherent.",
1667
+ "Patch fields on an existing region without replacing it. Accepts any subset of the region's own properties — only the supplied fields are changed, everything else is preserved. Overlay geometry (x/y/width/height) is PERCENT of canvas 0-100; values <= 1 are treated as 0-1 fractions and scaled. Works on zoom (depth, focusX, focusY, focusMode: 'static'|'auto' — set 'auto' for the cursor-follow camera), trim (startMs, endMs), speed (startMs, endMs, speed), fx (atMs, durationMs, opacity, blendMode, speed), annotation, overlay, clip-transform (startMs, endMs, preset, transitionMs), and audio-overlay (startMs, endMs, sourceStartMs, volume — lets you drag-trim an audio overlay without removing it). Time shifts on a link-grouped region (e.g. one half of a designed segment) propagate to every peer in the group so the pair stays coherent.",
1574
1668
  inputSchema: {
1575
1669
  type: "object",
1576
1670
  properties: {
@@ -1816,6 +1910,16 @@ const TOOLS = [
1816
1910
  description: "Fallback duration when neither endMs nor maxDurationMs is given.",
1817
1911
  },
1818
1912
  volume: { type: "number", description: "Volume multiplier 0–2. Default 0.8." },
1913
+ fadeIn: {
1914
+ type: "number",
1915
+ description:
1916
+ "Fade-in ramp at the overlay's audible start, in ms (e.g. 1000 = 1s). Applied by the export mixer. Omit for a hard start.",
1917
+ },
1918
+ fadeOut: {
1919
+ type: "number",
1920
+ description:
1921
+ "Fade-out ramp ending at the overlay's audible end, in ms. Applied for bounded overlays (endMs/maxDurationMs set); ignored on uncapped full-length overlays. Useful for seam-crossfading a looped music bed. Omit for a hard stop.",
1922
+ },
1819
1923
  maxDurationMs: {
1820
1924
  type: "number",
1821
1925
  description:
@@ -1983,7 +2087,7 @@ const TOOLS = [
1983
2087
  {
1984
2088
  name: "transcript_get",
1985
2089
  description:
1986
- "Return the merged edited-time transcript: every word with id, text, startMs, endMs in EDITED-TIMELINE coordinates. Use word IDs as input to transcript_delete_words. For a PODCAST composite, each word also carries `speaker` ('host' = speaker 1 / mediaPath, 'guest' = speaker 2 / webcamPath) — use these spans to drive project_set_clip_layout (cut to whoever is talking).",
2090
+ "Return the merged transcript. Each word carries BOTH time bases: startMs/endMs are SOURCE time (the raw recording; use for anchorSourceMs), and editedStartMs/editedEndMs are EDITED-timeline time with trims/speeds applied (null = the word sits inside a trim). Read the base you need; no conversion calls required. Word IDs feed transcript_delete_words. For a PODCAST composite, each word also carries `speaker` ('host' = speaker 1 / mediaPath, 'guest' = speaker 2 / webcamPath) — use these spans to drive project_set_clip_layout (cut to whoever is talking).",
1987
2091
  inputSchema: {
1988
2092
  type: "object",
1989
2093
  properties: {
@@ -2071,7 +2175,7 @@ const TOOLS = [
2071
2175
  {
2072
2176
  name: "transcript_find_issues",
2073
2177
  description:
2074
- "Scan the transcript for editorial problems the speaker created and recovered from: **duplicate-take** (a line re-recorded later — keep the cleaner second take, cut the first), **false-start** (an abandoned fragment before a restart), and **adjacent-repeat** (a stutter like 'the the'). Returns candidate `issues`, each with `type`, `severity`, the `wordIds` of the DISCARDED attempt (pass straight to transcript.delete-words), the edited-time span, the offending `text`, and a `note` explaining what's kept. CONSERVATIVE and read-only — it never edits; review each candidate against context before deleting, since some repeats are intentional. Run this after transcribe + remove-fillers as the content-cleanup pass. Filter with `types` (comma-separated) and cap with `limit`.",
2178
+ "Scan the transcript for editorial problems the speaker created and recovered from: **duplicate-take** (a line re-recorded later — keep the cleaner second take, cut the first), **false-start** (an abandoned fragment before a restart), and **adjacent-repeat** (a stutter like 'the the'). Returns candidate `issues`, each with `type`, `severity`, the `wordIds` of the DISCARDED attempt (pass straight to transcript.delete-words), the edited-time span, the offending `text`, and a `note` explaining what's kept. CONSERVATIVE and read-only — it never edits. Deliberate parallel structure ('one for transcription, one for outreach') is NOT flagged as a false start, and lone stopword repeats across pause tokens are skipped. A `severity: \"low\"` false-start is a REVIEW candidate (the restart diverges from the fragment — possibly an intentional list): KEEP it by default unless context clearly shows a flubbed take. Review every candidate against context before deleting. Run this after transcribe + remove-fillers as the content-cleanup pass. Filter with `types` (comma-separated) and cap with `limit`.",
2075
2179
  inputSchema: {
2076
2180
  type: "object",
2077
2181
  properties: {
@@ -2090,7 +2194,7 @@ const TOOLS = [
2090
2194
  {
2091
2195
  name: "transcript_find_replace",
2092
2196
  description:
2093
- "Find the first occurrence of a phrase in the transcript and replace it with new text. Useful for correcting mis-transcribed names, technical terms, or brand names without re-transcribing. Returns { matched, replacements } — matched is false if the phrase wasn't found.",
2197
+ "Find every occurrence of a phrase in the transcript and replace it with new text. Useful for correcting mis-transcribed names, technical terms, or brand names without re-transcribing. Timing is preserved — only the displayed/exported text changes. Returns { replacedCount, wordsPatched }; replacedCount is 0 if the phrase wasn't found.",
2094
2198
  inputSchema: {
2095
2199
  type: "object",
2096
2200
  properties: {
@@ -2099,7 +2203,7 @@ const TOOLS = [
2099
2203
  find: {
2100
2204
  type: "string",
2101
2205
  description:
2102
- 'Phrase to search for. Case-insensitive; digits and punctuation are ignored on both sides, so "than60" and "than" both match the word "than60". A pure-number/punctuation phrase can\'t be targeted (it normalizes to empty) — include a letter word.',
2206
+ 'Phrase to search for. Case-insensitive; punctuation is ignored. Digits are SIGNIFICANT when the phrase contains them ("try30" matches only the token "try30") and ignored otherwise ("than" also matches "than60"). A multi-word phrase also matches a single merged STT token ("Wispr Flow" as one token; "of $499" matches "of$499"), and a phrase equal to a token\'s raw text always matches — so space/digit/punctuation-bearing tokens (even pure numbers like "30%") are all targetable.',
2103
2207
  },
2104
2208
  replace: { type: "string", description: "Replacement text (replaces the whole phrase)" },
2105
2209
  expectedRevision: { type: "number" },
@@ -2193,7 +2297,7 @@ const TOOLS = [
2193
2297
  {
2194
2298
  name: "caption_set_style",
2195
2299
  description:
2196
- "Override specific style properties of the caption overlay without changing the template. All fields are optional — pass only what you want to change. positionY controls vertical placement (0=top, 100=bottom, default 85). fontFamily sets the caption font (a system font like 'Georgia' or a loaded custom font; unloaded fonts fall back). fontSize is in REM (e.g. '2.6rem'), matching the editor's Font size slider; it is clamped to 1.0–5.0rem (a px value is converted to rem), so you can't exceed the editor's max.",
2300
+ "Override specific style properties of the caption overlay without changing the template. All fields are optional — pass only what you want to change. positionY controls vertical placement in PERCENT of frame height (0=top, 100=bottom, default 85; a 0-1 fraction like 0.85 is auto-converted to percent). fontFamily sets the caption font (a system font like 'Georgia' or a loaded custom font; unloaded fonts fall back). fontSize is in REM (e.g. '2.6rem'), matching the editor's Font size slider; it is clamped to 1.0–5.0rem (a px value is converted to rem), so you can't exceed the editor's max.",
2197
2301
  inputSchema: {
2198
2302
  type: "object",
2199
2303
  properties: {
@@ -2203,7 +2307,11 @@ const TOOLS = [
2203
2307
  type: "number",
2204
2308
  description: "Max words shown at once. Default varies by template.",
2205
2309
  },
2206
- positionY: { type: "number", description: "Vertical position 0-100. Default 85." },
2310
+ positionY: {
2311
+ type: "number",
2312
+ description:
2313
+ "Vertical position as percent of frame height from the top (0-100). Default 85. Fractions \u2264 1.0 (e.g. 0.85) are treated as 0-1 and converted to percent.",
2314
+ },
2207
2315
  fontFamily: {
2208
2316
  type: "string",
2209
2317
  description:
@@ -2355,7 +2463,7 @@ const TOOLS = [
2355
2463
  {
2356
2464
  name: "motion_generate",
2357
2465
  description:
2358
- "Render a bundled template by id with your own slot values (text, colors, list items) and a background mode, then add the result to the timeline with project_add_motion_graphic (or project_add_designed_segment for split-panel). Async — returns { jobId, outputPath }; call job_wait, then pass that jobId as `fromJob` to the add tool. Discover templates + slots with motion_list. Prefer this over motion_render_html whenever a bundled template fits the brief — it's faster and already designed.",
2466
+ "Render a bundled template by id with your own slot values (text, colors, list items) and a background mode, then add the result to the timeline with project_add_motion_graphic (or project_add_designed_segment for split-panel). Async — returns { jobId, outputPath }; call job_wait, then pass that jobId as `fromJob` to the add tool. Discover templates + slots with motion_list. Prefer this over motion_render_html whenever a bundled template fits the brief — it's faster and already designed. The same automatic quality gates as motion_render_html apply (pre-render contract lint; STATIC_RENDER on frozen output).",
2359
2467
  inputSchema: {
2360
2468
  type: "object",
2361
2469
  properties: {
@@ -2428,7 +2536,7 @@ const TOOLS = [
2428
2536
  {
2429
2537
  name: "motion_render_html",
2430
2538
  description:
2431
- "Render arbitrary HTML/CSS/JS to video via the HyperFrames engine — frame-perfect, seekable capture through chrome-headless-shell's BeginFrame API. Use this when the bundled motion-graphic templates don't fit the brief. The HTML MUST expose a paused GSAP timeline registered as `window.__timelines[<data-composition-id>] = tl` — NOT CSS keyframes, NOT setTimeout, NOT `window.__hf.seek` (that lower-level protocol skips the compositor invalidation wrapper and renders with 1-second stalls). See the SKILL.md `Custom motion graphics — HTML authoring` section for the required `data-composition-id`/`data-width`/`data-height`/`data-duration` root element, the canonical template, and pacing rules. Pass either inline `html` or `htmlPath`. Set `transparent: true` for a WebM+alpha overlay (lower thirds, watermarks). IMPORTANT: renders are sequential — call job_wait to completion before starting another render, or you'll get a RENDER_BUSY error. Async — returns { jobId, outputPath }; poll job_wait for the final file.",
2539
+ "Render arbitrary HTML/CSS/JS to video via the HyperFrames engine — frame-perfect, seekable capture through chrome-headless-shell's BeginFrame API. Use this when the bundled motion-graphic templates don't fit the brief. The HTML MUST expose a paused GSAP timeline registered as `window.__timelines[<data-composition-id>] = tl` — NOT CSS keyframes, NOT setTimeout, NOT `window.__hf.seek` (that lower-level protocol skips the compositor invalidation wrapper and renders with 1-second stalls). See the SKILL.md `Custom motion graphics — HTML authoring` section for the required `data-composition-id`/`data-width`/`data-height`/`data-duration` root element, the canonical template, and pacing rules. Pass either inline `html` or `htmlPath`. Set `transparent: true` for a WebM+alpha overlay (lower thirds, watermarks). Renders queue automatically and execute serially in submission order — fire multiple renders back-to-back without waiting, then job_wait each. RENDER_BUSY appears only past a 16-deep runaway ceiling. Two automatic quality gates apply: a pre-render lint rejects contract violations (unpaused timeline, repeat:-1, Math.random, missing __timelines registration, shadow/blur radii over 500px), and a post-render check fails any clip frozen for 90%+ of its duration with STATIC_RENDER (fix the composition; never place a gated clip). Async — returns { jobId, outputPath }; poll job_wait for the final file.",
2432
2540
  inputSchema: {
2433
2541
  type: "object",
2434
2542
  properties: {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@writepanda/mcp",
3
- "version": "1.90.0",
3
+ "version": "1.98.0",
4
4
  "description": "Model Context Protocol server for PandaStudio. Exposes the desktop video editor's automation surface to Cursor, Continue, Cline, Claude Desktop, and any MCP-compliant client.",
5
5
  "keywords": [
6
6
  "pandastudio",