aimakeall-mcp 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -16,7 +16,7 @@ Claude Code·Codex 같은 MCP 클라이언트에서 자연어로 AImakeAll 영
16
16
  "mcpServers": {
17
17
  "aimakeall": {
18
18
  "command": "npx",
19
- "args": ["-y", "aimakeall-mcp@0.8.0"],
19
+ "args": ["-y", "aimakeall-mcp@0.9.0"],
20
20
  "env": { "AIMAKEALL_PAT": "aio_pat_..." }
21
21
  }
22
22
  }
@@ -23,16 +23,15 @@ import {
23
23
  import { mp3DurationSec } from "./mp3-duration.mjs";
24
24
  import { canonicalYoutubeWatchUrl } from "./payload-guard.mjs";
25
25
  import { createUsageEventId } from "./usage-event.mjs";
26
+ import { loudnessDbfsFromWav, silenceRatioFromWav } from "./wav-dsp.mjs";
26
27
  import { trimWavToSeconds } from "./wav-trim.mjs";
27
28
 
28
29
  // 여러 파일을 base64 로 싣는 요청의 인코딩 크기 합이 서버 JSON 상한을 넘지 않게 사전 검사.
29
30
  // 서버가 전체 업로드를 받은 뒤 413 을 내는 것을 막고 행동 가능한 한국어 안내를 준다.
30
31
  const BODY_BUDGET_BYTES = SERVER_JSON_BODY_LIMIT_BYTES - 512 * 1024; // 바디의 나머지 필드 여유
31
32
 
32
- // 컴패니언 계약 상수 — companion/index.cjs 의 MAX_SHOT_COUNT 및 오디오 추출 포맷과 맞춘다.
33
- // 오디오는 extractCompanionAudio 기본값(16kHz)·16bit·모노 = 초당 32,000 바이트.
33
+ // 컴패니언 계약 상수 — companion/index.cjs 의 MAX_SHOT_COUNT 와 맞춘다.
34
34
  const COMPANION_MAX_SHOTS = 18;
35
- const COMPANION_AUDIO_BYTES_PER_SECOND = 16_000 * 2;
36
35
 
37
36
  function assertEncodedBudget(filePaths, { context = "요청" } = {}) {
38
37
  let total = 0;
@@ -253,7 +252,7 @@ export function registerCloudTools(server, config, api) {
253
252
  categoryId: z.enum(["community-shorts", "viral-shorts", "insight"]).optional().describe("기본 community-shorts. insight=명언(통찰) 쇼츠"),
254
253
  targetCustomer: z.string().optional(),
255
254
  tone: z.string().optional().describe("기본 '자동 추천'"),
256
- sceneCount: z.number().int().min(1).max(10).optional().describe("기본 3"),
255
+ sceneCount: z.number().int().min(1).max(10).optional().describe("씬 수. 생략하면 channelFingerprint 의 목표 길이로 자동 산출(지문도 없으면 3)"),
257
256
  channelFingerprint: z.object({
258
257
  sampleSize: z.number().optional(),
259
258
  durationTargetSec: z.number().nullable().optional(),
@@ -276,7 +275,7 @@ export function registerCloudTools(server, config, api) {
276
275
  forbidden: z.string().optional().describe("금지 변형 (기본: different face, different hairstyle)"),
277
276
  })).max(4).optional().describe("등장 캐릭터 — 서버가 씬마다 identity_lock 을 imagePrompt 에 강제 주입해 캐릭터 일관성을 잠급니다"),
278
277
  },
279
- wrapCloudHandler(config, async ({ topic = "", copy = "", categoryId = "community-shorts", targetCustomer = "", tone = "자동 추천", sceneCount = 3, characters = [], channelFingerprint = null }) => {
278
+ wrapCloudHandler(config, async ({ topic = "", copy = "", categoryId = "community-shorts", targetCustomer = "", tone = "자동 추천", sceneCount = 0, characters = [], channelFingerprint = null }) => {
280
279
  if (!String(topic).trim() && !String(copy).trim()) {
281
280
  return textResult("topic 또는 copy 중 하나는 입력해야 합니다.", { isError: true });
282
281
  }
@@ -780,9 +779,14 @@ export function registerCloudTools(server, config, api) {
780
779
  emphasisKeywords: z.array(z.string()).max(12).optional().describe("강조 단어 목록 — 자막 라인 안에서 해당 단어만 다른 색/크기로 렌더 (plan_shorts_video의 emphasisKeywords 를 그대로 전달)"),
781
780
  emphasisColor: z.string().regex(/^#[0-9a-fA-F]{6}$/, "#RRGGBB 형식").optional().describe("강조 색 #RRGGBB (기본 #facc15 노랑)"),
782
781
  emphasisSizeScale: z.number().optional().describe("강조 크기 배율 0.5~2 (기본 1.2)"),
782
+ channelFingerprint: z.object({
783
+ subtitleYPct: z.number().nullable().optional(),
784
+ subtitleLines: z.number().nullable().optional(),
785
+ hasBgm: z.boolean().nullable().optional(),
786
+ }).passthrough().optional().describe("measure_channel_spec 이 반환한 fingerprint — 자막 세로 위치·BGM 유무를 참고 채널에 맞춥니다"),
783
787
  transitionMode: z.enum(["fade", "none"]).optional().describe("기본 fade"),
784
788
  },
785
- wrapCloudHandler(config, async ({ sceneVideos, featureKey = "commerceVideo", projectTitle = "aimakeall-mcp", aspectRatio = "9:16", ttsAudioPath = "", bgmAudioPath = "", audioDurationSec = 0, bgmVolume = null, muteVideoAudio = true, titleText = "", subtitleLines = [], emphasisKeywords = [], emphasisColor = "", emphasisSizeScale = 0, transitionMode = "fade" }) => {
789
+ wrapCloudHandler(config, async ({ sceneVideos, featureKey = "commerceVideo", projectTitle = "aimakeall-mcp", aspectRatio = "9:16", ttsAudioPath = "", bgmAudioPath = "", audioDurationSec = 0, bgmVolume = null, muteVideoAudio = true, titleText = "", subtitleLines = [], emphasisKeywords = [], emphasisColor = "", emphasisSizeScale = 0, transitionMode = "fade", channelFingerprint = null }) => {
786
790
  // 오디오 확장자 검증 + 합산 크기 예산(서버 413 사전 차단).
787
791
  if (ttsAudioPath) assertAllowedInputFile(ttsAudioPath, new Set([".mp3", ".wav"]), { kind: "TTS 오디오" });
788
792
  if (bgmAudioPath) assertAllowedInputFile(bgmAudioPath, new Set([".mp3", ".wav"]), { kind: "BGM 오디오" });
@@ -807,6 +811,7 @@ export function registerCloudTools(server, config, api) {
807
811
  })),
808
812
  emphasisColor,
809
813
  emphasisKeywords,
814
+ channelFingerprint,
810
815
  emphasisSizeScale,
811
816
  subtitleLines,
812
817
  titleText,
@@ -944,6 +949,17 @@ export function registerCloudTools(server, config, api) {
944
949
  christianContentType: z.enum(["youtube-narration", "prayer", "devotional", "comfort-message"]).optional().describe("styleId=christian일 때 유형"),
945
950
  insightFigureNames: z.string().optional().describe("styleId=insight일 때 인물명(쉼표 구분)"),
946
951
  userOpinion: z.string().optional().describe("본문에 자연스럽게 녹일 작성자 의견"),
952
+ channelFingerprint: z.object({
953
+ durationTargetSec: z.number().nullable().optional(),
954
+ cutIntervalSec: z.number().nullable().optional(),
955
+ cpsTarget: z.number().nullable().optional(),
956
+ subtitleYPct: z.number().nullable().optional(),
957
+ subtitleLines: z.number().nullable().optional(),
958
+ endingStyle: z.string().nullable().optional(),
959
+ emphasisCount: z.number().nullable().optional(),
960
+ hasBgm: z.boolean().nullable().optional(),
961
+ sampleSize: z.number().optional(),
962
+ }).passthrough().optional().describe("measure_channel_spec 의 fingerprint — 실측 규격(길이·컷·말끝)을 대본에 강제합니다"),
947
963
  },
948
964
  wrapCloudHandler(config, async ({
949
965
  title,
@@ -957,12 +973,15 @@ export function registerCloudTools(server, config, api) {
957
973
  christianContentType = "youtube-narration",
958
974
  insightFigureNames = "",
959
975
  userOpinion = "",
976
+ channelFingerprint = null,
960
977
  }) => {
961
978
  const SCRIPT_MODEL_MAP = { gemini: "gemini-3.5-flash", opus: "claude-opus-4-8", sonnet: "claude-sonnet-5" };
962
979
  const payload = await api.request("/api/tracker/ai/script", {
963
980
  body: {
964
981
  apiModel: SCRIPT_MODEL_MAP[model] || SCRIPT_MODEL_MAP.gemini,
965
- channelStyleProfile: null,
982
+ // 지문이 있으면 styleProfile 껍데기에 실어 보낸다 — 서버 script-prompts 가
983
+ // channelStyleProfile.fingerprint 에서 실측 규격 지시문을 뽑는다.
984
+ channelStyleProfile: channelFingerprint ? { fingerprint: channelFingerprint } : null,
966
985
  channelStyleReferenceText: "",
967
986
  christianContentType,
968
987
  contentFormat,
@@ -1445,13 +1464,11 @@ export function registerCloudTools(server, config, api) {
1445
1464
  wrapCloudHandler(config, async ({ videoUrls }) => {
1446
1465
  const specs = [];
1447
1466
  const notes = [];
1448
- // 서버 채널지문 라우트의 본문 상한은 16MB. 표본당 120초 WAV(16kHz·16bit 모노)는
1449
- // base64 ~5.1MB 라 표본 3개만 넘어도 합산이 상한을 넘겨 전체 413 으로 죽는다.
1450
- // 표본 수에 맞춰 오디오 트림 길이를 나눠 갖고, 누적 크기도 방어적으로 감시한다.
1451
- const audioBudgetChars = 9_000_000;
1452
- const base64CharsPerAudioSecond = Math.ceil(COMPANION_AUDIO_BYTES_PER_SECOND * 4 / 3);
1453
- const audioTrimSeconds = Math.max(20, Math.min(120, Math.floor(audioBudgetChars / videoUrls.length / base64CharsPerAudioSecond)));
1454
- let audioCharsUsed = 0;
1467
+ // 라우드니스·무음은 여기서 재고 "숫자만" 보낸다. 오디오를 서버로 올리면 표본 3편만
1468
+ // 넘어도 서버 본문 상한(16MB)을 넘겨 요청 전체가 죽고, 그걸 피하려 트림을 줄이면
1469
+ // 표본이 많을수록 측정 구간이 짧아진다. 웹(src/lib/channelFingerprint.js)도 같은 이유로
1470
+ // 브라우저에서 재고 숫자만 보낸다.
1471
+ const audioTrimSeconds = 120;
1455
1472
  for (const [index, rawUrl] of videoUrls.entries()) {
1456
1473
  const watchUrl = canonicalYoutubeWatchUrl(rawUrl);
1457
1474
  if (!watchUrl) {
@@ -1485,25 +1502,24 @@ export function registerCloudTools(server, config, api) {
1485
1502
  if (cutCountClamped) {
1486
1503
  notes.push(`표본 ${index + 1}: 감지 컷이 컴패니언 상한(${COMPANION_MAX_SHOTS})에 도달 — 컷 수는 하한값이라 컷 리듬 축에서 제외`);
1487
1504
  }
1488
- // 오디오 — 라우드니스/무음은 서버가 검증된 DSP 로 실측.
1489
- let audioBase64 = "";
1505
+ // 오디오 — 여기서 실측해 숫자만 싣는다(서버 DSP 와 동일 알고리즘, wav-dsp.mjs).
1506
+ let loudnessDbfs = null;
1507
+ let silenceRatio = null;
1490
1508
  try {
1491
1509
  const wav = await extractCompanionAudio(config.companionUrl, { url: watchUrl });
1492
- const encoded = trimWavToSeconds(wav, audioTrimSeconds).toString("base64");
1493
- if (audioCharsUsed + encoded.length > audioBudgetChars) {
1494
- notes.push(`표본 ${index + 1}: 본문 예산 초과로 오디오 제외 — 라우드니스 축 제외`);
1495
- } else {
1496
- audioBase64 = encoded;
1497
- audioCharsUsed += encoded.length;
1498
- }
1510
+ const trimmed = trimWavToSeconds(wav, audioTrimSeconds);
1511
+ loudnessDbfs = loudnessDbfsFromWav(trimmed);
1512
+ silenceRatio = silenceRatioFromWav(trimmed);
1513
+ if (loudnessDbfs === null) notes.push(`표본 ${index + 1}: WAV 파싱 실패 — 라우드니스 축 제외`);
1499
1514
  } catch (error) {
1500
1515
  notes.push(`표본 ${index + 1}: 오디오 추출 실패(${String(error?.message || error).slice(0, 80)}) — 라우드니스 축 제외`);
1501
1516
  }
1502
1517
  specs.push({
1503
- audioBase64,
1504
1518
  ...(cutCountClamped ? {} : { cutCount: shots.length }),
1505
1519
  durationSec,
1506
1520
  frames,
1521
+ ...(loudnessDbfs !== null ? { loudnessDbfs } : {}),
1522
+ ...(silenceRatio !== null ? { silenceRatio } : {}),
1507
1523
  videoId: watchUrl.slice(-11),
1508
1524
  });
1509
1525
  } catch (error) {
package/lib/config.mjs CHANGED
@@ -2,7 +2,7 @@ import { homedir } from "node:os";
2
2
  import path from "node:path";
3
3
 
4
4
  // 프록시 버전 — 서버가 X-AImakeAll-MCP-Version 으로 하한을 강제(426)할 수 있다.
5
- export const MCP_PROXY_VERSION = "0.8.0";
5
+ export const MCP_PROXY_VERSION = "0.9.0";
6
6
 
7
7
  export const DEFAULT_API_BASE = "https://aimakeall.com";
8
8
  export const DEFAULT_COMPANION_URL = "http://127.0.0.1:9876";
@@ -0,0 +1,82 @@
1
+ // WAV 라우드니스·무음 실측 — server/channel-fingerprint.mjs 의 동일 알고리즘을 MCP 로 이식한 것.
2
+ // 이식하는 이유: 지문 측정에서 오디오를 서버로 올리면 표본 3편만 넘어도 서버 본문 상한(16MB)을
3
+ // 넘겨 요청 전체가 죽는다. 웹(src/lib/channelFingerprint.js)도 같은 이유로 브라우저에서 재고
4
+ // 숫자만 보낸다. MCP 는 별도 npm 패키지라 server/ 를 import 할 수 없어 코드를 복제하되,
5
+ // wav-dsp.test.mjs 가 서버 구현과의 수치 일치를 계약으로 고정한다.
6
+ //
7
+ // 세 구현(서버·웹·MCP)은 반드시 같은 숫자를 내야 한다. 한쪽을 고치면 나머지도 같이 고칠 것.
8
+
9
+ export function parseWavPcm(buffer) {
10
+ if (!Buffer.isBuffer(buffer) || buffer.length < 44) return null;
11
+ if (buffer.toString("ascii", 0, 4) !== "RIFF" || buffer.toString("ascii", 8, 12) !== "WAVE") return null;
12
+ let offset = 12;
13
+ let fmt = null;
14
+ let dataOffset = -1;
15
+ let dataLen = 0;
16
+ while (offset + 8 <= buffer.length) {
17
+ const id = buffer.toString("ascii", offset, offset + 4);
18
+ const size = buffer.readUInt32LE(offset + 4);
19
+ const body = offset + 8;
20
+ // fmt 청크 헤더는 있으나 본문(16바이트)이 잘린 버퍼는 건너뛴다 — 읽기 범위 초과 대신 null.
21
+ if (id === "fmt " && body + 16 <= buffer.length) {
22
+ fmt = {
23
+ audioFormat: buffer.readUInt16LE(body),
24
+ channels: buffer.readUInt16LE(body + 2),
25
+ sampleRate: buffer.readUInt32LE(body + 4),
26
+ bitsPerSample: buffer.readUInt16LE(body + 14),
27
+ };
28
+ } else if (id === "data") {
29
+ dataOffset = body;
30
+ dataLen = Math.min(size, buffer.length - body);
31
+ }
32
+ offset = body + size + (size % 2); // chunks are word-aligned
33
+ }
34
+ if (!fmt || dataOffset < 0 || fmt.bitsPerSample !== 16 || fmt.audioFormat !== 1) return null;
35
+ return { ...fmt, dataOffset, dataLen };
36
+ }
37
+
38
+ // 16-bit PCM 표본을 순회하며 콜백. 첫 채널만 사용(모노 가정, 스테레오면 좌채널).
39
+ function forEachSample(buffer, wav, fn) {
40
+ const step = 2 * Math.max(1, wav.channels);
41
+ for (let i = wav.dataOffset; i + 1 < wav.dataOffset + wav.dataLen; i += step) {
42
+ fn(buffer.readInt16LE(i) / 32768);
43
+ }
44
+ }
45
+
46
+ // RMS → dBFS. 무음(전부 0)이면 -Infinity 대신 -100 으로 바닥 처리.
47
+ export function loudnessDbfsFromWav(buffer) {
48
+ const wav = parseWavPcm(buffer);
49
+ if (!wav) return null;
50
+ let sumSq = 0;
51
+ let n = 0;
52
+ forEachSample(buffer, wav, (s) => { sumSq += s * s; n += 1; });
53
+ if (!n) return null;
54
+ const rms = Math.sqrt(sumSq / n);
55
+ if (rms <= 0) return -100;
56
+ return Math.max(-100, Math.round(20 * Math.log10(rms) * 10) / 10);
57
+ }
58
+
59
+ // 무음 비율 — 짧은 창(20ms)의 RMS 가 임계(dBFS) 아래인 시간 비율. 0~1.
60
+ export function silenceRatioFromWav(buffer, { thresholdDb = -45, windowMs = 20 } = {}) {
61
+ const wav = parseWavPcm(buffer);
62
+ if (!wav) return null;
63
+ const win = Math.max(1, Math.floor((wav.sampleRate * windowMs) / 1000));
64
+ const thresholdRms = Math.pow(10, thresholdDb / 20);
65
+ let windows = 0;
66
+ let silent = 0;
67
+ let sumSq = 0;
68
+ let count = 0;
69
+ forEachSample(buffer, wav, (s) => {
70
+ sumSq += s * s;
71
+ count += 1;
72
+ if (count >= win) {
73
+ const rms = Math.sqrt(sumSq / count);
74
+ windows += 1;
75
+ if (rms < thresholdRms) silent += 1;
76
+ sumSq = 0;
77
+ count = 0;
78
+ }
79
+ });
80
+ if (!windows) return null;
81
+ return Math.round((silent / windows) * 1000) / 1000;
82
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "aimakeall-mcp",
3
- "version": "0.8.0",
3
+ "version": "0.9.0",
4
4
  "description": "AImakeAll MCP 서버 — Claude Code/Codex에서 자연어로 영상 기획·생성·렌더·퍼블리시 (렌더는 로컬 컴패니언)",
5
5
  "type": "module",
6
6
  "bin": {