aimakeall-mcp 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/lib/cloud-tools.mjs +40 -24
- package/lib/config.mjs +1 -1
- package/lib/wav-dsp.mjs +82 -0
- package/package.json +1 -1
package/README.md
CHANGED
package/lib/cloud-tools.mjs
CHANGED
|
@@ -23,16 +23,15 @@ import {
|
|
|
23
23
|
import { mp3DurationSec } from "./mp3-duration.mjs";
|
|
24
24
|
import { canonicalYoutubeWatchUrl } from "./payload-guard.mjs";
|
|
25
25
|
import { createUsageEventId } from "./usage-event.mjs";
|
|
26
|
+
import { loudnessDbfsFromWav, silenceRatioFromWav } from "./wav-dsp.mjs";
|
|
26
27
|
import { trimWavToSeconds } from "./wav-trim.mjs";
|
|
27
28
|
|
|
28
29
|
// 여러 파일을 base64 로 싣는 요청의 인코딩 크기 합이 서버 JSON 상한을 넘지 않게 사전 검사.
|
|
29
30
|
// 서버가 전체 업로드를 받은 뒤 413 을 내는 것을 막고 행동 가능한 한국어 안내를 준다.
|
|
30
31
|
const BODY_BUDGET_BYTES = SERVER_JSON_BODY_LIMIT_BYTES - 512 * 1024; // 바디의 나머지 필드 여유
|
|
31
32
|
|
|
32
|
-
// 컴패니언 계약 상수 — companion/index.cjs 의 MAX_SHOT_COUNT
|
|
33
|
-
// 오디오는 extractCompanionAudio 기본값(16kHz)·16bit·모노 = 초당 32,000 바이트.
|
|
33
|
+
// 컴패니언 계약 상수 — companion/index.cjs 의 MAX_SHOT_COUNT 와 맞춘다.
|
|
34
34
|
const COMPANION_MAX_SHOTS = 18;
|
|
35
|
-
const COMPANION_AUDIO_BYTES_PER_SECOND = 16_000 * 2;
|
|
36
35
|
|
|
37
36
|
function assertEncodedBudget(filePaths, { context = "요청" } = {}) {
|
|
38
37
|
let total = 0;
|
|
@@ -253,7 +252,7 @@ export function registerCloudTools(server, config, api) {
|
|
|
253
252
|
categoryId: z.enum(["community-shorts", "viral-shorts", "insight"]).optional().describe("기본 community-shorts. insight=명언(통찰) 쇼츠"),
|
|
254
253
|
targetCustomer: z.string().optional(),
|
|
255
254
|
tone: z.string().optional().describe("기본 '자동 추천'"),
|
|
256
|
-
sceneCount: z.number().int().min(1).max(10).optional().describe("
|
|
255
|
+
sceneCount: z.number().int().min(1).max(10).optional().describe("씬 수. 생략하면 channelFingerprint 의 목표 길이로 자동 산출(지문도 없으면 3)"),
|
|
257
256
|
channelFingerprint: z.object({
|
|
258
257
|
sampleSize: z.number().optional(),
|
|
259
258
|
durationTargetSec: z.number().nullable().optional(),
|
|
@@ -276,7 +275,7 @@ export function registerCloudTools(server, config, api) {
|
|
|
276
275
|
forbidden: z.string().optional().describe("금지 변형 (기본: different face, different hairstyle)"),
|
|
277
276
|
})).max(4).optional().describe("등장 캐릭터 — 서버가 씬마다 identity_lock 을 imagePrompt 에 강제 주입해 캐릭터 일관성을 잠급니다"),
|
|
278
277
|
},
|
|
279
|
-
wrapCloudHandler(config, async ({ topic = "", copy = "", categoryId = "community-shorts", targetCustomer = "", tone = "자동 추천", sceneCount =
|
|
278
|
+
wrapCloudHandler(config, async ({ topic = "", copy = "", categoryId = "community-shorts", targetCustomer = "", tone = "자동 추천", sceneCount = 0, characters = [], channelFingerprint = null }) => {
|
|
280
279
|
if (!String(topic).trim() && !String(copy).trim()) {
|
|
281
280
|
return textResult("topic 또는 copy 중 하나는 입력해야 합니다.", { isError: true });
|
|
282
281
|
}
|
|
@@ -780,9 +779,14 @@ export function registerCloudTools(server, config, api) {
|
|
|
780
779
|
emphasisKeywords: z.array(z.string()).max(12).optional().describe("강조 단어 목록 — 자막 라인 안에서 해당 단어만 다른 색/크기로 렌더 (plan_shorts_video의 emphasisKeywords 를 그대로 전달)"),
|
|
781
780
|
emphasisColor: z.string().regex(/^#[0-9a-fA-F]{6}$/, "#RRGGBB 형식").optional().describe("강조 색 #RRGGBB (기본 #facc15 노랑)"),
|
|
782
781
|
emphasisSizeScale: z.number().optional().describe("강조 크기 배율 0.5~2 (기본 1.2)"),
|
|
782
|
+
channelFingerprint: z.object({
|
|
783
|
+
subtitleYPct: z.number().nullable().optional(),
|
|
784
|
+
subtitleLines: z.number().nullable().optional(),
|
|
785
|
+
hasBgm: z.boolean().nullable().optional(),
|
|
786
|
+
}).passthrough().optional().describe("measure_channel_spec 이 반환한 fingerprint — 자막 세로 위치·BGM 유무를 참고 채널에 맞춥니다"),
|
|
783
787
|
transitionMode: z.enum(["fade", "none"]).optional().describe("기본 fade"),
|
|
784
788
|
},
|
|
785
|
-
wrapCloudHandler(config, async ({ sceneVideos, featureKey = "commerceVideo", projectTitle = "aimakeall-mcp", aspectRatio = "9:16", ttsAudioPath = "", bgmAudioPath = "", audioDurationSec = 0, bgmVolume = null, muteVideoAudio = true, titleText = "", subtitleLines = [], emphasisKeywords = [], emphasisColor = "", emphasisSizeScale = 0, transitionMode = "fade" }) => {
|
|
789
|
+
wrapCloudHandler(config, async ({ sceneVideos, featureKey = "commerceVideo", projectTitle = "aimakeall-mcp", aspectRatio = "9:16", ttsAudioPath = "", bgmAudioPath = "", audioDurationSec = 0, bgmVolume = null, muteVideoAudio = true, titleText = "", subtitleLines = [], emphasisKeywords = [], emphasisColor = "", emphasisSizeScale = 0, transitionMode = "fade", channelFingerprint = null }) => {
|
|
786
790
|
// 오디오 확장자 검증 + 합산 크기 예산(서버 413 사전 차단).
|
|
787
791
|
if (ttsAudioPath) assertAllowedInputFile(ttsAudioPath, new Set([".mp3", ".wav"]), { kind: "TTS 오디오" });
|
|
788
792
|
if (bgmAudioPath) assertAllowedInputFile(bgmAudioPath, new Set([".mp3", ".wav"]), { kind: "BGM 오디오" });
|
|
@@ -807,6 +811,7 @@ export function registerCloudTools(server, config, api) {
|
|
|
807
811
|
})),
|
|
808
812
|
emphasisColor,
|
|
809
813
|
emphasisKeywords,
|
|
814
|
+
channelFingerprint,
|
|
810
815
|
emphasisSizeScale,
|
|
811
816
|
subtitleLines,
|
|
812
817
|
titleText,
|
|
@@ -944,6 +949,17 @@ export function registerCloudTools(server, config, api) {
|
|
|
944
949
|
christianContentType: z.enum(["youtube-narration", "prayer", "devotional", "comfort-message"]).optional().describe("styleId=christian일 때 유형"),
|
|
945
950
|
insightFigureNames: z.string().optional().describe("styleId=insight일 때 인물명(쉼표 구분)"),
|
|
946
951
|
userOpinion: z.string().optional().describe("본문에 자연스럽게 녹일 작성자 의견"),
|
|
952
|
+
channelFingerprint: z.object({
|
|
953
|
+
durationTargetSec: z.number().nullable().optional(),
|
|
954
|
+
cutIntervalSec: z.number().nullable().optional(),
|
|
955
|
+
cpsTarget: z.number().nullable().optional(),
|
|
956
|
+
subtitleYPct: z.number().nullable().optional(),
|
|
957
|
+
subtitleLines: z.number().nullable().optional(),
|
|
958
|
+
endingStyle: z.string().nullable().optional(),
|
|
959
|
+
emphasisCount: z.number().nullable().optional(),
|
|
960
|
+
hasBgm: z.boolean().nullable().optional(),
|
|
961
|
+
sampleSize: z.number().optional(),
|
|
962
|
+
}).passthrough().optional().describe("measure_channel_spec 의 fingerprint — 실측 규격(길이·컷·말끝)을 대본에 강제합니다"),
|
|
947
963
|
},
|
|
948
964
|
wrapCloudHandler(config, async ({
|
|
949
965
|
title,
|
|
@@ -957,12 +973,15 @@ export function registerCloudTools(server, config, api) {
|
|
|
957
973
|
christianContentType = "youtube-narration",
|
|
958
974
|
insightFigureNames = "",
|
|
959
975
|
userOpinion = "",
|
|
976
|
+
channelFingerprint = null,
|
|
960
977
|
}) => {
|
|
961
978
|
const SCRIPT_MODEL_MAP = { gemini: "gemini-3.5-flash", opus: "claude-opus-4-8", sonnet: "claude-sonnet-5" };
|
|
962
979
|
const payload = await api.request("/api/tracker/ai/script", {
|
|
963
980
|
body: {
|
|
964
981
|
apiModel: SCRIPT_MODEL_MAP[model] || SCRIPT_MODEL_MAP.gemini,
|
|
965
|
-
|
|
982
|
+
// 지문이 있으면 styleProfile 껍데기에 실어 보낸다 — 서버 script-prompts 가
|
|
983
|
+
// channelStyleProfile.fingerprint 에서 실측 규격 지시문을 뽑는다.
|
|
984
|
+
channelStyleProfile: channelFingerprint ? { fingerprint: channelFingerprint } : null,
|
|
966
985
|
channelStyleReferenceText: "",
|
|
967
986
|
christianContentType,
|
|
968
987
|
contentFormat,
|
|
@@ -1445,13 +1464,11 @@ export function registerCloudTools(server, config, api) {
|
|
|
1445
1464
|
wrapCloudHandler(config, async ({ videoUrls }) => {
|
|
1446
1465
|
const specs = [];
|
|
1447
1466
|
const notes = [];
|
|
1448
|
-
//
|
|
1449
|
-
//
|
|
1450
|
-
//
|
|
1451
|
-
|
|
1452
|
-
const
|
|
1453
|
-
const audioTrimSeconds = Math.max(20, Math.min(120, Math.floor(audioBudgetChars / videoUrls.length / base64CharsPerAudioSecond)));
|
|
1454
|
-
let audioCharsUsed = 0;
|
|
1467
|
+
// 라우드니스·무음은 여기서 재고 "숫자만" 보낸다. 오디오를 서버로 올리면 표본 3편만
|
|
1468
|
+
// 넘어도 서버 본문 상한(16MB)을 넘겨 요청 전체가 죽고, 그걸 피하려 트림을 줄이면
|
|
1469
|
+
// 표본이 많을수록 측정 구간이 짧아진다. 웹(src/lib/channelFingerprint.js)도 같은 이유로
|
|
1470
|
+
// 브라우저에서 재고 숫자만 보낸다.
|
|
1471
|
+
const audioTrimSeconds = 120;
|
|
1455
1472
|
for (const [index, rawUrl] of videoUrls.entries()) {
|
|
1456
1473
|
const watchUrl = canonicalYoutubeWatchUrl(rawUrl);
|
|
1457
1474
|
if (!watchUrl) {
|
|
@@ -1485,25 +1502,24 @@ export function registerCloudTools(server, config, api) {
|
|
|
1485
1502
|
if (cutCountClamped) {
|
|
1486
1503
|
notes.push(`표본 ${index + 1}: 감지 컷이 컴패니언 상한(${COMPANION_MAX_SHOTS})에 도달 — 컷 수는 하한값이라 컷 리듬 축에서 제외`);
|
|
1487
1504
|
}
|
|
1488
|
-
// 오디오 —
|
|
1489
|
-
let
|
|
1505
|
+
// 오디오 — 여기서 실측해 숫자만 싣는다(서버 DSP 와 동일 알고리즘, wav-dsp.mjs).
|
|
1506
|
+
let loudnessDbfs = null;
|
|
1507
|
+
let silenceRatio = null;
|
|
1490
1508
|
try {
|
|
1491
1509
|
const wav = await extractCompanionAudio(config.companionUrl, { url: watchUrl });
|
|
1492
|
-
const
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
}
|
|
1496
|
-
audioBase64 = encoded;
|
|
1497
|
-
audioCharsUsed += encoded.length;
|
|
1498
|
-
}
|
|
1510
|
+
const trimmed = trimWavToSeconds(wav, audioTrimSeconds);
|
|
1511
|
+
loudnessDbfs = loudnessDbfsFromWav(trimmed);
|
|
1512
|
+
silenceRatio = silenceRatioFromWav(trimmed);
|
|
1513
|
+
if (loudnessDbfs === null) notes.push(`표본 ${index + 1}: WAV 파싱 실패 — 라우드니스 축 제외`);
|
|
1499
1514
|
} catch (error) {
|
|
1500
1515
|
notes.push(`표본 ${index + 1}: 오디오 추출 실패(${String(error?.message || error).slice(0, 80)}) — 라우드니스 축 제외`);
|
|
1501
1516
|
}
|
|
1502
1517
|
specs.push({
|
|
1503
|
-
audioBase64,
|
|
1504
1518
|
...(cutCountClamped ? {} : { cutCount: shots.length }),
|
|
1505
1519
|
durationSec,
|
|
1506
1520
|
frames,
|
|
1521
|
+
...(loudnessDbfs !== null ? { loudnessDbfs } : {}),
|
|
1522
|
+
...(silenceRatio !== null ? { silenceRatio } : {}),
|
|
1507
1523
|
videoId: watchUrl.slice(-11),
|
|
1508
1524
|
});
|
|
1509
1525
|
} catch (error) {
|
package/lib/config.mjs
CHANGED
|
@@ -2,7 +2,7 @@ import { homedir } from "node:os";
|
|
|
2
2
|
import path from "node:path";
|
|
3
3
|
|
|
4
4
|
// 프록시 버전 — 서버가 X-AImakeAll-MCP-Version 으로 하한을 강제(426)할 수 있다.
|
|
5
|
-
export const MCP_PROXY_VERSION = "0.
|
|
5
|
+
export const MCP_PROXY_VERSION = "0.9.0";
|
|
6
6
|
|
|
7
7
|
export const DEFAULT_API_BASE = "https://aimakeall.com";
|
|
8
8
|
export const DEFAULT_COMPANION_URL = "http://127.0.0.1:9876";
|
package/lib/wav-dsp.mjs
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
// WAV 라우드니스·무음 실측 — server/channel-fingerprint.mjs 의 동일 알고리즘을 MCP 로 이식한 것.
|
|
2
|
+
// 이식하는 이유: 지문 측정에서 오디오를 서버로 올리면 표본 3편만 넘어도 서버 본문 상한(16MB)을
|
|
3
|
+
// 넘겨 요청 전체가 죽는다. 웹(src/lib/channelFingerprint.js)도 같은 이유로 브라우저에서 재고
|
|
4
|
+
// 숫자만 보낸다. MCP 는 별도 npm 패키지라 server/ 를 import 할 수 없어 코드를 복제하되,
|
|
5
|
+
// wav-dsp.test.mjs 가 서버 구현과의 수치 일치를 계약으로 고정한다.
|
|
6
|
+
//
|
|
7
|
+
// 세 구현(서버·웹·MCP)은 반드시 같은 숫자를 내야 한다. 한쪽을 고치면 나머지도 같이 고칠 것.
|
|
8
|
+
|
|
9
|
+
export function parseWavPcm(buffer) {
|
|
10
|
+
if (!Buffer.isBuffer(buffer) || buffer.length < 44) return null;
|
|
11
|
+
if (buffer.toString("ascii", 0, 4) !== "RIFF" || buffer.toString("ascii", 8, 12) !== "WAVE") return null;
|
|
12
|
+
let offset = 12;
|
|
13
|
+
let fmt = null;
|
|
14
|
+
let dataOffset = -1;
|
|
15
|
+
let dataLen = 0;
|
|
16
|
+
while (offset + 8 <= buffer.length) {
|
|
17
|
+
const id = buffer.toString("ascii", offset, offset + 4);
|
|
18
|
+
const size = buffer.readUInt32LE(offset + 4);
|
|
19
|
+
const body = offset + 8;
|
|
20
|
+
// fmt 청크 헤더는 있으나 본문(16바이트)이 잘린 버퍼는 건너뛴다 — 읽기 범위 초과 대신 null.
|
|
21
|
+
if (id === "fmt " && body + 16 <= buffer.length) {
|
|
22
|
+
fmt = {
|
|
23
|
+
audioFormat: buffer.readUInt16LE(body),
|
|
24
|
+
channels: buffer.readUInt16LE(body + 2),
|
|
25
|
+
sampleRate: buffer.readUInt32LE(body + 4),
|
|
26
|
+
bitsPerSample: buffer.readUInt16LE(body + 14),
|
|
27
|
+
};
|
|
28
|
+
} else if (id === "data") {
|
|
29
|
+
dataOffset = body;
|
|
30
|
+
dataLen = Math.min(size, buffer.length - body);
|
|
31
|
+
}
|
|
32
|
+
offset = body + size + (size % 2); // chunks are word-aligned
|
|
33
|
+
}
|
|
34
|
+
if (!fmt || dataOffset < 0 || fmt.bitsPerSample !== 16 || fmt.audioFormat !== 1) return null;
|
|
35
|
+
return { ...fmt, dataOffset, dataLen };
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
// 16-bit PCM 표본을 순회하며 콜백. 첫 채널만 사용(모노 가정, 스테레오면 좌채널).
|
|
39
|
+
function forEachSample(buffer, wav, fn) {
|
|
40
|
+
const step = 2 * Math.max(1, wav.channels);
|
|
41
|
+
for (let i = wav.dataOffset; i + 1 < wav.dataOffset + wav.dataLen; i += step) {
|
|
42
|
+
fn(buffer.readInt16LE(i) / 32768);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// RMS → dBFS. 무음(전부 0)이면 -Infinity 대신 -100 으로 바닥 처리.
|
|
47
|
+
export function loudnessDbfsFromWav(buffer) {
|
|
48
|
+
const wav = parseWavPcm(buffer);
|
|
49
|
+
if (!wav) return null;
|
|
50
|
+
let sumSq = 0;
|
|
51
|
+
let n = 0;
|
|
52
|
+
forEachSample(buffer, wav, (s) => { sumSq += s * s; n += 1; });
|
|
53
|
+
if (!n) return null;
|
|
54
|
+
const rms = Math.sqrt(sumSq / n);
|
|
55
|
+
if (rms <= 0) return -100;
|
|
56
|
+
return Math.max(-100, Math.round(20 * Math.log10(rms) * 10) / 10);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// 무음 비율 — 짧은 창(20ms)의 RMS 가 임계(dBFS) 아래인 시간 비율. 0~1.
|
|
60
|
+
export function silenceRatioFromWav(buffer, { thresholdDb = -45, windowMs = 20 } = {}) {
|
|
61
|
+
const wav = parseWavPcm(buffer);
|
|
62
|
+
if (!wav) return null;
|
|
63
|
+
const win = Math.max(1, Math.floor((wav.sampleRate * windowMs) / 1000));
|
|
64
|
+
const thresholdRms = Math.pow(10, thresholdDb / 20);
|
|
65
|
+
let windows = 0;
|
|
66
|
+
let silent = 0;
|
|
67
|
+
let sumSq = 0;
|
|
68
|
+
let count = 0;
|
|
69
|
+
forEachSample(buffer, wav, (s) => {
|
|
70
|
+
sumSq += s * s;
|
|
71
|
+
count += 1;
|
|
72
|
+
if (count >= win) {
|
|
73
|
+
const rms = Math.sqrt(sumSq / count);
|
|
74
|
+
windows += 1;
|
|
75
|
+
if (rms < thresholdRms) silent += 1;
|
|
76
|
+
sumSq = 0;
|
|
77
|
+
count = 0;
|
|
78
|
+
}
|
|
79
|
+
});
|
|
80
|
+
if (!windows) return null;
|
|
81
|
+
return Math.round((silent / windows) * 1000) / 1000;
|
|
82
|
+
}
|