@lmliheng/lesson-video 0.0.0-stage → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,169 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * 逐帧渲染课件:把 deck.html 按 lesson.json 的时间轴定格成 30fps 的图片序列。
4
+ * 每一帧都由 seek(t) 从时间 t 纯函数地算出来,所以画面与配音、字幕严格对齐。
5
+ *
6
+ * 用法:
7
+ * PLAYWRIGHT_MODULE=<pw-core 路径> CHROME_PATH=<chrome.exe> \
8
+ * node lib/render.mjs --dir lessons/ep00-pilot [--fps 30] [--scale 0.5]
9
+ */
10
+ import fs from "node:fs";
11
+ import path from "node:path";
12
+ import { createRequire } from "node:module";
13
+ import { fileURLToPath, pathToFileURL } from "node:url";
14
+
15
+ const require = createRequire(import.meta.url);
16
+ const HERE = path.dirname(fileURLToPath(import.meta.url));
17
+ const PW = process.env.PLAYWRIGHT_MODULE || "playwright-core";
18
+ const CHROME = process.env.CHROME_PATH || "";
19
+ const pw = require(PW);
20
+ const { chromium } = pw;
21
+
22
+ const argv = process.argv.slice(2);
23
+ const arg = (k, d) => {
24
+ const i = argv.indexOf(k);
25
+ return i >= 0 ? argv[i + 1] : d;
26
+ };
27
+
28
+ const dir = path.resolve(arg("--dir", "lessons/ep00-pilot"));
29
+ // CSS 布局固定按 1080x1920 设计;--scale 只缩放光栅,用来快速预览
30
+ const dsf = parseFloat(arg("--scale", "1")) || 1;
31
+ const W = 1080,
32
+ H = 1920;
33
+
34
+ const lesson = JSON.parse(fs.readFileSync(path.join(dir, "lesson.json"), "utf8"));
35
+ const fps = parseInt(arg("--fps", String(lesson.fps || 30)), 10);
36
+ const framesDir = path.join(dir, "frames");
37
+ fs.rmSync(framesDir, { recursive: true, force: true });
38
+ fs.mkdirSync(framesDir, { recursive: true });
39
+
40
+ const total = Math.round(lesson.duration * fps);
41
+ console.log(
42
+ `渲染 ${lesson.meta.series} · ${lesson.meta.episode}: ${lesson.duration}s × ${fps}fps = ${total} 帧 @ ${W * dsf}x${H * dsf}`,
43
+ );
44
+
45
+ const browser = await chromium.launch({
46
+ executablePath: CHROME || undefined,
47
+ args: [
48
+ "--no-sandbox",
49
+ "--disable-dev-shm-usage",
50
+ "--hide-scrollbars",
51
+ "--force-color-profile=srgb",
52
+ ],
53
+ });
54
+ const ctx = await browser.newContext({ viewport: { width: W, height: H }, deviceScaleFactor: dsf });
55
+ const page = await ctx.newPage();
56
+ await page.addInitScript(`window.__LESSON__ = ${JSON.stringify(lesson)};`);
57
+ // 每集可以自带 deck.html 覆盖版式;没有就用 lib/deck.html 这份共用的
58
+ const localDeck = path.join(dir, "deck.html");
59
+ const deck = fs.existsSync(localDeck) ? localDeck : path.join(HERE, "deck.html");
60
+ await page.goto(pathToFileURL(deck).href, { waitUntil: "load" });
61
+ await page.evaluate(() => window.__init(window.__LESSON__));
62
+ await page.evaluate(() => document.fonts.ready);
63
+
64
+ // 安全区自检:字幕带从 y=1470 开始,任何一幕的内容越线都会盖住字幕。
65
+ // 逐幕快进到结束前,量一遍可见元素的底边。
66
+ const SAFE_BOTTOM = 1470;
67
+ const over = await page.evaluate((safe) => {
68
+ const bad = [];
69
+ const nodes = document.querySelectorAll(".scene");
70
+ window.__LESSON__.scenes.forEach((s, i) => {
71
+ window.seek(s.start + s.dur - 0.35);
72
+ let worst = 0,
73
+ who = "";
74
+ nodes[i].querySelectorAll("*").forEach((n) => {
75
+ const r = n.getBoundingClientRect();
76
+ if (r.height > 0 && r.bottom > worst) {
77
+ worst = r.bottom;
78
+ who = (n.className || n.tagName).toString();
79
+ }
80
+ });
81
+ if (worst > safe) bad.push({ scene: s.id, bottom: Math.round(worst), el: who.slice(0, 40) });
82
+ });
83
+ return bad;
84
+ }, SAFE_BOTTOM);
85
+ if (over.length) {
86
+ console.log("⚠ 越出字幕安全线:");
87
+ over.forEach((o) => console.log(` ${o.scene} 底边 ${o.bottom} (${o.el})`));
88
+ } else {
89
+ console.log(`安全区自检通过(所有内容底边 < ${SAFE_BOTTOM})`);
90
+ }
91
+
92
+ // 代码行比卡片宽就会被悄悄裁掉(.card-bd 是 overflow:hidden),这里逐幕量一遍
93
+ const clipped = await page.evaluate(() => {
94
+ const bad = [];
95
+ document.querySelectorAll(".scene").forEach((sc, i) => {
96
+ const bd = sc.querySelector(".card-bd");
97
+ if (bd && bd.scrollWidth > bd.clientWidth + 1) {
98
+ bad.push({
99
+ scene: sc.classList.contains("on") ? i : i,
100
+ w: bd.scrollWidth,
101
+ box: bd.clientWidth,
102
+ });
103
+ }
104
+ });
105
+ return bad;
106
+ });
107
+ if (clipped.length) {
108
+ console.log("⚠ 代码行被裁:");
109
+ clipped.forEach((o) => console.log(` 第 ${o.scene + 1} 幕 内容 ${o.w}px > 卡片 ${o.box}px`));
110
+ } else {
111
+ console.log("代码宽度自检通过(没有行被裁)");
112
+ }
113
+
114
+ // 标题折行后末行只剩一两个字(孤字)会被安全区挤出来,这里量一遍每个标题的行宽
115
+ const orphans = await page.evaluate(() => {
116
+ const bad = [];
117
+ const nodes = document.querySelectorAll(".scene");
118
+ nodes.forEach((sc, i) => {
119
+ // 先切到这一幕:.scene 默认 display:none,隐藏元素的 Range 量不到行框
120
+ const s = window.__LESSON__.scenes[i];
121
+ window.seek(s.start + s.dur * 0.5);
122
+ const h = sc.querySelector(".hero, .stitle, .outro");
123
+ if (!h) return;
124
+ const range = document.createRange();
125
+ range.selectNodeContents(h);
126
+ const rects = [...range.getClientRects()].filter((r) => r.width > 0);
127
+ if (rects.length < 2) return;
128
+ const widths = rects.map((r) => r.width);
129
+ const last = widths[widths.length - 1],
130
+ max = Math.max(...widths);
131
+ if (last < max * 0.45)
132
+ bad.push({ scene: i + 1, text: h.textContent, last: Math.round(last), max: Math.round(max) });
133
+ });
134
+ return bad;
135
+ });
136
+ if (orphans.length) {
137
+ console.log("⚠ 标题末行过短(孤字):");
138
+ orphans.forEach((o) =>
139
+ console.log(` 第 ${o.scene} 幕 「${o.text}」 末行 ${o.last}px / 最长行 ${o.max}px`),
140
+ );
141
+ } else {
142
+ console.log("标题折行自检通过(没有孤字行)");
143
+ }
144
+
145
+ const t0 = Date.now();
146
+ if (argv.includes("--check")) {
147
+ // 只做自检,不出帧
148
+ await browser.close();
149
+ process.exit(over.length || clipped.length || orphans.length ? 1 : 0);
150
+ }
151
+ for (let i = 0; i < total; i++) {
152
+ const t = i / fps;
153
+ await page.evaluate((tt) => window.seek(tt), t);
154
+ await page.screenshot({
155
+ path: path.join(framesDir, `f_${String(i).padStart(5, "0")}.jpg`),
156
+ type: "jpeg",
157
+ quality: 95,
158
+ });
159
+ if (i % 30 === 0 || i === total - 1) {
160
+ const done = i + 1,
161
+ el = (Date.now() - t0) / 1000;
162
+ process.stdout.write(
163
+ `\r ${done}/${total} ${((el / done) * 1000).toFixed(0)}ms/帧 eta ${(((el / done) * (total - done)) / 60).toFixed(1)} min `,
164
+ );
165
+ }
166
+ }
167
+ process.stdout.write("\n");
168
+ await browser.close();
169
+ console.log(`OK ${total} 帧 -> ${framesDir}`);
@@ -0,0 +1,223 @@
1
+ """把分镜脚本变成「配音 + 精确时间轴」。
2
+
3
+ 流程:edge-tts 逐幕合成配音(同时拿到逐词边界)→ 转 48k 单声道 wav 量长度 →
4
+ 拼出全局时间轴 → 写出 lesson.json(课件渲染与视频合成共用这一份数据)。
5
+
6
+ 用法: python lib/tts.py lessons/ep00-pilot
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import asyncio
11
+ import json
12
+ import re
13
+ import subprocess
14
+ import sys
15
+ import time
16
+ import wave
17
+ from pathlib import Path
18
+
19
+ import edge_tts
20
+ import imageio_ffmpeg
21
+
22
+ FFMPEG = imageio_ffmpeg.get_ffmpeg_exe()
23
+ PUNCT = ",。!?;:、"
24
+
25
+
26
+ async def synth(text: str, voice: str, rate: str, mp3: Path) -> list[dict]:
27
+ """合成一段配音,返回逐词边界(秒)。
28
+
29
+ edge-tts 7.x 默认 boundary='SentenceBoundary',拿不到逐词时间——必须显式要 WordBoundary,
30
+ 字幕才能贴着人声逐句走。
31
+ """
32
+ comm = edge_tts.Communicate(text, voice, rate=rate, boundary="WordBoundary")
33
+ words: list[dict] = []
34
+ with open(mp3, "wb") as f:
35
+ async for chunk in comm.stream():
36
+ if chunk["type"] == "audio":
37
+ f.write(chunk["data"])
38
+ elif chunk["type"] == "WordBoundary":
39
+ words.append({
40
+ "text": chunk["text"],
41
+ "start": chunk["offset"] / 1e7,
42
+ "dur": chunk["duration"] / 1e7,
43
+ })
44
+ return words
45
+
46
+
47
+ def synth_retry(text: str, voice: str, rate: str, mp3: Path, tries: int = 4) -> list[dict]:
48
+ """edge-tts 偶尔会空手而归(NoAudioReceived),重试几次就好。"""
49
+ last: Exception | None = None
50
+ for i in range(tries):
51
+ try:
52
+ words = asyncio.run(synth(text, voice, rate, mp3))
53
+ if words:
54
+ return words
55
+ last = RuntimeError("没有拿到逐词边界")
56
+ except Exception as e: # noqa: BLE001
57
+ last = e
58
+ time.sleep(1.5 * (i + 1))
59
+ print(f" 重试 {i + 1}/{tries - 1}:{type(last).__name__}")
60
+ raise SystemExit(f"配音合成失败:{last}")
61
+
62
+
63
+ def to_wav(mp3: Path, wav: Path) -> None:
64
+ subprocess.run([FFMPEG, "-y", "-hide_banner", "-loglevel", "error",
65
+ "-i", str(mp3), "-ar", "48000", "-ac", "1", "-c:a", "pcm_s16le", str(wav)], check=True)
66
+
67
+
68
+ def wav_duration(p: Path) -> float:
69
+ with wave.open(str(p), "rb") as w:
70
+ return w.getnframes() / w.getframerate()
71
+
72
+
73
+ def join_words(group: list[dict]) -> str:
74
+ """把词拼回一行;中文与西文交界处补一个空格,中英混排才不挤。"""
75
+ out = ""
76
+ for w in group:
77
+ t = w["text"]
78
+ if out and t:
79
+ a, b = out[-1], t[0]
80
+ ascii_a, ascii_b = a.isascii() and a.isalnum(), b.isascii() and b.isalnum()
81
+ if (ascii_a and ascii_b) or (ascii_a != ascii_b and (a.isalnum() or b.isalnum())):
82
+ out += " "
83
+ out += t
84
+ return out
85
+
86
+
87
+ def build_subtitles(text: str, words: list[dict], lead: float, max_len: int = 13) -> list[dict]:
88
+ """先按标点断句,再按**词**切短行——不能按字数硬切,否则会把 Kaggle 切一半。
89
+
90
+ 时间直接取该行首词与末词的 WordBoundary,所以字幕贴着人声走。
91
+ 句尾标点补回最后一行。
92
+ """
93
+ subs: list[dict] = []
94
+ wi = 0
95
+ for sent in re.findall(rf"[^{PUNCT}]+[{PUNCT}]*", text):
96
+ if not sent.strip():
97
+ continue
98
+ m = re.search(rf"[{PUNCT}]+$", sent)
99
+ tail = m.group(0) if m else ""
100
+ need = len(re.sub(r"\s", "", sent[: m.start()] if m else sent))
101
+
102
+ groups: list[list[dict]] = []
103
+ cur: list[dict] = []
104
+ consumed = 0 # 本句已吃掉的字数,用来判断这句的词吃完了没有
105
+ curlen = 0 # 当前这一行的字数,用来判断该不该换行
106
+ while consumed < need and wi < len(words):
107
+ t = words[wi]["text"]
108
+ if cur and curlen + len(t) > max_len:
109
+ groups.append(cur)
110
+ cur, curlen = [], 0
111
+ cur.append(words[wi])
112
+ curlen += len(t)
113
+ consumed += len(t)
114
+ wi += 1
115
+ if cur:
116
+ groups.append(cur)
117
+
118
+ for gi, g in enumerate(groups):
119
+ subs.append({
120
+ "text": join_words(g) + (tail if gi == len(groups) - 1 else ""),
121
+ "start": round(lead + g[0]["start"], 3),
122
+ "end": round(lead + g[-1]["start"] + g[-1]["dur"], 3),
123
+ })
124
+ return subs
125
+
126
+
127
+ def cue_time(cue: str, words: list[dict]) -> float | None:
128
+ """在逐词流里找到这句关键词,返回它第一个字被念到的时间(相对本幕)。
129
+
130
+ 刻意按「词」而不是按「字幕行」匹配:字幕行长度有限,像
131
+ 「X 转置 X 的逆」这种 cue 很可能正好横跨两行,按行匹配会匹配不到。
132
+ """
133
+ joined, owner = "", []
134
+ for wi, w in enumerate(words):
135
+ t = re.sub(r"\s", "", w["text"])
136
+ joined += t
137
+ owner.extend([wi] * len(t))
138
+ p = joined.find(re.sub(r"\s", "", cue))
139
+ return None if p < 0 else words[owner[p]]["start"]
140
+
141
+
142
+ def apply_cues(node, words: list[dict], lead: float = 0.15) -> None:
143
+ """把脚本里的 "cue": "关键词"(或 "<名>_cue")解析成出现秒数。
144
+
145
+ 写作时只写「念到这个词的时候,这块内容出现」,不用手算时间;
146
+ 配音合成后拿逐词时间反过来标定,画面和讲解自动咬合。已经写了 at 的不覆盖。
147
+ """
148
+ if isinstance(node, dict):
149
+ for k in [kk for kk in list(node) if kk.endswith("_cue") and isinstance(node[kk], str)]:
150
+ t = cue_time(node[k], words)
151
+ if t is not None:
152
+ node.setdefault(k[:-4] + "_at", max(0.0, round(t - lead, 2)))
153
+ if isinstance(node.get("cue"), str) and "at" not in node:
154
+ t = cue_time(node["cue"], words)
155
+ if t is not None:
156
+ node["at"] = max(0.0, round(t - lead, 2))
157
+ for v in node.values():
158
+ apply_cues(v, words, lead)
159
+ elif isinstance(node, list):
160
+ for v in node:
161
+ apply_cues(v, words, lead)
162
+
163
+
164
+ def main() -> None:
165
+ lesson_dir = Path(sys.argv[1]).resolve() if len(sys.argv) > 1 else Path("lessons/ep00-pilot").resolve()
166
+ script = json.loads((lesson_dir / "script.json").read_text(encoding="utf-8"))
167
+ audio_dir = lesson_dir / "audio"
168
+ audio_dir.mkdir(exist_ok=True)
169
+
170
+ voice, rate, gap = script["voice"], script["rate"], script.get("gap", 0.35)
171
+ scenes, subs, cursor = [], [], 0.0
172
+
173
+ for i, sc in enumerate(script["scenes"]):
174
+ mp3 = audio_dir / f"{sc['id']}.mp3"
175
+ wav = audio_dir / f"{sc['id']}.wav"
176
+ words = synth_retry(sc["narration"], voice, rate, mp3)
177
+ to_wav(mp3, wav)
178
+ dur = wav_duration(wav)
179
+
180
+ scene_subs = build_subtitles(sc["narration"], words, cursor)
181
+ entry = {k: v for k, v in sc.items() if k != "narration"}
182
+ apply_cues(entry, words)
183
+ entry.update({"start": round(cursor, 3), "dur": round(dur + gap, 3), "audio": round(dur, 3)})
184
+ scenes.append(entry)
185
+ subs.extend(scene_subs)
186
+ print(f" {sc['id']:>4} {dur:5.2f}s {sc['narration']}")
187
+ cursor += dur + gap
188
+
189
+ duration = round(cursor, 3)
190
+ # meta 里除 scenes 之外的顶层标量原样传下去:deck.html 用 meta.wm / meta.mark 渲染
191
+ # 右上角水印、台标字母,脚本里写什么就显示什么,模板本身不含任何固定文案。
192
+ meta = {"series": script["series"], "episode": script["episode"], "id": script["id"],
193
+ "brand": script.get("brand", script["series"]), "voice": voice, "rate": rate}
194
+ meta.update({k: v for k, v in script.items() if isinstance(v, (str, int, float, bool))})
195
+ lesson = {
196
+ "meta": meta,
197
+ "fps": 30,
198
+ "duration": duration,
199
+ "scenes": scenes,
200
+ "subtitles": [{"text": s["text"], "start": round(s["start"], 3), "end": round(s["end"], 3)} for s in subs],
201
+ }
202
+ (lesson_dir / "lesson.json").write_text(json.dumps(lesson, ensure_ascii=False, indent=2), encoding="utf-8")
203
+
204
+ # 配音轨:每幕之间补上 gap 长度的静音,最后也补一段
205
+ concat = audio_dir / "concat.txt"
206
+ lines = []
207
+ for i, sc in enumerate(scenes):
208
+ lines.append(f"file '{(audio_dir / (sc['id'] + '.wav')).as_posix()}'")
209
+ lines.append(f"file '{(audio_dir / 'gap.wav').as_posix()}'")
210
+ with wave.open(str(audio_dir / "gap.wav"), "wb") as w:
211
+ w.setnchannels(1); w.setsampwidth(2); w.setframerate(48000)
212
+ w.writeframes(b"\x00" * int(48000 * 2 * gap))
213
+ concat.write_text("\n".join(lines) + "\n", encoding="utf-8")
214
+ subprocess.run([FFMPEG, "-y", "-hide_banner", "-loglevel", "error",
215
+ "-f", "concat", "-safe", "0", "-i", str(concat),
216
+ str(audio_dir / "narration.wav")], check=True)
217
+
218
+ print(f"OK 时长 {duration:.2f}s -> {lesson_dir / 'lesson.json'}")
219
+ print(f" 配音 -> {audio_dir / 'narration.wav'} ({wav_duration(audio_dir / 'narration.wav'):.2f}s)")
220
+
221
+
222
+ if __name__ == "__main__":
223
+ main()