dsh-convert-core 0.1.0-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (115) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +149 -0
  3. package/bin/media-to-md.mjs +168 -0
  4. package/bin/tita-convert.mjs +85 -0
  5. package/docs/media-to-md.md +143 -0
  6. package/lib/adapters/anydoc.d.ts +73 -0
  7. package/lib/adapters/anydoc.d.ts.map +1 -0
  8. package/lib/adapters/anydoc.js +295 -0
  9. package/lib/adapters/anydoc.js.map +1 -0
  10. package/lib/adapters/iddoc.d.ts +28 -0
  11. package/lib/adapters/iddoc.d.ts.map +1 -0
  12. package/lib/adapters/iddoc.js +153 -0
  13. package/lib/adapters/iddoc.js.map +1 -0
  14. package/lib/adapters/index.d.ts +41 -0
  15. package/lib/adapters/index.d.ts.map +1 -0
  16. package/lib/adapters/index.js +115 -0
  17. package/lib/adapters/index.js.map +1 -0
  18. package/lib/adapters/scanned.d.ts +28 -0
  19. package/lib/adapters/scanned.d.ts.map +1 -0
  20. package/lib/adapters/scanned.js +121 -0
  21. package/lib/adapters/scanned.js.map +1 -0
  22. package/lib/adapters/text.d.ts +29 -0
  23. package/lib/adapters/text.d.ts.map +1 -0
  24. package/lib/adapters/text.js +135 -0
  25. package/lib/adapters/text.js.map +1 -0
  26. package/lib/adapters/types.d.ts +65 -0
  27. package/lib/adapters/types.d.ts.map +1 -0
  28. package/lib/adapters/types.js +11 -0
  29. package/lib/adapters/types.js.map +1 -0
  30. package/lib/contract.d.ts +185 -0
  31. package/lib/contract.d.ts.map +1 -0
  32. package/lib/contract.js +59 -0
  33. package/lib/contract.js.map +1 -0
  34. package/lib/delivery.d.ts +36 -0
  35. package/lib/delivery.d.ts.map +1 -0
  36. package/lib/delivery.js +97 -0
  37. package/lib/delivery.js.map +1 -0
  38. package/lib/http.d.ts +34 -0
  39. package/lib/http.d.ts.map +1 -0
  40. package/lib/http.js +372 -0
  41. package/lib/http.js.map +1 -0
  42. package/lib/markdown.d.ts +32 -0
  43. package/lib/markdown.d.ts.map +1 -0
  44. package/lib/markdown.js +145 -0
  45. package/lib/markdown.js.map +1 -0
  46. package/lib/media/binaries.d.ts +26 -0
  47. package/lib/media/binaries.d.ts.map +1 -0
  48. package/lib/media/binaries.js +60 -0
  49. package/lib/media/binaries.js.map +1 -0
  50. package/lib/media/convert.d.ts +19 -0
  51. package/lib/media/convert.d.ts.map +1 -0
  52. package/lib/media/convert.js +210 -0
  53. package/lib/media/convert.js.map +1 -0
  54. package/lib/media/index.d.ts +27 -0
  55. package/lib/media/index.d.ts.map +1 -0
  56. package/lib/media/index.js +27 -0
  57. package/lib/media/index.js.map +1 -0
  58. package/lib/media/markdown.d.ts +40 -0
  59. package/lib/media/markdown.d.ts.map +1 -0
  60. package/lib/media/markdown.js +89 -0
  61. package/lib/media/markdown.js.map +1 -0
  62. package/lib/media/media.d.ts +45 -0
  63. package/lib/media/media.d.ts.map +1 -0
  64. package/lib/media/media.js +167 -0
  65. package/lib/media/media.js.map +1 -0
  66. package/lib/media/models.d.ts +31 -0
  67. package/lib/media/models.d.ts.map +1 -0
  68. package/lib/media/models.js +87 -0
  69. package/lib/media/models.js.map +1 -0
  70. package/lib/media/scanned.d.ts +18 -0
  71. package/lib/media/scanned.d.ts.map +1 -0
  72. package/lib/media/scanned.js +179 -0
  73. package/lib/media/scanned.js.map +1 -0
  74. package/lib/media/types.d.ts +154 -0
  75. package/lib/media/types.d.ts.map +1 -0
  76. package/lib/media/types.js +23 -0
  77. package/lib/media/types.js.map +1 -0
  78. package/lib/server.d.ts +35 -0
  79. package/lib/server.d.ts.map +1 -0
  80. package/lib/server.js +64 -0
  81. package/lib/server.js.map +1 -0
  82. package/lib/service.d.ts +170 -0
  83. package/lib/service.d.ts.map +1 -0
  84. package/lib/service.js +562 -0
  85. package/lib/service.js.map +1 -0
  86. package/package.json +59 -0
  87. package/scripts/asr.py +55 -0
  88. package/scripts/ocr.py +40 -0
  89. package/scripts/pdf_pages.py +49 -0
  90. package/scripts/vendor.mjs +76 -0
  91. package/src/adapters/anydoc.ts +356 -0
  92. package/src/adapters/iddoc.ts +171 -0
  93. package/src/adapters/index.ts +147 -0
  94. package/src/adapters/scanned.ts +139 -0
  95. package/src/adapters/text.ts +146 -0
  96. package/src/adapters/types.ts +76 -0
  97. package/src/contract.ts +230 -0
  98. package/src/delivery.ts +124 -0
  99. package/src/http.ts +394 -0
  100. package/src/markdown.ts +141 -0
  101. package/src/media/binaries.ts +67 -0
  102. package/src/media/convert.ts +232 -0
  103. package/src/media/index.ts +43 -0
  104. package/src/media/markdown.ts +105 -0
  105. package/src/media/media.ts +204 -0
  106. package/src/media/models.ts +124 -0
  107. package/src/media/scanned.ts +200 -0
  108. package/src/media/types.ts +171 -0
  109. package/src/server.ts +85 -0
  110. package/src/service.ts +639 -0
  111. package/web/app.js +382 -0
  112. package/web/index.html +217 -0
  113. package/web/styles.css +263 -0
  114. package/web/vendor/icons.js +25 -0
  115. package/web/vendor/vue.global.prod.js +14 -0
@@ -0,0 +1,232 @@
1
+ /**
2
+ * 主编排:一个本地文件 → 知识就绪 Markdown + 本地图片。
3
+ *
4
+ * 档位规则(刻意保持简单可预测):
5
+ *
6
+ * | mode | 字幕 | 转写 | 说明 |
7
+ * | --------------- | ---- | ---- | ---- |
8
+ * | `auto`(默认) | 优先 | 仅当允许下载模型 | 有字幕就绝不碰模型 |
9
+ * | `subtitle-only` | 只用 | 永不 | 拿不到字幕就是"没有逐字稿",不报错 |
10
+ * | `transcribe` | 跳过 | 强制 | 必须允许下载模型 |
11
+ * | `frames-only` | 跳过 | 跳过 | 只要关键帧(连 OCR 也要显式开) |
12
+ */
13
+
14
+ import { createHash } from 'node:crypto'
15
+ import { createReadStream } from 'node:fs'
16
+ import { mkdtemp, readFile, rm, stat } from 'node:fs/promises'
17
+ import { basename, extname, join } from 'node:path'
18
+ import { tmpdir } from 'node:os'
19
+ import { probeTools, runFfmpeg, FFMPEG_HINT } from './binaries.js'
20
+ import { buildMarkdown } from './markdown.js'
21
+ import { extractKeyframes, extractSubtitles, parseSrt, probeMedia } from './media.js'
22
+ import { describeModel, ocrImages, transcribe } from './models.js'
23
+ import {
24
+ MediaToMarkdownError,
25
+ type MediaMeta,
26
+ type MediaMode,
27
+ type MediaSourceFormat,
28
+ type MediaToMarkdownOptions,
29
+ type MediaToMarkdownResult,
30
+ type MediaWarning,
31
+ } from './types.js'
32
+
33
+ export const VIDEO_EXTENSIONS = [
34
+ '.mp4', '.m4v', '.mov', '.mkv', '.webm', '.avi', '.flv', '.wmv', '.mpg', '.mpeg', '.ogv',
35
+ ]
36
+ export const AUDIO_EXTENSIONS = [
37
+ '.mp3', '.m4a', '.wav', '.aac', '.flac', '.ogg', '.opus', '.aiff', '.aif', '.amr', '.wma',
38
+ ]
39
+ export const IMAGE_EXTENSIONS = [
40
+ '.png', '.jpg', '.jpeg', '.webp', '.gif', '.bmp', '.tif', '.tiff', '.heic',
41
+ ]
42
+ export const MEDIA_EXTENSIONS = [...VIDEO_EXTENSIONS, ...AUDIO_EXTENSIONS, ...IMAGE_EXTENSIONS]
43
+
44
+ const IMAGE_MAGIC: Array<{ test: (data: Buffer) => boolean; extension: string; contentType: string }> = [
45
+ { test: data => data.subarray(0, 8).equals(Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])), extension: 'png', contentType: 'image/png' },
46
+ { test: data => data[0] === 0xff && data[1] === 0xd8, extension: 'jpg', contentType: 'image/jpeg' },
47
+ { test: data => data.subarray(0, 3).toString('latin1') === 'GIF', extension: 'gif', contentType: 'image/gif' },
48
+ { test: data => data.subarray(0, 4).toString('latin1') === 'RIFF' && data.subarray(8, 12).toString('latin1') === 'WEBP', extension: 'webp', contentType: 'image/webp' },
49
+ { test: data => data.subarray(0, 2).toString('latin1') === 'BM', extension: 'bmp', contentType: 'image/bmp' },
50
+ ]
51
+
52
+ async function sha256OfFile(file: string): Promise<string> {
53
+ const hash = createHash('sha256')
54
+ await new Promise<void>((resolve, reject) => {
55
+ createReadStream(file)
56
+ .on('data', chunk => hash.update(chunk))
57
+ .on('error', reject)
58
+ .on('end', () => resolve())
59
+ })
60
+ return `sha256:${hash.digest('hex')}`
61
+ }
62
+
63
+ function progressReporter(options: MediaToMarkdownOptions) {
64
+ return (stage: Parameters<NonNullable<MediaToMarkdownOptions['onProgress']>>[0]['stage'], progress: number | null, message?: string): void => {
65
+ options.onProgress?.({ stage, progress, ...(message ? { message } : {}) })
66
+ }
67
+ }
68
+
69
+ export async function convertMedia(options: MediaToMarkdownOptions): Promise<MediaToMarkdownResult> {
70
+ const file = options.file
71
+ const fileInfo = await stat(file).catch(() => undefined)
72
+ if (!fileInfo?.isFile()) {
73
+ throw new MediaToMarkdownError('INPUT_NOT_A_FILE', `输入不是文件:${file}`)
74
+ }
75
+ const versions = await probeTools({ ...(options.tools?.uvxPath ? { uvxPath: options.tools.uvxPath } : {}) })
76
+ if (!versions.ffmpeg || !versions.ffprobe) {
77
+ const missing = [!versions.ffprobe ? 'ffprobe' : null, !versions.ffmpeg ? 'ffmpeg' : null].filter(Boolean).join('、')
78
+ throw new MediaToMarkdownError('DEPENDENCY_MISSING', `缺少外部二进制:${missing}`, FFMPEG_HINT)
79
+ }
80
+
81
+ const extension = extname(file).toLowerCase()
82
+ const title = options.title?.trim() || basename(file, extname(file))
83
+ const mode: MediaMode = options.mode ?? 'auto'
84
+ const report = progressReporter(options)
85
+ const warnings: MediaWarning[] = []
86
+ const scratch = await mkdtemp(join(tmpdir(), 'media-to-md-'))
87
+ const dependencyVersions: Record<string, string> = { ffmpeg: versions.ffmpeg }
88
+ if (versions.uvx) dependencyVersions.uvx = versions.uvx
89
+
90
+ const externalId = await sha256OfFile(file)
91
+
92
+ try {
93
+ /* ------------------------------ 图片档 ------------------------------ */
94
+ if (IMAGE_EXTENSIONS.includes(extension)) {
95
+ report('probing', null, '读取图片')
96
+ const bytes = await readFile(file)
97
+ const magic = IMAGE_MAGIC.find(entry => entry.test(bytes))
98
+ const assetExtension = magic?.extension ?? (extension.replace('.', '') || 'bin')
99
+ const rel = `assets/image-${createHash('sha1').update(bytes).digest('hex').slice(0, 12)}.${assetExtension}`
100
+ let ocrText = ''
101
+ if (options.imageOcr) {
102
+ report('ocr', null, '本地 OCR')
103
+ const ocr = await ocrImages([file], options.tools, options.signal)
104
+ warnings.push(...ocr.warnings)
105
+ ocrText = (ocr.entries[0]?.text ?? '').trim()
106
+ warnings.push({ code: 'IMAGE_OCR_ENABLED', message: '已显式开启图片 OCR(imageOcr=true)。' })
107
+ } else {
108
+ warnings.push({ code: 'IMAGE_OCR_SKIPPED', message: '未做图片 OCR:图片直接收录、靠多模态召回;如需文字请设 imageOcr=true。' })
109
+ }
110
+ const meta: MediaMeta = {
111
+ sourceFormat: 'image',
112
+ containerExtension: assetExtension,
113
+ durationSeconds: 0,
114
+ sizeBytes: fileInfo.size,
115
+ hasVideo: false,
116
+ hasAudio: false,
117
+ hasSubtitleTrack: false,
118
+ transcriptSource: 'none',
119
+ }
120
+ return {
121
+ title: title.slice(0, 300),
122
+ markdown: buildMarkdown({ title, meta, frames: [], cues: [], image: { rel, ocrText } }),
123
+ assets: [{ rel, data: bytes, ...(magic ? { contentType: magic.contentType } : {}) }],
124
+ meta,
125
+ transcript: { source: 'none', cues: [] },
126
+ warnings,
127
+ dependencyVersions,
128
+ externalId,
129
+ }
130
+ }
131
+
132
+ /* ------------------------------ 音视频档 ------------------------------ */
133
+ report('probing', 0.05, '探测媒体信息')
134
+ const probe = await probeMedia(file, options.signal)
135
+ const sourceFormat: MediaSourceFormat = probe.hasVideo ? 'video' : 'audio'
136
+
137
+ let frames: Awaited<ReturnType<typeof extractKeyframes>> = []
138
+ if (probe.hasVideo) {
139
+ report('frames', 0.2, mode === 'frames-only' ? '抽取关键帧(只要帧)' : '抽取关键帧')
140
+ frames = await extractKeyframes(file, probe.durationSeconds, scratch, options.frames ?? {}, options.signal)
141
+ }
142
+ if (frames.length > 0) {
143
+ warnings.push({ code: 'KEYFRAMES_EXTRACTED', message: `已抽出 ${frames.length} 个关键帧并落地为 assets/。` })
144
+ }
145
+
146
+ let frameOcr = new Map<string, string>()
147
+ if (options.ocrFrames && frames.length > 0) {
148
+ report('ocr', 0.35, '关键帧画面文字 OCR')
149
+ // OCR 读的是落盘的真实路径(scratch 下),结果再按顺序映射回 rel。
150
+ const ocr = await ocrImages(frames.map(frame => frame.sourcePath), options.tools, options.signal)
151
+ warnings.push(...ocr.warnings)
152
+ const bySourcePath = new Map(ocr.entries.map(entry => [entry.path, entry.text.trim()]))
153
+ frameOcr = new Map(
154
+ frames
155
+ .map(frame => [frame.rel, bySourcePath.get(frame.sourcePath) ?? ''] as const)
156
+ .filter(([, text]) => text.length > 0),
157
+ )
158
+ }
159
+
160
+ let cues: MediaToMarkdownResult['transcript']['cues'] = []
161
+ let transcriptSource: MediaToMarkdownResult['transcript']['source'] = 'none'
162
+
163
+ if (mode !== 'frames-only') {
164
+ report('subtitles', 0.45, '查找字幕轨')
165
+ const subtitleText = await extractSubtitles(file, probe, scratch, options.signal)
166
+ if (subtitleText) {
167
+ cues = parseSrt(subtitleText)
168
+ transcriptSource = 'subtitle'
169
+ warnings.push({ code: 'SUBTITLE_USED', message: `使用字幕轨/旁挂字幕(${cues.length} 条),未下载任何转写模型。` })
170
+ } else if (!probe.hasAudio) {
171
+ warnings.push({ code: 'NO_AUDIO_TRACK', message: '该文件没有音频轨,也没有字幕。' })
172
+ } else if (mode === 'subtitle-only') {
173
+ warnings.push({ code: 'SUBTITLE_MISSING', message: '没有可用字幕;subtitle-only 档不做转写。' })
174
+ } else {
175
+ const asr = options.asr ?? {}
176
+ const model = asr.model?.trim() || 'small'
177
+ const mayDownload = asr.allowModelDownload === true || isLocalModelPath(model)
178
+ if (mode === 'auto' && !mayDownload) {
179
+ throw new MediaToMarkdownError(
180
+ 'ASR_MODEL_REQUIRED',
181
+ `无字幕,转写需要下载本地 whisper 模型(${model},${describeModel(model)});当前未允许下载,已放弃转换。`,
182
+ '确认后设 allowModelDownload=true 重试;中文建议 model=large-v3-turbo、language=zh。也可以先用 ffmpeg 抽字幕轨,或改用带字幕的来源。',
183
+ )
184
+ }
185
+ report('transcribing', 0.6, '本地转写(首次会下载模型)')
186
+ const audio = join(scratch, 'audio-16k.wav')
187
+ await runFfmpeg(['-hide_banner', '-loglevel', 'error', '-y', '-i', file, '-vn', '-ac', '1', '-ar', '16000', '-c:a', 'pcm_s16le', audio])
188
+ const result = await transcribe(audio, scratch, asr, options.tools, options.signal)
189
+ cues = result.cues
190
+ transcriptSource = cues.length > 0 ? 'transcription' : 'none'
191
+ warnings.push({
192
+ code: 'ASR_LOCAL',
193
+ message: `无字幕,已用本地 faster-whisper(${result.model},语言 ${result.language ?? 'auto'})转写;音频未离开本机。`,
194
+ })
195
+ }
196
+ }
197
+
198
+ report('assembling', 0.9, '装配 Markdown')
199
+ const meta = {
200
+ sourceFormat,
201
+ containerExtension: extension.replace('.', ''),
202
+ durationSeconds: probe.durationSeconds,
203
+ sizeBytes: probe.sizeBytes,
204
+ ...(probe.width ? { width: probe.width } : {}),
205
+ ...(probe.height ? { height: probe.height } : {}),
206
+ ...(probe.fps ? { fps: probe.fps } : {}),
207
+ ...(probe.videoCodec ? { videoCodec: probe.videoCodec } : {}),
208
+ ...(probe.audioCodec ? { audioCodec: probe.audioCodec } : {}),
209
+ hasVideo: probe.hasVideo,
210
+ hasAudio: probe.hasAudio,
211
+ hasSubtitleTrack: probe.hasSubtitleTrack,
212
+ transcriptSource,
213
+ }
214
+ return {
215
+ title: title.slice(0, 300),
216
+ markdown: buildMarkdown({ title, meta, frames, cues, frameOcr }),
217
+ assets: frames.map(frame => ({ rel: frame.rel, data: frame.data, contentType: 'image/jpeg' })),
218
+ meta,
219
+ transcript: { source: transcriptSource, cues },
220
+ warnings,
221
+ dependencyVersions,
222
+ externalId,
223
+ }
224
+ } finally {
225
+ if (!options.keepScratch) await rm(scratch, { recursive: true, force: true }).catch(() => undefined)
226
+ }
227
+ }
228
+
229
+ /** 模型名看着像路径(本地目录)时不视为需要下载。 */
230
+ function isLocalModelPath(model: string): boolean {
231
+ return model.includes('/') || model.startsWith('.') || model.startsWith('~')
232
+ }
@@ -0,0 +1,43 @@
1
+ /**
2
+ * media-to-md —— 本地音视频 / 图片 → Markdown + 本地图片。
3
+ *
4
+ * 纯本地:不调用云端 API、不调用大模型。两个专用小模型(whisper 语音识别、RapidOCR
5
+ * 画面文字)都是可选档位,是否允许下载由调用方显式决定。
6
+ *
7
+ * ```ts
8
+ * import { convertMedia } from 'dsh-convert-core/media'
9
+ *
10
+ * const result = await convertMedia({
11
+ * file: '/abs/path/clip.mp4',
12
+ * mode: 'auto',
13
+ * asr: { model: 'large-v3-turbo', language: 'zh', allowModelDownload: true },
14
+ * frames: { samples: 12, maxFrames: 40, sceneThreshold: 0.3 },
15
+ * onProgress: p => console.log(p.stage, p.progress),
16
+ * })
17
+ * // result.markdown / result.assets / result.warnings / result.meta
18
+ * ```
19
+ */
20
+
21
+ export { convertMedia, MEDIA_EXTENSIONS, VIDEO_EXTENSIONS, AUDIO_EXTENSIONS, IMAGE_EXTENSIONS } from './convert.js'
22
+ export { convertScannedDocument } from './scanned.js'
23
+ export { probeTools, FFMPEG_HINT, UVX_HINT, type ToolVersions } from './binaries.js'
24
+ export { probeMedia, extractKeyframes, extractSubtitles, parseSrt, type MediaProbe, type ExtractedFrame } from './media.js'
25
+ export { transcribe, ocrImages, ASR_MODEL_SIZES, describeModel, type TranscriptionResult, type OcrEntry } from './models.js'
26
+ export { buildMarkdown, buildScannedMarkdown, formatClock, type BuildMarkdownInput, type BuildScannedMarkdownInput } from './markdown.js'
27
+ export {
28
+ MediaToMarkdownError,
29
+ type MediaAsset,
30
+ type MediaAsrOptions,
31
+ type MediaCue,
32
+ type MediaFrameOptions,
33
+ type MediaMeta,
34
+ type MediaMode,
35
+ type MediaSourceFormat,
36
+ type MediaToMarkdownOptions,
37
+ type MediaToMarkdownProgress,
38
+ type MediaToMarkdownResult,
39
+ type MediaToolOptions,
40
+ type MediaWarning,
41
+ type ScannedPage,
42
+ type ScannedPdfOptions,
43
+ } from './types.js'
@@ -0,0 +1,105 @@
1
+ /**
2
+ * Markdown 装配 —— 纯模板,零模型、零网络。
3
+ *
4
+ * 产物只返回**正文**:front matter 是宿主的契约(转换内核自己写,CLI 自己写),
5
+ * 这个包不该替宿主决定 front matter 字段。
6
+ */
7
+
8
+ import type { ExtractedFrame } from './media.js'
9
+ import type { MediaCue, MediaMeta } from './types.js'
10
+
11
+ export function formatClock(seconds: number): string {
12
+ const total = Math.max(0, Math.floor(seconds))
13
+ const hours = Math.floor(total / 3600)
14
+ const minutes = Math.floor((total % 3600) / 60)
15
+ const secs = total % 60
16
+ const mm = String(minutes).padStart(2, '0')
17
+ const ss = String(secs).padStart(2, '0')
18
+ return hours > 0 ? `${String(hours).padStart(2, '0')}:${mm}:${ss}` : `${mm}:${ss}`
19
+ }
20
+
21
+ export interface BuildMarkdownInput {
22
+ title: string
23
+ meta: MediaMeta
24
+ frames: ExtractedFrame[]
25
+ cues: MediaCue[]
26
+ /** 关键帧 rel → 画面文字(仅在开启帧 OCR 时非空)。 */
27
+ frameOcr?: Map<string, string>
28
+ /** 图片输入的资产与文字。 */
29
+ image?: { rel: string; ocrText: string }
30
+ }
31
+
32
+ export function buildMarkdown(input: BuildMarkdownInput): string {
33
+ const { title, meta, frames, cues, image } = input
34
+ const frameOcr = input.frameOcr ?? new Map<string, string>()
35
+ const lines: string[] = [`# ${title}`, '']
36
+
37
+ const facts = [
38
+ meta.durationSeconds > 0 ? `时长 ${formatClock(meta.durationSeconds)}` : null,
39
+ meta.width && meta.height ? `${meta.width}×${meta.height}` : null,
40
+ meta.videoCodec ?? null,
41
+ meta.audioCodec ? `音频 ${meta.audioCodec}` : null,
42
+ ].filter((value): value is string => value !== null)
43
+ if (facts.length > 0) lines.push(`> ${facts.join(' · ')}`, '')
44
+
45
+ if (image) {
46
+ lines.push(`![${title}](${image.rel})`, '')
47
+ if (image.ocrText) lines.push('## 画面文字', '', image.ocrText, '')
48
+ else lines.push('_图片已原样收录(未做 OCR);以多模态方式参与召回。_', '')
49
+ }
50
+
51
+ if (frames.length > 0) {
52
+ lines.push('## 时间轴', '')
53
+ for (const [index, frame] of frames.entries()) {
54
+ const start = frame.seconds
55
+ const end = index + 1 < frames.length ? frames[index + 1]!.seconds : Number.POSITIVE_INFINITY
56
+ lines.push(`### ${formatClock(start)}`, '', `![${formatClock(start)}](${frame.rel})`, '')
57
+ const ocr = frameOcr.get(frame.rel)
58
+ if (ocr) lines.push('**画面文字**', '', ocr, '')
59
+ const inWindow = cues.filter(cue => cue.seconds >= start - 0.01 && cue.seconds < end)
60
+ for (const cue of inWindow) lines.push(`- \`${formatClock(cue.seconds)}\` ${cue.text}`)
61
+ if (inWindow.length > 0) lines.push('')
62
+ }
63
+ }
64
+
65
+ if (!image) {
66
+ lines.push('## 逐字稿', '')
67
+ if (cues.length === 0) {
68
+ lines.push(meta.transcriptSource === 'none' ? '_没有字幕轨,也未开启本地转写。_' : '_未识别到可用语音内容。_', '')
69
+ } else {
70
+ for (const cue of cues) lines.push(`- \`${formatClock(cue.seconds)}\` ${cue.text}`)
71
+ lines.push('')
72
+ }
73
+ }
74
+ return lines.join('\n')
75
+ }
76
+
77
+ export interface BuildScannedMarkdownInput {
78
+ title: string
79
+ totalPages: number
80
+ pageImages: boolean
81
+ pages: Array<{ index: number; rel: string; text: string }>
82
+ }
83
+
84
+ /**
85
+ * 扫描档的 Markdown:逐页一节,页内先图后文。
86
+ *
87
+ * 不承诺版面顺序与表格结构 —— 这一档只负责"把图片页里的字读出来"。
88
+ */
89
+ export function buildScannedMarkdown(input: BuildScannedMarkdownInput): string {
90
+ const { title, totalPages, pageImages, pages } = input
91
+ const lines: string[] = [`# ${title}`, '']
92
+ const facts = [`共 ${totalPages} 页`, `已识别 ${pages.length} 页`, '本地 OCR(轻量档,无版面还原)']
93
+ lines.push(`> ${facts.join(' · ')}`, '')
94
+ if (pages.length === 0) {
95
+ lines.push('_没有可识别的页面。_', '')
96
+ return lines.join('\n')
97
+ }
98
+ for (const page of pages) {
99
+ lines.push(`## 第 ${page.index} 页`, '')
100
+ if (pageImages) lines.push(`![第 ${page.index} 页](${page.rel})`, '')
101
+ if (page.text.trim()) lines.push(page.text.trim(), '')
102
+ else lines.push('_本页未识别到文字。_', '')
103
+ }
104
+ return lines.join('\n')
105
+ }
@@ -0,0 +1,204 @@
1
+ /**
2
+ * 媒体探测、关键帧与字幕 —— 全部走 ffmpeg/ffprobe,不含任何模型。
3
+ */
4
+
5
+ import { execFile } from 'node:child_process'
6
+ import { mkdir, readdir, readFile } from 'node:fs/promises'
7
+ import { basename, dirname, extname, join } from 'node:path'
8
+ import { promisify } from 'node:util'
9
+ import { MediaToMarkdownError, type MediaCue, type MediaFrameOptions } from './types.js'
10
+ import { runFfmpeg } from './binaries.js'
11
+
12
+ const execFileAsync = promisify(execFile)
13
+
14
+ export interface MediaStream {
15
+ codec_type?: string
16
+ codec_name?: string
17
+ width?: number
18
+ height?: number
19
+ r_frame_rate?: string
20
+ }
21
+
22
+ export interface MediaProbe {
23
+ durationSeconds: number
24
+ sizeBytes: number
25
+ hasVideo: boolean
26
+ hasAudio: boolean
27
+ hasSubtitleTrack: boolean
28
+ width?: number
29
+ height?: number
30
+ fps?: number
31
+ videoCodec?: string
32
+ audioCodec?: string
33
+ }
34
+
35
+ export async function probeMedia(file: string, signal?: AbortSignal): Promise<MediaProbe> {
36
+ let stdout: string
37
+ try {
38
+ ({ stdout } = await execFileAsync('ffprobe', [
39
+ '-v', 'error', '-print_format', 'json', '-show_format', '-show_streams', file,
40
+ ], { maxBuffer: 32 * 1024 * 1024, ...(signal ? { signal } : {}) }))
41
+ } catch (cause) {
42
+ throw new MediaToMarkdownError(
43
+ 'MEDIA_UNREADABLE',
44
+ `ffprobe 无法读取该文件:${cause instanceof Error ? cause.message : String(cause)}`,
45
+ )
46
+ }
47
+ const parsed = JSON.parse(stdout) as { format?: { duration?: string; size?: string }; streams?: MediaStream[] }
48
+ const streams = parsed.streams ?? []
49
+ const video = streams.find(stream => stream.codec_type === 'video')
50
+ const audio = streams.find(stream => stream.codec_type === 'audio')
51
+ const subtitle = streams.find(stream => stream.codec_type === 'subtitle')
52
+ const rate = video?.r_frame_rate?.split('/')
53
+ const fps = rate && Number(rate[1]) > 0 ? Number(rate[0]) / Number(rate[1]) : undefined
54
+ return {
55
+ durationSeconds: Number(parsed.format?.duration ?? 0),
56
+ sizeBytes: Number(parsed.format?.size ?? 0),
57
+ hasVideo: video !== undefined,
58
+ hasAudio: audio !== undefined,
59
+ hasSubtitleTrack: subtitle !== undefined,
60
+ ...(video?.width ? { width: video.width } : {}),
61
+ ...(video?.height ? { height: video.height } : {}),
62
+ ...(fps !== undefined && Number.isFinite(fps) ? { fps: Number(fps.toFixed(2)) } : {}),
63
+ ...(video?.codec_name ? { videoCodec: video.codec_name } : {}),
64
+ ...(audio?.codec_name ? { audioCodec: audio.codec_name } : {}),
65
+ }
66
+ }
67
+
68
+ export interface ExtractedFrame {
69
+ /** 产物内相对路径(含 `assets/` 前缀),写进 Markdown 用。 */
70
+ rel: string
71
+ /** 落盘的真实路径(scratch 下),给 OCR 之类的后续步骤用。 */
72
+ sourcePath: string
73
+ data: Buffer
74
+ seconds: number
75
+ }
76
+
77
+ async function readdirSafe(dir: string): Promise<string[]> {
78
+ try {
79
+ return await readdir(dir)
80
+ } catch {
81
+ return []
82
+ }
83
+ }
84
+
85
+ /**
86
+ * 均匀采样 + 场景变化两路合并,再按内容 sha1 去重。
87
+ *
88
+ * 只靠 `scene` 检测会漏内容:口播/图文类视频几乎没有硬切,阈值再低也可能只出 1 帧
89
+ * (实测:一个 3 段画面的测试视频,阈值降到 0.1 仍然只有 1 帧)。只靠均匀采样又会漏掉
90
+ * 真正的画面切换。两路都跑,按时间排序、近邻合并、内容去重。
91
+ */
92
+ export async function extractKeyframes(
93
+ file: string,
94
+ durationSeconds: number,
95
+ scratch: string,
96
+ options: MediaFrameOptions = {},
97
+ signal?: AbortSignal,
98
+ ): Promise<ExtractedFrame[]> {
99
+ const samples = Math.max(1, options.samples ?? 12)
100
+ const maxFrames = Math.max(1, options.maxFrames ?? 40)
101
+ const scene = clamp01(options.sceneThreshold ?? 0.3)
102
+ const raw = join(scratch, 'raw')
103
+ await mkdir(join(raw, 'scene'), { recursive: true })
104
+
105
+ const count = Math.max(1, Math.min(samples, maxFrames))
106
+ const step = durationSeconds > 0 ? Math.max(durationSeconds / count, 0.5) : 1
107
+ await runFfmpeg([
108
+ '-hide_banner', '-loglevel', 'error', '-i', file,
109
+ '-vf', `fps=1/${step.toFixed(3)}`, '-frames:v', String(count), '-q:v', '3',
110
+ join(raw, 'even-%04d.jpg'),
111
+ ], { allowFailure: true })
112
+
113
+ const sceneRun = await runFfmpeg([
114
+ '-hide_banner', '-loglevel', 'info', '-i', file,
115
+ '-vf', `select='gt(scene\\,${scene})',showinfo`, '-fps_mode', 'vfr',
116
+ '-frames:v', String(maxFrames), '-q:v', '3',
117
+ join(raw, 'scene', 'scene-%04d.jpg'),
118
+ ], { allowFailure: true })
119
+ const sceneTimes = [...(sceneRun.stderr ?? '').matchAll(/pts_time:([0-9.]+)/g)].map(match => Number(match[1]))
120
+
121
+ const evenFrames = (await readdirSafe(raw)).filter(name => name.startsWith('even-')).sort()
122
+ .map((name, index) => ({ seconds: Number((index * step).toFixed(3)), path: join(raw, name) }))
123
+ const sceneFrames = (await readdirSafe(join(raw, 'scene'))).filter(name => name.startsWith('scene-')).sort()
124
+ .map((name, index) => ({
125
+ seconds: Number.isFinite(sceneTimes[index]) ? Number(sceneTimes[index]) : 0,
126
+ path: join(raw, 'scene', name),
127
+ }))
128
+
129
+ const picked: Array<{ seconds: number; path: string }> = []
130
+ for (const candidate of [...sceneFrames, ...evenFrames].sort((a, b) => a.seconds - b.seconds)) {
131
+ if (picked.length >= maxFrames) break
132
+ if (picked.some(kept => Math.abs(kept.seconds - candidate.seconds) < 0.75)) continue
133
+ picked.push(candidate)
134
+ }
135
+
136
+ const { createHash } = await import('node:crypto')
137
+ const frames: ExtractedFrame[] = []
138
+ const seen = new Set<string>()
139
+ for (const item of picked) {
140
+ if (signal?.aborted) throw new MediaToMarkdownError('ABORTED', '已取消。')
141
+ const data = await readFile(item.path)
142
+ // 静止画面会被采样多次,按内容去重,别让 assets/ 里塞满一模一样的帧。
143
+ const digest = createHash('sha1').update(data).digest('hex')
144
+ if (seen.has(digest)) continue
145
+ seen.add(digest)
146
+ frames.push({
147
+ rel: `assets/frame-${String(frames.length + 1).padStart(4, '0')}.jpg`,
148
+ sourcePath: item.path,
149
+ data,
150
+ seconds: item.seconds,
151
+ })
152
+ }
153
+ return frames
154
+ }
155
+
156
+ /** 内嵌字幕轨优先,没有再看同名旁挂 `.srt`。 */
157
+ export async function extractSubtitles(file: string, probe: MediaProbe, scratch: string, signal?: AbortSignal): Promise<string | undefined> {
158
+ if (probe.hasSubtitleTrack) {
159
+ const target = join(scratch, 'subs.srt')
160
+ try {
161
+ await execFileAsync('ffmpeg', ['-hide_banner', '-loglevel', 'error', '-y', '-i', file, '-map', '0:s:0', target], { signal })
162
+ const text = await readFile(target, 'utf8')
163
+ if (text.trim()) return text
164
+ } catch {
165
+ /* 抽轨失败就退到旁挂字幕 */
166
+ }
167
+ }
168
+ const sidecar = join(dirname(file), `${basename(file, extname(file))}.srt`)
169
+ try {
170
+ const text = await readFile(sidecar, 'utf8')
171
+ if (text.trim()) return text
172
+ } catch {
173
+ /* 没有旁挂字幕 */
174
+ }
175
+ return undefined
176
+ }
177
+
178
+ /** 极简 SRT 解析:取起止时间与文本,顺带去掉自动字幕常见的滚动重复行。 */
179
+ export function parseSrt(source: string): MediaCue[] {
180
+ const cues: MediaCue[] = []
181
+ const range = /(\d{1,2}):(\d{2}):(\d{2})[,.](\d{1,3})\s*-->\s*(\d{1,2}):(\d{2}):(\d{2})[,.](\d{1,3})/
182
+ const toSeconds = (h: string, m: string, s: string, ms: string): number =>
183
+ Number(h) * 3600 + Number(m) * 60 + Number(s) + Number(ms.padEnd(3, '0')) / 1000
184
+ for (const block of source.replace(/\r/g, '').split(/\n{2,}/)) {
185
+ const lines = block.split('\n').map(line => line.trim()).filter(Boolean)
186
+ if (lines.length === 0) continue
187
+ const timeLine = lines.find(line => line.includes('-->'))
188
+ if (!timeLine) continue
189
+ const match = range.exec(timeLine)
190
+ const text = lines.filter(line => line !== timeLine && !/^\d+$/.test(line)).join(' ').trim()
191
+ if (!match || !text) continue
192
+ const seconds = toSeconds(match[1]!, match[2]!, match[3]!, match[4]!)
193
+ const endSeconds = toSeconds(match[5]!, match[6]!, match[7]!, match[8]!)
194
+ const previous = cues[cues.length - 1]
195
+ if (previous && previous.text === text) continue
196
+ cues.push({ seconds, endSeconds, text })
197
+ }
198
+ return cues
199
+ }
200
+
201
+ function clamp01(value: number): number {
202
+ if (!Number.isFinite(value)) return 0.3
203
+ return Math.min(1, Math.max(0, value))
204
+ }