dsh-convert-core 0.1.0-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +149 -0
- package/bin/media-to-md.mjs +168 -0
- package/bin/tita-convert.mjs +85 -0
- package/docs/media-to-md.md +143 -0
- package/lib/adapters/anydoc.d.ts +73 -0
- package/lib/adapters/anydoc.d.ts.map +1 -0
- package/lib/adapters/anydoc.js +295 -0
- package/lib/adapters/anydoc.js.map +1 -0
- package/lib/adapters/iddoc.d.ts +28 -0
- package/lib/adapters/iddoc.d.ts.map +1 -0
- package/lib/adapters/iddoc.js +153 -0
- package/lib/adapters/iddoc.js.map +1 -0
- package/lib/adapters/index.d.ts +41 -0
- package/lib/adapters/index.d.ts.map +1 -0
- package/lib/adapters/index.js +115 -0
- package/lib/adapters/index.js.map +1 -0
- package/lib/adapters/scanned.d.ts +28 -0
- package/lib/adapters/scanned.d.ts.map +1 -0
- package/lib/adapters/scanned.js +121 -0
- package/lib/adapters/scanned.js.map +1 -0
- package/lib/adapters/text.d.ts +29 -0
- package/lib/adapters/text.d.ts.map +1 -0
- package/lib/adapters/text.js +135 -0
- package/lib/adapters/text.js.map +1 -0
- package/lib/adapters/types.d.ts +65 -0
- package/lib/adapters/types.d.ts.map +1 -0
- package/lib/adapters/types.js +11 -0
- package/lib/adapters/types.js.map +1 -0
- package/lib/contract.d.ts +185 -0
- package/lib/contract.d.ts.map +1 -0
- package/lib/contract.js +59 -0
- package/lib/contract.js.map +1 -0
- package/lib/delivery.d.ts +36 -0
- package/lib/delivery.d.ts.map +1 -0
- package/lib/delivery.js +97 -0
- package/lib/delivery.js.map +1 -0
- package/lib/http.d.ts +34 -0
- package/lib/http.d.ts.map +1 -0
- package/lib/http.js +372 -0
- package/lib/http.js.map +1 -0
- package/lib/markdown.d.ts +32 -0
- package/lib/markdown.d.ts.map +1 -0
- package/lib/markdown.js +145 -0
- package/lib/markdown.js.map +1 -0
- package/lib/media/binaries.d.ts +26 -0
- package/lib/media/binaries.d.ts.map +1 -0
- package/lib/media/binaries.js +60 -0
- package/lib/media/binaries.js.map +1 -0
- package/lib/media/convert.d.ts +19 -0
- package/lib/media/convert.d.ts.map +1 -0
- package/lib/media/convert.js +210 -0
- package/lib/media/convert.js.map +1 -0
- package/lib/media/index.d.ts +27 -0
- package/lib/media/index.d.ts.map +1 -0
- package/lib/media/index.js +27 -0
- package/lib/media/index.js.map +1 -0
- package/lib/media/markdown.d.ts +40 -0
- package/lib/media/markdown.d.ts.map +1 -0
- package/lib/media/markdown.js +89 -0
- package/lib/media/markdown.js.map +1 -0
- package/lib/media/media.d.ts +45 -0
- package/lib/media/media.d.ts.map +1 -0
- package/lib/media/media.js +167 -0
- package/lib/media/media.js.map +1 -0
- package/lib/media/models.d.ts +31 -0
- package/lib/media/models.d.ts.map +1 -0
- package/lib/media/models.js +87 -0
- package/lib/media/models.js.map +1 -0
- package/lib/media/scanned.d.ts +18 -0
- package/lib/media/scanned.d.ts.map +1 -0
- package/lib/media/scanned.js +179 -0
- package/lib/media/scanned.js.map +1 -0
- package/lib/media/types.d.ts +154 -0
- package/lib/media/types.d.ts.map +1 -0
- package/lib/media/types.js +23 -0
- package/lib/media/types.js.map +1 -0
- package/lib/server.d.ts +35 -0
- package/lib/server.d.ts.map +1 -0
- package/lib/server.js +64 -0
- package/lib/server.js.map +1 -0
- package/lib/service.d.ts +170 -0
- package/lib/service.d.ts.map +1 -0
- package/lib/service.js +562 -0
- package/lib/service.js.map +1 -0
- package/package.json +59 -0
- package/scripts/asr.py +55 -0
- package/scripts/ocr.py +40 -0
- package/scripts/pdf_pages.py +49 -0
- package/scripts/vendor.mjs +76 -0
- package/src/adapters/anydoc.ts +356 -0
- package/src/adapters/iddoc.ts +171 -0
- package/src/adapters/index.ts +147 -0
- package/src/adapters/scanned.ts +139 -0
- package/src/adapters/text.ts +146 -0
- package/src/adapters/types.ts +76 -0
- package/src/contract.ts +230 -0
- package/src/delivery.ts +124 -0
- package/src/http.ts +394 -0
- package/src/markdown.ts +141 -0
- package/src/media/binaries.ts +67 -0
- package/src/media/convert.ts +232 -0
- package/src/media/index.ts +43 -0
- package/src/media/markdown.ts +105 -0
- package/src/media/media.ts +204 -0
- package/src/media/models.ts +124 -0
- package/src/media/scanned.ts +200 -0
- package/src/media/types.ts +171 -0
- package/src/server.ts +85 -0
- package/src/service.ts +639 -0
- package/web/app.js +382 -0
- package/web/index.html +217 -0
- package/web/styles.css +263 -0
- package/web/vendor/icons.js +25 -0
- package/web/vendor/vue.global.prod.js +14 -0
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 两个**专用小模型**的调用点:语音识别(whisper)与画面文字 OCR。
|
|
3
|
+
*
|
|
4
|
+
* 都不是大模型,也都不联网推理:模型下载一次后完全离线。helper 脚本随包发布
|
|
5
|
+
* (`scripts/*.py`,由 `uvx --with …` 按需拉起),所以这个包不把 Python/模型塞进 npm tarball,
|
|
6
|
+
* 但会把「要不要下载模型」这个决定**显式**交回调用方。
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { execFile } from 'node:child_process'
|
|
10
|
+
import { fileURLToPath } from 'node:url'
|
|
11
|
+
import { promisify } from 'node:util'
|
|
12
|
+
import { MediaToMarkdownError, type MediaAsrOptions, type MediaCue, type MediaToolOptions } from './types.js'
|
|
13
|
+
import { UVX_HINT } from './binaries.js'
|
|
14
|
+
|
|
15
|
+
const execFileAsync = promisify(execFile)
|
|
16
|
+
|
|
17
|
+
/** helper 脚本路径:`lib/*.js` → `scripts/*.py`。 */
|
|
18
|
+
function helperPath(name: string): string {
|
|
19
|
+
return fileURLToPath(new URL(`../../scripts/${name}`, import.meta.url))
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export const ASR_MODEL_SIZES: Record<string, string> = {
|
|
23
|
+
tiny: '约 75 MB',
|
|
24
|
+
base: '约 145 MB',
|
|
25
|
+
small: '约 460 MB',
|
|
26
|
+
medium: '约 1.5 GB',
|
|
27
|
+
'large-v3': '约 3 GB',
|
|
28
|
+
'large-v3-turbo': '约 1.5 GB',
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function describeModel(model: string): string {
|
|
32
|
+
return ASR_MODEL_SIZES[model] ?? '未知大小'
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function uvxArgs(tools: MediaToolOptions | undefined, packages: string[], script: string, args: string[]): [string, string[]] {
|
|
36
|
+
const uvx = tools?.uvxPath?.trim() || process.env.MEDIA_TO_MD_UVX?.trim() || 'uvx'
|
|
37
|
+
const python = tools?.pythonVersion?.trim() || process.env.MEDIA_TO_MD_PYTHON?.trim() || '3.12'
|
|
38
|
+
return [uvx, ['--python', python, ...packages.flatMap(pkg => ['--with', pkg]), 'python', script, ...args]]
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export interface TranscriptionResult {
|
|
42
|
+
cues: MediaCue[]
|
|
43
|
+
model: string
|
|
44
|
+
language?: string
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** 音频 → 带时间戳的转写。`allowModelDownload` 为 false 且本地无模型时抛 `ASR_MODEL_REQUIRED`。 */
|
|
48
|
+
export async function transcribe(
|
|
49
|
+
audioFile: string,
|
|
50
|
+
scratch: string,
|
|
51
|
+
asr: MediaAsrOptions = {},
|
|
52
|
+
tools?: MediaToolOptions,
|
|
53
|
+
signal?: AbortSignal,
|
|
54
|
+
): Promise<TranscriptionResult> {
|
|
55
|
+
const model = asr.model?.trim() || 'small'
|
|
56
|
+
const [cmd, args] = uvxArgs(tools, ['faster-whisper'], helperPath('asr.py'), [audioFile, model, asr.language?.trim() ?? ''])
|
|
57
|
+
let stdout: string
|
|
58
|
+
try {
|
|
59
|
+
({ stdout } = await execFileAsync(cmd, args, {
|
|
60
|
+
maxBuffer: 64 * 1024 * 1024,
|
|
61
|
+
env: {
|
|
62
|
+
...process.env,
|
|
63
|
+
...(asr.device ? { MEDIA_TO_MD_ASR_DEVICE: asr.device } : {}),
|
|
64
|
+
...(asr.computeType ? { MEDIA_TO_MD_ASR_COMPUTE: asr.computeType } : {}),
|
|
65
|
+
...(asr.beamSize ? { MEDIA_TO_MD_ASR_BEAM: String(asr.beamSize) } : {}),
|
|
66
|
+
},
|
|
67
|
+
...(signal ? { signal } : {}),
|
|
68
|
+
}))
|
|
69
|
+
} catch (cause) {
|
|
70
|
+
const detail = cause instanceof Error ? cause.message : String(cause)
|
|
71
|
+
const looksLikeDownload = /LocalEntryNotFoundError|ConnectError|SSL|Connection|snapshot_download/i.test(detail)
|
|
72
|
+
if (looksLikeDownload && asr.allowModelDownload === false) {
|
|
73
|
+
throw new MediaToMarkdownError(
|
|
74
|
+
'ASR_MODEL_REQUIRED',
|
|
75
|
+
`无字幕,转写需要下载本地 whisper 模型(${model},${describeModel(model)});当前未允许下载,已放弃转换。`,
|
|
76
|
+
`先取得模型再重试:设 allowModelDownload=true,或把模型放到本地目录后用 asr.model 指向它。中文建议 large-v3-turbo。`,
|
|
77
|
+
)
|
|
78
|
+
}
|
|
79
|
+
throw new MediaToMarkdownError(
|
|
80
|
+
'ASR_FAILED',
|
|
81
|
+
`本地转写失败:${detail}`,
|
|
82
|
+
'检查网络(首次需下载模型,国内可设 HF_ENDPOINT=https://hf-mirror.com)或预先下载模型目录并用 asr.model 指向它。',
|
|
83
|
+
)
|
|
84
|
+
}
|
|
85
|
+
const payload = JSON.parse(stdout.trim().split('\n').filter(Boolean).pop() ?? '{}') as {
|
|
86
|
+
segments?: Array<{ start: number; end?: number; text: string }>
|
|
87
|
+
language?: string
|
|
88
|
+
}
|
|
89
|
+
const cues = (payload.segments ?? [])
|
|
90
|
+
.map(segment => ({
|
|
91
|
+
seconds: Number(segment.start) || 0,
|
|
92
|
+
...(Number(segment.end) ? { endSeconds: Number(segment.end) } : {}),
|
|
93
|
+
text: (segment.text ?? '').trim(),
|
|
94
|
+
}))
|
|
95
|
+
.filter(cue => cue.text.length > 0)
|
|
96
|
+
return { cues, model, ...(payload.language ? { language: payload.language } : {}) }
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
export interface OcrEntry {
|
|
100
|
+
path: string
|
|
101
|
+
lines: string[]
|
|
102
|
+
text: string
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** 图片(或关键帧) → 画面文字。仅在调用方显式开启时才该被调到。 */
|
|
106
|
+
export async function ocrImages(
|
|
107
|
+
files: string[],
|
|
108
|
+
tools?: MediaToolOptions,
|
|
109
|
+
signal?: AbortSignal,
|
|
110
|
+
): Promise<{ entries: OcrEntry[]; warnings: Array<{ code: string; message: string }> }> {
|
|
111
|
+
if (files.length === 0) return { entries: [], warnings: [] }
|
|
112
|
+
const [cmd, args] = uvxArgs(tools, ['rapidocr-onnxruntime'], helperPath('ocr.py'), files)
|
|
113
|
+
try {
|
|
114
|
+
const { stdout } = await execFileAsync(cmd, args, { maxBuffer: 32 * 1024 * 1024, ...(signal ? { signal } : {}) })
|
|
115
|
+
const entries = JSON.parse(stdout.trim().split('\n').filter(Boolean).pop() ?? '[]') as OcrEntry[]
|
|
116
|
+
return { entries, warnings: [] }
|
|
117
|
+
} catch (cause) {
|
|
118
|
+
const detail = cause instanceof Error ? cause.message : String(cause)
|
|
119
|
+
if (/ENOENT/.test(detail)) {
|
|
120
|
+
return { entries: [], warnings: [{ code: 'OCR_UNAVAILABLE', message: `未找到 uvx,跳过 OCR。${UVX_HINT}` }] }
|
|
121
|
+
}
|
|
122
|
+
return { entries: [], warnings: [{ code: 'OCR_FAILED', message: `OCR 失败:${detail}` }] }
|
|
123
|
+
}
|
|
124
|
+
}
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 轻量扫描档:纯图片页 PDF → 逐页本地 OCR → Markdown。
|
|
3
|
+
*
|
|
4
|
+
* 解决的**唯一**问题:扫描件里的字读不出来(anydoc 只吃文本层 PDF)。
|
|
5
|
+
* 刻意不做哪些事:
|
|
6
|
+
*
|
|
7
|
+
* - **不引入 docling / PaddlePaddle / MinerU**:那是 issues/0017 的"可选高质量档",
|
|
8
|
+
* 用来做版面顺序、表格结构、公式还原;代价是几百 MB ~ 20 GB 的依赖。
|
|
9
|
+
* - **不承诺版面**:输出就是"第 N 页 + 该页图片 + 该页文字",顺序 = 页序。
|
|
10
|
+
* - **不联网**:整条链只有两个本地组件 —— pypdfium2(渲染页图)与 RapidOCR(识别文字)。
|
|
11
|
+
*
|
|
12
|
+
* 渲染用 pypdfium2 而不是 poppler,是因为 `pdftoppm` 是外部二进制,不是每台机器都有
|
|
13
|
+
* (本机就没有);pypdfium2 是 pip wheel,自带 pdfium,由 uvx 按需拉起。
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import { createHash } from 'node:crypto'
|
|
17
|
+
import { createReadStream } from 'node:fs'
|
|
18
|
+
import { execFile } from 'node:child_process'
|
|
19
|
+
import { mkdir, mkdtemp, readFile, rm, stat } from 'node:fs/promises'
|
|
20
|
+
import { basename, extname, join } from 'node:path'
|
|
21
|
+
import { tmpdir } from 'node:os'
|
|
22
|
+
import { fileURLToPath } from 'node:url'
|
|
23
|
+
import { promisify } from 'node:util'
|
|
24
|
+
import { UVX_HINT } from './binaries.js'
|
|
25
|
+
import { buildScannedMarkdown } from './markdown.js'
|
|
26
|
+
import { ocrImages } from './models.js'
|
|
27
|
+
import {
|
|
28
|
+
MediaToMarkdownError,
|
|
29
|
+
type MediaToMarkdownResult,
|
|
30
|
+
type MediaToolOptions,
|
|
31
|
+
type MediaWarning,
|
|
32
|
+
type ScannedPage,
|
|
33
|
+
type ScannedPdfOptions,
|
|
34
|
+
} from './types.js'
|
|
35
|
+
|
|
36
|
+
const execFileAsync = promisify(execFile)
|
|
37
|
+
|
|
38
|
+
function helperPath(name: string): string {
|
|
39
|
+
return fileURLToPath(new URL(`../../scripts/${name}`, import.meta.url))
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function uvxCommand(tools: MediaToolOptions | undefined): { cmd: string; python: string } {
|
|
43
|
+
return {
|
|
44
|
+
cmd: tools?.uvxPath?.trim() || process.env.MEDIA_TO_MD_UVX?.trim() || 'uvx',
|
|
45
|
+
python: tools?.pythonVersion?.trim() || process.env.MEDIA_TO_MD_PYTHON?.trim() || '3.12',
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
async function sha256OfFile(file: string): Promise<string> {
|
|
50
|
+
const hash = createHash('sha256')
|
|
51
|
+
await new Promise<void>((resolve, reject) => {
|
|
52
|
+
createReadStream(file)
|
|
53
|
+
.on('data', chunk => hash.update(chunk))
|
|
54
|
+
.on('error', reject)
|
|
55
|
+
.on('end', () => resolve())
|
|
56
|
+
})
|
|
57
|
+
return `sha256:${hash.digest('hex')}`
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
interface RenderedPages {
|
|
61
|
+
pages: Array<{ index: number; path: string; width?: number; height?: number }>
|
|
62
|
+
totalPages: number
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
async function renderPages(file: string, scratch: string, dpi: number, maxPages: number, tools?: MediaToolOptions, signal?: AbortSignal): Promise<RenderedPages & { uvxVersion?: string }> {
|
|
66
|
+
const { cmd, python } = uvxCommand(tools)
|
|
67
|
+
const outDir = join(scratch, 'pages')
|
|
68
|
+
await mkdir(outDir, { recursive: true })
|
|
69
|
+
let stdout: string
|
|
70
|
+
try {
|
|
71
|
+
({ stdout } = await execFileAsync(cmd, [
|
|
72
|
+
'--python', python, '--with', 'pypdfium2', '--with', 'pillow', 'python', helperPath('pdf_pages.py'),
|
|
73
|
+
file, outDir, String(dpi), String(maxPages),
|
|
74
|
+
], { maxBuffer: 32 * 1024 * 1024, ...(signal ? { signal } : {}) }))
|
|
75
|
+
} catch (cause) {
|
|
76
|
+
const detail = cause instanceof Error ? cause.message : String(cause)
|
|
77
|
+
if (/ENOENT/.test(detail)) {
|
|
78
|
+
throw new MediaToMarkdownError(
|
|
79
|
+
'DEPENDENCY_MISSING',
|
|
80
|
+
'扫描档需要 uvx(用于拉起 pypdfium2 渲染页图与 RapidOCR 识别),本机没有找到。',
|
|
81
|
+
UVX_HINT,
|
|
82
|
+
)
|
|
83
|
+
}
|
|
84
|
+
throw new MediaToMarkdownError('PDF_RENDER_FAILED', `页图渲染失败:${detail}`)
|
|
85
|
+
}
|
|
86
|
+
const payload = JSON.parse(stdout.trim().split('\n').filter(Boolean).pop() ?? '{}') as RenderedPages
|
|
87
|
+
// uvx 版本只在能力上报里用;这里顺手探测一次,失败不影响主流程
|
|
88
|
+
let uvxVersion: string | undefined
|
|
89
|
+
try {
|
|
90
|
+
const version = await execFileAsync(cmd, ['--version'])
|
|
91
|
+
uvxVersion = /(\d+\.\d+\.\d+)/.exec(version.stdout)?.[1]
|
|
92
|
+
} catch {
|
|
93
|
+
/* 可选 */
|
|
94
|
+
}
|
|
95
|
+
return { ...payload, ...(uvxVersion ? { uvxVersion } : {}) }
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/** 纯图片页 PDF → 逐页 OCR 的 Markdown(轻量档)。 */
|
|
99
|
+
export async function convertScannedDocument(options: ScannedPdfOptions): Promise<MediaToMarkdownResult> {
|
|
100
|
+
const file = options.file
|
|
101
|
+
const fileInfo = await stat(file).catch(() => undefined)
|
|
102
|
+
if (!fileInfo?.isFile()) throw new MediaToMarkdownError('INPUT_NOT_A_FILE', `输入不是文件:${file}`)
|
|
103
|
+
if (extname(file).toLowerCase() !== '.pdf') {
|
|
104
|
+
throw new MediaToMarkdownError('INPUT_NOT_PDF', `扫描档只处理 PDF,收到:${extname(file) || '(无扩展名)'}`)
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const title = options.title?.trim() || basename(file, extname(file))
|
|
108
|
+
const dpi = Math.min(300, Math.max(72, options.dpi ?? 150))
|
|
109
|
+
const maxPages = Math.max(1, options.maxPages ?? 50)
|
|
110
|
+
const pageImages = options.pageImages !== false
|
|
111
|
+
const report = (stage: Parameters<NonNullable<ScannedPdfOptions['onProgress']>>[0]['stage'], progress: number | null, message?: string): void => {
|
|
112
|
+
options.onProgress?.({ stage, progress, ...(message ? { message } : {}) })
|
|
113
|
+
}
|
|
114
|
+
const warnings: MediaWarning[] = []
|
|
115
|
+
const scratch = await mkdtemp(join(tmpdir(), 'media-to-md-scan-'))
|
|
116
|
+
|
|
117
|
+
try {
|
|
118
|
+
report('probing', 0.05, '渲染 PDF 页图')
|
|
119
|
+
const rendered = await renderPages(file, scratch, dpi, maxPages, options.tools, options.signal)
|
|
120
|
+
if (rendered.totalPages > rendered.pages.length) {
|
|
121
|
+
warnings.push({
|
|
122
|
+
code: 'PAGE_LIMIT_REACHED',
|
|
123
|
+
message: `PDF 共 ${rendered.totalPages} 页,本次只渲染并识别前 ${rendered.pages.length} 页(上限 ${maxPages});调大 maxPages 可覆盖更多。`,
|
|
124
|
+
})
|
|
125
|
+
}
|
|
126
|
+
if (rendered.pages.length === 0) {
|
|
127
|
+
warnings.push({ code: 'NO_PAGES', message: '该 PDF 没有可渲染的页面。' })
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
report('ocr', 0.4, `本地 OCR(${rendered.pages.length} 页)`)
|
|
131
|
+
const ocr = await ocrImages(rendered.pages.map(page => page.path), options.tools, options.signal)
|
|
132
|
+
warnings.push(...ocr.warnings)
|
|
133
|
+
const byPath = new Map(ocr.entries.map(entry => [entry.path, entry]))
|
|
134
|
+
|
|
135
|
+
const pages: ScannedPage[] = []
|
|
136
|
+
const assets: MediaToMarkdownResult['assets'] = []
|
|
137
|
+
let emptyPages = 0
|
|
138
|
+
for (const page of rendered.pages) {
|
|
139
|
+
const rel = `assets/page-${String(page.index).padStart(4, '0')}.png`
|
|
140
|
+
const entry = byPath.get(page.path)
|
|
141
|
+
const text = (entry?.text ?? '').trim()
|
|
142
|
+
if (!text) emptyPages += 1
|
|
143
|
+
pages.push({
|
|
144
|
+
index: page.index,
|
|
145
|
+
rel,
|
|
146
|
+
...(page.width ? { width: page.width } : {}),
|
|
147
|
+
...(page.height ? { height: page.height } : {}),
|
|
148
|
+
text,
|
|
149
|
+
ocrLines: entry?.lines.length ?? 0,
|
|
150
|
+
})
|
|
151
|
+
if (pageImages) {
|
|
152
|
+
assets.push({ rel, data: await readFile(page.path), contentType: 'image/png' })
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
const identified = pages.filter(page => page.text).length
|
|
156
|
+
warnings.push({
|
|
157
|
+
code: 'SCANNED_OCR_LOCAL',
|
|
158
|
+
message: `扫描档走本地 OCR:${identified}/${pages.length} 页识别到文字(${dpi} DPI);不联网、不做版面还原。`,
|
|
159
|
+
})
|
|
160
|
+
if (emptyPages > 0) {
|
|
161
|
+
warnings.push({ code: 'OCR_EMPTY_PAGES', message: `有 ${emptyPages} 页没有识别到文字(可能是空白页、手写体或图片型内容)。` })
|
|
162
|
+
}
|
|
163
|
+
if (!pageImages) {
|
|
164
|
+
warnings.push({ code: 'PAGE_IMAGES_SKIPPED', message: 'pageImages=false:只保留文字,未落地页图。' })
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
report('assembling', 0.9, '装配 Markdown')
|
|
168
|
+
const meta = {
|
|
169
|
+
sourceFormat: 'pdf' as const,
|
|
170
|
+
containerExtension: 'pdf',
|
|
171
|
+
durationSeconds: 0,
|
|
172
|
+
sizeBytes: fileInfo.size,
|
|
173
|
+
hasVideo: false,
|
|
174
|
+
hasAudio: false,
|
|
175
|
+
hasSubtitleTrack: false,
|
|
176
|
+
transcriptSource: 'none' as const,
|
|
177
|
+
}
|
|
178
|
+
const dependencyVersions: Record<string, string> = { pypdfium2: 'via uvx', rapidocr: 'via uvx' }
|
|
179
|
+
if (rendered.uvxVersion) dependencyVersions.uvx = rendered.uvxVersion
|
|
180
|
+
|
|
181
|
+
return {
|
|
182
|
+
title: title.slice(0, 300),
|
|
183
|
+
markdown: buildScannedMarkdown({
|
|
184
|
+
title,
|
|
185
|
+
totalPages: rendered.totalPages,
|
|
186
|
+
pageImages,
|
|
187
|
+
pages: pages.map(page => ({ index: page.index, rel: page.rel, text: page.text })),
|
|
188
|
+
}),
|
|
189
|
+
assets,
|
|
190
|
+
meta,
|
|
191
|
+
transcript: { source: 'none', cues: [] },
|
|
192
|
+
warnings,
|
|
193
|
+
dependencyVersions,
|
|
194
|
+
externalId: await sha256OfFile(file),
|
|
195
|
+
pages,
|
|
196
|
+
}
|
|
197
|
+
} finally {
|
|
198
|
+
if (!options.keepScratch) await rm(scratch, { recursive: true, force: true }).catch(() => undefined)
|
|
199
|
+
}
|
|
200
|
+
}
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* media-to-md 的公开契约。
|
|
3
|
+
*
|
|
4
|
+
* 这个包只做一件事:把一个**本地**音视频/图片文件变成「知识就绪 Markdown + 本地图片」。
|
|
5
|
+
* 它刻意不认识任何宿主(转换内核、CLI、未来的桌面端),宿主拿到的就是一个纯函数:
|
|
6
|
+
*
|
|
7
|
+
* convertMedia(options) → { markdown, assets[], meta, transcript, warnings, … }
|
|
8
|
+
*
|
|
9
|
+
* 全程本地:不调用云端 API、不调用大模型。用到的两个**专用小模型**(语音识别 whisper、
|
|
10
|
+
* 画面文字 OCR)都是可选档位,且是否允许下载由调用方显式给出(`allowModelDownload`)。
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
export type MediaSourceFormat = 'video' | 'audio' | 'image' | 'pdf'
|
|
14
|
+
|
|
15
|
+
/** 解析档位。`auto` = 有字幕用字幕,没有才考虑转写(且必须允许下载模型)。 */
|
|
16
|
+
export type MediaMode = 'auto' | 'subtitle-only' | 'transcribe' | 'frames-only'
|
|
17
|
+
|
|
18
|
+
export interface MediaAsrOptions {
|
|
19
|
+
/** whisper 模型名(tiny/base/small/medium/large-v3/large-v3-turbo)。 */
|
|
20
|
+
model?: string
|
|
21
|
+
/** 语言代码(zh/en…);省略则自动检测。 */
|
|
22
|
+
language?: string
|
|
23
|
+
/** faster-whisper 设备与量化,默认 cpu/int8。 */
|
|
24
|
+
device?: string
|
|
25
|
+
computeType?: string
|
|
26
|
+
beamSize?: number
|
|
27
|
+
/**
|
|
28
|
+
* 是否允许下载模型。默认 **false**:
|
|
29
|
+
* 无字幕又没有本地模型时会抛 `ASR_MODEL_REQUIRED`,把「要下多大」摊在调用方面前。
|
|
30
|
+
*/
|
|
31
|
+
allowModelDownload?: boolean
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export interface MediaFrameOptions {
|
|
35
|
+
/** 均匀采样帧数(保证任何视频都有基础覆盖)。 */
|
|
36
|
+
samples?: number
|
|
37
|
+
/** 关键帧上限。 */
|
|
38
|
+
maxFrames?: number
|
|
39
|
+
/** 场景切换阈值(0–1);硬切/翻页会额外抽帧。 */
|
|
40
|
+
sceneThreshold?: number
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export interface MediaToolOptions {
|
|
44
|
+
/** uvx 可执行文件;默认 `uvx`。 */
|
|
45
|
+
uvxPath?: string
|
|
46
|
+
/** uvx 使用的 Python 版本;默认 `3.12`。 */
|
|
47
|
+
pythonVersion?: string
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
export interface MediaToMarkdownOptions {
|
|
51
|
+
/** 本地文件绝对路径。 */
|
|
52
|
+
file: string
|
|
53
|
+
/** 产物标题;默认取文件名(去扩展名)。 */
|
|
54
|
+
title?: string
|
|
55
|
+
mode?: MediaMode
|
|
56
|
+
asr?: MediaAsrOptions
|
|
57
|
+
frames?: MediaFrameOptions
|
|
58
|
+
tools?: MediaToolOptions
|
|
59
|
+
/** 对**图片**输入开启本地 OCR;默认 false(issue 0013:图片直接收录、靠多模态召回)。 */
|
|
60
|
+
imageOcr?: boolean
|
|
61
|
+
/** 对视频**关键帧**做画面文字 OCR;默认 false。 */
|
|
62
|
+
ocrFrames?: boolean
|
|
63
|
+
/** 保留中间目录(排障用)。 */
|
|
64
|
+
keepScratch?: boolean
|
|
65
|
+
/** 取消信号。 */
|
|
66
|
+
signal?: AbortSignal
|
|
67
|
+
onProgress?: (progress: MediaToMarkdownProgress) => void
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* 轻量扫描档:纯图片页 PDF → 逐页本地 OCR。
|
|
72
|
+
*
|
|
73
|
+
* 刻意**不引入 docling / PaddlePaddle 框架**:它只解决"扫描件里的字读不出来",
|
|
74
|
+
* 版面顺序与表格结构不在这一档的承诺范围内(见 issues/0017 的可选高质量档)。
|
|
75
|
+
*/
|
|
76
|
+
export interface ScannedPdfOptions {
|
|
77
|
+
file: string
|
|
78
|
+
title?: string
|
|
79
|
+
/** 最多渲染并识别的页数(默认 50);超出会写 warning,不静默截断。 */
|
|
80
|
+
maxPages?: number
|
|
81
|
+
/** 渲染 DPI(默认 150);越高越准,也越慢越大。 */
|
|
82
|
+
dpi?: number
|
|
83
|
+
/** 是否把逐页图片作为 assets 落地(默认 true;false 时只保留文字)。 */
|
|
84
|
+
pageImages?: boolean
|
|
85
|
+
tools?: MediaToolOptions
|
|
86
|
+
keepScratch?: boolean
|
|
87
|
+
signal?: AbortSignal
|
|
88
|
+
onProgress?: (progress: MediaToMarkdownProgress) => void
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** 扫描档的页级结果。 */
|
|
92
|
+
export interface ScannedPage {
|
|
93
|
+
index: number
|
|
94
|
+
/** 该页图片在产物里的相对路径(未落地图片时为 `page-NNNN.png` 的约定名)。 */
|
|
95
|
+
rel: string
|
|
96
|
+
width?: number
|
|
97
|
+
height?: number
|
|
98
|
+
text: string
|
|
99
|
+
ocrLines: number
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export interface MediaToMarkdownProgress {
|
|
103
|
+
stage: 'probing' | 'frames' | 'subtitles' | 'transcribing' | 'ocr' | 'assembling'
|
|
104
|
+
/** 0–1,未知时为 null。 */
|
|
105
|
+
progress: number | null
|
|
106
|
+
message?: string
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export interface MediaAsset {
|
|
110
|
+
/** 相对产物根的路径,**含 `assets/` 前缀**,可直接写进 Markdown。 */
|
|
111
|
+
rel: string
|
|
112
|
+
data: Buffer
|
|
113
|
+
contentType?: string
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
export interface MediaCue {
|
|
117
|
+
seconds: number
|
|
118
|
+
/** 结束时间(字幕/SRT 有;自动转写也有)。缺失时按下一条的开始时间处理。 */
|
|
119
|
+
endSeconds?: number
|
|
120
|
+
text: string
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
export interface MediaMeta {
|
|
124
|
+
sourceFormat: MediaSourceFormat
|
|
125
|
+
/** 容器/图片扩展名(不含点),例如 `mp4`、`png` —— 宿主用它拼 parser 标识。 */
|
|
126
|
+
containerExtension?: string
|
|
127
|
+
durationSeconds: number
|
|
128
|
+
sizeBytes: number
|
|
129
|
+
width?: number
|
|
130
|
+
height?: number
|
|
131
|
+
fps?: number
|
|
132
|
+
videoCodec?: string
|
|
133
|
+
audioCodec?: string
|
|
134
|
+
hasVideo: boolean
|
|
135
|
+
hasAudio: boolean
|
|
136
|
+
hasSubtitleTrack: boolean
|
|
137
|
+
/** 实际用到的字幕来源。 */
|
|
138
|
+
transcriptSource: 'subtitle' | 'transcription' | 'none'
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
export interface MediaWarning {
|
|
142
|
+
code: string
|
|
143
|
+
message: string
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
export interface MediaToMarkdownResult {
|
|
147
|
+
title: string
|
|
148
|
+
/** Markdown 正文;**不含 front matter**(front matter 是宿主的契约)。 */
|
|
149
|
+
markdown: string
|
|
150
|
+
assets: MediaAsset[]
|
|
151
|
+
meta: MediaMeta
|
|
152
|
+
transcript: { source: MediaMeta['transcriptSource']; cues: MediaCue[] }
|
|
153
|
+
warnings: MediaWarning[]
|
|
154
|
+
dependencyVersions: Record<string, string>
|
|
155
|
+
/** 文件内容 hash(sha256:…),给宿主做幂等。 */
|
|
156
|
+
externalId: string
|
|
157
|
+
/** 扫描档才有:逐页文本与图片。 */
|
|
158
|
+
pages?: ScannedPage[]
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/** 带可操作建议的错误:宿主(内核/CLI)应当原样展示 `code` 与 `suggestion`。 */export class MediaToMarkdownError extends Error {
|
|
162
|
+
readonly code: string
|
|
163
|
+
readonly suggestion?: string
|
|
164
|
+
|
|
165
|
+
constructor(code: string, message: string, suggestion?: string) {
|
|
166
|
+
super(message)
|
|
167
|
+
this.name = 'MediaToMarkdownError'
|
|
168
|
+
this.code = code
|
|
169
|
+
if (suggestion !== undefined) this.suggestion = suggestion
|
|
170
|
+
}
|
|
171
|
+
}
|
package/src/server.ts
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Public entry point of the conversion kernel.
|
|
3
|
+
*
|
|
4
|
+
* Two integration shapes, one implementation:
|
|
5
|
+
* const service = await createConvertService({ rootDir })
|
|
6
|
+
* const server = await startConvertServer({ port: 8787 }) // standalone
|
|
7
|
+
* const handler = createRequestHandler({ service, basePath }) // embedded
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { createServer, type Server } from 'node:http'
|
|
11
|
+
import { createRequestHandler, type HandlerOptions } from './http.js'
|
|
12
|
+
import { ConvertService, defaultConvertRoot, type ConvertServiceOptions } from './service.js'
|
|
13
|
+
|
|
14
|
+
export * from './contract.js'
|
|
15
|
+
export * from './service.js'
|
|
16
|
+
export * from './delivery.js'
|
|
17
|
+
export * from './adapters/index.js'
|
|
18
|
+
export * from './adapters/types.js'
|
|
19
|
+
export { createRequestHandler, type HandlerOptions } from './http.js'
|
|
20
|
+
|
|
21
|
+
export interface CreateServiceOptions extends ConvertServiceOptions {}
|
|
22
|
+
|
|
23
|
+
export async function createConvertService(options: CreateServiceOptions = {}): Promise<ConvertService> {
|
|
24
|
+
const service = new ConvertService({ rootDir: options.rootDir ?? defaultConvertRoot(), ...options })
|
|
25
|
+
await service.start()
|
|
26
|
+
return service
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export interface StartServerOptions extends CreateServiceOptions {
|
|
30
|
+
port?: number
|
|
31
|
+
host?: string
|
|
32
|
+
basePath?: string
|
|
33
|
+
webRoot?: string
|
|
34
|
+
apiKey?: string
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export interface RunningConvertServer {
|
|
38
|
+
service: ConvertService
|
|
39
|
+
server: Server
|
|
40
|
+
url: string
|
|
41
|
+
close: () => Promise<void>
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** Start the kernel standalone — the console is at `url`, the API under `url/api`. */
|
|
45
|
+
export async function startConvertServer(options: StartServerOptions = {}): Promise<RunningConvertServer> {
|
|
46
|
+
const service = await createConvertService(options)
|
|
47
|
+
const handlerOptions: HandlerOptions = {
|
|
48
|
+
service,
|
|
49
|
+
basePath: options.basePath ?? '',
|
|
50
|
+
...(options.webRoot ? { webRoot: options.webRoot } : {}),
|
|
51
|
+
...(options.apiKey ? { apiKey: options.apiKey } : {}),
|
|
52
|
+
}
|
|
53
|
+
const handler = createRequestHandler(handlerOptions)
|
|
54
|
+
const server = createServer((req, res) => {
|
|
55
|
+
void handler(req, res).catch(() => {
|
|
56
|
+
if (!res.headersSent) {
|
|
57
|
+
res.statusCode = 500
|
|
58
|
+
res.setHeader('content-type', 'application/json; charset=utf-8')
|
|
59
|
+
}
|
|
60
|
+
res.end(JSON.stringify({ ok: false, error: { code: 'INTERNAL_ERROR', message: '未捕获的服务端错误' } }))
|
|
61
|
+
})
|
|
62
|
+
})
|
|
63
|
+
|
|
64
|
+
const port = options.port ?? 8787
|
|
65
|
+
const host = options.host ?? '127.0.0.1'
|
|
66
|
+
await new Promise<void>((resolve, reject) => {
|
|
67
|
+
server.once('error', reject)
|
|
68
|
+
server.listen(port, host, () => {
|
|
69
|
+
server.off('error', reject)
|
|
70
|
+
resolve()
|
|
71
|
+
})
|
|
72
|
+
})
|
|
73
|
+
const address = server.address()
|
|
74
|
+
const actualPort = typeof address === 'object' && address ? address.port : port
|
|
75
|
+
const basePath = (options.basePath ?? '').replace(/\/+$/, '')
|
|
76
|
+
return {
|
|
77
|
+
service,
|
|
78
|
+
server,
|
|
79
|
+
url: `http://${host === '0.0.0.0' ? '127.0.0.1' : host}:${actualPort}${basePath}`,
|
|
80
|
+
close: async () => {
|
|
81
|
+
await service.close()
|
|
82
|
+
await new Promise<void>(resolve => server.close(() => resolve()))
|
|
83
|
+
},
|
|
84
|
+
}
|
|
85
|
+
}
|