dsh-convert-core 0.1.0-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (115) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +149 -0
  3. package/bin/media-to-md.mjs +168 -0
  4. package/bin/tita-convert.mjs +85 -0
  5. package/docs/media-to-md.md +143 -0
  6. package/lib/adapters/anydoc.d.ts +73 -0
  7. package/lib/adapters/anydoc.d.ts.map +1 -0
  8. package/lib/adapters/anydoc.js +295 -0
  9. package/lib/adapters/anydoc.js.map +1 -0
  10. package/lib/adapters/iddoc.d.ts +28 -0
  11. package/lib/adapters/iddoc.d.ts.map +1 -0
  12. package/lib/adapters/iddoc.js +153 -0
  13. package/lib/adapters/iddoc.js.map +1 -0
  14. package/lib/adapters/index.d.ts +41 -0
  15. package/lib/adapters/index.d.ts.map +1 -0
  16. package/lib/adapters/index.js +115 -0
  17. package/lib/adapters/index.js.map +1 -0
  18. package/lib/adapters/scanned.d.ts +28 -0
  19. package/lib/adapters/scanned.d.ts.map +1 -0
  20. package/lib/adapters/scanned.js +121 -0
  21. package/lib/adapters/scanned.js.map +1 -0
  22. package/lib/adapters/text.d.ts +29 -0
  23. package/lib/adapters/text.d.ts.map +1 -0
  24. package/lib/adapters/text.js +135 -0
  25. package/lib/adapters/text.js.map +1 -0
  26. package/lib/adapters/types.d.ts +65 -0
  27. package/lib/adapters/types.d.ts.map +1 -0
  28. package/lib/adapters/types.js +11 -0
  29. package/lib/adapters/types.js.map +1 -0
  30. package/lib/contract.d.ts +185 -0
  31. package/lib/contract.d.ts.map +1 -0
  32. package/lib/contract.js +59 -0
  33. package/lib/contract.js.map +1 -0
  34. package/lib/delivery.d.ts +36 -0
  35. package/lib/delivery.d.ts.map +1 -0
  36. package/lib/delivery.js +97 -0
  37. package/lib/delivery.js.map +1 -0
  38. package/lib/http.d.ts +34 -0
  39. package/lib/http.d.ts.map +1 -0
  40. package/lib/http.js +372 -0
  41. package/lib/http.js.map +1 -0
  42. package/lib/markdown.d.ts +32 -0
  43. package/lib/markdown.d.ts.map +1 -0
  44. package/lib/markdown.js +145 -0
  45. package/lib/markdown.js.map +1 -0
  46. package/lib/media/binaries.d.ts +26 -0
  47. package/lib/media/binaries.d.ts.map +1 -0
  48. package/lib/media/binaries.js +60 -0
  49. package/lib/media/binaries.js.map +1 -0
  50. package/lib/media/convert.d.ts +19 -0
  51. package/lib/media/convert.d.ts.map +1 -0
  52. package/lib/media/convert.js +210 -0
  53. package/lib/media/convert.js.map +1 -0
  54. package/lib/media/index.d.ts +27 -0
  55. package/lib/media/index.d.ts.map +1 -0
  56. package/lib/media/index.js +27 -0
  57. package/lib/media/index.js.map +1 -0
  58. package/lib/media/markdown.d.ts +40 -0
  59. package/lib/media/markdown.d.ts.map +1 -0
  60. package/lib/media/markdown.js +89 -0
  61. package/lib/media/markdown.js.map +1 -0
  62. package/lib/media/media.d.ts +45 -0
  63. package/lib/media/media.d.ts.map +1 -0
  64. package/lib/media/media.js +167 -0
  65. package/lib/media/media.js.map +1 -0
  66. package/lib/media/models.d.ts +31 -0
  67. package/lib/media/models.d.ts.map +1 -0
  68. package/lib/media/models.js +87 -0
  69. package/lib/media/models.js.map +1 -0
  70. package/lib/media/scanned.d.ts +18 -0
  71. package/lib/media/scanned.d.ts.map +1 -0
  72. package/lib/media/scanned.js +179 -0
  73. package/lib/media/scanned.js.map +1 -0
  74. package/lib/media/types.d.ts +154 -0
  75. package/lib/media/types.d.ts.map +1 -0
  76. package/lib/media/types.js +23 -0
  77. package/lib/media/types.js.map +1 -0
  78. package/lib/server.d.ts +35 -0
  79. package/lib/server.d.ts.map +1 -0
  80. package/lib/server.js +64 -0
  81. package/lib/server.js.map +1 -0
  82. package/lib/service.d.ts +170 -0
  83. package/lib/service.d.ts.map +1 -0
  84. package/lib/service.js +562 -0
  85. package/lib/service.js.map +1 -0
  86. package/package.json +59 -0
  87. package/scripts/asr.py +55 -0
  88. package/scripts/ocr.py +40 -0
  89. package/scripts/pdf_pages.py +49 -0
  90. package/scripts/vendor.mjs +76 -0
  91. package/src/adapters/anydoc.ts +356 -0
  92. package/src/adapters/iddoc.ts +171 -0
  93. package/src/adapters/index.ts +147 -0
  94. package/src/adapters/scanned.ts +139 -0
  95. package/src/adapters/text.ts +146 -0
  96. package/src/adapters/types.ts +76 -0
  97. package/src/contract.ts +230 -0
  98. package/src/delivery.ts +124 -0
  99. package/src/http.ts +394 -0
  100. package/src/markdown.ts +141 -0
  101. package/src/media/binaries.ts +67 -0
  102. package/src/media/convert.ts +232 -0
  103. package/src/media/index.ts +43 -0
  104. package/src/media/markdown.ts +105 -0
  105. package/src/media/media.ts +204 -0
  106. package/src/media/models.ts +124 -0
  107. package/src/media/scanned.ts +200 -0
  108. package/src/media/types.ts +171 -0
  109. package/src/server.ts +85 -0
  110. package/src/service.ts +639 -0
  111. package/web/app.js +382 -0
  112. package/web/index.html +217 -0
  113. package/web/styles.css +263 -0
  114. package/web/vendor/icons.js +25 -0
  115. package/web/vendor/vue.global.prod.js +14 -0
@@ -0,0 +1,171 @@
1
+ /**
2
+ * iddoc —— 音视频 / 图片 adapter(issue 0017 的 video 档)。
3
+ *
4
+ * 这个文件**只做契约映射**:能力全在同包的 `src/media/` 里(纯本地管线:
5
+ * ffmpeg 抽帧 + 可选的字幕/转写 + 可选的画面 OCR),这里负责
6
+ *
7
+ * 1. 把内核的 `AdapterContext` 翻译成该包的 options(**含 env 旋钮**);
8
+ * 2. 把 `MediaToMarkdownResult` 映射成 `AdapterOutput`(assets 交给内核写盘);
9
+ * 3. 把它抛出的错误原样转成 `AdapterError`,保留 `code` 与可操作建议。
10
+ *
11
+ * 与文档档(anydoc)一样:引擎独立于 adapter,内核只管契约。
12
+ *
13
+ * 环境变量(都是**显式**开关,默认不下载模型、不对图片做 OCR):
14
+ * IDDOC_ASR=1 允许在无字幕时下载模型并本地转写
15
+ * IDDOC_ASR_MODEL whisper 模型(默认 small;中文建议 large-v3-turbo)
16
+ * IDDOC_ASR_LANG 语言代码(默认自动检测;中文建议 zh)
17
+ * IDDOC_OCR_FRAMES=1 对关键帧做画面文字 OCR
18
+ * IDDOC_IMAGE_OCR=1 对图片输入做 OCR(默认关闭:图片直接收录)
19
+ * IDDOC_UVX / IDDOC_PYTHON 自定义 uvx 与 Python 版本
20
+ * IDDOC_SAMPLES / IDDOC_MAX_FRAMES / IDDOC_SCENE 抽帧参数
21
+ */
22
+
23
+ import { stat } from 'node:fs/promises'
24
+ import {
25
+ IMAGE_EXTENSIONS,
26
+ MediaToMarkdownError,
27
+ convertMedia,
28
+ probeTools,
29
+ type MediaAsset,
30
+ type MediaToMarkdownOptions,
31
+ } from '../media/index.js'
32
+ import type { ConvertWarning } from '../contract.js'
33
+ import { AdapterError, type Adapter, type AdapterAvailability, type AdapterContext, type AdapterOutput, type PreparedAsset } from './types.js'
34
+
35
+ export const IDDOC_VIDEO_EXTENSIONS = [
36
+ '.mp4', '.m4v', '.mov', '.mkv', '.webm', '.avi', '.flv', '.wmv', '.mpg', '.mpeg', '.ogv',
37
+ ]
38
+ export const IDDOC_AUDIO_EXTENSIONS = [
39
+ '.mp3', '.m4a', '.wav', '.aac', '.flac', '.ogg', '.opus', '.aiff', '.aif', '.amr', '.wma',
40
+ ]
41
+ export const IDDOC_IMAGE_EXTENSIONS = IMAGE_EXTENSIONS
42
+ export const IDDOC_EXTENSIONS = [...IDDOC_VIDEO_EXTENSIONS, ...IDDOC_AUDIO_EXTENSIONS, ...IDDOC_IMAGE_EXTENSIONS]
43
+
44
+ const INSTALL_HINT = '装 ffmpeg(含 ffprobe)后重试:macOS `brew install ffmpeg`,Debian/Ubuntu `apt install ffmpeg`。'
45
+
46
+ function envFlag(name: string): boolean {
47
+ const value = (process.env[name] ?? '').trim().toLowerCase()
48
+ return value === '1' || value === 'true' || value === 'yes' || value === 'on'
49
+ }
50
+
51
+ function envInt(name: string): number | undefined {
52
+ const parsed = Number.parseInt((process.env[name] ?? '').trim(), 10)
53
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : undefined
54
+ }
55
+
56
+ function envFloat(name: string): number | undefined {
57
+ const parsed = Number.parseFloat((process.env[name] ?? '').trim())
58
+ return Number.isFinite(parsed) && parsed >= 0 && parsed <= 1 ? parsed : undefined
59
+ }
60
+
61
+ /** 路由用的廉价同步检查;真正的探测在 probe() 与首次使用时。 */
62
+ function availabilitySync(): AdapterAvailability {
63
+ return { available: true }
64
+ }
65
+
66
+ async function probeAvailability(): Promise<AdapterAvailability> {
67
+ const versions = await probeTools({ ...(process.env.IDDOC_UVX ? { uvxPath: process.env.IDDOC_UVX } : {}) })
68
+ if (!versions.ffprobe || !versions.ffmpeg) {
69
+ const missing = [!versions.ffprobe ? 'ffprobe' : null, !versions.ffmpeg ? 'ffmpeg' : null].filter(Boolean).join('、')
70
+ return { available: false, reason: `缺少外部二进制:${missing}`, suggestion: INSTALL_HINT }
71
+ }
72
+ const reported: Record<string, string> = {}
73
+ if (versions.ffmpeg) reported.ffmpeg = versions.ffmpeg
74
+ if (versions.ffprobe) reported.ffprobe = versions.ffprobe
75
+ if (versions.uvx) reported.uvx = versions.uvx
76
+ return { available: true, versions: reported }
77
+ }
78
+
79
+ function toOptions(context: AdapterContext, file: string): MediaToMarkdownOptions {
80
+ const model = (process.env.IDDOC_ASR_MODEL ?? '').trim() || 'small'
81
+ const language = (process.env.IDDOC_ASR_LANG ?? '').trim()
82
+ const frames = {
83
+ ...(envInt('IDDOC_SAMPLES') ? { samples: envInt('IDDOC_SAMPLES') } : {}),
84
+ ...(envInt('IDDOC_MAX_FRAMES') ? { maxFrames: envInt('IDDOC_MAX_FRAMES') } : {}),
85
+ ...(envFloat('IDDOC_SCENE') !== undefined ? { sceneThreshold: envFloat('IDDOC_SCENE') } : {}),
86
+ }
87
+ const tools = {
88
+ ...(process.env.IDDOC_UVX?.trim() ? { uvxPath: process.env.IDDOC_UVX.trim() } : {}),
89
+ ...(process.env.IDDOC_PYTHON?.trim() ? { pythonVersion: process.env.IDDOC_PYTHON.trim() } : {}),
90
+ }
91
+ return {
92
+ file,
93
+ ...(context.request.title?.trim() ? { title: context.request.title.trim() } : {}),
94
+ mode: 'auto',
95
+ asr: {
96
+ model,
97
+ ...(language ? { language } : {}),
98
+ allowModelDownload: envFlag('IDDOC_ASR'),
99
+ },
100
+ ...(Object.keys(frames).length > 0 ? { frames } : {}),
101
+ ...(Object.keys(tools).length > 0 ? { tools } : {}),
102
+ ...(envFlag('IDDOC_IMAGE_OCR') ? { imageOcr: true } : {}),
103
+ ...(envFlag('IDDOC_OCR_FRAMES') ? { ocrFrames: true } : {}),
104
+ ...(context.signal ? { signal: context.signal } : {}),
105
+ onProgress: progress => context.report(progress.progress, progress.stage, progress.message),
106
+ }
107
+ }
108
+
109
+ function toPreparedAssets(assets: MediaAsset[]): PreparedAsset[] {
110
+ return assets.map(asset => ({
111
+ rel: asset.rel,
112
+ data: asset.data,
113
+ ...(asset.contentType ? { contentType: asset.contentType } : {}),
114
+ }))
115
+ }
116
+
117
+ async function run(context: AdapterContext): Promise<AdapterOutput> {
118
+ if (!context.inputPath) {
119
+ throw new AdapterError('INPUT_REQUIRED', 'iddoc 只处理本地文件(kind=file):音频、视频或图片。')
120
+ }
121
+ const file = context.inputPath
122
+ const fileInfo = await stat(file).catch(() => undefined)
123
+ if (!fileInfo?.isFile()) throw new AdapterError('INPUT_NOT_A_FILE', `输入不是文件:${file}`)
124
+
125
+ const versions = await probeTools({ ...(process.env.IDDOC_UVX ? { uvxPath: process.env.IDDOC_UVX } : {}) })
126
+ if (!versions.ffprobe || !versions.ffmpeg) {
127
+ throw new AdapterError('DEPENDENCY_MISSING', 'iddoc 需要 ffmpeg 与 ffprobe。', INSTALL_HINT)
128
+ }
129
+
130
+ let result
131
+ try {
132
+ result = await convertMedia(toOptions(context, file))
133
+ } catch (cause) {
134
+ if (cause instanceof MediaToMarkdownError) {
135
+ throw new AdapterError(cause.code, cause.message, cause.suggestion)
136
+ }
137
+ throw new AdapterError('ADAPTER_FAILED', cause instanceof Error ? cause.message : String(cause))
138
+ }
139
+
140
+ const warnings: ConvertWarning[] = result.warnings.map(warning => ({ code: warning.code, message: warning.message }))
141
+ const suffix = result.meta.transcriptSource === 'subtitle'
142
+ ? '+subtitle'
143
+ : result.meta.transcriptSource === 'transcription' ? '+asr' : ''
144
+ // 文档类按格式命名(pdf/office),媒体类按容器命名(mp4/png),与既有 parser 约定保持一致。
145
+ const parserKind = result.meta.sourceFormat === 'image'
146
+ ? (result.meta.containerExtension ?? 'image')
147
+ : `${result.meta.sourceFormat}${suffix}`
148
+ const assets = toPreparedAssets(result.assets)
149
+
150
+ return {
151
+ title: result.title,
152
+ markdown: result.markdown,
153
+ sourceFormat: result.meta.sourceFormat,
154
+ parser: `iddoc@1/${parserKind}`,
155
+ sourcePath: file,
156
+ warnings,
157
+ dependencyVersions: result.dependencyVersions,
158
+ externalId: result.externalId,
159
+ ...(assets.length > 0 ? { assets } : {}),
160
+ }
161
+ }
162
+
163
+ export const iddocAdapter: Adapter = {
164
+ id: 'iddoc',
165
+ description: '本地音视频 / 图片 → Markdown 与本地图片(能力来自内核自带的 src/media/:字幕优先;转写需显式开启;图片直接收录)',
166
+ kinds: ['file'],
167
+ extensions: IDDOC_EXTENSIONS,
168
+ availability: availabilitySync,
169
+ probe: probeAvailability,
170
+ run,
171
+ }
@@ -0,0 +1,147 @@
1
+ /**
2
+ * Adapter registry.
3
+ *
4
+ * Resolution order stays trivial and explicit: an explicit override, then the
5
+ * media adapter (`iddoc`, by extension), then the document engine, then the text
6
+ * pass-through. Unknown extensions are only treated as text when the bytes are
7
+ * actually textual — guessing there is how "converted successfully" becomes a lie.
8
+ *
9
+ * `.pdf` 默认仍走 anydoc(文本层最好、零依赖);只有显式 `SCANNED_OCR=1` 才改走
10
+ * `scanned`(纯图片页的轻量 OCR 档)。
11
+ */
12
+
13
+ import { open } from 'node:fs/promises'
14
+ import { extname } from 'node:path'
15
+ import type { ConvertRequest } from '../contract.js'
16
+ import { ANYDOC_EXTENSIONS, anydocAdapter, anydocAvailability } from './anydoc.js'
17
+ import { IDDOC_EXTENSIONS, iddocAdapter } from './iddoc.js'
18
+ import { SCANNED_EXTENSIONS, scannedAdapter, scannedPreferred } from './scanned.js'
19
+ import { textAdapter, TEXT_EXTENSIONS } from './text.js'
20
+ import { AdapterError, type Adapter, type AdapterAvailability } from './types.js'
21
+
22
+ export const ADAPTERS: Adapter[] = [anydocAdapter, iddocAdapter, scannedAdapter, textAdapter]
23
+
24
+ export function findAdapter(id: string): Adapter | undefined {
25
+ return ADAPTERS.find(adapter => adapter.id === id)
26
+ }
27
+
28
+ /** Cheap binary sniff: NUL bytes or invalid UTF-8 in the head means "not text". */
29
+ export async function looksTextual(path: string): Promise<boolean> {
30
+ let handle
31
+ try {
32
+ handle = await open(path, 'r')
33
+ } catch {
34
+ return false
35
+ }
36
+ try {
37
+ const buffer = Buffer.alloc(4096)
38
+ const { bytesRead } = await handle.read(buffer, 0, buffer.length, 0)
39
+ if (bytesRead === 0) return true
40
+ const head = buffer.subarray(0, bytesRead)
41
+ if (head.includes(0)) return false
42
+ return !head.toString('utf8').includes('\uFFFD')
43
+ } finally {
44
+ await handle.close().catch(() => undefined)
45
+ }
46
+ }
47
+
48
+ export const SUPPORTED_EXTENSIONS = [...ANYDOC_EXTENSIONS, ...IDDOC_EXTENSIONS, ...TEXT_EXTENSIONS]
49
+
50
+ export async function resolveAdapter(
51
+ request: ConvertRequest,
52
+ filename: string,
53
+ inputPath: string | undefined,
54
+ ): Promise<Adapter> {
55
+ if (request.adapter) {
56
+ const explicit = findAdapter(request.adapter)
57
+ if (!explicit) {
58
+ throw new AdapterError(
59
+ 'ADAPTER_UNKNOWN',
60
+ `未知的 adapter:${request.adapter}`,
61
+ `可用 adapter:${ADAPTERS.map(adapter => adapter.id).join(', ')}`,
62
+ )
63
+ }
64
+ const availability = await (explicit.probe?.() ?? Promise.resolve(explicit.availability()))
65
+ if (!availability.available) {
66
+ throw new AdapterError('DEPENDENCY_MISSING', availability.reason ?? `${explicit.id} 当前不可用`, availability.suggestion)
67
+ }
68
+ return explicit
69
+ }
70
+
71
+ if (request.kind === 'text') return textAdapter
72
+
73
+ const extension = extname(filename).toLowerCase()
74
+ // 扫描档是显式 opt-in:默认仍让 anydoc 试文本层,失败了错误建议再指向 scanned。
75
+ if (SCANNED_EXTENSIONS.includes(extension) && scannedPreferred()) return scannedAdapter
76
+ if (ANYDOC_EXTENSIONS.includes(extension)) return anydocAdapter
77
+ if (IDDOC_EXTENSIONS.includes(extension)) return iddocAdapter
78
+ if (TEXT_EXTENSIONS.includes(extension)) return textAdapter
79
+ if (inputPath && (await looksTextual(inputPath))) return textAdapter
80
+
81
+ throw new AdapterError(
82
+ 'FORMAT_UNSUPPORTED',
83
+ `不支持的输入格式:${extension || '(无扩展名)'}`,
84
+ `本版做文件类:${ANYDOC_EXTENSIONS.join(' ')}(文档)、${SCANNED_EXTENSIONS.join(' ')}(扫描件 OCR 档)、${IDDOC_EXTENSIONS.join(' ')}(音视频/图片)以及 ${TEXT_EXTENSIONS.join(' ')}(直通);网页链接仍在 issues/0017。`,
85
+ )
86
+ }
87
+
88
+ export interface AdapterCapability {
89
+ id: string
90
+ description: string
91
+ kinds: ConvertRequest['kind'][]
92
+ extensions: string[]
93
+ availability: AdapterAvailability
94
+ }
95
+
96
+ export interface CapabilityReport {
97
+ adapters: AdapterCapability[]
98
+ /** Engine-level runtime facts; documents need no external binary, iddoc needs ffmpeg. */
99
+ runtime: Record<string, { available: boolean; version?: string; error?: string }>
100
+ defaults: Record<string, string>
101
+ /** Extensions the document engine covers. */
102
+ extensions: string[]
103
+ }
104
+
105
+ /** Everything the console and the DSH settings panel need to tell the truth. */
106
+ export async function capabilityReport(): Promise<CapabilityReport> {
107
+ const adapters: AdapterCapability[] = []
108
+ const runtime: CapabilityReport['runtime'] = {}
109
+
110
+ for (const adapter of ADAPTERS) {
111
+ const availability = await (adapter.probe?.() ?? Promise.resolve(adapter.availability()))
112
+ adapters.push({
113
+ id: adapter.id,
114
+ description: adapter.description,
115
+ kinds: adapter.kinds,
116
+ extensions: adapter.extensions,
117
+ availability,
118
+ })
119
+ for (const [name, version] of Object.entries(availability.versions ?? {})) {
120
+ runtime[name] = { available: true, version }
121
+ }
122
+ if (adapter.probe && !availability.available) {
123
+ runtime[adapter.id] = { available: false, error: availability.reason }
124
+ }
125
+ }
126
+
127
+ const anydoc = await anydocAvailability()
128
+ if (runtime['@firecrawl/anydoc'] === undefined) {
129
+ runtime['@firecrawl/anydoc'] = anydoc.available
130
+ ? { available: true, ...(anydoc.versions?.['@firecrawl/anydoc'] ? { version: anydoc.versions['@firecrawl/anydoc'] } : {}) }
131
+ : { available: false, error: anydoc.reason }
132
+ }
133
+
134
+ return {
135
+ adapters,
136
+ runtime,
137
+ defaults: {
138
+ document: anydoc.available ? 'anydoc' : '(不可用)',
139
+ text: 'text',
140
+ video: 'iddoc',
141
+ audio: 'iddoc',
142
+ image: 'iddoc',
143
+ scannedPdf: 'scanned',
144
+ },
145
+ extensions: ANYDOC_EXTENSIONS,
146
+ }
147
+ }
@@ -0,0 +1,139 @@
1
+ /**
2
+ * scanned —— 扫描件 / 纯图片页 PDF 的**轻量 OCR 档**(issue 0017 的扫描件线,轻量版)。
3
+ *
4
+ * 与 `anydoc` 的分工是明确的:
5
+ *
6
+ * - `.pdf` **默认仍走 anydoc**(文本层 PDF,零依赖、质量最好);
7
+ * - anydoc 报 `needsOcr`(整份是图片页)时,错误建议会指向本档;
8
+ * - 设 `SCANNED_OCR=1` 可让 `.pdf` 直接走本档(显式 opt-in,不做静默降级)。
9
+ *
10
+ * 本档**不做**版面还原:输出是"第 N 页 + 该页图片 + 该页文字"。版面顺序、表格结构、
11
+ * 公式属于 0017 的"可选高质量档"(docling / MinerU),不在这一档的承诺里。
12
+ *
13
+ * 能力来自同包 `src/media/` 的 `convertScannedDocument()`:
14
+ * pypdfium2 渲染页图(无系统 poppler 依赖)+ RapidOCR 本地识别,全程不联网。
15
+ *
16
+ * 环境变量:
17
+ * SCANNED_OCR=1 让 `.pdf` 直接走本档(默认 false:先让 anydoc 试文本层)
18
+ * SCANNED_MAX_PAGES 最多处理页数(默认 50)
19
+ * SCANNED_DPI 渲染 DPI(默认 150)
20
+ * SCANNED_NO_PAGE_IMAGES=1 只保留文字,不落地页图
21
+ * SCANNED_UVX / SCANNED_PYTHON 自定义 uvx 与 Python 版本
22
+ */
23
+
24
+ import { stat } from 'node:fs/promises'
25
+ import {
26
+ MediaToMarkdownError,
27
+ convertScannedDocument,
28
+ probeTools,
29
+ UVX_HINT,
30
+ type MediaAsset,
31
+ } from '../media/index.js'
32
+ import type { ConvertWarning } from '../contract.js'
33
+ import { AdapterError, type Adapter, type AdapterAvailability, type AdapterContext, type AdapterOutput, type PreparedAsset } from './types.js'
34
+
35
+ export const SCANNED_EXTENSIONS = ['.pdf']
36
+
37
+ function envFlag(name: string): boolean {
38
+ const value = (process.env[name] ?? '').trim().toLowerCase()
39
+ return value === '1' || value === 'true' || value === 'yes' || value === 'on'
40
+ }
41
+
42
+ function envInt(name: string): number | undefined {
43
+ const parsed = Number.parseInt((process.env[name] ?? '').trim(), 10)
44
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : undefined
45
+ }
46
+
47
+ /** 廉价同步检查:真正的依赖探测在 probe() 与首次使用时。 */
48
+ function availabilitySync(): AdapterAvailability {
49
+ return { available: true }
50
+ }
51
+
52
+ async function probeAvailability(): Promise<AdapterAvailability> {
53
+ const versions = await probeTools({ ...(process.env.SCANNED_UVX ? { uvxPath: process.env.SCANNED_UVX } : {}) })
54
+ if (!versions.uvx) {
55
+ return {
56
+ available: false,
57
+ reason: '扫描档需要 uvx(用于拉起 pypdfium2 渲染页图与 RapidOCR 识别文字)',
58
+ suggestion: UVX_HINT,
59
+ }
60
+ }
61
+ return { available: true, versions: { uvx: versions.uvx } }
62
+ }
63
+
64
+ function toPreparedAssets(assets: MediaAsset[]): PreparedAsset[] {
65
+ return assets.map(asset => ({
66
+ rel: asset.rel,
67
+ data: asset.data,
68
+ ...(asset.contentType ? { contentType: asset.contentType } : {}),
69
+ }))
70
+ }
71
+
72
+ async function run(context: AdapterContext): Promise<AdapterOutput> {
73
+ if (!context.inputPath) {
74
+ throw new AdapterError('INPUT_REQUIRED', 'scanned 只处理本地 PDF 文件(kind=file)。')
75
+ }
76
+ const file = context.inputPath
77
+ const fileInfo = await stat(file).catch(() => undefined)
78
+ if (!fileInfo?.isFile()) throw new AdapterError('INPUT_NOT_A_FILE', `输入不是文件:${file}`)
79
+
80
+ const uvxPath = process.env.SCANNED_UVX?.trim()
81
+ const tools = {
82
+ ...(uvxPath ? { uvxPath } : {}),
83
+ ...(process.env.SCANNED_PYTHON?.trim() ? { pythonVersion: process.env.SCANNED_PYTHON.trim() } : {}),
84
+ }
85
+
86
+ let result
87
+ try {
88
+ result = await convertScannedDocument({
89
+ file,
90
+ ...(context.request.title?.trim() ? { title: context.request.title.trim() } : {}),
91
+ ...(envInt('SCANNED_MAX_PAGES') ? { maxPages: envInt('SCANNED_MAX_PAGES') } : {}),
92
+ ...(envInt('SCANNED_DPI') ? { dpi: envInt('SCANNED_DPI') } : {}),
93
+ ...(envFlag('SCANNED_NO_PAGE_IMAGES') ? { pageImages: false } : {}),
94
+ ...(Object.keys(tools).length > 0 ? { tools } : {}),
95
+ ...(context.signal ? { signal: context.signal } : {}),
96
+ onProgress: progress => context.report(progress.progress, progress.stage, progress.message),
97
+ })
98
+ } catch (cause) {
99
+ if (cause instanceof MediaToMarkdownError) {
100
+ throw new AdapterError(cause.code, cause.message, cause.suggestion)
101
+ }
102
+ throw new AdapterError('ADAPTER_FAILED', cause instanceof Error ? cause.message : String(cause))
103
+ }
104
+
105
+ const warnings: ConvertWarning[] = result.warnings.map(warning => ({ code: warning.code, message: warning.message }))
106
+ const identified = result.pages?.filter(page => page.text).length ?? 0
107
+ warnings.push({
108
+ code: 'SCANNED_TIER',
109
+ message: `扫描档(轻量 OCR):共 ${result.pages?.length ?? 0} 页,${identified} 页识别到文字;不做版面还原。如版面/表格错乱,见 issues/0017 的可选高质量档。`,
110
+ })
111
+
112
+ const assets = toPreparedAssets(result.assets)
113
+ return {
114
+ title: result.title,
115
+ markdown: result.markdown,
116
+ sourceFormat: 'pdf',
117
+ parser: 'scanned@1/pdf+ocr',
118
+ sourcePath: file,
119
+ warnings,
120
+ dependencyVersions: result.dependencyVersions,
121
+ externalId: result.externalId,
122
+ ...(assets.length > 0 ? { assets } : {}),
123
+ }
124
+ }
125
+
126
+ export const scannedAdapter: Adapter = {
127
+ id: 'scanned',
128
+ description: '扫描件 / 纯图片页 PDF → 逐页本地 OCR Markdown(轻量档:不出网、不做版面还原;需 uvx)',
129
+ kinds: ['file'],
130
+ extensions: SCANNED_EXTENSIONS,
131
+ availability: availabilitySync,
132
+ probe: probeAvailability,
133
+ run,
134
+ }
135
+
136
+ /** 路由用:是否让 `.pdf` 直接走扫描档(显式 opt-in)。 */
137
+ export function scannedPreferred(): boolean {
138
+ return envFlag('SCANNED_OCR')
139
+ }
@@ -0,0 +1,146 @@
1
+ /**
2
+ * Text-family adapter: Markdown, plain text, JSON/TSV and friends.
3
+ *
4
+ * These inputs need no parsing, so the adapter's whole job is normalization:
5
+ * drop any pre-existing front matter (the service writes the authoritative
6
+ * block) and report the carrier format truthfully.
7
+ *
8
+ * `.csv` is deliberately absent: the document engine owns it, because it
9
+ * produces a real GFM table where this adapter would only pass bytes through.
10
+ */
11
+
12
+ import { readFile, stat } from 'node:fs/promises'
13
+ import { basename, extname } from 'node:path'
14
+ import type { ConvertWarning, SourceFormat } from '../contract.js'
15
+ import { parseMarkdownDocument, sha1, sha256 } from '../markdown.js'
16
+ import { AdapterError, type Adapter, type AdapterContext, type AdapterOutput, type PreparedAsset } from './types.js'
17
+
18
+ const INLINE_IMAGE_PATTERN = /!\[([^\]]*)\]\(\s*(?:<([^>]+)>|([^\s)]+))\s*\)/g
19
+
20
+ const MAGIC_EXTENSIONS: Array<{ test: (data: Buffer) => boolean; extension: string; contentType: string }> = [
21
+ { test: data => data.subarray(0, 8).equals(Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])), extension: 'png', contentType: 'image/png' },
22
+ { test: data => data[0] === 0xff && data[1] === 0xd8, extension: 'jpg', contentType: 'image/jpeg' },
23
+ { test: data => data.subarray(0, 3).toString('latin1') === 'GIF', extension: 'gif', contentType: 'image/gif' },
24
+ { test: data => data.subarray(0, 4).toString('latin1') === 'RIFF' && data.subarray(8, 12).toString('latin1') === 'WEBP', extension: 'webp', contentType: 'image/webp' },
25
+ ]
26
+
27
+ /**
28
+ * Turn `data:` image URIs into real `assets/` files.
29
+ *
30
+ * A data URI is not a network fetch — it is bytes already in hand — but it has
31
+ * to become a local file anyway, because retrieval skips `data:` exactly like it
32
+ * skips `http(s):`. Without this, a Markdown source with a pasted screenshot
33
+ * would upload "successfully" and be unsearchable.
34
+ */
35
+ export function materializeInlineImages(markdown: string): { markdown: string; assets: PreparedAsset[]; warnings: ConvertWarning[] } {
36
+ const pieces = []
37
+ const assets: PreparedAsset[] = []
38
+ const warnings: ConvertWarning[] = []
39
+ let cursor = 0
40
+ for (const match of markdown.matchAll(INLINE_IMAGE_PATTERN)) {
41
+ const index = match.index ?? 0
42
+ pieces.push(markdown.slice(cursor, index))
43
+ cursor = index + match[0].length
44
+ const alt = match[1] ?? ''
45
+ const url = (match[2] ?? match[3] ?? '').trim()
46
+ if (!url.startsWith('data:')) {
47
+ pieces.push(match[0])
48
+ continue
49
+ }
50
+ const parsed = /^data:([^;,]*)((?:;[^,]*)*),(.*)$/s.exec(url)
51
+ const data = parsed === null
52
+ ? Buffer.alloc(0)
53
+ : parsed[2]!.includes('base64')
54
+ ? Buffer.from(parsed[3]!.replace(/\s+/g, ''), 'base64')
55
+ : Buffer.from(decodeURIComponent(parsed[3]!), 'utf8')
56
+ const magic = data.length > 0 ? MAGIC_EXTENSIONS.find(entry => entry.test(data)) : undefined
57
+ if (!magic) {
58
+ warnings.push({ code: 'INLINE_IMAGE_UNSUPPORTED', message: '无法识别内联图片 data URI 的类型,已丢弃该引用。' })
59
+ continue
60
+ }
61
+ const digest = sha1(data)
62
+ const rel = `assets/img-${digest.slice(0, 12)}.${magic.extension}`
63
+ assets.push({ rel, data, contentType: magic.contentType })
64
+ pieces.push(`![${alt}](${rel})`)
65
+ warnings.push({ code: 'INLINE_IMAGE_MATERIALIZED', message: `内联图片已落地为 ${rel}。` })
66
+ }
67
+ pieces.push(markdown.slice(cursor))
68
+ return { markdown: pieces.join(''), assets, warnings }
69
+ }
70
+
71
+ export const TEXT_EXTENSIONS = [
72
+ '.md', '.markdown', '.mdx', '.txt', '.text', '.log', '.tsv',
73
+ '.json', '.jsonl', '.ndjson', '.yml', '.yaml', '.toml', '.ini',
74
+ '.adoc', '.rst', '.org', '.tex',
75
+ ]
76
+
77
+ export const MARKDOWN_EXTENSIONS = ['.md', '.markdown', '.mdx', '.adoc', '.rst', '.org']
78
+
79
+ function formatFor(filename: string, content: string): SourceFormat {
80
+ const extension = extname(filename).toLowerCase()
81
+ if (MARKDOWN_EXTENSIONS.includes(extension)) return 'markdown'
82
+ if (/^\s*#{1,6}\s+\S/m.test(content) && /\[[^\]]+\]\([^)]+\)/.test(content)) return 'markdown'
83
+ return 'text'
84
+ }
85
+
86
+ function stripFrontMatter(content: string): string {
87
+ return content.startsWith('---') ? parseMarkdownDocument(content).content : content.trim()
88
+ }
89
+
90
+ async function run(context: AdapterContext): Promise<AdapterOutput> {
91
+ let content: string
92
+ let filename: string
93
+ let sourcePath: string | undefined
94
+
95
+ if (context.inputPath) {
96
+ const info = await stat(context.inputPath)
97
+ if (!info.isFile()) throw new AdapterError('INPUT_NOT_A_FILE', `输入不是文件:${context.inputPath}`)
98
+ content = await readFile(context.inputPath, 'utf8')
99
+ filename = context.filename || basename(context.inputPath)
100
+ sourcePath = context.inputPath
101
+ } else {
102
+ content = context.text ?? ''
103
+ filename = context.filename || 'inline.txt'
104
+ }
105
+
106
+ if (!content.trim()) throw new AdapterError('CONTENT_REQUIRED', '输入内容为空,无法转换。')
107
+ const warnings: AdapterOutput['warnings'] = []
108
+ const format = formatFor(filename, content)
109
+
110
+ // Detect a real front matter block instead of comparing lengths: a trailing
111
+ // newline alone used to trigger this warning on ordinary Markdown.
112
+ const hadFrontMatter = format === 'markdown' && content.startsWith('---') && content.indexOf('\n---', 3) > 0
113
+ const stripped = format === 'markdown' ? stripFrontMatter(content) : content.trim()
114
+ if (hadFrontMatter) {
115
+ warnings.push({ code: 'FRONT_MATTER_REPLACED', message: '输入已带 front matter,已由转换内核重写为统一契约。' })
116
+ }
117
+ const materialized = materializeInlineImages(stripped)
118
+ const body = materialized.markdown
119
+ warnings.push(...materialized.warnings)
120
+
121
+ const title =
122
+ context.request.title?.trim() ||
123
+ (format === 'markdown' ? /^#{1,6}\s+(.+)$/m.exec(body)?.[1]?.trim() : undefined) ||
124
+ basename(filename, extname(filename))
125
+
126
+ return {
127
+ title: title.slice(0, 300),
128
+ markdown: body,
129
+ sourceFormat: format,
130
+ parser: `convert-core/passthrough@1/${format}`,
131
+ ...(sourcePath ? { sourcePath } : {}),
132
+ warnings,
133
+ dependencyVersions: {},
134
+ externalId: `sha256:${sha256(content)}`,
135
+ ...(materialized.assets.length > 0 ? { assets: materialized.assets } : {}),
136
+ }
137
+ }
138
+
139
+ export const textAdapter: Adapter = {
140
+ id: 'text',
141
+ description: 'Markdown / 纯文本 / JSON / TSV 直通(只做契约归一化,不改变内容)',
142
+ kinds: ['text', 'file'],
143
+ extensions: TEXT_EXTENSIONS,
144
+ availability: () => ({ available: true }),
145
+ run,
146
+ }
@@ -0,0 +1,76 @@
1
+ import type { ConvertRequest, ConvertWarning, SourceFormat } from '../contract.js'
2
+
3
+ /**
4
+ * An asset an adapter hands to the service as bytes. The service owns the
5
+ * delivery directory, so it computes the path, size and hash.
6
+ */
7
+ export interface PreparedAsset {
8
+ rel: string
9
+ data: Buffer
10
+ contentType?: string
11
+ }
12
+
13
+ export interface AdapterContext {
14
+ request: ConvertRequest
15
+ /** Staged local path for `kind=file`. */
16
+ inputPath?: string
17
+ /** Inline payload for `kind=text`. */
18
+ text?: string
19
+ filename: string
20
+ signal: AbortSignal
21
+ report: (progress: number | null, stage: string, message?: string) => void
22
+ }
23
+
24
+ export interface AdapterOutput {
25
+ title: string
26
+ /** Markdown body only — front matter is the service's job. */
27
+ markdown: string
28
+ sourceFormat: SourceFormat
29
+ parser: string
30
+ sourcePath?: string
31
+ warnings: ConvertWarning[]
32
+ dependencyVersions: Record<string, string>
33
+ /** Stable identity for idempotency (file hash). */
34
+ externalId?: string
35
+ assets?: PreparedAsset[]
36
+ }
37
+
38
+ export interface AdapterAvailability {
39
+ available: boolean
40
+ /** Populated when `available` is false. */
41
+ reason?: string
42
+ /** Actionable next step shown verbatim to the user. */
43
+ suggestion?: string
44
+ /** Versions of the pieces involved, when known. */
45
+ versions?: Record<string, string>
46
+ }
47
+
48
+ export interface Adapter {
49
+ id: string
50
+ /** Human description shown in the console and in `GET /api/capabilities`. */
51
+ description: string
52
+ /** Task kinds this adapter can serve. */
53
+ kinds: ConvertRequest['kind'][]
54
+ /** File extensions (lowercase, with dot) this adapter claims. */
55
+ extensions: string[]
56
+ /** Cheap synchronous check used for routing decisions. */
57
+ availability: () => AdapterAvailability
58
+ /**
59
+ * Authoritative asynchronous check used by `GET /api/capabilities`: it really
60
+ * loads the engine (and the platform binary) instead of assuming it works.
61
+ */
62
+ probe?: () => Promise<AdapterAvailability>
63
+ run: (context: AdapterContext) => Promise<AdapterOutput>
64
+ }
65
+
66
+ export class AdapterError extends Error {
67
+ readonly code: string
68
+ readonly suggestion?: string
69
+
70
+ constructor(code: string, message: string, suggestion?: string) {
71
+ super(message)
72
+ this.name = 'AdapterError'
73
+ this.code = code
74
+ this.suggestion = suggestion
75
+ }
76
+ }