dsh-convert-core 0.1.0-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +149 -0
- package/bin/media-to-md.mjs +168 -0
- package/bin/tita-convert.mjs +85 -0
- package/docs/media-to-md.md +143 -0
- package/lib/adapters/anydoc.d.ts +73 -0
- package/lib/adapters/anydoc.d.ts.map +1 -0
- package/lib/adapters/anydoc.js +295 -0
- package/lib/adapters/anydoc.js.map +1 -0
- package/lib/adapters/iddoc.d.ts +28 -0
- package/lib/adapters/iddoc.d.ts.map +1 -0
- package/lib/adapters/iddoc.js +153 -0
- package/lib/adapters/iddoc.js.map +1 -0
- package/lib/adapters/index.d.ts +41 -0
- package/lib/adapters/index.d.ts.map +1 -0
- package/lib/adapters/index.js +115 -0
- package/lib/adapters/index.js.map +1 -0
- package/lib/adapters/scanned.d.ts +28 -0
- package/lib/adapters/scanned.d.ts.map +1 -0
- package/lib/adapters/scanned.js +121 -0
- package/lib/adapters/scanned.js.map +1 -0
- package/lib/adapters/text.d.ts +29 -0
- package/lib/adapters/text.d.ts.map +1 -0
- package/lib/adapters/text.js +135 -0
- package/lib/adapters/text.js.map +1 -0
- package/lib/adapters/types.d.ts +65 -0
- package/lib/adapters/types.d.ts.map +1 -0
- package/lib/adapters/types.js +11 -0
- package/lib/adapters/types.js.map +1 -0
- package/lib/contract.d.ts +185 -0
- package/lib/contract.d.ts.map +1 -0
- package/lib/contract.js +59 -0
- package/lib/contract.js.map +1 -0
- package/lib/delivery.d.ts +36 -0
- package/lib/delivery.d.ts.map +1 -0
- package/lib/delivery.js +97 -0
- package/lib/delivery.js.map +1 -0
- package/lib/http.d.ts +34 -0
- package/lib/http.d.ts.map +1 -0
- package/lib/http.js +372 -0
- package/lib/http.js.map +1 -0
- package/lib/markdown.d.ts +32 -0
- package/lib/markdown.d.ts.map +1 -0
- package/lib/markdown.js +145 -0
- package/lib/markdown.js.map +1 -0
- package/lib/media/binaries.d.ts +26 -0
- package/lib/media/binaries.d.ts.map +1 -0
- package/lib/media/binaries.js +60 -0
- package/lib/media/binaries.js.map +1 -0
- package/lib/media/convert.d.ts +19 -0
- package/lib/media/convert.d.ts.map +1 -0
- package/lib/media/convert.js +210 -0
- package/lib/media/convert.js.map +1 -0
- package/lib/media/index.d.ts +27 -0
- package/lib/media/index.d.ts.map +1 -0
- package/lib/media/index.js +27 -0
- package/lib/media/index.js.map +1 -0
- package/lib/media/markdown.d.ts +40 -0
- package/lib/media/markdown.d.ts.map +1 -0
- package/lib/media/markdown.js +89 -0
- package/lib/media/markdown.js.map +1 -0
- package/lib/media/media.d.ts +45 -0
- package/lib/media/media.d.ts.map +1 -0
- package/lib/media/media.js +167 -0
- package/lib/media/media.js.map +1 -0
- package/lib/media/models.d.ts +31 -0
- package/lib/media/models.d.ts.map +1 -0
- package/lib/media/models.js +87 -0
- package/lib/media/models.js.map +1 -0
- package/lib/media/scanned.d.ts +18 -0
- package/lib/media/scanned.d.ts.map +1 -0
- package/lib/media/scanned.js +179 -0
- package/lib/media/scanned.js.map +1 -0
- package/lib/media/types.d.ts +154 -0
- package/lib/media/types.d.ts.map +1 -0
- package/lib/media/types.js +23 -0
- package/lib/media/types.js.map +1 -0
- package/lib/server.d.ts +35 -0
- package/lib/server.d.ts.map +1 -0
- package/lib/server.js +64 -0
- package/lib/server.js.map +1 -0
- package/lib/service.d.ts +170 -0
- package/lib/service.d.ts.map +1 -0
- package/lib/service.js +562 -0
- package/lib/service.js.map +1 -0
- package/package.json +59 -0
- package/scripts/asr.py +55 -0
- package/scripts/ocr.py +40 -0
- package/scripts/pdf_pages.py +49 -0
- package/scripts/vendor.mjs +76 -0
- package/src/adapters/anydoc.ts +356 -0
- package/src/adapters/iddoc.ts +171 -0
- package/src/adapters/index.ts +147 -0
- package/src/adapters/scanned.ts +139 -0
- package/src/adapters/text.ts +146 -0
- package/src/adapters/types.ts +76 -0
- package/src/contract.ts +230 -0
- package/src/delivery.ts +124 -0
- package/src/http.ts +394 -0
- package/src/markdown.ts +141 -0
- package/src/media/binaries.ts +67 -0
- package/src/media/convert.ts +232 -0
- package/src/media/index.ts +43 -0
- package/src/media/markdown.ts +105 -0
- package/src/media/media.ts +204 -0
- package/src/media/models.ts +124 -0
- package/src/media/scanned.ts +200 -0
- package/src/media/types.ts +171 -0
- package/src/server.ts +85 -0
- package/src/service.ts +639 -0
- package/web/app.js +382 -0
- package/web/index.html +217 -0
- package/web/styles.css +263 -0
- package/web/vendor/icons.js +25 -0
- package/web/vendor/vue.global.prod.js +14 -0
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* iddoc —— 音视频 / 图片 adapter(issue 0017 的 video 档)。
|
|
3
|
+
*
|
|
4
|
+
* 这个文件**只做契约映射**:能力全在同包的 `src/media/` 里(纯本地管线:
|
|
5
|
+
* ffmpeg 抽帧 + 可选的字幕/转写 + 可选的画面 OCR),这里负责
|
|
6
|
+
*
|
|
7
|
+
* 1. 把内核的 `AdapterContext` 翻译成该包的 options(**含 env 旋钮**);
|
|
8
|
+
* 2. 把 `MediaToMarkdownResult` 映射成 `AdapterOutput`(assets 交给内核写盘);
|
|
9
|
+
* 3. 把它抛出的错误原样转成 `AdapterError`,保留 `code` 与可操作建议。
|
|
10
|
+
*
|
|
11
|
+
* 与文档档(anydoc)一样:引擎独立于 adapter,内核只管契约。
|
|
12
|
+
*
|
|
13
|
+
* 环境变量(都是**显式**开关,默认不下载模型、不对图片做 OCR):
|
|
14
|
+
* IDDOC_ASR=1 允许在无字幕时下载模型并本地转写
|
|
15
|
+
* IDDOC_ASR_MODEL whisper 模型(默认 small;中文建议 large-v3-turbo)
|
|
16
|
+
* IDDOC_ASR_LANG 语言代码(默认自动检测;中文建议 zh)
|
|
17
|
+
* IDDOC_OCR_FRAMES=1 对关键帧做画面文字 OCR
|
|
18
|
+
* IDDOC_IMAGE_OCR=1 对图片输入做 OCR(默认关闭:图片直接收录)
|
|
19
|
+
* IDDOC_UVX / IDDOC_PYTHON 自定义 uvx 与 Python 版本
|
|
20
|
+
* IDDOC_SAMPLES / IDDOC_MAX_FRAMES / IDDOC_SCENE 抽帧参数
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import { stat } from 'node:fs/promises'
|
|
24
|
+
import {
|
|
25
|
+
IMAGE_EXTENSIONS,
|
|
26
|
+
MediaToMarkdownError,
|
|
27
|
+
convertMedia,
|
|
28
|
+
probeTools,
|
|
29
|
+
type MediaAsset,
|
|
30
|
+
type MediaToMarkdownOptions,
|
|
31
|
+
} from '../media/index.js'
|
|
32
|
+
import type { ConvertWarning } from '../contract.js'
|
|
33
|
+
import { AdapterError, type Adapter, type AdapterAvailability, type AdapterContext, type AdapterOutput, type PreparedAsset } from './types.js'
|
|
34
|
+
|
|
35
|
+
export const IDDOC_VIDEO_EXTENSIONS = [
|
|
36
|
+
'.mp4', '.m4v', '.mov', '.mkv', '.webm', '.avi', '.flv', '.wmv', '.mpg', '.mpeg', '.ogv',
|
|
37
|
+
]
|
|
38
|
+
export const IDDOC_AUDIO_EXTENSIONS = [
|
|
39
|
+
'.mp3', '.m4a', '.wav', '.aac', '.flac', '.ogg', '.opus', '.aiff', '.aif', '.amr', '.wma',
|
|
40
|
+
]
|
|
41
|
+
export const IDDOC_IMAGE_EXTENSIONS = IMAGE_EXTENSIONS
|
|
42
|
+
export const IDDOC_EXTENSIONS = [...IDDOC_VIDEO_EXTENSIONS, ...IDDOC_AUDIO_EXTENSIONS, ...IDDOC_IMAGE_EXTENSIONS]
|
|
43
|
+
|
|
44
|
+
const INSTALL_HINT = '装 ffmpeg(含 ffprobe)后重试:macOS `brew install ffmpeg`,Debian/Ubuntu `apt install ffmpeg`。'
|
|
45
|
+
|
|
46
|
+
function envFlag(name: string): boolean {
|
|
47
|
+
const value = (process.env[name] ?? '').trim().toLowerCase()
|
|
48
|
+
return value === '1' || value === 'true' || value === 'yes' || value === 'on'
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function envInt(name: string): number | undefined {
|
|
52
|
+
const parsed = Number.parseInt((process.env[name] ?? '').trim(), 10)
|
|
53
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : undefined
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
function envFloat(name: string): number | undefined {
|
|
57
|
+
const parsed = Number.parseFloat((process.env[name] ?? '').trim())
|
|
58
|
+
return Number.isFinite(parsed) && parsed >= 0 && parsed <= 1 ? parsed : undefined
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** 路由用的廉价同步检查;真正的探测在 probe() 与首次使用时。 */
|
|
62
|
+
function availabilitySync(): AdapterAvailability {
|
|
63
|
+
return { available: true }
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
async function probeAvailability(): Promise<AdapterAvailability> {
|
|
67
|
+
const versions = await probeTools({ ...(process.env.IDDOC_UVX ? { uvxPath: process.env.IDDOC_UVX } : {}) })
|
|
68
|
+
if (!versions.ffprobe || !versions.ffmpeg) {
|
|
69
|
+
const missing = [!versions.ffprobe ? 'ffprobe' : null, !versions.ffmpeg ? 'ffmpeg' : null].filter(Boolean).join('、')
|
|
70
|
+
return { available: false, reason: `缺少外部二进制:${missing}`, suggestion: INSTALL_HINT }
|
|
71
|
+
}
|
|
72
|
+
const reported: Record<string, string> = {}
|
|
73
|
+
if (versions.ffmpeg) reported.ffmpeg = versions.ffmpeg
|
|
74
|
+
if (versions.ffprobe) reported.ffprobe = versions.ffprobe
|
|
75
|
+
if (versions.uvx) reported.uvx = versions.uvx
|
|
76
|
+
return { available: true, versions: reported }
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function toOptions(context: AdapterContext, file: string): MediaToMarkdownOptions {
|
|
80
|
+
const model = (process.env.IDDOC_ASR_MODEL ?? '').trim() || 'small'
|
|
81
|
+
const language = (process.env.IDDOC_ASR_LANG ?? '').trim()
|
|
82
|
+
const frames = {
|
|
83
|
+
...(envInt('IDDOC_SAMPLES') ? { samples: envInt('IDDOC_SAMPLES') } : {}),
|
|
84
|
+
...(envInt('IDDOC_MAX_FRAMES') ? { maxFrames: envInt('IDDOC_MAX_FRAMES') } : {}),
|
|
85
|
+
...(envFloat('IDDOC_SCENE') !== undefined ? { sceneThreshold: envFloat('IDDOC_SCENE') } : {}),
|
|
86
|
+
}
|
|
87
|
+
const tools = {
|
|
88
|
+
...(process.env.IDDOC_UVX?.trim() ? { uvxPath: process.env.IDDOC_UVX.trim() } : {}),
|
|
89
|
+
...(process.env.IDDOC_PYTHON?.trim() ? { pythonVersion: process.env.IDDOC_PYTHON.trim() } : {}),
|
|
90
|
+
}
|
|
91
|
+
return {
|
|
92
|
+
file,
|
|
93
|
+
...(context.request.title?.trim() ? { title: context.request.title.trim() } : {}),
|
|
94
|
+
mode: 'auto',
|
|
95
|
+
asr: {
|
|
96
|
+
model,
|
|
97
|
+
...(language ? { language } : {}),
|
|
98
|
+
allowModelDownload: envFlag('IDDOC_ASR'),
|
|
99
|
+
},
|
|
100
|
+
...(Object.keys(frames).length > 0 ? { frames } : {}),
|
|
101
|
+
...(Object.keys(tools).length > 0 ? { tools } : {}),
|
|
102
|
+
...(envFlag('IDDOC_IMAGE_OCR') ? { imageOcr: true } : {}),
|
|
103
|
+
...(envFlag('IDDOC_OCR_FRAMES') ? { ocrFrames: true } : {}),
|
|
104
|
+
...(context.signal ? { signal: context.signal } : {}),
|
|
105
|
+
onProgress: progress => context.report(progress.progress, progress.stage, progress.message),
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function toPreparedAssets(assets: MediaAsset[]): PreparedAsset[] {
|
|
110
|
+
return assets.map(asset => ({
|
|
111
|
+
rel: asset.rel,
|
|
112
|
+
data: asset.data,
|
|
113
|
+
...(asset.contentType ? { contentType: asset.contentType } : {}),
|
|
114
|
+
}))
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
async function run(context: AdapterContext): Promise<AdapterOutput> {
|
|
118
|
+
if (!context.inputPath) {
|
|
119
|
+
throw new AdapterError('INPUT_REQUIRED', 'iddoc 只处理本地文件(kind=file):音频、视频或图片。')
|
|
120
|
+
}
|
|
121
|
+
const file = context.inputPath
|
|
122
|
+
const fileInfo = await stat(file).catch(() => undefined)
|
|
123
|
+
if (!fileInfo?.isFile()) throw new AdapterError('INPUT_NOT_A_FILE', `输入不是文件:${file}`)
|
|
124
|
+
|
|
125
|
+
const versions = await probeTools({ ...(process.env.IDDOC_UVX ? { uvxPath: process.env.IDDOC_UVX } : {}) })
|
|
126
|
+
if (!versions.ffprobe || !versions.ffmpeg) {
|
|
127
|
+
throw new AdapterError('DEPENDENCY_MISSING', 'iddoc 需要 ffmpeg 与 ffprobe。', INSTALL_HINT)
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
let result
|
|
131
|
+
try {
|
|
132
|
+
result = await convertMedia(toOptions(context, file))
|
|
133
|
+
} catch (cause) {
|
|
134
|
+
if (cause instanceof MediaToMarkdownError) {
|
|
135
|
+
throw new AdapterError(cause.code, cause.message, cause.suggestion)
|
|
136
|
+
}
|
|
137
|
+
throw new AdapterError('ADAPTER_FAILED', cause instanceof Error ? cause.message : String(cause))
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
const warnings: ConvertWarning[] = result.warnings.map(warning => ({ code: warning.code, message: warning.message }))
|
|
141
|
+
const suffix = result.meta.transcriptSource === 'subtitle'
|
|
142
|
+
? '+subtitle'
|
|
143
|
+
: result.meta.transcriptSource === 'transcription' ? '+asr' : ''
|
|
144
|
+
// 文档类按格式命名(pdf/office),媒体类按容器命名(mp4/png),与既有 parser 约定保持一致。
|
|
145
|
+
const parserKind = result.meta.sourceFormat === 'image'
|
|
146
|
+
? (result.meta.containerExtension ?? 'image')
|
|
147
|
+
: `${result.meta.sourceFormat}${suffix}`
|
|
148
|
+
const assets = toPreparedAssets(result.assets)
|
|
149
|
+
|
|
150
|
+
return {
|
|
151
|
+
title: result.title,
|
|
152
|
+
markdown: result.markdown,
|
|
153
|
+
sourceFormat: result.meta.sourceFormat,
|
|
154
|
+
parser: `iddoc@1/${parserKind}`,
|
|
155
|
+
sourcePath: file,
|
|
156
|
+
warnings,
|
|
157
|
+
dependencyVersions: result.dependencyVersions,
|
|
158
|
+
externalId: result.externalId,
|
|
159
|
+
...(assets.length > 0 ? { assets } : {}),
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
export const iddocAdapter: Adapter = {
|
|
164
|
+
id: 'iddoc',
|
|
165
|
+
description: '本地音视频 / 图片 → Markdown 与本地图片(能力来自内核自带的 src/media/:字幕优先;转写需显式开启;图片直接收录)',
|
|
166
|
+
kinds: ['file'],
|
|
167
|
+
extensions: IDDOC_EXTENSIONS,
|
|
168
|
+
availability: availabilitySync,
|
|
169
|
+
probe: probeAvailability,
|
|
170
|
+
run,
|
|
171
|
+
}
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Adapter registry.
|
|
3
|
+
*
|
|
4
|
+
* Resolution order stays trivial and explicit: an explicit override, then the
|
|
5
|
+
* media adapter (`iddoc`, by extension), then the document engine, then the text
|
|
6
|
+
* pass-through. Unknown extensions are only treated as text when the bytes are
|
|
7
|
+
* actually textual — guessing there is how "converted successfully" becomes a lie.
|
|
8
|
+
*
|
|
9
|
+
* `.pdf` 默认仍走 anydoc(文本层最好、零依赖);只有显式 `SCANNED_OCR=1` 才改走
|
|
10
|
+
* `scanned`(纯图片页的轻量 OCR 档)。
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { open } from 'node:fs/promises'
|
|
14
|
+
import { extname } from 'node:path'
|
|
15
|
+
import type { ConvertRequest } from '../contract.js'
|
|
16
|
+
import { ANYDOC_EXTENSIONS, anydocAdapter, anydocAvailability } from './anydoc.js'
|
|
17
|
+
import { IDDOC_EXTENSIONS, iddocAdapter } from './iddoc.js'
|
|
18
|
+
import { SCANNED_EXTENSIONS, scannedAdapter, scannedPreferred } from './scanned.js'
|
|
19
|
+
import { textAdapter, TEXT_EXTENSIONS } from './text.js'
|
|
20
|
+
import { AdapterError, type Adapter, type AdapterAvailability } from './types.js'
|
|
21
|
+
|
|
22
|
+
export const ADAPTERS: Adapter[] = [anydocAdapter, iddocAdapter, scannedAdapter, textAdapter]
|
|
23
|
+
|
|
24
|
+
export function findAdapter(id: string): Adapter | undefined {
|
|
25
|
+
return ADAPTERS.find(adapter => adapter.id === id)
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** Cheap binary sniff: NUL bytes or invalid UTF-8 in the head means "not text". */
|
|
29
|
+
export async function looksTextual(path: string): Promise<boolean> {
|
|
30
|
+
let handle
|
|
31
|
+
try {
|
|
32
|
+
handle = await open(path, 'r')
|
|
33
|
+
} catch {
|
|
34
|
+
return false
|
|
35
|
+
}
|
|
36
|
+
try {
|
|
37
|
+
const buffer = Buffer.alloc(4096)
|
|
38
|
+
const { bytesRead } = await handle.read(buffer, 0, buffer.length, 0)
|
|
39
|
+
if (bytesRead === 0) return true
|
|
40
|
+
const head = buffer.subarray(0, bytesRead)
|
|
41
|
+
if (head.includes(0)) return false
|
|
42
|
+
return !head.toString('utf8').includes('\uFFFD')
|
|
43
|
+
} finally {
|
|
44
|
+
await handle.close().catch(() => undefined)
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export const SUPPORTED_EXTENSIONS = [...ANYDOC_EXTENSIONS, ...IDDOC_EXTENSIONS, ...TEXT_EXTENSIONS]
|
|
49
|
+
|
|
50
|
+
export async function resolveAdapter(
|
|
51
|
+
request: ConvertRequest,
|
|
52
|
+
filename: string,
|
|
53
|
+
inputPath: string | undefined,
|
|
54
|
+
): Promise<Adapter> {
|
|
55
|
+
if (request.adapter) {
|
|
56
|
+
const explicit = findAdapter(request.adapter)
|
|
57
|
+
if (!explicit) {
|
|
58
|
+
throw new AdapterError(
|
|
59
|
+
'ADAPTER_UNKNOWN',
|
|
60
|
+
`未知的 adapter:${request.adapter}`,
|
|
61
|
+
`可用 adapter:${ADAPTERS.map(adapter => adapter.id).join(', ')}`,
|
|
62
|
+
)
|
|
63
|
+
}
|
|
64
|
+
const availability = await (explicit.probe?.() ?? Promise.resolve(explicit.availability()))
|
|
65
|
+
if (!availability.available) {
|
|
66
|
+
throw new AdapterError('DEPENDENCY_MISSING', availability.reason ?? `${explicit.id} 当前不可用`, availability.suggestion)
|
|
67
|
+
}
|
|
68
|
+
return explicit
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
if (request.kind === 'text') return textAdapter
|
|
72
|
+
|
|
73
|
+
const extension = extname(filename).toLowerCase()
|
|
74
|
+
// 扫描档是显式 opt-in:默认仍让 anydoc 试文本层,失败了错误建议再指向 scanned。
|
|
75
|
+
if (SCANNED_EXTENSIONS.includes(extension) && scannedPreferred()) return scannedAdapter
|
|
76
|
+
if (ANYDOC_EXTENSIONS.includes(extension)) return anydocAdapter
|
|
77
|
+
if (IDDOC_EXTENSIONS.includes(extension)) return iddocAdapter
|
|
78
|
+
if (TEXT_EXTENSIONS.includes(extension)) return textAdapter
|
|
79
|
+
if (inputPath && (await looksTextual(inputPath))) return textAdapter
|
|
80
|
+
|
|
81
|
+
throw new AdapterError(
|
|
82
|
+
'FORMAT_UNSUPPORTED',
|
|
83
|
+
`不支持的输入格式:${extension || '(无扩展名)'}`,
|
|
84
|
+
`本版做文件类:${ANYDOC_EXTENSIONS.join(' ')}(文档)、${SCANNED_EXTENSIONS.join(' ')}(扫描件 OCR 档)、${IDDOC_EXTENSIONS.join(' ')}(音视频/图片)以及 ${TEXT_EXTENSIONS.join(' ')}(直通);网页链接仍在 issues/0017。`,
|
|
85
|
+
)
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
export interface AdapterCapability {
|
|
89
|
+
id: string
|
|
90
|
+
description: string
|
|
91
|
+
kinds: ConvertRequest['kind'][]
|
|
92
|
+
extensions: string[]
|
|
93
|
+
availability: AdapterAvailability
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export interface CapabilityReport {
|
|
97
|
+
adapters: AdapterCapability[]
|
|
98
|
+
/** Engine-level runtime facts; documents need no external binary, iddoc needs ffmpeg. */
|
|
99
|
+
runtime: Record<string, { available: boolean; version?: string; error?: string }>
|
|
100
|
+
defaults: Record<string, string>
|
|
101
|
+
/** Extensions the document engine covers. */
|
|
102
|
+
extensions: string[]
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** Everything the console and the DSH settings panel need to tell the truth. */
|
|
106
|
+
export async function capabilityReport(): Promise<CapabilityReport> {
|
|
107
|
+
const adapters: AdapterCapability[] = []
|
|
108
|
+
const runtime: CapabilityReport['runtime'] = {}
|
|
109
|
+
|
|
110
|
+
for (const adapter of ADAPTERS) {
|
|
111
|
+
const availability = await (adapter.probe?.() ?? Promise.resolve(adapter.availability()))
|
|
112
|
+
adapters.push({
|
|
113
|
+
id: adapter.id,
|
|
114
|
+
description: adapter.description,
|
|
115
|
+
kinds: adapter.kinds,
|
|
116
|
+
extensions: adapter.extensions,
|
|
117
|
+
availability,
|
|
118
|
+
})
|
|
119
|
+
for (const [name, version] of Object.entries(availability.versions ?? {})) {
|
|
120
|
+
runtime[name] = { available: true, version }
|
|
121
|
+
}
|
|
122
|
+
if (adapter.probe && !availability.available) {
|
|
123
|
+
runtime[adapter.id] = { available: false, error: availability.reason }
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
const anydoc = await anydocAvailability()
|
|
128
|
+
if (runtime['@firecrawl/anydoc'] === undefined) {
|
|
129
|
+
runtime['@firecrawl/anydoc'] = anydoc.available
|
|
130
|
+
? { available: true, ...(anydoc.versions?.['@firecrawl/anydoc'] ? { version: anydoc.versions['@firecrawl/anydoc'] } : {}) }
|
|
131
|
+
: { available: false, error: anydoc.reason }
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
return {
|
|
135
|
+
adapters,
|
|
136
|
+
runtime,
|
|
137
|
+
defaults: {
|
|
138
|
+
document: anydoc.available ? 'anydoc' : '(不可用)',
|
|
139
|
+
text: 'text',
|
|
140
|
+
video: 'iddoc',
|
|
141
|
+
audio: 'iddoc',
|
|
142
|
+
image: 'iddoc',
|
|
143
|
+
scannedPdf: 'scanned',
|
|
144
|
+
},
|
|
145
|
+
extensions: ANYDOC_EXTENSIONS,
|
|
146
|
+
}
|
|
147
|
+
}
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* scanned —— 扫描件 / 纯图片页 PDF 的**轻量 OCR 档**(issue 0017 的扫描件线,轻量版)。
|
|
3
|
+
*
|
|
4
|
+
* 与 `anydoc` 的分工是明确的:
|
|
5
|
+
*
|
|
6
|
+
* - `.pdf` **默认仍走 anydoc**(文本层 PDF,零依赖、质量最好);
|
|
7
|
+
* - anydoc 报 `needsOcr`(整份是图片页)时,错误建议会指向本档;
|
|
8
|
+
* - 设 `SCANNED_OCR=1` 可让 `.pdf` 直接走本档(显式 opt-in,不做静默降级)。
|
|
9
|
+
*
|
|
10
|
+
* 本档**不做**版面还原:输出是"第 N 页 + 该页图片 + 该页文字"。版面顺序、表格结构、
|
|
11
|
+
* 公式属于 0017 的"可选高质量档"(docling / MinerU),不在这一档的承诺里。
|
|
12
|
+
*
|
|
13
|
+
* 能力来自同包 `src/media/` 的 `convertScannedDocument()`:
|
|
14
|
+
* pypdfium2 渲染页图(无系统 poppler 依赖)+ RapidOCR 本地识别,全程不联网。
|
|
15
|
+
*
|
|
16
|
+
* 环境变量:
|
|
17
|
+
* SCANNED_OCR=1 让 `.pdf` 直接走本档(默认 false:先让 anydoc 试文本层)
|
|
18
|
+
* SCANNED_MAX_PAGES 最多处理页数(默认 50)
|
|
19
|
+
* SCANNED_DPI 渲染 DPI(默认 150)
|
|
20
|
+
* SCANNED_NO_PAGE_IMAGES=1 只保留文字,不落地页图
|
|
21
|
+
* SCANNED_UVX / SCANNED_PYTHON 自定义 uvx 与 Python 版本
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { stat } from 'node:fs/promises'
|
|
25
|
+
import {
|
|
26
|
+
MediaToMarkdownError,
|
|
27
|
+
convertScannedDocument,
|
|
28
|
+
probeTools,
|
|
29
|
+
UVX_HINT,
|
|
30
|
+
type MediaAsset,
|
|
31
|
+
} from '../media/index.js'
|
|
32
|
+
import type { ConvertWarning } from '../contract.js'
|
|
33
|
+
import { AdapterError, type Adapter, type AdapterAvailability, type AdapterContext, type AdapterOutput, type PreparedAsset } from './types.js'
|
|
34
|
+
|
|
35
|
+
export const SCANNED_EXTENSIONS = ['.pdf']
|
|
36
|
+
|
|
37
|
+
function envFlag(name: string): boolean {
|
|
38
|
+
const value = (process.env[name] ?? '').trim().toLowerCase()
|
|
39
|
+
return value === '1' || value === 'true' || value === 'yes' || value === 'on'
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function envInt(name: string): number | undefined {
|
|
43
|
+
const parsed = Number.parseInt((process.env[name] ?? '').trim(), 10)
|
|
44
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : undefined
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** 廉价同步检查:真正的依赖探测在 probe() 与首次使用时。 */
|
|
48
|
+
function availabilitySync(): AdapterAvailability {
|
|
49
|
+
return { available: true }
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
async function probeAvailability(): Promise<AdapterAvailability> {
|
|
53
|
+
const versions = await probeTools({ ...(process.env.SCANNED_UVX ? { uvxPath: process.env.SCANNED_UVX } : {}) })
|
|
54
|
+
if (!versions.uvx) {
|
|
55
|
+
return {
|
|
56
|
+
available: false,
|
|
57
|
+
reason: '扫描档需要 uvx(用于拉起 pypdfium2 渲染页图与 RapidOCR 识别文字)',
|
|
58
|
+
suggestion: UVX_HINT,
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
return { available: true, versions: { uvx: versions.uvx } }
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
function toPreparedAssets(assets: MediaAsset[]): PreparedAsset[] {
|
|
65
|
+
return assets.map(asset => ({
|
|
66
|
+
rel: asset.rel,
|
|
67
|
+
data: asset.data,
|
|
68
|
+
...(asset.contentType ? { contentType: asset.contentType } : {}),
|
|
69
|
+
}))
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
async function run(context: AdapterContext): Promise<AdapterOutput> {
|
|
73
|
+
if (!context.inputPath) {
|
|
74
|
+
throw new AdapterError('INPUT_REQUIRED', 'scanned 只处理本地 PDF 文件(kind=file)。')
|
|
75
|
+
}
|
|
76
|
+
const file = context.inputPath
|
|
77
|
+
const fileInfo = await stat(file).catch(() => undefined)
|
|
78
|
+
if (!fileInfo?.isFile()) throw new AdapterError('INPUT_NOT_A_FILE', `输入不是文件:${file}`)
|
|
79
|
+
|
|
80
|
+
const uvxPath = process.env.SCANNED_UVX?.trim()
|
|
81
|
+
const tools = {
|
|
82
|
+
...(uvxPath ? { uvxPath } : {}),
|
|
83
|
+
...(process.env.SCANNED_PYTHON?.trim() ? { pythonVersion: process.env.SCANNED_PYTHON.trim() } : {}),
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
let result
|
|
87
|
+
try {
|
|
88
|
+
result = await convertScannedDocument({
|
|
89
|
+
file,
|
|
90
|
+
...(context.request.title?.trim() ? { title: context.request.title.trim() } : {}),
|
|
91
|
+
...(envInt('SCANNED_MAX_PAGES') ? { maxPages: envInt('SCANNED_MAX_PAGES') } : {}),
|
|
92
|
+
...(envInt('SCANNED_DPI') ? { dpi: envInt('SCANNED_DPI') } : {}),
|
|
93
|
+
...(envFlag('SCANNED_NO_PAGE_IMAGES') ? { pageImages: false } : {}),
|
|
94
|
+
...(Object.keys(tools).length > 0 ? { tools } : {}),
|
|
95
|
+
...(context.signal ? { signal: context.signal } : {}),
|
|
96
|
+
onProgress: progress => context.report(progress.progress, progress.stage, progress.message),
|
|
97
|
+
})
|
|
98
|
+
} catch (cause) {
|
|
99
|
+
if (cause instanceof MediaToMarkdownError) {
|
|
100
|
+
throw new AdapterError(cause.code, cause.message, cause.suggestion)
|
|
101
|
+
}
|
|
102
|
+
throw new AdapterError('ADAPTER_FAILED', cause instanceof Error ? cause.message : String(cause))
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
const warnings: ConvertWarning[] = result.warnings.map(warning => ({ code: warning.code, message: warning.message }))
|
|
106
|
+
const identified = result.pages?.filter(page => page.text).length ?? 0
|
|
107
|
+
warnings.push({
|
|
108
|
+
code: 'SCANNED_TIER',
|
|
109
|
+
message: `扫描档(轻量 OCR):共 ${result.pages?.length ?? 0} 页,${identified} 页识别到文字;不做版面还原。如版面/表格错乱,见 issues/0017 的可选高质量档。`,
|
|
110
|
+
})
|
|
111
|
+
|
|
112
|
+
const assets = toPreparedAssets(result.assets)
|
|
113
|
+
return {
|
|
114
|
+
title: result.title,
|
|
115
|
+
markdown: result.markdown,
|
|
116
|
+
sourceFormat: 'pdf',
|
|
117
|
+
parser: 'scanned@1/pdf+ocr',
|
|
118
|
+
sourcePath: file,
|
|
119
|
+
warnings,
|
|
120
|
+
dependencyVersions: result.dependencyVersions,
|
|
121
|
+
externalId: result.externalId,
|
|
122
|
+
...(assets.length > 0 ? { assets } : {}),
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
export const scannedAdapter: Adapter = {
|
|
127
|
+
id: 'scanned',
|
|
128
|
+
description: '扫描件 / 纯图片页 PDF → 逐页本地 OCR Markdown(轻量档:不出网、不做版面还原;需 uvx)',
|
|
129
|
+
kinds: ['file'],
|
|
130
|
+
extensions: SCANNED_EXTENSIONS,
|
|
131
|
+
availability: availabilitySync,
|
|
132
|
+
probe: probeAvailability,
|
|
133
|
+
run,
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/** 路由用:是否让 `.pdf` 直接走扫描档(显式 opt-in)。 */
|
|
137
|
+
export function scannedPreferred(): boolean {
|
|
138
|
+
return envFlag('SCANNED_OCR')
|
|
139
|
+
}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Text-family adapter: Markdown, plain text, JSON/TSV and friends.
|
|
3
|
+
*
|
|
4
|
+
* These inputs need no parsing, so the adapter's whole job is normalization:
|
|
5
|
+
* drop any pre-existing front matter (the service writes the authoritative
|
|
6
|
+
* block) and report the carrier format truthfully.
|
|
7
|
+
*
|
|
8
|
+
* `.csv` is deliberately absent: the document engine owns it, because it
|
|
9
|
+
* produces a real GFM table where this adapter would only pass bytes through.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { readFile, stat } from 'node:fs/promises'
|
|
13
|
+
import { basename, extname } from 'node:path'
|
|
14
|
+
import type { ConvertWarning, SourceFormat } from '../contract.js'
|
|
15
|
+
import { parseMarkdownDocument, sha1, sha256 } from '../markdown.js'
|
|
16
|
+
import { AdapterError, type Adapter, type AdapterContext, type AdapterOutput, type PreparedAsset } from './types.js'
|
|
17
|
+
|
|
18
|
+
const INLINE_IMAGE_PATTERN = /!\[([^\]]*)\]\(\s*(?:<([^>]+)>|([^\s)]+))\s*\)/g
|
|
19
|
+
|
|
20
|
+
const MAGIC_EXTENSIONS: Array<{ test: (data: Buffer) => boolean; extension: string; contentType: string }> = [
|
|
21
|
+
{ test: data => data.subarray(0, 8).equals(Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])), extension: 'png', contentType: 'image/png' },
|
|
22
|
+
{ test: data => data[0] === 0xff && data[1] === 0xd8, extension: 'jpg', contentType: 'image/jpeg' },
|
|
23
|
+
{ test: data => data.subarray(0, 3).toString('latin1') === 'GIF', extension: 'gif', contentType: 'image/gif' },
|
|
24
|
+
{ test: data => data.subarray(0, 4).toString('latin1') === 'RIFF' && data.subarray(8, 12).toString('latin1') === 'WEBP', extension: 'webp', contentType: 'image/webp' },
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Turn `data:` image URIs into real `assets/` files.
|
|
29
|
+
*
|
|
30
|
+
* A data URI is not a network fetch — it is bytes already in hand — but it has
|
|
31
|
+
* to become a local file anyway, because retrieval skips `data:` exactly like it
|
|
32
|
+
* skips `http(s):`. Without this, a Markdown source with a pasted screenshot
|
|
33
|
+
* would upload "successfully" and be unsearchable.
|
|
34
|
+
*/
|
|
35
|
+
export function materializeInlineImages(markdown: string): { markdown: string; assets: PreparedAsset[]; warnings: ConvertWarning[] } {
|
|
36
|
+
const pieces = []
|
|
37
|
+
const assets: PreparedAsset[] = []
|
|
38
|
+
const warnings: ConvertWarning[] = []
|
|
39
|
+
let cursor = 0
|
|
40
|
+
for (const match of markdown.matchAll(INLINE_IMAGE_PATTERN)) {
|
|
41
|
+
const index = match.index ?? 0
|
|
42
|
+
pieces.push(markdown.slice(cursor, index))
|
|
43
|
+
cursor = index + match[0].length
|
|
44
|
+
const alt = match[1] ?? ''
|
|
45
|
+
const url = (match[2] ?? match[3] ?? '').trim()
|
|
46
|
+
if (!url.startsWith('data:')) {
|
|
47
|
+
pieces.push(match[0])
|
|
48
|
+
continue
|
|
49
|
+
}
|
|
50
|
+
const parsed = /^data:([^;,]*)((?:;[^,]*)*),(.*)$/s.exec(url)
|
|
51
|
+
const data = parsed === null
|
|
52
|
+
? Buffer.alloc(0)
|
|
53
|
+
: parsed[2]!.includes('base64')
|
|
54
|
+
? Buffer.from(parsed[3]!.replace(/\s+/g, ''), 'base64')
|
|
55
|
+
: Buffer.from(decodeURIComponent(parsed[3]!), 'utf8')
|
|
56
|
+
const magic = data.length > 0 ? MAGIC_EXTENSIONS.find(entry => entry.test(data)) : undefined
|
|
57
|
+
if (!magic) {
|
|
58
|
+
warnings.push({ code: 'INLINE_IMAGE_UNSUPPORTED', message: '无法识别内联图片 data URI 的类型,已丢弃该引用。' })
|
|
59
|
+
continue
|
|
60
|
+
}
|
|
61
|
+
const digest = sha1(data)
|
|
62
|
+
const rel = `assets/img-${digest.slice(0, 12)}.${magic.extension}`
|
|
63
|
+
assets.push({ rel, data, contentType: magic.contentType })
|
|
64
|
+
pieces.push(``)
|
|
65
|
+
warnings.push({ code: 'INLINE_IMAGE_MATERIALIZED', message: `内联图片已落地为 ${rel}。` })
|
|
66
|
+
}
|
|
67
|
+
pieces.push(markdown.slice(cursor))
|
|
68
|
+
return { markdown: pieces.join(''), assets, warnings }
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export const TEXT_EXTENSIONS = [
|
|
72
|
+
'.md', '.markdown', '.mdx', '.txt', '.text', '.log', '.tsv',
|
|
73
|
+
'.json', '.jsonl', '.ndjson', '.yml', '.yaml', '.toml', '.ini',
|
|
74
|
+
'.adoc', '.rst', '.org', '.tex',
|
|
75
|
+
]
|
|
76
|
+
|
|
77
|
+
export const MARKDOWN_EXTENSIONS = ['.md', '.markdown', '.mdx', '.adoc', '.rst', '.org']
|
|
78
|
+
|
|
79
|
+
function formatFor(filename: string, content: string): SourceFormat {
|
|
80
|
+
const extension = extname(filename).toLowerCase()
|
|
81
|
+
if (MARKDOWN_EXTENSIONS.includes(extension)) return 'markdown'
|
|
82
|
+
if (/^\s*#{1,6}\s+\S/m.test(content) && /\[[^\]]+\]\([^)]+\)/.test(content)) return 'markdown'
|
|
83
|
+
return 'text'
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function stripFrontMatter(content: string): string {
|
|
87
|
+
return content.startsWith('---') ? parseMarkdownDocument(content).content : content.trim()
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
async function run(context: AdapterContext): Promise<AdapterOutput> {
|
|
91
|
+
let content: string
|
|
92
|
+
let filename: string
|
|
93
|
+
let sourcePath: string | undefined
|
|
94
|
+
|
|
95
|
+
if (context.inputPath) {
|
|
96
|
+
const info = await stat(context.inputPath)
|
|
97
|
+
if (!info.isFile()) throw new AdapterError('INPUT_NOT_A_FILE', `输入不是文件:${context.inputPath}`)
|
|
98
|
+
content = await readFile(context.inputPath, 'utf8')
|
|
99
|
+
filename = context.filename || basename(context.inputPath)
|
|
100
|
+
sourcePath = context.inputPath
|
|
101
|
+
} else {
|
|
102
|
+
content = context.text ?? ''
|
|
103
|
+
filename = context.filename || 'inline.txt'
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
if (!content.trim()) throw new AdapterError('CONTENT_REQUIRED', '输入内容为空,无法转换。')
|
|
107
|
+
const warnings: AdapterOutput['warnings'] = []
|
|
108
|
+
const format = formatFor(filename, content)
|
|
109
|
+
|
|
110
|
+
// Detect a real front matter block instead of comparing lengths: a trailing
|
|
111
|
+
// newline alone used to trigger this warning on ordinary Markdown.
|
|
112
|
+
const hadFrontMatter = format === 'markdown' && content.startsWith('---') && content.indexOf('\n---', 3) > 0
|
|
113
|
+
const stripped = format === 'markdown' ? stripFrontMatter(content) : content.trim()
|
|
114
|
+
if (hadFrontMatter) {
|
|
115
|
+
warnings.push({ code: 'FRONT_MATTER_REPLACED', message: '输入已带 front matter,已由转换内核重写为统一契约。' })
|
|
116
|
+
}
|
|
117
|
+
const materialized = materializeInlineImages(stripped)
|
|
118
|
+
const body = materialized.markdown
|
|
119
|
+
warnings.push(...materialized.warnings)
|
|
120
|
+
|
|
121
|
+
const title =
|
|
122
|
+
context.request.title?.trim() ||
|
|
123
|
+
(format === 'markdown' ? /^#{1,6}\s+(.+)$/m.exec(body)?.[1]?.trim() : undefined) ||
|
|
124
|
+
basename(filename, extname(filename))
|
|
125
|
+
|
|
126
|
+
return {
|
|
127
|
+
title: title.slice(0, 300),
|
|
128
|
+
markdown: body,
|
|
129
|
+
sourceFormat: format,
|
|
130
|
+
parser: `convert-core/passthrough@1/${format}`,
|
|
131
|
+
...(sourcePath ? { sourcePath } : {}),
|
|
132
|
+
warnings,
|
|
133
|
+
dependencyVersions: {},
|
|
134
|
+
externalId: `sha256:${sha256(content)}`,
|
|
135
|
+
...(materialized.assets.length > 0 ? { assets: materialized.assets } : {}),
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
export const textAdapter: Adapter = {
|
|
140
|
+
id: 'text',
|
|
141
|
+
description: 'Markdown / 纯文本 / JSON / TSV 直通(只做契约归一化,不改变内容)',
|
|
142
|
+
kinds: ['text', 'file'],
|
|
143
|
+
extensions: TEXT_EXTENSIONS,
|
|
144
|
+
availability: () => ({ available: true }),
|
|
145
|
+
run,
|
|
146
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
import type { ConvertRequest, ConvertWarning, SourceFormat } from '../contract.js'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* An asset an adapter hands to the service as bytes. The service owns the
|
|
5
|
+
* delivery directory, so it computes the path, size and hash.
|
|
6
|
+
*/
|
|
7
|
+
export interface PreparedAsset {
|
|
8
|
+
rel: string
|
|
9
|
+
data: Buffer
|
|
10
|
+
contentType?: string
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export interface AdapterContext {
|
|
14
|
+
request: ConvertRequest
|
|
15
|
+
/** Staged local path for `kind=file`. */
|
|
16
|
+
inputPath?: string
|
|
17
|
+
/** Inline payload for `kind=text`. */
|
|
18
|
+
text?: string
|
|
19
|
+
filename: string
|
|
20
|
+
signal: AbortSignal
|
|
21
|
+
report: (progress: number | null, stage: string, message?: string) => void
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export interface AdapterOutput {
|
|
25
|
+
title: string
|
|
26
|
+
/** Markdown body only — front matter is the service's job. */
|
|
27
|
+
markdown: string
|
|
28
|
+
sourceFormat: SourceFormat
|
|
29
|
+
parser: string
|
|
30
|
+
sourcePath?: string
|
|
31
|
+
warnings: ConvertWarning[]
|
|
32
|
+
dependencyVersions: Record<string, string>
|
|
33
|
+
/** Stable identity for idempotency (file hash). */
|
|
34
|
+
externalId?: string
|
|
35
|
+
assets?: PreparedAsset[]
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export interface AdapterAvailability {
|
|
39
|
+
available: boolean
|
|
40
|
+
/** Populated when `available` is false. */
|
|
41
|
+
reason?: string
|
|
42
|
+
/** Actionable next step shown verbatim to the user. */
|
|
43
|
+
suggestion?: string
|
|
44
|
+
/** Versions of the pieces involved, when known. */
|
|
45
|
+
versions?: Record<string, string>
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export interface Adapter {
|
|
49
|
+
id: string
|
|
50
|
+
/** Human description shown in the console and in `GET /api/capabilities`. */
|
|
51
|
+
description: string
|
|
52
|
+
/** Task kinds this adapter can serve. */
|
|
53
|
+
kinds: ConvertRequest['kind'][]
|
|
54
|
+
/** File extensions (lowercase, with dot) this adapter claims. */
|
|
55
|
+
extensions: string[]
|
|
56
|
+
/** Cheap synchronous check used for routing decisions. */
|
|
57
|
+
availability: () => AdapterAvailability
|
|
58
|
+
/**
|
|
59
|
+
* Authoritative asynchronous check used by `GET /api/capabilities`: it really
|
|
60
|
+
* loads the engine (and the platform binary) instead of assuming it works.
|
|
61
|
+
*/
|
|
62
|
+
probe?: () => Promise<AdapterAvailability>
|
|
63
|
+
run: (context: AdapterContext) => Promise<AdapterOutput>
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export class AdapterError extends Error {
|
|
67
|
+
readonly code: string
|
|
68
|
+
readonly suggestion?: string
|
|
69
|
+
|
|
70
|
+
constructor(code: string, message: string, suggestion?: string) {
|
|
71
|
+
super(message)
|
|
72
|
+
this.name = 'AdapterError'
|
|
73
|
+
this.code = code
|
|
74
|
+
this.suggestion = suggestion
|
|
75
|
+
}
|
|
76
|
+
}
|