dsh-convert-core 0.1.0-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +149 -0
- package/bin/media-to-md.mjs +168 -0
- package/bin/tita-convert.mjs +85 -0
- package/docs/media-to-md.md +143 -0
- package/lib/adapters/anydoc.d.ts +73 -0
- package/lib/adapters/anydoc.d.ts.map +1 -0
- package/lib/adapters/anydoc.js +295 -0
- package/lib/adapters/anydoc.js.map +1 -0
- package/lib/adapters/iddoc.d.ts +28 -0
- package/lib/adapters/iddoc.d.ts.map +1 -0
- package/lib/adapters/iddoc.js +153 -0
- package/lib/adapters/iddoc.js.map +1 -0
- package/lib/adapters/index.d.ts +41 -0
- package/lib/adapters/index.d.ts.map +1 -0
- package/lib/adapters/index.js +115 -0
- package/lib/adapters/index.js.map +1 -0
- package/lib/adapters/scanned.d.ts +28 -0
- package/lib/adapters/scanned.d.ts.map +1 -0
- package/lib/adapters/scanned.js +121 -0
- package/lib/adapters/scanned.js.map +1 -0
- package/lib/adapters/text.d.ts +29 -0
- package/lib/adapters/text.d.ts.map +1 -0
- package/lib/adapters/text.js +135 -0
- package/lib/adapters/text.js.map +1 -0
- package/lib/adapters/types.d.ts +65 -0
- package/lib/adapters/types.d.ts.map +1 -0
- package/lib/adapters/types.js +11 -0
- package/lib/adapters/types.js.map +1 -0
- package/lib/contract.d.ts +185 -0
- package/lib/contract.d.ts.map +1 -0
- package/lib/contract.js +59 -0
- package/lib/contract.js.map +1 -0
- package/lib/delivery.d.ts +36 -0
- package/lib/delivery.d.ts.map +1 -0
- package/lib/delivery.js +97 -0
- package/lib/delivery.js.map +1 -0
- package/lib/http.d.ts +34 -0
- package/lib/http.d.ts.map +1 -0
- package/lib/http.js +372 -0
- package/lib/http.js.map +1 -0
- package/lib/markdown.d.ts +32 -0
- package/lib/markdown.d.ts.map +1 -0
- package/lib/markdown.js +145 -0
- package/lib/markdown.js.map +1 -0
- package/lib/media/binaries.d.ts +26 -0
- package/lib/media/binaries.d.ts.map +1 -0
- package/lib/media/binaries.js +60 -0
- package/lib/media/binaries.js.map +1 -0
- package/lib/media/convert.d.ts +19 -0
- package/lib/media/convert.d.ts.map +1 -0
- package/lib/media/convert.js +210 -0
- package/lib/media/convert.js.map +1 -0
- package/lib/media/index.d.ts +27 -0
- package/lib/media/index.d.ts.map +1 -0
- package/lib/media/index.js +27 -0
- package/lib/media/index.js.map +1 -0
- package/lib/media/markdown.d.ts +40 -0
- package/lib/media/markdown.d.ts.map +1 -0
- package/lib/media/markdown.js +89 -0
- package/lib/media/markdown.js.map +1 -0
- package/lib/media/media.d.ts +45 -0
- package/lib/media/media.d.ts.map +1 -0
- package/lib/media/media.js +167 -0
- package/lib/media/media.js.map +1 -0
- package/lib/media/models.d.ts +31 -0
- package/lib/media/models.d.ts.map +1 -0
- package/lib/media/models.js +87 -0
- package/lib/media/models.js.map +1 -0
- package/lib/media/scanned.d.ts +18 -0
- package/lib/media/scanned.d.ts.map +1 -0
- package/lib/media/scanned.js +179 -0
- package/lib/media/scanned.js.map +1 -0
- package/lib/media/types.d.ts +154 -0
- package/lib/media/types.d.ts.map +1 -0
- package/lib/media/types.js +23 -0
- package/lib/media/types.js.map +1 -0
- package/lib/server.d.ts +35 -0
- package/lib/server.d.ts.map +1 -0
- package/lib/server.js +64 -0
- package/lib/server.js.map +1 -0
- package/lib/service.d.ts +170 -0
- package/lib/service.d.ts.map +1 -0
- package/lib/service.js +562 -0
- package/lib/service.js.map +1 -0
- package/package.json +59 -0
- package/scripts/asr.py +55 -0
- package/scripts/ocr.py +40 -0
- package/scripts/pdf_pages.py +49 -0
- package/scripts/vendor.mjs +76 -0
- package/src/adapters/anydoc.ts +356 -0
- package/src/adapters/iddoc.ts +171 -0
- package/src/adapters/index.ts +147 -0
- package/src/adapters/scanned.ts +139 -0
- package/src/adapters/text.ts +146 -0
- package/src/adapters/types.ts +76 -0
- package/src/contract.ts +230 -0
- package/src/delivery.ts +124 -0
- package/src/http.ts +394 -0
- package/src/markdown.ts +141 -0
- package/src/media/binaries.ts +67 -0
- package/src/media/convert.ts +232 -0
- package/src/media/index.ts +43 -0
- package/src/media/markdown.ts +105 -0
- package/src/media/media.ts +204 -0
- package/src/media/models.ts +124 -0
- package/src/media/scanned.ts +200 -0
- package/src/media/types.ts +171 -0
- package/src/server.ts +85 -0
- package/src/service.ts +639 -0
- package/web/app.js +382 -0
- package/web/index.html +217 -0
- package/web/styles.css +263 -0
- package/web/vendor/icons.js +25 -0
- package/web/vendor/vue.global.prod.js +14 -0
package/scripts/asr.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""media-to-md 转写 helper:音频 → 带时间戳的转写 JSON(本地 faster-whisper,不调用云端/LLM)。
|
|
3
|
+
|
|
4
|
+
由 `src/models.ts` 通过 `uvx --with faster-whisper python` 拉起(转换内核的 iddoc adapter 也走这里)。
|
|
5
|
+
|
|
6
|
+
用法:
|
|
7
|
+
python asr.py <audio> [model] [lang]
|
|
8
|
+
|
|
9
|
+
- model: faster-whisper 模型名(small / large-v3-turbo / medium)或**本地模型目录**。
|
|
10
|
+
- lang: 语言代码(zh / en …),省略则自动检测。
|
|
11
|
+
|
|
12
|
+
输出(stdout,单行 JSON):
|
|
13
|
+
{"model","language","language_probability","duration","segments":[{"start","end","text"}]}
|
|
14
|
+
"""
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import sys
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def main() -> int:
|
|
21
|
+
if len(sys.argv) < 2:
|
|
22
|
+
print("usage: asr.py <media> [model] [lang]", file=sys.stderr)
|
|
23
|
+
return 2
|
|
24
|
+
media = sys.argv[1]
|
|
25
|
+
model_name = sys.argv[2] if len(sys.argv) > 2 and sys.argv[2] else os.environ.get("MEDIA_TO_MD_ASR_MODEL", "small")
|
|
26
|
+
lang = sys.argv[3] if len(sys.argv) > 3 and sys.argv[3] else None
|
|
27
|
+
|
|
28
|
+
from faster_whisper import WhisperModel
|
|
29
|
+
|
|
30
|
+
device = os.environ.get("MEDIA_TO_MD_ASR_DEVICE", "cpu")
|
|
31
|
+
compute_type = os.environ.get("MEDIA_TO_MD_ASR_COMPUTE", "int8")
|
|
32
|
+
model = WhisperModel(model_name, device=device, compute_type=compute_type)
|
|
33
|
+
segments, info = model.transcribe(
|
|
34
|
+
media,
|
|
35
|
+
language=lang,
|
|
36
|
+
beam_size=int(os.environ.get("MEDIA_TO_MD_ASR_BEAM", "1")),
|
|
37
|
+
vad_filter=True,
|
|
38
|
+
)
|
|
39
|
+
payload = {
|
|
40
|
+
"model": model_name,
|
|
41
|
+
"language": getattr(info, "language", None),
|
|
42
|
+
"language_probability": getattr(info, "language_probability", None),
|
|
43
|
+
"duration": getattr(info, "duration", None),
|
|
44
|
+
"segments": [
|
|
45
|
+
{"start": round(s.start, 3), "end": round(s.end, 3), "text": (s.text or "").strip()}
|
|
46
|
+
for s in segments
|
|
47
|
+
if (s.text or "").strip()
|
|
48
|
+
],
|
|
49
|
+
}
|
|
50
|
+
print(json.dumps(payload, ensure_ascii=False))
|
|
51
|
+
return 0
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
if __name__ == "__main__":
|
|
55
|
+
raise SystemExit(main())
|
package/scripts/ocr.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""media-to-md 图片 OCR helper:图片 → 画面文字 JSON(本地 RapidOCR,默认不启用)。
|
|
3
|
+
|
|
4
|
+
只有在 `imageOcr=true` 时才会被 `src/convert.ts` 调用(约定:图片默认直接收录、不做 OCR)。
|
|
5
|
+
|
|
6
|
+
用法:
|
|
7
|
+
python ocr.py <img1> [img2 ...]
|
|
8
|
+
输出(stdout,单行 JSON 数组):
|
|
9
|
+
[{"path","lines":[...],"text":"..."}]
|
|
10
|
+
"""
|
|
11
|
+
import json
|
|
12
|
+
import sys
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def main() -> int:
|
|
16
|
+
paths = sys.argv[1:]
|
|
17
|
+
if not paths:
|
|
18
|
+
print("usage: ocr.py <img...>", file=sys.stderr)
|
|
19
|
+
return 2
|
|
20
|
+
|
|
21
|
+
from rapidocr_onnxruntime import RapidOCR
|
|
22
|
+
|
|
23
|
+
engine = RapidOCR()
|
|
24
|
+
results = []
|
|
25
|
+
for path in paths:
|
|
26
|
+
entry = {"path": path, "lines": [], "text": "", "error": None}
|
|
27
|
+
try:
|
|
28
|
+
res, _elapse = engine(path)
|
|
29
|
+
lines = [(item[1] or "").strip() for item in (res or [])]
|
|
30
|
+
entry["lines"] = [line for line in lines if line]
|
|
31
|
+
entry["text"] = "\n".join(entry["lines"])
|
|
32
|
+
except Exception as exc: # 单张失败不影响整体
|
|
33
|
+
entry["error"] = f"{type(exc).__name__}: {exc}"
|
|
34
|
+
results.append(entry)
|
|
35
|
+
print(json.dumps(results, ensure_ascii=False))
|
|
36
|
+
return 0
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
if __name__ == "__main__":
|
|
40
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""media-to-md 轻量扫描档 helper:PDF → 逐页 PNG(pypdfium2,无系统依赖)。
|
|
3
|
+
|
|
4
|
+
为什么不用 poppler:`pdftoppm` 是外部二进制,不是每台机器都有(本机就没有);
|
|
5
|
+
pypdfium2 是 pip wheel,自带 pdfium,跨平台、可被 uvx 按需拉起。
|
|
6
|
+
|
|
7
|
+
用法:
|
|
8
|
+
python pdf_pages.py <in.pdf> <out_dir> [dpi] [max_pages]
|
|
9
|
+
|
|
10
|
+
输出(stdout,单行 JSON):
|
|
11
|
+
{"pages":[{"index":1,"path":"...","width":..,"height":..}],"totalPages":N,"renderedPages":M}
|
|
12
|
+
"""
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import sys
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def main() -> int:
|
|
19
|
+
if len(sys.argv) < 3:
|
|
20
|
+
print("usage: pdf_pages.py <in.pdf> <out_dir> [dpi] [max_pages]", file=sys.stderr)
|
|
21
|
+
return 2
|
|
22
|
+
pdf_path = sys.argv[1]
|
|
23
|
+
out_dir = sys.argv[2]
|
|
24
|
+
dpi = int(sys.argv[3]) if len(sys.argv) > 3 and sys.argv[3] else 150
|
|
25
|
+
max_pages = int(sys.argv[4]) if len(sys.argv) > 4 and sys.argv[4] else 50
|
|
26
|
+
|
|
27
|
+
import pypdfium2 as pdfium
|
|
28
|
+
|
|
29
|
+
os.makedirs(out_dir, exist_ok=True)
|
|
30
|
+
document = pdfium.PdfDocument(pdf_path)
|
|
31
|
+
total = len(document)
|
|
32
|
+
limit = max(0, min(total, max_pages))
|
|
33
|
+
scale = max(0.1, dpi / 72.0)
|
|
34
|
+
|
|
35
|
+
pages = []
|
|
36
|
+
for index in range(limit):
|
|
37
|
+
page = document[index]
|
|
38
|
+
bitmap = page.render(scale=scale)
|
|
39
|
+
image = bitmap.to_pil()
|
|
40
|
+
path = os.path.join(out_dir, f"page-{index + 1:04d}.png")
|
|
41
|
+
image.save(path)
|
|
42
|
+
pages.append({"index": index + 1, "path": path, "width": image.width, "height": image.height})
|
|
43
|
+
|
|
44
|
+
print(json.dumps({"pages": pages, "totalPages": total, "renderedPages": len(pages)}))
|
|
45
|
+
return 0
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
if __name__ == "__main__":
|
|
49
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Copy the browser dependencies into `web/vendor/` during `pnpm build`.
|
|
3
|
+
*
|
|
4
|
+
* The console is deliberately buildless: one HTML file + one app.js using the
|
|
5
|
+
* Vue 3 Options API against the global `Vue` object. That keeps the kernel
|
|
6
|
+
* runnable from a plain `node bin/tita-convert.mjs` with no bundler in the
|
|
7
|
+
* loop, which matters because the DSH plugin serves this page as static assets.
|
|
8
|
+
*
|
|
9
|
+
* Lucide is vendored as a plain SVG map rather than as the runtime
|
|
10
|
+
* `createIcons()` scanner: the scanner replaces DOM nodes behind Vue's back,
|
|
11
|
+
* while a map lets the template use `v-html` on a node Vue owns.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { mkdir, readFile, writeFile } from 'node:fs/promises'
|
|
15
|
+
import { createRequire } from 'node:module'
|
|
16
|
+
import { dirname, resolve } from 'node:path'
|
|
17
|
+
import { fileURLToPath } from 'node:url'
|
|
18
|
+
|
|
19
|
+
const require = createRequire(import.meta.url)
|
|
20
|
+
const root = resolve(dirname(fileURLToPath(import.meta.url)), '..')
|
|
21
|
+
const vendorDir = resolve(root, 'web', 'vendor')
|
|
22
|
+
|
|
23
|
+
/** Every icon the console template references. */
|
|
24
|
+
export const CONSOLE_ICONS = [
|
|
25
|
+
'layers', 'file-up', 'link', 'video', 'file-text', 'clipboard-type',
|
|
26
|
+
'refresh-cw', 'hard-drive', 'alert-triangle', 'check-circle', 'activity',
|
|
27
|
+
'inbox', 'package', 'scan-eye', 'x', 'rotate-ccw', 'loader', 'play',
|
|
28
|
+
'copy', 'download', 'circle-alert', 'circle-check',
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
await mkdir(vendorDir, { recursive: true })
|
|
32
|
+
|
|
33
|
+
async function vendorScript(specifier, filename, label) {
|
|
34
|
+
try {
|
|
35
|
+
const source = require.resolve(specifier)
|
|
36
|
+
const data = await readFile(source)
|
|
37
|
+
await writeFile(resolve(vendorDir, filename), data)
|
|
38
|
+
console.log(`[convert-core] vendor: ${label.padEnd(20)} ${(data.length / 1024).toFixed(0)} KiB`)
|
|
39
|
+
return true
|
|
40
|
+
} catch (cause) {
|
|
41
|
+
console.warn(`[convert-core] vendor: ${label} 跳过 —— ${cause.message}`)
|
|
42
|
+
return false
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
await vendorScript('vue/dist/vue.global.prod.js', 'vue.global.prod.js', 'vue (Options API)')
|
|
47
|
+
|
|
48
|
+
/** Lucide renamed a few icons; keep the template names stable. */
|
|
49
|
+
const ALIASES = {
|
|
50
|
+
'alert-triangle': 'triangle-alert',
|
|
51
|
+
'check-circle': 'circle-check',
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// Lucide ships raw SVGs in `lucide-static`; read them and emit one JS map.
|
|
55
|
+
const icons = {}
|
|
56
|
+
let missing = []
|
|
57
|
+
for (const name of CONSOLE_ICONS) {
|
|
58
|
+
try {
|
|
59
|
+
const file = ALIASES[name] ?? name
|
|
60
|
+
const path = require.resolve(`lucide-static/icons/${file}.svg`)
|
|
61
|
+
const svg = (await readFile(path, 'utf8'))
|
|
62
|
+
.replace(/\s+width="24"/, '')
|
|
63
|
+
.replace(/\s+height="24"/, '')
|
|
64
|
+
.replace(/\n+/g, '')
|
|
65
|
+
.trim()
|
|
66
|
+
icons[name] = svg
|
|
67
|
+
} catch {
|
|
68
|
+
missing.push(name)
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
await writeFile(
|
|
72
|
+
resolve(vendorDir, 'icons.js'),
|
|
73
|
+
`// Generated by scripts/vendor.mjs from the lucide-static icon set.\nwindow.__CONVERT_ICONS__ = ${JSON.stringify(icons, null, 2)};\n`,
|
|
74
|
+
)
|
|
75
|
+
console.log(`[convert-core] vendor: ${'lucide icons'.padEnd(20)} ${Object.keys(icons).length}/${CONSOLE_ICONS.length} icons`)
|
|
76
|
+
if (missing.length > 0) console.warn(`[convert-core] vendor: 缺少图标 ${missing.join(', ')}`)
|
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The document engine: `@firecrawl/anydoc` (MIT, Rust, prebuilt Node binding).
|
|
3
|
+
*
|
|
4
|
+
* One adapter replaces what used to be five (pdfjs text-layer reader, mammoth,
|
|
5
|
+
* the high-quality docling tier, plus the format routing between them). anydoc
|
|
6
|
+
* covers doc/docx/docm, ppt/pps/pot/pptx/pptm/ppsx/ppsm, xls/xlsx/xlsm/xlsb,
|
|
7
|
+
* odt/ods/odp, rtf, epub, csv and text-layer pdf, detects the format from the
|
|
8
|
+
* bytes, and needs no Python, no models and no external binary.
|
|
9
|
+
*
|
|
10
|
+
* The one thing it cannot do for us is the 0013 image contract: Markdown has no
|
|
11
|
+
* way to embed bytes, so an embedded image renders as its alt text while the
|
|
12
|
+
* bytes stay on `document.assets`. Retrieval only reads local relative image
|
|
13
|
+
* paths, so this adapter re-materialises every embedded image into `assets/`
|
|
14
|
+
* and puts a real `` reference back where the document had it.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { createRequire } from 'node:module'
|
|
18
|
+
import { readFile } from 'node:fs/promises'
|
|
19
|
+
import { basename, extname } from 'node:path'
|
|
20
|
+
import type { ConvertWarning, SourceFormat } from '../contract.js'
|
|
21
|
+
import { sha1, sha256 } from '../markdown.js'
|
|
22
|
+
import { AdapterError, type Adapter, type AdapterAvailability, type AdapterContext, type AdapterOutput, type PreparedAsset } from './types.js'
|
|
23
|
+
|
|
24
|
+
export const ANYDOC_EXTENSIONS = [
|
|
25
|
+
// Word
|
|
26
|
+
'.doc', '.docx', '.docm',
|
|
27
|
+
// PowerPoint
|
|
28
|
+
'.ppt', '.pps', '.pot', '.pptx', '.pptm', '.ppsx', '.ppsm',
|
|
29
|
+
// Excel
|
|
30
|
+
'.xls', '.xlsx', '.xlsm', '.xlsb',
|
|
31
|
+
// OpenDocument
|
|
32
|
+
'.odt', '.ods', '.odp',
|
|
33
|
+
// Other containers anydoc reads
|
|
34
|
+
'.rtf', '.epub', '.csv', '.pdf',
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
interface AnyDocAsset {
|
|
38
|
+
id?: number
|
|
39
|
+
mediaType?: string
|
|
40
|
+
originPart?: string
|
|
41
|
+
data?: Uint8Array
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
interface AnyDocInline {
|
|
45
|
+
kind: string
|
|
46
|
+
alt?: string
|
|
47
|
+
source?: { kind?: string; assetId?: number; url?: string }
|
|
48
|
+
text?: string
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
interface AnyDocBlock {
|
|
52
|
+
kind: string
|
|
53
|
+
content?: AnyDocInline[]
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
interface AnyDocDocument {
|
|
57
|
+
blocks?: AnyDocBlock[]
|
|
58
|
+
notes?: unknown[]
|
|
59
|
+
assets?: AnyDocAsset[]
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
interface AnyDocModule {
|
|
63
|
+
toDocument(bytes: Uint8Array): Promise<AnyDocDocument>
|
|
64
|
+
toMarkdown(path: string, options?: Record<string, unknown>): Promise<string>
|
|
65
|
+
toMarkdownBytes(bytes: Uint8Array, format?: string): Promise<string>
|
|
66
|
+
formatFromBytes(bytes: Uint8Array): string | null
|
|
67
|
+
formatFromPath(path: string): string | null
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
const MEDIA_EXTENSIONS: Record<string, string> = {
|
|
71
|
+
'image/png': 'png',
|
|
72
|
+
'image/jpeg': 'jpg',
|
|
73
|
+
'image/jpg': 'jpg',
|
|
74
|
+
'image/gif': 'gif',
|
|
75
|
+
'image/webp': 'webp',
|
|
76
|
+
'image/tiff': 'tiff',
|
|
77
|
+
'image/bmp': 'bmp',
|
|
78
|
+
'image/svg+xml': 'svg',
|
|
79
|
+
'image/avif': 'avif',
|
|
80
|
+
'image/heic': 'heic',
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
let modulePromise: Promise<AnyDocModule> | undefined
|
|
84
|
+
let moduleVersion: string | undefined
|
|
85
|
+
let moduleError: string | undefined
|
|
86
|
+
|
|
87
|
+
/** Load the binding once; the platform binary ships as an optional dependency. */
|
|
88
|
+
export async function loadAnydoc(): Promise<AnyDocModule> {
|
|
89
|
+
if (!modulePromise) {
|
|
90
|
+
modulePromise = (async () => {
|
|
91
|
+
const loaded = (await import('@firecrawl/anydoc')) as unknown as AnyDocModule & { default?: AnyDocModule }
|
|
92
|
+
const candidate = typeof loaded.toMarkdown === 'function' ? loaded : loaded.default
|
|
93
|
+
if (!candidate || typeof candidate.toMarkdown !== 'function') throw new Error('找不到 toMarkdown 导出')
|
|
94
|
+
try {
|
|
95
|
+
const require = createRequire(import.meta.url)
|
|
96
|
+
const pkg = JSON.parse(await readFile(require.resolve('@firecrawl/anydoc/package.json'), 'utf8')) as { version?: string }
|
|
97
|
+
moduleVersion = pkg.version
|
|
98
|
+
} catch {
|
|
99
|
+
/* version is cosmetic */
|
|
100
|
+
}
|
|
101
|
+
return candidate
|
|
102
|
+
})().catch(error => {
|
|
103
|
+
moduleError = error instanceof Error ? error.message : String(error)
|
|
104
|
+
modulePromise = undefined
|
|
105
|
+
throw error
|
|
106
|
+
})
|
|
107
|
+
}
|
|
108
|
+
return await modulePromise
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
export async function anydocAvailability(): Promise<AdapterAvailability> {
|
|
112
|
+
try {
|
|
113
|
+
await loadAnydoc()
|
|
114
|
+
return { available: true, ...(moduleVersion ? { versions: { '@firecrawl/anydoc': moduleVersion } } : {}) }
|
|
115
|
+
} catch (cause) {
|
|
116
|
+
return {
|
|
117
|
+
available: false,
|
|
118
|
+
reason: `无法加载 @firecrawl/anydoc:${moduleError ?? (cause instanceof Error ? cause.message : String(cause))}`,
|
|
119
|
+
suggestion: '在 packages/dsh-convert-core 目录执行 pnpm install;Apple Silicon 需要 @firecrawl/anydoc-darwin-arm64(随 optionalDependencies 自动安装)。',
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
function extensionFor(asset: AnyDocAsset): string | undefined {
|
|
125
|
+
const mediaType = asset.mediaType?.split(';')[0]?.trim().toLowerCase()
|
|
126
|
+
if (mediaType && MEDIA_EXTENSIONS[mediaType]) return MEDIA_EXTENSIONS[mediaType]
|
|
127
|
+
const fromPart = asset.originPart ? /\.([a-zA-Z0-9]{2,5})$/.exec(asset.originPart)?.[1]?.toLowerCase() : undefined
|
|
128
|
+
if (fromPart && fromPart !== 'bin') return fromPart
|
|
129
|
+
return undefined
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/** In document order, every inline image that points at an embedded asset. */
|
|
133
|
+
function collectEmbeddedImages(document: AnyDocDocument): Array<{ assetId: number; alt: string; asset: AnyDocAsset }> {
|
|
134
|
+
const found: Array<{ assetId: number; alt: string; asset: AnyDocAsset }> = []
|
|
135
|
+
for (const block of document.blocks ?? []) {
|
|
136
|
+
for (const inline of block.content ?? []) {
|
|
137
|
+
if (inline.kind !== 'image') continue
|
|
138
|
+
const source = inline.source
|
|
139
|
+
if (!source || source.kind !== 'asset' || typeof source.assetId !== 'number') continue
|
|
140
|
+
const asset = document.assets?.[source.assetId]
|
|
141
|
+
if (!asset) continue
|
|
142
|
+
found.push({ assetId: source.assetId, alt: (inline.alt ?? '').trim(), asset })
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
return found
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export interface ImagePlacement {
|
|
149
|
+
markdown: string
|
|
150
|
+
placed: number
|
|
151
|
+
appended: number
|
|
152
|
+
warnings: string[]
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* Put real image references back into the Markdown.
|
|
157
|
+
*
|
|
158
|
+
* anydoc renders an embedded image as its alt text, so placement is a match on
|
|
159
|
+
* that text. A standalone-line match is used only when it is unambiguous;
|
|
160
|
+
* anything that cannot be located is appended under a `## 图片` heading rather
|
|
161
|
+
* than guessed at, and the count is reported.
|
|
162
|
+
*/
|
|
163
|
+
export function placeImageReferences(markdown: string, images: Array<{ rel: string; alt: string }>): ImagePlacement {
|
|
164
|
+
const lines = markdown.split('\n')
|
|
165
|
+
const consumed = new Set<number>()
|
|
166
|
+
const remaining: Array<{ rel: string; alt: string }> = []
|
|
167
|
+
const warnings: string[] = []
|
|
168
|
+
|
|
169
|
+
for (const image of images) {
|
|
170
|
+
const alt = image.alt.trim()
|
|
171
|
+
let index = -1
|
|
172
|
+
if (alt) {
|
|
173
|
+
// Only accept a line that is exactly the alt text, and only once.
|
|
174
|
+
const candidates = lines
|
|
175
|
+
.map((line, position) => (line.trim() === alt && !consumed.has(position) ? position : -1))
|
|
176
|
+
.filter(position => position >= 0)
|
|
177
|
+
if (candidates.length === 1) index = candidates[0]!
|
|
178
|
+
}
|
|
179
|
+
if (index >= 0) {
|
|
180
|
+
consumed.add(index)
|
|
181
|
+
lines[index] = ``
|
|
182
|
+
continue
|
|
183
|
+
}
|
|
184
|
+
remaining.push(image)
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
let output = lines.join('\n')
|
|
188
|
+
if (remaining.length > 0) {
|
|
189
|
+
const section = remaining.map(image => ``).join('\n\n')
|
|
190
|
+
output = `${output.trimEnd()}\n\n## 图片\n\n${section}\n`
|
|
191
|
+
warnings.push(
|
|
192
|
+
`${remaining.length} 张内嵌图片在正文中没有可定位的占位(无 alt 文本或 alt 不唯一),已追加到文末「图片」小节。`,
|
|
193
|
+
)
|
|
194
|
+
}
|
|
195
|
+
return { markdown: output, placed: images.length - remaining.length, appended: remaining.length, warnings }
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
function sourceFormatFor(anydocFormat: string | null, filename: string): SourceFormat {
|
|
199
|
+
const format = anydocFormat ?? extname(filename).replace('.', '').toLowerCase()
|
|
200
|
+
if (format === 'pdf') return 'pdf'
|
|
201
|
+
if (format === 'csv') return 'text'
|
|
202
|
+
return 'office'
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** anydoc's error codes are already precise; keep them instead of flattening. */
|
|
206
|
+
const ANYDOC_SUGGESTIONS: Record<string, string> = {
|
|
207
|
+
needsOcr:
|
|
208
|
+
'该 PDF 是扫描件/纯图片页,anydoc 本地不做 OCR(它只覆盖文本层 PDF)。走扫描档:把 adapter 指定为 "scanned",' +
|
|
209
|
+
'或设环境变量 SCANNED_OCR=1 让 .pdf 直接走它 —— 扫描档用 pypdfium2 渲染页图 + 本地 RapidOCR 识别文字,不出网,' +
|
|
210
|
+
'但不做版面还原;需要版面/表格/公式时见 issues/0017 的可选高质量档。' +
|
|
211
|
+
'不要使用 anydoc 的 ocr:"hosted":它会把整份文档发往 Firecrawl Parse 云端,与 0015 的本地优先原则冲突。',
|
|
212
|
+
encrypted: '文档有密码保护,请先去掉密码再转换。',
|
|
213
|
+
unsupported: '该格式不在本版范围内(doc/docx/ppt/pptx/xls/xlsx/odt/ods/odp/rtf/epub/csv/文本层 PDF)。',
|
|
214
|
+
malformed: '文件结构损坏,无法提取有意义的内容。',
|
|
215
|
+
resourceLimit: '文件触发了 anydoc 的解压/嵌套安全上限,可能是文档炸弹或异常嵌套。',
|
|
216
|
+
missingPart: '文档缺少必需的结构部件(可能被截断或由非标准工具生成)。',
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
function toAdapterError(cause: unknown): AdapterError {
|
|
220
|
+
const error = cause as Error & { code?: string; pages?: number[] }
|
|
221
|
+
const code = typeof error?.code === 'string' ? error.code : undefined
|
|
222
|
+
if (code && code !== 'io') {
|
|
223
|
+
const pages = Array.isArray(error.pages) && error.pages.length > 0 ? `(第 ${error.pages.slice(0, 8).join(', ')} 页)` : ''
|
|
224
|
+
return new AdapterError(
|
|
225
|
+
`ANYDOC_${code.replace(/([a-z])([A-Z])/g, '$1_$2').toUpperCase()}`,
|
|
226
|
+
`${error.message || code}${pages}`,
|
|
227
|
+
ANYDOC_SUGGESTIONS[code],
|
|
228
|
+
)
|
|
229
|
+
}
|
|
230
|
+
return new AdapterError('ANYDOC_FAILED', error instanceof Error ? error.message : String(cause), ANYDOC_SUGGESTIONS.malformed)
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
async function run(context: AdapterContext): Promise<AdapterOutput> {
|
|
234
|
+
const sourcePath = context.inputPath
|
|
235
|
+
if (!sourcePath) throw new AdapterError('INPUT_REQUIRED', '文档适配器需要文件路径。')
|
|
236
|
+
|
|
237
|
+
const anydoc = await loadAnydoc().catch(cause => {
|
|
238
|
+
throw new AdapterError(
|
|
239
|
+
'DEPENDENCY_MISSING',
|
|
240
|
+
`无法加载 @firecrawl/anydoc:${cause instanceof Error ? cause.message : String(cause)}`,
|
|
241
|
+
'在 packages/dsh-convert-core 目录执行 pnpm install。',
|
|
242
|
+
)
|
|
243
|
+
})
|
|
244
|
+
|
|
245
|
+
const filename = context.filename || basename(sourcePath)
|
|
246
|
+
const bytes = await readFile(sourcePath)
|
|
247
|
+
context.report(0.15, 'detect', '识别文档格式')
|
|
248
|
+
|
|
249
|
+
let detected: string | null = null
|
|
250
|
+
try {
|
|
251
|
+
detected = anydoc.formatFromBytes(new Uint8Array(bytes))
|
|
252
|
+
} catch {
|
|
253
|
+
detected = null
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
context.report(0.35, 'convert', `解析 ${detected ?? extname(filename)}`)
|
|
257
|
+
let markdown: string
|
|
258
|
+
let document: AnyDocDocument = {}
|
|
259
|
+
const assetNote: string[] = []
|
|
260
|
+
try {
|
|
261
|
+
markdown = await anydoc.toMarkdown(sourcePath)
|
|
262
|
+
} catch (cause) {
|
|
263
|
+
throw toAdapterError(cause)
|
|
264
|
+
}
|
|
265
|
+
// `toDocument` covers the OOXML / OpenDocument / RTF / EPUB / CSV family. PDF
|
|
266
|
+
// renders straight to Markdown through pdf-inspector and *rejects* the
|
|
267
|
+
// document model, so a PDF simply has no embedded assets to place.
|
|
268
|
+
if (detected !== 'pdf') {
|
|
269
|
+
try {
|
|
270
|
+
document = await anydoc.toDocument(new Uint8Array(bytes))
|
|
271
|
+
} catch (cause) {
|
|
272
|
+
const error = cause as Error & { code?: string }
|
|
273
|
+
// Never lose a good Markdown because the asset model was unavailable.
|
|
274
|
+
assetNote.push(
|
|
275
|
+
`内嵌资源枚举失败(${error?.code ?? 'unknown'}: ${error?.message ?? String(cause)}),本次只交付 Markdown,未落地图片。`,
|
|
276
|
+
)
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
if (!markdown.trim()) {
|
|
280
|
+
throw new AdapterError('CONTENT_REQUIRED', '文档解析后没有内容(可能全为图片或受保护内容)。')
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
// ---- asset glue: bytes live on the document model, not in the Markdown ----
|
|
284
|
+
const embedded = collectEmbeddedImages(document)
|
|
285
|
+
const images: Array<{ rel: string; alt: string }> = []
|
|
286
|
+
const assets: PreparedAsset[] = []
|
|
287
|
+
const warnings: ConvertWarning[] = []
|
|
288
|
+
let skippedObjects = 0
|
|
289
|
+
|
|
290
|
+
for (const entry of embedded) {
|
|
291
|
+
const data = entry.asset.data ? Buffer.from(entry.asset.data) : undefined
|
|
292
|
+
if (!data || data.length === 0) {
|
|
293
|
+
skippedObjects += 1
|
|
294
|
+
continue
|
|
295
|
+
}
|
|
296
|
+
if (!entry.asset.mediaType?.startsWith('image/')) {
|
|
297
|
+
// 0013 only consumes images; other embedded objects are reported, not dropped.
|
|
298
|
+
skippedObjects += 1
|
|
299
|
+
continue
|
|
300
|
+
}
|
|
301
|
+
const extension = extensionFor(entry.asset)
|
|
302
|
+
if (!extension) {
|
|
303
|
+
skippedObjects += 1
|
|
304
|
+
continue
|
|
305
|
+
}
|
|
306
|
+
const digest = sha1(data)
|
|
307
|
+
const rel = `assets/img-${digest.slice(0, 12)}.${extension}`
|
|
308
|
+
images.push({ rel, alt: entry.alt })
|
|
309
|
+
assets.push({
|
|
310
|
+
rel,
|
|
311
|
+
data,
|
|
312
|
+
...(entry.asset.mediaType ? { contentType: entry.asset.mediaType } : {}),
|
|
313
|
+
})
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
const placement = placeImageReferences(markdown, images)
|
|
317
|
+
for (const note of assetNote) warnings.push({ code: 'ASSET_ENUMERATION_FAILED', message: note })
|
|
318
|
+
if (placement.appended > 0) {
|
|
319
|
+
warnings.push({ code: 'IMAGES_APPENDED', message: placement.warnings[0]! })
|
|
320
|
+
}
|
|
321
|
+
if (skippedObjects > 0) {
|
|
322
|
+
warnings.push({
|
|
323
|
+
code: 'EMBEDDED_OBJECTS_SKIPPED',
|
|
324
|
+
message: `文档里有 ${skippedObjects} 个内嵌对象不是可落地的图片(非 image/* 或缺少字节),未写入 assets/。`,
|
|
325
|
+
})
|
|
326
|
+
}
|
|
327
|
+
if (embedded.length > 0) {
|
|
328
|
+
warnings.push({
|
|
329
|
+
code: 'IMAGES_LOCALIZED',
|
|
330
|
+
message: `内嵌图片 ${embedded.length} 张已落地为本地 assets/ 并改写正文引用(${placement.placed} 张原位、${placement.appended} 张追加)。`,
|
|
331
|
+
})
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
const heading = /^#{1,6}\s+(.+)$/m.exec(placement.markdown)?.[1]?.trim()
|
|
335
|
+
return {
|
|
336
|
+
title: (context.request.title?.trim() || heading || basename(filename, extname(filename))).slice(0, 300),
|
|
337
|
+
markdown: placement.markdown.trim(),
|
|
338
|
+
sourceFormat: sourceFormatFor(detected, filename),
|
|
339
|
+
parser: `anydoc@${moduleVersion ?? 'unknown'}/${detected ?? 'unknown'}`,
|
|
340
|
+
sourcePath,
|
|
341
|
+
warnings,
|
|
342
|
+
dependencyVersions: { '@firecrawl/anydoc': moduleVersion ?? 'unknown' },
|
|
343
|
+
externalId: `sha256:${sha256(bytes)}`,
|
|
344
|
+
assets,
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
export const anydocAdapter: Adapter = {
|
|
349
|
+
id: 'anydoc',
|
|
350
|
+
description: 'DOC/DOCX/PPT/PPTX/XLS/XLSX/ODT/ODS/ODP/RTF/EPUB/CSV/文本层 PDF → GFM Markdown(Rust 内核,无模型、无外部二进制;内嵌图片落地到 assets/)',
|
|
351
|
+
kinds: ['file'],
|
|
352
|
+
extensions: ANYDOC_EXTENSIONS,
|
|
353
|
+
availability: () => ({ available: true, ...(moduleVersion ? { versions: { '@firecrawl/anydoc': moduleVersion } } : {}) }),
|
|
354
|
+
probe: anydocAvailability,
|
|
355
|
+
run,
|
|
356
|
+
}
|