dsh-convert-core 0.1.0-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (115) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +149 -0
  3. package/bin/media-to-md.mjs +168 -0
  4. package/bin/tita-convert.mjs +85 -0
  5. package/docs/media-to-md.md +143 -0
  6. package/lib/adapters/anydoc.d.ts +73 -0
  7. package/lib/adapters/anydoc.d.ts.map +1 -0
  8. package/lib/adapters/anydoc.js +295 -0
  9. package/lib/adapters/anydoc.js.map +1 -0
  10. package/lib/adapters/iddoc.d.ts +28 -0
  11. package/lib/adapters/iddoc.d.ts.map +1 -0
  12. package/lib/adapters/iddoc.js +153 -0
  13. package/lib/adapters/iddoc.js.map +1 -0
  14. package/lib/adapters/index.d.ts +41 -0
  15. package/lib/adapters/index.d.ts.map +1 -0
  16. package/lib/adapters/index.js +115 -0
  17. package/lib/adapters/index.js.map +1 -0
  18. package/lib/adapters/scanned.d.ts +28 -0
  19. package/lib/adapters/scanned.d.ts.map +1 -0
  20. package/lib/adapters/scanned.js +121 -0
  21. package/lib/adapters/scanned.js.map +1 -0
  22. package/lib/adapters/text.d.ts +29 -0
  23. package/lib/adapters/text.d.ts.map +1 -0
  24. package/lib/adapters/text.js +135 -0
  25. package/lib/adapters/text.js.map +1 -0
  26. package/lib/adapters/types.d.ts +65 -0
  27. package/lib/adapters/types.d.ts.map +1 -0
  28. package/lib/adapters/types.js +11 -0
  29. package/lib/adapters/types.js.map +1 -0
  30. package/lib/contract.d.ts +185 -0
  31. package/lib/contract.d.ts.map +1 -0
  32. package/lib/contract.js +59 -0
  33. package/lib/contract.js.map +1 -0
  34. package/lib/delivery.d.ts +36 -0
  35. package/lib/delivery.d.ts.map +1 -0
  36. package/lib/delivery.js +97 -0
  37. package/lib/delivery.js.map +1 -0
  38. package/lib/http.d.ts +34 -0
  39. package/lib/http.d.ts.map +1 -0
  40. package/lib/http.js +372 -0
  41. package/lib/http.js.map +1 -0
  42. package/lib/markdown.d.ts +32 -0
  43. package/lib/markdown.d.ts.map +1 -0
  44. package/lib/markdown.js +145 -0
  45. package/lib/markdown.js.map +1 -0
  46. package/lib/media/binaries.d.ts +26 -0
  47. package/lib/media/binaries.d.ts.map +1 -0
  48. package/lib/media/binaries.js +60 -0
  49. package/lib/media/binaries.js.map +1 -0
  50. package/lib/media/convert.d.ts +19 -0
  51. package/lib/media/convert.d.ts.map +1 -0
  52. package/lib/media/convert.js +210 -0
  53. package/lib/media/convert.js.map +1 -0
  54. package/lib/media/index.d.ts +27 -0
  55. package/lib/media/index.d.ts.map +1 -0
  56. package/lib/media/index.js +27 -0
  57. package/lib/media/index.js.map +1 -0
  58. package/lib/media/markdown.d.ts +40 -0
  59. package/lib/media/markdown.d.ts.map +1 -0
  60. package/lib/media/markdown.js +89 -0
  61. package/lib/media/markdown.js.map +1 -0
  62. package/lib/media/media.d.ts +45 -0
  63. package/lib/media/media.d.ts.map +1 -0
  64. package/lib/media/media.js +167 -0
  65. package/lib/media/media.js.map +1 -0
  66. package/lib/media/models.d.ts +31 -0
  67. package/lib/media/models.d.ts.map +1 -0
  68. package/lib/media/models.js +87 -0
  69. package/lib/media/models.js.map +1 -0
  70. package/lib/media/scanned.d.ts +18 -0
  71. package/lib/media/scanned.d.ts.map +1 -0
  72. package/lib/media/scanned.js +179 -0
  73. package/lib/media/scanned.js.map +1 -0
  74. package/lib/media/types.d.ts +154 -0
  75. package/lib/media/types.d.ts.map +1 -0
  76. package/lib/media/types.js +23 -0
  77. package/lib/media/types.js.map +1 -0
  78. package/lib/server.d.ts +35 -0
  79. package/lib/server.d.ts.map +1 -0
  80. package/lib/server.js +64 -0
  81. package/lib/server.js.map +1 -0
  82. package/lib/service.d.ts +170 -0
  83. package/lib/service.d.ts.map +1 -0
  84. package/lib/service.js +562 -0
  85. package/lib/service.js.map +1 -0
  86. package/package.json +59 -0
  87. package/scripts/asr.py +55 -0
  88. package/scripts/ocr.py +40 -0
  89. package/scripts/pdf_pages.py +49 -0
  90. package/scripts/vendor.mjs +76 -0
  91. package/src/adapters/anydoc.ts +356 -0
  92. package/src/adapters/iddoc.ts +171 -0
  93. package/src/adapters/index.ts +147 -0
  94. package/src/adapters/scanned.ts +139 -0
  95. package/src/adapters/text.ts +146 -0
  96. package/src/adapters/types.ts +76 -0
  97. package/src/contract.ts +230 -0
  98. package/src/delivery.ts +124 -0
  99. package/src/http.ts +394 -0
  100. package/src/markdown.ts +141 -0
  101. package/src/media/binaries.ts +67 -0
  102. package/src/media/convert.ts +232 -0
  103. package/src/media/index.ts +43 -0
  104. package/src/media/markdown.ts +105 -0
  105. package/src/media/media.ts +204 -0
  106. package/src/media/models.ts +124 -0
  107. package/src/media/scanned.ts +200 -0
  108. package/src/media/types.ts +171 -0
  109. package/src/server.ts +85 -0
  110. package/src/service.ts +639 -0
  111. package/web/app.js +382 -0
  112. package/web/index.html +217 -0
  113. package/web/styles.css +263 -0
  114. package/web/vendor/icons.js +25 -0
  115. package/web/vendor/vue.global.prod.js +14 -0
package/scripts/asr.py ADDED
@@ -0,0 +1,55 @@
1
+ #!/usr/bin/env python3
2
+ """media-to-md 转写 helper:音频 → 带时间戳的转写 JSON(本地 faster-whisper,不调用云端/LLM)。
3
+
4
+ 由 `src/models.ts` 通过 `uvx --with faster-whisper python` 拉起(转换内核的 iddoc adapter 也走这里)。
5
+
6
+ 用法:
7
+ python asr.py <audio> [model] [lang]
8
+
9
+ - model: faster-whisper 模型名(small / large-v3-turbo / medium)或**本地模型目录**。
10
+ - lang: 语言代码(zh / en …),省略则自动检测。
11
+
12
+ 输出(stdout,单行 JSON):
13
+ {"model","language","language_probability","duration","segments":[{"start","end","text"}]}
14
+ """
15
+ import json
16
+ import os
17
+ import sys
18
+
19
+
20
+ def main() -> int:
21
+ if len(sys.argv) < 2:
22
+ print("usage: asr.py <media> [model] [lang]", file=sys.stderr)
23
+ return 2
24
+ media = sys.argv[1]
25
+ model_name = sys.argv[2] if len(sys.argv) > 2 and sys.argv[2] else os.environ.get("MEDIA_TO_MD_ASR_MODEL", "small")
26
+ lang = sys.argv[3] if len(sys.argv) > 3 and sys.argv[3] else None
27
+
28
+ from faster_whisper import WhisperModel
29
+
30
+ device = os.environ.get("MEDIA_TO_MD_ASR_DEVICE", "cpu")
31
+ compute_type = os.environ.get("MEDIA_TO_MD_ASR_COMPUTE", "int8")
32
+ model = WhisperModel(model_name, device=device, compute_type=compute_type)
33
+ segments, info = model.transcribe(
34
+ media,
35
+ language=lang,
36
+ beam_size=int(os.environ.get("MEDIA_TO_MD_ASR_BEAM", "1")),
37
+ vad_filter=True,
38
+ )
39
+ payload = {
40
+ "model": model_name,
41
+ "language": getattr(info, "language", None),
42
+ "language_probability": getattr(info, "language_probability", None),
43
+ "duration": getattr(info, "duration", None),
44
+ "segments": [
45
+ {"start": round(s.start, 3), "end": round(s.end, 3), "text": (s.text or "").strip()}
46
+ for s in segments
47
+ if (s.text or "").strip()
48
+ ],
49
+ }
50
+ print(json.dumps(payload, ensure_ascii=False))
51
+ return 0
52
+
53
+
54
+ if __name__ == "__main__":
55
+ raise SystemExit(main())
package/scripts/ocr.py ADDED
@@ -0,0 +1,40 @@
1
+ #!/usr/bin/env python3
2
+ """media-to-md 图片 OCR helper:图片 → 画面文字 JSON(本地 RapidOCR,默认不启用)。
3
+
4
+ 只有在 `imageOcr=true` 时才会被 `src/convert.ts` 调用(约定:图片默认直接收录、不做 OCR)。
5
+
6
+ 用法:
7
+ python ocr.py <img1> [img2 ...]
8
+ 输出(stdout,单行 JSON 数组):
9
+ [{"path","lines":[...],"text":"..."}]
10
+ """
11
+ import json
12
+ import sys
13
+
14
+
15
+ def main() -> int:
16
+ paths = sys.argv[1:]
17
+ if not paths:
18
+ print("usage: ocr.py <img...>", file=sys.stderr)
19
+ return 2
20
+
21
+ from rapidocr_onnxruntime import RapidOCR
22
+
23
+ engine = RapidOCR()
24
+ results = []
25
+ for path in paths:
26
+ entry = {"path": path, "lines": [], "text": "", "error": None}
27
+ try:
28
+ res, _elapse = engine(path)
29
+ lines = [(item[1] or "").strip() for item in (res or [])]
30
+ entry["lines"] = [line for line in lines if line]
31
+ entry["text"] = "\n".join(entry["lines"])
32
+ except Exception as exc: # 单张失败不影响整体
33
+ entry["error"] = f"{type(exc).__name__}: {exc}"
34
+ results.append(entry)
35
+ print(json.dumps(results, ensure_ascii=False))
36
+ return 0
37
+
38
+
39
+ if __name__ == "__main__":
40
+ raise SystemExit(main())
@@ -0,0 +1,49 @@
1
+ #!/usr/bin/env python3
2
+ """media-to-md 轻量扫描档 helper:PDF → 逐页 PNG(pypdfium2,无系统依赖)。
3
+
4
+ 为什么不用 poppler:`pdftoppm` 是外部二进制,不是每台机器都有(本机就没有);
5
+ pypdfium2 是 pip wheel,自带 pdfium,跨平台、可被 uvx 按需拉起。
6
+
7
+ 用法:
8
+ python pdf_pages.py <in.pdf> <out_dir> [dpi] [max_pages]
9
+
10
+ 输出(stdout,单行 JSON):
11
+ {"pages":[{"index":1,"path":"...","width":..,"height":..}],"totalPages":N,"renderedPages":M}
12
+ """
13
+ import json
14
+ import os
15
+ import sys
16
+
17
+
18
+ def main() -> int:
19
+ if len(sys.argv) < 3:
20
+ print("usage: pdf_pages.py <in.pdf> <out_dir> [dpi] [max_pages]", file=sys.stderr)
21
+ return 2
22
+ pdf_path = sys.argv[1]
23
+ out_dir = sys.argv[2]
24
+ dpi = int(sys.argv[3]) if len(sys.argv) > 3 and sys.argv[3] else 150
25
+ max_pages = int(sys.argv[4]) if len(sys.argv) > 4 and sys.argv[4] else 50
26
+
27
+ import pypdfium2 as pdfium
28
+
29
+ os.makedirs(out_dir, exist_ok=True)
30
+ document = pdfium.PdfDocument(pdf_path)
31
+ total = len(document)
32
+ limit = max(0, min(total, max_pages))
33
+ scale = max(0.1, dpi / 72.0)
34
+
35
+ pages = []
36
+ for index in range(limit):
37
+ page = document[index]
38
+ bitmap = page.render(scale=scale)
39
+ image = bitmap.to_pil()
40
+ path = os.path.join(out_dir, f"page-{index + 1:04d}.png")
41
+ image.save(path)
42
+ pages.append({"index": index + 1, "path": path, "width": image.width, "height": image.height})
43
+
44
+ print(json.dumps({"pages": pages, "totalPages": total, "renderedPages": len(pages)}))
45
+ return 0
46
+
47
+
48
+ if __name__ == "__main__":
49
+ raise SystemExit(main())
@@ -0,0 +1,76 @@
1
+ /**
2
+ * Copy the browser dependencies into `web/vendor/` during `pnpm build`.
3
+ *
4
+ * The console is deliberately buildless: one HTML file + one app.js using the
5
+ * Vue 3 Options API against the global `Vue` object. That keeps the kernel
6
+ * runnable from a plain `node bin/tita-convert.mjs` with no bundler in the
7
+ * loop, which matters because the DSH plugin serves this page as static assets.
8
+ *
9
+ * Lucide is vendored as a plain SVG map rather than as the runtime
10
+ * `createIcons()` scanner: the scanner replaces DOM nodes behind Vue's back,
11
+ * while a map lets the template use `v-html` on a node Vue owns.
12
+ */
13
+
14
+ import { mkdir, readFile, writeFile } from 'node:fs/promises'
15
+ import { createRequire } from 'node:module'
16
+ import { dirname, resolve } from 'node:path'
17
+ import { fileURLToPath } from 'node:url'
18
+
19
+ const require = createRequire(import.meta.url)
20
+ const root = resolve(dirname(fileURLToPath(import.meta.url)), '..')
21
+ const vendorDir = resolve(root, 'web', 'vendor')
22
+
23
+ /** Every icon the console template references. */
24
+ export const CONSOLE_ICONS = [
25
+ 'layers', 'file-up', 'link', 'video', 'file-text', 'clipboard-type',
26
+ 'refresh-cw', 'hard-drive', 'alert-triangle', 'check-circle', 'activity',
27
+ 'inbox', 'package', 'scan-eye', 'x', 'rotate-ccw', 'loader', 'play',
28
+ 'copy', 'download', 'circle-alert', 'circle-check',
29
+ ]
30
+
31
+ await mkdir(vendorDir, { recursive: true })
32
+
33
+ async function vendorScript(specifier, filename, label) {
34
+ try {
35
+ const source = require.resolve(specifier)
36
+ const data = await readFile(source)
37
+ await writeFile(resolve(vendorDir, filename), data)
38
+ console.log(`[convert-core] vendor: ${label.padEnd(20)} ${(data.length / 1024).toFixed(0)} KiB`)
39
+ return true
40
+ } catch (cause) {
41
+ console.warn(`[convert-core] vendor: ${label} 跳过 —— ${cause.message}`)
42
+ return false
43
+ }
44
+ }
45
+
46
+ await vendorScript('vue/dist/vue.global.prod.js', 'vue.global.prod.js', 'vue (Options API)')
47
+
48
+ /** Lucide renamed a few icons; keep the template names stable. */
49
+ const ALIASES = {
50
+ 'alert-triangle': 'triangle-alert',
51
+ 'check-circle': 'circle-check',
52
+ }
53
+
54
+ // Lucide ships raw SVGs in `lucide-static`; read them and emit one JS map.
55
+ const icons = {}
56
+ let missing = []
57
+ for (const name of CONSOLE_ICONS) {
58
+ try {
59
+ const file = ALIASES[name] ?? name
60
+ const path = require.resolve(`lucide-static/icons/${file}.svg`)
61
+ const svg = (await readFile(path, 'utf8'))
62
+ .replace(/\s+width="24"/, '')
63
+ .replace(/\s+height="24"/, '')
64
+ .replace(/\n+/g, '')
65
+ .trim()
66
+ icons[name] = svg
67
+ } catch {
68
+ missing.push(name)
69
+ }
70
+ }
71
+ await writeFile(
72
+ resolve(vendorDir, 'icons.js'),
73
+ `// Generated by scripts/vendor.mjs from the lucide-static icon set.\nwindow.__CONVERT_ICONS__ = ${JSON.stringify(icons, null, 2)};\n`,
74
+ )
75
+ console.log(`[convert-core] vendor: ${'lucide icons'.padEnd(20)} ${Object.keys(icons).length}/${CONSOLE_ICONS.length} icons`)
76
+ if (missing.length > 0) console.warn(`[convert-core] vendor: 缺少图标 ${missing.join(', ')}`)
@@ -0,0 +1,356 @@
1
+ /**
2
+ * The document engine: `@firecrawl/anydoc` (MIT, Rust, prebuilt Node binding).
3
+ *
4
+ * One adapter replaces what used to be five (pdfjs text-layer reader, mammoth,
5
+ * the high-quality docling tier, plus the format routing between them). anydoc
6
+ * covers doc/docx/docm, ppt/pps/pot/pptx/pptm/ppsx/ppsm, xls/xlsx/xlsm/xlsb,
7
+ * odt/ods/odp, rtf, epub, csv and text-layer pdf, detects the format from the
8
+ * bytes, and needs no Python, no models and no external binary.
9
+ *
10
+ * The one thing it cannot do for us is the 0013 image contract: Markdown has no
11
+ * way to embed bytes, so an embedded image renders as its alt text while the
12
+ * bytes stay on `document.assets`. Retrieval only reads local relative image
13
+ * paths, so this adapter re-materialises every embedded image into `assets/`
14
+ * and puts a real `![](assets/img-…)` reference back where the document had it.
15
+ */
16
+
17
+ import { createRequire } from 'node:module'
18
+ import { readFile } from 'node:fs/promises'
19
+ import { basename, extname } from 'node:path'
20
+ import type { ConvertWarning, SourceFormat } from '../contract.js'
21
+ import { sha1, sha256 } from '../markdown.js'
22
+ import { AdapterError, type Adapter, type AdapterAvailability, type AdapterContext, type AdapterOutput, type PreparedAsset } from './types.js'
23
+
24
+ export const ANYDOC_EXTENSIONS = [
25
+ // Word
26
+ '.doc', '.docx', '.docm',
27
+ // PowerPoint
28
+ '.ppt', '.pps', '.pot', '.pptx', '.pptm', '.ppsx', '.ppsm',
29
+ // Excel
30
+ '.xls', '.xlsx', '.xlsm', '.xlsb',
31
+ // OpenDocument
32
+ '.odt', '.ods', '.odp',
33
+ // Other containers anydoc reads
34
+ '.rtf', '.epub', '.csv', '.pdf',
35
+ ]
36
+
37
+ interface AnyDocAsset {
38
+ id?: number
39
+ mediaType?: string
40
+ originPart?: string
41
+ data?: Uint8Array
42
+ }
43
+
44
+ interface AnyDocInline {
45
+ kind: string
46
+ alt?: string
47
+ source?: { kind?: string; assetId?: number; url?: string }
48
+ text?: string
49
+ }
50
+
51
+ interface AnyDocBlock {
52
+ kind: string
53
+ content?: AnyDocInline[]
54
+ }
55
+
56
+ interface AnyDocDocument {
57
+ blocks?: AnyDocBlock[]
58
+ notes?: unknown[]
59
+ assets?: AnyDocAsset[]
60
+ }
61
+
62
+ interface AnyDocModule {
63
+ toDocument(bytes: Uint8Array): Promise<AnyDocDocument>
64
+ toMarkdown(path: string, options?: Record<string, unknown>): Promise<string>
65
+ toMarkdownBytes(bytes: Uint8Array, format?: string): Promise<string>
66
+ formatFromBytes(bytes: Uint8Array): string | null
67
+ formatFromPath(path: string): string | null
68
+ }
69
+
70
+ const MEDIA_EXTENSIONS: Record<string, string> = {
71
+ 'image/png': 'png',
72
+ 'image/jpeg': 'jpg',
73
+ 'image/jpg': 'jpg',
74
+ 'image/gif': 'gif',
75
+ 'image/webp': 'webp',
76
+ 'image/tiff': 'tiff',
77
+ 'image/bmp': 'bmp',
78
+ 'image/svg+xml': 'svg',
79
+ 'image/avif': 'avif',
80
+ 'image/heic': 'heic',
81
+ }
82
+
83
+ let modulePromise: Promise<AnyDocModule> | undefined
84
+ let moduleVersion: string | undefined
85
+ let moduleError: string | undefined
86
+
87
+ /** Load the binding once; the platform binary ships as an optional dependency. */
88
+ export async function loadAnydoc(): Promise<AnyDocModule> {
89
+ if (!modulePromise) {
90
+ modulePromise = (async () => {
91
+ const loaded = (await import('@firecrawl/anydoc')) as unknown as AnyDocModule & { default?: AnyDocModule }
92
+ const candidate = typeof loaded.toMarkdown === 'function' ? loaded : loaded.default
93
+ if (!candidate || typeof candidate.toMarkdown !== 'function') throw new Error('找不到 toMarkdown 导出')
94
+ try {
95
+ const require = createRequire(import.meta.url)
96
+ const pkg = JSON.parse(await readFile(require.resolve('@firecrawl/anydoc/package.json'), 'utf8')) as { version?: string }
97
+ moduleVersion = pkg.version
98
+ } catch {
99
+ /* version is cosmetic */
100
+ }
101
+ return candidate
102
+ })().catch(error => {
103
+ moduleError = error instanceof Error ? error.message : String(error)
104
+ modulePromise = undefined
105
+ throw error
106
+ })
107
+ }
108
+ return await modulePromise
109
+ }
110
+
111
+ export async function anydocAvailability(): Promise<AdapterAvailability> {
112
+ try {
113
+ await loadAnydoc()
114
+ return { available: true, ...(moduleVersion ? { versions: { '@firecrawl/anydoc': moduleVersion } } : {}) }
115
+ } catch (cause) {
116
+ return {
117
+ available: false,
118
+ reason: `无法加载 @firecrawl/anydoc:${moduleError ?? (cause instanceof Error ? cause.message : String(cause))}`,
119
+ suggestion: '在 packages/dsh-convert-core 目录执行 pnpm install;Apple Silicon 需要 @firecrawl/anydoc-darwin-arm64(随 optionalDependencies 自动安装)。',
120
+ }
121
+ }
122
+ }
123
+
124
+ function extensionFor(asset: AnyDocAsset): string | undefined {
125
+ const mediaType = asset.mediaType?.split(';')[0]?.trim().toLowerCase()
126
+ if (mediaType && MEDIA_EXTENSIONS[mediaType]) return MEDIA_EXTENSIONS[mediaType]
127
+ const fromPart = asset.originPart ? /\.([a-zA-Z0-9]{2,5})$/.exec(asset.originPart)?.[1]?.toLowerCase() : undefined
128
+ if (fromPart && fromPart !== 'bin') return fromPart
129
+ return undefined
130
+ }
131
+
132
+ /** In document order, every inline image that points at an embedded asset. */
133
+ function collectEmbeddedImages(document: AnyDocDocument): Array<{ assetId: number; alt: string; asset: AnyDocAsset }> {
134
+ const found: Array<{ assetId: number; alt: string; asset: AnyDocAsset }> = []
135
+ for (const block of document.blocks ?? []) {
136
+ for (const inline of block.content ?? []) {
137
+ if (inline.kind !== 'image') continue
138
+ const source = inline.source
139
+ if (!source || source.kind !== 'asset' || typeof source.assetId !== 'number') continue
140
+ const asset = document.assets?.[source.assetId]
141
+ if (!asset) continue
142
+ found.push({ assetId: source.assetId, alt: (inline.alt ?? '').trim(), asset })
143
+ }
144
+ }
145
+ return found
146
+ }
147
+
148
+ export interface ImagePlacement {
149
+ markdown: string
150
+ placed: number
151
+ appended: number
152
+ warnings: string[]
153
+ }
154
+
155
+ /**
156
+ * Put real image references back into the Markdown.
157
+ *
158
+ * anydoc renders an embedded image as its alt text, so placement is a match on
159
+ * that text. A standalone-line match is used only when it is unambiguous;
160
+ * anything that cannot be located is appended under a `## 图片` heading rather
161
+ * than guessed at, and the count is reported.
162
+ */
163
+ export function placeImageReferences(markdown: string, images: Array<{ rel: string; alt: string }>): ImagePlacement {
164
+ const lines = markdown.split('\n')
165
+ const consumed = new Set<number>()
166
+ const remaining: Array<{ rel: string; alt: string }> = []
167
+ const warnings: string[] = []
168
+
169
+ for (const image of images) {
170
+ const alt = image.alt.trim()
171
+ let index = -1
172
+ if (alt) {
173
+ // Only accept a line that is exactly the alt text, and only once.
174
+ const candidates = lines
175
+ .map((line, position) => (line.trim() === alt && !consumed.has(position) ? position : -1))
176
+ .filter(position => position >= 0)
177
+ if (candidates.length === 1) index = candidates[0]!
178
+ }
179
+ if (index >= 0) {
180
+ consumed.add(index)
181
+ lines[index] = `![${alt}](${image.rel})`
182
+ continue
183
+ }
184
+ remaining.push(image)
185
+ }
186
+
187
+ let output = lines.join('\n')
188
+ if (remaining.length > 0) {
189
+ const section = remaining.map(image => `![${image.alt || basename(image.rel)}](${image.rel})`).join('\n\n')
190
+ output = `${output.trimEnd()}\n\n## 图片\n\n${section}\n`
191
+ warnings.push(
192
+ `${remaining.length} 张内嵌图片在正文中没有可定位的占位(无 alt 文本或 alt 不唯一),已追加到文末「图片」小节。`,
193
+ )
194
+ }
195
+ return { markdown: output, placed: images.length - remaining.length, appended: remaining.length, warnings }
196
+ }
197
+
198
+ function sourceFormatFor(anydocFormat: string | null, filename: string): SourceFormat {
199
+ const format = anydocFormat ?? extname(filename).replace('.', '').toLowerCase()
200
+ if (format === 'pdf') return 'pdf'
201
+ if (format === 'csv') return 'text'
202
+ return 'office'
203
+ }
204
+
205
+ /** anydoc's error codes are already precise; keep them instead of flattening. */
206
+ const ANYDOC_SUGGESTIONS: Record<string, string> = {
207
+ needsOcr:
208
+ '该 PDF 是扫描件/纯图片页,anydoc 本地不做 OCR(它只覆盖文本层 PDF)。走扫描档:把 adapter 指定为 "scanned",' +
209
+ '或设环境变量 SCANNED_OCR=1 让 .pdf 直接走它 —— 扫描档用 pypdfium2 渲染页图 + 本地 RapidOCR 识别文字,不出网,' +
210
+ '但不做版面还原;需要版面/表格/公式时见 issues/0017 的可选高质量档。' +
211
+ '不要使用 anydoc 的 ocr:"hosted":它会把整份文档发往 Firecrawl Parse 云端,与 0015 的本地优先原则冲突。',
212
+ encrypted: '文档有密码保护,请先去掉密码再转换。',
213
+ unsupported: '该格式不在本版范围内(doc/docx/ppt/pptx/xls/xlsx/odt/ods/odp/rtf/epub/csv/文本层 PDF)。',
214
+ malformed: '文件结构损坏,无法提取有意义的内容。',
215
+ resourceLimit: '文件触发了 anydoc 的解压/嵌套安全上限,可能是文档炸弹或异常嵌套。',
216
+ missingPart: '文档缺少必需的结构部件(可能被截断或由非标准工具生成)。',
217
+ }
218
+
219
+ function toAdapterError(cause: unknown): AdapterError {
220
+ const error = cause as Error & { code?: string; pages?: number[] }
221
+ const code = typeof error?.code === 'string' ? error.code : undefined
222
+ if (code && code !== 'io') {
223
+ const pages = Array.isArray(error.pages) && error.pages.length > 0 ? `(第 ${error.pages.slice(0, 8).join(', ')} 页)` : ''
224
+ return new AdapterError(
225
+ `ANYDOC_${code.replace(/([a-z])([A-Z])/g, '$1_$2').toUpperCase()}`,
226
+ `${error.message || code}${pages}`,
227
+ ANYDOC_SUGGESTIONS[code],
228
+ )
229
+ }
230
+ return new AdapterError('ANYDOC_FAILED', error instanceof Error ? error.message : String(cause), ANYDOC_SUGGESTIONS.malformed)
231
+ }
232
+
233
+ async function run(context: AdapterContext): Promise<AdapterOutput> {
234
+ const sourcePath = context.inputPath
235
+ if (!sourcePath) throw new AdapterError('INPUT_REQUIRED', '文档适配器需要文件路径。')
236
+
237
+ const anydoc = await loadAnydoc().catch(cause => {
238
+ throw new AdapterError(
239
+ 'DEPENDENCY_MISSING',
240
+ `无法加载 @firecrawl/anydoc:${cause instanceof Error ? cause.message : String(cause)}`,
241
+ '在 packages/dsh-convert-core 目录执行 pnpm install。',
242
+ )
243
+ })
244
+
245
+ const filename = context.filename || basename(sourcePath)
246
+ const bytes = await readFile(sourcePath)
247
+ context.report(0.15, 'detect', '识别文档格式')
248
+
249
+ let detected: string | null = null
250
+ try {
251
+ detected = anydoc.formatFromBytes(new Uint8Array(bytes))
252
+ } catch {
253
+ detected = null
254
+ }
255
+
256
+ context.report(0.35, 'convert', `解析 ${detected ?? extname(filename)}`)
257
+ let markdown: string
258
+ let document: AnyDocDocument = {}
259
+ const assetNote: string[] = []
260
+ try {
261
+ markdown = await anydoc.toMarkdown(sourcePath)
262
+ } catch (cause) {
263
+ throw toAdapterError(cause)
264
+ }
265
+ // `toDocument` covers the OOXML / OpenDocument / RTF / EPUB / CSV family. PDF
266
+ // renders straight to Markdown through pdf-inspector and *rejects* the
267
+ // document model, so a PDF simply has no embedded assets to place.
268
+ if (detected !== 'pdf') {
269
+ try {
270
+ document = await anydoc.toDocument(new Uint8Array(bytes))
271
+ } catch (cause) {
272
+ const error = cause as Error & { code?: string }
273
+ // Never lose a good Markdown because the asset model was unavailable.
274
+ assetNote.push(
275
+ `内嵌资源枚举失败(${error?.code ?? 'unknown'}: ${error?.message ?? String(cause)}),本次只交付 Markdown,未落地图片。`,
276
+ )
277
+ }
278
+ }
279
+ if (!markdown.trim()) {
280
+ throw new AdapterError('CONTENT_REQUIRED', '文档解析后没有内容(可能全为图片或受保护内容)。')
281
+ }
282
+
283
+ // ---- asset glue: bytes live on the document model, not in the Markdown ----
284
+ const embedded = collectEmbeddedImages(document)
285
+ const images: Array<{ rel: string; alt: string }> = []
286
+ const assets: PreparedAsset[] = []
287
+ const warnings: ConvertWarning[] = []
288
+ let skippedObjects = 0
289
+
290
+ for (const entry of embedded) {
291
+ const data = entry.asset.data ? Buffer.from(entry.asset.data) : undefined
292
+ if (!data || data.length === 0) {
293
+ skippedObjects += 1
294
+ continue
295
+ }
296
+ if (!entry.asset.mediaType?.startsWith('image/')) {
297
+ // 0013 only consumes images; other embedded objects are reported, not dropped.
298
+ skippedObjects += 1
299
+ continue
300
+ }
301
+ const extension = extensionFor(entry.asset)
302
+ if (!extension) {
303
+ skippedObjects += 1
304
+ continue
305
+ }
306
+ const digest = sha1(data)
307
+ const rel = `assets/img-${digest.slice(0, 12)}.${extension}`
308
+ images.push({ rel, alt: entry.alt })
309
+ assets.push({
310
+ rel,
311
+ data,
312
+ ...(entry.asset.mediaType ? { contentType: entry.asset.mediaType } : {}),
313
+ })
314
+ }
315
+
316
+ const placement = placeImageReferences(markdown, images)
317
+ for (const note of assetNote) warnings.push({ code: 'ASSET_ENUMERATION_FAILED', message: note })
318
+ if (placement.appended > 0) {
319
+ warnings.push({ code: 'IMAGES_APPENDED', message: placement.warnings[0]! })
320
+ }
321
+ if (skippedObjects > 0) {
322
+ warnings.push({
323
+ code: 'EMBEDDED_OBJECTS_SKIPPED',
324
+ message: `文档里有 ${skippedObjects} 个内嵌对象不是可落地的图片(非 image/* 或缺少字节),未写入 assets/。`,
325
+ })
326
+ }
327
+ if (embedded.length > 0) {
328
+ warnings.push({
329
+ code: 'IMAGES_LOCALIZED',
330
+ message: `内嵌图片 ${embedded.length} 张已落地为本地 assets/ 并改写正文引用(${placement.placed} 张原位、${placement.appended} 张追加)。`,
331
+ })
332
+ }
333
+
334
+ const heading = /^#{1,6}\s+(.+)$/m.exec(placement.markdown)?.[1]?.trim()
335
+ return {
336
+ title: (context.request.title?.trim() || heading || basename(filename, extname(filename))).slice(0, 300),
337
+ markdown: placement.markdown.trim(),
338
+ sourceFormat: sourceFormatFor(detected, filename),
339
+ parser: `anydoc@${moduleVersion ?? 'unknown'}/${detected ?? 'unknown'}`,
340
+ sourcePath,
341
+ warnings,
342
+ dependencyVersions: { '@firecrawl/anydoc': moduleVersion ?? 'unknown' },
343
+ externalId: `sha256:${sha256(bytes)}`,
344
+ assets,
345
+ }
346
+ }
347
+
348
+ export const anydocAdapter: Adapter = {
349
+ id: 'anydoc',
350
+ description: 'DOC/DOCX/PPT/PPTX/XLS/XLSX/ODT/ODS/ODP/RTF/EPUB/CSV/文本层 PDF → GFM Markdown(Rust 内核,无模型、无外部二进制;内嵌图片落地到 assets/)',
351
+ kinds: ['file'],
352
+ extensions: ANYDOC_EXTENSIONS,
353
+ availability: () => ({ available: true, ...(moduleVersion ? { versions: { '@firecrawl/anydoc': moduleVersion } } : {}) }),
354
+ probe: anydocAvailability,
355
+ run,
356
+ }