opencode-ufr 0.2.9 → 0.2.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +55 -62
- package/package.json +14 -6
- package/skills/pdf2md/SKILL.md +32 -0
- package/src/client/gateway.ts +101 -0
- package/src/client/pdf2md.ts +747 -0
- package/src/daemon/catalog-source.ts +9 -0
- package/src/daemon/daemon.ts +70 -18
- package/src/daemon/keypool.ts +12 -7
- package/src/daemon/main.ts +7 -1
- package/src/daemon/probe-all.ts +60 -38
- package/src/daemon/router.ts +215 -17
- package/src/daemon/server.ts +33 -1
- package/src/daemon/stats.ts +13 -1
- package/src/daemon/upstream.ts +14 -3
- package/src/daemon/vpn/fortinet.ts +42 -29
- package/src/daemon/vpn/manager.ts +50 -16
- package/src/daemon/vpn/proxy.ts +3 -1
- package/src/plugin/connect.ts +4 -4
- package/src/plugin/ensure-daemon.ts +48 -19
- package/src/plugin/index.ts +121 -46
- package/src/plugin/tui.tsx +77 -0
- package/src/shared/config.ts +53 -4
- package/src/shared/connect.ts +3 -4
- package/src/shared/daemon-client.ts +29 -0
- package/src/shared/errors.ts +1 -1
- package/src/shared/fs.ts +8 -3
- package/src/shared/keys.ts +65 -0
- package/src/shared/models-file.ts +1 -1
- package/src/shared/secrets.ts +2 -6
- package/bin/ufr.ts +0 -4
- package/src/cli/catalog.ts +0 -44
- package/src/cli/connect.ts +0 -146
- package/src/cli/context-probe.ts +0 -96
- package/src/cli/daemon-client.ts +0 -19
- package/src/cli/index.ts +0 -105
- package/src/cli/io.ts +0 -98
- package/src/cli/keys.ts +0 -130
- package/src/cli/stats.ts +0 -37
- package/src/cli/status.ts +0 -48
- package/src/cli/stop.ts +0 -8
- package/src/cli/ufr-check.ts +0 -43
|
@@ -0,0 +1,747 @@
|
|
|
1
|
+
#!/usr/bin/env bun
|
|
2
|
+
/**
|
|
3
|
+
* pdf2md — agentic PDF-to-Markdown converter on UFR vision models.
|
|
4
|
+
*
|
|
5
|
+
* TypeScript port of the author's former standalone Python script, integrated
|
|
6
|
+
* with the opencode-ufr gateway: every model call goes through the gateway's
|
|
7
|
+
* relay (POST /v1/_relay), so key rotation and soft rate limiting are handled
|
|
8
|
+
* centrally — pages run at full parallelism and the caller never paces itself.
|
|
9
|
+
*
|
|
10
|
+
* Pipeline per page:
|
|
11
|
+
* 1. Render at 300 DPI (pdftoppm)
|
|
12
|
+
* 2. Deduplicate incremental-reveal pages (PowerPoint-style, magick compare)
|
|
13
|
+
* 3. Classify: TEXT / TABLE_MATH / DIAGRAM / MIXED (qwen-3.8-27b)
|
|
14
|
+
* 4. Route to the best model per type:
|
|
15
|
+
* - TEXT → glm-5.3-flash (structure + LaTeX, 3-4x faster than the old OCR model)
|
|
16
|
+
* - TABLE_MATH → glm-5.3-flash + refinement
|
|
17
|
+
* - DIAGRAM → 3-model ensemble (glm-5.3-flash, deepseek-v4.1-flash,
|
|
18
|
+
* qwen-3.5-397b) full-page + quadrant zoom, adjudicated
|
|
19
|
+
* - MIXED → OCR + diagram ensemble, merged
|
|
20
|
+
* 5. Refinement loop (gemma-4-31b compares markdown against the page image)
|
|
21
|
+
* 6. Cross-page table merge
|
|
22
|
+
*
|
|
23
|
+
* External tools: pdftoppm/pdfinfo (poppler-utils) and magick (ImageMagick).
|
|
24
|
+
*
|
|
25
|
+
* Usage:
|
|
26
|
+
* bun src/client/pdf2md.ts <input.pdf> [output.md] [--workers N] [--dpi 300]
|
|
27
|
+
* [--no-dedup] [--dedup-threshold 0.98] [--no-refine] [--no-merge] [--verbose]
|
|
28
|
+
*/
|
|
29
|
+
import { execFile as execFileCb } from "node:child_process"
|
|
30
|
+
import { mkdir, mkdtemp, readdir, rm } from "node:fs/promises"
|
|
31
|
+
import { tmpdir } from "node:os"
|
|
32
|
+
import { join } from "node:path"
|
|
33
|
+
import { promisify } from "node:util"
|
|
34
|
+
import { connectGateway, type Gateway, relay } from "./gateway"
|
|
35
|
+
|
|
36
|
+
const execFile = promisify(execFileCb)
|
|
37
|
+
|
|
38
|
+
// ─── Configuration ───────────────────────────────────────────────────────────
|
|
39
|
+
|
|
40
|
+
const OCR_MODEL = process.env.PDF2MD_OCR_MODEL ?? "glm-5.3-flash-llmlb"
|
|
41
|
+
const CLASSIFY_MODEL = process.env.PDF2MD_CLASSIFY_MODEL ?? "qwen-3.8-27b-llmlb"
|
|
42
|
+
const REFINE_MODEL = process.env.PDF2MD_REFINE_MODEL ?? "gemma-4-31b-llmlb"
|
|
43
|
+
|
|
44
|
+
const MIME: Record<string, string> = {
|
|
45
|
+
".png": "image/png",
|
|
46
|
+
".jpg": "image/jpeg",
|
|
47
|
+
".jpeg": "image/jpeg",
|
|
48
|
+
".webp": "image/webp",
|
|
49
|
+
}
|
|
50
|
+
const DEFAULT_DPI = 300
|
|
51
|
+
const MAX_REFINE_ROUNDS = 3
|
|
52
|
+
const DEDUP_SIMILARITY_THRESHOLD = 0.90
|
|
53
|
+
const REQUEST_TIMEOUT_MS = 600_000 // diagrams can think for minutes
|
|
54
|
+
const DEFAULT_MAX_TOKENS = 8192 // generous budget so reasoning models don't run out before visible output
|
|
55
|
+
|
|
56
|
+
// ─── Logging ─────────────────────────────────────────────────────────────────
|
|
57
|
+
|
|
58
|
+
let VERBOSE = false
|
|
59
|
+
const log = (msg: string, force = false): void => {
|
|
60
|
+
if (force || VERBOSE) console.error(`[pdf2md] ${msg}`)
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
const logProgress = (current: number, total: number, label = ""): void => {
|
|
64
|
+
const pct = total ? Math.floor((current * 100) / total) : 0
|
|
65
|
+
const tag = label ? ` ${label}` : ""
|
|
66
|
+
process.stderr.write(`\r[pdf2md] ${current}/${total} (${pct}%)${tag} `)
|
|
67
|
+
if (current >= total) process.stderr.write("\n")
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// ─── Model calls (all through the gateway's relay: key rotation + pacing) ────
|
|
71
|
+
|
|
72
|
+
type Msg = { role: string; content: unknown }
|
|
73
|
+
|
|
74
|
+
/** One raw model call, returning the upstream JSON body as text. Injectable for tests. */
|
|
75
|
+
export type RelayCall = (model: string, messages: Msg[], maxTokens: number) => Promise<{ status: number; body: string; retryAfterMs?: number }>
|
|
76
|
+
|
|
77
|
+
let gateway: Gateway | null = null
|
|
78
|
+
|
|
79
|
+
const gatewayCall: RelayCall = async (model, messages, maxTokens) => {
|
|
80
|
+
gateway ??= await connectGateway()
|
|
81
|
+
return relay(gateway, { model, messages, max_tokens: maxTokens, temperature: 0 }, { timeoutMs: REQUEST_TIMEOUT_MS })
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
type ChatOptions = {
|
|
85
|
+
maxTokens?: number
|
|
86
|
+
/** Classification only: when the model thinks past its budget and emits no
|
|
87
|
+
* visible content, its reasoning_content is parsed for the category instead.
|
|
88
|
+
* For content generation reasoning is deliberation, never the answer. */
|
|
89
|
+
allowReasoningFallback?: boolean
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** Chat completion with the Python script's retry ladder, via the relay.
|
|
93
|
+
* The gateway already waits for key slots — a 429 here means a wall, a pool
|
|
94
|
+
* cap or an exhausted budget, all of which are worth a real backoff. */
|
|
95
|
+
export async function apiChat(call: RelayCall, model: string, messages: Msg[], o: ChatOptions = {}): Promise<string> {
|
|
96
|
+
const base = { model, temperature: 0, messages }
|
|
97
|
+
let effectiveMax = o.maxTokens ?? DEFAULT_MAX_TOKENS
|
|
98
|
+
let lastError = ""
|
|
99
|
+
for (let attempt = 0; attempt < 3; attempt++) {
|
|
100
|
+
if (attempt > 0) effectiveMax *= 2 ** attempt
|
|
101
|
+
let res: Awaited<ReturnType<RelayCall>>
|
|
102
|
+
try {
|
|
103
|
+
res = await call(model, messages, effectiveMax)
|
|
104
|
+
} catch (e) {
|
|
105
|
+
lastError = `network error: ${(e as Error).message}`
|
|
106
|
+
await Bun.sleep(2 ** attempt * 1000)
|
|
107
|
+
continue
|
|
108
|
+
}
|
|
109
|
+
if (res.status === 429 || res.status === 503) {
|
|
110
|
+
const waitS = Math.min(res.retryAfterMs ? Math.round(res.retryAfterMs / 1000) : 2 ** attempt * 5, 120)
|
|
111
|
+
log(`rate-limited (${res.status}), waiting ${waitS}s before retry...`)
|
|
112
|
+
await Bun.sleep(waitS * 1000)
|
|
113
|
+
lastError = `HTTP ${res.status}`
|
|
114
|
+
continue
|
|
115
|
+
}
|
|
116
|
+
if (res.status !== 200) {
|
|
117
|
+
lastError = `HTTP ${res.status}: ${res.body.slice(0, 500)}`
|
|
118
|
+
continue
|
|
119
|
+
}
|
|
120
|
+
let data: {
|
|
121
|
+
choices?: { message?: { content?: unknown; reasoning_content?: unknown }; finish_reason?: string }[]
|
|
122
|
+
}
|
|
123
|
+
try {
|
|
124
|
+
data = JSON.parse(res.body)
|
|
125
|
+
} catch {
|
|
126
|
+
lastError = "response was not JSON"
|
|
127
|
+
continue
|
|
128
|
+
}
|
|
129
|
+
const choice = data.choices?.[0]
|
|
130
|
+
if (!choice) {
|
|
131
|
+
lastError = "no choices in response"
|
|
132
|
+
continue
|
|
133
|
+
}
|
|
134
|
+
const content = choice.message?.content
|
|
135
|
+
if (typeof content === "string" && content.trim()) return content
|
|
136
|
+
const reasoning = choice.message?.reasoning_content
|
|
137
|
+
if (typeof reasoning === "string" && reasoning.trim() && o.allowReasoningFallback) {
|
|
138
|
+
log(` model ${model}: content empty, using reasoning_content (${reasoning.length} chars)`)
|
|
139
|
+
return reasoning.trim()
|
|
140
|
+
}
|
|
141
|
+
if (choice.finish_reason === "length") {
|
|
142
|
+
log(` model ${model}: truncated (finish=length), retrying with larger budget`)
|
|
143
|
+
lastError = "truncated (finish=length)"
|
|
144
|
+
continue
|
|
145
|
+
}
|
|
146
|
+
lastError = `empty content (finish=${choice.finish_reason})`
|
|
147
|
+
}
|
|
148
|
+
throw new Error(`API call failed after 3 attempts (model=${model}): ${lastError}`)
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const imagePart = (b64: string, mime: string) => ({ type: "image_url", image_url: { url: `data:${mime};base64,${b64}` } })
|
|
152
|
+
|
|
153
|
+
const apiChatImage = async (call: RelayCall, model: string, b64: string, mime: string, prompt: string, o: ChatOptions = {}) =>
|
|
154
|
+
apiChat(call, model, [{ role: "user", content: [{ type: "text", text: prompt }, imagePart(b64, mime)] }], o)
|
|
155
|
+
|
|
156
|
+
const apiChatImages = (call: RelayCall, model: string, images: [string, string][], prompt: string, o: ChatOptions = {}) =>
|
|
157
|
+
apiChat(call, model, [{ role: "user", content: [{ type: "text", text: prompt }, ...images.map(([b64, mime]) => imagePart(b64, mime))] }], o)
|
|
158
|
+
|
|
159
|
+
// ─── Image utilities (no image library, use ImageMagick) ─────────────────────
|
|
160
|
+
|
|
161
|
+
const cropImage = async (srcPng: string, outputPng: string, x: number, y: number, w: number, h: number): Promise<boolean> => {
|
|
162
|
+
try {
|
|
163
|
+
await execFile("magick", ["convert", srcPng, "-crop", `${w}x${h}+${x}+${y}`, "+repage", outputPng], { timeout: 30_000 })
|
|
164
|
+
return true
|
|
165
|
+
} catch (e) {
|
|
166
|
+
log(`crop failed: ${(e as Error).message}`)
|
|
167
|
+
return false
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
const imageToB64 = async (path: string): Promise<[string, string]> => {
|
|
172
|
+
const mime = MIME[path.slice(path.lastIndexOf(".")).toLowerCase()] ?? "image/png"
|
|
173
|
+
const buf = await Bun.file(path).arrayBuffer()
|
|
174
|
+
return [Buffer.from(buf).toString("base64"), mime]
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
const renderPdfPages = async (pdfPath: string, dpi: number, outDir: string): Promise<string[]> => {
|
|
178
|
+
const prefix = join(outDir, "page")
|
|
179
|
+
try {
|
|
180
|
+
await execFile("pdftoppm", ["-png", "-r", String(dpi), pdfPath, prefix], { timeout: 600_000 })
|
|
181
|
+
} catch (e) {
|
|
182
|
+
throw new Error(`pdftoppm failed (${(e as Error).message}). Install poppler-utils.`)
|
|
183
|
+
}
|
|
184
|
+
const names = (await readdir(outDir)).filter((n) => n.startsWith("page") && n.endsWith(".png")).sort()
|
|
185
|
+
if (names.length === 0) throw new Error("pdftoppm produced no output images")
|
|
186
|
+
return names.map((n) => join(outDir, n))
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const getPageCount = async (pdfPath: string): Promise<number> => {
|
|
190
|
+
try {
|
|
191
|
+
const { stdout } = await execFile("pdfinfo", [pdfPath], { timeout: 30_000 })
|
|
192
|
+
for (const line of stdout.split("\n")) {
|
|
193
|
+
if (line.startsWith("Pages:")) return Number.parseInt(line.split(":")[1]!.trim(), 10)
|
|
194
|
+
}
|
|
195
|
+
} catch {
|
|
196
|
+
// fall through
|
|
197
|
+
}
|
|
198
|
+
return 0
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
const getImageDims = async (pngPath: string): Promise<[number, number]> => {
|
|
202
|
+
try {
|
|
203
|
+
const { stdout } = await execFile("magick", ["identify", "-format", "%w %h", pngPath], { timeout: 10_000 })
|
|
204
|
+
const [w, h] = stdout.trim().split(/\s+/).map(Number)
|
|
205
|
+
return [w ?? 0, h ?? 0]
|
|
206
|
+
} catch {
|
|
207
|
+
return [0, 0]
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
// ─── Stage 1b: Incremental-reveal deduplication ──────────────────────────────
|
|
212
|
+
|
|
213
|
+
/** Normalized RMSE via ImageMagick: similarity = 1 - rmse/255, in [0, 1]. */
|
|
214
|
+
const imageSimilarity = async (imgA: string, imgB: string): Promise<number> => {
|
|
215
|
+
try {
|
|
216
|
+
// magick compare writes the metric to stderr and exits non-zero on difference
|
|
217
|
+
const res = await execFile("magick", ["compare", "-metric", "RMSE", imgA, imgB, "null:"], { timeout: 30_000 }).catch(
|
|
218
|
+
(e: { stderr?: string }) => ({ stderr: e.stderr ?? "" }),
|
|
219
|
+
)
|
|
220
|
+
const line = (res.stderr ?? "").trim()
|
|
221
|
+
const normalized = /\(([\d.]+)\)/.exec(line)?.[1]
|
|
222
|
+
if (normalized !== undefined) return Math.max(0, 1 - Number.parseFloat(normalized))
|
|
223
|
+
const raw = Number.parseFloat(line.split(/\s+/)[0] ?? "65535")
|
|
224
|
+
return Math.max(0, 1 - (Number.isFinite(raw) ? raw : 65535) / 65535)
|
|
225
|
+
} catch {
|
|
226
|
+
return 0
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/** PowerPoint-style incremental reveals: consecutive near-identical pages — keep the last of each run. */
|
|
231
|
+
export const dedupIncrementalReveals = async (pagePngs: string[], threshold = DEDUP_SIMILARITY_THRESHOLD): Promise<string[]> => {
|
|
232
|
+
if (pagePngs.length <= 1) return pagePngs
|
|
233
|
+
const kept: string[] = []
|
|
234
|
+
let skipped = 0
|
|
235
|
+
for (let i = 0; i < pagePngs.length; i++) {
|
|
236
|
+
const current = pagePngs[i]!
|
|
237
|
+
if (i === pagePngs.length - 1) {
|
|
238
|
+
kept.push(current) // the last page of a run has all the content
|
|
239
|
+
break
|
|
240
|
+
}
|
|
241
|
+
const sim = await imageSimilarity(current, pagePngs[i + 1]!)
|
|
242
|
+
if (sim >= threshold) {
|
|
243
|
+
log(` page ${i + 1} → ${i + 2}: similarity ${sim.toFixed(4)} ≥ ${threshold}, dropping page ${i + 1}`)
|
|
244
|
+
skipped++
|
|
245
|
+
} else {
|
|
246
|
+
kept.push(current)
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
log(skipped > 0 ? `dedup: removed ${skipped} incremental-reveal duplicates, ${kept.length} pages remain` : "dedup: no incremental-reveal duplicates detected")
|
|
250
|
+
return kept
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
// ─── Stage 2: Page classification ────────────────────────────────────────────
|
|
254
|
+
|
|
255
|
+
const CLASSIFY_PROMPT = `Analyze this document page image and classify it into exactly ONE of these categories:
|
|
256
|
+
|
|
257
|
+
- TEXT: Mostly text (paragraphs, headings, body text). May contain simple inline figures.
|
|
258
|
+
- TABLE_MATH: Contains significant tables, mathematical equations, or structured data.
|
|
259
|
+
- DIAGRAM: The page's meaning is carried by a drawing: technical drawings, circuit diagrams, schematics, flowcharts, block diagrams, node graphs, state machines, org charts, or complex figures with labels. Boxes connected by arrows count as DIAGRAM even when drawn simply.
|
|
260
|
+
- MIXED: Contains both significant text AND a technical diagram/figure.
|
|
261
|
+
|
|
262
|
+
Respond with ONLY the category name (TEXT, TABLE_MATH, DIAGRAM, or MIXED). No other text.`
|
|
263
|
+
|
|
264
|
+
export type PageType = "TEXT" | "TABLE_MATH" | "DIAGRAM" | "MIXED"
|
|
265
|
+
|
|
266
|
+
export const classifyPage = async (call: RelayCall, pagePng: string): Promise<PageType> => {
|
|
267
|
+
const [b64, mime] = await imageToB64(pagePng)
|
|
268
|
+
try {
|
|
269
|
+
const result = (await apiChatImage(call, CLASSIFY_MODEL, b64, mime, CLASSIFY_PROMPT, { maxTokens: 50, allowReasoningFallback: true })).trim().toUpperCase()
|
|
270
|
+
for (const valid of ["TEXT", "TABLE_MATH", "DIAGRAM", "MIXED"] as const) {
|
|
271
|
+
if (result.includes(valid)) return valid
|
|
272
|
+
}
|
|
273
|
+
return "TEXT"
|
|
274
|
+
} catch (e) {
|
|
275
|
+
log(`classification failed for ${pagePng.split("/").pop()}: ${(e as Error).message}, defaulting to TEXT`)
|
|
276
|
+
return "TEXT"
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
// ─── Stage 3: Per-page-type transcription ────────────────────────────────────
|
|
281
|
+
|
|
282
|
+
const OCR_PROMPT = `Convert this document page to clean Markdown.
|
|
283
|
+
Rules:
|
|
284
|
+
- Preserve all headings, lists, tables (as GitHub markdown tables), and code blocks
|
|
285
|
+
- For math, use LaTeX: $inline$ and $$block$$
|
|
286
|
+
- For figures, output  with a caption if present
|
|
287
|
+
- For footnotes, use [^N] syntax
|
|
288
|
+
- Output ONLY the markdown, no commentary
|
|
289
|
+
- If a table spans the page width, reproduce every cell accurately
|
|
290
|
+
- Page {page_info}`
|
|
291
|
+
|
|
292
|
+
const DIAGRAM_PROMPT = `You are analyzing a page from a technical document that contains a circuit diagram, schematic, flowchart, or node-based graph.
|
|
293
|
+
|
|
294
|
+
Your task:
|
|
295
|
+
1. First, transcribe ALL text on the page (headings, paragraphs, captions, labels) as Markdown.
|
|
296
|
+
2. For every visible component (resistors, capacitors, ICs, connectors, ...), read: its reference designator (e.g. R1, C2, U3), its value (e.g. 10kΩ, 100nF), and any pin labels. Output a structured component table:
|
|
297
|
+
|
|
298
|
+
| Ref | Type | Value | Notes |
|
|
299
|
+
|-----|------|-------|-------|
|
|
300
|
+
| R1 | Resistor | 10kΩ | |
|
|
301
|
+
|
|
302
|
+
3. Describe the wiring: every connection you can trace as "A connects to B" (use reference designators and net labels like VIN, GND).
|
|
303
|
+
4. If the page contains a flowchart or node-based graph, add a "## Graph" section:
|
|
304
|
+
- every node: its label text exactly as written, and its type (process / decision / input / output / start / end)
|
|
305
|
+
- every edge as "source -> target" with the edge label (e.g. yes/no) if present
|
|
306
|
+
- the whole graph as a Mermaid flowchart (graph TD) with the exact labels
|
|
307
|
+
5. End with: 
|
|
308
|
+
6. Output ONLY Markdown, no commentary.
|
|
309
|
+
|
|
310
|
+
Page {page_info}`
|
|
311
|
+
|
|
312
|
+
const QUADRANT_PROMPT = (label: string): string =>
|
|
313
|
+
`Focus on this ${label} quadrant of a technical diagram. List every component label, value, and connection you can read. Output as a Markdown table with columns: Ref, Type, Value, Connection. Only output the table, nothing else.`
|
|
314
|
+
|
|
315
|
+
/** The ensemble: three independent vision models at two zoom levels disagree in
|
|
316
|
+
* complementary ways (measured 2026-10-06: glm-5.3-flash read resistor values
|
|
317
|
+
* right on the full page where deepseek/qwen misread, all three agreed on the
|
|
318
|
+
* quadrant crops that fixed the full-page misreads). The adjudicator sees the
|
|
319
|
+
* page plus every reading and resolves conflicts against the image. */
|
|
320
|
+
const DIAGRAM_ENSEMBLE = (process.env.PDF2MD_DIAGRAM_ENSEMBLE ?? "glm-5.3-flash-llmlb,deepseek-v4.1-flash-llmlb,qwen-3.5-397b-llmlb")
|
|
321
|
+
.split(",").map((s) => s.trim()).filter(Boolean)
|
|
322
|
+
const ADJUDICATOR_MODEL = process.env.PDF2MD_ADJUDICATOR_MODEL ?? "glm-5.3-flash-llmlb"
|
|
323
|
+
|
|
324
|
+
const pageInfo = (page: number, total: number): string => (total > 1 ? `(Seite ${page} von ${total})` : "")
|
|
325
|
+
|
|
326
|
+
const transcribeTextPage = async (call: RelayCall, pagePng: string, page: number, total: number): Promise<string> => {
|
|
327
|
+
const [b64, mime] = await imageToB64(pagePng)
|
|
328
|
+
const info = pageInfo(page, total)
|
|
329
|
+
const prompt = OCR_PROMPT.replace("{page_info}", info ? ` ${info}` : "")
|
|
330
|
+
return apiChatImage(call, OCR_MODEL, b64, mime, prompt)
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
/** Diagram/graph page: the full-accuracy path.
|
|
334
|
+
*
|
|
335
|
+
* Three independent vision models read the full page in parallel, then all
|
|
336
|
+
* three read each of the four native-resolution quadrant crops (12 more
|
|
337
|
+
* calls). Each scale/model combination misreads different labels — measured
|
|
338
|
+
* 2026-10-06 on a schematic with R1 10kΩ/R2 4.7kΩ/C1 10µF: glm-5.3-flash read
|
|
339
|
+
* the resistors right on the full page where the others dropped the "k", the
|
|
340
|
+
* quadrant crops fixed R1/C1 for everyone but all three dropped the "k" there
|
|
341
|
+
* instead. The adjudicator then sees the page image plus every reading,
|
|
342
|
+
* re-examines each disagreement against the image, normalizes equivalent
|
|
343
|
+
* units (0.1uF = 100nF) and emits the final structure. */
|
|
344
|
+
export const transcribeDiagramPage = async (call: RelayCall, pagePng: string, page: number, total: number): Promise<string> => {
|
|
345
|
+
const [b64, mime] = await imageToB64(pagePng)
|
|
346
|
+
const prompt = DIAGRAM_PROMPT.replace("{page_info}", pageInfo(page, total)).replace("{page_num}", String(page))
|
|
347
|
+
|
|
348
|
+
// 1. Full-page structured pass: all ensemble models in parallel.
|
|
349
|
+
const fullReads = await Promise.allSettled(DIAGRAM_ENSEMBLE.map((m) => apiChatImage(call, m, b64, mime, prompt)))
|
|
350
|
+
const fullOk: string[] = []
|
|
351
|
+
for (const [i, r] of fullReads.entries()) {
|
|
352
|
+
if (r.status === "fulfilled") fullOk.push(`### Full-page reading (${DIAGRAM_ENSEMBLE[i]})\n\n${r.value}`)
|
|
353
|
+
else log(`full-page pass failed for ${DIAGRAM_ENSEMBLE[i]}: ${(r.reason as Error).message}`)
|
|
354
|
+
}
|
|
355
|
+
if (fullOk.length === 0) throw new Error("every diagram model failed on the full page")
|
|
356
|
+
|
|
357
|
+
// 2. Quadrant zoom pass: each model × each native-resolution crop, in parallel.
|
|
358
|
+
const [w, h] = await getImageDims(pagePng)
|
|
359
|
+
const quadrantReads: string[] = []
|
|
360
|
+
if (w > 0 && h > 0) {
|
|
361
|
+
const halfW = Math.floor(w / 2)
|
|
362
|
+
const halfH = Math.floor(h / 2)
|
|
363
|
+
const crops: [string, number, number, number, number][] = [
|
|
364
|
+
["top-left", 0, 0, halfW, halfH],
|
|
365
|
+
["top-right", halfW, 0, halfW, halfH],
|
|
366
|
+
["bottom-left", 0, halfH, halfW, halfH],
|
|
367
|
+
["bottom-right", halfW, halfH, halfW, halfH],
|
|
368
|
+
]
|
|
369
|
+
const tempDir = join(pagePng.slice(0, pagePng.lastIndexOf("/")), `quadrants_${pagePng.slice(pagePng.lastIndexOf("/") + 1, -4)}`)
|
|
370
|
+
try {
|
|
371
|
+
await mkdir(tempDir, { recursive: true })
|
|
372
|
+
const jobs: Promise<void>[] = []
|
|
373
|
+
for (const [label, x, y, cw, ch] of crops) {
|
|
374
|
+
const cropPath = join(tempDir, `${label}.png`)
|
|
375
|
+
if (!(await cropImage(pagePng, cropPath, x, y, cw, ch))) continue
|
|
376
|
+
const [qb64, qmime] = await imageToB64(cropPath)
|
|
377
|
+
for (const m of DIAGRAM_ENSEMBLE) {
|
|
378
|
+
jobs.push(
|
|
379
|
+
apiChatImage(call, m, qb64, qmime, QUADRANT_PROMPT(label))
|
|
380
|
+
.then((res) => {
|
|
381
|
+
quadrantReads.push(`### Quadrant ${label} (${m})\n\n${res}`)
|
|
382
|
+
})
|
|
383
|
+
.catch((e) => log(`quadrant pass failed for ${label} (${m}): ${(e as Error).message}`))
|
|
384
|
+
.finally(() => rm(cropPath, { force: true }).catch(() => {})),
|
|
385
|
+
)
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
await Promise.allSettled(jobs)
|
|
389
|
+
} finally {
|
|
390
|
+
await rm(tempDir, { force: true, recursive: true }).catch(() => {})
|
|
391
|
+
}
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
// 3. Adjudication: the strongest reader re-examines every disagreement
|
|
395
|
+
// against the page image and emits the final structure.
|
|
396
|
+
const adjudication =
|
|
397
|
+
"You are the adjudicator of a diagram transcription. Below are independent readings " +
|
|
398
|
+
"of the same page: several models read the full page, and several models read each " +
|
|
399
|
+
"quadrant crop at higher effective resolution. The readings DISAGREE in places — each " +
|
|
400
|
+
"reader misread different labels. Re-examine the attached page image yourself and " +
|
|
401
|
+
"resolve every conflict: prefer a value confirmed by multiple readings, but trust the " +
|
|
402
|
+
"image over any reading when you can read the label clearly. Normalize equivalent " +
|
|
403
|
+
"units (0.1uF = 100nF), drop readings of things that are not on the page, and keep " +
|
|
404
|
+
"every correct detail from all readings.\n\n" +
|
|
405
|
+
"Output the final merged Markdown with exactly these sections where applicable: the " +
|
|
406
|
+
"page's text transcription, the component table (Ref/Type/Value/Notes), the wiring " +
|
|
407
|
+
"connections, a '## Graph' section (node list, edge list as source -> target with " +
|
|
408
|
+
"labels, Mermaid graph TD) when the page shows a flowchart or node graph, and the " +
|
|
409
|
+
"placeholder image reference. No commentary about the merging.\n\n" +
|
|
410
|
+
fullOk.join("\n\n") +
|
|
411
|
+
(quadrantReads.length > 0 ? `\n\n## Quadrant readings\n\n${quadrantReads.join("\n\n")}` : "")
|
|
412
|
+
try {
|
|
413
|
+
return await apiChatImage(call, ADJUDICATOR_MODEL, b64, mime, adjudication)
|
|
414
|
+
} catch (e) {
|
|
415
|
+
log(`adjudication failed (${(e as Error).message}), merging text-only`)
|
|
416
|
+
const mergePrompt =
|
|
417
|
+
"Merge these independent readings of the same diagram page into a single accurate " +
|
|
418
|
+
"Markdown document. Deduplicate components, resolve conflicting values in favour of " +
|
|
419
|
+
"the majority and normalize equivalent units (0.1uF = 100nF). Keep every correct " +
|
|
420
|
+
"detail. Output the final merged Markdown only.\n\n" +
|
|
421
|
+
fullOk.join("\n\n") +
|
|
422
|
+
(quadrantReads.length > 0 ? `\n\n## Quadrant readings\n\n${quadrantReads.join("\n\n")}` : "")
|
|
423
|
+
return apiChat(call, REFINE_MODEL, [{ role: "user", content: mergePrompt }])
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
/** Mixed page: OCR for the text, diagram analysis for the figures, merged. */
|
|
428
|
+
export const transcribeMixedPage = async (call: RelayCall, pagePng: string, page: number, total: number): Promise<string> => {
|
|
429
|
+
const ocrResult = await transcribeTextPage(call, pagePng, page, total)
|
|
430
|
+
const diagramResult = await transcribeDiagramPage(call, pagePng, page, total)
|
|
431
|
+
const mergePrompt =
|
|
432
|
+
"Below are two transcriptions of the same document page. The first was done " +
|
|
433
|
+
"by an OCR model (good at text), the second by a vision model (good at diagrams). " +
|
|
434
|
+
"Merge them into a single accurate Markdown document. Use the OCR version as the " +
|
|
435
|
+
"base for text content. If the diagram version found component tables or technical " +
|
|
436
|
+
"details missing from the OCR version, incorporate them. Deduplicate. " +
|
|
437
|
+
"Output the final merged Markdown only.\n\n" +
|
|
438
|
+
`## OCR Transcription\n\n${ocrResult}\n\n## Diagram Transcription\n\n${diagramResult}`
|
|
439
|
+
return apiChat(call, REFINE_MODEL, [{ role: "user", content: mergePrompt }])
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
// ─── Stage 4: Refinement loop ────────────────────────────────────────────────
|
|
443
|
+
|
|
444
|
+
const REFINE_PROMPT = `You are a meticulous editor. Below is Markdown extracted from a document page, and the original page image.
|
|
445
|
+
Compare the Markdown against the image and fix any issues:
|
|
446
|
+
- Broken or incomplete tables (compare every cell against the image)
|
|
447
|
+
- Missing text or labels
|
|
448
|
+
- Garbled or misread characters (especially numbers and symbols)
|
|
449
|
+
- Inconsistent heading hierarchy
|
|
450
|
+
- Missing footnotes or annotations
|
|
451
|
+
- Lost or incorrect formatting (bold, italic, code blocks)
|
|
452
|
+
- Math equations that don't match the image
|
|
453
|
+
|
|
454
|
+
Output the corrected Markdown ONLY. No commentary, no explanations about what you changed.
|
|
455
|
+
|
|
456
|
+
Current Markdown:
|
|
457
|
+
{markdown}`
|
|
458
|
+
|
|
459
|
+
const refinePage = async (call: RelayCall, pagePng: string, markdown: string, round: number): Promise<string> => {
|
|
460
|
+
const [b64, mime] = await imageToB64(pagePng)
|
|
461
|
+
const result = await apiChatImage(call, REFINE_MODEL, b64, mime, REFINE_PROMPT.replace("{markdown}", markdown))
|
|
462
|
+
// Sanity check: if refinement produced empty or much shorter output, keep original
|
|
463
|
+
if (result.trim().length < markdown.trim().length * 0.3) {
|
|
464
|
+
log(` refine round ${round}: output suspiciously short (${result.length} vs ${markdown.length} chars), keeping previous`)
|
|
465
|
+
return markdown
|
|
466
|
+
}
|
|
467
|
+
return result
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
export const refineLoop = async (call: RelayCall, pagePng: string, markdown: string, maxRounds = MAX_REFINE_ROUNDS): Promise<string> => {
|
|
471
|
+
if (!markdown.trim()) return markdown
|
|
472
|
+
let current = markdown
|
|
473
|
+
for (let r = 1; r <= maxRounds; r++) {
|
|
474
|
+
log(` refine round ${r}/${maxRounds}...`)
|
|
475
|
+
const refined = await refinePage(call, pagePng, current, r)
|
|
476
|
+
if (refined.trim() === current.trim()) {
|
|
477
|
+
log(` refine round ${r}: no changes, stopping early`)
|
|
478
|
+
break
|
|
479
|
+
}
|
|
480
|
+
if (refined.trim().length === current.trim().length) {
|
|
481
|
+
log(` refine round ${r}: same length, likely converged`)
|
|
482
|
+
current = refined
|
|
483
|
+
break
|
|
484
|
+
}
|
|
485
|
+
current = refined
|
|
486
|
+
}
|
|
487
|
+
return current
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
// ─── Full per-page pipeline ──────────────────────────────────────────────────
|
|
491
|
+
|
|
492
|
+
/** Remove an outermost ```markdown fence that wraps the ENTIRE output — inner
|
|
493
|
+
* code blocks are preserved. */
|
|
494
|
+
export const stripCodeFences = (text: string): string => {
|
|
495
|
+
const stripped = text.trim()
|
|
496
|
+
if (!stripped.startsWith("```")) return text
|
|
497
|
+
const lines = stripped.split("\n")
|
|
498
|
+
if (lines.length < 2) return text
|
|
499
|
+
if (!["```markdown", "```md", "```"].includes(lines[0]!.trim())) return text
|
|
500
|
+
if (lines[lines.length - 1]!.trim() !== "```") return text
|
|
501
|
+
return lines.slice(1, -1).join("\n").trim()
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
export const processPage = async (call: RelayCall, pagePng: string, page: number, total: number, doRefine: boolean): Promise<[number, string]> => {
|
|
505
|
+
const pageType = await classifyPage(call, pagePng)
|
|
506
|
+
log(`page ${page}: classified as ${pageType}`)
|
|
507
|
+
let markdown: string
|
|
508
|
+
try {
|
|
509
|
+
if (pageType === "TEXT" || pageType === "TABLE_MATH") {
|
|
510
|
+
markdown = await transcribeTextPage(call, pagePng, page, total)
|
|
511
|
+
} else if (pageType === "DIAGRAM") {
|
|
512
|
+
try {
|
|
513
|
+
markdown = await transcribeDiagramPage(call, pagePng, page, total)
|
|
514
|
+
} catch (e) {
|
|
515
|
+
log(`page ${page}: diagram model failed (${(e as Error).message}), falling back to OCR`)
|
|
516
|
+
markdown = await transcribeTextPage(call, pagePng, page, total)
|
|
517
|
+
}
|
|
518
|
+
} else {
|
|
519
|
+
try {
|
|
520
|
+
markdown = await transcribeMixedPage(call, pagePng, page, total)
|
|
521
|
+
} catch (e) {
|
|
522
|
+
log(`page ${page}: mixed model failed (${(e as Error).message}), falling back to OCR`)
|
|
523
|
+
markdown = await transcribeTextPage(call, pagePng, page, total)
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
} catch (e) {
|
|
527
|
+
log(`page ${page}: transcription failed (${(e as Error).message}), using placeholder`)
|
|
528
|
+
return [page, `<!-- Page ${page}: transcription failed: ${(e as Error).message} -->\n`]
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
try {
|
|
532
|
+
if (doRefine) {
|
|
533
|
+
markdown = await refineLoop(call, pagePng, markdown, pageType === "TEXT" ? 1 : MAX_REFINE_ROUNDS)
|
|
534
|
+
}
|
|
535
|
+
} catch (e) {
|
|
536
|
+
log(`page ${page}: refinement failed (${(e as Error).message}), using unrefined output`)
|
|
537
|
+
}
|
|
538
|
+
return [page, stripCodeFences(markdown)]
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
// ─── Concurrency auto-tuning ─────────────────────────────────────────────────
|
|
542
|
+
|
|
543
|
+
/** Probe throughput at increasing parallelism; the gateway's key pool absorbs
|
|
544
|
+
* the load, so this measures what the pipeline will actually see. Probes up
|
|
545
|
+
* to 16 workers — with several keys the pool admits far more than 8 in
|
|
546
|
+
* parallel, and long documents want every slot. */
|
|
547
|
+
export const probeConcurrency = async (call: RelayCall): Promise<number> => {
|
|
548
|
+
log("auto-tuning concurrency...")
|
|
549
|
+
const probeMsg: Msg[] = [{ role: "user", content: "Reply with exactly: OK" }]
|
|
550
|
+
const best = { workers: 3, throughput: 0 }
|
|
551
|
+
for (const n of [1, 3, 5, 8, 12, 16]) {
|
|
552
|
+
try {
|
|
553
|
+
const start = performance.now()
|
|
554
|
+
const results = await Promise.allSettled(Array.from({ length: n }, () => apiChat(call, REFINE_MODEL, probeMsg, { maxTokens: 10 })))
|
|
555
|
+
if (results.some((r) => r.status === "rejected")) throw new Error("a probe request failed")
|
|
556
|
+
const elapsed = (performance.now() - start) / 1000
|
|
557
|
+
const throughput = n / elapsed
|
|
558
|
+
log(` ${n} parallel requests: ${elapsed.toFixed(1)}s → ${throughput.toFixed(2)} req/s`)
|
|
559
|
+
if (throughput > best.throughput) {
|
|
560
|
+
best.throughput = throughput
|
|
561
|
+
best.workers = n
|
|
562
|
+
}
|
|
563
|
+
} catch (e) {
|
|
564
|
+
log(` ${n} parallel requests: failed (${(e as Error).message}), backing off`)
|
|
565
|
+
break
|
|
566
|
+
}
|
|
567
|
+
}
|
|
568
|
+
const workers = Math.min(16, Math.max(2, best.workers))
|
|
569
|
+
log(`optimal concurrency: ${workers} workers (${best.throughput.toFixed(2)} req/s)`)
|
|
570
|
+
return workers
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
/** Fixed-size worker pool over an index cursor. */
|
|
574
|
+
async function runPool<T>(items: T[], workers: number, fn: (item: T, index: number) => Promise<void>): Promise<void> {
|
|
575
|
+
let next = 0
|
|
576
|
+
const worker = async (): Promise<void> => {
|
|
577
|
+
for (;;) {
|
|
578
|
+
const i = next++
|
|
579
|
+
if (i >= items.length) return
|
|
580
|
+
await fn(items[i]!, i)
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
await Promise.all(Array.from({ length: Math.min(workers, items.length) }, worker))
|
|
584
|
+
}
|
|
585
|
+
|
|
586
|
+
// ─── Cross-page table merging ────────────────────────────────────────────────
|
|
587
|
+
|
|
588
|
+
const TABLE_MERGE_PROMPT = `Below are consecutive pages of Markdown extracted from a PDF.
|
|
589
|
+
Some tables may span across page boundaries. Your job:
|
|
590
|
+
1. Merge any tables that were split across pages into single coherent tables.
|
|
591
|
+
2. Fix heading hierarchy (ensure H1→H2→H3 flow makes sense across pages).
|
|
592
|
+
3. Do NOT change the content of individual cells, only merge structure.
|
|
593
|
+
4. Do NOT delete anything: keep every heading, paragraph, list, footnote, figure placeholder, caption, code block and component table from every page. If two pages contain the same figure placeholder, keep one and keep its caption.
|
|
594
|
+
5. Output the complete merged Markdown.
|
|
595
|
+
|
|
596
|
+
Pages:
|
|
597
|
+
{pages}`
|
|
598
|
+
|
|
599
|
+
export const crossPageMerge = async (call: RelayCall, pageMarkdowns: string[]): Promise<string> => {
|
|
600
|
+
const joinFinal = (parts: string[]) => parts.join("\n\n---\n\n")
|
|
601
|
+
if (pageMarkdowns.length <= 1) return joinFinal(pageMarkdowns)
|
|
602
|
+
const mergeChunk = async (chunk: string[]): Promise<string> => {
|
|
603
|
+
const text = chunk.join("\n\n---PAGE BREAK---\n\n")
|
|
604
|
+
try {
|
|
605
|
+
return await apiChat(call, REFINE_MODEL, [{ role: "user", content: TABLE_MERGE_PROMPT.replace("{pages}", text) }])
|
|
606
|
+
} catch {
|
|
607
|
+
return joinFinal(chunk)
|
|
608
|
+
}
|
|
609
|
+
}
|
|
610
|
+
const joined = pageMarkdowns.join("\n\n---PAGE BREAK---\n\n")
|
|
611
|
+
// For large documents, merge in chunks to avoid context overflow
|
|
612
|
+
const maxChunkChars = 100_000 // ~25K tokens, safe for 128K context models
|
|
613
|
+
if (joined.length <= maxChunkChars) {
|
|
614
|
+
const merged = await mergeChunk(pageMarkdowns)
|
|
615
|
+
return merged.replaceAll("\n\n---PAGE BREAK---\n\n", "\n\n---\n\n").replaceAll("---PAGE BREAK---", "---")
|
|
616
|
+
}
|
|
617
|
+
log(`document is large (${joined.length} chars), merging in chunks...`)
|
|
618
|
+
const chunks: string[] = []
|
|
619
|
+
let current: string[] = []
|
|
620
|
+
let size = 0
|
|
621
|
+
for (const md of pageMarkdowns) {
|
|
622
|
+
if (size + md.length > maxChunkChars && current.length > 0) {
|
|
623
|
+
chunks.push(await mergeChunk(current))
|
|
624
|
+
current = [md]
|
|
625
|
+
size = md.length
|
|
626
|
+
} else {
|
|
627
|
+
current.push(md)
|
|
628
|
+
size += md.length
|
|
629
|
+
}
|
|
630
|
+
}
|
|
631
|
+
if (current.length > 0) chunks.push(await mergeChunk(current))
|
|
632
|
+
return chunks.join("\n\n---\n\n").replaceAll("---PAGE BREAK---", "---")
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
// ─── Main pipeline ───────────────────────────────────────────────────────────
|
|
636
|
+
|
|
637
|
+
const argValue = (args: string[], name: string): string | undefined => {
|
|
638
|
+
const i = args.indexOf(name)
|
|
639
|
+
return i >= 0 ? args[i + 1] : undefined
|
|
640
|
+
}
|
|
641
|
+
|
|
642
|
+
const VALUE_FLAGS = new Set(["--workers", "--dpi", "--dedup-threshold"])
|
|
643
|
+
|
|
644
|
+
/** Positional arguments = everything that is not a flag or a flag's value. */
|
|
645
|
+
const positionals = (args: string[]): string[] => {
|
|
646
|
+
const out: string[] = []
|
|
647
|
+
for (let i = 0; i < args.length; i++) {
|
|
648
|
+
const a = args[i]!
|
|
649
|
+
if (a.startsWith("--")) {
|
|
650
|
+
if (VALUE_FLAGS.has(a)) i++ // skip the flag's value
|
|
651
|
+
continue
|
|
652
|
+
}
|
|
653
|
+
if (VALUE_FLAGS.has(args[i - 1] ?? "")) continue // a flag value that looks positional
|
|
654
|
+
out.push(a)
|
|
655
|
+
}
|
|
656
|
+
return out
|
|
657
|
+
}
|
|
658
|
+
|
|
659
|
+
export async function main(argv: string[], call: RelayCall = gatewayCall): Promise<number> {
|
|
660
|
+
const [input, output] = positionals(argv)
|
|
661
|
+
if (!input) {
|
|
662
|
+
console.error("usage: pdf2md.ts <input.pdf> [output.md] [--workers N] [--dpi 300] [--no-dedup] [--dedup-threshold 0.98] [--no-refine] [--no-merge] [--verbose]")
|
|
663
|
+
return 2
|
|
664
|
+
}
|
|
665
|
+
VERBOSE = argv.includes("--verbose") || argv.includes("-v")
|
|
666
|
+
const dpi = Number(argValue(argv, "--dpi") ?? DEFAULT_DPI)
|
|
667
|
+
const dedupThreshold = Number(argValue(argv, "--dedup-threshold") ?? DEDUP_SIMILARITY_THRESHOLD)
|
|
668
|
+
const noDedup = argv.includes("--no-dedup")
|
|
669
|
+
const noRefine = argv.includes("--no-refine")
|
|
670
|
+
const noMerge = argv.includes("--no-merge")
|
|
671
|
+
|
|
672
|
+
if (!(await Bun.file(input).exists())) {
|
|
673
|
+
console.error(`input not found: ${input}`)
|
|
674
|
+
return 2
|
|
675
|
+
}
|
|
676
|
+
const pageCount = await getPageCount(input)
|
|
677
|
+
if (pageCount === 0) {
|
|
678
|
+
console.error("could not determine page count (pdfinfo not available or failed)")
|
|
679
|
+
return 2
|
|
680
|
+
}
|
|
681
|
+
log(`PDF: ${input.split("/").pop()} (${pageCount} pages)`)
|
|
682
|
+
|
|
683
|
+
const tempDir = await mkdtemp(join(tmpdir(), "pdf2md-"))
|
|
684
|
+
try {
|
|
685
|
+
log(`rendering ${pageCount} pages at ${dpi} DPI...`)
|
|
686
|
+
let pagePngs = await renderPdfPages(input, dpi, tempDir)
|
|
687
|
+
log(`rendered ${pagePngs.length} page images`)
|
|
688
|
+
if (pagePngs.length !== pageCount) {
|
|
689
|
+
log(`WARNING: pdfinfo said ${pageCount} pages but rendered ${pagePngs.length} images`)
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
if (!noDedup && pagePngs.length > 1) {
|
|
693
|
+
log("deduplicating incremental-reveal pages...")
|
|
694
|
+
pagePngs = await dedupIncrementalReveals(pagePngs, dedupThreshold)
|
|
695
|
+
if (pagePngs.length === 0) {
|
|
696
|
+
console.error("all pages were deduplicated — nothing to process")
|
|
697
|
+
return 2
|
|
698
|
+
}
|
|
699
|
+
}
|
|
700
|
+
const total = pagePngs.length
|
|
701
|
+
|
|
702
|
+
let workers = Number(argValue(argv, "--workers") ?? 0)
|
|
703
|
+
if (!Number.isFinite(workers) || workers <= 0) {
|
|
704
|
+
workers = total <= 2 ? total : await probeConcurrency(call)
|
|
705
|
+
}
|
|
706
|
+
log(`processing ${total} pages with ${workers} workers...`)
|
|
707
|
+
|
|
708
|
+
const results = new Map<number, string>()
|
|
709
|
+
let completed = 0
|
|
710
|
+
await runPool(pagePngs, workers, async (png, idx) => {
|
|
711
|
+
try {
|
|
712
|
+
const [page, markdown] = await processPage(call, png, idx + 1, total, !noRefine)
|
|
713
|
+
results.set(idx, markdown)
|
|
714
|
+
} catch (e) {
|
|
715
|
+
log(`page ${idx + 1} FAILED: ${(e as Error).message}`, true)
|
|
716
|
+
results.set(idx, `<!-- Page ${idx + 1} failed: ${(e as Error).message} -->\n`)
|
|
717
|
+
}
|
|
718
|
+
completed++
|
|
719
|
+
logProgress(completed, total, `page ${idx + 1}`)
|
|
720
|
+
})
|
|
721
|
+
|
|
722
|
+
const ordered = Array.from({ length: total }, (_, i) => results.get(i) ?? `<!-- Page ${i + 1} missing -->\n`)
|
|
723
|
+
|
|
724
|
+
let finalMd: string
|
|
725
|
+
if (!noMerge && total > 1 && total <= 200) {
|
|
726
|
+
log("cross-page table merging...")
|
|
727
|
+
finalMd = await crossPageMerge(call, ordered)
|
|
728
|
+
} else {
|
|
729
|
+
if (total > 200) log(`skipping cross-page merge (document too large: ${total} pages)`)
|
|
730
|
+
finalMd = ordered.join("\n\n---\n\n")
|
|
731
|
+
}
|
|
732
|
+
|
|
733
|
+
if (output) {
|
|
734
|
+
await Bun.write(output, finalMd)
|
|
735
|
+
log(`output written to ${output} (${finalMd.length} chars)`, true)
|
|
736
|
+
} else {
|
|
737
|
+
process.stdout.write(finalMd)
|
|
738
|
+
}
|
|
739
|
+
} finally {
|
|
740
|
+
await rm(tempDir, { force: true, recursive: true }).catch(() => {})
|
|
741
|
+
}
|
|
742
|
+
return 0
|
|
743
|
+
}
|
|
744
|
+
|
|
745
|
+
if (import.meta.main) {
|
|
746
|
+
process.exit(await main(process.argv.slice(2)))
|
|
747
|
+
}
|