opencode-ufr 0.2.9 → 0.2.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,747 @@
1
+ #!/usr/bin/env bun
2
+ /**
3
+ * pdf2md — agentic PDF-to-Markdown converter on UFR vision models.
4
+ *
5
+ * TypeScript port of the author's former standalone Python script, integrated
6
+ * with the opencode-ufr gateway: every model call goes through the gateway's
7
+ * relay (POST /v1/_relay), so key rotation and soft rate limiting are handled
8
+ * centrally — pages run at full parallelism and the caller never paces itself.
9
+ *
10
+ * Pipeline per page:
11
+ * 1. Render at 300 DPI (pdftoppm)
12
+ * 2. Deduplicate incremental-reveal pages (PowerPoint-style, magick compare)
13
+ * 3. Classify: TEXT / TABLE_MATH / DIAGRAM / MIXED (qwen-3.8-27b)
14
+ * 4. Route to the best model per type:
15
+ * - TEXT → glm-5.3-flash (structure + LaTeX, 3-4x faster than the old OCR model)
16
+ * - TABLE_MATH → glm-5.3-flash + refinement
17
+ * - DIAGRAM → 3-model ensemble (glm-5.3-flash, deepseek-v4.1-flash,
18
+ * qwen-3.5-397b) full-page + quadrant zoom, adjudicated
19
+ * - MIXED → OCR + diagram ensemble, merged
20
+ * 5. Refinement loop (gemma-4-31b compares markdown against the page image)
21
+ * 6. Cross-page table merge
22
+ *
23
+ * External tools: pdftoppm/pdfinfo (poppler-utils) and magick (ImageMagick).
24
+ *
25
+ * Usage:
26
+ * bun src/client/pdf2md.ts <input.pdf> [output.md] [--workers N] [--dpi 300]
27
+ * [--no-dedup] [--dedup-threshold 0.98] [--no-refine] [--no-merge] [--verbose]
28
+ */
29
+ import { execFile as execFileCb } from "node:child_process"
30
+ import { mkdir, mkdtemp, readdir, rm } from "node:fs/promises"
31
+ import { tmpdir } from "node:os"
32
+ import { join } from "node:path"
33
+ import { promisify } from "node:util"
34
+ import { connectGateway, type Gateway, relay } from "./gateway"
35
+
36
+ const execFile = promisify(execFileCb)
37
+
38
+ // ─── Configuration ───────────────────────────────────────────────────────────
39
+
40
+ const OCR_MODEL = process.env.PDF2MD_OCR_MODEL ?? "glm-5.3-flash-llmlb"
41
+ const CLASSIFY_MODEL = process.env.PDF2MD_CLASSIFY_MODEL ?? "qwen-3.8-27b-llmlb"
42
+ const REFINE_MODEL = process.env.PDF2MD_REFINE_MODEL ?? "gemma-4-31b-llmlb"
43
+
44
+ const MIME: Record<string, string> = {
45
+ ".png": "image/png",
46
+ ".jpg": "image/jpeg",
47
+ ".jpeg": "image/jpeg",
48
+ ".webp": "image/webp",
49
+ }
50
+ const DEFAULT_DPI = 300
51
+ const MAX_REFINE_ROUNDS = 3
52
+ const DEDUP_SIMILARITY_THRESHOLD = 0.90
53
+ const REQUEST_TIMEOUT_MS = 600_000 // diagrams can think for minutes
54
+ const DEFAULT_MAX_TOKENS = 8192 // generous budget so reasoning models don't run out before visible output
55
+
56
+ // ─── Logging ─────────────────────────────────────────────────────────────────
57
+
58
+ let VERBOSE = false
59
+ const log = (msg: string, force = false): void => {
60
+ if (force || VERBOSE) console.error(`[pdf2md] ${msg}`)
61
+ }
62
+
63
+ const logProgress = (current: number, total: number, label = ""): void => {
64
+ const pct = total ? Math.floor((current * 100) / total) : 0
65
+ const tag = label ? ` ${label}` : ""
66
+ process.stderr.write(`\r[pdf2md] ${current}/${total} (${pct}%)${tag} `)
67
+ if (current >= total) process.stderr.write("\n")
68
+ }
69
+
70
+ // ─── Model calls (all through the gateway's relay: key rotation + pacing) ────
71
+
72
+ type Msg = { role: string; content: unknown }
73
+
74
+ /** One raw model call, returning the upstream JSON body as text. Injectable for tests. */
75
+ export type RelayCall = (model: string, messages: Msg[], maxTokens: number) => Promise<{ status: number; body: string; retryAfterMs?: number }>
76
+
77
+ let gateway: Gateway | null = null
78
+
79
+ const gatewayCall: RelayCall = async (model, messages, maxTokens) => {
80
+ gateway ??= await connectGateway()
81
+ return relay(gateway, { model, messages, max_tokens: maxTokens, temperature: 0 }, { timeoutMs: REQUEST_TIMEOUT_MS })
82
+ }
83
+
84
+ type ChatOptions = {
85
+ maxTokens?: number
86
+ /** Classification only: when the model thinks past its budget and emits no
87
+ * visible content, its reasoning_content is parsed for the category instead.
88
+ * For content generation reasoning is deliberation, never the answer. */
89
+ allowReasoningFallback?: boolean
90
+ }
91
+
92
+ /** Chat completion with the Python script's retry ladder, via the relay.
93
+ * The gateway already waits for key slots — a 429 here means a wall, a pool
94
+ * cap or an exhausted budget, all of which are worth a real backoff. */
95
+ export async function apiChat(call: RelayCall, model: string, messages: Msg[], o: ChatOptions = {}): Promise<string> {
96
+ const base = { model, temperature: 0, messages }
97
+ let effectiveMax = o.maxTokens ?? DEFAULT_MAX_TOKENS
98
+ let lastError = ""
99
+ for (let attempt = 0; attempt < 3; attempt++) {
100
+ if (attempt > 0) effectiveMax *= 2 ** attempt
101
+ let res: Awaited<ReturnType<RelayCall>>
102
+ try {
103
+ res = await call(model, messages, effectiveMax)
104
+ } catch (e) {
105
+ lastError = `network error: ${(e as Error).message}`
106
+ await Bun.sleep(2 ** attempt * 1000)
107
+ continue
108
+ }
109
+ if (res.status === 429 || res.status === 503) {
110
+ const waitS = Math.min(res.retryAfterMs ? Math.round(res.retryAfterMs / 1000) : 2 ** attempt * 5, 120)
111
+ log(`rate-limited (${res.status}), waiting ${waitS}s before retry...`)
112
+ await Bun.sleep(waitS * 1000)
113
+ lastError = `HTTP ${res.status}`
114
+ continue
115
+ }
116
+ if (res.status !== 200) {
117
+ lastError = `HTTP ${res.status}: ${res.body.slice(0, 500)}`
118
+ continue
119
+ }
120
+ let data: {
121
+ choices?: { message?: { content?: unknown; reasoning_content?: unknown }; finish_reason?: string }[]
122
+ }
123
+ try {
124
+ data = JSON.parse(res.body)
125
+ } catch {
126
+ lastError = "response was not JSON"
127
+ continue
128
+ }
129
+ const choice = data.choices?.[0]
130
+ if (!choice) {
131
+ lastError = "no choices in response"
132
+ continue
133
+ }
134
+ const content = choice.message?.content
135
+ if (typeof content === "string" && content.trim()) return content
136
+ const reasoning = choice.message?.reasoning_content
137
+ if (typeof reasoning === "string" && reasoning.trim() && o.allowReasoningFallback) {
138
+ log(` model ${model}: content empty, using reasoning_content (${reasoning.length} chars)`)
139
+ return reasoning.trim()
140
+ }
141
+ if (choice.finish_reason === "length") {
142
+ log(` model ${model}: truncated (finish=length), retrying with larger budget`)
143
+ lastError = "truncated (finish=length)"
144
+ continue
145
+ }
146
+ lastError = `empty content (finish=${choice.finish_reason})`
147
+ }
148
+ throw new Error(`API call failed after 3 attempts (model=${model}): ${lastError}`)
149
+ }
150
+
151
+ const imagePart = (b64: string, mime: string) => ({ type: "image_url", image_url: { url: `data:${mime};base64,${b64}` } })
152
+
153
+ const apiChatImage = async (call: RelayCall, model: string, b64: string, mime: string, prompt: string, o: ChatOptions = {}) =>
154
+ apiChat(call, model, [{ role: "user", content: [{ type: "text", text: prompt }, imagePart(b64, mime)] }], o)
155
+
156
+ const apiChatImages = (call: RelayCall, model: string, images: [string, string][], prompt: string, o: ChatOptions = {}) =>
157
+ apiChat(call, model, [{ role: "user", content: [{ type: "text", text: prompt }, ...images.map(([b64, mime]) => imagePart(b64, mime))] }], o)
158
+
159
+ // ─── Image utilities (no image library, use ImageMagick) ─────────────────────
160
+
161
+ const cropImage = async (srcPng: string, outputPng: string, x: number, y: number, w: number, h: number): Promise<boolean> => {
162
+ try {
163
+ await execFile("magick", ["convert", srcPng, "-crop", `${w}x${h}+${x}+${y}`, "+repage", outputPng], { timeout: 30_000 })
164
+ return true
165
+ } catch (e) {
166
+ log(`crop failed: ${(e as Error).message}`)
167
+ return false
168
+ }
169
+ }
170
+
171
+ const imageToB64 = async (path: string): Promise<[string, string]> => {
172
+ const mime = MIME[path.slice(path.lastIndexOf(".")).toLowerCase()] ?? "image/png"
173
+ const buf = await Bun.file(path).arrayBuffer()
174
+ return [Buffer.from(buf).toString("base64"), mime]
175
+ }
176
+
177
+ const renderPdfPages = async (pdfPath: string, dpi: number, outDir: string): Promise<string[]> => {
178
+ const prefix = join(outDir, "page")
179
+ try {
180
+ await execFile("pdftoppm", ["-png", "-r", String(dpi), pdfPath, prefix], { timeout: 600_000 })
181
+ } catch (e) {
182
+ throw new Error(`pdftoppm failed (${(e as Error).message}). Install poppler-utils.`)
183
+ }
184
+ const names = (await readdir(outDir)).filter((n) => n.startsWith("page") && n.endsWith(".png")).sort()
185
+ if (names.length === 0) throw new Error("pdftoppm produced no output images")
186
+ return names.map((n) => join(outDir, n))
187
+ }
188
+
189
+ const getPageCount = async (pdfPath: string): Promise<number> => {
190
+ try {
191
+ const { stdout } = await execFile("pdfinfo", [pdfPath], { timeout: 30_000 })
192
+ for (const line of stdout.split("\n")) {
193
+ if (line.startsWith("Pages:")) return Number.parseInt(line.split(":")[1]!.trim(), 10)
194
+ }
195
+ } catch {
196
+ // fall through
197
+ }
198
+ return 0
199
+ }
200
+
201
+ const getImageDims = async (pngPath: string): Promise<[number, number]> => {
202
+ try {
203
+ const { stdout } = await execFile("magick", ["identify", "-format", "%w %h", pngPath], { timeout: 10_000 })
204
+ const [w, h] = stdout.trim().split(/\s+/).map(Number)
205
+ return [w ?? 0, h ?? 0]
206
+ } catch {
207
+ return [0, 0]
208
+ }
209
+ }
210
+
211
+ // ─── Stage 1b: Incremental-reveal deduplication ──────────────────────────────
212
+
213
+ /** Normalized RMSE via ImageMagick: similarity = 1 - rmse/255, in [0, 1]. */
214
+ const imageSimilarity = async (imgA: string, imgB: string): Promise<number> => {
215
+ try {
216
+ // magick compare writes the metric to stderr and exits non-zero on difference
217
+ const res = await execFile("magick", ["compare", "-metric", "RMSE", imgA, imgB, "null:"], { timeout: 30_000 }).catch(
218
+ (e: { stderr?: string }) => ({ stderr: e.stderr ?? "" }),
219
+ )
220
+ const line = (res.stderr ?? "").trim()
221
+ const normalized = /\(([\d.]+)\)/.exec(line)?.[1]
222
+ if (normalized !== undefined) return Math.max(0, 1 - Number.parseFloat(normalized))
223
+ const raw = Number.parseFloat(line.split(/\s+/)[0] ?? "65535")
224
+ return Math.max(0, 1 - (Number.isFinite(raw) ? raw : 65535) / 65535)
225
+ } catch {
226
+ return 0
227
+ }
228
+ }
229
+
230
+ /** PowerPoint-style incremental reveals: consecutive near-identical pages — keep the last of each run. */
231
+ export const dedupIncrementalReveals = async (pagePngs: string[], threshold = DEDUP_SIMILARITY_THRESHOLD): Promise<string[]> => {
232
+ if (pagePngs.length <= 1) return pagePngs
233
+ const kept: string[] = []
234
+ let skipped = 0
235
+ for (let i = 0; i < pagePngs.length; i++) {
236
+ const current = pagePngs[i]!
237
+ if (i === pagePngs.length - 1) {
238
+ kept.push(current) // the last page of a run has all the content
239
+ break
240
+ }
241
+ const sim = await imageSimilarity(current, pagePngs[i + 1]!)
242
+ if (sim >= threshold) {
243
+ log(` page ${i + 1} → ${i + 2}: similarity ${sim.toFixed(4)} ≥ ${threshold}, dropping page ${i + 1}`)
244
+ skipped++
245
+ } else {
246
+ kept.push(current)
247
+ }
248
+ }
249
+ log(skipped > 0 ? `dedup: removed ${skipped} incremental-reveal duplicates, ${kept.length} pages remain` : "dedup: no incremental-reveal duplicates detected")
250
+ return kept
251
+ }
252
+
253
+ // ─── Stage 2: Page classification ────────────────────────────────────────────
254
+
255
+ const CLASSIFY_PROMPT = `Analyze this document page image and classify it into exactly ONE of these categories:
256
+
257
+ - TEXT: Mostly text (paragraphs, headings, body text). May contain simple inline figures.
258
+ - TABLE_MATH: Contains significant tables, mathematical equations, or structured data.
259
+ - DIAGRAM: The page's meaning is carried by a drawing: technical drawings, circuit diagrams, schematics, flowcharts, block diagrams, node graphs, state machines, org charts, or complex figures with labels. Boxes connected by arrows count as DIAGRAM even when drawn simply.
260
+ - MIXED: Contains both significant text AND a technical diagram/figure.
261
+
262
+ Respond with ONLY the category name (TEXT, TABLE_MATH, DIAGRAM, or MIXED). No other text.`
263
+
264
+ export type PageType = "TEXT" | "TABLE_MATH" | "DIAGRAM" | "MIXED"
265
+
266
+ export const classifyPage = async (call: RelayCall, pagePng: string): Promise<PageType> => {
267
+ const [b64, mime] = await imageToB64(pagePng)
268
+ try {
269
+ const result = (await apiChatImage(call, CLASSIFY_MODEL, b64, mime, CLASSIFY_PROMPT, { maxTokens: 50, allowReasoningFallback: true })).trim().toUpperCase()
270
+ for (const valid of ["TEXT", "TABLE_MATH", "DIAGRAM", "MIXED"] as const) {
271
+ if (result.includes(valid)) return valid
272
+ }
273
+ return "TEXT"
274
+ } catch (e) {
275
+ log(`classification failed for ${pagePng.split("/").pop()}: ${(e as Error).message}, defaulting to TEXT`)
276
+ return "TEXT"
277
+ }
278
+ }
279
+
280
+ // ─── Stage 3: Per-page-type transcription ────────────────────────────────────
281
+
282
+ const OCR_PROMPT = `Convert this document page to clean Markdown.
283
+ Rules:
284
+ - Preserve all headings, lists, tables (as GitHub markdown tables), and code blocks
285
+ - For math, use LaTeX: $inline$ and $$block$$
286
+ - For figures, output ![Figure N](placeholder_fig_N.png) with a caption if present
287
+ - For footnotes, use [^N] syntax
288
+ - Output ONLY the markdown, no commentary
289
+ - If a table spans the page width, reproduce every cell accurately
290
+ - Page {page_info}`
291
+
292
+ const DIAGRAM_PROMPT = `You are analyzing a page from a technical document that contains a circuit diagram, schematic, flowchart, or node-based graph.
293
+
294
+ Your task:
295
+ 1. First, transcribe ALL text on the page (headings, paragraphs, captions, labels) as Markdown.
296
+ 2. For every visible component (resistors, capacitors, ICs, connectors, ...), read: its reference designator (e.g. R1, C2, U3), its value (e.g. 10kΩ, 100nF), and any pin labels. Output a structured component table:
297
+
298
+ | Ref | Type | Value | Notes |
299
+ |-----|------|-------|-------|
300
+ | R1 | Resistor | 10kΩ | |
301
+
302
+ 3. Describe the wiring: every connection you can trace as "A connects to B" (use reference designators and net labels like VIN, GND).
303
+ 4. If the page contains a flowchart or node-based graph, add a "## Graph" section:
304
+ - every node: its label text exactly as written, and its type (process / decision / input / output / start / end)
305
+ - every edge as "source -> target" with the edge label (e.g. yes/no) if present
306
+ - the whole graph as a Mermaid flowchart (graph TD) with the exact labels
307
+ 5. End with: ![Diagram](placeholder_diagram_{page_num}.png)
308
+ 6. Output ONLY Markdown, no commentary.
309
+
310
+ Page {page_info}`
311
+
312
+ const QUADRANT_PROMPT = (label: string): string =>
313
+ `Focus on this ${label} quadrant of a technical diagram. List every component label, value, and connection you can read. Output as a Markdown table with columns: Ref, Type, Value, Connection. Only output the table, nothing else.`
314
+
315
+ /** The ensemble: three independent vision models at two zoom levels disagree in
316
+ * complementary ways (measured 2026-10-06: glm-5.3-flash read resistor values
317
+ * right on the full page where deepseek/qwen misread, all three agreed on the
318
+ * quadrant crops that fixed the full-page misreads). The adjudicator sees the
319
+ * page plus every reading and resolves conflicts against the image. */
320
+ const DIAGRAM_ENSEMBLE = (process.env.PDF2MD_DIAGRAM_ENSEMBLE ?? "glm-5.3-flash-llmlb,deepseek-v4.1-flash-llmlb,qwen-3.5-397b-llmlb")
321
+ .split(",").map((s) => s.trim()).filter(Boolean)
322
+ const ADJUDICATOR_MODEL = process.env.PDF2MD_ADJUDICATOR_MODEL ?? "glm-5.3-flash-llmlb"
323
+
324
+ const pageInfo = (page: number, total: number): string => (total > 1 ? `(Seite ${page} von ${total})` : "")
325
+
326
+ const transcribeTextPage = async (call: RelayCall, pagePng: string, page: number, total: number): Promise<string> => {
327
+ const [b64, mime] = await imageToB64(pagePng)
328
+ const info = pageInfo(page, total)
329
+ const prompt = OCR_PROMPT.replace("{page_info}", info ? ` ${info}` : "")
330
+ return apiChatImage(call, OCR_MODEL, b64, mime, prompt)
331
+ }
332
+
333
+ /** Diagram/graph page: the full-accuracy path.
334
+ *
335
+ * Three independent vision models read the full page in parallel, then all
336
+ * three read each of the four native-resolution quadrant crops (12 more
337
+ * calls). Each scale/model combination misreads different labels — measured
338
+ * 2026-10-06 on a schematic with R1 10kΩ/R2 4.7kΩ/C1 10µF: glm-5.3-flash read
339
+ * the resistors right on the full page where the others dropped the "k", the
340
+ * quadrant crops fixed R1/C1 for everyone but all three dropped the "k" there
341
+ * instead. The adjudicator then sees the page image plus every reading,
342
+ * re-examines each disagreement against the image, normalizes equivalent
343
+ * units (0.1uF = 100nF) and emits the final structure. */
344
+ export const transcribeDiagramPage = async (call: RelayCall, pagePng: string, page: number, total: number): Promise<string> => {
345
+ const [b64, mime] = await imageToB64(pagePng)
346
+ const prompt = DIAGRAM_PROMPT.replace("{page_info}", pageInfo(page, total)).replace("{page_num}", String(page))
347
+
348
+ // 1. Full-page structured pass: all ensemble models in parallel.
349
+ const fullReads = await Promise.allSettled(DIAGRAM_ENSEMBLE.map((m) => apiChatImage(call, m, b64, mime, prompt)))
350
+ const fullOk: string[] = []
351
+ for (const [i, r] of fullReads.entries()) {
352
+ if (r.status === "fulfilled") fullOk.push(`### Full-page reading (${DIAGRAM_ENSEMBLE[i]})\n\n${r.value}`)
353
+ else log(`full-page pass failed for ${DIAGRAM_ENSEMBLE[i]}: ${(r.reason as Error).message}`)
354
+ }
355
+ if (fullOk.length === 0) throw new Error("every diagram model failed on the full page")
356
+
357
+ // 2. Quadrant zoom pass: each model × each native-resolution crop, in parallel.
358
+ const [w, h] = await getImageDims(pagePng)
359
+ const quadrantReads: string[] = []
360
+ if (w > 0 && h > 0) {
361
+ const halfW = Math.floor(w / 2)
362
+ const halfH = Math.floor(h / 2)
363
+ const crops: [string, number, number, number, number][] = [
364
+ ["top-left", 0, 0, halfW, halfH],
365
+ ["top-right", halfW, 0, halfW, halfH],
366
+ ["bottom-left", 0, halfH, halfW, halfH],
367
+ ["bottom-right", halfW, halfH, halfW, halfH],
368
+ ]
369
+ const tempDir = join(pagePng.slice(0, pagePng.lastIndexOf("/")), `quadrants_${pagePng.slice(pagePng.lastIndexOf("/") + 1, -4)}`)
370
+ try {
371
+ await mkdir(tempDir, { recursive: true })
372
+ const jobs: Promise<void>[] = []
373
+ for (const [label, x, y, cw, ch] of crops) {
374
+ const cropPath = join(tempDir, `${label}.png`)
375
+ if (!(await cropImage(pagePng, cropPath, x, y, cw, ch))) continue
376
+ const [qb64, qmime] = await imageToB64(cropPath)
377
+ for (const m of DIAGRAM_ENSEMBLE) {
378
+ jobs.push(
379
+ apiChatImage(call, m, qb64, qmime, QUADRANT_PROMPT(label))
380
+ .then((res) => {
381
+ quadrantReads.push(`### Quadrant ${label} (${m})\n\n${res}`)
382
+ })
383
+ .catch((e) => log(`quadrant pass failed for ${label} (${m}): ${(e as Error).message}`))
384
+ .finally(() => rm(cropPath, { force: true }).catch(() => {})),
385
+ )
386
+ }
387
+ }
388
+ await Promise.allSettled(jobs)
389
+ } finally {
390
+ await rm(tempDir, { force: true, recursive: true }).catch(() => {})
391
+ }
392
+ }
393
+
394
+ // 3. Adjudication: the strongest reader re-examines every disagreement
395
+ // against the page image and emits the final structure.
396
+ const adjudication =
397
+ "You are the adjudicator of a diagram transcription. Below are independent readings " +
398
+ "of the same page: several models read the full page, and several models read each " +
399
+ "quadrant crop at higher effective resolution. The readings DISAGREE in places — each " +
400
+ "reader misread different labels. Re-examine the attached page image yourself and " +
401
+ "resolve every conflict: prefer a value confirmed by multiple readings, but trust the " +
402
+ "image over any reading when you can read the label clearly. Normalize equivalent " +
403
+ "units (0.1uF = 100nF), drop readings of things that are not on the page, and keep " +
404
+ "every correct detail from all readings.\n\n" +
405
+ "Output the final merged Markdown with exactly these sections where applicable: the " +
406
+ "page's text transcription, the component table (Ref/Type/Value/Notes), the wiring " +
407
+ "connections, a '## Graph' section (node list, edge list as source -> target with " +
408
+ "labels, Mermaid graph TD) when the page shows a flowchart or node graph, and the " +
409
+ "placeholder image reference. No commentary about the merging.\n\n" +
410
+ fullOk.join("\n\n") +
411
+ (quadrantReads.length > 0 ? `\n\n## Quadrant readings\n\n${quadrantReads.join("\n\n")}` : "")
412
+ try {
413
+ return await apiChatImage(call, ADJUDICATOR_MODEL, b64, mime, adjudication)
414
+ } catch (e) {
415
+ log(`adjudication failed (${(e as Error).message}), merging text-only`)
416
+ const mergePrompt =
417
+ "Merge these independent readings of the same diagram page into a single accurate " +
418
+ "Markdown document. Deduplicate components, resolve conflicting values in favour of " +
419
+ "the majority and normalize equivalent units (0.1uF = 100nF). Keep every correct " +
420
+ "detail. Output the final merged Markdown only.\n\n" +
421
+ fullOk.join("\n\n") +
422
+ (quadrantReads.length > 0 ? `\n\n## Quadrant readings\n\n${quadrantReads.join("\n\n")}` : "")
423
+ return apiChat(call, REFINE_MODEL, [{ role: "user", content: mergePrompt }])
424
+ }
425
+ }
426
+
427
+ /** Mixed page: OCR for the text, diagram analysis for the figures, merged. */
428
+ export const transcribeMixedPage = async (call: RelayCall, pagePng: string, page: number, total: number): Promise<string> => {
429
+ const ocrResult = await transcribeTextPage(call, pagePng, page, total)
430
+ const diagramResult = await transcribeDiagramPage(call, pagePng, page, total)
431
+ const mergePrompt =
432
+ "Below are two transcriptions of the same document page. The first was done " +
433
+ "by an OCR model (good at text), the second by a vision model (good at diagrams). " +
434
+ "Merge them into a single accurate Markdown document. Use the OCR version as the " +
435
+ "base for text content. If the diagram version found component tables or technical " +
436
+ "details missing from the OCR version, incorporate them. Deduplicate. " +
437
+ "Output the final merged Markdown only.\n\n" +
438
+ `## OCR Transcription\n\n${ocrResult}\n\n## Diagram Transcription\n\n${diagramResult}`
439
+ return apiChat(call, REFINE_MODEL, [{ role: "user", content: mergePrompt }])
440
+ }
441
+
442
+ // ─── Stage 4: Refinement loop ────────────────────────────────────────────────
443
+
444
+ const REFINE_PROMPT = `You are a meticulous editor. Below is Markdown extracted from a document page, and the original page image.
445
+ Compare the Markdown against the image and fix any issues:
446
+ - Broken or incomplete tables (compare every cell against the image)
447
+ - Missing text or labels
448
+ - Garbled or misread characters (especially numbers and symbols)
449
+ - Inconsistent heading hierarchy
450
+ - Missing footnotes or annotations
451
+ - Lost or incorrect formatting (bold, italic, code blocks)
452
+ - Math equations that don't match the image
453
+
454
+ Output the corrected Markdown ONLY. No commentary, no explanations about what you changed.
455
+
456
+ Current Markdown:
457
+ {markdown}`
458
+
459
+ const refinePage = async (call: RelayCall, pagePng: string, markdown: string, round: number): Promise<string> => {
460
+ const [b64, mime] = await imageToB64(pagePng)
461
+ const result = await apiChatImage(call, REFINE_MODEL, b64, mime, REFINE_PROMPT.replace("{markdown}", markdown))
462
+ // Sanity check: if refinement produced empty or much shorter output, keep original
463
+ if (result.trim().length < markdown.trim().length * 0.3) {
464
+ log(` refine round ${round}: output suspiciously short (${result.length} vs ${markdown.length} chars), keeping previous`)
465
+ return markdown
466
+ }
467
+ return result
468
+ }
469
+
470
+ export const refineLoop = async (call: RelayCall, pagePng: string, markdown: string, maxRounds = MAX_REFINE_ROUNDS): Promise<string> => {
471
+ if (!markdown.trim()) return markdown
472
+ let current = markdown
473
+ for (let r = 1; r <= maxRounds; r++) {
474
+ log(` refine round ${r}/${maxRounds}...`)
475
+ const refined = await refinePage(call, pagePng, current, r)
476
+ if (refined.trim() === current.trim()) {
477
+ log(` refine round ${r}: no changes, stopping early`)
478
+ break
479
+ }
480
+ if (refined.trim().length === current.trim().length) {
481
+ log(` refine round ${r}: same length, likely converged`)
482
+ current = refined
483
+ break
484
+ }
485
+ current = refined
486
+ }
487
+ return current
488
+ }
489
+
490
+ // ─── Full per-page pipeline ──────────────────────────────────────────────────
491
+
492
+ /** Remove an outermost ```markdown fence that wraps the ENTIRE output — inner
493
+ * code blocks are preserved. */
494
+ export const stripCodeFences = (text: string): string => {
495
+ const stripped = text.trim()
496
+ if (!stripped.startsWith("```")) return text
497
+ const lines = stripped.split("\n")
498
+ if (lines.length < 2) return text
499
+ if (!["```markdown", "```md", "```"].includes(lines[0]!.trim())) return text
500
+ if (lines[lines.length - 1]!.trim() !== "```") return text
501
+ return lines.slice(1, -1).join("\n").trim()
502
+ }
503
+
504
+ export const processPage = async (call: RelayCall, pagePng: string, page: number, total: number, doRefine: boolean): Promise<[number, string]> => {
505
+ const pageType = await classifyPage(call, pagePng)
506
+ log(`page ${page}: classified as ${pageType}`)
507
+ let markdown: string
508
+ try {
509
+ if (pageType === "TEXT" || pageType === "TABLE_MATH") {
510
+ markdown = await transcribeTextPage(call, pagePng, page, total)
511
+ } else if (pageType === "DIAGRAM") {
512
+ try {
513
+ markdown = await transcribeDiagramPage(call, pagePng, page, total)
514
+ } catch (e) {
515
+ log(`page ${page}: diagram model failed (${(e as Error).message}), falling back to OCR`)
516
+ markdown = await transcribeTextPage(call, pagePng, page, total)
517
+ }
518
+ } else {
519
+ try {
520
+ markdown = await transcribeMixedPage(call, pagePng, page, total)
521
+ } catch (e) {
522
+ log(`page ${page}: mixed model failed (${(e as Error).message}), falling back to OCR`)
523
+ markdown = await transcribeTextPage(call, pagePng, page, total)
524
+ }
525
+ }
526
+ } catch (e) {
527
+ log(`page ${page}: transcription failed (${(e as Error).message}), using placeholder`)
528
+ return [page, `<!-- Page ${page}: transcription failed: ${(e as Error).message} -->\n`]
529
+ }
530
+
531
+ try {
532
+ if (doRefine) {
533
+ markdown = await refineLoop(call, pagePng, markdown, pageType === "TEXT" ? 1 : MAX_REFINE_ROUNDS)
534
+ }
535
+ } catch (e) {
536
+ log(`page ${page}: refinement failed (${(e as Error).message}), using unrefined output`)
537
+ }
538
+ return [page, stripCodeFences(markdown)]
539
+ }
540
+
541
+ // ─── Concurrency auto-tuning ─────────────────────────────────────────────────
542
+
543
+ /** Probe throughput at increasing parallelism; the gateway's key pool absorbs
544
+ * the load, so this measures what the pipeline will actually see. Probes up
545
+ * to 16 workers — with several keys the pool admits far more than 8 in
546
+ * parallel, and long documents want every slot. */
547
+ export const probeConcurrency = async (call: RelayCall): Promise<number> => {
548
+ log("auto-tuning concurrency...")
549
+ const probeMsg: Msg[] = [{ role: "user", content: "Reply with exactly: OK" }]
550
+ const best = { workers: 3, throughput: 0 }
551
+ for (const n of [1, 3, 5, 8, 12, 16]) {
552
+ try {
553
+ const start = performance.now()
554
+ const results = await Promise.allSettled(Array.from({ length: n }, () => apiChat(call, REFINE_MODEL, probeMsg, { maxTokens: 10 })))
555
+ if (results.some((r) => r.status === "rejected")) throw new Error("a probe request failed")
556
+ const elapsed = (performance.now() - start) / 1000
557
+ const throughput = n / elapsed
558
+ log(` ${n} parallel requests: ${elapsed.toFixed(1)}s → ${throughput.toFixed(2)} req/s`)
559
+ if (throughput > best.throughput) {
560
+ best.throughput = throughput
561
+ best.workers = n
562
+ }
563
+ } catch (e) {
564
+ log(` ${n} parallel requests: failed (${(e as Error).message}), backing off`)
565
+ break
566
+ }
567
+ }
568
+ const workers = Math.min(16, Math.max(2, best.workers))
569
+ log(`optimal concurrency: ${workers} workers (${best.throughput.toFixed(2)} req/s)`)
570
+ return workers
571
+ }
572
+
573
+ /** Fixed-size worker pool over an index cursor. */
574
+ async function runPool<T>(items: T[], workers: number, fn: (item: T, index: number) => Promise<void>): Promise<void> {
575
+ let next = 0
576
+ const worker = async (): Promise<void> => {
577
+ for (;;) {
578
+ const i = next++
579
+ if (i >= items.length) return
580
+ await fn(items[i]!, i)
581
+ }
582
+ }
583
+ await Promise.all(Array.from({ length: Math.min(workers, items.length) }, worker))
584
+ }
585
+
586
+ // ─── Cross-page table merging ────────────────────────────────────────────────
587
+
588
+ const TABLE_MERGE_PROMPT = `Below are consecutive pages of Markdown extracted from a PDF.
589
+ Some tables may span across page boundaries. Your job:
590
+ 1. Merge any tables that were split across pages into single coherent tables.
591
+ 2. Fix heading hierarchy (ensure H1→H2→H3 flow makes sense across pages).
592
+ 3. Do NOT change the content of individual cells, only merge structure.
593
+ 4. Do NOT delete anything: keep every heading, paragraph, list, footnote, figure placeholder, caption, code block and component table from every page. If two pages contain the same figure placeholder, keep one and keep its caption.
594
+ 5. Output the complete merged Markdown.
595
+
596
+ Pages:
597
+ {pages}`
598
+
599
+ export const crossPageMerge = async (call: RelayCall, pageMarkdowns: string[]): Promise<string> => {
600
+ const joinFinal = (parts: string[]) => parts.join("\n\n---\n\n")
601
+ if (pageMarkdowns.length <= 1) return joinFinal(pageMarkdowns)
602
+ const mergeChunk = async (chunk: string[]): Promise<string> => {
603
+ const text = chunk.join("\n\n---PAGE BREAK---\n\n")
604
+ try {
605
+ return await apiChat(call, REFINE_MODEL, [{ role: "user", content: TABLE_MERGE_PROMPT.replace("{pages}", text) }])
606
+ } catch {
607
+ return joinFinal(chunk)
608
+ }
609
+ }
610
+ const joined = pageMarkdowns.join("\n\n---PAGE BREAK---\n\n")
611
+ // For large documents, merge in chunks to avoid context overflow
612
+ const maxChunkChars = 100_000 // ~25K tokens, safe for 128K context models
613
+ if (joined.length <= maxChunkChars) {
614
+ const merged = await mergeChunk(pageMarkdowns)
615
+ return merged.replaceAll("\n\n---PAGE BREAK---\n\n", "\n\n---\n\n").replaceAll("---PAGE BREAK---", "---")
616
+ }
617
+ log(`document is large (${joined.length} chars), merging in chunks...`)
618
+ const chunks: string[] = []
619
+ let current: string[] = []
620
+ let size = 0
621
+ for (const md of pageMarkdowns) {
622
+ if (size + md.length > maxChunkChars && current.length > 0) {
623
+ chunks.push(await mergeChunk(current))
624
+ current = [md]
625
+ size = md.length
626
+ } else {
627
+ current.push(md)
628
+ size += md.length
629
+ }
630
+ }
631
+ if (current.length > 0) chunks.push(await mergeChunk(current))
632
+ return chunks.join("\n\n---\n\n").replaceAll("---PAGE BREAK---", "---")
633
+ }
634
+
635
+ // ─── Main pipeline ───────────────────────────────────────────────────────────
636
+
637
+ const argValue = (args: string[], name: string): string | undefined => {
638
+ const i = args.indexOf(name)
639
+ return i >= 0 ? args[i + 1] : undefined
640
+ }
641
+
642
+ const VALUE_FLAGS = new Set(["--workers", "--dpi", "--dedup-threshold"])
643
+
644
+ /** Positional arguments = everything that is not a flag or a flag's value. */
645
+ const positionals = (args: string[]): string[] => {
646
+ const out: string[] = []
647
+ for (let i = 0; i < args.length; i++) {
648
+ const a = args[i]!
649
+ if (a.startsWith("--")) {
650
+ if (VALUE_FLAGS.has(a)) i++ // skip the flag's value
651
+ continue
652
+ }
653
+ if (VALUE_FLAGS.has(args[i - 1] ?? "")) continue // a flag value that looks positional
654
+ out.push(a)
655
+ }
656
+ return out
657
+ }
658
+
659
+ export async function main(argv: string[], call: RelayCall = gatewayCall): Promise<number> {
660
+ const [input, output] = positionals(argv)
661
+ if (!input) {
662
+ console.error("usage: pdf2md.ts <input.pdf> [output.md] [--workers N] [--dpi 300] [--no-dedup] [--dedup-threshold 0.98] [--no-refine] [--no-merge] [--verbose]")
663
+ return 2
664
+ }
665
+ VERBOSE = argv.includes("--verbose") || argv.includes("-v")
666
+ const dpi = Number(argValue(argv, "--dpi") ?? DEFAULT_DPI)
667
+ const dedupThreshold = Number(argValue(argv, "--dedup-threshold") ?? DEDUP_SIMILARITY_THRESHOLD)
668
+ const noDedup = argv.includes("--no-dedup")
669
+ const noRefine = argv.includes("--no-refine")
670
+ const noMerge = argv.includes("--no-merge")
671
+
672
+ if (!(await Bun.file(input).exists())) {
673
+ console.error(`input not found: ${input}`)
674
+ return 2
675
+ }
676
+ const pageCount = await getPageCount(input)
677
+ if (pageCount === 0) {
678
+ console.error("could not determine page count (pdfinfo not available or failed)")
679
+ return 2
680
+ }
681
+ log(`PDF: ${input.split("/").pop()} (${pageCount} pages)`)
682
+
683
+ const tempDir = await mkdtemp(join(tmpdir(), "pdf2md-"))
684
+ try {
685
+ log(`rendering ${pageCount} pages at ${dpi} DPI...`)
686
+ let pagePngs = await renderPdfPages(input, dpi, tempDir)
687
+ log(`rendered ${pagePngs.length} page images`)
688
+ if (pagePngs.length !== pageCount) {
689
+ log(`WARNING: pdfinfo said ${pageCount} pages but rendered ${pagePngs.length} images`)
690
+ }
691
+
692
+ if (!noDedup && pagePngs.length > 1) {
693
+ log("deduplicating incremental-reveal pages...")
694
+ pagePngs = await dedupIncrementalReveals(pagePngs, dedupThreshold)
695
+ if (pagePngs.length === 0) {
696
+ console.error("all pages were deduplicated — nothing to process")
697
+ return 2
698
+ }
699
+ }
700
+ const total = pagePngs.length
701
+
702
+ let workers = Number(argValue(argv, "--workers") ?? 0)
703
+ if (!Number.isFinite(workers) || workers <= 0) {
704
+ workers = total <= 2 ? total : await probeConcurrency(call)
705
+ }
706
+ log(`processing ${total} pages with ${workers} workers...`)
707
+
708
+ const results = new Map<number, string>()
709
+ let completed = 0
710
+ await runPool(pagePngs, workers, async (png, idx) => {
711
+ try {
712
+ const [page, markdown] = await processPage(call, png, idx + 1, total, !noRefine)
713
+ results.set(idx, markdown)
714
+ } catch (e) {
715
+ log(`page ${idx + 1} FAILED: ${(e as Error).message}`, true)
716
+ results.set(idx, `<!-- Page ${idx + 1} failed: ${(e as Error).message} -->\n`)
717
+ }
718
+ completed++
719
+ logProgress(completed, total, `page ${idx + 1}`)
720
+ })
721
+
722
+ const ordered = Array.from({ length: total }, (_, i) => results.get(i) ?? `<!-- Page ${i + 1} missing -->\n`)
723
+
724
+ let finalMd: string
725
+ if (!noMerge && total > 1 && total <= 200) {
726
+ log("cross-page table merging...")
727
+ finalMd = await crossPageMerge(call, ordered)
728
+ } else {
729
+ if (total > 200) log(`skipping cross-page merge (document too large: ${total} pages)`)
730
+ finalMd = ordered.join("\n\n---\n\n")
731
+ }
732
+
733
+ if (output) {
734
+ await Bun.write(output, finalMd)
735
+ log(`output written to ${output} (${finalMd.length} chars)`, true)
736
+ } else {
737
+ process.stdout.write(finalMd)
738
+ }
739
+ } finally {
740
+ await rm(tempDir, { force: true, recursive: true }).catch(() => {})
741
+ }
742
+ return 0
743
+ }
744
+
745
+ if (import.meta.main) {
746
+ process.exit(await main(process.argv.slice(2)))
747
+ }