@youweichen/pi-harness 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/README.md +14 -0
  2. package/dist/server/document-conversion/extension.js +164 -0
  3. package/dist/server/document-conversion/python/bridge.py +424 -0
  4. package/dist/server/document-conversion/python/fixtures.py +201 -0
  5. package/dist/server/document-conversion/python/native_code.py +149 -0
  6. package/dist/server/document-conversion/runtime.js +321 -0
  7. package/dist/server/document-conversion/service.js +340 -0
  8. package/dist/server/document-conversion/settings.js +63 -0
  9. package/dist/server/index.js +2 -0
  10. package/dist/server/native-tools.js +4 -0
  11. package/dist/server/okf/extension.js +118 -0
  12. package/dist/server/okf/markdown-sources.js +58 -0
  13. package/dist/server/okf/service.js +713 -0
  14. package/dist/server/okf/storage.js +165 -0
  15. package/dist/server/okf/types.js +1 -0
  16. package/dist/server/wiki-links.js +24 -0
  17. package/dist/server/wiki-service.js +37 -5
  18. package/package.json +2 -2
  19. package/web/dist/assets/{NodeWorkbench-BiP2ktQP.js → NodeWorkbench-DM6YTS5e.js} +1 -1
  20. package/web/dist/assets/{TerminalPanel-CKaSUsN-.js → TerminalPanel-aFJylyxg.js} +1 -1
  21. package/web/dist/assets/{abnfDiagram-VCTEODGH-CnwRJDRL.js → abnfDiagram-VCTEODGH-CcM7f7oX.js} +1 -1
  22. package/web/dist/assets/{arc-CG4Q7-W_.js → arc-Bv7vS5l5.js} +1 -1
  23. package/web/dist/assets/{architectureDiagram-5GKGNRK7-BrGqX_Rg.js → architectureDiagram-5GKGNRK7-Cl8QcB3w.js} +1 -1
  24. package/web/dist/assets/{blockDiagram-I7D4REHJ-C2t0PjZT.js → blockDiagram-I7D4REHJ-DarS2-4G.js} +1 -1
  25. package/web/dist/assets/{c4Diagram-7LVT6UL2-CadrcgGL.js → c4Diagram-7LVT6UL2-BY0_ttBt.js} +1 -1
  26. package/web/dist/assets/channel-DJGMgVTh.js +1 -0
  27. package/web/dist/assets/{chunk-2Q5K7J3B-CC3V2ITZ.js → chunk-2Q5K7J3B-Cz-zS7_X.js} +1 -1
  28. package/web/dist/assets/{chunk-5VM5RSS4-DqKSbGmE.js → chunk-5VM5RSS4-B7AyqkmI.js} +1 -1
  29. package/web/dist/assets/{chunk-F27PBJKO-_CDr3JbY.js → chunk-F27PBJKO-B7hVduzL.js} +1 -1
  30. package/web/dist/assets/{chunk-IMKFNOWR-BIR4H4fX.js → chunk-IMKFNOWR-DNGJqYn5.js} +1 -1
  31. package/web/dist/assets/{chunk-JWPE2WC7-CrsdBGXT.js → chunk-JWPE2WC7-B5iJgXvl.js} +1 -1
  32. package/web/dist/assets/{chunk-POPQ4Y6H-C6q9kQmf.js → chunk-POPQ4Y6H-BvWCkriR.js} +1 -1
  33. package/web/dist/assets/{chunk-SVP7TREG-l2XXTr2v.js → chunk-SVP7TREG-BcdlBuOQ.js} +1 -1
  34. package/web/dist/assets/{chunk-TICWLB2K-29I_mZgB.js → chunk-TICWLB2K-B1igqjQd.js} +1 -1
  35. package/web/dist/assets/{chunk-XXDRQBXY-htV9FBqi.js → chunk-XXDRQBXY-DKGDquaw.js} +1 -1
  36. package/web/dist/assets/classDiagram-ZZMXUADV-D9nxgF0M.js +1 -0
  37. package/web/dist/assets/classDiagram-v2-VYDZK3BY-D9nxgF0M.js +1 -0
  38. package/web/dist/assets/{cose-bilkent-JH36ORCC-Cop-ihyx.js → cose-bilkent-JH36ORCC-BU_9nu88.js} +1 -1
  39. package/web/dist/assets/{cynefin-OW5HDTMX-IEk8qBRq.js → cynefin-OW5HDTMX-DY0BDLjc.js} +1 -1
  40. package/web/dist/assets/{cynefinDiagram-5FMLGOSQ-Bmzt6Qb5.js → cynefinDiagram-5FMLGOSQ-B-zM25Fk.js} +1 -1
  41. package/web/dist/assets/{dagre-GXQ25YYZ-DiXTJYYU.js → dagre-GXQ25YYZ-BxooDyBc.js} +1 -1
  42. package/web/dist/assets/{diagram-S7CK7UJ4-BT7CLuNk.js → diagram-S7CK7UJ4-CCO-8de9.js} +1 -1
  43. package/web/dist/assets/{diagram-UQ7AKVKN-BZZ3nTCO.js → diagram-UQ7AKVKN-CmGsH525.js} +1 -1
  44. package/web/dist/assets/{diagram-VSXAHHWV-Dal8LeZh.js → diagram-VSXAHHWV-CSQJYo3m.js} +1 -1
  45. package/web/dist/assets/{diagram-VX7I27RA-D6V9FfNA.js → diagram-VX7I27RA-CwuJsF66.js} +1 -1
  46. package/web/dist/assets/{diagram-Z3DM3KII-KUAIcFiD.js → diagram-Z3DM3KII-BF13T_lj.js} +1 -1
  47. package/web/dist/assets/{ebnfDiagram-PWID7BFC-s0FEPZrV.js → ebnfDiagram-PWID7BFC-ZtoDLro6.js} +1 -1
  48. package/web/dist/assets/{erDiagram-RLTQ6QDP-CiIQPFCD.js → erDiagram-RLTQ6QDP-BBo-qmNH.js} +1 -1
  49. package/web/dist/assets/{flowDiagram-HODETNUW-CvG4MYWn.js → flowDiagram-HODETNUW-DAq9qdP8.js} +1 -1
  50. package/web/dist/assets/{ganttDiagram-EL5Y4UJY-CUf115ii.js → ganttDiagram-EL5Y4UJY-C53eRlMu.js} +1 -1
  51. package/web/dist/assets/{gitGraphDiagram-WWUBYQGX-o3K1r6fL.js → gitGraphDiagram-WWUBYQGX-BviBMJxj.js} +1 -1
  52. package/web/dist/assets/{index-C5or_GAq.js → index-Bab44wX8.js} +81 -81
  53. package/web/dist/assets/{infoDiagram-27XIBGKW-BxiBeQ8E.js → infoDiagram-27XIBGKW-DI95SO3j.js} +1 -1
  54. package/web/dist/assets/{ishikawaDiagram-5VMMS53U-B1ONuCkH.js → ishikawaDiagram-5VMMS53U-Cl8IDXwf.js} +1 -1
  55. package/web/dist/assets/{journeyDiagram-3NMN7TZE-DwmuoXcX.js → journeyDiagram-3NMN7TZE-Co6aJeTZ.js} +1 -1
  56. package/web/dist/assets/{kanban-definition-UXKFOSKX-pMKHubxH.js → kanban-definition-UXKFOSKX-CknQQpwW.js} +1 -1
  57. package/web/dist/assets/{layout-CEwfjQR0.js → layout-CFtUmtTW.js} +1 -1
  58. package/web/dist/assets/{linear-D22Gdbt1.js → linear-mRlGUAKy.js} +1 -1
  59. package/web/dist/assets/{mermaid.core-CO2nrD04.js → mermaid.core-CvQEO3ue.js} +5 -5
  60. package/web/dist/assets/{mindmap-definition-YA3MSWOX-9MQn9ZAp.js → mindmap-definition-YA3MSWOX-DpqhsxUl.js} +1 -1
  61. package/web/dist/assets/{pegDiagram-XKGWAZYB-Dy6fLjM-.js → pegDiagram-XKGWAZYB-C9eawB5u.js} +1 -1
  62. package/web/dist/assets/{pieDiagram-E7YTZNPT-EDLez6nA.js → pieDiagram-E7YTZNPT-8vaqXVD6.js} +1 -1
  63. package/web/dist/assets/{quadrantDiagram-AXDQQJYC-Cgcd9mvy.js → quadrantDiagram-AXDQQJYC-Diwr8ynm.js} +1 -1
  64. package/web/dist/assets/{railroadDiagram-O6MQD6OU-B4Td0tjj.js → railroadDiagram-O6MQD6OU-krW50Cgi.js} +1 -1
  65. package/web/dist/assets/{requirementDiagram-BXWQKSXE-aF2QNZ78.js → requirementDiagram-BXWQKSXE-BPLI6eMG.js} +1 -1
  66. package/web/dist/assets/{sankeyDiagram-P5KCCOFB-DgTYwYMa.js → sankeyDiagram-P5KCCOFB-Bix_elmH.js} +1 -1
  67. package/web/dist/assets/{sequenceDiagram-WJ2MYXX4-EUcddWUc.js → sequenceDiagram-WJ2MYXX4-0m0xaUKi.js} +1 -1
  68. package/web/dist/assets/{sizeCapture-INFHLROL-CXHOv9Sx.js → sizeCapture-INFHLROL-BtzJ_oOR.js} +1 -1
  69. package/web/dist/assets/{stateDiagram-D77RDMKH-B-xlgfaK.js → stateDiagram-D77RDMKH-Clq5bN0v.js} +1 -1
  70. package/web/dist/assets/stateDiagram-v2-MP3YSRHH-Bn1oW4gp.js +1 -0
  71. package/web/dist/assets/{swimlanes-42K2YHIH-BtFBOCDh.js → swimlanes-42K2YHIH-11MDGnYA.js} +1 -1
  72. package/web/dist/assets/swimlanesDiagram-VR7AAH4N-BW96Edgo.js +8 -0
  73. package/web/dist/assets/{timeline-definition-24CTP7MA-CZlQY6pb.js → timeline-definition-24CTP7MA-BeQfWP95.js} +1 -1
  74. package/web/dist/assets/{vennDiagram-4TSXK5OY-Cn61dJvL.js → vennDiagram-4TSXK5OY-BMYkX0KL.js} +1 -1
  75. package/web/dist/assets/{wardleyDiagram-VM6X3IG4-fiMNrWpr.js → wardleyDiagram-VM6X3IG4-BV3fTUkU.js} +1 -1
  76. package/web/dist/assets/{xychartDiagram-S5SC5T6Z-CuzM_sof.js → xychartDiagram-S5SC5T6Z-Ct7ezLnc.js} +1 -1
  77. package/web/dist/index.html +1 -1
  78. package/web/dist/assets/channel-C27Jizb6.js +0 -1
  79. package/web/dist/assets/classDiagram-ZZMXUADV-DYTBrAEP.js +0 -1
  80. package/web/dist/assets/classDiagram-v2-VYDZK3BY-DYTBrAEP.js +0 -1
  81. package/web/dist/assets/stateDiagram-v2-MP3YSRHH-kwAwAlWX.js +0 -1
  82. package/web/dist/assets/swimlanesDiagram-VR7AAH4N-D5j5Cj3S.js +0 -8
package/README.md CHANGED
@@ -40,6 +40,18 @@ pi-harness --port 9000 --cwd /path/to/project
40
40
  - 中文和英文界面、声音提醒、可选插件及 pi 扩展。
41
41
  - CLI、开机自启服务、Docker 与 Electron 桌面版。
42
42
 
43
+ ## 文档转换与 OKF Wiki
44
+
45
+ 内置 PDF 转 Markdown 扩展支持扫描件 OCR、表格、代码和来源定位;OKF 扩展让当前 Agent 整理 PDF、Markdown、文本和 Office 文件,提取候选知识、核对重复与冲突,再生成带原文证据的 OKF Wiki。结果可在现有文档工作台中打开。
46
+
47
+ ```text
48
+ /pdf-md setup
49
+ /pdf-md convert ./raw/技术说明.pdf
50
+ /okf ingest ./raw,输出到 ./knowledge
51
+ ```
52
+
53
+ PDF 与 Office 解析需要单独的本机 Python 3.12 环境,首次运行 `setup` 会安装依赖并下载模型;可用 `/pdf-md cancel` 取消、`/pdf-md doctor` 检查。MD/TXT 不需要此环境。知识整理使用当前会话选定的模型,疑点与冲突保留草稿,原文快照随知识目录保存。设置、恢复与资料移交见 [文档知识扩展说明](https://github.com/youweichen0208/pi-harness/blob/develop/docs/architecture-document-knowledge.md)。
54
+
43
55
  ## 运行与部署
44
56
 
45
57
  ```bash
@@ -68,6 +80,8 @@ npm run test:smoke
68
80
 
69
81
  pi-harness is an independently maintained web and Electron interface for the pi coding agent SDK. It includes chat, file browsing and editing, attachments, a local terminal, model management, and a built-in SSH node workbench with grouped hosts, multiple PTY tabs, SFTP files, and a separate Agent conversation for each node.
70
82
 
83
+ Built-in document extensions provide local PDF-to-Markdown conversion and an Agent-driven OKF Wiki workflow for PDF, Markdown, text and Office files. Run `/pdf-md setup` to prepare the separate local parser, `/pdf-md convert <path>` to convert a PDF, or `/okf ingest <directory>` to organize sourced knowledge with the current conversation's model. See the [document extensions guide](https://github.com/youweichen0208/pi-harness/blob/develop/docs/architecture-document-knowledge.md).
84
+
71
85
  Requires Node.js **>= 22.19.0** and a configured pi model provider. Install with `npm install -g @youweichen/pi-harness`, then run `pi-harness`. See the [SSH workbench guide](https://github.com/youweichen0208/pi-harness/blob/develop/docs/ssh-workbench.md), [Xshell guide](https://github.com/youweichen0208/pi-harness/blob/develop/docs/xshell.md), and [deployment guide](https://github.com/youweichen0208/pi-harness/blob/develop/docs/deployment.md).
72
86
 
73
87
  ## License
@@ -0,0 +1,164 @@
1
+ import { createHash } from "node:crypto";
2
+ import { lstat, realpath, stat } from "node:fs/promises";
3
+ import { basename, dirname, extname, isAbsolute, relative, resolve, sep } from "node:path";
4
+ import { Type } from "typebox";
5
+ import { convertDocument } from "./service.js";
6
+ import { doctorRuntime, setupRuntime } from "./runtime.js";
7
+ import { getDocumentSettings, updateDocumentSettings } from "./settings.js";
8
+ const pathParameter = Type.String({ minLength: 1, maxLength: 4096 });
9
+ let activeSetup;
10
+ export const pdfMarkdownParameters = Type.Object({
11
+ inputPath: pathParameter,
12
+ outputDir: Type.Optional(Type.String({ minLength: 1, maxLength: 4096, description: "Output directory inside the current workspace. Defaults to converted/<filename>-<path hash>. Existing edited files are never overwritten." })),
13
+ });
14
+ export const pdfMarkdownOutput = Type.Object({
15
+ status: Type.Union([Type.Literal("complete"), Type.Literal("partial")]),
16
+ markdownPath: Type.String(), structurePath: Type.String(), sourceMapPath: Type.String(), assetsDir: Type.String(),
17
+ sourceHash: Type.String(), parserVersion: Type.String(), warnings: Type.Array(Type.String()),
18
+ });
19
+ /** Output files must remain accessible through the workspace's existing file UI. */
20
+ async function outputDirectory(cwd, input, requested) {
21
+ const root = await realpath(cwd);
22
+ const hash = createHash("sha256").update(input).digest("hex").slice(0, 8);
23
+ const destination = resolve(root, requested ?? `converted/${basename(input, extname(input))}-${hash}`);
24
+ const within = (target) => { const rel = relative(root, target); return rel !== ".." && !rel.startsWith(`..${sep}`) && !isAbsolute(rel); };
25
+ if (!within(destination) || destination === root)
26
+ throw Error("Choose an output directory inside the current workspace.");
27
+ let ancestor = destination;
28
+ while (true) {
29
+ try {
30
+ await lstat(ancestor);
31
+ }
32
+ catch (error) {
33
+ if (error.code !== "ENOENT")
34
+ throw error;
35
+ const parent = dirname(ancestor);
36
+ if (parent === ancestor)
37
+ throw error;
38
+ ancestor = parent;
39
+ continue;
40
+ }
41
+ // A dangling symlink exists according to lstat: its failed realpath must not
42
+ // be mistaken for a missing directory and skipped to the parent.
43
+ if (!within(await realpath(ancestor)) || !(await stat(ancestor)).isDirectory())
44
+ throw Error("Output directory leaves the workspace or is not a directory.");
45
+ return destination;
46
+ }
47
+ }
48
+ async function pdfSettings(args, ctx) {
49
+ if (!ctx.isIdle()) {
50
+ ctx.ui.notify("请等待当前任务结束后修改扩展设置。", "warning");
51
+ return;
52
+ }
53
+ const current = getDocumentSettings();
54
+ let action = args.trim();
55
+ if (!action && ctx.hasUI)
56
+ action = await ctx.ui.select("PDF 转 Markdown 设置", ["启用", "停用", "Python 路径"]) ?? "";
57
+ if (["on", "启用"].includes(action))
58
+ await updateDocumentSettings({ pdfEnabled: true });
59
+ else if (["off", "停用"].includes(action))
60
+ await updateDocumentSettings({ pdfEnabled: false });
61
+ else if (action === "Python 路径" || action.startsWith("python ")) {
62
+ const path = action.startsWith("python ") ? action.slice(7).trim() : await ctx.ui.input("Python 可执行文件路径(留空恢复自动发现)", current.pythonPath ?? "");
63
+ if (path === undefined)
64
+ return;
65
+ await updateDocumentSettings({ pythonPath: path.trim() || undefined });
66
+ }
67
+ else {
68
+ ctx.ui.notify(`PDF 转 Markdown:${current.pdfEnabled ? "启用" : "停用"};Python:${current.pythonPath ?? "自动发现"}。用 /pdf-md settings on|off|python <路径> 修改。`, "info");
69
+ return;
70
+ }
71
+ ctx.ui.notify("设置已保存。请使用 /reload 或新建会话更新工具列表;停用后现有工具调用会立即被拒绝。", "info");
72
+ }
73
+ export const createPdfMarkdownExtension = () => pi => {
74
+ pi.registerTool({
75
+ name: "pdf_to_markdown", label: "PDF to Markdown",
76
+ description: "Convert a local PDF, including scans, to structured Markdown with local Docling OCR, tables, code, images and source locations. Uses the local parser without extra chat model calls. Returns file paths and quality warnings. Read the Markdown and warnings before reporting quality; do not invent missing content. Link the resulting workspace-relative Markdown file in your final answer.",
77
+ exposure: getDocumentSettings().pdfEnabled ? "deferred" : "hidden", executionMode: "sequential",
78
+ annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: false },
79
+ parameters: pdfMarkdownParameters, outputSchema: pdfMarkdownOutput,
80
+ async execute(_id, args, signal, onUpdate, ctx) {
81
+ if (!getDocumentSettings().pdfEnabled)
82
+ throw Error("PDF conversion is disabled. Enable it with /pdf-md settings on.");
83
+ signal?.throwIfAborted();
84
+ const cwd = await realpath(ctx.cwd);
85
+ const inputPath = await realpath(resolve(cwd, args.inputPath));
86
+ if (extname(inputPath).toLowerCase() !== ".pdf")
87
+ throw Error("pdf_to_markdown accepts PDF files; use okf_ingest for other document formats.");
88
+ const outputDir = await outputDirectory(cwd, inputPath, args.outputDir);
89
+ const result = await convertDocument({ inputPath, outputDir, signal, onProgress: text => onUpdate?.({ content: [{ type: "text", text }], details: {} }) });
90
+ const localPath = (path) => relative(cwd, path).split(sep).join("/");
91
+ const data = { status: result.status, markdownPath: localPath(result.markdownPath), structurePath: localPath(result.structurePath), sourceMapPath: localPath(result.sourceMapPath), assetsDir: localPath(result.assetsDir), sourceHash: result.sourceHash, parserVersion: result.parserVersion, warnings: result.warnings };
92
+ return { content: [{ type: "text", text: `${JSON.stringify(data, null, 2)}\nRead the generated Markdown and conversion warnings. In your final reply, link [converted document](${data.markdownPath}).` }], structuredContent: data, details: {} };
93
+ },
94
+ });
95
+ pi.registerCommand("pdf-md", {
96
+ description: "PDF 转 Markdown;convert、setup、cancel、doctor、settings",
97
+ async handler(args, ctx) {
98
+ const [, action = "", rest = ""] = /^(\S+)?\s*([\s\S]*)$/.exec(args.trim()) ?? [];
99
+ if (action === "settings") {
100
+ await pdfSettings(rest, ctx);
101
+ return;
102
+ }
103
+ if (action === "cancel") {
104
+ if (!activeSetup) {
105
+ ctx.ui.notify("没有正在安装的文档解析环境。", "info");
106
+ return;
107
+ }
108
+ if (activeSetup.ownerSessionId !== ctx.sessionManager.getSessionId()) {
109
+ ctx.ui.notify("解析环境正在其他会话中安装,请在发起安装的会话中运行 /pdf-md cancel。", "warning");
110
+ return;
111
+ }
112
+ activeSetup.controller.abort();
113
+ ctx.ui.notify("已请求取消解析环境安装,正在清理安装进程。", "info");
114
+ return;
115
+ }
116
+ if (action === "doctor" || action === "setup") {
117
+ if (!ctx.isIdle()) {
118
+ ctx.ui.notify("请等待当前任务结束后检查或安装解析环境。", "warning");
119
+ return;
120
+ }
121
+ if (activeSetup) {
122
+ ctx.ui.notify("文档解析环境已经在安装中。可在发起安装的会话运行 /pdf-md cancel 取消。", "warning");
123
+ return;
124
+ }
125
+ const setup = action === "setup" ? { ownerSessionId: ctx.sessionManager.getSessionId(), controller: new AbortController() } : undefined;
126
+ if (setup)
127
+ activeSetup = setup;
128
+ const run = async () => {
129
+ try {
130
+ const signal = setup ? (ctx.signal ? AbortSignal.any([setup.controller.signal, ctx.signal]) : setup.controller.signal) : ctx.signal;
131
+ const result = setup ? await setupRuntime({ signal, onProgress: text => ctx.ui.setStatus("document-runtime", text) }) : await doctorRuntime({ signal });
132
+ ctx.ui.notify(JSON.stringify(result, null, 2), result.ready ? "info" : "warning");
133
+ }
134
+ catch (error) {
135
+ ctx.ui.notify(error.message, "error");
136
+ }
137
+ finally {
138
+ if (setup && activeSetup === setup)
139
+ activeSetup = undefined;
140
+ ctx.ui.setStatus("document-runtime", undefined);
141
+ }
142
+ };
143
+ if (setup) {
144
+ // Acknowledge this explicit command immediately so the same WebUI
145
+ // input remains available for /pdf-md cancel during installation.
146
+ ctx.ui.notify("已开始安装文档解析环境。进度显示在状态栏;可用 /pdf-md cancel 取消。", "info");
147
+ setup.task = run();
148
+ }
149
+ else
150
+ await run();
151
+ return;
152
+ }
153
+ if (action !== "convert" || !rest) {
154
+ ctx.ui.notify("用法:/pdf-md convert <PDF 路径及可选输出要求>;/pdf-md setup;/pdf-md cancel;/pdf-md doctor;/pdf-md settings", "info");
155
+ return;
156
+ }
157
+ if (!getDocumentSettings().pdfEnabled) {
158
+ ctx.ui.notify("PDF 转换已停用,请先用 /pdf-md settings on 启用。", "warning");
159
+ return;
160
+ }
161
+ pi.sendUserMessage(`Convert the PDF described below with the native pdf_to_markdown tool. Use tool_search to discover it if needed. Preserve the source document, inspect the conversion warnings, and give a clickable relative link to the resulting Markdown. Do not install dependencies unless I explicitly request setup.\n\n${rest}`, ctx.isIdle() ? undefined : { deliverAs: "followUp" });
162
+ },
163
+ });
164
+ };
@@ -0,0 +1,424 @@
1
+ """Versioned, local-only Docling worker. stdout is one JSON response; logs use stderr."""
2
+ import hashlib
3
+ import importlib.metadata
4
+ import json
5
+ import logging
6
+ import os
7
+ from pathlib import Path
8
+ import platform
9
+ import re
10
+ import shutil
11
+ import sys
12
+ import tempfile
13
+ from contextlib import redirect_stdout
14
+ from concurrent.futures import ThreadPoolExecutor
15
+
16
+ VERSION = "2.136.0"
17
+ PROFILE = "docling-cpu-zh-v1"
18
+ MODEL_REVISIONS = {
19
+ "docling-project/docling-layout-heron": "8f39ad3c0b4c58e9c2d2c84a38465abf757272d8",
20
+ "docling-project/docling-models": "fc0f2d45e2218ea24bce5045f58a389aed16dc23",
21
+ "docling-project/CodeFormulaV2": "ecedbe111d15c2dc60bfd4a823cbe80127b58af4",
22
+ }
23
+ logging.basicConfig(stream=sys.stderr, level=logging.WARNING)
24
+ # Hugging Face's HTTP downloader respects the host's proxy configuration and can
25
+ # resume downloads without requiring Xet's separate network transport.
26
+ os.environ.setdefault("HF_HUB_DISABLE_XET", "1")
27
+
28
+
29
+ def emit(value):
30
+ print(json.dumps(value, ensure_ascii=False), flush=True)
31
+
32
+
33
+ def sha256(path):
34
+ hash_value = hashlib.sha256()
35
+ with path.open("rb") as stream:
36
+ for chunk in iter(lambda: stream.read(1024 * 1024), b""):
37
+ hash_value.update(chunk)
38
+ return hash_value.hexdigest()
39
+
40
+
41
+ def parser_assets():
42
+ return {path.name: sha256(path) for path in sorted(Path(__file__).parent.glob("*.py"))}
43
+
44
+
45
+ LOADED_ASSET_HASHES = parser_assets()
46
+
47
+
48
+ def offline():
49
+ for name in ("HF_HUB_OFFLINE", "TRANSFORMERS_OFFLINE", "HF_DATASETS_OFFLINE", "HF_HUB_DISABLE_TELEMETRY"):
50
+ os.environ[name] = "1"
51
+ # Protect all optional backends as well as Hugging Face. Conversion never opens
52
+ # TCP connections, including those to loopback or inherited proxy endpoints.
53
+ def audit(event, args):
54
+ if event in ("socket.connect", "socket.getaddrinfo", "socket.sendto"):
55
+ raise PermissionError("Network access is disabled during document conversion")
56
+ sys.addaudithook(audit)
57
+
58
+
59
+ def doctor(root):
60
+ missing = []
61
+ warnings = []
62
+ try:
63
+ actual = importlib.metadata.version("docling")
64
+ if actual != VERSION:
65
+ missing.append("Expected Docling " + VERSION + "; found " + actual)
66
+ except importlib.metadata.PackageNotFoundError:
67
+ actual = None
68
+ missing.append("Docling is not installed")
69
+ if sys.version_info[:2] != (3, 12):
70
+ missing.append("This runtime profile requires Python 3.12")
71
+ try:
72
+ receipt = json.loads((root / "ready.json").read_text("utf-8"))
73
+ if receipt.get("profile") != PROFILE or receipt.get("docling") != VERSION or not receipt.get("smokePassed"):
74
+ missing.append("Runtime qualification receipt is invalid")
75
+ if receipt.get("parserAssets") != parser_assets():
76
+ missing.append("Parser implementation changed; run /pdf-md setup to requalify the offline profile")
77
+ for relative, metadata in receipt.get("models", {}).items():
78
+ path = root / "models" / relative
79
+ if not path.is_file() or path.stat().st_size != metadata["size"] or sha256(path) != metadata["sha256"]:
80
+ missing.append("Missing or changed model: " + relative)
81
+ for name, version in receipt.get("packages", {}).items():
82
+ try:
83
+ if importlib.metadata.version(name) != version:
84
+ missing.append("Runtime dependency changed: " + name)
85
+ except importlib.metadata.PackageNotFoundError:
86
+ missing.append("Runtime dependency missing: " + name)
87
+ if not receipt.get("models"):
88
+ missing.append("No model artifacts were recorded")
89
+ except (OSError, ValueError, KeyError):
90
+ missing.append("Run /pdf-md setup to download models and qualify offline conversion")
91
+ return {"ready": not missing, "parserVersion": actual, "missing": missing, "warnings": warnings}
92
+
93
+
94
+ def converter(root):
95
+ from docling.datamodel.accelerator_options import AcceleratorDevice, AcceleratorOptions
96
+ from docling.datamodel.base_models import InputFormat
97
+ from docling.datamodel.pipeline_options import (
98
+ CodeFormulaVlmOptions, OcrMode, PdfPipelineOptions, RapidOcrOptions,
99
+ TableFormerMode, TableStructureOptions,
100
+ )
101
+ from docling.datamodel.vlm_engine_options import TransformersVlmEngineOptions
102
+ from docling.document_converter import DocumentConverter, PdfFormatOption
103
+ options = PdfPipelineOptions(
104
+ artifacts_path=root / "models",
105
+ enable_remote_services=False,
106
+ allow_external_plugins=False,
107
+ accelerator_options=AcceleratorOptions(device=AcceleratorDevice.CPU, num_threads=4),
108
+ do_ocr=True,
109
+ ocr_options=RapidOcrOptions(backend="onnxruntime", lang=["ch"], model_size="small", mode=OcrMode.PDF_AWARE_LAYOUT_REGIONS),
110
+ do_table_structure=True,
111
+ table_structure_options=TableStructureOptions(mode=TableFormerMode.ACCURATE, do_cell_matching=True),
112
+ do_code_enrichment=True,
113
+ do_formula_enrichment=True,
114
+ code_formula_options=CodeFormulaVlmOptions.from_preset("codeformulav2", engine_options=TransformersVlmEngineOptions(
115
+ device=AcceleratorDevice.CPU, torch_dtype="float32", quantized=False, compile_model=False,
116
+ )),
117
+ generate_picture_images=True,
118
+ generate_page_images=True,
119
+ )
120
+ options.heading_hierarchy_options.enabled = True
121
+ return DocumentConverter(allowed_formats=[InputFormat.PDF, InputFormat.DOCX, InputFormat.PPTX, InputFormat.XLSX], format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=options)})
122
+
123
+
124
+ def extract_blocks(document, suffix):
125
+ from docling_core.types.doc import TableItem
126
+ blocks = []
127
+ head = None
128
+ for item, level in document.iterate_items():
129
+ label = str(getattr(item, "label", ""))
130
+ text = getattr(item, "text", "")
131
+ if isinstance(item, TableItem):
132
+ text = item.export_to_markdown(doc=document)
133
+ if not text:
134
+ continue
135
+ if "section_header" in label or "title" in label:
136
+ head = text
137
+ locator = {"heading": head} if head else {}
138
+ provenance = [p.model_dump(mode="json") for p in getattr(item, "prov", [])]
139
+ if provenance:
140
+ page = provenance[0].get("page_no")
141
+ if page:
142
+ locator["slide" if suffix == ".pptx" else "page"] = page
143
+ blocks.append({"id": getattr(item, "self_ref", "block-" + str(len(blocks) + 1)), "text": text, "locator": locator, "provenance": provenance, "kind": label})
144
+ return blocks
145
+
146
+
147
+ def excel_provenance(source, warnings):
148
+ from openpyxl import load_workbook
149
+ formulas = load_workbook(source, read_only=True, data_only=False)
150
+ values = load_workbook(source, read_only=True, data_only=True)
151
+ blocks = []
152
+ metadata = []
153
+ try:
154
+ for sheet in formulas:
155
+ cached = values[sheet.title]
156
+ for row, cached_row in zip(sheet.iter_rows(), cached.iter_rows()):
157
+ for cell, cached_cell in zip(row, cached_row):
158
+ if cell.value is None:
159
+ continue
160
+ value = cached_cell.value
161
+ is_formula = cell.data_type == "f"
162
+ if is_formula:
163
+ metadata.append({"sheet": sheet.title, "cell": cell.coordinate, "formula": str(cell.value), "cachedValue": value})
164
+ if value is None:
165
+ warnings.append("Missing formula cache: " + sheet.title + "!" + cell.coordinate)
166
+ blocks.append({"id": "cell-" + str(len(blocks) + 1), "text": str(value) if value is not None else "[formula result unavailable]", "locator": {"sheet": sheet.title, "cell": cell.coordinate}})
167
+ if metadata:
168
+ warnings.append("Spreadsheet formula results are saved cached values; no formulas were evaluated and cache freshness cannot be verified.")
169
+ finally:
170
+ formulas.close()
171
+ values.close()
172
+ return blocks, metadata
173
+
174
+
175
+ def convert(root, request, engine=None):
176
+ from docling_core.types.doc import ImageRefMode, TableItem
177
+ source, output = Path(request["inputPath"]), Path(request["outputDir"])
178
+ output.mkdir(parents=True, exist_ok=True)
179
+ assets = output / "assets"
180
+ assets.mkdir(exist_ok=True)
181
+ warnings = []
182
+ enrichment_failed = False
183
+ material_defect = False
184
+ native_verified_ids = set()
185
+ expected_pages = None
186
+ if source.suffix.lower() == ".pdf":
187
+ import pypdfium2
188
+ pdf = pypdfium2.PdfDocument(source)
189
+ try:
190
+ expected_pages = len(pdf)
191
+ native = []
192
+ for index in range(expected_pages):
193
+ page = pdf[index]
194
+ try:
195
+ textpage = page.get_textpage()
196
+ try:
197
+ native.append({"page": index + 1, "text": textpage.get_text_range()})
198
+ finally:
199
+ textpage.close()
200
+ finally:
201
+ page.close()
202
+ (output / "pdf-text-layer.json").write_text(json.dumps({"sourceHash": request["sourceHash"], "pages": native}, ensure_ascii=False), "utf-8")
203
+ finally:
204
+ pdf.close()
205
+ class Capture(logging.Handler):
206
+ def emit(self, record):
207
+ nonlocal enrichment_failed
208
+ if record.levelno >= logging.ERROR:
209
+ enrichment_failed = True
210
+ if record.levelno >= logging.WARNING and len(warnings) < 100:
211
+ warnings.append(record.getMessage()[:1000])
212
+ capture = Capture()
213
+ logging.getLogger().addHandler(capture)
214
+ try:
215
+ result = (engine or converter(root)).convert(source, raises_on_error=False)
216
+ finally:
217
+ logging.getLogger().removeHandler(capture)
218
+ state = str(getattr(result.status, "value", result.status))
219
+ if state not in ("success", "partial_success"):
220
+ raise RuntimeError("Docling conversion failed: " + state + "; " + "; ".join(str(error) for error in result.errors))
221
+ for error in result.errors:
222
+ warnings.append(str(error))
223
+ material_defect = bool(result.errors)
224
+ document = result.document
225
+ if source.suffix.lower() == ".pdf":
226
+ from native_code import recover_native_code
227
+ recovery = recover_native_code(document, source)
228
+ native_verified_ids = {item["id"] for item in recovery["recoveries"]}
229
+ warnings.extend(recovery["warnings"])
230
+ material_defect = material_defect or recovery["needs_review"]
231
+ (output / "native-code.json").write_text(json.dumps({"sourceHash": request["sourceHash"], **recovery}, ensure_ascii=False, indent=2), "utf-8")
232
+ if expected_pages is not None and len(result.pages) != expected_pages:
233
+ warnings.append("Incomplete PDF page coverage: expected " + str(expected_pages) + ", received " + str(len(result.pages)))
234
+ material_defect = True
235
+ document.save_as_markdown(output / "document.md", artifacts_dir=assets, image_mode=ImageRefMode.REFERENCED)
236
+ document.save_as_json(output / "structure.json", artifacts_dir=assets, image_mode=ImageRefMode.REFERENCED)
237
+ # Staging directories are renamed by the host. Never persist absolute staging
238
+ # URIs even if a future serializer changes its reference-path default.
239
+ markdown_path = output / "document.md"
240
+ markdown = markdown_path.read_text("utf-8").replace(output.as_uri() + "/", "").replace(str(output) + os.sep, "")
241
+ markdown_path.write_text(markdown, "utf-8")
242
+ def relative_images(value):
243
+ if isinstance(value, dict):
244
+ for key, child in value.items():
245
+ if key in ("uri", "url") and isinstance(child, str):
246
+ value[key] = child.replace(output.as_uri() + "/", "").replace(str(output) + os.sep, "")
247
+ else:
248
+ relative_images(child)
249
+ elif isinstance(value, list):
250
+ for child in value:
251
+ relative_images(child)
252
+ structure = json.loads((output / "structure.json").read_text("utf-8"))
253
+ relative_images(structure)
254
+ (output / "structure.json").write_text(json.dumps(structure, ensure_ascii=False, indent=2), "utf-8")
255
+ blocks = extract_blocks(document, source.suffix.lower())
256
+ for block in blocks:
257
+ if "code" in block.get("kind", "") and len(block["text"]) >= 6000 and block["id"] not in native_verified_ids:
258
+ warnings.append("Long recognized code block may reach the parser token limit; compare " + block["id"] + " with the original PDF.")
259
+ material_defect = True
260
+ if source.suffix.lower() == ".pdf" and "code" not in block.get("kind", "") and re.match(r"^\s*(?:(?:async\s+)?def\s+\w+\([^)]*\)\s*:|(?:export\s+)?(?:async\s+)?function\s+\w*\s*\()", block["text"]):
261
+ warnings.append("Possible code region was parsed as plain text: " + block["id"] + "; compare the original PDF for line breaks and indentation.")
262
+ material_defect = True
263
+ formula_metadata = []
264
+ if source.suffix.lower() == ".xlsx":
265
+ blocks, formula_metadata = excel_provenance(source, warnings)
266
+ for index, table in enumerate(document.tables):
267
+ # Preserve merged cells in the native JSON and an HTML companion; GFM cannot
268
+ # represent row/column spans. CSV is provided for machine validation.
269
+ table.export_to_dataframe(doc=document).to_csv(assets / ("table-" + str(index + 1) + ".csv"), index=False)
270
+ (assets / ("table-" + str(index + 1) + ".html")).write_text(table.export_to_html(doc=document), encoding="utf-8")
271
+ cells = getattr(table.data, "table_cells", [])
272
+ if any(getattr(cell, "row_span", 1) > 1 or getattr(cell, "col_span", 1) > 1 for cell in cells):
273
+ warnings.append("Table " + str(index + 1) + " contains merged cells; Markdown is flattened. Refer to structure.json and table HTML/CSV.")
274
+ material_defect = True
275
+ try:
276
+ picture = table.get_image(document)
277
+ if picture:
278
+ picture.save(assets / ("table-" + str(index + 1) + ".png"))
279
+ except Exception:
280
+ warnings.append("Table " + str(index + 1) + " has no source image crop.")
281
+ if source.suffix.lower() == ".pdf":
282
+ warnings.append("OCR and code recognition require review; code enrichment covers detected regions only and long code blocks can be truncated.")
283
+ if not blocks:
284
+ warnings.append("No readable text blocks were extracted.")
285
+ status = "partial" if state == "partial_success" or not blocks or enrichment_failed or material_defect else "complete"
286
+ if any(warning.startswith("Missing formula cache:") for warning in warnings):
287
+ status = "partial"
288
+ (output / "source-map.json").write_text(json.dumps({"schemaVersion": 1, "source": str(source), "sourceHash": request["sourceHash"], "blocks": blocks, "formulas": formula_metadata}, ensure_ascii=False, indent=2, default=str), "utf-8")
289
+ return {"status": status, "warnings": list(dict.fromkeys(warnings))}
290
+
291
+
292
+ def smoke(root):
293
+ from fixtures import make_fixtures
294
+ folder = Path(tempfile.mkdtemp(prefix="qualification-", dir=root))
295
+ try:
296
+ files = make_fixtures(folder)
297
+ engine = converter(root)
298
+ for source, expected in files:
299
+ print("Offline fixture: " + source.name, file=sys.stderr, flush=True)
300
+ output = folder / (source.stem + "-output")
301
+ response = convert(root, {"inputPath": str(source), "outputDir": str(output), "sourceHash": sha256(source)}, engine)
302
+ text = (output / "document.md").read_text("utf-8")
303
+ structure = json.loads((output / "structure.json").read_text("utf-8"))
304
+ source_map = json.loads((output / "source-map.json").read_text("utf-8"))
305
+ known_short_code_limit = source.name == "digital-short.pdf" and response["status"] == "partial" and any(warning.startswith("Possible code region was parsed as plain text:") for warning in response["warnings"])
306
+ ambiguous_code_is_partial = source.name == "digital-ambiguous.pdf" and response["status"] == "partial" and any(warning.startswith("Native code layout could not be verified") for warning in response["warnings"])
307
+ if source.name == "digital-ambiguous.pdf" and not ambiguous_code_is_partial:
308
+ raise RuntimeError("Ambiguous PDF code characters were not reported for review")
309
+ if (response["status"] != "complete" and not known_short_code_limit and not ambiguous_code_is_partial) or expected not in text:
310
+ raise RuntimeError("Offline conversion fixture failed: " + source.name)
311
+ if source.name in ("digital.pdf", "digital-short.pdf") and not known_short_code_limit:
312
+ for marker in ("公司技术资料", "Name", "Value", "calculate_total", "return"):
313
+ if marker not in text:
314
+ raise RuntimeError("PDF structure fixture omitted " + marker)
315
+ if not structure.get("tables"):
316
+ raise RuntimeError("PDF table structure was not detected")
317
+ if "```" not in text:
318
+ raise RuntimeError("PDF code block was not recognized; this profile has not passed code fidelity qualification")
319
+ if "def calculate_total(values):\n return sum(values)" not in text:
320
+ raise RuntimeError("PDF code text or indentation did not survive recognition")
321
+ if source.name == "scanned.pdf":
322
+ for marker in ("Name", "Value", "Total", "123"):
323
+ if marker not in text:
324
+ raise RuntimeError("OCR table fixture omitted " + marker)
325
+ if not structure.get("tables"):
326
+ raise RuntimeError("OCR table structure was not detected")
327
+ if not all(marker in re.sub(r"\s+", "", text) for marker in ("公司资料", "扫描表格")):
328
+ raise RuntimeError("Chinese OCR text was not recognized")
329
+ if known_short_code_limit:
330
+ native = json.loads((output / "pdf-text-layer.json").read_text("utf-8"))
331
+ if not all(marker in native["pages"][0]["text"] for marker in ("def calculate_total(values):", "return sum(values)")):
332
+ raise RuntimeError("Unrecognized short code lost its original text-layer evidence")
333
+ if source.suffix == ".pdf":
334
+ expected_cells = {(0, 0): "Name", (0, 1): "Value", (1, 0): "Total", (1, 1): expected}
335
+ grids = [{(cell["start_row_offset_idx"], cell["start_col_offset_idx"]): cell["text"].strip() for cell in table["data"]["table_cells"]} for table in structure.get("tables", [])]
336
+ if not any(all(grid.get(position) == value for position, value in expected_cells.items()) for grid in grids):
337
+ raise RuntimeError("PDF fixture table cell values or row/column positions are incorrect")
338
+ if source.suffix == ".pdf" and not all(block["locator"].get("page") == 1 for block in source_map["blocks"]):
339
+ raise RuntimeError("PDF source blocks lost page locators")
340
+ if source.suffix == ".xlsx" and not all(block["locator"].get("sheet") and block["locator"].get("cell") for block in source_map["blocks"]):
341
+ raise RuntimeError("Spreadsheet source blocks lost cell locators")
342
+ if source.suffix == ".pptx" and not all(block["locator"].get("slide") == 1 for block in source_map["blocks"]):
343
+ raise RuntimeError("Presentation source blocks lost slide locators")
344
+ except Exception as error:
345
+ raise RuntimeError(str(error) + "; qualification artifacts retained at " + str(folder)) from error
346
+ shutil.rmtree(folder)
347
+ return [source.name for source, expected in files]
348
+
349
+
350
+ def setup(root):
351
+ if importlib.metadata.version("docling") != VERSION:
352
+ raise RuntimeError("Unexpected Docling version")
353
+ (root / "ready.json").unlink(missing_ok=True)
354
+ from docling.utils.model_downloader import download_models
355
+ from huggingface_hub import snapshot_download
356
+ def fetch_model(repo):
357
+ print("Downloading CPU profile model: " + repo, file=sys.stderr, flush=True)
358
+ # The generic downloader includes both ONNX/Torch layout and fast/accurate
359
+ # tables. This profile uses the Torch layout and accurate table model only.
360
+ patterns = ["model_artifacts/tableformer/accurate/*", "README.md", "config.json"] if repo.endswith("/docling-models") else None
361
+ snapshot_download(repo_id=repo, revision=MODEL_REVISIONS[repo], local_dir=root / "models" / repo.replace("/", "--"), allow_patterns=patterns)
362
+ print("Downloaded CPU profile model: " + repo, file=sys.stderr, flush=True)
363
+ def fetch_ocr():
364
+ print("Downloading Chinese/English OCR models", file=sys.stderr, flush=True)
365
+ download_models(output_dir=root / "models", with_layout=False, with_tableformer=False, with_code_formula=False, with_picture_classifier=False, rapidocr_models=["onnxruntime:ch"], rapidocr_model_size="small", progress=True)
366
+ def report_failure(future):
367
+ if future.exception():
368
+ print("Model setup failed: " + str(future.exception()), file=sys.stderr, flush=True)
369
+ with ThreadPoolExecutor(max_workers=4) as pool:
370
+ futures = [pool.submit(fetch_model, repo) for repo in MODEL_REVISIONS] + [pool.submit(fetch_ocr)]
371
+ for future in futures:
372
+ future.add_done_callback(report_failure)
373
+ for future in futures:
374
+ future.result()
375
+ models = {}
376
+ for path in sorted((root / "models").rglob("*")):
377
+ if path.is_file() and ".cache" not in path.parts:
378
+ models[str(path.relative_to(root / "models"))] = {"sha256": sha256(path), "size": path.stat().st_size}
379
+ offline()
380
+ passed = smoke(root)
381
+ if parser_assets() != LOADED_ASSET_HASHES:
382
+ raise RuntimeError("Parser implementation changed during setup; retry qualification")
383
+ packages = {dist.metadata["Name"]: dist.version for dist in importlib.metadata.distributions()}
384
+ receipt = {"profile": PROFILE, "docling": VERSION, "python": platform.python_version(), "platform": platform.platform(), "parserAssets": LOADED_ASSET_HASHES, "modelRevisions": MODEL_REVISIONS, "models": models, "packages": packages, "smokePassed": passed}
385
+ (root / "ready.json").write_text(json.dumps(receipt, indent=2, sort_keys=True), "utf-8")
386
+ return doctor(root)
387
+
388
+
389
+ def main():
390
+ action, root = sys.argv[1], Path(sys.argv[2])
391
+ if action == "setup":
392
+ with redirect_stdout(sys.stderr):
393
+ response = setup(root)
394
+ emit(response)
395
+ else:
396
+ offline()
397
+ if action == "doctor":
398
+ emit(doctor(root))
399
+ elif action == "convert":
400
+ request = json.load(sys.stdin)
401
+ with redirect_stdout(sys.stderr):
402
+ response = convert(root, request)
403
+ emit(response)
404
+ elif action == "serve":
405
+ engine = None
406
+ for line in sys.stdin:
407
+ try:
408
+ with redirect_stdout(sys.stderr):
409
+ if engine is None:
410
+ engine = converter(root)
411
+ response = convert(root, json.loads(line), engine)
412
+ emit(response)
413
+ except Exception as error:
414
+ emit({"error": str(error)})
415
+ elif action == "smoke":
416
+ with redirect_stdout(sys.stderr):
417
+ passed = smoke(root)
418
+ emit({"passed": passed})
419
+ else:
420
+ raise ValueError("Unknown operation")
421
+
422
+
423
+ if __name__ == "__main__":
424
+ main()