@youweichen/pi-harness 1.0.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +1 -1
- package/README.md +14 -0
- package/dist/server/agent-service.js +87 -7
- package/dist/server/control-socket.js +1 -0
- package/dist/server/document-conversion/extension.js +164 -0
- package/dist/server/document-conversion/python/bridge.py +424 -0
- package/dist/server/document-conversion/python/fixtures.py +201 -0
- package/dist/server/document-conversion/python/native_code.py +149 -0
- package/dist/server/document-conversion/runtime.js +321 -0
- package/dist/server/document-conversion/service.js +340 -0
- package/dist/server/document-conversion/settings.js +63 -0
- package/dist/server/extensions-worker.js +3 -2
- package/dist/server/files-service.js +8 -16
- package/dist/server/index.js +9 -3
- package/dist/server/native-tools.js +4 -0
- package/dist/server/node-agent.js +262 -0
- package/dist/server/node-workbench.js +62 -1
- package/dist/server/okf/extension.js +118 -0
- package/dist/server/okf/markdown-sources.js +58 -0
- package/dist/server/okf/service.js +713 -0
- package/dist/server/okf/storage.js +165 -0
- package/dist/server/okf/types.js +1 -0
- package/dist/server/plan/extension.js +4 -4
- package/dist/server/plan/state.js +11 -1
- package/dist/server/slash-commands.js +6 -0
- package/dist/server/wiki-links.js +24 -0
- package/dist/server/wiki-routes.js +8 -1
- package/dist/server/wiki-service.js +50 -5
- package/package.json +2 -2
- package/web/dist/assets/NodeWorkbench-DM6YTS5e.js +3 -0
- package/web/dist/assets/{TerminalPanel-CWh6vw_N.js → TerminalPanel-aFJylyxg.js} +1 -1
- package/web/dist/assets/{abnfDiagram-VCTEODGH-D5Ti0Kpa.js → abnfDiagram-VCTEODGH-CcM7f7oX.js} +1 -1
- package/web/dist/assets/{arc-DKJilLd6.js → arc-Bv7vS5l5.js} +1 -1
- package/web/dist/assets/{architectureDiagram-5GKGNRK7-Cf0w38g-.js → architectureDiagram-5GKGNRK7-Cl8QcB3w.js} +1 -1
- package/web/dist/assets/{blockDiagram-I7D4REHJ-BOTLWLIL.js → blockDiagram-I7D4REHJ-DarS2-4G.js} +1 -1
- package/web/dist/assets/{c4Diagram-7LVT6UL2-CLCCoJs3.js → c4Diagram-7LVT6UL2-BY0_ttBt.js} +1 -1
- package/web/dist/assets/channel-DJGMgVTh.js +1 -0
- package/web/dist/assets/{chunk-2Q5K7J3B-BHjga2Ge.js → chunk-2Q5K7J3B-Cz-zS7_X.js} +1 -1
- package/web/dist/assets/{chunk-5VM5RSS4-CQJ5VfDZ.js → chunk-5VM5RSS4-B7AyqkmI.js} +1 -1
- package/web/dist/assets/{chunk-F27PBJKO-C67qoPJH.js → chunk-F27PBJKO-B7hVduzL.js} +1 -1
- package/web/dist/assets/{chunk-IMKFNOWR-CgXXGwI8.js → chunk-IMKFNOWR-DNGJqYn5.js} +1 -1
- package/web/dist/assets/{chunk-JWPE2WC7-D0RPyByP.js → chunk-JWPE2WC7-B5iJgXvl.js} +1 -1
- package/web/dist/assets/{chunk-POPQ4Y6H-BeYC9s9x.js → chunk-POPQ4Y6H-BvWCkriR.js} +1 -1
- package/web/dist/assets/{chunk-SVP7TREG-eQPl4UBQ.js → chunk-SVP7TREG-BcdlBuOQ.js} +1 -1
- package/web/dist/assets/{chunk-TICWLB2K-DT73arPU.js → chunk-TICWLB2K-B1igqjQd.js} +1 -1
- package/web/dist/assets/{chunk-XXDRQBXY-B4FwQcP-.js → chunk-XXDRQBXY-DKGDquaw.js} +1 -1
- package/web/dist/assets/classDiagram-ZZMXUADV-D9nxgF0M.js +1 -0
- package/web/dist/assets/classDiagram-v2-VYDZK3BY-D9nxgF0M.js +1 -0
- package/web/dist/assets/{cose-bilkent-JH36ORCC-Bnn5nlQN.js → cose-bilkent-JH36ORCC-BU_9nu88.js} +1 -1
- package/web/dist/assets/{cynefin-OW5HDTMX-CStWX9kE.js → cynefin-OW5HDTMX-DY0BDLjc.js} +1 -1
- package/web/dist/assets/{cynefinDiagram-5FMLGOSQ-BMVHC03f.js → cynefinDiagram-5FMLGOSQ-B-zM25Fk.js} +1 -1
- package/web/dist/assets/{dagre-GXQ25YYZ-ClSaO4ml.js → dagre-GXQ25YYZ-BxooDyBc.js} +1 -1
- package/web/dist/assets/{diagram-S7CK7UJ4-C43LL08C.js → diagram-S7CK7UJ4-CCO-8de9.js} +1 -1
- package/web/dist/assets/{diagram-UQ7AKVKN-BPVn_ySW.js → diagram-UQ7AKVKN-CmGsH525.js} +1 -1
- package/web/dist/assets/{diagram-VSXAHHWV-C34FFhRi.js → diagram-VSXAHHWV-CSQJYo3m.js} +1 -1
- package/web/dist/assets/{diagram-VX7I27RA-BRk4bncf.js → diagram-VX7I27RA-CwuJsF66.js} +1 -1
- package/web/dist/assets/{diagram-Z3DM3KII-O5hYPLEd.js → diagram-Z3DM3KII-BF13T_lj.js} +1 -1
- package/web/dist/assets/{ebnfDiagram-PWID7BFC-DVGEBm99.js → ebnfDiagram-PWID7BFC-ZtoDLro6.js} +1 -1
- package/web/dist/assets/{erDiagram-RLTQ6QDP-M-BLpTEV.js → erDiagram-RLTQ6QDP-BBo-qmNH.js} +1 -1
- package/web/dist/assets/{flowDiagram-HODETNUW-BwdLopEx.js → flowDiagram-HODETNUW-DAq9qdP8.js} +1 -1
- package/web/dist/assets/{ganttDiagram-EL5Y4UJY-DYxx2ppA.js → ganttDiagram-EL5Y4UJY-C53eRlMu.js} +1 -1
- package/web/dist/assets/{gitGraphDiagram-WWUBYQGX-DUU0wMYB.js → gitGraphDiagram-WWUBYQGX-BviBMJxj.js} +1 -1
- package/web/dist/assets/index-Bab44wX8.js +276 -0
- package/web/dist/assets/index-C3vDFSEV.css +10 -0
- package/web/dist/assets/{infoDiagram-27XIBGKW-CFaW5C0C.js → infoDiagram-27XIBGKW-DI95SO3j.js} +1 -1
- package/web/dist/assets/{ishikawaDiagram-5VMMS53U--ENP1MTB.js → ishikawaDiagram-5VMMS53U-Cl8IDXwf.js} +1 -1
- package/web/dist/assets/{journeyDiagram-3NMN7TZE-DgnAzSTj.js → journeyDiagram-3NMN7TZE-Co6aJeTZ.js} +1 -1
- package/web/dist/assets/{kanban-definition-UXKFOSKX-_STZN1vU.js → kanban-definition-UXKFOSKX-CknQQpwW.js} +1 -1
- package/web/dist/assets/{layout-CkUClkXt.js → layout-CFtUmtTW.js} +1 -1
- package/web/dist/assets/{linear-sYHiUOxa.js → linear-mRlGUAKy.js} +1 -1
- package/web/dist/assets/{markdown-vNTCZJ6d.js → markdown-BhJ_LInD.js} +1 -1
- package/web/dist/assets/{mermaid.core-B8fKGWsc.js → mermaid.core-CvQEO3ue.js} +5 -5
- package/web/dist/assets/{mindmap-definition-YA3MSWOX-DP0BYOEx.js → mindmap-definition-YA3MSWOX-DpqhsxUl.js} +1 -1
- package/web/dist/assets/{pegDiagram-XKGWAZYB-DkWTmZre.js → pegDiagram-XKGWAZYB-C9eawB5u.js} +1 -1
- package/web/dist/assets/{pieDiagram-E7YTZNPT-DfZgMV4u.js → pieDiagram-E7YTZNPT-8vaqXVD6.js} +1 -1
- package/web/dist/assets/{quadrantDiagram-AXDQQJYC-Djl4TT7q.js → quadrantDiagram-AXDQQJYC-Diwr8ynm.js} +1 -1
- package/web/dist/assets/{railroadDiagram-O6MQD6OU-DbKF6zXn.js → railroadDiagram-O6MQD6OU-krW50Cgi.js} +1 -1
- package/web/dist/assets/{react-BNj3n7jX.js → react-BddUxV-9.js} +1 -1
- package/web/dist/assets/{requirementDiagram-BXWQKSXE-idola5qb.js → requirementDiagram-BXWQKSXE-BPLI6eMG.js} +1 -1
- package/web/dist/assets/{sankeyDiagram-P5KCCOFB-KA0hQOle.js → sankeyDiagram-P5KCCOFB-Bix_elmH.js} +1 -1
- package/web/dist/assets/{sequenceDiagram-WJ2MYXX4-BXSKvxzP.js → sequenceDiagram-WJ2MYXX4-0m0xaUKi.js} +1 -1
- package/web/dist/assets/{sizeCapture-INFHLROL-C_Ewv3S5.js → sizeCapture-INFHLROL-BtzJ_oOR.js} +1 -1
- package/web/dist/assets/{stateDiagram-D77RDMKH-DIf5X8oo.js → stateDiagram-D77RDMKH-Clq5bN0v.js} +1 -1
- package/web/dist/assets/stateDiagram-v2-MP3YSRHH-Bn1oW4gp.js +1 -0
- package/web/dist/assets/{swimlanes-42K2YHIH-Cx_Rc8M2.js → swimlanes-42K2YHIH-11MDGnYA.js} +1 -1
- package/web/dist/assets/swimlanesDiagram-VR7AAH4N-BW96Edgo.js +8 -0
- package/web/dist/assets/{timeline-definition-24CTP7MA-BRrDhWVz.js → timeline-definition-24CTP7MA-BeQfWP95.js} +1 -1
- package/web/dist/assets/{vennDiagram-4TSXK5OY-Ad6G0YwD.js → vennDiagram-4TSXK5OY-BMYkX0KL.js} +1 -1
- package/web/dist/assets/{wardleyDiagram-VM6X3IG4-X6-ENpSi.js → wardleyDiagram-VM6X3IG4-BV3fTUkU.js} +1 -1
- package/web/dist/assets/{xychartDiagram-S5SC5T6Z-DhegNS6d.js → xychartDiagram-S5SC5T6Z-Ct7ezLnc.js} +1 -1
- package/web/dist/index.html +4 -4
- package/web/dist/assets/NodeWorkbench-C4U1PldE.js +0 -3
- package/web/dist/assets/channel-D8Hd7T4P.js +0 -1
- package/web/dist/assets/classDiagram-ZZMXUADV-B_btgYRB.js +0 -1
- package/web/dist/assets/classDiagram-v2-VYDZK3BY-B_btgYRB.js +0 -1
- package/web/dist/assets/index-CI2KfVSr.css +0 -10
- package/web/dist/assets/index-DQcpggFU.js +0 -275
- package/web/dist/assets/stateDiagram-v2-MP3YSRHH-DYwe_2mB.js +0 -1
- package/web/dist/assets/swimlanesDiagram-VR7AAH4N-52Y8tgRV.js +0 -8
|
@@ -0,0 +1,424 @@
|
|
|
1
|
+
"""Versioned, local-only Docling worker. stdout is one JSON response; logs use stderr."""
|
|
2
|
+
import hashlib
|
|
3
|
+
import importlib.metadata
|
|
4
|
+
import json
|
|
5
|
+
import logging
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
import platform
|
|
9
|
+
import re
|
|
10
|
+
import shutil
|
|
11
|
+
import sys
|
|
12
|
+
import tempfile
|
|
13
|
+
from contextlib import redirect_stdout
|
|
14
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
15
|
+
|
|
16
|
+
VERSION = "2.136.0"
|
|
17
|
+
PROFILE = "docling-cpu-zh-v1"
|
|
18
|
+
MODEL_REVISIONS = {
|
|
19
|
+
"docling-project/docling-layout-heron": "8f39ad3c0b4c58e9c2d2c84a38465abf757272d8",
|
|
20
|
+
"docling-project/docling-models": "fc0f2d45e2218ea24bce5045f58a389aed16dc23",
|
|
21
|
+
"docling-project/CodeFormulaV2": "ecedbe111d15c2dc60bfd4a823cbe80127b58af4",
|
|
22
|
+
}
|
|
23
|
+
logging.basicConfig(stream=sys.stderr, level=logging.WARNING)
|
|
24
|
+
# Hugging Face's HTTP downloader respects the host's proxy configuration and can
|
|
25
|
+
# resume downloads without requiring Xet's separate network transport.
|
|
26
|
+
os.environ.setdefault("HF_HUB_DISABLE_XET", "1")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def emit(value):
|
|
30
|
+
print(json.dumps(value, ensure_ascii=False), flush=True)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def sha256(path):
|
|
34
|
+
hash_value = hashlib.sha256()
|
|
35
|
+
with path.open("rb") as stream:
|
|
36
|
+
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
|
37
|
+
hash_value.update(chunk)
|
|
38
|
+
return hash_value.hexdigest()
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def parser_assets():
|
|
42
|
+
return {path.name: sha256(path) for path in sorted(Path(__file__).parent.glob("*.py"))}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
LOADED_ASSET_HASHES = parser_assets()
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def offline():
|
|
49
|
+
for name in ("HF_HUB_OFFLINE", "TRANSFORMERS_OFFLINE", "HF_DATASETS_OFFLINE", "HF_HUB_DISABLE_TELEMETRY"):
|
|
50
|
+
os.environ[name] = "1"
|
|
51
|
+
# Protect all optional backends as well as Hugging Face. Conversion never opens
|
|
52
|
+
# TCP connections, including those to loopback or inherited proxy endpoints.
|
|
53
|
+
def audit(event, args):
|
|
54
|
+
if event in ("socket.connect", "socket.getaddrinfo", "socket.sendto"):
|
|
55
|
+
raise PermissionError("Network access is disabled during document conversion")
|
|
56
|
+
sys.addaudithook(audit)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def doctor(root):
|
|
60
|
+
missing = []
|
|
61
|
+
warnings = []
|
|
62
|
+
try:
|
|
63
|
+
actual = importlib.metadata.version("docling")
|
|
64
|
+
if actual != VERSION:
|
|
65
|
+
missing.append("Expected Docling " + VERSION + "; found " + actual)
|
|
66
|
+
except importlib.metadata.PackageNotFoundError:
|
|
67
|
+
actual = None
|
|
68
|
+
missing.append("Docling is not installed")
|
|
69
|
+
if sys.version_info[:2] != (3, 12):
|
|
70
|
+
missing.append("This runtime profile requires Python 3.12")
|
|
71
|
+
try:
|
|
72
|
+
receipt = json.loads((root / "ready.json").read_text("utf-8"))
|
|
73
|
+
if receipt.get("profile") != PROFILE or receipt.get("docling") != VERSION or not receipt.get("smokePassed"):
|
|
74
|
+
missing.append("Runtime qualification receipt is invalid")
|
|
75
|
+
if receipt.get("parserAssets") != parser_assets():
|
|
76
|
+
missing.append("Parser implementation changed; run /pdf-md setup to requalify the offline profile")
|
|
77
|
+
for relative, metadata in receipt.get("models", {}).items():
|
|
78
|
+
path = root / "models" / relative
|
|
79
|
+
if not path.is_file() or path.stat().st_size != metadata["size"] or sha256(path) != metadata["sha256"]:
|
|
80
|
+
missing.append("Missing or changed model: " + relative)
|
|
81
|
+
for name, version in receipt.get("packages", {}).items():
|
|
82
|
+
try:
|
|
83
|
+
if importlib.metadata.version(name) != version:
|
|
84
|
+
missing.append("Runtime dependency changed: " + name)
|
|
85
|
+
except importlib.metadata.PackageNotFoundError:
|
|
86
|
+
missing.append("Runtime dependency missing: " + name)
|
|
87
|
+
if not receipt.get("models"):
|
|
88
|
+
missing.append("No model artifacts were recorded")
|
|
89
|
+
except (OSError, ValueError, KeyError):
|
|
90
|
+
missing.append("Run /pdf-md setup to download models and qualify offline conversion")
|
|
91
|
+
return {"ready": not missing, "parserVersion": actual, "missing": missing, "warnings": warnings}
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def converter(root):
|
|
95
|
+
from docling.datamodel.accelerator_options import AcceleratorDevice, AcceleratorOptions
|
|
96
|
+
from docling.datamodel.base_models import InputFormat
|
|
97
|
+
from docling.datamodel.pipeline_options import (
|
|
98
|
+
CodeFormulaVlmOptions, OcrMode, PdfPipelineOptions, RapidOcrOptions,
|
|
99
|
+
TableFormerMode, TableStructureOptions,
|
|
100
|
+
)
|
|
101
|
+
from docling.datamodel.vlm_engine_options import TransformersVlmEngineOptions
|
|
102
|
+
from docling.document_converter import DocumentConverter, PdfFormatOption
|
|
103
|
+
options = PdfPipelineOptions(
|
|
104
|
+
artifacts_path=root / "models",
|
|
105
|
+
enable_remote_services=False,
|
|
106
|
+
allow_external_plugins=False,
|
|
107
|
+
accelerator_options=AcceleratorOptions(device=AcceleratorDevice.CPU, num_threads=4),
|
|
108
|
+
do_ocr=True,
|
|
109
|
+
ocr_options=RapidOcrOptions(backend="onnxruntime", lang=["ch"], model_size="small", mode=OcrMode.PDF_AWARE_LAYOUT_REGIONS),
|
|
110
|
+
do_table_structure=True,
|
|
111
|
+
table_structure_options=TableStructureOptions(mode=TableFormerMode.ACCURATE, do_cell_matching=True),
|
|
112
|
+
do_code_enrichment=True,
|
|
113
|
+
do_formula_enrichment=True,
|
|
114
|
+
code_formula_options=CodeFormulaVlmOptions.from_preset("codeformulav2", engine_options=TransformersVlmEngineOptions(
|
|
115
|
+
device=AcceleratorDevice.CPU, torch_dtype="float32", quantized=False, compile_model=False,
|
|
116
|
+
)),
|
|
117
|
+
generate_picture_images=True,
|
|
118
|
+
generate_page_images=True,
|
|
119
|
+
)
|
|
120
|
+
options.heading_hierarchy_options.enabled = True
|
|
121
|
+
return DocumentConverter(allowed_formats=[InputFormat.PDF, InputFormat.DOCX, InputFormat.PPTX, InputFormat.XLSX], format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=options)})
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def extract_blocks(document, suffix):
|
|
125
|
+
from docling_core.types.doc import TableItem
|
|
126
|
+
blocks = []
|
|
127
|
+
head = None
|
|
128
|
+
for item, level in document.iterate_items():
|
|
129
|
+
label = str(getattr(item, "label", ""))
|
|
130
|
+
text = getattr(item, "text", "")
|
|
131
|
+
if isinstance(item, TableItem):
|
|
132
|
+
text = item.export_to_markdown(doc=document)
|
|
133
|
+
if not text:
|
|
134
|
+
continue
|
|
135
|
+
if "section_header" in label or "title" in label:
|
|
136
|
+
head = text
|
|
137
|
+
locator = {"heading": head} if head else {}
|
|
138
|
+
provenance = [p.model_dump(mode="json") for p in getattr(item, "prov", [])]
|
|
139
|
+
if provenance:
|
|
140
|
+
page = provenance[0].get("page_no")
|
|
141
|
+
if page:
|
|
142
|
+
locator["slide" if suffix == ".pptx" else "page"] = page
|
|
143
|
+
blocks.append({"id": getattr(item, "self_ref", "block-" + str(len(blocks) + 1)), "text": text, "locator": locator, "provenance": provenance, "kind": label})
|
|
144
|
+
return blocks
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def excel_provenance(source, warnings):
|
|
148
|
+
from openpyxl import load_workbook
|
|
149
|
+
formulas = load_workbook(source, read_only=True, data_only=False)
|
|
150
|
+
values = load_workbook(source, read_only=True, data_only=True)
|
|
151
|
+
blocks = []
|
|
152
|
+
metadata = []
|
|
153
|
+
try:
|
|
154
|
+
for sheet in formulas:
|
|
155
|
+
cached = values[sheet.title]
|
|
156
|
+
for row, cached_row in zip(sheet.iter_rows(), cached.iter_rows()):
|
|
157
|
+
for cell, cached_cell in zip(row, cached_row):
|
|
158
|
+
if cell.value is None:
|
|
159
|
+
continue
|
|
160
|
+
value = cached_cell.value
|
|
161
|
+
is_formula = cell.data_type == "f"
|
|
162
|
+
if is_formula:
|
|
163
|
+
metadata.append({"sheet": sheet.title, "cell": cell.coordinate, "formula": str(cell.value), "cachedValue": value})
|
|
164
|
+
if value is None:
|
|
165
|
+
warnings.append("Missing formula cache: " + sheet.title + "!" + cell.coordinate)
|
|
166
|
+
blocks.append({"id": "cell-" + str(len(blocks) + 1), "text": str(value) if value is not None else "[formula result unavailable]", "locator": {"sheet": sheet.title, "cell": cell.coordinate}})
|
|
167
|
+
if metadata:
|
|
168
|
+
warnings.append("Spreadsheet formula results are saved cached values; no formulas were evaluated and cache freshness cannot be verified.")
|
|
169
|
+
finally:
|
|
170
|
+
formulas.close()
|
|
171
|
+
values.close()
|
|
172
|
+
return blocks, metadata
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def convert(root, request, engine=None):
|
|
176
|
+
from docling_core.types.doc import ImageRefMode, TableItem
|
|
177
|
+
source, output = Path(request["inputPath"]), Path(request["outputDir"])
|
|
178
|
+
output.mkdir(parents=True, exist_ok=True)
|
|
179
|
+
assets = output / "assets"
|
|
180
|
+
assets.mkdir(exist_ok=True)
|
|
181
|
+
warnings = []
|
|
182
|
+
enrichment_failed = False
|
|
183
|
+
material_defect = False
|
|
184
|
+
native_verified_ids = set()
|
|
185
|
+
expected_pages = None
|
|
186
|
+
if source.suffix.lower() == ".pdf":
|
|
187
|
+
import pypdfium2
|
|
188
|
+
pdf = pypdfium2.PdfDocument(source)
|
|
189
|
+
try:
|
|
190
|
+
expected_pages = len(pdf)
|
|
191
|
+
native = []
|
|
192
|
+
for index in range(expected_pages):
|
|
193
|
+
page = pdf[index]
|
|
194
|
+
try:
|
|
195
|
+
textpage = page.get_textpage()
|
|
196
|
+
try:
|
|
197
|
+
native.append({"page": index + 1, "text": textpage.get_text_range()})
|
|
198
|
+
finally:
|
|
199
|
+
textpage.close()
|
|
200
|
+
finally:
|
|
201
|
+
page.close()
|
|
202
|
+
(output / "pdf-text-layer.json").write_text(json.dumps({"sourceHash": request["sourceHash"], "pages": native}, ensure_ascii=False), "utf-8")
|
|
203
|
+
finally:
|
|
204
|
+
pdf.close()
|
|
205
|
+
class Capture(logging.Handler):
|
|
206
|
+
def emit(self, record):
|
|
207
|
+
nonlocal enrichment_failed
|
|
208
|
+
if record.levelno >= logging.ERROR:
|
|
209
|
+
enrichment_failed = True
|
|
210
|
+
if record.levelno >= logging.WARNING and len(warnings) < 100:
|
|
211
|
+
warnings.append(record.getMessage()[:1000])
|
|
212
|
+
capture = Capture()
|
|
213
|
+
logging.getLogger().addHandler(capture)
|
|
214
|
+
try:
|
|
215
|
+
result = (engine or converter(root)).convert(source, raises_on_error=False)
|
|
216
|
+
finally:
|
|
217
|
+
logging.getLogger().removeHandler(capture)
|
|
218
|
+
state = str(getattr(result.status, "value", result.status))
|
|
219
|
+
if state not in ("success", "partial_success"):
|
|
220
|
+
raise RuntimeError("Docling conversion failed: " + state + "; " + "; ".join(str(error) for error in result.errors))
|
|
221
|
+
for error in result.errors:
|
|
222
|
+
warnings.append(str(error))
|
|
223
|
+
material_defect = bool(result.errors)
|
|
224
|
+
document = result.document
|
|
225
|
+
if source.suffix.lower() == ".pdf":
|
|
226
|
+
from native_code import recover_native_code
|
|
227
|
+
recovery = recover_native_code(document, source)
|
|
228
|
+
native_verified_ids = {item["id"] for item in recovery["recoveries"]}
|
|
229
|
+
warnings.extend(recovery["warnings"])
|
|
230
|
+
material_defect = material_defect or recovery["needs_review"]
|
|
231
|
+
(output / "native-code.json").write_text(json.dumps({"sourceHash": request["sourceHash"], **recovery}, ensure_ascii=False, indent=2), "utf-8")
|
|
232
|
+
if expected_pages is not None and len(result.pages) != expected_pages:
|
|
233
|
+
warnings.append("Incomplete PDF page coverage: expected " + str(expected_pages) + ", received " + str(len(result.pages)))
|
|
234
|
+
material_defect = True
|
|
235
|
+
document.save_as_markdown(output / "document.md", artifacts_dir=assets, image_mode=ImageRefMode.REFERENCED)
|
|
236
|
+
document.save_as_json(output / "structure.json", artifacts_dir=assets, image_mode=ImageRefMode.REFERENCED)
|
|
237
|
+
# Staging directories are renamed by the host. Never persist absolute staging
|
|
238
|
+
# URIs even if a future serializer changes its reference-path default.
|
|
239
|
+
markdown_path = output / "document.md"
|
|
240
|
+
markdown = markdown_path.read_text("utf-8").replace(output.as_uri() + "/", "").replace(str(output) + os.sep, "")
|
|
241
|
+
markdown_path.write_text(markdown, "utf-8")
|
|
242
|
+
def relative_images(value):
|
|
243
|
+
if isinstance(value, dict):
|
|
244
|
+
for key, child in value.items():
|
|
245
|
+
if key in ("uri", "url") and isinstance(child, str):
|
|
246
|
+
value[key] = child.replace(output.as_uri() + "/", "").replace(str(output) + os.sep, "")
|
|
247
|
+
else:
|
|
248
|
+
relative_images(child)
|
|
249
|
+
elif isinstance(value, list):
|
|
250
|
+
for child in value:
|
|
251
|
+
relative_images(child)
|
|
252
|
+
structure = json.loads((output / "structure.json").read_text("utf-8"))
|
|
253
|
+
relative_images(structure)
|
|
254
|
+
(output / "structure.json").write_text(json.dumps(structure, ensure_ascii=False, indent=2), "utf-8")
|
|
255
|
+
blocks = extract_blocks(document, source.suffix.lower())
|
|
256
|
+
for block in blocks:
|
|
257
|
+
if "code" in block.get("kind", "") and len(block["text"]) >= 6000 and block["id"] not in native_verified_ids:
|
|
258
|
+
warnings.append("Long recognized code block may reach the parser token limit; compare " + block["id"] + " with the original PDF.")
|
|
259
|
+
material_defect = True
|
|
260
|
+
if source.suffix.lower() == ".pdf" and "code" not in block.get("kind", "") and re.match(r"^\s*(?:(?:async\s+)?def\s+\w+\([^)]*\)\s*:|(?:export\s+)?(?:async\s+)?function\s+\w*\s*\()", block["text"]):
|
|
261
|
+
warnings.append("Possible code region was parsed as plain text: " + block["id"] + "; compare the original PDF for line breaks and indentation.")
|
|
262
|
+
material_defect = True
|
|
263
|
+
formula_metadata = []
|
|
264
|
+
if source.suffix.lower() == ".xlsx":
|
|
265
|
+
blocks, formula_metadata = excel_provenance(source, warnings)
|
|
266
|
+
for index, table in enumerate(document.tables):
|
|
267
|
+
# Preserve merged cells in the native JSON and an HTML companion; GFM cannot
|
|
268
|
+
# represent row/column spans. CSV is provided for machine validation.
|
|
269
|
+
table.export_to_dataframe(doc=document).to_csv(assets / ("table-" + str(index + 1) + ".csv"), index=False)
|
|
270
|
+
(assets / ("table-" + str(index + 1) + ".html")).write_text(table.export_to_html(doc=document), encoding="utf-8")
|
|
271
|
+
cells = getattr(table.data, "table_cells", [])
|
|
272
|
+
if any(getattr(cell, "row_span", 1) > 1 or getattr(cell, "col_span", 1) > 1 for cell in cells):
|
|
273
|
+
warnings.append("Table " + str(index + 1) + " contains merged cells; Markdown is flattened. Refer to structure.json and table HTML/CSV.")
|
|
274
|
+
material_defect = True
|
|
275
|
+
try:
|
|
276
|
+
picture = table.get_image(document)
|
|
277
|
+
if picture:
|
|
278
|
+
picture.save(assets / ("table-" + str(index + 1) + ".png"))
|
|
279
|
+
except Exception:
|
|
280
|
+
warnings.append("Table " + str(index + 1) + " has no source image crop.")
|
|
281
|
+
if source.suffix.lower() == ".pdf":
|
|
282
|
+
warnings.append("OCR and code recognition require review; code enrichment covers detected regions only and long code blocks can be truncated.")
|
|
283
|
+
if not blocks:
|
|
284
|
+
warnings.append("No readable text blocks were extracted.")
|
|
285
|
+
status = "partial" if state == "partial_success" or not blocks or enrichment_failed or material_defect else "complete"
|
|
286
|
+
if any(warning.startswith("Missing formula cache:") for warning in warnings):
|
|
287
|
+
status = "partial"
|
|
288
|
+
(output / "source-map.json").write_text(json.dumps({"schemaVersion": 1, "source": str(source), "sourceHash": request["sourceHash"], "blocks": blocks, "formulas": formula_metadata}, ensure_ascii=False, indent=2, default=str), "utf-8")
|
|
289
|
+
return {"status": status, "warnings": list(dict.fromkeys(warnings))}
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def smoke(root):
|
|
293
|
+
from fixtures import make_fixtures
|
|
294
|
+
folder = Path(tempfile.mkdtemp(prefix="qualification-", dir=root))
|
|
295
|
+
try:
|
|
296
|
+
files = make_fixtures(folder)
|
|
297
|
+
engine = converter(root)
|
|
298
|
+
for source, expected in files:
|
|
299
|
+
print("Offline fixture: " + source.name, file=sys.stderr, flush=True)
|
|
300
|
+
output = folder / (source.stem + "-output")
|
|
301
|
+
response = convert(root, {"inputPath": str(source), "outputDir": str(output), "sourceHash": sha256(source)}, engine)
|
|
302
|
+
text = (output / "document.md").read_text("utf-8")
|
|
303
|
+
structure = json.loads((output / "structure.json").read_text("utf-8"))
|
|
304
|
+
source_map = json.loads((output / "source-map.json").read_text("utf-8"))
|
|
305
|
+
known_short_code_limit = source.name == "digital-short.pdf" and response["status"] == "partial" and any(warning.startswith("Possible code region was parsed as plain text:") for warning in response["warnings"])
|
|
306
|
+
ambiguous_code_is_partial = source.name == "digital-ambiguous.pdf" and response["status"] == "partial" and any(warning.startswith("Native code layout could not be verified") for warning in response["warnings"])
|
|
307
|
+
if source.name == "digital-ambiguous.pdf" and not ambiguous_code_is_partial:
|
|
308
|
+
raise RuntimeError("Ambiguous PDF code characters were not reported for review")
|
|
309
|
+
if (response["status"] != "complete" and not known_short_code_limit and not ambiguous_code_is_partial) or expected not in text:
|
|
310
|
+
raise RuntimeError("Offline conversion fixture failed: " + source.name)
|
|
311
|
+
if source.name in ("digital.pdf", "digital-short.pdf") and not known_short_code_limit:
|
|
312
|
+
for marker in ("公司技术资料", "Name", "Value", "calculate_total", "return"):
|
|
313
|
+
if marker not in text:
|
|
314
|
+
raise RuntimeError("PDF structure fixture omitted " + marker)
|
|
315
|
+
if not structure.get("tables"):
|
|
316
|
+
raise RuntimeError("PDF table structure was not detected")
|
|
317
|
+
if "```" not in text:
|
|
318
|
+
raise RuntimeError("PDF code block was not recognized; this profile has not passed code fidelity qualification")
|
|
319
|
+
if "def calculate_total(values):\n return sum(values)" not in text:
|
|
320
|
+
raise RuntimeError("PDF code text or indentation did not survive recognition")
|
|
321
|
+
if source.name == "scanned.pdf":
|
|
322
|
+
for marker in ("Name", "Value", "Total", "123"):
|
|
323
|
+
if marker not in text:
|
|
324
|
+
raise RuntimeError("OCR table fixture omitted " + marker)
|
|
325
|
+
if not structure.get("tables"):
|
|
326
|
+
raise RuntimeError("OCR table structure was not detected")
|
|
327
|
+
if not all(marker in re.sub(r"\s+", "", text) for marker in ("公司资料", "扫描表格")):
|
|
328
|
+
raise RuntimeError("Chinese OCR text was not recognized")
|
|
329
|
+
if known_short_code_limit:
|
|
330
|
+
native = json.loads((output / "pdf-text-layer.json").read_text("utf-8"))
|
|
331
|
+
if not all(marker in native["pages"][0]["text"] for marker in ("def calculate_total(values):", "return sum(values)")):
|
|
332
|
+
raise RuntimeError("Unrecognized short code lost its original text-layer evidence")
|
|
333
|
+
if source.suffix == ".pdf":
|
|
334
|
+
expected_cells = {(0, 0): "Name", (0, 1): "Value", (1, 0): "Total", (1, 1): expected}
|
|
335
|
+
grids = [{(cell["start_row_offset_idx"], cell["start_col_offset_idx"]): cell["text"].strip() for cell in table["data"]["table_cells"]} for table in structure.get("tables", [])]
|
|
336
|
+
if not any(all(grid.get(position) == value for position, value in expected_cells.items()) for grid in grids):
|
|
337
|
+
raise RuntimeError("PDF fixture table cell values or row/column positions are incorrect")
|
|
338
|
+
if source.suffix == ".pdf" and not all(block["locator"].get("page") == 1 for block in source_map["blocks"]):
|
|
339
|
+
raise RuntimeError("PDF source blocks lost page locators")
|
|
340
|
+
if source.suffix == ".xlsx" and not all(block["locator"].get("sheet") and block["locator"].get("cell") for block in source_map["blocks"]):
|
|
341
|
+
raise RuntimeError("Spreadsheet source blocks lost cell locators")
|
|
342
|
+
if source.suffix == ".pptx" and not all(block["locator"].get("slide") == 1 for block in source_map["blocks"]):
|
|
343
|
+
raise RuntimeError("Presentation source blocks lost slide locators")
|
|
344
|
+
except Exception as error:
|
|
345
|
+
raise RuntimeError(str(error) + "; qualification artifacts retained at " + str(folder)) from error
|
|
346
|
+
shutil.rmtree(folder)
|
|
347
|
+
return [source.name for source, expected in files]
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def setup(root):
|
|
351
|
+
if importlib.metadata.version("docling") != VERSION:
|
|
352
|
+
raise RuntimeError("Unexpected Docling version")
|
|
353
|
+
(root / "ready.json").unlink(missing_ok=True)
|
|
354
|
+
from docling.utils.model_downloader import download_models
|
|
355
|
+
from huggingface_hub import snapshot_download
|
|
356
|
+
def fetch_model(repo):
|
|
357
|
+
print("Downloading CPU profile model: " + repo, file=sys.stderr, flush=True)
|
|
358
|
+
# The generic downloader includes both ONNX/Torch layout and fast/accurate
|
|
359
|
+
# tables. This profile uses the Torch layout and accurate table model only.
|
|
360
|
+
patterns = ["model_artifacts/tableformer/accurate/*", "README.md", "config.json"] if repo.endswith("/docling-models") else None
|
|
361
|
+
snapshot_download(repo_id=repo, revision=MODEL_REVISIONS[repo], local_dir=root / "models" / repo.replace("/", "--"), allow_patterns=patterns)
|
|
362
|
+
print("Downloaded CPU profile model: " + repo, file=sys.stderr, flush=True)
|
|
363
|
+
def fetch_ocr():
|
|
364
|
+
print("Downloading Chinese/English OCR models", file=sys.stderr, flush=True)
|
|
365
|
+
download_models(output_dir=root / "models", with_layout=False, with_tableformer=False, with_code_formula=False, with_picture_classifier=False, rapidocr_models=["onnxruntime:ch"], rapidocr_model_size="small", progress=True)
|
|
366
|
+
def report_failure(future):
|
|
367
|
+
if future.exception():
|
|
368
|
+
print("Model setup failed: " + str(future.exception()), file=sys.stderr, flush=True)
|
|
369
|
+
with ThreadPoolExecutor(max_workers=4) as pool:
|
|
370
|
+
futures = [pool.submit(fetch_model, repo) for repo in MODEL_REVISIONS] + [pool.submit(fetch_ocr)]
|
|
371
|
+
for future in futures:
|
|
372
|
+
future.add_done_callback(report_failure)
|
|
373
|
+
for future in futures:
|
|
374
|
+
future.result()
|
|
375
|
+
models = {}
|
|
376
|
+
for path in sorted((root / "models").rglob("*")):
|
|
377
|
+
if path.is_file() and ".cache" not in path.parts:
|
|
378
|
+
models[str(path.relative_to(root / "models"))] = {"sha256": sha256(path), "size": path.stat().st_size}
|
|
379
|
+
offline()
|
|
380
|
+
passed = smoke(root)
|
|
381
|
+
if parser_assets() != LOADED_ASSET_HASHES:
|
|
382
|
+
raise RuntimeError("Parser implementation changed during setup; retry qualification")
|
|
383
|
+
packages = {dist.metadata["Name"]: dist.version for dist in importlib.metadata.distributions()}
|
|
384
|
+
receipt = {"profile": PROFILE, "docling": VERSION, "python": platform.python_version(), "platform": platform.platform(), "parserAssets": LOADED_ASSET_HASHES, "modelRevisions": MODEL_REVISIONS, "models": models, "packages": packages, "smokePassed": passed}
|
|
385
|
+
(root / "ready.json").write_text(json.dumps(receipt, indent=2, sort_keys=True), "utf-8")
|
|
386
|
+
return doctor(root)
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def main():
|
|
390
|
+
action, root = sys.argv[1], Path(sys.argv[2])
|
|
391
|
+
if action == "setup":
|
|
392
|
+
with redirect_stdout(sys.stderr):
|
|
393
|
+
response = setup(root)
|
|
394
|
+
emit(response)
|
|
395
|
+
else:
|
|
396
|
+
offline()
|
|
397
|
+
if action == "doctor":
|
|
398
|
+
emit(doctor(root))
|
|
399
|
+
elif action == "convert":
|
|
400
|
+
request = json.load(sys.stdin)
|
|
401
|
+
with redirect_stdout(sys.stderr):
|
|
402
|
+
response = convert(root, request)
|
|
403
|
+
emit(response)
|
|
404
|
+
elif action == "serve":
|
|
405
|
+
engine = None
|
|
406
|
+
for line in sys.stdin:
|
|
407
|
+
try:
|
|
408
|
+
with redirect_stdout(sys.stderr):
|
|
409
|
+
if engine is None:
|
|
410
|
+
engine = converter(root)
|
|
411
|
+
response = convert(root, json.loads(line), engine)
|
|
412
|
+
emit(response)
|
|
413
|
+
except Exception as error:
|
|
414
|
+
emit({"error": str(error)})
|
|
415
|
+
elif action == "smoke":
|
|
416
|
+
with redirect_stdout(sys.stderr):
|
|
417
|
+
passed = smoke(root)
|
|
418
|
+
emit({"passed": passed})
|
|
419
|
+
else:
|
|
420
|
+
raise ValueError("Unknown operation")
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
if __name__ == "__main__":
|
|
424
|
+
main()
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Small synthetic fixtures; these verify engine wiring, not real-document fidelity."""
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from base64 import b64decode
|
|
4
|
+
from io import BytesIO
|
|
5
|
+
|
|
6
|
+
# Self-authored raster caption keeps the Chinese OCR check independent of OS fonts.
|
|
7
|
+
CHINESE_CAPTION_PNG = (
|
|
8
|
+
"iVBORw0KGgoAAAANSUhEUgAAAa4AAAA8CAIAAACxTskxAAAegUlEQVR42u1dZ1gUV9uepYgsiIii2AVsoIIVFSwgIMUCEV+NEWKi"
|
|
9
|
+
"gmBDMKBIFLGESwFRQBGNSoIGX5RLUQELFlBiFBSVKgLSO7vALkvb8v04n+eazO7Ozi6L0Tfn/sE1zJw5c+bszD3PeSpNIBBgCAgI"
|
|
10
|
+
"CP9uKKApQEBAQFBCU4CA8OVj69atysrKbm5uU6ZMQbOBpMJ/Ej/99JOjo2NsbGx+fj6Px5Njz/v27Zs5c+by5cudnJx+/fVXqc69"
|
|
11
|
+
"cuXKuHHjFixYYGtru3r16qqqKnmNau/evQkJCQ0NDT09PZ9hegUCgb6+vpmZ2dKlS+3t7W/evCmyWVJSEo1Go9FoampqWVlZ+EOL"
|
|
12
|
+
"Fi0Ch1asWCHfsf3888/6+voLFixYtmyZr6+vvLotKChQVVXV0tJSU1MbNWpUZ2enuJY8Hi8+Pj48PHzq1KkjR468evUqeh/75BFE"
|
|
13
|
+
"kIi2trYBAwbASfvll1/k2HlERAToduTIkc3NzVKdm5OTA0cVHBwsx1F5eHjAnv/zn/98hkmeNGkSuNycOXNaWlpEtnn9+jVoY2pq"
|
|
14
|
+
"Sji0bt06cCg5OVm+AwsNDYVT8fr1a3l1i+c+8t/u3r17sKWFhUV7ezt6JeUO6aRCNptNk4Tly5fLzMuOjo40eYPJZJJc8cGDB5aW"
|
|
15
|
+
"lmlpaeQDO336NIvFAtuRkZF+fn5y/Bppa2uDDTMzMy0tLXHNcnNz16xZ09XVhd85atQouL148WI5jmrIkCFgQ09P77fffvsMX+XB"
|
|
16
|
+
"gweDjR07dgwcOFBkG01NTbBBo9EIhzQ0NMQd6iUGDRoENuh0+vTp0+XVrYqKirq6OtjW0dEhafnHH3/AkcTFxdHpdCTDfQUL5D//"
|
|
17
|
+
"/PPLsUorKirC51gYNTU1zs7Ojx49Mjc3Nzc3F0eIZWVlhw8fBtt79uzZtm2byGZ8Pr+np6ejo4PFYjGZzIaGBnEs3NLSEhYWBoUC"
|
|
18
|
+
"+GQrKyuDDQ6HM3PmzLNnz5aUlIA9Fy9eNDExuXbt2r59+/BdDRw4EL758qUASEZTpkxRVVUlHK2oqHj8+HFhYWFNTU1bW1tXVxeP"
|
|
19
|
+
"x4O/u0Ag6O7uZrFYdXV1BQUFqampFy5cIIyc5Ip4AZwANTU1cYeEBykvQMJSVVWV7ySTPJwQTCYzPj4ebB88eHDYsGGItv6BBXJ7"
|
|
20
|
+
"e7uFhUVcXFxXV5dAIICSETnevXsnm4zq4OAg37uj0+kkl3N1dSW0t7e3LywsxLfh8XjW1tayXf3MmTPCF/X29gbv89WrV8GeBw8e"
|
|
21
|
+
"gPbr168He/BasMuXLwsEgpMnT/7/t0tB4fnz5/gOITtkZmbKcb1w+vRp0K2Dg4Pw0djYWGlnQ09Pj/yKq1atAi1v3LgBdzY0NKSn"
|
|
22
|
+
"pxcVFTU1NXE4nObmZihBE07fs2cPOJSSksLj8fBcnJGRQbIe7yP4+/tT1wnExsaKaxMSEkL9ovr6+nw+H612ZYAEKoTaa3V1dWtr"
|
|
23
|
+
"awaDIa4ln88fMWIEaHzgwIF/9q78/f2hEEHSjMvlRkdHE9YmdDq9trYWtpF5LdyvX7+mpibhi65evRrSLtjz9OlTAhWGh4eDPebm"
|
|
24
|
+
"5nB6LSwsoJjW09MDO4TClHyp8MKFC/KlwtmzZ5Nfcf369cJUmJKS0ntWGjt27JdJhdOmTSOnQj6fP378eOoXDQ0NRaTWJ7pCSIVd"
|
|
25
|
+
"XV3btm0jkedpNJq9vT3Y/losXIqKim5ubh8+fPD29oYLn4CAAEiOERERQUFBJIs1AwMDXV3dESNGaGlp0el0RUVFeNTOzg4qv/DY"
|
|
26
|
+
"vHkz2Lh//35TU5PInh8+fAjNuHB6o6Ki+vXrh2FYXl4eFNkwDFNS6hOPKLl3K1HDBfUD/x5InOTExMTi4mKKvQ0YMGDTpk1opSt/"
|
|
27
|
+
"XSGLxbp27RrY9vPzk7h6/eabb8BGUVHR3bt3v5YpUFdXDw0NTU5OHjp0qJ2dnY+PD9h/8uTJnTt3Yp/sldBsV1hYCHULT58+LS0t"
|
|
28
|
+
"ra6ubm5ubm9vf/78OezWxcVF5OWsra2BrYPL5YoUedhs9v379zEMGz16NH5tPmnSJDiewMBABoMhx0l4/fr127dvq6qqWltbu7u7"
|
|
29
|
+
"+Xw+XikGFpuNjY0lJSUZGRlsNtvCwuLp06cFBQWVlZWNjY0tLS0cDqerq6unp4fL5fJxsLGxgWKyxC/T5/zdo6Ojnzx5kpWVVVBQ"
|
|
30
|
+
"AH/EtrY2cCNcLpfH44FbuHHjBjTsgD08Hq+np6e7u5vD4bBYLAaDUV9fX1FRUVRUlJOT8/z58wcPHgg/AF1dXUlJSW/evKmrq2Oz"
|
|
31
|
+
"2VwuF04yn89vbW2trKzMzs6+c+cOl8sFTwh+UfLy5UuR4szatWtBA1dXV3HmJoRe6QrPnj0L2owYMYLNZkuUMLlc7siRI7FP1swv"
|
|
32
|
+
"f4FMQG1tbWNjI9gODg6GUzR9+nQmkwmb7dixAx7av38/vgdbW1vsk6Gzs7NT3IUgo61Zs0Z4gRwTEwP+9fHxIZzIYDCgYP7TTz+B"
|
|
33
|
+
"nVD27M0CGRIWRs0hToaebWxsCIfS09Nfv35dWVnZ0tLS3d3t5uYGWl6/fp3BYBQXFz98+LC6ujojI6OoqKiurg7Yo8TpCuGPnpyc"
|
|
34
|
+
"3NPTAxSLpaWlmZmZ9+/f783jhKfC3vTT2NhIcYZZLJZAIIiKisLv/P7774X7LCgoUFBQACIhfHoRZIBY+ZzL5UI68PPzI7Hc4b/q"
|
|
35
|
+
"mzdvDgwMxDAsLS3t5s2bjo6OVH749vb2pKQkTU1NTU1NdXV1dXV1Op3er18/ZWVlZWVlBQUF8GPjbbXAXAue+La2tubm5vr6emtr"
|
|
36
|
+
"ayrjFAe80tDLy2vatGk7d+5UVVVNTU2FPhwfPnyIjo4Gyz0OhxMcHLxu3ToDAwNwF5MnT1ZWVi4pKbGyslJRURF3IScnp/DwcHV1"
|
|
37
|
+
"deiwggekwqVLlwobHHfu3BkYGEin0+FX5+uFl5fXq1evhPdDdSogBVNTU7zILLFbGo2mpKSkpKQEHJh1dXVJGpeUlOTn52tqampo"
|
|
38
|
+
"aIBnr3///uDBU1RUBP5YIgUIIBt2d3eDh5DNZrNYrJaWltbWVicnp95PDpvNPnjwIH7P1atXDx06NHbsWPxOT09PPp+PYdju3btF"
|
|
39
|
+
"Pk4IvZUKL168CEXCjo4Oisza3NwMPSFGjRrV1tZG5ayPHz/K5V4+fvzYG6kQuIAEBQUdPHgQ/Mtms/E+z11dXbNnzwY9JyQkAH9A"
|
|
40
|
+
"IyMjvMxIBTwe7/r16xwOR5zZpKys7Pz587ABQbJwcnKCd/pVS4WzZs2iKB9BQB8GEqkwJSWF+vAIklefvlNSSYXQxZ1Op0OP0R9+"
|
|
41
|
+
"+AHfIfT3HD16NGGiEORjQWYymdB96dSpU1L1+PPPP8NfdOXKlVRM+18IFT569AhY9IRfM7D8h197W1tbfBCCiYlJfX29zL+BMBVK"
|
|
42
|
+
"BblQYWZmZnZ2dkVFBZPJ7Ozs5HK50Ebs4ODQ2dnJZDJramoKCwvT0tKkeusQFeLDS27evPnXX38VFxc3Nzd3dnYaGxuDU3777Tc2"
|
|
43
|
+
"m11TU/Pu3bt79+7h4w5jY2PfvXsHhFMajfbs2TPQW35+PnB4pNFoqampiMv6hAqh1kZXV1dYJKypqSF3RcTL8N7e3l8+Fb548cLK"
|
|
44
|
+
"ygq/wiovLyd8G6AeUEdHp6qqCpCjiYkJ3JmYmEhFar5z505mZmZFRUVLSwvwTMZTITBQ1NfXl5aWvnr1KiUlBSqAeDyeSIdNuVAh"
|
|
45
|
+
"ibuMSGcauVDhvXv3nj179uHDh8bGRg6Hs2XLFtDy2rVrDQ0N+fn5SUlJeLchILaHhYWFhYXFx8cLax6DgoKCgoJKS0u/TCokoLq6"
|
|
46
|
+
"Gmp+8M40ubm5MHLG3d0d7ISm4YkTJ7a2tjY1NUFnIE9PT0RkfUKF0GoMFNiEoz09PYMHD161alVWVpa4TpOSkvBPhqurK5fLJRkE"
|
|
47
|
+
"m82+fv36/fv3X7x4kZeXV1ZWBo2SeIskUNDw+Xwul9vd3d3e3s5gMCorKwsLC58/f56YmAhtOxSpkM/n37p1y9zcHD/aiRMnEoJY"
|
|
48
|
+
"k5OTIblraGi8ePEC/zSPGTMGnjt37tyYmBhx8bNA+JL2vXr69Ck4cfbs2erq6gSO/qqpkABIhXi/QiogBPlUVVW9f/8+MzPzwYMH"
|
|
49
|
+
"cgkZlpfZhIDjx4/jRT+4v6OjIyEhwcLCYtq0aVAQYTAY8CG0s7ODipolS5aA8AcEOVNhTk4O/CKJtALjaW7WrFkiAyoEAsHu3bvx"
|
|
50
|
+
"77OFhYXwO/wPWpDLysoCAwMJCnU1NbVjx47hH6zMzEw7Ozu8XUXYoaG0tJSQN0lZWXnWrFnu7u4nTpwgRIbIRoWdnZ3Dhw8XR0z/"
|
|
51
|
+
"81T4/v17Jycnc3PzGTNm6OvrDx8+XFNTExg3CPY0TFTWny+TCvl8/oQJE/BUWFlZuX37dhC3ChdY+FNevnxJsAoaGxu3trYiFpM/"
|
|
52
|
+
"FVZUVEC7ZL9+/XJycoRPgM6DYCEJLQzCH2pLS0vs79HyERERn+cLJpEKYTIYiFWrVlVUVMAGaWlpCxYsIPgDitMMtLW1ubq6Cpsa"
|
|
53
|
+
"VVVVCXMos1SIT4jw8OHDf4oKwXL+M1NhQUGBzGvVY8eOiVRT3Lhx4/Hjx69evSoqKqqqqmpqampra+vo6ABLECpUCL0LOzs72Ww2"
|
|
54
|
+
"EEjLy8sLCwuzs7OfPn2anJxMorUErqN4oznwuyT3QsMr4jEMW7duXUNDA2IxOVNhcXExXko6cuSIcOvKykp8SEBISIiANLEVFOMh"
|
|
55
|
+
"xowZk5aW9o9TIZ/Ph0w3bty4pKQkQgMOhwPTLkyePPm///2vxItmZGTA2DiAw4cPE9q0tLSkpqYCAwWDweBwOOXl5TDkAOgK2Ww2"
|
|
56
|
+
"0BVmZ2c/fPgQmLD5fL6RkRH2KYfVP0KFd+/eNTIywjOamZkZdVaSlgpbW1sPHjzY2dnZ0NAgMxWGhYXJRU0hGyZNmiTuZvG6aQzD"
|
|
57
|
+
"IiIioMfozZs3hduXlpbipRC8hOHj4yOVTR+BjApfvnwJl2AYhllaWoJEIwTASAwMw+bNmyfROtzU1DRnzhz8L7d27dru7m5x7YFP"
|
|
58
|
+
"ogyIiYmR1sX62bNnysrKvr6+JNnfoqOj7927J3IqxCErK8vNzW3IkCF0Op0kZBsiICAA3gW5BRnKJiAe63NSYXJy8vz588G/y5cv"
|
|
59
|
+
"72sqjIqK8vf3HzRokIaGBvimimMWGLZM4IKwsDDsU0a1L5AKX7x4gf09GU9sbOzRo0ehuICPaMjPz3d1dcW7qYp0dTQ2Nt61a9eN"
|
|
60
|
+
"GzdKS0tRRgYZqfDWrVv4iR43bpxI1xAmkwl/ORqN9tdff1G5AIvFgh/AqVOnki+Qpc3hjOHSDsoQbVJWVga3YeqX3gDSBI/HKy4u"
|
|
61
|
+
"pqLvhzksIBXy+fzjx49XVlYKhLwRYWT+/Pnz+5oKf//9d5HxcFZWVn1NhXin/c7OTpiiUQYqjIqK+gKpEFjqjI2NoTtRbGxsfX09"
|
|
62
|
+
"8Ot2d3dvaWnp6OiIj4+3tbUlEN+UKVMyMjJu3rwpMsIduiKKTMaDICBPx2BraxsYGAjSvWlqaiYlJQ0dOlR4fk+cONHa2gq2nZ2d"
|
|
63
|
+
"586dSzHCNyUlxdPTE8OwM2fOkAeikiSqIwc09WBSJizBhPKn9gYw8aqCgoK+vr7E9teuXaupqcH+nspw5cqVvr6+0J8JQkFBAUwj"
|
|
64
|
+
"hmGtra1QVpIvPn78GBERYWNjAzOYgeoF/fr1c3R0jIuLS0xMxGvi0tPTc3JySktLa2pqmpubWSwWIYaX3Hmby+W+evUqNDTU3t6e"
|
|
65
|
+
"kO1m5syZe/fu5XK5EoOXMXnnyBF88qiHAT9KSkqPHz8GmhPqbxcMV8cjLi7uyZMn4CXC7x86dOjRo0cfPnxobm7u6empo6OzZs2a"
|
|
66
|
+
"u3fvwiyQAwcOPHHixJs3b0xNTR0cHHJzc2H0sXDc1Lx581D8iIw+UC9evBg+fDjB4onXNEO6GTBgALlroUCMI7HENrdv36buD4hP"
|
|
67
|
+
"nkiwTsjgYn39+vXeT6abm5tAmsgWQJdQtlq/fj14SeA7IxDyOhozZkxwcDBeyYCXCpubmxMTEz08PCi+7fjoF3d3d5H5oIYPH37q"
|
|
68
|
+
"1CmRCcd6aTbZvHmzyHQ1Li4uhKyR8pUKKfIgPmw0Kiqqrq7O09Nz+PDh4tgwJiZGoh68ubkZyBnKysq1tbV4qVDw9/h0PPr37+/t"
|
|
69
|
+
"7S3yJ3jy5MnChQsJ7eVbcOLfaEEmybng5eWFSarDwGAwEhISejMamKlFWiokvAkyUCGJw0RLS0tycjJJrDvM2bNlyxbqNwuTEp4/"
|
|
70
|
+
"fx6/QP7xxx9hyKNw5CJBw9De3g7V7aNHj8ZntCbRyQqDJE993znTiCuZJOxX+JmpsLW1dcmSJXA8R48e7enpgc4V58+fF4jK7AkC"
|
|
71
|
+
"n48fP06iqoOS4Lp16wS4eBtIhd3d3fj1lpqampeXV3V1NfmA09PTv/322+nTpy9btkxFRYWKcgaBLF+huFwGZWVlZ86cgeZUuEyD"
|
|
72
|
+
"6OzsDA4O1tfXX716NZTs/meQl5dnb2+vra09aNCgWbNmlZaW9r5PBoMBigTMnz9/8uTJ+EMhISFgoV1TU3Po0CHCiRwOJy0tLSws"
|
|
73
|
+
"zMXFZcqUKRoaGjBZC1AvQmG/vr6e+nhgFkWgKwgICMDn5ukjLFq0CIarBwcHf/fdd1/Cz52fnz9v3rxHjx6Bf4OCgvbt26ekpLRr"
|
|
74
|
+
"1y7sU/Y2kAEB4tdff129ejWXy+Vyub6+vkuWLBFZy+H8+fOXL18G2yKlPyAtxsXFaWhozJgxw8DAoL29PSwsbOTIkeQFfBYtWnT1"
|
|
75
|
+
"6tU5c+bcuXOHyWRSUc4gYDLUQfb29oaq68jISOEUmx8/fty/fz9o88MPP2RnZ+NjML52VFdXQ0Weg4ODnp5e7/vcsmULCM4X/q5o"
|
|
76
|
+
"aWkFBAR4enrq6urOmjUrKysLZMHLzc199+5dRUUFRloexMDAQE9Pb+zYsSSpcTBRWRTBpf38/LZu3Uqn0+FL23cYN27c2LFj/fz8"
|
|
77
|
+
"Nm7cqKys7O7uTt7+/fv34kqLgORAvVcWnTt3ztvbm8PhAKNNZGQkHNWOHTsiIyPLy8vz8vLi4uKAQNrW1rZz507o9TlmzJjt27fP"
|
|
78
|
+
"nz+/f//+hM7fvn0L07uZm5uT6PJ0dXXfv3+vo6Pj6uoqlU8lWB/0XY2XfzsVpqamwsWjs7MzwXEaPoiBgYEg5TKDwVi3bl16evpn"
|
|
79
|
+
"zsTZd/jw4QP2yWXa29u79x3GxMQA1eT48eNXr16NT/gKiZLFYnl5eW3atIk8JXj//v1BplUMw+7cubNs2TLZhjRx4kQfHx8/Pz8q"
|
|
80
|
+
"hYfkaGD98OHDF5K8+s2bN9u3b8/IyIDS2a1bt2DgOYZhKioqhw8f/v777zEM27t3r6OjY2pq6rZt28CXUkdH58CBA5s2bRJn5Bk1"
|
|
81
|
+
"atSgQYPq6uowDNu/fz9GIV+ctKT2L0wDjn22inccDmfr1q3wm4MvCIsJVUyHuQn+/PNPkjz4GLUMhhKrespsbpYW0Ats6dKlsPiZ"
|
|
82
|
+
"zMjJyYGLI19fX5EfDBUVFX9/fzqdTkhOBzBhwgRnZ+fIyMjMzMy2tjaYS7GXtdCOHz/+OXkQmGWlentl0BVSQXZ29po1a2bOnAl5"
|
|
83
|
+
"EMMwDQ0NPA9CUQD4wVRVVU2bNs3R0bG6ulpNTS0gIKC4uNjDw4PE2D148GCQC9nCwgKviCR7OT+FFZI7nAo7ISHIXyrcs2cPlImC"
|
|
84
|
+
"g4NFOtlgn1zAoqKi5syZAySUwMBAW1tb4WgTcvT09Mi8tOnlRBC0Pxgul/2zZ88woZSiIgH8TkhQXl5ua2sLDD6GhobQQoKRuvvQ"
|
|
85
|
+
"aDRjY2Nzc/NFixYtXLiQkKETrx9EDzR1NDY2JiYmxsTE4BlQXV2dJDssjUa7dOnStGnT2Gz2x48faTSai4tLUFAQ3jmUBA4ODosX"
|
|
86
|
+
"L8ZnYUD4OqTCx48fw3JCNjY2EivIzJw5E36guFyui4sLLPVLER0dHbLdRnd3dy8nglBqHcNZM0EVEXV19ZUrVwKNQU5OjlSdQFHX"
|
|
87
|
+
"xsYGOhKeOHFCYpWfhQsXXr58ubGxMTs7Oyws7JtvvhHOVAzqYJCwOQImVJR1xowZQBmH58FVq1a9fv0ak6TfhGsjZWXlFStWUORB"
|
|
88
|
+
"7JMnqbTyAcI/TIUsFuvHH38EgoaGhgZ0+CAQUFNTE4iWTUtLu337tqGhIVRsFxYWHjhwQKrRQBduaZ1pxBVfl+HSBP0AVOt4eHho"
|
|
89
|
+
"aGhUVFQ4OTkZGRlZWFjcuHGDIAaK7ATDGeihXOnk5EQlcfTUqVPXr19PEleAF0UlyqTyQlFRUWpqKrDn1NbWgtAIkIftq3jonZyc"
|
|
90
|
+
"Wltb8V+OSZMmJSUlJSQk4MNPMfHeo+Cj2N3dvWbNmn379lFfzcjFkx/hcyyQi4uLGxoaWlpaYmJiysvLsU8Oa/7+/iDCAQBskwtB"
|
|
91
|
+
"QPBxcnKiGJeCYZi4YpgSQYjZwGQqWwycvKDqjcfjbd68GegHtLW1Qfz15s2bQZjHkydPnjx5Mnbs2G3bti1YsGDcuHFAi0d+lSNH"
|
|
92
|
+
"jjQ0NCQmJsoxYyhkQCge9jXi4+MlKv7lqMToDcWLvOiwYcOio6NBJMmIESP8/f1dXV0lai3LysrOnTtnaWlpaWkZFxdnaWkJYk+D"
|
|
93
|
+
"goJu37598uRJkRZFihohpNz44qgwMjLy1KlTmJBjXV5enmwP8aZNm7Kzsylqx6XKQaKqqgpdsg0NDakvVEVi48aNGzduxHB+G25u"
|
|
94
|
+
"bunp6UBDdOHCBfA9/+WXX4YNGxYfHw+W5OXl5b6+vnQ6fcOGDV5eXhKpEMOw8PDwDRs24KUD+KrLtsLtIyqE3YrzX5H5tf9sJ5JM"
|
|
95
|
+
"qbW19XfffWdmZrZx40ZhxxfCPCQnJ589e/bu3buwQDudTk9KSlq0aBF4L3Jzc62srBYvXrxr1y57e3vqkYLw1sjp/sqVK1euXEFs"
|
|
96
|
+
"9VmpcOfOneHh4VS+UUpKSkOHDtXW1oZ/tbS0tLS0Bg8eXFtbC51O8vLyQkJC8OVcMQoefBi1gu7CNj7w9b516xbF2rvCyMzMjIiI"
|
|
97
|
+
"iIuLA8yioKAQHR29YsUK7FP4amxsbEhIyKlTp06fPg0kRA6HExUVBZr5+PiQJyno378/oQGkMBneeYFAAE8H3nC9oSpFRUVotXz5"
|
|
98
|
+
"8iX2yZwthcJFQUFBQQGY+EEMMnyWyO+Oz+dDAx2BfOF9yeBXSHJRieRSW1t74cKFc+fOVVZWYrhkkeCTqaWllZGR4ezsfOfOHXAo"
|
|
99
|
+
"LS0tLS1t4MCBVlZWpqamBgYGJiYm5MoNODyRlC2tqNh7jTmGYpDxwKfVotPphoaG9vb27u7uR44cuXTp0r1793JychoaGsgTAeEr"
|
|
100
|
+
"+GhoaJDkuBfIIyGwmZnZ0KFDR48eTdDFGBgYUOmhrKwsISHB09OTEISrra1NklyeyWQeOnQIpmDAcLYOQkkAipHX+PxXFIGnvz/+"
|
|
101
|
+
"+KM301hdXa2lpaWtra2np4d/gbdt20ZomZeXl5KSkpWVVVJSUl9fD0oviKvZAFWiJiYmhEMdHR1Tp07V09MzNDTEX5Ewe1BRIwNE"
|
|
102
|
+
"pt2kooBWUFAgGLU0NTU9PDwIdSz4fP7BgwdFmr+WLl0qsXo4NLmcPn1a+CjMmEkR27dvR/Fz8qxtEh8fHxoampqaCooZyQaY3cDa"
|
|
103
|
+
"2jovL6+vb0NcMoUdO3YISEv8GBkZifRPVFFRcXV1pZKDgMlk7t27l+AN++2331If/KVLlzAxVdwEFOpGwYuePXu2l9NIKLwL8Pvv"
|
|
104
|
+
"v8slBllfX19kCgNMVOgbvs3bt29lpsLdu3dTHGdJSUl4eLiwvxSNRrOysoqLiyMpgZuXlwfCdTBc9kCJPCgQCOCzJzLbNtTYUPQr"
|
|
105
|
+
"dHFxQYwm54p3vQeXyzU3NxeZj7cvwOFwhL/MkyZNqqurE5CmCxR2dp0wYUJAQAD5iSJFqg0bNoDl2/jx4ykWgAY4duyYVDKsQEyx"
|
|
106
|
+
"wKNHj/ZyGoWj+tauXUteoos6FQ4cOFBkeh68IK+kpLR3715Cm9TUVIlJoQmALtbAC4IK3r59S3iEBg8e7OPjU1JSQrGHpKQkkMZc"
|
|
107
|
+
"UVHx7du3AgrZKuG1hO9aIBCsWrVKKirEp5JEkBZKfbTuVlRUfPz48Wdb5quqqhoYGOTk5CgpKeno6EydOnXZsmUbN24UmQMK/+Jd"
|
|
108
|
+
"vHjRxMRkzJgxxsbGc+fONTc3p2L6EMaIESNiYmJcXV1379595swZqSJhhgwZAvhC2GcQo1AmuPf2d4jRo0cbGhoqKioaGBjMmDHD"
|
|
109
|
+
"zs4OFuqVGRMnTgQDA9pDQnQNcM179+7dnDlzTE1NbWxshN1NQLCaVBgwYABIJCOxDhSEkZGRh4cHKHpjaGjo7e29fv16cosKAfb2"
|
|
110
|
+
"9vb29rm5uenp6bD6AkaakgNu46V7iKqqKirXnTJlCnh+RMYmIVAE7X/Jii8QCORl7kT4F6K4uNjZ2Xnfvn0rVqxADxKiQgQEBAQM"
|
|
111
|
+
"RZsgICAgICpEQEBAQFSIgICAgKgQAQEBAVEhAgICAqJCBAQEBESFCAgICIgKERAQEBAVIiAgICAqREBAQEBUiICAgICoEAEBAQFR"
|
|
112
|
+
"IQICAgKiQgQEBAREhQgICAiIChEQEBAQFSIgICAgKkRTgICAgICoEAEBAQH7PzsHvqY55zH8AAAAAElFTkSuQmCC"
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def make_fixtures(root):
|
|
117
|
+
from PIL import Image, ImageDraw
|
|
118
|
+
from docx import Document
|
|
119
|
+
from openpyxl import Workbook
|
|
120
|
+
from pptx import Presentation
|
|
121
|
+
from pptx.util import Inches
|
|
122
|
+
files = []
|
|
123
|
+
# Raster PDF exercises the actual OCR path without a hidden text layer.
|
|
124
|
+
image = Image.new("RGB", (1000, 600), "white")
|
|
125
|
+
draw = ImageDraw.Draw(image)
|
|
126
|
+
draw.text((80, 70), "TOTAL 123", fill="black", font_size=60)
|
|
127
|
+
draw.text((80, 180), "Document conversion fixture", fill="black", font_size=32)
|
|
128
|
+
image.paste(Image.open(BytesIO(b64decode(CHINESE_CAPTION_PNG))), (80, 260))
|
|
129
|
+
draw.rectangle((75, 350, 700, 520), outline="black", width=3)
|
|
130
|
+
draw.line((75, 430, 700, 430), fill="black", width=3)
|
|
131
|
+
draw.line((400, 350, 400, 520), fill="black", width=3)
|
|
132
|
+
draw.text((95, 375), "Name", fill="black", font_size=32)
|
|
133
|
+
draw.text((425, 375), "Value", fill="black", font_size=32)
|
|
134
|
+
draw.text((95, 455), "Total", fill="black", font_size=32)
|
|
135
|
+
draw.text((425, 455), "123", fill="black", font_size=32)
|
|
136
|
+
path = root / "scanned.pdf"
|
|
137
|
+
image.save(path, "PDF", resolution=120)
|
|
138
|
+
files.append((path, "123"))
|
|
139
|
+
# A born-digital PDF preserves text independently of OCR.
|
|
140
|
+
objects = [b"<< /Type /Catalog /Pages 2 0 R >>", b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>", b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 4 0 R /F2 6 0 R /F3 7 0 R >> >> /Contents 5 0 R >>", b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"]
|
|
141
|
+
chinese = "公司技术资料".encode("utf-16-be").hex().encode()
|
|
142
|
+
stream = b"BT /F1 18 Tf 72 720 Td (Document conversion fixture 456) Tj ET\n"
|
|
143
|
+
stream += b"BT /F3 16 Tf 72 680 Td <" + chinese + b"> Tj ET\n"
|
|
144
|
+
stream += b"72 500 400 100 re 72 550 m 472 550 l 272 500 m 272 600 l S\n"
|
|
145
|
+
stream += b"BT /F1 14 Tf 85 570 Td (Name) Tj 200 0 Td (Value) Tj ET\nBT /F1 14 Tf 85 520 Td (Total) Tj 200 0 Td (456) Tj ET\n"
|
|
146
|
+
header_stream = stream
|
|
147
|
+
code_lines = ["def calculate_total(values):", " return sum(values)", "", "def print_report(values):", " total = calculate_total(values)", " print('Total:', total)", "", "print_report([1, 2, 3])"]
|
|
148
|
+
stream += b"0.95 g 62 284 470 164 re f 0 g\nBT /F2 12 Tf 72 430 Td\n"
|
|
149
|
+
for index, line in enumerate(code_lines):
|
|
150
|
+
if index:
|
|
151
|
+
stream += b"0 -18 Td\n"
|
|
152
|
+
escaped = line.replace("\\", "\\\\").replace("(", "\\(").replace(")", "\\)")
|
|
153
|
+
stream += b"(" + escaped.encode("ascii") + b") Tj\n"
|
|
154
|
+
stream += b"ET\n"
|
|
155
|
+
objects.append(b"<< /Length " + str(len(stream)).encode() + b" >>\nstream\n" + stream + b"\nendstream")
|
|
156
|
+
objects.extend([b"<< /Type /Font /Subtype /Type1 /BaseFont /Courier >>", b"<< /Type /Font /Subtype /Type0 /BaseFont /STSong-Light /Encoding /UniGB-UCS2-H /DescendantFonts [8 0 R] >>", b"<< /Type /Font /Subtype /CIDFontType0 /BaseFont /STSong-Light /CIDSystemInfo << /Registry (Adobe) /Ordering (GB1) /Supplement 4 >> /DW 1000 >>"])
|
|
157
|
+
def write_pdf(name, content):
|
|
158
|
+
objects[4] = b"<< /Length " + str(len(content)).encode() + b" >>\nstream\n" + content + b"\nendstream"
|
|
159
|
+
data, offsets = bytearray(b"%PDF-1.4\n"), [0]
|
|
160
|
+
for index, value in enumerate(objects, 1):
|
|
161
|
+
offsets.append(len(data))
|
|
162
|
+
data.extend(str(index).encode() + b" 0 obj\n" + value + b"\nendobj\n")
|
|
163
|
+
xref = len(data)
|
|
164
|
+
data.extend(("xref\n0 %d\n0000000000 65535 f \n" % (len(objects) + 1)).encode())
|
|
165
|
+
for offset in offsets[1:]:
|
|
166
|
+
data.extend(("%010d 00000 n \n" % offset).encode())
|
|
167
|
+
data.extend(("trailer\n<< /Root 1 0 R /Size %d >>\nstartxref\n%d\n%%%%EOF\n" % (len(objects) + 1, xref)).encode())
|
|
168
|
+
path = root / name
|
|
169
|
+
path.write_bytes(data)
|
|
170
|
+
files.append((path, "456"))
|
|
171
|
+
objects[5] = b"<< /Type /Font /Subtype /Type1 /BaseFont /Courier /Encoding /WinAnsiEncoding >>"
|
|
172
|
+
write_pdf("digital.pdf", stream)
|
|
173
|
+
# StandardEncoding maps this font's quote byte to U+2019. Retain the
|
|
174
|
+
# ambiguous original case: a recognizer must not silently turn it into ASCII.
|
|
175
|
+
objects[5] = b"<< /Type /Font /Subtype /Type1 /BaseFont /Courier >>"
|
|
176
|
+
write_pdf("digital-ambiguous.pdf", stream)
|
|
177
|
+
# Keep the short, unboxed snippet which the layout model misclassified in the
|
|
178
|
+
# first real qualification: this must be preserved or explicitly report partial.
|
|
179
|
+
write_pdf("digital-short.pdf", header_stream + b"BT /F2 12 Tf 72 430 Td (def calculate_total\\(values\\):) Tj 0 -18 Td ( return sum\\(values\\)) Tj ET\n")
|
|
180
|
+
document = Document()
|
|
181
|
+
document.add_heading("公司资料 Fixture document", 1)
|
|
182
|
+
document.add_paragraph("Office text 789")
|
|
183
|
+
table = document.add_table(rows=2, cols=2)
|
|
184
|
+
for cell, value in zip([cell for row in table.rows for cell in row.cells], ["Name", "Value", "Total", "789"]):
|
|
185
|
+
cell.text = value
|
|
186
|
+
path = root / "office.docx"
|
|
187
|
+
document.save(path)
|
|
188
|
+
files.append((path, "789"))
|
|
189
|
+
book = Workbook()
|
|
190
|
+
book.active.append(["Name", "Value"])
|
|
191
|
+
book.active.append(["Total", 321])
|
|
192
|
+
path = root / "sheet.xlsx"
|
|
193
|
+
book.save(path)
|
|
194
|
+
files.append((path, "321"))
|
|
195
|
+
presentation = Presentation()
|
|
196
|
+
slide = presentation.slides.add_slide(presentation.slide_layouts[6])
|
|
197
|
+
slide.shapes.add_textbox(Inches(1), Inches(1), Inches(7), Inches(2)).text = "Presentation fixture 654"
|
|
198
|
+
path = root / "slides.pptx"
|
|
199
|
+
presentation.save(path)
|
|
200
|
+
files.append((path, "654"))
|
|
201
|
+
return files
|