@jackgreen2018/pdf-engine 1.0.45 → 1.0.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +143 -33
- package/dist/cli.d.ts +3 -0
- package/dist/cli.d.ts.map +1 -0
- package/dist/cli.js +60 -0
- package/dist/index.d.ts +17 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +49 -18
- package/package.json +38 -14
- package/pkg/README.md +143 -33
- package/pkg/package.json +1 -2
- package/pkg/pdf_engine.d.ts +6 -8
- package/pkg/pdf_engine.js +1 -1
- package/pkg/pdf_engine_bg.js +60 -78
- package/pkg/pdf_engine_bg.wasm +0 -0
- package/pkg/pdf_engine_bg.wasm.d.ts +5 -6
- package/src/cli.ts +66 -0
- package/src/index.ts +62 -0
- package/src/lib.rs +294 -0
- package/LICENSE +0 -21
- package/index.d.ts +0 -6
- package/pkg/LICENSE +0 -21
package/pkg/pdf_engine.d.ts
CHANGED
|
@@ -1,16 +1,14 @@
|
|
|
1
1
|
/* tslint:disable */
|
|
2
2
|
/* eslint-disable */
|
|
3
3
|
|
|
4
|
-
export function
|
|
4
|
+
export function extract_text(buffer: Uint8Array): string;
|
|
5
5
|
|
|
6
|
-
export function
|
|
6
|
+
export function get_page_count(buffer: Uint8Array): number;
|
|
7
7
|
|
|
8
|
-
export function
|
|
8
|
+
export function get_pdf_info(buffer: Uint8Array): string;
|
|
9
9
|
|
|
10
|
-
export function
|
|
10
|
+
export function init(): void;
|
|
11
11
|
|
|
12
|
-
export function
|
|
12
|
+
export function merge(pdf_buffers: Array<any>): Uint8Array;
|
|
13
13
|
|
|
14
|
-
export function
|
|
15
|
-
|
|
16
|
-
export function split_pdf(input: Uint8Array, pages: Array<any>): Array<any>;
|
|
14
|
+
export function split(buffer: Uint8Array, pages: Array<any>): Array<any>;
|
package/pkg/pdf_engine.js
CHANGED
|
@@ -5,5 +5,5 @@ import { __wbg_set_wasm } from "./pdf_engine_bg.js";
|
|
|
5
5
|
__wbg_set_wasm(wasm);
|
|
6
6
|
wasm.__wbindgen_start();
|
|
7
7
|
export {
|
|
8
|
-
get_page_count, get_pdf_info, init,
|
|
8
|
+
extract_text, get_page_count, get_pdf_info, init, merge, split
|
|
9
9
|
} from "./pdf_engine_bg.js";
|
package/pkg/pdf_engine_bg.js
CHANGED
|
@@ -1,106 +1,100 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @param {Uint8Array}
|
|
3
|
-
* @returns {
|
|
2
|
+
* @param {Uint8Array} buffer
|
|
3
|
+
* @returns {string}
|
|
4
4
|
*/
|
|
5
|
-
export function
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
5
|
+
export function extract_text(buffer) {
|
|
6
|
+
let deferred3_0;
|
|
7
|
+
let deferred3_1;
|
|
8
|
+
try {
|
|
9
|
+
const ptr0 = passArray8ToWasm0(buffer, wasm.__wbindgen_malloc);
|
|
10
|
+
const len0 = WASM_VECTOR_LEN;
|
|
11
|
+
const ret = wasm.extract_text(ptr0, len0);
|
|
12
|
+
var ptr2 = ret[0];
|
|
13
|
+
var len2 = ret[1];
|
|
14
|
+
if (ret[3]) {
|
|
15
|
+
ptr2 = 0; len2 = 0;
|
|
16
|
+
throw takeFromExternrefTable0(ret[2]);
|
|
17
|
+
}
|
|
18
|
+
deferred3_0 = ptr2;
|
|
19
|
+
deferred3_1 = len2;
|
|
20
|
+
return getStringFromWasm0(ptr2, len2);
|
|
21
|
+
} finally {
|
|
22
|
+
wasm.__wbindgen_free(deferred3_0, deferred3_1, 1);
|
|
11
23
|
}
|
|
12
|
-
return ret[0] >>> 0;
|
|
13
24
|
}
|
|
14
25
|
|
|
15
26
|
/**
|
|
16
|
-
* @param {Uint8Array}
|
|
17
|
-
* @returns {
|
|
27
|
+
* @param {Uint8Array} buffer
|
|
28
|
+
* @returns {number}
|
|
18
29
|
*/
|
|
19
|
-
export function
|
|
20
|
-
const ptr0 = passArray8ToWasm0(
|
|
30
|
+
export function get_page_count(buffer) {
|
|
31
|
+
const ptr0 = passArray8ToWasm0(buffer, wasm.__wbindgen_malloc);
|
|
21
32
|
const len0 = WASM_VECTOR_LEN;
|
|
22
|
-
const ret = wasm.
|
|
33
|
+
const ret = wasm.get_page_count(ptr0, len0);
|
|
23
34
|
if (ret[2]) {
|
|
24
35
|
throw takeFromExternrefTable0(ret[1]);
|
|
25
36
|
}
|
|
26
|
-
return
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
/**
|
|
30
|
-
* @returns {Promise<any>}
|
|
31
|
-
*/
|
|
32
|
-
export function init() {
|
|
33
|
-
const ret = wasm.init();
|
|
34
|
-
return ret;
|
|
37
|
+
return ret[0] >>> 0;
|
|
35
38
|
}
|
|
36
39
|
|
|
37
40
|
/**
|
|
38
|
-
* @param {Uint8Array}
|
|
39
|
-
* @
|
|
40
|
-
* @returns {Uint8Array}
|
|
41
|
+
* @param {Uint8Array} buffer
|
|
42
|
+
* @returns {string}
|
|
41
43
|
*/
|
|
42
|
-
export function
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
44
|
+
export function get_pdf_info(buffer) {
|
|
45
|
+
let deferred3_0;
|
|
46
|
+
let deferred3_1;
|
|
47
|
+
try {
|
|
48
|
+
const ptr0 = passArray8ToWasm0(buffer, wasm.__wbindgen_malloc);
|
|
49
|
+
const len0 = WASM_VECTOR_LEN;
|
|
50
|
+
const ret = wasm.get_pdf_info(ptr0, len0);
|
|
51
|
+
var ptr2 = ret[0];
|
|
52
|
+
var len2 = ret[1];
|
|
53
|
+
if (ret[3]) {
|
|
54
|
+
ptr2 = 0; len2 = 0;
|
|
55
|
+
throw takeFromExternrefTable0(ret[2]);
|
|
56
|
+
}
|
|
57
|
+
deferred3_0 = ptr2;
|
|
58
|
+
deferred3_1 = len2;
|
|
59
|
+
return getStringFromWasm0(ptr2, len2);
|
|
60
|
+
} finally {
|
|
61
|
+
wasm.__wbindgen_free(deferred3_0, deferred3_1, 1);
|
|
50
62
|
}
|
|
51
|
-
var v3 = getArrayU8FromWasm0(ret[0], ret[1]).slice();
|
|
52
|
-
wasm.__wbindgen_free(ret[0], ret[1] * 1, 1);
|
|
53
|
-
return v3;
|
|
54
63
|
}
|
|
55
64
|
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
export function merge_pdfs_vec(docs) {
|
|
61
|
-
const ret = wasm.merge_pdfs_vec(docs);
|
|
62
|
-
if (ret[3]) {
|
|
63
|
-
throw takeFromExternrefTable0(ret[2]);
|
|
65
|
+
export function init() {
|
|
66
|
+
const ret = wasm.init();
|
|
67
|
+
if (ret[1]) {
|
|
68
|
+
throw takeFromExternrefTable0(ret[0]);
|
|
64
69
|
}
|
|
65
|
-
var v1 = getArrayU8FromWasm0(ret[0], ret[1]).slice();
|
|
66
|
-
wasm.__wbindgen_free(ret[0], ret[1] * 1, 1);
|
|
67
|
-
return v1;
|
|
68
70
|
}
|
|
69
71
|
|
|
70
72
|
/**
|
|
71
|
-
* @param {
|
|
73
|
+
* @param {Array<any>} pdf_buffers
|
|
72
74
|
* @returns {Uint8Array}
|
|
73
75
|
*/
|
|
74
|
-
export function
|
|
75
|
-
const
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
if (ret[3]) {
|
|
79
|
-
throw takeFromExternrefTable0(ret[2]);
|
|
76
|
+
export function merge(pdf_buffers) {
|
|
77
|
+
const ret = wasm.merge(pdf_buffers);
|
|
78
|
+
if (ret[2]) {
|
|
79
|
+
throw takeFromExternrefTable0(ret[1]);
|
|
80
80
|
}
|
|
81
|
-
|
|
82
|
-
wasm.__wbindgen_free(ret[0], ret[1] * 1, 1);
|
|
83
|
-
return v2;
|
|
81
|
+
return takeFromExternrefTable0(ret[0]);
|
|
84
82
|
}
|
|
85
83
|
|
|
86
84
|
/**
|
|
87
|
-
* @param {Uint8Array}
|
|
85
|
+
* @param {Uint8Array} buffer
|
|
88
86
|
* @param {Array<any>} pages
|
|
89
87
|
* @returns {Array<any>}
|
|
90
88
|
*/
|
|
91
|
-
export function
|
|
92
|
-
const ptr0 = passArray8ToWasm0(
|
|
89
|
+
export function split(buffer, pages) {
|
|
90
|
+
const ptr0 = passArray8ToWasm0(buffer, wasm.__wbindgen_malloc);
|
|
93
91
|
const len0 = WASM_VECTOR_LEN;
|
|
94
|
-
const ret = wasm.
|
|
92
|
+
const ret = wasm.split(ptr0, len0, pages);
|
|
95
93
|
if (ret[2]) {
|
|
96
94
|
throw takeFromExternrefTable0(ret[1]);
|
|
97
95
|
}
|
|
98
96
|
return takeFromExternrefTable0(ret[0]);
|
|
99
97
|
}
|
|
100
|
-
export function __wbg_Error_92b29b0548f8b746(arg0, arg1) {
|
|
101
|
-
const ret = Error(getStringFromWasm0(arg0, arg1));
|
|
102
|
-
return ret;
|
|
103
|
-
}
|
|
104
98
|
export function __wbg___wbindgen_number_get_394265ed1e1b84ee(arg0, arg1) {
|
|
105
99
|
const obj = arg1;
|
|
106
100
|
const ret = typeof(obj) === 'number' ? obj : undefined;
|
|
@@ -136,10 +130,6 @@ export function __wbg_new_32b398fb48b6d94a() {
|
|
|
136
130
|
const ret = new Array();
|
|
137
131
|
return ret;
|
|
138
132
|
}
|
|
139
|
-
export function __wbg_new_7796ffc7ed656783() {
|
|
140
|
-
const ret = new Map();
|
|
141
|
-
return ret;
|
|
142
|
-
}
|
|
143
133
|
export function __wbg_new_from_slice_77cdfb7977362f3c(arg0, arg1) {
|
|
144
134
|
const ret = new Uint8Array(getArrayU8FromWasm0(arg0, arg1));
|
|
145
135
|
return ret;
|
|
@@ -151,14 +141,6 @@ export function __wbg_push_d2ae3af0c1217ae6(arg0, arg1) {
|
|
|
151
141
|
const ret = arg0.push(arg1);
|
|
152
142
|
return ret;
|
|
153
143
|
}
|
|
154
|
-
export function __wbg_resolve_2191a4dfe481c25b(arg0) {
|
|
155
|
-
const ret = Promise.resolve(arg0);
|
|
156
|
-
return ret;
|
|
157
|
-
}
|
|
158
|
-
export function __wbg_set_575dd786d51585f8(arg0, arg1, arg2) {
|
|
159
|
-
const ret = arg0.set(arg1, arg2);
|
|
160
|
-
return ret;
|
|
161
|
-
}
|
|
162
144
|
export function __wbindgen_cast_0000000000000001(arg0, arg1) {
|
|
163
145
|
// Cast intrinsic for `Ref(String) -> Externref`.
|
|
164
146
|
const ret = getStringFromWasm0(arg0, arg1);
|
package/pkg/pdf_engine_bg.wasm
CHANGED
|
Binary file
|
|
@@ -1,13 +1,12 @@
|
|
|
1
1
|
/* tslint:disable */
|
|
2
2
|
/* eslint-disable */
|
|
3
3
|
export const memory: WebAssembly.Memory;
|
|
4
|
+
export const extract_text: (a: number, b: number) => [number, number, number, number];
|
|
4
5
|
export const get_page_count: (a: number, b: number) => [number, number, number];
|
|
5
|
-
export const get_pdf_info: (a: number, b: number) => [number, number, number];
|
|
6
|
-
export const init: () =>
|
|
7
|
-
export const
|
|
8
|
-
export const
|
|
9
|
-
export const optimize_pdf: (a: number, b: number) => [number, number, number, number];
|
|
10
|
-
export const split_pdf: (a: number, b: number, c: any) => [number, number, number];
|
|
6
|
+
export const get_pdf_info: (a: number, b: number) => [number, number, number, number];
|
|
7
|
+
export const init: () => [number, number];
|
|
8
|
+
export const merge: (a: any) => [number, number, number];
|
|
9
|
+
export const split: (a: number, b: number, c: any) => [number, number, number];
|
|
11
10
|
export const __wbindgen_externrefs: WebAssembly.Table;
|
|
12
11
|
export const __wbindgen_malloc: (a: number, b: number) => number;
|
|
13
12
|
export const __externref_table_dealloc: (a: number) => void;
|
package/src/cli.ts
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { createRequire } from 'module';
|
|
3
|
+
import pdfEngine from './index.js';
|
|
4
|
+
const require = createRequire(import.meta.url);
|
|
5
|
+
|
|
6
|
+
const USAGE = `pdf-engine — Rust/WASM PDF processing (v${require('../package.json').version})
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
pdf-engine <command> [args]
|
|
10
|
+
|
|
11
|
+
Commands:
|
|
12
|
+
page-count <file> Print the page count of a PDF
|
|
13
|
+
extract-text <file> Extract text from a PDF
|
|
14
|
+
merge <file1> <file2> ... Merge PDFs into a single output
|
|
15
|
+
split <file> <page1> ... Split a PDF by 0-based page indices
|
|
16
|
+
help Print this message
|
|
17
|
+
`;
|
|
18
|
+
|
|
19
|
+
async function main(): Promise<number> {
|
|
20
|
+
const [cmd, ...args] = process.argv.slice(2);
|
|
21
|
+
if (!cmd || cmd === '-h' || cmd === '--help' || cmd === 'help') {
|
|
22
|
+
process.stdout.write(USAGE);
|
|
23
|
+
return 0;
|
|
24
|
+
}
|
|
25
|
+
const fs = await import('fs/promises');
|
|
26
|
+
const read = async (p: string) => new Uint8Array(await fs.readFile(p));
|
|
27
|
+
await pdfEngine.init();
|
|
28
|
+
switch (cmd) {
|
|
29
|
+
case 'page-count': {
|
|
30
|
+
const n = await pdfEngine.getPageCount(await read(args[0]));
|
|
31
|
+
process.stdout.write(String(n) + '\n');
|
|
32
|
+
return 0;
|
|
33
|
+
}
|
|
34
|
+
case 'extract-text': {
|
|
35
|
+
const t = await pdfEngine.extractText(await read(args[0]));
|
|
36
|
+
process.stdout.write(t + '\n');
|
|
37
|
+
return 0;
|
|
38
|
+
}
|
|
39
|
+
case 'merge': {
|
|
40
|
+
const out = await pdfEngine.merge(await Promise.all(args.map(read)));
|
|
41
|
+
await fs.writeFile('merged.pdf', out);
|
|
42
|
+
process.stdout.write(`wrote merged.pdf (${out.byteLength} bytes)\n`);
|
|
43
|
+
return 0;
|
|
44
|
+
}
|
|
45
|
+
case 'split': {
|
|
46
|
+
const [file, ...pages] = args;
|
|
47
|
+
const out = await pdfEngine.split(
|
|
48
|
+
await read(file),
|
|
49
|
+
pages.map(Number),
|
|
50
|
+
);
|
|
51
|
+
for (let i = 0; i < out.length; i++) {
|
|
52
|
+
await fs.writeFile(`part-${i}.pdf`, out[i]);
|
|
53
|
+
}
|
|
54
|
+
process.stdout.write(`wrote ${out.length} part-*.pdf files\n`);
|
|
55
|
+
return 0;
|
|
56
|
+
}
|
|
57
|
+
default:
|
|
58
|
+
process.stderr.write(`unknown command: ${cmd}\n${USAGE}`);
|
|
59
|
+
return 1;
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
main().then((c) => process.exit(c)).catch((e) => {
|
|
64
|
+
process.stderr.write(String(e) + '\n');
|
|
65
|
+
process.exit(1);
|
|
66
|
+
});
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
// PDF Engine TypeScript bindings for Rust/WASM PDF processing library
|
|
2
|
+
|
|
3
|
+
import * as wasm from '../pkg/pdf_engine.js';
|
|
4
|
+
|
|
5
|
+
export async function initialize(): Promise<void> {
|
|
6
|
+
await wasm.init();
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
export async function getPageCount(buffer: Uint8Array): Promise<number> {
|
|
10
|
+
return await wasm.get_page_count(buffer);
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export async function extractText(buffer: Uint8Array): Promise<string> {
|
|
14
|
+
return await wasm.extract_text(buffer);
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
export async function merge(pdfBuffers: Uint8Array[]): Promise<Uint8Array> {
|
|
18
|
+
const result = await wasm.merge(pdfBuffers);
|
|
19
|
+
return result;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export async function split(buffer: Uint8Array, pages: number[]): Promise<Uint8Array[]> {
|
|
23
|
+
const resultArray = await wasm.split(buffer, pages);
|
|
24
|
+
return Array.from(resultArray).map(arr => arr as Uint8Array);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export class PdfEngineImpl {
|
|
28
|
+
constructor() {
|
|
29
|
+
this.initialized = false;
|
|
30
|
+
}
|
|
31
|
+
private initialized = false;
|
|
32
|
+
|
|
33
|
+
async init(): Promise<void> {
|
|
34
|
+
if (!this.initialized) {
|
|
35
|
+
await initialize();
|
|
36
|
+
this.initialized = true;
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
async getPageCount(buffer: Uint8Array): Promise<number> {
|
|
41
|
+
await this.init();
|
|
42
|
+
return await getPageCount(buffer);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
async extractText(buffer: Uint8Array): Promise<string> {
|
|
46
|
+
await this.init();
|
|
47
|
+
return await extractText(buffer);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
async merge(pdfBuffers: Uint8Array[]): Promise<Uint8Array> {
|
|
51
|
+
await this.init();
|
|
52
|
+
return await merge(pdfBuffers);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
async split(buffer: Uint8Array, pages: number[]): Promise<Uint8Array[]> {
|
|
56
|
+
await this.init();
|
|
57
|
+
return await split(buffer, pages);
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const pdfEngine = new PdfEngineImpl();
|
|
62
|
+
export default pdfEngine;
|
package/src/lib.rs
ADDED
|
@@ -0,0 +1,294 @@
|
|
|
1
|
+
use wasm_bindgen::prelude::*;
|
|
2
|
+
use js_sys::{Array, Uint8Array};
|
|
3
|
+
use lopdf::{Document, Object, ObjectId};
|
|
4
|
+
use std::collections::BTreeMap;
|
|
5
|
+
|
|
6
|
+
#[wasm_bindgen]
|
|
7
|
+
pub fn init() -> Result<(), JsValue> {
|
|
8
|
+
Ok(())
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
#[wasm_bindgen]
|
|
12
|
+
pub fn get_page_count(buffer: &[u8]) -> Result<u32, JsValue> {
|
|
13
|
+
let doc = Document::load_mem(buffer)
|
|
14
|
+
.map_err(|e| JsValue::from_str(&format!("PDF error: {}", e)))?;
|
|
15
|
+
Ok(doc.get_pages().len().max(1) as u32)
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/// Decode PDF hex strings (<HHHH...>) in content stream operators.
|
|
19
|
+
fn decode_hex_strings(content: &str) -> String {
|
|
20
|
+
let mut result = String::new();
|
|
21
|
+
let bytes = content.as_bytes();
|
|
22
|
+
let mut i = 0;
|
|
23
|
+
while i < bytes.len() {
|
|
24
|
+
if bytes[i] == b'<' {
|
|
25
|
+
let mut j = i + 1;
|
|
26
|
+
while j < bytes.len() && bytes[j] != b'>' {
|
|
27
|
+
j += 1;
|
|
28
|
+
}
|
|
29
|
+
if j > i + 1 && j < bytes.len() {
|
|
30
|
+
let hex = &content[i+1..j];
|
|
31
|
+
if hex.len() % 2 == 0 && hex.chars().all(|c| c.is_ascii_hexdigit()) {
|
|
32
|
+
let mut buf = Vec::with_capacity(hex.len() / 2);
|
|
33
|
+
for k in (0..hex.len()).step_by(2) {
|
|
34
|
+
if let Ok(byte) = u8::from_str_radix(&hex[k..k+2], 16) {
|
|
35
|
+
buf.push(byte);
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
if let Ok(s) = String::from_utf8(buf) {
|
|
39
|
+
result.push_str(&s);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
i = j + 1;
|
|
44
|
+
} else {
|
|
45
|
+
i += 1;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
result
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/// Extract text from all content streams by decoding hex strings.
|
|
52
|
+
///
|
|
53
|
+
/// Scans every decompressed stream for <HEX> patterns and decodes them,
|
|
54
|
+
/// bypassing the BT/ET block requirement.
|
|
55
|
+
fn collect_page_text(doc: &Document) -> String {
|
|
56
|
+
let mut text = String::new();
|
|
57
|
+
|
|
58
|
+
for (_id, obj) in doc.objects.iter() {
|
|
59
|
+
let Object::Stream(ref stream) = obj else { continue };
|
|
60
|
+
let mut s = stream.clone();
|
|
61
|
+
let _ = s.decompress();
|
|
62
|
+
let content = String::from_utf8_lossy(&s.content);
|
|
63
|
+
if content.is_empty() {
|
|
64
|
+
continue;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
let decoded = decode_hex_strings(&content);
|
|
68
|
+
if !decoded.is_empty() {
|
|
69
|
+
if !text.is_empty() {
|
|
70
|
+
text.push('\n');
|
|
71
|
+
}
|
|
72
|
+
text.push_str(&decoded);
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
text
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
#[wasm_bindgen]
|
|
80
|
+
pub fn extract_text(buffer: &[u8]) -> Result<String, JsValue> {
|
|
81
|
+
let doc = Document::load_mem(buffer)
|
|
82
|
+
.map_err(|e| JsValue::from_str(&format!("PDF error: {}", e)))?;
|
|
83
|
+
|
|
84
|
+
let text = collect_page_text(&doc);
|
|
85
|
+
|
|
86
|
+
if text.is_empty() {
|
|
87
|
+
Ok("No text extracted.".to_string())
|
|
88
|
+
} else {
|
|
89
|
+
Ok(text)
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/// Remap all ObjectId references in `obj` using `id_map`.
|
|
94
|
+
fn remap_refs(obj: &mut Object, id_map: &BTreeMap<ObjectId, ObjectId>) {
|
|
95
|
+
match obj {
|
|
96
|
+
Object::Reference(r) => {
|
|
97
|
+
if let Some(&new_id) = id_map.get(r) {
|
|
98
|
+
*r = new_id;
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
Object::Array(arr) => {
|
|
102
|
+
for item in arr {
|
|
103
|
+
remap_refs(item, id_map);
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
Object::Dictionary(dict) => {
|
|
107
|
+
for (_, v) in dict.iter_mut() {
|
|
108
|
+
remap_refs(v, id_map);
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
Object::Stream(stream) => {
|
|
112
|
+
for (_, v) in stream.dict.iter_mut() {
|
|
113
|
+
remap_refs(v, id_map);
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
_ => {}
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/// Collect all object IDs reachable from `start` in `doc` (including start itself).
|
|
121
|
+
fn collect_reachable(start: ObjectId, doc: &Document, visited: &mut BTreeMap<ObjectId, Object>) {
|
|
122
|
+
if visited.contains_key(&start) {
|
|
123
|
+
return;
|
|
124
|
+
}
|
|
125
|
+
if let Ok(obj) = doc.get_object(start) {
|
|
126
|
+
visited.insert(start, obj.clone());
|
|
127
|
+
walk_refs(&obj, doc, visited);
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/// Walk all references in `obj` and collect reachable objects.
|
|
132
|
+
fn walk_refs(obj: &Object, doc: &Document, visited: &mut BTreeMap<ObjectId, Object>) {
|
|
133
|
+
match obj {
|
|
134
|
+
Object::Reference(r) => collect_reachable(*r, doc, visited),
|
|
135
|
+
Object::Array(arr) => {
|
|
136
|
+
for item in arr {
|
|
137
|
+
walk_refs(item, doc, visited);
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
Object::Dictionary(dict) => {
|
|
141
|
+
for (_, v) in dict.iter() {
|
|
142
|
+
walk_refs(v, doc, visited);
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
Object::Stream(stream) => {
|
|
146
|
+
for (_, v) in stream.dict.iter() {
|
|
147
|
+
walk_refs(v, doc, visited);
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
_ => {}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/// Copy page at `page_obj_id` from `src` into `dst`, updating all internal references.
|
|
155
|
+
fn copy_page_into(src: &Document, dst: &mut Document, page_obj_id: ObjectId) -> ObjectId {
|
|
156
|
+
let mut to_copy: BTreeMap<ObjectId, Object> = BTreeMap::new();
|
|
157
|
+
collect_reachable(page_obj_id, src, &mut to_copy);
|
|
158
|
+
|
|
159
|
+
let mut id_map: BTreeMap<ObjectId, ObjectId> = BTreeMap::new();
|
|
160
|
+
let mut next_id = dst.max_id + 1;
|
|
161
|
+
|
|
162
|
+
let mut ids_sorted: Vec<ObjectId> = to_copy.keys().cloned().collect();
|
|
163
|
+
ids_sorted.sort_by_key(|(x, _)| *x);
|
|
164
|
+
|
|
165
|
+
for old_id in ids_sorted {
|
|
166
|
+
let new_id = loop {
|
|
167
|
+
let candidate = (next_id, 0);
|
|
168
|
+
if !dst.objects.contains_key(&candidate) {
|
|
169
|
+
break candidate;
|
|
170
|
+
}
|
|
171
|
+
next_id += 1;
|
|
172
|
+
};
|
|
173
|
+
id_map.insert(old_id, new_id);
|
|
174
|
+
let mut obj = to_copy[&old_id].clone();
|
|
175
|
+
remap_refs(&mut obj, &id_map);
|
|
176
|
+
dst.objects.insert(new_id, obj);
|
|
177
|
+
dst.max_id = next_id.max(dst.max_id);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
*id_map.get(&page_obj_id).unwrap_or(&(0, 0))
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
#[wasm_bindgen]
|
|
184
|
+
pub fn merge(pdf_buffers: Array) -> Result<Uint8Array, JsValue> {
|
|
185
|
+
let len = pdf_buffers.length();
|
|
186
|
+
if len == 0 {
|
|
187
|
+
return Ok(Uint8Array::new_from_slice(&[]));
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
let first_buf = pdf_buffers.get(0).dyn_into::<Uint8Array>()
|
|
191
|
+
.map_err(|_| JsValue::from_str("merge: first item not Uint8Array"))?;
|
|
192
|
+
let mut doc = Document::load_mem(&first_buf.to_vec())
|
|
193
|
+
.map_err(|e| JsValue::from_str(&format!("PDF load error: {}", e)))?;
|
|
194
|
+
|
|
195
|
+
for i in 1..len {
|
|
196
|
+
let buf = pdf_buffers.get(i).dyn_into::<Uint8Array>()
|
|
197
|
+
.map_err(|_| JsValue::from_str("merge: item not Uint8Array"))?;
|
|
198
|
+
let other = Document::load_mem(&buf.to_vec())
|
|
199
|
+
.map_err(|e| JsValue::from_str(&format!("PDF load error: {}", e)))?;
|
|
200
|
+
|
|
201
|
+
let page_count = doc.get_pages().len() as i64;
|
|
202
|
+
|
|
203
|
+
for &page_obj_id in other.get_pages().values() {
|
|
204
|
+
let new_page_id = copy_page_into(&other, &mut doc, page_obj_id);
|
|
205
|
+
if let Ok(root_ref) = doc.trailer.get(b"Root").and_then(|r| r.as_reference()) {
|
|
206
|
+
if let Some(Object::Dictionary(ref mut root_dict)) = doc.objects.get_mut(&root_ref) {
|
|
207
|
+
if let Ok(pages_ref) = root_dict.get(b"Pages").and_then(|p| p.as_reference()) {
|
|
208
|
+
if let Some(Object::Dictionary(ref mut pages_dict)) = doc.objects.get_mut(&pages_ref) {
|
|
209
|
+
if let Ok(kids) = pages_dict.get_mut(b"Kids") {
|
|
210
|
+
if let Object::Array(ref mut kids_arr) = kids {
|
|
211
|
+
kids_arr.push(Object::Reference(new_page_id));
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
if let Ok(count) = pages_dict.get_mut(b"Count") {
|
|
215
|
+
*count = Object::Integer(page_count + 1);
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
let mut buf = Vec::new();
|
|
225
|
+
doc.save_to(&mut buf).map_err(|e| JsValue::from_str(&format!("PDF save error: {}", e)))?;
|
|
226
|
+
Ok(Uint8Array::new_from_slice(&buf))
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
#[wasm_bindgen]
|
|
230
|
+
pub fn split(buffer: &[u8], pages: Array) -> Result<Array, JsValue> {
|
|
231
|
+
let doc = Document::load_mem(buffer)
|
|
232
|
+
.map_err(|e| JsValue::from_str(&format!("PDF error: {}", e)))?;
|
|
233
|
+
let page_len = pages.length();
|
|
234
|
+
|
|
235
|
+
let page_map: BTreeMap<u32, ObjectId> = doc.get_pages();
|
|
236
|
+
let results = Array::new();
|
|
237
|
+
|
|
238
|
+
for i in 0..page_len {
|
|
239
|
+
let page_num = pages.get(i).as_f64().unwrap_or(0.0) as u32;
|
|
240
|
+
// PDF page numbers are 1-based internally
|
|
241
|
+
let Some(&page_obj_id) = page_map.get(&(page_num + 1)) else { continue };
|
|
242
|
+
|
|
243
|
+
let mut new_doc = Document::with_version("1.5");
|
|
244
|
+
let new_page_id = copy_page_into(&doc, &mut new_doc, page_obj_id);
|
|
245
|
+
|
|
246
|
+
// Build the page tree: Pages -> [page], Catalog -> Pages
|
|
247
|
+
let pages_id = (new_doc.max_id + 1, 0);
|
|
248
|
+
let catalog_id = (new_doc.max_id + 2, 0);
|
|
249
|
+
|
|
250
|
+
let pages_dict = lopdf::dictionary! {
|
|
251
|
+
"Type" => "Pages",
|
|
252
|
+
"Kids" => vec![Object::Reference(new_page_id)],
|
|
253
|
+
"Count" => 1
|
|
254
|
+
};
|
|
255
|
+
new_doc.objects.insert(pages_id, Object::Dictionary(pages_dict));
|
|
256
|
+
new_doc.max_id = pages_id.0.max(new_doc.max_id);
|
|
257
|
+
|
|
258
|
+
let catalog_dict = lopdf::dictionary! {
|
|
259
|
+
"Type" => "Catalog",
|
|
260
|
+
"Pages" => Object::Reference(pages_id)
|
|
261
|
+
};
|
|
262
|
+
new_doc.objects.insert(catalog_id, Object::Dictionary(catalog_dict));
|
|
263
|
+
new_doc.max_id = catalog_id.0.max(new_doc.max_id);
|
|
264
|
+
new_doc.trailer.set("Root", Object::Reference(catalog_id));
|
|
265
|
+
|
|
266
|
+
let mut pdf_bytes = Vec::new();
|
|
267
|
+
let _ = new_doc.save_to(&mut pdf_bytes);
|
|
268
|
+
results.push(&Uint8Array::new_from_slice(&pdf_bytes).into());
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
if results.length() == 0 {
|
|
272
|
+
results.push(&Uint8Array::new_from_slice(&[]).into());
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
Ok(results)
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
#[wasm_bindgen]
|
|
279
|
+
pub fn get_pdf_info(buffer: &[u8]) -> Result<String, JsValue> {
|
|
280
|
+
let count = get_page_count(buffer)?;
|
|
281
|
+
Ok(format!("PDF with {} pages", count))
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
#[cfg(test)]
|
|
285
|
+
mod tests {
|
|
286
|
+
use super::*;
|
|
287
|
+
|
|
288
|
+
#[test]
|
|
289
|
+
fn page_count_of_empty_doc_is_one() {
|
|
290
|
+
// ponytail: lopdf get_pages() returns empty map for docs with no pages
|
|
291
|
+
let doc = Document::with_version("1.5");
|
|
292
|
+
assert_eq!(doc.get_pages().len().max(1), 1);
|
|
293
|
+
}
|
|
294
|
+
}
|
package/LICENSE
DELETED
|
@@ -1,21 +0,0 @@
|
|
|
1
|
-
MIT License
|
|
2
|
-
|
|
3
|
-
Copyright (c) 2024 Jack Green
|
|
4
|
-
|
|
5
|
-
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
-
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
-
in the Software without restriction, including without limitation the rights
|
|
8
|
-
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
-
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
-
furnished to do so, subject to the following conditions:
|
|
11
|
-
|
|
12
|
-
The above copyright notice and this permission notice shall be included in all
|
|
13
|
-
copies or substantial portions of the Software.
|
|
14
|
-
|
|
15
|
-
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
-
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
-
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
-
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
-
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
-
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
-
SOFTWARE.
|