@social-mail/social-mail-web-server 1.8.39 → 1.8.41
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/common/BufferHelper.d.ts +4 -0
- package/dist/common/BufferHelper.d.ts.map +1 -0
- package/dist/common/BufferHelper.js +15 -0
- package/dist/common/BufferHelper.js.map +1 -0
- package/dist/isolated/extract/IExtract.d.ts +3 -0
- package/dist/isolated/extract/IExtract.d.ts.map +1 -1
- package/dist/isolated/extract/index.d.ts +4 -2
- package/dist/isolated/extract/index.d.ts.map +1 -1
- package/dist/isolated/extract/index.js +9 -15
- package/dist/isolated/extract/index.js.map +1 -1
- package/dist/server/seed/ui/seed-ui.js +1 -1
- package/dist/server/services/convert/Convert.d.ts.map +1 -1
- package/dist/server/services/convert/Convert.js +6 -3
- package/dist/server/services/convert/Convert.js.map +1 -1
- package/dist/server/services/extract/Extract.d.ts +1 -1
- package/dist/server/services/extract/Extract.d.ts.map +1 -1
- package/dist/server/services/extract/Extract.js +24 -21
- package/dist/server/services/extract/Extract.js.map +1 -1
- package/dist/server/services/extract/pdf/PdfDoc.d.ts +1 -1
- package/dist/server/services/extract/pdf/PdfDoc.d.ts.map +1 -1
- package/dist/server/services/extract/pdf/PdfDoc.js +4 -4
- package/dist/server/services/extract/pdf/PdfDoc.js.map +1 -1
- package/dist/server/workflows/search/IndexAppFilesWorkflow.d.ts.map +1 -1
- package/dist/server/workflows/search/IndexAppFilesWorkflow.js +30 -39
- package/dist/server/workflows/search/IndexAppFilesWorkflow.js.map +1 -1
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/package.json +1 -1
- package/src/common/BufferHelper.ts +18 -0
- package/src/isolated/extract/IExtract.ts +3 -0
- package/src/isolated/extract/index.ts +9 -15
- package/src/server/seed/ui/seed-ui.ts +1 -1
- package/src/server/services/convert/Convert.ts +9 -3
- package/src/server/services/extract/Extract.ts +24 -22
- package/src/server/services/extract/pdf/PdfDoc.ts +4 -4
- package/src/server/workflows/search/IndexAppFilesWorkflow.ts +39 -42
package/package.json
CHANGED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
export default class BufferHelper {
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
static replace00(b: Buffer, replace: Buffer) {
|
|
5
|
+
let start = 0;
|
|
6
|
+
for(;;) {
|
|
7
|
+
const i = b.indexOf(0x0, start);
|
|
8
|
+
if (i === -1) {
|
|
9
|
+
break;
|
|
10
|
+
}
|
|
11
|
+
b.set(replace, i);
|
|
12
|
+
start = i + 1;
|
|
13
|
+
}
|
|
14
|
+
return b;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
}
|
|
@@ -1,36 +1,30 @@
|
|
|
1
1
|
/* eslint-disable no-console */
|
|
2
|
-
import { readFile } from "fs/promises";
|
|
2
|
+
import { copyFile, readFile, writeFile } from "fs/promises";
|
|
3
3
|
import tesseract from "tesseract.js";
|
|
4
4
|
import FileInfo from "../FileInfo.js";
|
|
5
5
|
import IExtract from "./IExtract.js";
|
|
6
6
|
const { recognize } = tesseract;
|
|
7
7
|
|
|
8
8
|
export class ExtractWorker {
|
|
9
|
-
static async text({ input: file, useOcr }: IExtract) {
|
|
9
|
+
static async text({ input: file, useOcr, output }: IExtract) {
|
|
10
10
|
try {
|
|
11
11
|
if (/^image\//.test(file.contentType)) {
|
|
12
12
|
if (useOcr) {
|
|
13
|
-
return this.ocr(file);
|
|
13
|
+
return this.ocr(file, output);
|
|
14
14
|
}
|
|
15
15
|
}
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
await copyFile(file.path, output.path);
|
|
18
|
+
return;
|
|
18
19
|
} catch (error) {
|
|
19
20
|
console.error(error);
|
|
20
21
|
}
|
|
21
|
-
return "";
|
|
22
|
-
|
|
23
22
|
}
|
|
24
23
|
|
|
25
|
-
static async ocr(file: FileInfo) {
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
return result.data.text;
|
|
30
|
-
} catch (error) {
|
|
31
|
-
console.error(error);
|
|
32
|
-
}
|
|
33
|
-
return "";
|
|
24
|
+
static async ocr(file: FileInfo, output: { path: string}) {
|
|
25
|
+
const buffer = await readFile(file.path);
|
|
26
|
+
const result = await recognize(buffer);
|
|
27
|
+
await writeFile(output.path, result.data.text, "utf-8");
|
|
34
28
|
}
|
|
35
29
|
};
|
|
36
30
|
|
|
@@ -5,7 +5,7 @@ export default async function seedUI(config: DBConfig) {
|
|
|
5
5
|
await config.saveVersion(UIPackageConfig, {
|
|
6
6
|
package: "@social-mail/social-mail-client",
|
|
7
7
|
view: "dist/web/AppIndex",
|
|
8
|
-
version: "1.8.
|
|
8
|
+
version: "1.8.216"
|
|
9
9
|
});
|
|
10
10
|
|
|
11
11
|
await config.saveVersion(WebComponentsPackageConfig, {
|
|
@@ -40,10 +40,16 @@ export const Convert = {
|
|
|
40
40
|
if (args[1] === "txt") {
|
|
41
41
|
// txt hangs, so lets use to html and convert
|
|
42
42
|
args[1] = "html";
|
|
43
|
+
|
|
43
44
|
await this.execute(file, args, /\.html$/, output);
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
45
|
+
|
|
46
|
+
let text = "";
|
|
47
|
+
if (file.contentSize < 10*1024*1024) {
|
|
48
|
+
|
|
49
|
+
const htmlText = await readFile(output, "utf-8");
|
|
50
|
+
const { window } = new JSDOM(htmlText);
|
|
51
|
+
text = window.document.documentElement.innerText ?? window.document.body.innerText ?? htmlText;
|
|
52
|
+
}
|
|
47
53
|
await writeFile(output, text);
|
|
48
54
|
return output;
|
|
49
55
|
}
|
|
@@ -1,25 +1,20 @@
|
|
|
1
1
|
/* eslint-disable no-console */
|
|
2
2
|
import { LocalFile } from "@entity-access/server-pages/dist/core/LocalFile.js";
|
|
3
3
|
import { FileType } from "../../../common/FileType.js";
|
|
4
|
-
import { readFile } from "fs/promises";
|
|
4
|
+
import { copyFile, readFile } from "fs/promises";
|
|
5
5
|
import { Convert } from "../convert/Convert.js";
|
|
6
6
|
import LockFile from "../../storage/LockFile.js";
|
|
7
7
|
import PdfDoc from "./pdf/PdfDoc.js";
|
|
8
8
|
import IsolatedProcess from "../../../isolated/IsolatedProcess.js";
|
|
9
9
|
import { tempDiskCache } from "../../storage/tempDiskCache.js";
|
|
10
|
-
import { AppStringHelper } from "../../../common/AppStringHelper.js";
|
|
11
10
|
|
|
12
11
|
export class Extract {
|
|
13
12
|
|
|
14
|
-
static async text(src: LocalFile, useOcr = false) {
|
|
15
|
-
|
|
16
|
-
if (text) {
|
|
17
|
-
return AppStringHelper.remove00(text);
|
|
18
|
-
}
|
|
19
|
-
return text;
|
|
13
|
+
static async text(src: LocalFile, outputFile: LocalFile, useOcr = false) {
|
|
14
|
+
await this.internalText(src, outputFile, useOcr);
|
|
20
15
|
}
|
|
21
16
|
|
|
22
|
-
private static async internalText(src: LocalFile, useOcr = false) {
|
|
17
|
+
private static async internalText(src: LocalFile, outputFile: LocalFile, useOcr = false) {
|
|
23
18
|
|
|
24
19
|
if (FileType.isMedia(src)) {
|
|
25
20
|
// currently we are not going to support
|
|
@@ -28,7 +23,8 @@ export class Extract {
|
|
|
28
23
|
// in future might provide some sort of
|
|
29
24
|
// caption extraction or title information
|
|
30
25
|
// extraction
|
|
31
|
-
|
|
26
|
+
await outputFile.writeAllText("");
|
|
27
|
+
return;
|
|
32
28
|
}
|
|
33
29
|
|
|
34
30
|
using folder = tempDiskCache.newFolder("to-text-"+ Date.now());
|
|
@@ -36,11 +32,13 @@ export class Extract {
|
|
|
36
32
|
await src.copyTo(file);
|
|
37
33
|
|
|
38
34
|
if (FileType.isPdf(file)) {
|
|
39
|
-
|
|
35
|
+
await PdfDoc.extract(file, outputFile);
|
|
36
|
+
return;
|
|
40
37
|
}
|
|
41
38
|
|
|
42
39
|
if (FileType.isCode(file)) {
|
|
43
|
-
|
|
40
|
+
await copyFile(file.path, outputFile.path);
|
|
41
|
+
return;
|
|
44
42
|
}
|
|
45
43
|
|
|
46
44
|
if (!FileType.isImage(file)) {
|
|
@@ -49,18 +47,22 @@ export class Extract {
|
|
|
49
47
|
return "";
|
|
50
48
|
}
|
|
51
49
|
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
await Convert.convert(file, "text", output.path);
|
|
55
|
-
const { contentSize } = output;
|
|
56
|
-
if (contentSize > 10*1024*1024) {
|
|
57
|
-
throw new Error(`File size too big ${contentSize}`);
|
|
58
|
-
}
|
|
59
|
-
return await output.readAsText();
|
|
50
|
+
await Convert.convert(file, "text", outputFile.path);
|
|
51
|
+
return;
|
|
60
52
|
}
|
|
61
53
|
|
|
62
|
-
using
|
|
54
|
+
using _lock = await LockFile.lock("extract-worker-lock");
|
|
63
55
|
|
|
64
|
-
return await IsolatedProcess.extractText({
|
|
56
|
+
return await IsolatedProcess.extractText({
|
|
57
|
+
input: {
|
|
58
|
+
fileName: file.fileName,
|
|
59
|
+
contentType: file.contentType,
|
|
60
|
+
path: file.path
|
|
61
|
+
},
|
|
62
|
+
output: {
|
|
63
|
+
path: outputFile.path
|
|
64
|
+
},
|
|
65
|
+
useOcr
|
|
66
|
+
});
|
|
65
67
|
}
|
|
66
68
|
};
|
|
@@ -7,15 +7,15 @@ import { tempDiskCache } from "../../../storage/tempDiskCache.js";
|
|
|
7
7
|
|
|
8
8
|
export default class PdfDoc {
|
|
9
9
|
|
|
10
|
-
static async extract(file: LocalFile) {
|
|
11
|
-
|
|
10
|
+
static async extract(file: LocalFile, outputFile: LocalFile) {
|
|
11
|
+
await spawnPromise("pdftotext", [
|
|
12
12
|
file.path,
|
|
13
|
-
|
|
13
|
+
outputFile.path,
|
|
14
14
|
], {
|
|
15
15
|
logData: false,
|
|
16
16
|
timeout: 5*60*1000
|
|
17
17
|
});
|
|
18
|
-
return
|
|
18
|
+
return outputFile;
|
|
19
19
|
}
|
|
20
20
|
|
|
21
21
|
static async extractAsHtmlToFile(src: LocalFile, output: LocalFile) {
|
|
@@ -11,6 +11,8 @@ import EmailParserService from "../../smtp/services/EmailParserService.js";
|
|
|
11
11
|
import { parse } from "path";
|
|
12
12
|
import { Extract } from "../../services/extract/Extract.js";
|
|
13
13
|
import { AppWorkflowContext } from "../AppWorkflowContext.js";
|
|
14
|
+
import { appendFile } from "fs/promises";
|
|
15
|
+
import BufferHelper from "../../../common/BufferHelper.js";
|
|
14
16
|
|
|
15
17
|
export default class IndexAppFilesWorkflow extends Workflow {
|
|
16
18
|
|
|
@@ -73,80 +75,75 @@ export default class IndexAppFilesWorkflow extends Workflow {
|
|
|
73
75
|
|
|
74
76
|
private async extractText(file: LocalFile, tempFileService: TempFileService, db: SocialMailContext) {
|
|
75
77
|
|
|
78
|
+
|
|
79
|
+
const outputFile = await tempFileService.createTempFile(".txt");
|
|
80
|
+
|
|
76
81
|
const isEmail = file.contentType === "message/rfc822";
|
|
77
82
|
if(isEmail) {
|
|
78
83
|
const parser = ServiceProvider.resolve(this, EmailParserService);
|
|
79
84
|
const parsedEmail = await parser.parse(file, true);
|
|
80
|
-
|
|
85
|
+
await outputFile.appendLine(parsedEmail.text);
|
|
81
86
|
for (const iterator of parsedEmail.attachments) {
|
|
82
87
|
const { ext } = parse(iterator.filename || "file.dat");
|
|
83
88
|
const tf = await tempFileService.createTempFile(ext, iterator.filename, iterator.contentType);
|
|
84
89
|
await tf.writeAll(iterator.content);
|
|
85
90
|
const t = await this.extractText(tf, tempFileService, db);
|
|
86
91
|
if (t) {
|
|
87
|
-
|
|
92
|
+
await outputFile.appendLine(await t.readAsText());
|
|
88
93
|
}
|
|
89
94
|
}
|
|
90
|
-
|
|
95
|
+
|
|
96
|
+
return outputFile;
|
|
91
97
|
}
|
|
92
98
|
|
|
93
99
|
if (/image\//i.test(file.contentType)) {
|
|
94
|
-
|
|
100
|
+
await Extract.text(file, outputFile, true);
|
|
101
|
+
return outputFile;
|
|
95
102
|
} if (FileType.canBeCompressed(file)) {
|
|
96
|
-
|
|
103
|
+
await Extract.text(file, outputFile);
|
|
104
|
+
return outputFile;
|
|
97
105
|
}
|
|
98
|
-
|
|
99
|
-
return
|
|
106
|
+
await outputFile.writeAllText("");
|
|
107
|
+
return outputFile;
|
|
100
108
|
}
|
|
101
109
|
|
|
102
110
|
|
|
103
111
|
private async saveTextContent(appFile: AppFile, tempFileService: TempFileService, db: SocialMailContext) {
|
|
104
|
-
let searchable = "";
|
|
105
|
-
|
|
106
|
-
// const isEmail = appFile.contentType === "message/rfc822";
|
|
107
|
-
|
|
108
|
-
// if (isEmail) {
|
|
109
|
-
// ServiceProvider.resolve(this, MailPar)
|
|
110
|
-
// // this is email... it will contain the attachments..
|
|
111
|
-
// // const email = await db.emails.where(appFile, (p) => (x) => x.originalFileID === p.appFileID)
|
|
112
|
-
// // .include((x) => x.attachments)
|
|
113
|
-
// // .first();
|
|
114
|
-
// // if (email) {
|
|
115
|
-
// // const text = [email.textBody];
|
|
116
|
-
// // if (email.attachments) {
|
|
117
|
-
// // for (const attachment of email.attachments) {
|
|
118
|
-
// // text.push(attachment.name);
|
|
119
|
-
// // const t = await this.saveTextContent(attachment, tempFileService, db);
|
|
120
|
-
// // if (t) {
|
|
121
|
-
// // text.push(t);
|
|
122
|
-
// // }
|
|
123
|
-
// // }
|
|
124
|
-
// // }
|
|
125
|
-
// // searchable = text.join("\n");
|
|
126
|
-
// // }
|
|
127
|
-
// }
|
|
128
112
|
|
|
129
113
|
const changes: Partial<FileContent> = {
|
|
130
114
|
textDocumentID: 0,
|
|
131
115
|
fileContentID: appFile.fileContentID
|
|
132
116
|
};
|
|
133
117
|
|
|
134
|
-
|
|
135
|
-
const file = await tempFileService.downloadAppFile(appFile);
|
|
118
|
+
const file = await tempFileService.downloadAppFile(appFile);
|
|
136
119
|
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
120
|
+
const fx = await this.extractText(file, tempFileService, db);
|
|
121
|
+
|
|
122
|
+
let searchable = "";
|
|
123
|
+
|
|
124
|
+
if (fx.contentSize < 10*1024*1024) {
|
|
125
|
+
|
|
126
|
+
// we need to first remove 00 byte sequence...
|
|
127
|
+
const newFile = await tempFileService.createTempFile("txt");
|
|
128
|
+
const empty = Buffer.from("\n");
|
|
129
|
+
for await (const buffer of fx.readBuffers()) {
|
|
130
|
+
BufferHelper.replace00(buffer, empty);
|
|
131
|
+
await appendFile(newFile.path, buffer);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
searchable = await newFile.readAsText();
|
|
145
135
|
}
|
|
136
|
+
|
|
137
|
+
const td = await db.textDocuments.statements.upsert({
|
|
138
|
+
textDocumentID: changes.fileContentID,
|
|
139
|
+
searchable
|
|
140
|
+
});
|
|
141
|
+
changes.textDocumentID = td.textDocumentID;
|
|
142
|
+
|
|
146
143
|
await db.fileContents.statements.update(
|
|
147
144
|
changes
|
|
148
145
|
);
|
|
149
146
|
|
|
150
|
-
return
|
|
147
|
+
return;
|
|
151
148
|
}
|
|
152
149
|
}
|