@social-mail/social-mail-web-server 1.8.39 → 1.8.41

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/common/BufferHelper.d.ts +4 -0
  2. package/dist/common/BufferHelper.d.ts.map +1 -0
  3. package/dist/common/BufferHelper.js +15 -0
  4. package/dist/common/BufferHelper.js.map +1 -0
  5. package/dist/isolated/extract/IExtract.d.ts +3 -0
  6. package/dist/isolated/extract/IExtract.d.ts.map +1 -1
  7. package/dist/isolated/extract/index.d.ts +4 -2
  8. package/dist/isolated/extract/index.d.ts.map +1 -1
  9. package/dist/isolated/extract/index.js +9 -15
  10. package/dist/isolated/extract/index.js.map +1 -1
  11. package/dist/server/seed/ui/seed-ui.js +1 -1
  12. package/dist/server/services/convert/Convert.d.ts.map +1 -1
  13. package/dist/server/services/convert/Convert.js +6 -3
  14. package/dist/server/services/convert/Convert.js.map +1 -1
  15. package/dist/server/services/extract/Extract.d.ts +1 -1
  16. package/dist/server/services/extract/Extract.d.ts.map +1 -1
  17. package/dist/server/services/extract/Extract.js +24 -21
  18. package/dist/server/services/extract/Extract.js.map +1 -1
  19. package/dist/server/services/extract/pdf/PdfDoc.d.ts +1 -1
  20. package/dist/server/services/extract/pdf/PdfDoc.d.ts.map +1 -1
  21. package/dist/server/services/extract/pdf/PdfDoc.js +4 -4
  22. package/dist/server/services/extract/pdf/PdfDoc.js.map +1 -1
  23. package/dist/server/workflows/search/IndexAppFilesWorkflow.d.ts.map +1 -1
  24. package/dist/server/workflows/search/IndexAppFilesWorkflow.js +30 -39
  25. package/dist/server/workflows/search/IndexAppFilesWorkflow.js.map +1 -1
  26. package/dist/tsconfig.tsbuildinfo +1 -1
  27. package/package.json +1 -1
  28. package/src/common/BufferHelper.ts +18 -0
  29. package/src/isolated/extract/IExtract.ts +3 -0
  30. package/src/isolated/extract/index.ts +9 -15
  31. package/src/server/seed/ui/seed-ui.ts +1 -1
  32. package/src/server/services/convert/Convert.ts +9 -3
  33. package/src/server/services/extract/Extract.ts +24 -22
  34. package/src/server/services/extract/pdf/PdfDoc.ts +4 -4
  35. package/src/server/workflows/search/IndexAppFilesWorkflow.ts +39 -42
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@social-mail/social-mail-web-server",
3
- "version": "1.8.39",
3
+ "version": "1.8.41",
4
4
  "description": "## Phase 1",
5
5
  "main": "index.js",
6
6
  "type": "module",
@@ -0,0 +1,18 @@
1
+ export default class BufferHelper {
2
+
3
+
4
+ static replace00(b: Buffer, replace: Buffer) {
5
+ let start = 0;
6
+ for(;;) {
7
+ const i = b.indexOf(0x0, start);
8
+ if (i === -1) {
9
+ break;
10
+ }
11
+ b.set(replace, i);
12
+ start = i + 1;
13
+ }
14
+ return b;
15
+ }
16
+
17
+
18
+ }
@@ -1,4 +1,7 @@
1
1
  export default interface IExtract {
2
2
  input: { contentType, fileName, path },
3
+ output: {
4
+ path: string,
5
+ },
3
6
  useOcr: boolean;
4
7
  }
@@ -1,36 +1,30 @@
1
1
  /* eslint-disable no-console */
2
- import { readFile } from "fs/promises";
2
+ import { copyFile, readFile, writeFile } from "fs/promises";
3
3
  import tesseract from "tesseract.js";
4
4
  import FileInfo from "../FileInfo.js";
5
5
  import IExtract from "./IExtract.js";
6
6
  const { recognize } = tesseract;
7
7
 
8
8
  export class ExtractWorker {
9
- static async text({ input: file, useOcr }: IExtract) {
9
+ static async text({ input: file, useOcr, output }: IExtract) {
10
10
  try {
11
11
  if (/^image\//.test(file.contentType)) {
12
12
  if (useOcr) {
13
- return this.ocr(file);
13
+ return this.ocr(file, output);
14
14
  }
15
15
  }
16
16
 
17
- return await readFile(file.path, "utf-8");
17
+ await copyFile(file.path, output.path);
18
+ return;
18
19
  } catch (error) {
19
20
  console.error(error);
20
21
  }
21
- return "";
22
-
23
22
  }
24
23
 
25
- static async ocr(file: FileInfo) {
26
- try {
27
- const buffer = await readFile(file.path);
28
- const result = await recognize(buffer);
29
- return result.data.text;
30
- } catch (error) {
31
- console.error(error);
32
- }
33
- return "";
24
+ static async ocr(file: FileInfo, output: { path: string}) {
25
+ const buffer = await readFile(file.path);
26
+ const result = await recognize(buffer);
27
+ await writeFile(output.path, result.data.text, "utf-8");
34
28
  }
35
29
  };
36
30
 
@@ -5,7 +5,7 @@ export default async function seedUI(config: DBConfig) {
5
5
  await config.saveVersion(UIPackageConfig, {
6
6
  package: "@social-mail/social-mail-client",
7
7
  view: "dist/web/AppIndex",
8
- version: "1.8.215"
8
+ version: "1.8.216"
9
9
  });
10
10
 
11
11
  await config.saveVersion(WebComponentsPackageConfig, {
@@ -40,10 +40,16 @@ export const Convert = {
40
40
  if (args[1] === "txt") {
41
41
  // txt hangs, so lets use to html and convert
42
42
  args[1] = "html";
43
+
43
44
  await this.execute(file, args, /\.html$/, output);
44
- const htmlText = await readFile(output, "utf-8");
45
- const { window } = new JSDOM(htmlText);
46
- const text = window.document.documentElement.innerText ?? window.document.body.innerText ?? htmlText;
45
+
46
+ let text = "";
47
+ if (file.contentSize < 10*1024*1024) {
48
+
49
+ const htmlText = await readFile(output, "utf-8");
50
+ const { window } = new JSDOM(htmlText);
51
+ text = window.document.documentElement.innerText ?? window.document.body.innerText ?? htmlText;
52
+ }
47
53
  await writeFile(output, text);
48
54
  return output;
49
55
  }
@@ -1,25 +1,20 @@
1
1
  /* eslint-disable no-console */
2
2
  import { LocalFile } from "@entity-access/server-pages/dist/core/LocalFile.js";
3
3
  import { FileType } from "../../../common/FileType.js";
4
- import { readFile } from "fs/promises";
4
+ import { copyFile, readFile } from "fs/promises";
5
5
  import { Convert } from "../convert/Convert.js";
6
6
  import LockFile from "../../storage/LockFile.js";
7
7
  import PdfDoc from "./pdf/PdfDoc.js";
8
8
  import IsolatedProcess from "../../../isolated/IsolatedProcess.js";
9
9
  import { tempDiskCache } from "../../storage/tempDiskCache.js";
10
- import { AppStringHelper } from "../../../common/AppStringHelper.js";
11
10
 
12
11
  export class Extract {
13
12
 
14
- static async text(src: LocalFile, useOcr = false) {
15
- const text = await this.internalText(src, useOcr);
16
- if (text) {
17
- return AppStringHelper.remove00(text);
18
- }
19
- return text;
13
+ static async text(src: LocalFile, outputFile: LocalFile, useOcr = false) {
14
+ await this.internalText(src, outputFile, useOcr);
20
15
  }
21
16
 
22
- private static async internalText(src: LocalFile, useOcr = false) {
17
+ private static async internalText(src: LocalFile, outputFile: LocalFile, useOcr = false) {
23
18
 
24
19
  if (FileType.isMedia(src)) {
25
20
  // currently we are not going to support
@@ -28,7 +23,8 @@ export class Extract {
28
23
  // in future might provide some sort of
29
24
  // caption extraction or title information
30
25
  // extraction
31
- return "";
26
+ await outputFile.writeAllText("");
27
+ return;
32
28
  }
33
29
 
34
30
  using folder = tempDiskCache.newFolder("to-text-"+ Date.now());
@@ -36,11 +32,13 @@ export class Extract {
36
32
  await src.copyTo(file);
37
33
 
38
34
  if (FileType.isPdf(file)) {
39
- return await PdfDoc.extract(file);
35
+ await PdfDoc.extract(file, outputFile);
36
+ return;
40
37
  }
41
38
 
42
39
  if (FileType.isCode(file)) {
43
- return await readFile(file.path, "utf-8");
40
+ await copyFile(file.path, outputFile.path);
41
+ return;
44
42
  }
45
43
 
46
44
  if (!FileType.isImage(file)) {
@@ -49,18 +47,22 @@ export class Extract {
49
47
  return "";
50
48
  }
51
49
 
52
-
53
- const output = folder.get(file.fileName + ".txt", "text/plain");
54
- await Convert.convert(file, "text", output.path);
55
- const { contentSize } = output;
56
- if (contentSize > 10*1024*1024) {
57
- throw new Error(`File size too big ${contentSize}`);
58
- }
59
- return await output.readAsText();
50
+ await Convert.convert(file, "text", outputFile.path);
51
+ return;
60
52
  }
61
53
 
62
- using lock = await LockFile.lock("extract-worker-lock");
54
+ using _lock = await LockFile.lock("extract-worker-lock");
63
55
 
64
- return await IsolatedProcess.extractText({ input: { fileName: file.fileName, contentType: file.contentType, path: file.path }, useOcr });
56
+ return await IsolatedProcess.extractText({
57
+ input: {
58
+ fileName: file.fileName,
59
+ contentType: file.contentType,
60
+ path: file.path
61
+ },
62
+ output: {
63
+ path: outputFile.path
64
+ },
65
+ useOcr
66
+ });
65
67
  }
66
68
  };
@@ -7,15 +7,15 @@ import { tempDiskCache } from "../../../storage/tempDiskCache.js";
7
7
 
8
8
  export default class PdfDoc {
9
9
 
10
- static async extract(file: LocalFile) {
11
- const output = await spawnPromise("pdftotext", [
10
+ static async extract(file: LocalFile, outputFile: LocalFile) {
11
+ await spawnPromise("pdftotext", [
12
12
  file.path,
13
- "-"
13
+ outputFile.path,
14
14
  ], {
15
15
  logData: false,
16
16
  timeout: 5*60*1000
17
17
  });
18
- return output.all;
18
+ return outputFile;
19
19
  }
20
20
 
21
21
  static async extractAsHtmlToFile(src: LocalFile, output: LocalFile) {
@@ -11,6 +11,8 @@ import EmailParserService from "../../smtp/services/EmailParserService.js";
11
11
  import { parse } from "path";
12
12
  import { Extract } from "../../services/extract/Extract.js";
13
13
  import { AppWorkflowContext } from "../AppWorkflowContext.js";
14
+ import { appendFile } from "fs/promises";
15
+ import BufferHelper from "../../../common/BufferHelper.js";
14
16
 
15
17
  export default class IndexAppFilesWorkflow extends Workflow {
16
18
 
@@ -73,80 +75,75 @@ export default class IndexAppFilesWorkflow extends Workflow {
73
75
 
74
76
  private async extractText(file: LocalFile, tempFileService: TempFileService, db: SocialMailContext) {
75
77
 
78
+
79
+ const outputFile = await tempFileService.createTempFile(".txt");
80
+
76
81
  const isEmail = file.contentType === "message/rfc822";
77
82
  if(isEmail) {
78
83
  const parser = ServiceProvider.resolve(this, EmailParserService);
79
84
  const parsedEmail = await parser.parse(file, true);
80
- const text = [parsedEmail.text];
85
+ await outputFile.appendLine(parsedEmail.text);
81
86
  for (const iterator of parsedEmail.attachments) {
82
87
  const { ext } = parse(iterator.filename || "file.dat");
83
88
  const tf = await tempFileService.createTempFile(ext, iterator.filename, iterator.contentType);
84
89
  await tf.writeAll(iterator.content);
85
90
  const t = await this.extractText(tf, tempFileService, db);
86
91
  if (t) {
87
- text.push(t);
92
+ await outputFile.appendLine(await t.readAsText());
88
93
  }
89
94
  }
90
- return text.join("\n");
95
+
96
+ return outputFile;
91
97
  }
92
98
 
93
99
  if (/image\//i.test(file.contentType)) {
94
- return Extract.text(file, true);
100
+ await Extract.text(file, outputFile, true);
101
+ return outputFile;
95
102
  } if (FileType.canBeCompressed(file)) {
96
- return Extract.text(file);
103
+ await Extract.text(file, outputFile);
104
+ return outputFile;
97
105
  }
98
-
99
- return "";
106
+ await outputFile.writeAllText("");
107
+ return outputFile;
100
108
  }
101
109
 
102
110
 
103
111
  private async saveTextContent(appFile: AppFile, tempFileService: TempFileService, db: SocialMailContext) {
104
- let searchable = "";
105
-
106
- // const isEmail = appFile.contentType === "message/rfc822";
107
-
108
- // if (isEmail) {
109
- // ServiceProvider.resolve(this, MailPar)
110
- // // this is email... it will contain the attachments..
111
- // // const email = await db.emails.where(appFile, (p) => (x) => x.originalFileID === p.appFileID)
112
- // // .include((x) => x.attachments)
113
- // // .first();
114
- // // if (email) {
115
- // // const text = [email.textBody];
116
- // // if (email.attachments) {
117
- // // for (const attachment of email.attachments) {
118
- // // text.push(attachment.name);
119
- // // const t = await this.saveTextContent(attachment, tempFileService, db);
120
- // // if (t) {
121
- // // text.push(t);
122
- // // }
123
- // // }
124
- // // }
125
- // // searchable = text.join("\n");
126
- // // }
127
- // }
128
112
 
129
113
  const changes: Partial<FileContent> = {
130
114
  textDocumentID: 0,
131
115
  fileContentID: appFile.fileContentID
132
116
  };
133
117
 
134
- if (!searchable) {
135
- const file = await tempFileService.downloadAppFile(appFile);
118
+ const file = await tempFileService.downloadAppFile(appFile);
136
119
 
137
- searchable = await this.extractText(file, tempFileService, db);
138
- }
139
- if (searchable) {
140
- const td = await db.textDocuments.statements.upsert({
141
- textDocumentID: changes.fileContentID,
142
- searchable
143
- });
144
- changes.textDocumentID = td.textDocumentID;
120
+ const fx = await this.extractText(file, tempFileService, db);
121
+
122
+ let searchable = "";
123
+
124
+ if (fx.contentSize < 10*1024*1024) {
125
+
126
+ // we need to first remove 00 byte sequence...
127
+ const newFile = await tempFileService.createTempFile("txt");
128
+ const empty = Buffer.from("\n");
129
+ for await (const buffer of fx.readBuffers()) {
130
+ BufferHelper.replace00(buffer, empty);
131
+ await appendFile(newFile.path, buffer);
132
+ }
133
+
134
+ searchable = await newFile.readAsText();
145
135
  }
136
+
137
+ const td = await db.textDocuments.statements.upsert({
138
+ textDocumentID: changes.fileContentID,
139
+ searchable
140
+ });
141
+ changes.textDocumentID = td.textDocumentID;
142
+
146
143
  await db.fileContents.statements.update(
147
144
  changes
148
145
  );
149
146
 
150
- return searchable;
147
+ return;
151
148
  }
152
149
  }