officeparser 3.1.4 → 3.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -12,6 +12,7 @@ A Node.js library to parse text out of any office file.
12
12
 
13
13
 
14
14
  #### Update
15
+ * 2023/04/07 - Added typings to methods to help with Typescript projects.
15
16
  * 2022/12/28 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
16
17
  * 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling.
17
18
  * 2021/11/21 - Added promise way to existing callback functions.
package/officeParser.js CHANGED
@@ -9,11 +9,11 @@ const rimraf = require('rimraf');
9
9
  const ERRORHEADER = "[OfficeParser]: ";
10
10
  /** Error messages */
11
11
  const ERRORMSG = {
12
- extensionUnsupported: (ext) => `${ERRORHEADER}Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
13
- fileCorrupted: (filename) => `${ERRORHEADER}Your file ${filename} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
14
- fileDoesNotExist: (filename) => `${ERRORHEADER}File ${filename} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
15
- locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
16
- improperArguments: `${ERRORHEADER}Improper arguments`
12
+ extensionUnsupported: (ext) => `${ERRORHEADER}Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
13
+ fileCorrupted: (filename) => `${ERRORHEADER}Your file ${filename} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
14
+ fileDoesNotExist: (filename) => `${ERRORHEADER}File ${filename} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
15
+ locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
16
+ improperArguments: `${ERRORHEADER}Improper arguments`
17
17
  }
18
18
  /** Default sublocation for decompressing files under the current directory. */
19
19
  const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
@@ -22,13 +22,20 @@ let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
22
22
  /** Flag to output errors to console other than normal error handling. Default is false as we anyway push the message for error handling. */
23
23
  let outputErrorToConsole = false;
24
24
 
25
- /** Console error if allowed */
25
+ /** Console error if allowed
26
+ * @param {string} errorMessage Error message to show on the console
27
+ * @returns {void}
28
+ */
26
29
  function consoleError(errorMessage) {
27
30
  if (outputErrorToConsole)
28
31
  console.error(errorMessage);
29
32
  }
30
33
 
31
- /** Custom parseString promise as the native has bugs */
34
+ /** Custom parseString promise as the native has bugs
35
+ * @param {string} xml The xml string from the doc file
36
+ * @param {boolean} [ignoreAttrs=true] Optional: Ignore attributes part of xml and focus only on the content.
37
+ * @returns {Promise<string>}
38
+ */
32
39
  const parseStringPromise = (xml, ignoreAttrs = true) => new Promise((resolve, reject) => {
33
40
  xml2js.parseString(xml, { "ignoreAttrs": ignoreAttrs }, (err, result) => {
34
41
  if (err)
@@ -38,9 +45,12 @@ const parseStringPromise = (xml, ignoreAttrs = true) => new Promise((resolve, re
38
45
  });
39
46
 
40
47
 
41
-
42
-
43
- /** Main function for parsing text from word files */
48
+ /** Main function for parsing text from word files
49
+ * @param {string} filename File path
50
+ * @param {function} callback Callback function that returns value or error
51
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
52
+ * @returns {void}
53
+ */
44
54
  function parseWord(filename, callback, deleteOfficeDist = true) {
45
55
  if (!fs.existsSync(filename)) {
46
56
  consoleError(ERRORMSG.fileDoesNotExist(filename));
@@ -113,7 +123,12 @@ function parseWord(filename, callback, deleteOfficeDist = true) {
113
123
  });
114
124
  }
115
125
 
116
- /** Main function for parsing text from PowerPoint files */
126
+ /** Main function for parsing text from PowerPoint files
127
+ * @param {string} filename File path
128
+ * @param {function} callback Callback function that returns value or error
129
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
130
+ * @returns {void}
131
+ */
117
132
  function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
118
133
  if (!fs.existsSync(filename)) {
119
134
  consoleError(ERRORMSG.fileDoesNotExist(filename));
@@ -196,7 +211,12 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
196
211
  });
197
212
  }
198
213
 
199
- /** Main function for parsing text from Excel files */
214
+ /** Main function for parsing text from Excel files
215
+ * @param {string} filename File path
216
+ * @param {function} callback Callback function that returns value or error
217
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
218
+ * @returns {void}
219
+ */
200
220
  function parseExcel(filename, callback, deleteOfficeDist = true) {
201
221
  if (!fs.existsSync(filename)) {
202
222
  consoleError(ERRORMSG.fileDoesNotExist(filename));
@@ -335,7 +355,12 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
335
355
  }
336
356
 
337
357
 
338
- /** Main function for parsing text from open office files */
358
+ /** Main function for parsing text from open office files
359
+ * @param {string} filename File path
360
+ * @param {function} callback Callback function that returns value or error
361
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
362
+ * @returns {void}
363
+ */
339
364
  function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
340
365
  if (!fs.existsSync(filename)) {
341
366
  consoleError(ERRORMSG.fileDoesNotExist(filename));
@@ -408,7 +433,12 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
408
433
  }
409
434
 
410
435
 
411
- /** Main async function with callback to execute parseOffice for supported files */
436
+ /** Main async function with callback to execute parseOffice for supported files
437
+ * @param {string} filename File path
438
+ * @param {function} callback Callback function that returns value or error
439
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
440
+ * @returns {void}
441
+ */
412
442
  function parseOffice(filename, callback, deleteOfficeDist = true) {
413
443
  if (!fs.existsSync(filename)) {
414
444
  consoleError(ERRORMSG.fileDoesNotExist(filename));
@@ -442,6 +472,7 @@ function parseOffice(filename, callback, deleteOfficeDist = true) {
442
472
  /**
443
473
  * Set decompression directory. The final decompressed data will be put inside officeDist folder within your directory
444
474
  * @param {string} newLocation Relative path to the directory that will contain officeDist folder with decompressed data
475
+ * @returns {void}
445
476
  */
446
477
  function setDecompressionLocation(newLocation) {
447
478
  if (newLocation != undefined) {
@@ -455,18 +486,28 @@ function setDecompressionLocation(newLocation) {
455
486
  decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
456
487
  }
457
488
 
458
- /** Enable console output */
489
+ /** Enable console output
490
+ * @returns {void}
491
+ */
459
492
  function enableConsoleOutput() {
460
493
  outputErrorToConsole = true;
461
494
  }
462
495
 
463
- /** Disabled console output */
496
+ /** Disabled console output
497
+ * @returns {void}
498
+ */
464
499
  function disableConsoleOutput() {
465
500
  outputErrorToConsole = false;
466
501
  }
467
502
 
468
503
 
469
504
  // #region Promise versions of above functions
505
+
506
+ /** Async function that can be used with await to execute parseWord. Or it can be used with promises.
507
+ * @param {string} filename File path
508
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
509
+ * @returns {Promise<string>}
510
+ */
470
511
  var parseWordAsync = function (filename, deleteOfficeDist = true) {
471
512
  return new Promise((resolve, reject) => {
472
513
  try {
@@ -482,6 +523,11 @@ var parseWordAsync = function (filename, deleteOfficeDist = true) {
482
523
  })
483
524
  }
484
525
 
526
+ /** Async function that can be used with await to execute parsePowerPoint. Or it can be used with promises.
527
+ * @param {string} filename File path
528
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
529
+ * @returns {Promise<string>}
530
+ */
485
531
  var parsePowerPointAsync = function (filename, deleteOfficeDist = true) {
486
532
  return new Promise((resolve, reject) => {
487
533
  try {
@@ -489,7 +535,7 @@ var parsePowerPointAsync = function (filename, deleteOfficeDist = true) {
489
535
  if (err)
490
536
  return reject(err);
491
537
  return resolve(data);
492
- },deleteOfficeDist);
538
+ }, deleteOfficeDist);
493
539
  }
494
540
  catch (error) {
495
541
  return reject(error);
@@ -497,6 +543,11 @@ var parsePowerPointAsync = function (filename, deleteOfficeDist = true) {
497
543
  })
498
544
  }
499
545
 
546
+ /** Async function that can be used with await to execute parseExcel. Or it can be used with promises.
547
+ * @param {string} filename File path
548
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
549
+ * @returns {Promise<string>}
550
+ */
500
551
  var parseExcelAsync = function (filename, deleteOfficeDist = true) {
501
552
  return new Promise((resolve, reject) => {
502
553
  try {
@@ -504,7 +555,7 @@ var parseExcelAsync = function (filename, deleteOfficeDist = true) {
504
555
  if (err)
505
556
  return reject(err);
506
557
  return resolve(data);
507
- },deleteOfficeDist);
558
+ }, deleteOfficeDist);
508
559
  }
509
560
  catch (error) {
510
561
  return reject(error);
@@ -512,6 +563,11 @@ var parseExcelAsync = function (filename, deleteOfficeDist = true) {
512
563
  })
513
564
  }
514
565
 
566
+ /** Async function that can be used with await to execute parseOpenOffice. Or it can be used with promises.
567
+ * @param {string} filename File path
568
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
569
+ * @returns {Promise<string>}
570
+ */
515
571
  var parseOpenOfficeAsync = function (filename, deleteOfficeDist = true) {
516
572
  return new Promise((resolve, reject) => {
517
573
  try {
@@ -519,7 +575,7 @@ var parseOpenOfficeAsync = function (filename, deleteOfficeDist = true) {
519
575
  if (err)
520
576
  return reject(err);
521
577
  return resolve(data);
522
- },deleteOfficeDist);
578
+ }, deleteOfficeDist);
523
579
  }
524
580
  catch (error) {
525
581
  return reject(error);
@@ -527,6 +583,12 @@ var parseOpenOfficeAsync = function (filename, deleteOfficeDist = true) {
527
583
  })
528
584
  }
529
585
 
586
+ /**
587
+ * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
588
+ * @param {string} filename File path
589
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
590
+ * @returns {Promise<string>}
591
+ */
530
592
  var parseOfficeAsync = function (filename, deleteOfficeDist = true) {
531
593
  return new Promise((resolve, reject) => {
532
594
  try {
@@ -534,7 +596,7 @@ var parseOfficeAsync = function (filename, deleteOfficeDist = true) {
534
596
  if (err)
535
597
  return reject(err);
536
598
  return resolve(data);
537
- },deleteOfficeDist);
599
+ }, deleteOfficeDist);
538
600
  }
539
601
  catch (error) {
540
602
  return reject(error);
@@ -557,6 +619,8 @@ module.exports.setDecompressionLocation = setDecompressionLocation;
557
619
  module.exports.enableConsoleOutput = enableConsoleOutput;
558
620
  module.exports.disableConsoleOutput = disableConsoleOutput;
559
621
 
622
+
623
+ // Run this library on CLI
560
624
  if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
561
625
  if (process.argv.length == 2) {
562
626
  // continue
package/package.json CHANGED
@@ -1,8 +1,13 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "3.1.4",
3
+ "version": "3.2.1",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp and ods files.",
5
5
  "main": "officeParser.js",
6
+ "files": [
7
+ "officeParser.js",
8
+ "typings/officeParser.d.ts"
9
+ ],
10
+ "types": "typings/officeParser.d.ts",
6
11
  "scripts": {
7
12
  "test": "node test/testOfficeParser.js"
8
13
  },
@@ -39,5 +44,8 @@
39
44
  "decompress": "^4.2.0",
40
45
  "rimraf": "^2.6.3",
41
46
  "xml2js": "^0.4.19"
47
+ },
48
+ "devDependencies": {
49
+ "typescript": "^5.0.3"
42
50
  }
43
- }
51
+ }
@@ -0,0 +1,82 @@
1
+ #!/usr/bin/env node
2
+ /** Main function for parsing text from word files
3
+ * @param {string} filename File path
4
+ * @param {function} callback Callback function that returns value or error
5
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
6
+ * @returns {void}
7
+ */
8
+ export function parseWord(filename: string, callback: Function, deleteOfficeDist?: boolean): void;
9
+ /** Main function for parsing text from PowerPoint files
10
+ * @param {string} filename File path
11
+ * @param {function} callback Callback function that returns value or error
12
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
13
+ * @returns {void}
14
+ */
15
+ export function parsePowerPoint(filename: string, callback: Function, deleteOfficeDist?: boolean): void;
16
+ /** Main function for parsing text from Excel files
17
+ * @param {string} filename File path
18
+ * @param {function} callback Callback function that returns value or error
19
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
20
+ * @returns {void}
21
+ */
22
+ export function parseExcel(filename: string, callback: Function, deleteOfficeDist?: boolean): void;
23
+ /** Main function for parsing text from open office files
24
+ * @param {string} filename File path
25
+ * @param {function} callback Callback function that returns value or error
26
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
27
+ * @returns {void}
28
+ */
29
+ export function parseOpenOffice(filename: string, callback: Function, deleteOfficeDist?: boolean): void;
30
+ /** Main async function with callback to execute parseOffice for supported files
31
+ * @param {string} filename File path
32
+ * @param {function} callback Callback function that returns value or error
33
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
34
+ * @returns {void}
35
+ */
36
+ export function parseOffice(filename: string, callback: Function, deleteOfficeDist?: boolean): void;
37
+ /** Async function that can be used with await to execute parseWord. Or it can be used with promises.
38
+ * @param {string} filename File path
39
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
40
+ * @returns {Promise<string>}
41
+ */
42
+ export function parseWordAsync(filename: string, deleteOfficeDist?: boolean): Promise<string>;
43
+ /** Async function that can be used with await to execute parsePowerPoint. Or it can be used with promises.
44
+ * @param {string} filename File path
45
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
46
+ * @returns {Promise<string>}
47
+ */
48
+ export function parsePowerPointAsync(filename: string, deleteOfficeDist?: boolean): Promise<string>;
49
+ /** Async function that can be used with await to execute parseExcel. Or it can be used with promises.
50
+ * @param {string} filename File path
51
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
52
+ * @returns {Promise<string>}
53
+ */
54
+ export function parseExcelAsync(filename: string, deleteOfficeDist?: boolean): Promise<string>;
55
+ /** Async function that can be used with await to execute parseOpenOffice. Or it can be used with promises.
56
+ * @param {string} filename File path
57
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
58
+ * @returns {Promise<string>}
59
+ */
60
+ export function parseOpenOfficeAsync(filename: string, deleteOfficeDist?: boolean): Promise<string>;
61
+ /**
62
+ * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
63
+ * @param {string} filename File path
64
+ * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
65
+ * @returns {Promise<string>}
66
+ */
67
+ export function parseOfficeAsync(filename: string, deleteOfficeDist?: boolean): Promise<string>;
68
+ /**
69
+ * Set decompression directory. The final decompressed data will be put inside officeDist folder within your directory
70
+ * @param {string} newLocation Relative path to the directory that will contain officeDist folder with decompressed data
71
+ * @returns {void}
72
+ */
73
+ export function setDecompressionLocation(newLocation: string): void;
74
+ /** Enable console output
75
+ * @returns {void}
76
+ */
77
+ export function enableConsoleOutput(): void;
78
+ /** Disabled console output
79
+ * @returns {void}
80
+ */
81
+ export function disableConsoleOutput(): void;
82
+ //# sourceMappingURL=officeParser.d.ts.map
@@ -1,76 +0,0 @@
1
- # Contributor Covenant Code of Conduct
2
-
3
- ## Our Pledge
4
-
5
- In the interest of fostering an open and welcoming environment, we as
6
- contributors and maintainers pledge to making participation in our project and
7
- our community a harassment-free experience for everyone, regardless of age, body
8
- size, disability, ethnicity, sex characteristics, gender identity and expression,
9
- level of experience, education, socio-economic status, nationality, personal
10
- appearance, race, religion, or sexual identity and orientation.
11
-
12
- ## Our Standards
13
-
14
- Examples of behavior that contributes to creating a positive environment
15
- include:
16
-
17
- * Using welcoming and inclusive language
18
- * Being respectful of differing viewpoints and experiences
19
- * Gracefully accepting constructive criticism
20
- * Focusing on what is best for the community
21
- * Showing empathy towards other community members
22
-
23
- Examples of unacceptable behavior by participants include:
24
-
25
- * The use of sexualized language or imagery and unwelcome sexual attention or
26
- advances
27
- * Trolling, insulting/derogatory comments, and personal or political attacks
28
- * Public or private harassment
29
- * Publishing others' private information, such as a physical or electronic
30
- address, without explicit permission
31
- * Other conduct which could reasonably be considered inappropriate in a
32
- professional setting
33
-
34
- ## Our Responsibilities
35
-
36
- Project maintainers are responsible for clarifying the standards of acceptable
37
- behavior and are expected to take appropriate and fair corrective action in
38
- response to any instances of unacceptable behavior.
39
-
40
- Project maintainers have the right and responsibility to remove, edit, or
41
- reject comments, commits, code, wiki edits, issues, and other contributions
42
- that are not aligned to this Code of Conduct, or to ban temporarily or
43
- permanently any contributor for other behaviors that they deem inappropriate,
44
- threatening, offensive, or harmful.
45
-
46
- ## Scope
47
-
48
- This Code of Conduct applies both within project spaces and in public spaces
49
- when an individual is representing the project or its community. Examples of
50
- representing a project or community include using an official project e-mail
51
- address, posting via an official social media account, or acting as an appointed
52
- representative at an online or offline event. Representation of a project may be
53
- further defined and clarified by project maintainers.
54
-
55
- ## Enforcement
56
-
57
- Instances of abusive, harassing, or otherwise unacceptable behavior may be
58
- reported by contacting the project team at harshankur@outlook.com. All
59
- complaints will be reviewed and investigated and will result in a response that
60
- is deemed necessary and appropriate to the circumstances. The project team is
61
- obligated to maintain confidentiality with regard to the reporter of an incident.
62
- Further details of specific enforcement policies may be posted separately.
63
-
64
- Project maintainers who do not follow or enforce the Code of Conduct in good
65
- faith may face temporary or permanent repercussions as determined by other
66
- members of the project's leadership.
67
-
68
- ## Attribution
69
-
70
- This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4,
71
- available at https://www.contributor-covenant.org/version/1/4/code-of-conduct.html
72
-
73
- [homepage]: https://www.contributor-covenant.org
74
-
75
- For answers to common questions about this code of conduct, see
76
- https://www.contributor-covenant.org/faq
@@ -1,10 +0,0 @@
1
- const supportedExtensions = [
2
- "docx",
3
- "pptx",
4
- "xlsx",
5
- "odt",
6
- "odp",
7
- "ods",
8
- ];
9
-
10
- module.exports = supportedExtensions;
@@ -1,27 +0,0 @@
1
- const officeParser = require("../officeParser");
2
- const fs = require("fs");
3
- const supportedExtensions = require("../supportedExtensions");
4
-
5
- // File names of test files and their text output content
6
- // test file name style => test.<ext>
7
- // test content output => test.<ext>.txt
8
-
9
- /** Get filename for an extension */
10
- function getFilename(ext, isContentFile = false) {
11
- return `test/files/test.${ext}` + (isContentFile ? `.txt` : '');
12
- }
13
-
14
- /** Create content file the test file with passed extension */
15
- function createContentFile(ext) {
16
- return officeParser.parseOfficeAsync(getFilename(ext))
17
- .then(text => fs.writeFileSync(getFilename(ext, true), text, 'utf8'))
18
- }
19
-
20
-
21
- process.argv.length == 3
22
- ? supportedExtensions.includes(process.argv[2])
23
- ? createContentFile(process.argv[2])
24
- .then(() => console.log(`Created text content file for ${process.argv[2]} => ${getFilename(process.argv[2], true)}`))
25
- .catch((error) => console.error(error))
26
- : console.error("The requested extension test is not currently available.")
27
- : console.error("Arguments missing");
Binary file
@@ -1 +0,0 @@
1
- Demonstration of DOCX support in calibre This document demonstrates the ability of the calibre DOCX Input plugin to convert the various typographic features in a Microsoft Word (2007 and newer) document. Convert this document to a modern ebook format, such as AZW3 for Kindles or EPUB for other ebook readers, to see it in action. There is support for images, tables, lists, footnotes, endnotes, links, dropcaps and various types of text and paragraph level formatting. To see the DOCX conversion in action, simply add this file to calibre using the “Add Books” button and then click “ Convert”. Set the output format in the top right corner of the conversion dialog to EPUB or AZW3 and click “OK” . Text Formatting Inline formatting Here, we demonstrate various types of inline text formatting and the use of embedded fonts. Here is some bold, italic, bold-italic, underlined and struck out text. Then, we have a super script and a sub script . Now we see some red , green and blue text. Some text with a yellow highlight . Some text in a box . Some text in inverse video . A paragraph with styled text: subtle emphasis f ollowed by strong text a nd intense emphasis . This paragraph uses document wide styles for styling rather than inline text properties as demonstrated in the previous paragraph — calibre can handle both with equal ease. Fun with fonts This document has embedded the Ubuntu font family. The body text is in the Ubuntu typeface, here is some text in the Ubuntu Mono typeface , notice how every letter has the same width, even i and m . Every embedded font will automatically be embedded in the output ebook during conversion. Paragraph level formatting You can do crazy things with paragraphs, if the urge strikes you. For instance this paragraph is right aligned and has a right border. It has also been given a light gray background. For the lovers of poetry amongst you, paragraphs with hanging indents, like this often come in handy. You can use hanging indents to ensure that a line of poetry retains its individual identity as a line even when the screen is too narrow to display it as a single line. Not only does this paragraph have a hanging indent, it is also has an extra top margin, setting it apart from the preceding paragraph. Tables Tables in Word can vary from the extremely simple to the extremely complex. calibre tries to do its best when converting tables. While you may run into trouble with the occasional table, the vast majority of common cases should be converted very well, as demonstrated in this section. Note that for optimum results, when creating tables in Word, you should set their widths using percentages, rather than absolute units. To the left of this paragraph is a floating two column table with a nice green border and header row. Now let’s look at a fancier table—one with alternating row colors and partial borders. This table is stretched out to take 100% of the available width. Next, we see a table with special formatting in various locations. Notice how the f ormatting for the h eader row and sub header rows is preserved. Source: Fictitious data, for illustration purposes only Next , we have something a little more complex, a nested table, i.e. a table inside another table. Additionally, the inner table has some of its cells merged. The table is displayed horizontally centered. W e end with a fancy calendar, note how much of the original formatting is preserved . Note that this table will only display correctly on relatively wide screens. In general, very wide tables or tables whose cells have fixed width requirements don’t fare well in ebooks. Structural Elements Miscellaneous structural elements you can add to your document, like footnotes, endnotes, dropcaps and the like. Footnotes & Endnotes Footnotes and endnotes are automatically recognized and both are converted to endnotes, with backlinks for maximum ease of use in ebook devices. Dropcaps D rop caps are used to emphasize the leading paragraph at the start of a section. In Word it is possible to s p ecify how many lines of text a drop-cap should use. Because of limitations in ebook technology, this is not possible when converting. Instead, the converted drop cap will use font size and line height to simulate the effect as well as possible. While not as good as the original, the result is usually tolerable. This paragraph has a “D” dropcap set to occupy three lines of text with a font size of 58.5 pts. Depending on the screen width and capabilities of the device you view the book on, this dropcap can look anything from perfect to ugly. Links Two kinds of links are possible, those that refer to an external website and those that refer to locations inside the document itself. Both are supported by calibre. For example, here is a link pointing to the . Then we have a link that points back to the section on in this document. calibre download page paragraph level formatting Table of Contents There are two approaches that calibre takes when generating a Table of Contents. The first is if the Word document has a Table of Contents itself. Provided that the Table of Contents uses hyperlinks, calibre will automatically use it. The levels of the Table of Contents are identified by their left indent, so if you want the ebook to have a multi-level Table of Contents, make sure you create a properly indented Table of Contents in Word. If no Table of Contents is found in the document, then a table of contents is automatically generated from the headings in the document. A heading is identified as something that has the Heading 1 or Heading 2, etc. style applied to it. These headings are turned into a Table of Contents with Heading 1 being the topmost level, Heading 2 the second level and so on. You can see the Table of Contents created by calibre by clicking the Table of Contents button in whatever viewer you are using to view the converted ebook. Demonstration of DOCX support in calibre 1 Text Formatting 2 Inline formatting 2 Fun with fonts 2 Paragraph level formatting 2 Tables 3 Structural Elements 5 Footnotes & Endnotes 5 Dropcaps 5 Links 5 Table of Contents 5 Images 7 Lists 8 Bulleted List 8 Numbered List 8 Multi-level Lists 8 Continued Lists 8 Images Images can be of three main types. Inline images are images that are part of the normal text flow, like this image of a green dot . Inline images do not cause breaks in the text and are usually small in size. The next category of image is a floating image, one that “floats “ on the page and is surrounded by text. Word supports more types of floating images than are possible with current ebook technology, so the conversion maps floating images to simple left and right floats, as you can see with the left and right arrow images on the sides of this paragraph. The final type of image is a “block” image, one that becomes a paragraph on its own and has no text on either side. Below is a centered green dot. Centered images like this are useful for large pictures that should be a focus of attention. Generally, it is not possible to translate the exact positioning of images from a Word document to an ebook. That is because in Word, image positioning is specified in absolute units from the page boundaries. There is no analogous technology in ebooks, so the conversion will usually end up placing the image either centered or floating close to the point in the text where it was inserted , not necessarily where it appears on the page in Word. Lists All types of lists are supported by the conversion, with the exception of lists that use fancy bullets, these get converted to regular bullets. Bulleted List One Two Numbered List One, with a very long line to demonstrate that the hanging indent for the list is working correctly Two Multi-level List s One Two Three Four with a very long line to demonstrate that the hanging indent for the list is working correctly. Five Six A Multi-level list with bullets: One Two This bullet uses an image as the bullet item Four Five Continued Lists One Two An interruption in our regularly scheduled listing, for this essential and very relevant public service announcement. We now resume our normal programming Four ITEM NEEDED Books 1 Pens 3 Pencils 2 Highlighter 2 colors Scissors 1 pair City or Town Point A Point B Point C Point D Point E Point A — Point B 87 — Point C 64 56 — Point D 37 32 91 — Point E 93 35 54 43 — College New students Graduating students Change Undergraduate Cedar University 110 103 +7 Oak Institute 202 210 -8 Graduate Cedar University 24 20 +4 Elm College 43 53 -10 Total 998 908 90 One Three Two Four To the left is a table inside a table, with some cells merged. December 2007 Sun Mon Tue Wed Thu Fri Sat 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31
Binary file
@@ -1 +0,0 @@
1
- Lorem ipsum Lorem ipsum dolor sit amet, consectetur adipiscing elit. Nunc ac faucibus odio. Vestibulum neque massa, scelerisque sit amet ligula eu, congue molestie mi. Praesent ut varius sem. Nullam at porttitor arcu, nec lacinia nisi. Ut ac dolor vitae odio interdum condimentum. Vivamus dapibus sodales ex, vitae malesuada ipsum cursus convallis. Maecenas sed egestas nulla, ac condimentum orci. Mauris diam felis, vulputate ac suscipit et, iaculis non est. Curabitur semper arcu ac ligula semper, nec luctus nisl blandit. Integer lacinia ante ac libero lobortis imperdiet. Nullam mollis convallis ipsum, ac accumsan nunc vehicula vitae. Nulla eget justo in felis tristique fringilla. Morbi sit amet tortor quis risus auctor condimentum. Morbi in ullamcorper elit. Nulla iaculis tellus sit amet mauris tempus fringilla. Maecenas mauris lectus, lobortis et purus mattis, blandit dictum tellus. Maecenas non lorem quis tellus placerat varius. Nulla facilisi. Aenean congue fringilla justo ut aliquam. Mauris id ex erat. Nunc vulputate neque vitae justo facilisis, non condimentum ante sagittis. Morbi viverra semper lorem nec molestie. Maecenas tincidunt est efficitur ligula euismod, sit amet ornare est vulputate. Chart Table Column 1 Column 2 Column 3 Column 4 Column 5 Photo
Binary file
@@ -1 +0,0 @@
1
- / / Hello Testing 0 First Name Last Name Gender Country Age Date Id 1 Dulce Abril Female United States 32 15/10/2017 1562 2 Mara Hashimoto Female Great Britain 25 16/08/2016 1582 3 Philip Gent Male France 36 21/05/2015 2587 4 Kathleen Hanner Female United States 25 15/10/2017 3549 5 Nereida Magwood Female United States 58 16/08/2016 2468 96 Roma Lafollette Female United States 34 15/10/2017 2654 97 Felisa Cail Female United States 28 16/08/2016 6525 98 Demetria Abbey Female United States 32 21/05/2015 3265 99 Jeromy Danz Male United States 39 15/10/2017 3265 100 Rasheeda Alkire Female United States 29 16/08/2016 6125
Binary file
@@ -1 +0,0 @@
1
- Text file Digital archaeologists find occasional gems in the vast European information repository. Taken from the mountain of uncatalogued ephemera, this text is written in a style suggesting composition in the early 21st century, possibly even on paper. Diary fragment 1: Dear Diary, Day 1 of our boating holiday on the Sûre in Luxembourg! Up at the crack of dawn to get to the departure point. Ten minutes to learn the ropes and then off we set in our canoes! There was not a breath of wind as we drifted downstream, sometimes passing an angler staring intently at his float, or a group of children paddling. When I commented that the signs on the two sides of the river looked like they were written in different languages, Dad grinned: “They’re two different countries – Germany on our left and Luxembourg on our right! We’re navigating pretty much along the border!” Analytical note 1: In this period, geographical information was provided in the form of metal panels or ‘signs’ mounted on poles. Numerous examples can now be found in museums. Diary fragment 2: I couldn’t get over that – I mean, if we want to go abroad we have to get on a plane or a ferry, but here you can cross to another country without even having to show a passport. The joy of Schengen! Analytical note 2: Schengen was an early, and highly significant, stage in the dismantling of European border barriers. European union was long and complex, encountering frequent obstacles. Indeed many predicted that it would not survive the Covid pandemic. Diary fragment 3: After last night’s hotel (really scraping the bottom of the barrel!), I was prepared for the worst tonight, but I have my own room with its own bathroom and everything! In the lounge, I scraped together enough German to introduce myself to an Austrian group I met, and soon we were chatting away nineteen to the dozen. Dinner was hearty – fine by me as I had worked up a ravenous appetite. But soon I felt my eyelids drooping and excused myself. So here I am, writing by the light of my bedside lamp, looking forward to new adventures on the waterways of Europe tomorrow. Analytical note 3: For the unknown writer of this text, Schengen represented a way for holiday-makers to enjoy Europe’s rivers. Looking back on this tumultuous century, it is now clear that working in union was vital for European nations to face the challenges that lay ahead.
Binary file
@@ -1 +0,0 @@
1
- Sample PowerPoint File St. Cloud Technical College This is a Sample Slide Here is an outline of bulleted points You can print out PPT files as handouts using the PRINT > PRINT WHAT > HANDOUTS option hello testing
Binary file
@@ -1 +0,0 @@
1
- 0 1 32 1562 2 25 1582 3 36 2587 4 25 3549 5 58 2468 96 34 2654 97 28 6525 98 32 3265 99 39 3265 100 29 6125 First Name Last Name Gender Country Age Date Id Dulce Abril Female United States 15/10/2017 Mara Hashimoto Great Britain 16/08/2016 Philip Gent Male France 21/05/2015 Kathleen Hanner Nereida Magwood Roma Lafollette Felisa Cail Demetria Abbey Jeromy Danz Rasheeda Alkire Hello Testing
@@ -1,97 +0,0 @@
1
- const officeParser = require("../officeParser");
2
- const fs = require("fs");
3
- const supportedExtensions = require("../supportedExtensions");
4
-
5
- // File names of test files and their text output content
6
- // test file name style => test.<ext>
7
- // test content output => test.<ext>.txt
8
-
9
- /** List of all supported extensions with office Parser */
10
- const supportedExtensionTests = [
11
- {
12
- ext: "docx",
13
- testAvailable: true,
14
- },
15
- {
16
- ext: "xlsx",
17
- testAvailable: true,
18
- },
19
- {
20
- ext: "pptx",
21
- testAvailable: true,
22
- },
23
- {
24
- ext: "odt",
25
- testAvailable: true,
26
- },
27
- {
28
- ext: "odp",
29
- testAvailable: true,
30
- },
31
- {
32
- ext: "ods",
33
- testAvailable: true,
34
- },
35
- ];
36
-
37
- /** Local list of supported extensions in test file */
38
- const localSupportedExtensionsList = supportedExtensionTests.map(test => test.ext);
39
-
40
- /** Get filename for an extension */
41
- function getFilename(ext, isContentFile = false) {
42
- return `test/files/test.${ext}` + (isContentFile ? `.txt` : '');
43
- }
44
-
45
- /** Run test for a passed extension */
46
- function runTest(ext) {
47
- return officeParser.parseOfficeAsync(getFilename(ext))
48
- .then(text =>
49
- fs.readFileSync(getFilename(ext, true), 'utf8') == text
50
- ? console.log(`[${ext}]=> Passed`)
51
- : console.log(`[${ext}]=> Failed`)
52
- )
53
- .catch(error => console.log("ERROR: " + error))
54
- }
55
-
56
- async function runAllTests() {
57
- for (let i = 0; i < supportedExtensionTests.length; i++)
58
- {
59
- const test = supportedExtensionTests[i];
60
- if (test.testAvailable)
61
- await runTest(test.ext)
62
- else
63
- console.log(`[${test.ext}]=> Skipped`);
64
- }
65
- }
66
-
67
- // Enable console output in case something fails
68
- // officeParser.enableConsoleOutput();
69
-
70
- // Run all test files with test content if no argument passed.
71
- if (process.argv.length == 2)
72
- {
73
- // Test to check all items in local extension list are present in supportedExtensions.js file
74
- localSupportedExtensionsList
75
- .every(ext => supportedExtensions.includes(ext))
76
- ? console.log("All extensions in test files found in primary supportedExtensions.js file")
77
- : console.warn("Extension in test files missing from primary supportedExtensions.js file");
78
-
79
- // Test to check all items in supportedExtensions.js file are present in local extension list
80
- supportedExtensions
81
- .every(ext => localSupportedExtensionsList.includes(ext))
82
- ? console.log("All extensions in primary supportedExtensions.js file found in test file")
83
- : console.warn("Extension in primary supportedExtensions.js file missing from test file");
84
-
85
- runAllTests();
86
- }
87
- else if (process.argv.length == 3)
88
- {
89
- if (localSupportedExtensionsList.includes(process.argv[2]))
90
- officeParser.parseOfficeAsync(getFilename(process.argv[2]))
91
- .then(text => console.log(text))
92
- .catch(error => console.log("ERROR: " + error))
93
- else
94
- console.error("The requested extension test is not currently available.");
95
- }
96
- else
97
- console.error("Invalid arguments");