echogarden 0.5.1 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1 @@
1
- {"version":3,"file":"Timeline.js","sourceRoot":"","sources":["../../src/utilities/Timeline.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,MAAM,EAAE,iBAAiB,EAAE,gBAAgB,EAAE,MAAM,wBAAwB,CAAA;AACpF,OAAO,EAAE,SAAS,EAAE,MAAM,sBAAsB,CAAA;AAChD,OAAO,EAAE,aAAa,EAAE,MAAM,gBAAgB,CAAA;AAE9C,MAAM,UAAU,uBAAuB,CAAC,cAAwB,EAAE,UAAkB;IACnF,IAAI,CAAC,cAAc,EAAE;QACpB,OAAO,cAAc,CAAA;KACrB;IAED,MAAM,WAAW,GAAG,SAAS,CAAC,cAAc,CAAC,CAAA;IAE7C,KAAK,MAAM,oBAAoB,IAAI,WAAW,EAAE;QAC/C,oBAAoB,CAAC,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,oBAAoB,CAAC,SAAS,GAAG,UAAU,EAAE,CAAC,CAAC,CAAA;QACzF,oBAAoB,CAAC,OAAO,GAAG,IAAI,CAAC,GAAG,CAAC,oBAAoB,CAAC,OAAO,GAAG,UAAU,EAAE,CAAC,CAAC,CAAA;QAErF,IAAI,oBAAoB,CAAC,QAAQ,EAAE;YAClC,oBAAoB,CAAC,QAAQ,GAAG,uBAAuB,CAAC,oBAAoB,CAAC,QAAQ,EAAE,UAAU,CAAC,CAAA;SAClG;KACD;IAED,OAAO,WAAW,CAAA;AACnB,CAAC;AAGD,MAAM,UAAU,wBAAwB,CAAC,cAAwB,EAAE,MAAc;IAChF,MAAM,WAAW,GAAG,SAAS,CAAC,cAAc,CAAC,CAAA;IAE7C,KAAK,MAAM,oBAAoB,IAAI,WAAW,EAAE;QAC/C,oBAAoB,CAAC,SAAS,GAAG,oBAAoB,CAAC,SAAS,GAAG,MAAM,CAAA;QACxE,oBAAoB,CAAC,OAAO,GAAG,oBAAoB,CAAC,OAAO,GAAG,MAAM,CAAA;QAEpE,IAAI,oBAAoB,CAAC,QAAQ,EAAE;YAClC,oBAAoB,CAAC,QAAQ,GAAG,wBAAwB,CAAC,oBAAoB,CAAC,QAAQ,EAAE,MAAM,CAAC,CAAA;SAC/F;KACD;IAED,OAAO,WAAW,CAAA;AACnB,CAAC;AAED,MAAM,UAAU,uBAAuB,CAAC,cAAwB,EAAE,aAAa,GAAG,CAAC;IAClF,MAAM,eAAe,GAAG,SAAS,CAAC,cAAc,CAAC,CAAA;IAEjD,KAAK,MAAM,KAAK,IAAI,eAAe,EAAE;QACpC,IAAI,KAAK,CAAC,SAAS,EAAE;YACpB,KAAK,CAAC,SAAS,GAAG,aAAa,CAAC,KAAK,CAAC,SAAS,EAAE,aAAa,CAAC,CAAA;SAC/D;QAED,IAAI,KAAK,CAAC,OAAO,EAAE;YAClB,KAAK,CAAC,OAAO,GAAG,aAAa,CAAC,KAAK,CAAC,OAAO,EAAE,aAAa,CAAC,CAAA;SAC3D;QAED,IAAI,KAAK,CAAC,QAAQ,EAAE;YACnB,KAAK,CAAC,QAAQ,GAAG,uBAAuB,CAAC,KAAK,CAAC,QAAQ,CAAC,CAAA;SACxD;KACD;IAED,OAAO,eAAe,CAAA;AACvB,CAAC;AAED,MAAM,CAAC,KAAK,UAAU,qCAAqC,CAAC,YAAsB,EAAE,UAAkB,EAAE,QAAgB;IACvH,MAAM,UAAU,GAAG,MAAM,iBAAiB,CAAC,UAAU,CAAC,CAAA;IAEtD,MAAM,QAAQ,GAAG,UAAU;SACxB,GAAG,CAAC,OAAO,CAAC,EAAE,CACd,gBAAgB,CAAC,OAAO,EAAE,QAAQ,CAAC,CAAC,GAAG,CAAC,QAAQ,CAAC,EAAE,CAClD,QAAQ,CAAC,IAAI,EAAE,GAAG,GAAG,CAAC,CAAC,CAAA;IAE3B,IAAI,IAAI,GAAG,EAAE,CAAA;IACb,MAAM,+BAA+B,GAAoB,EAAE,CAAA;IAE3D,MAAM,eAAe,GAAa,EAAE,CAAA;IAEpC,KAAK,MAAM,OAAO,IAAI,QAAQ,EAAE;QAC/B,MAAM,kBAAkB,GAAa,EAAE,CAAA;QAEvC,MAAM,YAAY,GAAkB;YACnC,IAAI,EAAE,SAAS;YACf,IAAI,EAAE,EAAE;YACR,SAAS,EAAE,CAAC,CAAC;YACb,OAAO,EAAE,CAAC,CAAC;YACX,QAAQ,EAAE,kBAAkB;SAC5B,CAAA;QAED,KAAK,MAAM,QAAQ,IAAI,OAAO,EAAE;YAC/B,MAAM,aAAa,GAAkB;gBACpC,IAAI,EAAE,UAAU;gBAChB,IAAI,EAAE,QAAQ;gBACd,SAAS,EAAE,CAAC,CAAC;gBACb,OAAO,EAAE,CAAC,CAAC;gBACX,QAAQ,EAAE,EAAE;aACZ,CAAA;YAED,KAAK,MAAM,IAAI,IAAI,QAAQ,EAAE;gBAC5B,IAAI,IAAI,IAAI,CAAA;gBACZ,+BAA+B,CAAC,IAAI,CAAC,aAAa,CAAC,CAAA;aACnD;YAED,kBAAkB,CAAC,IAAI,CAAC,aAAa,CAAC,CAAA;SACtC;QAED,eAAe,CAAC,IAAI,CAAC,YAAY,CAAC,CAAA;KAClC;IAED,IAAI,qBAAqB,GAAG,CAAC,CAAA;IAE7B,KAAK,IAAI,SAAS,GAAG,CAAC,EAAE,SAAS,GAAG,YAAY,CAAC,MAAM,EAAE,SAAS,EAAE,EAAE;QACrE,MAAM,SAAS,GAAG,YAAY,CAAC,SAAS,CAAC,CAAA;QACzC,MAAM,QAAQ,GAAG,SAAS,CAAC,IAAI,CAAA;QAE/B,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,EAAE;YACtB,SAAQ;SACR;QAED,MAAM,iBAAiB,GAAG,IAAI,CAAC,OAAO,CAAC,QAAQ,EAAE,qBAAqB,CAAC,CAAA;QAEvE,IAAI,iBAAiB,IAAI,CAAC,CAAC,EAAE;YAC5B,MAAM,IAAI,KAAK,CAAC,2BAA2B,QAAQ,mCAAmC,qBAAqB,EAAE,CAAC,CAAA;SAC9G;QAED,MAAM,mBAAmB,GAAG,+BAA+B,CAAC,iBAAiB,CAAC,CAAA;QAC9E,mBAAmB,CAAC,QAAS,CAAC,IAAI,CAAC,SAAS,CAAC,CAAA;QAE7C,qBAAqB,GAAG,iBAAiB,GAAG,QAAQ,CAAC,MAAM,CAAA;KAC3D;IAED,KAAK,MAAM,YAAY,IAAI,eAAe,EAAE;QAC3C,MAAM,gBAAgB,GAAG,YAAY,CAAC,QAAS,CAAA;QAE/C,IAAI,gBAAgB,CAAC,MAAM,IAAI,CAAC,EAAE;YACjC,MAAM,IAAI,KAAK,CAAC,iCAAiC,CAAC,CAAA;SAClD;QAED,KAAK,MAAM,aAAa,IAAI,gBAAgB,EAAE;YAC7C,MAAM,YAAY,GAAG,aAAa,CAAC,QAAS,CAAA;YAE5C,IAAI,YAAY,CAAC,MAAM,IAAI,CAAC,EAAE;gBAC7B,MAAM,IAAI,KAAK,CAAC,8BAA8B,CAAC,CAAA;aAC/C;YAED,aAAa,CAAC,SAAS,GAAG,YAAY,CAAC,CAAC,CAAC,CAAC,SAAS,CAAA;YACnD,aAAa,CAAC,OAAO,GAAG,YAAY,CAAC,YAAY,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,OAAO,CAAA;SACrE;QAED,YAAY,CAAC,IAAI,GAAG,gBAAgB,CAAC,GAAG,CAAC,aAAa,CAAC,EAAE,CAAC,aAAa,CAAC,IAAI,CAAC,CAAC,IAAI,CAAC,EAAE,CAAC,CAAA;QAEtF,YAAY,CAAC,SAAS,GAAG,gBAAgB,CAAC,CAAC,CAAC,CAAC,SAAS,CAAA;QACtD,YAAY,CAAC,OAAO,GAAG,gBAAgB,CAAC,gBAAgB,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,OAAO,CAAA;KAC5E;IAED,OAAO,EAAE,eAAe,EAAE,CAAA;AAC3B,CAAC"}
1
+ {"version":3,"file":"Timeline.js","sourceRoot":"","sources":["../../src/utilities/Timeline.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,MAAM,EAAE,iBAAiB,EAAE,gBAAgB,EAAE,MAAM,wBAAwB,CAAA;AACpF,OAAO,EAAE,SAAS,EAAE,MAAM,sBAAsB,CAAA;AAChD,OAAO,EAAE,aAAa,EAAE,MAAM,gBAAgB,CAAA;AAE9C,MAAM,UAAU,uBAAuB,CAAC,cAAwB,EAAE,UAAkB;IACnF,IAAI,CAAC,cAAc,EAAE;QACpB,OAAO,cAAc,CAAA;KACrB;IAED,MAAM,WAAW,GAAG,SAAS,CAAC,cAAc,CAAC,CAAA;IAE7C,KAAK,MAAM,oBAAoB,IAAI,WAAW,EAAE;QAC/C,oBAAoB,CAAC,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,oBAAoB,CAAC,SAAS,GAAG,UAAU,EAAE,CAAC,CAAC,CAAA;QACzF,oBAAoB,CAAC,OAAO,GAAG,IAAI,CAAC,GAAG,CAAC,oBAAoB,CAAC,OAAO,GAAG,UAAU,EAAE,CAAC,CAAC,CAAA;QAErF,IAAI,oBAAoB,CAAC,QAAQ,EAAE;YAClC,oBAAoB,CAAC,QAAQ,GAAG,uBAAuB,CAAC,oBAAoB,CAAC,QAAQ,EAAE,UAAU,CAAC,CAAA;SAClG;KACD;IAED,OAAO,WAAW,CAAA;AACnB,CAAC;AAGD,MAAM,UAAU,wBAAwB,CAAC,cAAwB,EAAE,MAAc;IAChF,MAAM,WAAW,GAAG,SAAS,CAAC,cAAc,CAAC,CAAA;IAE7C,KAAK,MAAM,oBAAoB,IAAI,WAAW,EAAE;QAC/C,oBAAoB,CAAC,SAAS,GAAG,oBAAoB,CAAC,SAAS,GAAG,MAAM,CAAA;QACxE,oBAAoB,CAAC,OAAO,GAAG,oBAAoB,CAAC,OAAO,GAAG,MAAM,CAAA;QAEpE,IAAI,oBAAoB,CAAC,QAAQ,EAAE;YAClC,oBAAoB,CAAC,QAAQ,GAAG,wBAAwB,CAAC,oBAAoB,CAAC,QAAQ,EAAE,MAAM,CAAC,CAAA;SAC/F;KACD;IAED,OAAO,WAAW,CAAA;AACnB,CAAC;AAED,MAAM,UAAU,uBAAuB,CAAC,cAAwB,EAAE,aAAa,GAAG,CAAC;IAClF,MAAM,eAAe,GAAG,SAAS,CAAC,cAAc,CAAC,CAAA;IAEjD,KAAK,MAAM,KAAK,IAAI,eAAe,EAAE;QACpC,IAAI,KAAK,CAAC,SAAS,EAAE;YACpB,KAAK,CAAC,SAAS,GAAG,aAAa,CAAC,KAAK,CAAC,SAAS,EAAE,aAAa,CAAC,CAAA;SAC/D;QAED,IAAI,KAAK,CAAC,OAAO,EAAE;YAClB,KAAK,CAAC,OAAO,GAAG,aAAa,CAAC,KAAK,CAAC,OAAO,EAAE,aAAa,CAAC,CAAA;SAC3D;QAED,IAAI,KAAK,CAAC,QAAQ,EAAE;YACnB,KAAK,CAAC,QAAQ,GAAG,uBAAuB,CAAC,KAAK,CAAC,QAAQ,CAAC,CAAA;SACxD;KACD;IAED,OAAO,eAAe,CAAA;AACvB,CAAC;AAED,MAAM,CAAC,KAAK,UAAU,qCAAqC,CAAC,YAAsB,EAAE,UAAkB,EAAE,QAAgB;IACvH,MAAM,UAAU,GAAG,MAAM,iBAAiB,CAAC,UAAU,EAAE,QAAQ,EAAE,KAAK,CAAC,CAAA;IAEvE,MAAM,QAAQ,GAAG,UAAU;SACxB,GAAG,CAAC,OAAO,CAAC,EAAE,CACd,gBAAgB,CAAC,OAAO,EAAE,QAAQ,CAAC,CAAC,GAAG,CAAC,QAAQ,CAAC,EAAE,CAClD,QAAQ,CAAC,IAAI,EAAE,CAAC,CAAC,CAAA;IAErB,IAAI,IAAI,GAAG,EAAE,CAAA;IACb,MAAM,+BAA+B,GAAoB,EAAE,CAAA;IAE3D,MAAM,eAAe,GAAa,EAAE,CAAA;IAEpC,KAAK,MAAM,OAAO,IAAI,QAAQ,EAAE;QAC/B,MAAM,kBAAkB,GAAa,EAAE,CAAA;QAEvC,MAAM,YAAY,GAAkB;YACnC,IAAI,EAAE,SAAS;YACf,IAAI,EAAE,EAAE;YACR,SAAS,EAAE,CAAC,CAAC;YACb,OAAO,EAAE,CAAC,CAAC;YACX,QAAQ,EAAE,kBAAkB;SAC5B,CAAA;QAED,KAAK,MAAM,QAAQ,IAAI,OAAO,EAAE;YAC/B,MAAM,aAAa,GAAkB;gBACpC,IAAI,EAAE,UAAU;gBAChB,IAAI,EAAE,QAAQ;gBACd,SAAS,EAAE,CAAC,CAAC;gBACb,OAAO,EAAE,CAAC,CAAC;gBACX,QAAQ,EAAE,EAAE;aACZ,CAAA;YAED,KAAK,MAAM,IAAI,IAAI,QAAQ,GAAG,GAAG,EAAE;gBAClC,IAAI,IAAI,IAAI,CAAA;gBACZ,+BAA+B,CAAC,IAAI,CAAC,aAAa,CAAC,CAAA;aACnD;YAED,kBAAkB,CAAC,IAAI,CAAC,aAAa,CAAC,CAAA;SACtC;QAED,eAAe,CAAC,IAAI,CAAC,YAAY,CAAC,CAAA;KAClC;IAED,IAAI,qBAAqB,GAAG,CAAC,CAAA;IAE7B,KAAK,IAAI,SAAS,GAAG,CAAC,EAAE,SAAS,GAAG,YAAY,CAAC,MAAM,EAAE,SAAS,EAAE,EAAE;QACrE,MAAM,SAAS,GAAG,YAAY,CAAC,SAAS,CAAC,CAAA;QACzC,MAAM,QAAQ,GAAG,SAAS,CAAC,IAAI,CAAA;QAE/B,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,EAAE;YACtB,SAAQ;SACR;QAED,MAAM,iBAAiB,GAAG,IAAI,CAAC,OAAO,CAAC,QAAQ,EAAE,qBAAqB,CAAC,CAAA;QAEvE,IAAI,iBAAiB,IAAI,CAAC,CAAC,EAAE;YAC5B,MAAM,IAAI,KAAK,CAAC,2BAA2B,QAAQ,mCAAmC,qBAAqB,EAAE,CAAC,CAAA;SAC9G;QAED,MAAM,mBAAmB,GAAG,+BAA+B,CAAC,iBAAiB,CAAC,CAAA;QAC9E,mBAAmB,CAAC,QAAS,CAAC,IAAI,CAAC,SAAS,CAAC,CAAA;QAE7C,qBAAqB,GAAG,iBAAiB,GAAG,QAAQ,CAAC,MAAM,CAAA;KAC3D;IAED,KAAK,MAAM,YAAY,IAAI,eAAe,EAAE;QAC3C,MAAM,gBAAgB,GAAG,YAAY,CAAC,QAAS,CAAA;QAE/C,IAAI,gBAAgB,CAAC,MAAM,IAAI,CAAC,EAAE;YACjC,MAAM,IAAI,KAAK,CAAC,iCAAiC,CAAC,CAAA;SAClD;QAED,KAAK,MAAM,aAAa,IAAI,gBAAgB,EAAE;YAC7C,MAAM,YAAY,GAAG,aAAa,CAAC,QAAS,CAAA;YAE5C,IAAI,YAAY,CAAC,MAAM,IAAI,CAAC,EAAE;gBAC7B,MAAM,IAAI,KAAK,CAAC,8BAA8B,CAAC,CAAA;aAC/C;YAED,aAAa,CAAC,SAAS,GAAG,YAAY,CAAC,CAAC,CAAC,CAAC,SAAS,CAAA;YACnD,aAAa,CAAC,OAAO,GAAG,YAAY,CAAC,YAAY,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,OAAO,CAAA;SACrE;QAED,YAAY,CAAC,IAAI,GAAG,gBAAgB,CAAC,GAAG,CAAC,aAAa,CAAC,EAAE,CAAC,aAAa,CAAC,IAAI,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAA;QAEvF,YAAY,CAAC,SAAS,GAAG,gBAAgB,CAAC,CAAC,CAAC,CAAC,SAAS,CAAA;QACtD,YAAY,CAAC,OAAO,GAAG,gBAAgB,CAAC,gBAAgB,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,OAAO,CAAA;KAC5E;IAED,OAAO,EAAE,eAAe,EAAE,CAAA;AAC3B,CAAC"}
@@ -15,9 +15,7 @@ export async function parseWikipediaArticle(articleName, language) {
15
15
  if (wordCharacterPattern.test(sectionTitle)) {
16
16
  sectionsText.push(sectionTitle);
17
17
  }
18
- //sectionsText.push()
19
- //const sectionParagraphs = section.paragraphs()
20
- const sectionParagraphs = splitToParagraphs(section.text());
18
+ const sectionParagraphs = splitToParagraphs(section.text(), 'single', true);
21
19
  for (const paragraph of sectionParagraphs) {
22
20
  const paragraphText = paragraph;
23
21
  if (wordCharacterPattern.test(paragraphText)) {
@@ -1 +1 @@
1
- {"version":3,"file":"WikipediaReader.js","sourceRoot":"","sources":["../../src/utilities/WikipediaReader.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,iBAAiB,EAAE,oBAAoB,EAAE,MAAM,wBAAwB,CAAA;AAChF,OAAO,EAAE,MAAM,EAAE,MAAM,aAAa,CAAA;AAEpC,MAAM,CAAC,KAAK,UAAU,qBAAqB,CAAC,WAAmB,EAAE,QAAgB;IAChF,MAAM,MAAM,GAAG,IAAI,MAAM,EAAE,CAAA;IAE3B,MAAM,CAAC,UAAU,CAAC,4BAA4B,CAAC,CAAA;IAE/C,MAAM,EAAE,OAAO,EAAE,GAAG,EAAE,GAAG,MAAM,MAAM,CAAC,eAAe,CAAC,CAAA;IAEtD,MAAM,QAAQ,GAAG,MAAM,GAAG,CAAC,KAAK,CAAC,WAAW,EAAE,QAAQ,CAAC,CAAA;IAEvD,IAAI,CAAC,QAAQ,EAAE;QACd,MAAM,IAAI,KAAK,CAAC,kCAAkC,CAAC,CAAA;KACnD;IAED,MAAM,QAAQ,GAAG,QAAQ,CAAC,QAAQ,EAAE,CAAA;IACpC,MAAM,YAAY,GAAa,EAAE,CAAA;IAEjC,KAAK,MAAM,OAAO,IAAI,QAAQ,EAAE;QAC/B,MAAM,YAAY,GAAG,OAAO,CAAC,KAAK,EAAE,CAAA;QAEpC,IAAI,oBAAoB,CAAC,IAAI,CAAC,YAAY,CAAC,EAAE;YAC5C,YAAY,CAAC,IAAI,CAAC,YAAY,CAAC,CAAA;SAC/B;QAED,qBAAqB;QAErB,gDAAgD;QAChD,MAAM,iBAAiB,GAAG,iBAAiB,CAAC,OAAO,CAAC,IAAI,EAAE,CAAC,CAAA;QAE3D,KAAK,MAAM,SAAS,IAAI,iBAAiB,EAAE;YAC1C,MAAM,aAAa,GAAG,SAAS,CAAA;YAE/B,IAAI,oBAAoB,CAAC,IAAI,CAAC,aAAa,CAAC,EAAE;gBAC7C,YAAY,CAAC,IAAI,CAAC,aAAa,CAAC,CAAA;aAChC;SACD;KACD;IAED,MAAM,CAAC,GAAG,EAAE,CAAA;IAEZ,OAAO,YAAY,CAAA;AACpB,CAAC"}
1
+ {"version":3,"file":"WikipediaReader.js","sourceRoot":"","sources":["../../src/utilities/WikipediaReader.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,iBAAiB,EAAE,oBAAoB,EAAE,MAAM,wBAAwB,CAAA;AAChF,OAAO,EAAE,MAAM,EAAE,MAAM,aAAa,CAAA;AAEpC,MAAM,CAAC,KAAK,UAAU,qBAAqB,CAAC,WAAmB,EAAE,QAAgB;IAChF,MAAM,MAAM,GAAG,IAAI,MAAM,EAAE,CAAA;IAE3B,MAAM,CAAC,UAAU,CAAC,4BAA4B,CAAC,CAAA;IAE/C,MAAM,EAAE,OAAO,EAAE,GAAG,EAAE,GAAG,MAAM,MAAM,CAAC,eAAe,CAAC,CAAA;IAEtD,MAAM,QAAQ,GAAG,MAAM,GAAG,CAAC,KAAK,CAAC,WAAW,EAAE,QAAQ,CAAC,CAAA;IAEvD,IAAI,CAAC,QAAQ,EAAE;QACd,MAAM,IAAI,KAAK,CAAC,kCAAkC,CAAC,CAAA;KACnD;IAED,MAAM,QAAQ,GAAG,QAAQ,CAAC,QAAQ,EAAE,CAAA;IACpC,MAAM,YAAY,GAAa,EAAE,CAAA;IAEjC,KAAK,MAAM,OAAO,IAAI,QAAQ,EAAE;QAC/B,MAAM,YAAY,GAAG,OAAO,CAAC,KAAK,EAAE,CAAA;QAEpC,IAAI,oBAAoB,CAAC,IAAI,CAAC,YAAY,CAAC,EAAE;YAC5C,YAAY,CAAC,IAAI,CAAC,YAAY,CAAC,CAAA;SAC/B;QAED,MAAM,iBAAiB,GAAG,iBAAiB,CAAC,OAAO,CAAC,IAAI,EAAE,EAAE,QAAQ,EAAE,IAAI,CAAC,CAAA;QAE3E,KAAK,MAAM,SAAS,IAAI,iBAAiB,EAAE;YAC1C,MAAM,aAAa,GAAG,SAAS,CAAA;YAE/B,IAAI,oBAAoB,CAAC,IAAI,CAAC,aAAa,CAAC,EAAE;gBAC7C,YAAY,CAAC,IAAI,CAAC,aAAa,CAAC,CAAA;aAChC;SACD;KACD;IAED,MAAM,CAAC,GAAG,EAAE,CAAA;IAEZ,OAAO,YAAY,CAAA;AACpB,CAAC"}
package/docs/CLI.md CHANGED
@@ -8,9 +8,11 @@ echogarden [command] [one or more inputs..] [one or more outputs...] [options...
8
8
 
9
9
  Here's a quick tour of the main operations available via the CLI.
10
10
 
11
- Each command can accepts one or more options, in the form `--[optionName]=[value]` (The `=` is required). A detailed reference of all the available options can be found [here](Options.md).
11
+ Each command can accept one or more options, in the form `--[optionName]=[value]` (The `=` is required). A detailed reference of all the available options can be found [here](Options.md).
12
12
 
13
- While the program is running, you can press `esc` to immediately exit, or, during audio playback, `enter` to skip it.
13
+ **Keyboard shortcuts**:
14
+ * While the program is running, you can press `esc` to exit immediately
15
+ * When audio is playing, you can press `enter` to skip it
14
16
 
15
17
  ## Text to speech
16
18
 
@@ -31,7 +33,7 @@ This would save the resulting audio to `result.mp3`:
31
33
  echogarden speak "Hello world!" result.mp3 --language=en
32
34
  ```
33
35
 
34
- `speak-file` synthesizes text loaded from a textual file, which can be either `.txt`, `.html`, `.srt`, `.vtt`:
36
+ `speak-file` synthesizes text loaded from a textual file, which can have the extensions `txt`, `html`, `xml`, `ssml`, `srt`, `vtt`:
35
37
  ```bash
36
38
  echogarden speak-file text.txt result.mp3 --language=en
37
39
  ```
@@ -46,12 +48,12 @@ The CLI supports multiple output files. This would synthesize a text file, and s
46
48
  echogarden speak-file text.txt result.mp3 result.wav result.srt --engine=vits --speed=1.1
47
49
  ```
48
50
 
49
- Synthesize a web page (will try to extract its main article parts and omit the rest):
51
+ Synthesize a web page (it will try to extract its main article parts and omit the rest):
50
52
  ```bash
51
53
  echogarden speak-url https://example.com/hola
52
54
  ```
53
55
 
54
- Synthesize a Wikipedia article in any of its language editions:
56
+ Synthesize a Wikipedia article, in any of its language editions:
55
57
  ```bash
56
58
  echogarden speak-wikipedia "Psychologie" --language=fr
57
59
  ```
@@ -128,6 +130,9 @@ engine = sapi
128
130
  # Voice for synthesis (case-insensitive, can be a search pattern):
129
131
  voice = zira
130
132
 
133
+ # VITS custom lexicon paths:
134
+ vits.customLexiconPaths = ["lexicon1.json", "lexicon2.json"]
135
+
131
136
  [transcribe]
132
137
 
133
138
  # Engine for recognition:
@@ -146,7 +151,10 @@ Name your file `echogarden.config.json`:
146
151
  {
147
152
  "speak": {
148
153
  "engine": "sapi",
149
- "voice": "zira"
154
+ "voice": "zira",
155
+ "vits": {
156
+ "customLexiconPaths": ["lexicon1.json", "lexicon2.json"]
157
+ }
150
158
  },
151
159
 
152
160
  "transcribe": {
@@ -200,7 +208,7 @@ Try to identify the language of a text file, and print the probabilities to the
200
208
  echogarden detect-text-language story.txt
201
209
  ```
202
210
 
203
- Try to identify the language of a text file, and store the probabilities in a JSON file:
211
+ Try to identify the language of a text file, and store the detailed probabilities in a JSON file:
204
212
  ```bash
205
213
  echogarden detect-text-language story.txt detection-results.json
206
214
  ```
@@ -1,19 +1,25 @@
1
1
  # How to help
2
2
 
3
- So far, this project has been the solo work of a single person (yours truly).
3
+ So far, this project has been the solo work of a single person.
4
4
 
5
5
  However, there are many areas where contributions can be made.
6
6
 
7
- ## Reporting any issue or bug you encounter
7
+ ## Report any issue or bug you encounter
8
8
 
9
- Especially if you're using the macOS platform, since I don't have access to a macOS machine.
9
+ First, check out the [task list](Tasklist.md) to see if the problem is already known to me. The task list allows me to efficiently document and organize a large quantity of small issues, enhancements or ideas that would otherwise flood an issue tracker with lots of unimportant entries, be ignored, or forgotten entirely.
10
10
 
11
- ## Reporting odd TTS pronunciations and other fail cases
11
+ If you find the issue you're encountering in the task list, you can still open an issue to discuss it. This allows me to know that someone cares about a particular issue, and I may give it higher priority.
12
12
 
13
- Though that may be an issue with model training, which should be forwarded to the original authors
13
+ There might be some obvious errors that have gone unreported. Especially if you're using the macOS platform, since I don't have access to a macOS machine, and thus almost no real testing has been done over that platform.
14
14
 
15
- ## Extending the pronunciation lexicons
15
+ In any case, please let me know if you get any unexpected error message or surprising behavior that you care about, and I'll try to prioritize it, if possible.
16
16
 
17
- Especially if you are proficient in a language other than English. In many cases, the default phonemizations produced by the eSpeak engine are incorrect
17
+ ## Report odd TTS pronunciations and other fail cases
18
18
 
19
- You can also add rules to help resolve the pronunciations of heteronyms (words that are written the same but can be pronounced in different ways depending on their context) based on their context.
19
+ When you encounter an odd pronunciation, there are several possible causes:
20
+
21
+ 1. An incorrect phonemization produced by the eSpeak engine. Fortunately, it can be overridden by adding a corrected pronunciations in the Echogarden lexicon.
22
+ 1. This word has multiple different pronunciations based on context (a heteronym). In that case, it may be possible resolve the pronunciations based on context, by using the preceding and succeeding words as indicators. This is supported by the Echogarden lexicon.
23
+ 1. An issue with model training, which may need to be forwarded to the original authors.
24
+
25
+ If the problem is serious, you can report it and we'll see what we can do.
package/docs/Options.md CHANGED
@@ -1,6 +1,10 @@
1
1
  # Configuration options reference
2
2
 
3
- For a comprehensive list of all supported engines: see [this page](Engines.md).
3
+ Here is a detailed reference for the options accepted by the Echogarden API and CLI.
4
+
5
+ Related resources:
6
+ * [A comprehensive list of all supported engines](Engines.md)
7
+ * [A quick guide for using the command line interface](CLI.md)
4
8
 
5
9
  ## Synthesis
6
10
 
@@ -16,19 +20,24 @@ General:
16
20
  * `pitchVariation`: pitch variation factor. In the range `0.1`..`10.0`. Defaults to `1.0`
17
21
  * `ssml`: the input is SSML. Defaults to `false`
18
22
  * `sentenceEndPause`: pause duration (seconds) at end of sentence. Defaults to `0.75`
19
- * `segmentEndPause` pause duration (seconds) at end of segment. Defaults to `1.0`
23
+ * `segmentEndPause`: pause duration (seconds) at end of segment. Defaults to `1.0`
24
+
25
+ Plain text preprocessing:
26
+ * `plainText.paragraphBreaks`: split to paragraphs based on single (`single`), or double (`double`) line breaks. Defaults to `double`
27
+ * `plainText.preserveLineBreaks`: preserve line breaks within paragraphs. Defaults to `false`
20
28
 
21
29
  Post-processing:
22
- * `postProcessing.normalizeAudio` should normalize output audio. Defaults to `true`
23
- * `postProcessing.targetPeakDb` target peak (decibels) for normalization. Defaults to `-3`
24
- * `postProcessing.maxIncreaseDb` max gain increase (decibels) when performing normalization. Defaults to `30`
30
+ * `postProcessing.normalizeAudio`: should normalize output audio. Defaults to `true`
31
+ * `postProcessing.targetPeakDb`: target peak (decibels) for normalization. Defaults to `-3`
32
+ * `postProcessing.maxIncreaseDb`: max gain increase (decibels) when performing normalization. Defaults to `30`
25
33
  * `postProcessing.speed`: target speed for time stretching. Defaults to `1.0`
26
34
  * `postProcessing.pitch`: target pitch for pitch shifting. Defaults to `1.0`
27
- * `postProcessing.timePitchShiftingMethod` method for time and pitch shifting. Can be `sonic` or `rubberband`. Defaults to `sonic`
35
+ * `postProcessing.timePitchShiftingMethod`: method for time and pitch shifting. Can be `sonic` or `rubberband`. Defaults to `sonic`
28
36
  * `postProcessing.rubberband`: prefix for RubberBand options (TODO)
29
37
 
30
38
  VITS:
31
- * `vits.speakerId`: speaker ID, for VITS models that support multiple speakers
39
+ * `vits.speakerId`: speaker ID, for VITS models that support multiple speakers. Optional
40
+ * `vits.customLexiconPaths`: an array of custom lexicon file paths. Optional
32
41
 
33
42
  eSpeak-ng:
34
43
  * `espeak.rate`: speech rate, in eSpeak units. Overrides `speed` when set
@@ -178,10 +187,10 @@ General:
178
187
  * `engine`: can only be `rnnoise`
179
188
 
180
189
  Postprocessing:
181
- * `postProcessing.normalizeAudio` should normalize output audio. Defaults to `false`
182
- * `postProcessing.targetPeakDb` target peak (decibels) for normalization. Defaults to `-3`
183
- * `postProcessing.maxIncreaseDb` max gain increase (decibels) when performing normalization. Defaults to `30`
184
- * `postProcessing.dryMixGainDb` gain (decibels) of dry (original) signal to mix back to the denoised output. Defaults to `-20`
190
+ * `postProcessing.normalizeAudio`: should normalize output audio. Defaults to `false`
191
+ * `postProcessing.targetPeakDb`: target peak (decibels) for normalization. Defaults to `-3`
192
+ * `postProcessing.maxIncreaseDb`: max gain increase (decibels) when performing normalization. Defaults to `30`
193
+ * `postProcessing.dryMixGainDb`: gain (decibels) of dry (original) signal to mix back to the denoised output. Defaults to `-20`
185
194
 
186
195
  ## Voice list request
187
196
 
package/docs/Tasklist.md CHANGED
@@ -39,6 +39,7 @@
39
39
  * Button or keyboard shortcut to show and hide handles
40
40
  * Show blinking placeholder when synthesis is loading for a particular text node
41
41
  * Navigate paragraphs or sentences with keyboard shortcuts
42
+ * Minimum size when iterating text nodes to get handle
42
43
 
43
44
  ### Worker
44
45
  * Optionally omit unnecessary data from the response (decoded input, segment data, etc.)
@@ -47,26 +48,24 @@
47
48
  * Support more operations
48
49
 
49
50
  ### CLI
50
- * Don't show an error when `--help` option is given, instead suggest to type `echogarden` or `echogarden help` to get help
51
51
  * Find a way to ensure that a user who typed `align audio.mp3 transcript.txt` and then changed to `transcribe audio.mp3 transcript.txt` won't accidently overwrite their transcript file. Simple solution, but possibly not the best solution: `align audio.mp3 --reference=transcript.txt`. Other solution: on `transcribe` and `translate-speech`, ask if output file already exist or require an `--overwrite` flag to ensure that the user intended to overwrite the existing file.
52
- * Restrict input media file extensions to a set list to avoid cases where a media file would be overwritten due to user error
52
+ * Restrict input media file extensions to a set list to avoid cases where an output media file would be overwritten due to user error
53
+ * Mode to print IPA words when speaking
53
54
  * Show a message when a new version is available
54
55
  * Figure out which terminal outputs should go to stdout, or if that's a good idea at all
55
56
  * Option to set audio output codec options
56
57
  * Option to set audio output device
57
58
  * Print available synthesis voices when no voice matches (or suggest near matches)
58
- * `transcribe` can also accept `http://` and `https://` URLs
59
+ * `transcribe` may also accept `http://` and `https://` URLs and pull the remote media file
59
60
  * Use a file type detector like `file-type` that uses magic numbers to detect the type of a binary file regardless of its extension. This would help giving better error messages when the given file type is wrong.
60
- * Consider adding the text offset to each segment, sentence and word in the resulting timeline with respect to the original file (even if it is, say, HTML or captions file)
61
+ * Consider adding the input text offset to each segment, sentence and word in the resulting timeline with respect to the original file (even if it is, say, an HTML or captions file)
61
62
  * Add phone playback support
62
63
  * More fine-grained intermediate progress report for operations
63
64
  * Suggest possible correction on the error of not using `=`, e.g. `speed 0.9` instead of `speed=0.9`
64
65
  * Multiple configuration files in `--config=..` taking precedence by order
65
- * Support comments in JSON configuration file
66
+ * Support comments in the JSON configuration file
66
67
  * Generate JSON configuration file schema
67
68
  * Make enum options case-insensitive if possible
68
- * Mode to print IPA words when speaking
69
- * Support list-typed properties configuration files (already supported in JSON)
70
69
 
71
70
  ### CLI / `speak-wikipedia`
72
71
  * Correctly detect language when a Wikipedia URL is passed instead of an article name
@@ -78,11 +77,7 @@
78
77
  ### CLI / `list-packages`
79
78
  * Support filters
80
79
 
81
- ### CLI / Configuration file
82
- * Support arrays
83
-
84
80
  ### CLI / New commands
85
- * `help`: Show help for a particular command, like `help transcribe`
86
81
  * `list-engines`: List available engines for a particular command, like `list-engines speak`
87
82
  * `play-with-captions`: Preview captions in terminal
88
83
  * `play-with-timeline`: Preview timeline in terminal
@@ -91,14 +86,13 @@
91
86
  * `text-to-ipa`, `arpabet-to-ipa`, `ipa-to-arpabet`
92
87
  * `phonemize-text`
93
88
  * `normalize-text`
94
- * `pos-tag-text`
95
89
  * `remove-nonspeech`
96
90
  * `speak-youtube`: To speak the transcript of a YouTube video
97
91
 
98
92
  ### API
99
93
  * Option to control logging verbosity
94
+ * Accept full language names as language identifiers
100
95
  * Add support to accept caption options in API and CLI
101
- * Support full language names as inputs
102
96
  * Retry on error when connecting to cloud providers, including WebSocket disconnection with `microsoft-edge` (already supported by `gaxios`, not sure about `ws` - decide on default setting)
103
97
  * Validate timelines to ensure timestamps are always increasing, no -1 timestamps or timestamps over the time of the audio, no sentences without words, etc. and correct if needed
104
98
  * Time/pitch shifting for recognition and alignment results
@@ -110,7 +104,6 @@
110
104
  * When using Whisper for language detection of speech, apply it to the entire audio, not just the first 30 seconds
111
105
 
112
106
  ### Segmentation
113
- * Option to split segment on single line break as well as double line break (when splitting on double line breaks, there might be cases where a single line break should be seen as a sentence boundary - a line in song lyrics). Maybe a better approach is to optionally preprocess the text and merge subsequent lines than are intended to be a part of the same paragraph. In this way, paragraph breaks would always be single line breaks, and there's no need to carry settings for this detail within the program.
114
107
  * Split long words
115
108
  * See if it's possible to reliably use eSpeak as a segmentation engine
116
109
  * Path to `kuromoji` dictionaries can be found more reliably than current
@@ -128,35 +121,33 @@
128
121
  * Find places to add commas (",") to improve speech fluency. VITS voices don't normally add phrasing breaks if there is no punctuation
129
122
  * An isolated dash " - " can be converted to a " , " to ensure there's a break in the speech.
130
123
  * Ensure abbreviations like "Ph.d" or names like are segmented and read correctly (why doesn't `cldr` treat it as a word? Maybe it's not getting the right parameters, or it's not included in the list?) and "C#"
131
- * Use preprocessed eSpeak in places other than VITS
132
- * Way to manually reset voice list cache
124
+ * Find way to manually reset voice list cache
133
125
  * When synthesized text isn't pre-split to sentences, apply sentence splits by using the existing method to convert the output of word timelines to sentence/segment timelines
134
126
  * Log full language of selected voice (it may have a different dialect than expected)
135
127
  * Add partial SSML support for all engines. In particular, allow changing language or voice using the `<voice>` and `<lang>` tags, `<say-as>` and `<phoneme>` where possible.
136
128
  * Some `sapi` voices and `msspeech` languages output phones that are converted to Microsoft alphabet, not IPA symbols. Try to see if these can be translated to IPA
137
129
  * Decide whether asterisk `*` should be spoken when using `speak-url` or `speak-wikipedia`
138
130
  * Decide what to do with `«` and `»` punctuation characters (guillemets) when parsing and playing
139
- * Use VAD on the synthesized audio file to get more accurate sentence or word segmentation
140
131
  * Try to remove reliance on `()` after `.` character hack in `EspeakTTS.synthesizeFragments`.
141
132
  * eSpeak IPA output puts stress marks on vowels, not syllables - which is the standard for IPA. Consider how to make a conversion to and from these two approaches (possibly detect it automatically).
142
133
  * Investigate if `espeak` can be made to correctly support phonemizing and pronouncing the dot character like in `object.key`
143
134
  * Speaker-specific voice option
144
135
  * Decide if `msspeech` engine should be selected if available. This would require attempting to load a matching voice, and falling back if it is not installed
145
136
  * Option to disable alignment
137
+ * Use VAD on the synthesized audio file to get more accurate sentence or word segmentation
146
138
 
147
139
  ### Synthesis / preprocessing
148
- * Extend the heteronyms JSON document with more words
149
- * Full date normalization (e.g. `21 August 2023`)
150
- * Support custom lexicons. For example, a lexicon for general pronunciation corrections (for example `ee-lon` instead of `eh-lon`)
151
- * Add support for capitalized only rules, and possibly also all uppercase / all lowercase rules.
140
+ * Extend the heteronyms JSON document with additional words like "conducts", "survey", "protest", "transport", "abuse", "combat", "combats", "affect", "contest", "detail", "marked", "contrast", "construct", "constructs", "console", "recall", "permit", "permits", "prospect", "prospects", "proceed", "proceeds", "invite", "reject", "deserts", "transcript", "transcripts", "compact", "impact", "impacts"
141
+ * Full date normalization (e.g. `21 August 2023`, `21 Aug 2023`)
142
+ * Use preprocessed eSpeak in places other than VITS
143
+ * Add support for capitalized-only rules, and possibly also all uppercase / all lowercase rules.
152
144
  * Support normalizing to graphemes, not only phonemes
145
+ * Cache lexicons to avoid parsing the JSON each time it is loaded (this may not be needed for if the file is relatively small)
153
146
  * Is it possible to pre-phonemize common words like "the" or is it a bad idea / not necessary?
154
- * Add support for user-defined lexicons for VITS synthesis (or any other one that supports preprocessing)
155
- * Add support for text normalization preprocessing for all engines that can benefit from it (possibly including cloud engines).
147
+ * Add support for text preprocessing for all engines that can benefit from it (possibly including cloud engines).
156
148
  * Add SAPI pronunciation to lexicons (the information is already there for `en_US` and `en_GB`)
157
149
  * Try to use entity recognition to detect years, dates, currencies etc., which would disambiguate cases where it is not clear, like "in 1993" in "She was born in 1993" and "It searched in 1993 websites"
158
150
  * Option to add POS tags to timeline, if available
159
- * Cache lexicons to avoid parsing the JSON each time it is loaded (this may not be needed for now since the existing one is relatively small)
160
151
 
161
152
  ### VITS
162
153
  * Allow to limit how many models are cached in memory
@@ -191,6 +182,7 @@
191
182
  * Start thinking about some modules being available in the browser. Which node core APIs the use? Which of them can be polyfilled, an which cannot?
192
183
  * Change all the Emscripten WASM modules to use the `EXPORT_ES6=1` flag to all of them and rebuild them. Support for node.js was only added in September 2022 (https://github.com/emscripten-core/emscripten/pull/17915), so maybe wait a little bit until it is stable.
193
184
  * Remove built-in voices from `flite` to reduce size?
185
+ * Slim down `kuromoji` package to the reduce base install size
194
186
 
195
187
  ## External bugs
196
188
 
@@ -227,9 +219,6 @@
227
219
  ### Text enhancement
228
220
  * Add capitalization and punctuation when to recognition outputs (Silero has a model for it for `en`, `de`, `ru`, `es`, but in `.pt` format only)
229
221
 
230
- ### Synthesis / preprocessing
231
- * Extend preprocessing to other language versions of `compromise`. There are versions for French, German, Italian and Spanish
232
-
233
222
  ### Recognition
234
223
  * Low latency recognition mode. Make the partial transcription available as fast as possible
235
224
  * Live input / microphone recognition
@@ -246,6 +235,7 @@
246
235
  * Align audio file to audio file
247
236
  * Alignment with speech translation assistance, which would enable multilingual subtitle replacement for translated captions
248
237
  * Make `dtw` mode work with more speech synthesizers to produce its reference
238
+ * Predict timing for individual letters (graphemes) based on phoneme timestamps
249
239
 
250
240
  ## Documentation
251
241
 
@@ -285,4 +275,4 @@
285
275
  * Special method to use time stretching to project between different utterances of the same text
286
276
  * Is it possible to combine the Silero speech recognizer and a language model and try to perform Viterbi decoding to find alignments?
287
277
  * Voice replacement
288
- * Predict timing for individual letters (graphemes) based on phoneme timestamps
278
+
package/docs/Technical.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  * Echogarden is written in TypeScript and targets the Node.js platform.
4
4
  * It uses ESM modules and latest ECMAScript and TypeScript features.
5
- * It does not depend on essential binary executables. Instead, all of its engines either use pure JavaScript, WebAssembly, WASI, or the ONNX runtime (with some exceptions: the CLI does invoke a few binary executables, loaded from expansion packages, for the `SoX` and `ffmpeg` tools.Using expansion packages simplifies the installation and ensures non-buggy version are used. Since SoX `v14.4.2` is broken on Windows, it bundles `v14.4.1`).
5
+ * It does not depend on essential binary executables. Instead, all of its engines either use pure JavaScript, WebAssembly, WASI, or the ONNX runtime, with some exceptions: the CLI does invoke a few binary executables, loaded from expansion packages, for the `SoX` and `ffmpeg` tools. Using expansion packages simplifies the installation and ensures non-buggy version are used. Since SoX `v14.4.2` is broken on Windows, it bundles `v14.4.1`.
6
6
  * It does not depend on essential native node.js modules requiring compilation with `node-gyp`. This greatly simplifies the installation experience for end-users (the ONNX runtime bundles precompiled NAPI modules for all supported platforms - it doesn't require any compilation during its installation).
7
7
 
8
8
  ## Package system
@@ -34,7 +34,7 @@ Currently, the biggest contributors to the size are:
34
34
 
35
35
  `onnxruntime-node` is big because it bundles pre-compiled binaries for multiple platforms. `kuromoji` is large because of its dictionary files and some unessential test code it bundles. The other three packages include large WASM binaries.
36
36
 
37
- So, yes, in the future it may be possible to reduce the core installed size by dynamically installing some of these dependencies, or using modified, "slimmed-down" custom versions.
37
+ So, yes, in the future it may be possible to reduce the core installed size by dynamically installing some of these dependencies, or using modified, "slimmed-down" versions of some packages.
38
38
 
39
39
  ## Since the code is almost all JavaScript and WASM, why can't it just run in a web browser?
40
40
 
package/package.json CHANGED
@@ -1,15 +1,15 @@
1
1
  {
2
2
  "name": "echogarden",
3
- "version": "0.5.1",
4
- "description": "A fully open-source speech system designed with end-users in mind.",
3
+ "version": "0.6.0",
4
+ "description": "An integrated speech system, providing a range of speech generation, recognition and processing tools designed to be directly usable by end-users.",
5
5
  "author": "Rotem Dan",
6
6
  "license": "GPL-3.0-only",
7
7
  "keywords": [
8
+ "speech",
8
9
  "text-to-speech",
9
10
  "speech synthesis",
10
11
  "speech-to-text",
11
12
  "speech recognition",
12
- "speech processing",
13
13
  "speech alignment",
14
14
  "forced alignment",
15
15
  "speech translation",