@remigius42/morg 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,6 +13,8 @@ import { unescapeBackslashCommands } from "./core/backslashCommands.js";
13
13
  import { unescapeTablePipes } from "./core/tablePipes.js";
14
14
  import { parseOrg } from "./core/bracedScripts.js";
15
15
  import { dropUnderscoreBulletGuards, guardUnderscoreBullets } from "./core/underscoreBullets.js";
16
+ import { conversionContext, fragmentSides, readOrgWriteMarkdown, takeOver } from "./presets/hooks.js";
17
+ import { resolveSides } from "./presets/sides.js";
16
18
  // a list item's paragraph after its nested list needs a blank line, or
17
19
  // md reads it as a lazy continuation of the nested list's last item
18
20
  function separateTextAfterNestedList(left, right, parent) {
@@ -29,12 +31,15 @@ function separateTextAfterNestedList(left, right, parent) {
29
31
  * @returns The converted Markdown string.
30
32
  */
31
33
  export function convertOrgToMarkdown(org, options = {}) {
32
- const convertOrg = options.preset?.convertOrg;
33
- return convertOrg
34
- ? convertOrg(org, (fragment, preset) => convertOrgToMarkdown(fragment, { ...options, preset }))
35
- : convertOrgDocument(org, options);
34
+ return convertOrgSides(org, options, resolveSides(options, "org", "markdown"));
36
35
  }
37
- function convertOrgDocument(org, options) {
36
+ function convertOrgSides(org, options, sides) {
37
+ const over = takeOver(sides, "convertOrg");
38
+ return over
39
+ ? over.preset.convertOrg(org, (fragment, preset, carried) => convertOrgSides(fragment, options, fragmentSides(sides, over.side, preset, carried)), conversionContext(over, options))
40
+ : convertOrgDocument(org, options, sides);
41
+ }
42
+ function convertOrgDocument(org, options, sides) {
38
43
  // Phase 1: Parse Org-mode to uniorg-ast
39
44
  // md text has no scripts, so ^:{} is implied there and consumed here
40
45
  // (see markdownToOrg); uniorg misreads `_.` lines (see underscoreBullets)
@@ -53,10 +58,9 @@ function convertOrgDocument(org, options) {
53
58
  unescapeFootnoteReferences(uniorgAst);
54
59
  unescapeBackslashCommands(uniorgAst);
55
60
  unescapeTablePipes(uniorgAst);
56
- // Phase 2: Extract dialect preset conventions, if any
57
- if (options.preset?.extractFromUniorg) {
58
- uniorgAst = options.preset.extractFromUniorg(uniorgAst);
59
- }
61
+ // Phase 2: Extract dialect preset conventions, if any: read the org
62
+ // dialect, then write the md dialect
63
+ uniorgAst = readOrgWriteMarkdown(uniorgAst, sides);
60
64
  // Phase 3: Generic uniorg-ast to mdast transformation
61
65
  const mdast = transformUniorgAstToMdast(uniorgAst, {
62
66
  ...(options.preserveOrgisms !== undefined && {
@@ -0,0 +1,49 @@
1
+ import type { OrgData } from "uniorg";
2
+ import type { Sides } from "./sides.js";
3
+ import type { ConversionContext, Preset } from "./types.js";
4
+ /**
5
+ * md→org's dialect step: reads the Markdown dialect out of the generic
6
+ * tree, then writes the org dialect into it.
7
+ * @param uniorgAst The tree the generic transform produced.
8
+ * @param sides The preset per side.
9
+ * @returns The tree in the org dialect.
10
+ */
11
+ export declare function readMarkdownWriteOrg(uniorgAst: OrgData, { input, output }: Sides): OrgData;
12
+ /**
13
+ * org→md's dialect step: reads the org dialect out of the parsed tree,
14
+ * then writes the Markdown dialect into it.
15
+ * @param uniorgAst The parsed org tree.
16
+ * @param sides The preset per side.
17
+ * @returns The tree, ready for the generic transform.
18
+ */
19
+ export declare function readOrgWriteMarkdown(uniorgAst: OrgData, { input, output }: Sides): OrgData;
20
+ /** A preset that takes a whole conversion over, and the side it is on. */
21
+ export interface TakeOver {
22
+ preset: Preset;
23
+ side: ConversionContext["side"];
24
+ }
25
+ /**
26
+ * Finds the preset that takes the conversion over, if any: the input's,
27
+ * else the output's.
28
+ * @param sides The preset per side.
29
+ * @param hook The hook that takes this direction over.
30
+ * @returns The preset and its side, or `undefined` for none.
31
+ */
32
+ export declare function takeOver({ input, output }: Sides, hook: "convertOrg" | "convertMarkdown"): TakeOver | undefined;
33
+ /**
34
+ * The sides a fragment converts with: the fragment's preset on the
35
+ * taking-over preset's side or sides, the other side as it was.
36
+ * @param sides The conversion's preset per side.
37
+ * @param side The taking-over preset's side.
38
+ * @param preset The fragment's preset.
39
+ * @param carried The preset for the other side where it is Vanilla.
40
+ * @returns The fragment's preset per side.
41
+ */
42
+ export declare function fragmentSides(sides: Sides, side: ConversionContext["side"], preset: Preset | undefined, carried?: Preset): Sides;
43
+ /**
44
+ * The context a taking-over preset converts in.
45
+ * @param over The taking-over preset and its side.
46
+ * @param options The conversion's options.
47
+ * @returns The context.
48
+ */
49
+ export declare function conversionContext(over: TakeOver, options: Pick<ConversionContext, "onWarning" | "orgismKeys">): ConversionContext;
@@ -0,0 +1,72 @@
1
+ /**
2
+ * md→org's dialect step: reads the Markdown dialect out of the generic
3
+ * tree, then writes the org dialect into it.
4
+ * @param uniorgAst The tree the generic transform produced.
5
+ * @param sides The preset per side.
6
+ * @returns The tree in the org dialect.
7
+ */
8
+ export function readMarkdownWriteOrg(uniorgAst, { input, output }) {
9
+ const read = input?.markdown?.read?.org ?? same;
10
+ const write = output?.org?.write ?? same;
11
+ return write(read(uniorgAst));
12
+ }
13
+ /**
14
+ * org→md's dialect step: reads the org dialect out of the parsed tree,
15
+ * then writes the Markdown dialect into it.
16
+ * @param uniorgAst The parsed org tree.
17
+ * @param sides The preset per side.
18
+ * @returns The tree, ready for the generic transform.
19
+ */
20
+ export function readOrgWriteMarkdown(uniorgAst, { input, output }) {
21
+ const read = input?.org?.read ?? same;
22
+ const write = output?.markdown?.write ?? same;
23
+ return write(read(uniorgAst));
24
+ }
25
+ function same(uniorgAst) {
26
+ return uniorgAst;
27
+ }
28
+ /**
29
+ * Finds the preset that takes the conversion over, if any: the input's,
30
+ * else the output's.
31
+ * @param sides The preset per side.
32
+ * @param hook The hook that takes this direction over.
33
+ * @returns The preset and its side, or `undefined` for none.
34
+ */
35
+ export function takeOver({ input, output }, hook) {
36
+ if (input?.[hook]) {
37
+ return {
38
+ preset: input,
39
+ side: input.name === output?.name ? "both" : "input"
40
+ };
41
+ }
42
+ return output?.[hook] ? { preset: output, side: "output" } : undefined;
43
+ }
44
+ /**
45
+ * The sides a fragment converts with: the fragment's preset on the
46
+ * taking-over preset's side or sides, the other side as it was.
47
+ * @param sides The conversion's preset per side.
48
+ * @param side The taking-over preset's side.
49
+ * @param preset The fragment's preset.
50
+ * @param carried The preset for the other side where it is Vanilla.
51
+ * @returns The fragment's preset per side.
52
+ */
53
+ export function fragmentSides(sides, side, preset, carried) {
54
+ // only a Vanilla side carries another preset's syntax
55
+ return {
56
+ input: side === "output" ? (sides.input ?? carried) : preset,
57
+ output: side === "input" ? (sides.output ?? carried) : preset
58
+ };
59
+ }
60
+ /**
61
+ * The context a taking-over preset converts in.
62
+ * @param over The taking-over preset and its side.
63
+ * @param options The conversion's options.
64
+ * @returns The context.
65
+ */
66
+ export function conversionContext(over, options) {
67
+ return {
68
+ side: over.side,
69
+ ...(options.onWarning && { onWarning: options.onWarning }),
70
+ ...(options.orgismKeys && { orgismKeys: options.orgismKeys })
71
+ };
72
+ }
@@ -1,50 +1,79 @@
1
1
  import { visit } from "unist-util-visit";
2
2
  import { toString } from "orgast-util-to-string";
3
- import { isScalar, isSeq } from "yaml";
4
- import { fitsKeywordLine, isFrontmatterNode, isModeLine, isModeLineComment, takeFrontmatterEntries } from "../core/frontmatterBlock.js";
3
+ import { isScalar, isSeq, stringify as stringifyYaml } from "yaml";
4
+ import { fitsKeywordLine, FRONTMATTER_BLOCK_BEGIN, frontmatterBlock, isFrontmatterNode, isModeLine, isModeLineComment, takeFrontmatterEntries } from "../core/frontmatterBlock.js";
5
5
  import { keyValueEntries } from "../core/keyValueLines.js";
6
6
  import { tryParse } from "../core/render.js";
7
+ import { orgNodeToText } from "../core/uniorgToMdast/shared.js";
7
8
  import { markdownOutlineToOrg, orgOutlineToMarkdown } from "./logseqOutline.js";
8
9
  /**
9
10
  * Logseq dialect preset: the outline of blocks, page properties, page
10
11
  * and block references, highlights and hiccup.
11
12
  */
12
13
  export function logseq() {
13
- const page = {
14
+ const page = pagePreset();
15
+ const block = blockPreset();
16
+ const presets = {
17
+ page,
18
+ block,
19
+ vanillaReader,
20
+ vanillaPage: pagePropertiesToFrontmatter,
21
+ vanillaInline
22
+ };
23
+ return {
24
+ ...page,
25
+ convertOrg: (org, convert, context) => orgOutlineToMarkdown(org, convert, presets, context),
26
+ convertMarkdown: (markdown, convert, context) => markdownOutlineToOrg(markdown, convert, presets, context)
27
+ };
28
+ }
29
+ // the hooks for a page's properties
30
+ function pagePreset() {
31
+ return {
14
32
  name: "logseq",
15
- applyToMdast: keepPagePropertySource,
16
- applyToUniorg: uniorgAst => {
17
- rewriteLabeledPageRefs(uniorgAst);
18
- takePageProperties(uniorgAst);
19
- return uniorgAst;
33
+ markdown: {
34
+ read: {
35
+ mdast: keepPagePropertySource,
36
+ org: rewriteLabeledPageRefs
37
+ },
38
+ write: uniorgAst => {
39
+ codeToQueryBlocks(uniorgAst);
40
+ pageProperties(uniorgAst);
41
+ return extractInlineSpecifics(uniorgAst);
42
+ }
20
43
  },
21
- extractFromUniorg: uniorgAst => {
22
- pageProperties(uniorgAst);
23
- return extractInlineSpecifics(uniorgAst);
44
+ org: {
45
+ read: readOrgInline,
46
+ write: takePageProperties
24
47
  }
25
48
  };
49
+ }
50
+ // the hooks for a block's content
51
+ function blockPreset() {
26
52
  const bareUrls = new Map();
27
- const block = {
53
+ return {
28
54
  name: "logseq",
29
- applyToMdast: mdast => {
30
- countBareUrls(mdast, bareUrls);
31
- emailLinksToText(mdast);
32
- },
33
- applyToUniorg: uniorgAst => {
34
- rewriteLabeledPageRefs(uniorgAst);
35
- restoreBareUrls(uniorgAst, bareUrls);
36
- return uniorgAst;
55
+ markdown: {
56
+ read: {
57
+ mdast: mdast => {
58
+ countBareUrls(mdast, bareUrls);
59
+ emailLinksToText(mdast);
60
+ },
61
+ org: rewriteLabeledPageRefs
62
+ },
63
+ write: uniorgAst => {
64
+ codeToQueryBlocks(uniorgAst);
65
+ bareUrlsToText(uniorgAst);
66
+ return extractInlineSpecifics(uniorgAst);
67
+ }
37
68
  },
38
- extractFromUniorg: uniorgAst => {
39
- bareUrlsToText(uniorgAst);
40
- return extractInlineSpecifics(uniorgAst);
69
+ org: {
70
+ read: uniorgAst => {
71
+ bareUrlsToText(uniorgAst);
72
+ return readOrgInline(uniorgAst);
73
+ },
74
+ write: uniorgAst => restoreBareUrls(uniorgAst, bareUrls)
41
75
  }
42
76
  };
43
- return {
44
- ...page,
45
- convertOrg: (org, convert) => orgOutlineToMarkdown(org, convert, { page, block }),
46
- convertMarkdown: (markdown, convert) => markdownOutlineToOrg(markdown, convert, { page, block })
47
- };
48
77
  }
49
78
  // Logseq md writes a url bare, as org writes a plain link; the core
50
79
  // carries a md link as a [[url]] bracket link, so md→org counts the
@@ -103,6 +132,7 @@ function restoreBareUrls(uniorgAst, counts) {
103
132
  counts.set(node.rawLink, count - 1);
104
133
  }
105
134
  });
135
+ return uniorgAst;
106
136
  }
107
137
  // org→md: a plain http(s) link stays a bare url, unescaped
108
138
  // an email address is text in Logseq org and written bare in Logseq md,
@@ -163,6 +193,7 @@ function rewriteLabeledPageRefs(uniorgAst) {
163
193
  node.rawLink = page;
164
194
  node.path = page;
165
195
  });
196
+ return uniorgAst;
166
197
  }
167
198
  // Logseq reads a page's first block of `key:: value` lines (md) and its
168
199
  // leading `#+key: value` lines (org) as page properties (ADR 0005)
@@ -204,6 +235,7 @@ function takePageProperties(uniorgAst) {
204
235
  // ahead of the block, so all of them lead the page
205
236
  const at = children.findIndex(node => node.type !== "keyword");
206
237
  children.splice(at === -1 ? children.length : at, 0, ...keywords.map(([key, value]) => ({ type: "keyword", key, value })));
238
+ return uniorgAst;
207
239
  }
208
240
  function takeFrontmatterKeywords(children) {
209
241
  const frontmatter = children.find(isFrontmatterNode);
@@ -264,6 +296,43 @@ function flatValue(value, text) {
264
296
  ? items.join(", ")
265
297
  : null;
266
298
  }
299
+ const PAGE_PROPERTY_LINE_RE = /^#\+([a-z0-9_-]+):(?: (.*))?$/;
300
+ // Vanilla Markdown: a page's properties, its leading lower-case keywords
301
+ // as Logseq writes them, become plain frontmatter, as Markdown tools
302
+ // read metadata (ADR 0006); the way back takes them as page properties
303
+ // (takeFlatEntries). A keyword that acts in Emacs stays one
304
+ function pagePropertiesToFrontmatter(lines) {
305
+ const begin = lines.findIndex(line => line.startsWith(FRONTMATTER_BLOCK_BEGIN));
306
+ const end = begin === -1 ? -1 : lines.indexOf("#+end_comment", begin);
307
+ // a key the frontmatter block has already stays a keyword
308
+ const taken = new Set(lines
309
+ .slice(begin + 1, begin === -1 ? 0 : end)
310
+ .map(line => /^([^\s:#][^:]*):/.exec(line)?.[1]));
311
+ const properties = {};
312
+ const rest = lines.filter((line, i) => {
313
+ const [, key, value = ""] = PAGE_PROPERTY_LINE_RE.exec(line) ?? [];
314
+ if (!key ||
315
+ (i > begin && i < end) ||
316
+ ACTING_KEYWORD_RE.test(key) ||
317
+ taken.has(key) ||
318
+ key in properties) {
319
+ return true;
320
+ }
321
+ properties[key] = value;
322
+ return false;
323
+ });
324
+ if (!Object.keys(properties).length) {
325
+ return lines;
326
+ }
327
+ // unfolded, or a long value would come back as no flat one
328
+ const yaml = stringifyYaml(properties, { lineWidth: 0 }).replace(/\n$/, "");
329
+ const block = frontmatterBlock(yaml).replace(/\n$/, "").split("\n");
330
+ const at = rest.findIndex(line => line.startsWith(FRONTMATTER_BLOCK_BEGIN));
331
+ // ahead of what the block holds, so the page properties lead
332
+ return at === -1
333
+ ? [...block, ...rest]
334
+ : [...rest.slice(0, at + 1), ...block.slice(1, -1), ...rest.slice(at + 1)];
335
+ }
267
336
  // the reverse: leading keywords become the first block, verbatim, keys
268
337
  // lower-cased (uniorg upper-cases them, Logseq reads only lower case)
269
338
  function pageProperties(uniorgAst) {
@@ -288,9 +357,97 @@ function pageProperties(uniorgAst) {
288
357
  contentsEnd: 0
289
358
  });
290
359
  }
291
- function extractInlineSpecifics(uniorgAst) {
360
+ // Logseq org's own inline syntax, carried in its Markdown form, which
361
+ // Vanilla Markdown keeps as text (ADR 0006)
362
+ function readOrgInline(uniorgAst) {
363
+ // first: a query's body is Logseq's query language, not org text
364
+ queryBlocksToCode(uniorgAst);
292
365
  keepVerbatimText(uniorgAst);
293
366
  repairHighlights(uniorgAst);
367
+ return uniorgAst;
368
+ }
369
+ // reading Vanilla Markdown as carrying Logseq Markdown's syntax: a
370
+ // `query` code block is a query block (ADR 0006), which Logseq Markdown
371
+ // writes as the block itself, so its code blocks stay code
372
+ function vanillaReader(preset) {
373
+ const read = preset.markdown?.read;
374
+ return {
375
+ ...preset,
376
+ markdown: {
377
+ ...preset.markdown,
378
+ read: {
379
+ ...read,
380
+ org: uniorgAst => {
381
+ const tree = read?.org?.(uniorgAst) ?? uniorgAst;
382
+ codeToQueryBlocks(tree, "QUERY");
383
+ return tree;
384
+ }
385
+ }
386
+ }
387
+ };
388
+ }
389
+ // a page ref stays a link for the output's Markdown dialect to write
390
+ // (Obsidian's wikilink); Vanilla Markdown carries Logseq's syntax. A
391
+ // hiccup paragraph is taken whole after its links are written, or a
392
+ // page ref in it would be lost
393
+ const vanillaInline = {
394
+ name: "logseq",
395
+ markdown: {
396
+ write: uniorgAst => {
397
+ fuzzyLinksToPageRefs(uniorgAst);
398
+ markHiccupParagraphs(uniorgAst);
399
+ return uniorgAst;
400
+ }
401
+ }
402
+ };
403
+ // a query block: Vanilla Markdown has none, so a code block in the
404
+ // query language carries it (ADR 0006); Logseq Markdown writes the
405
+ // block itself, as Logseq does
406
+ const QUERY_LANGUAGE = "query";
407
+ function queryBlocksToCode(uniorgAst) {
408
+ visit(uniorgAst, "special-block", (node, index, parent) => {
409
+ if (node.blockType.toUpperCase() !== "QUERY" ||
410
+ !parent ||
411
+ index === undefined) {
412
+ return undefined;
413
+ }
414
+ const lines = orgNodeToText(node).split("\n");
415
+ parent.children[index] = {
416
+ type: "src-block",
417
+ affiliated: node.affiliated,
418
+ language: QUERY_LANGUAGE,
419
+ // the block as written, for Logseq Markdown to write it back
420
+ blockType: node.blockType,
421
+ switches: null,
422
+ parameters: null,
423
+ value: `${lines.slice(1, -1).join("\n")}\n`
424
+ };
425
+ return undefined;
426
+ });
427
+ }
428
+ // a query block read as a code block (queryBlocksToCode) goes back as
429
+ // it was written; with a fallback type, so does any `query` code block,
430
+ // as Vanilla Markdown carries a query block
431
+ function codeToQueryBlocks(uniorgAst, fallback) {
432
+ visit(uniorgAst, "src-block", (node, index, parent) => {
433
+ const type = node.blockType ?? fallback;
434
+ if (node.language !== QUERY_LANGUAGE ||
435
+ !type ||
436
+ !parent ||
437
+ index === undefined) {
438
+ return undefined;
439
+ }
440
+ // a Markdown code block's value ends short of its last line break
441
+ const body = node.value.replace(/\n?$/, "\n");
442
+ const [block] = tryParse(`#+begin_${type}\n${body}#+end_${type}\n`)?.children ?? [];
443
+ if (block) {
444
+ parent.children[index] = block;
445
+ }
446
+ return undefined;
447
+ });
448
+ }
449
+ function extractInlineSpecifics(uniorgAst) {
450
+ keepVerbatimText(uniorgAst);
294
451
  fuzzyLinksToPageRefs(uniorgAst);
295
452
  markHiccupParagraphs(uniorgAst);
296
453
  return uniorgAst;
@@ -337,6 +494,7 @@ function repairHighlights(uniorgAst) {
337
494
  }
338
495
  }
339
496
  });
497
+ return uniorgAst;
340
498
  }
341
499
  // Logseq page and block references: [[page]] stays a wikilink,
342
500
  // [[page][label]] becomes [label]([[page]]), [[((uuid))][label]]
@@ -1,22 +1,47 @@
1
- import type { FragmentConverter, Preset } from "./types.js";
1
+ import type { ConversionContext, FragmentConverter, Preset } from "./types.js";
2
+ export type Meta = {
3
+ lines: string[];
4
+ } | {
5
+ key: string;
6
+ value: string;
7
+ };
8
+ /** A block of the outline, read from either format. */
9
+ export interface Block {
10
+ level: number;
11
+ heading: number;
12
+ metaFirst: boolean;
13
+ meta: Meta[];
14
+ content: string[];
15
+ }
16
+ /** A page: its properties' source lines, then its blocks. */
17
+ export interface Outline {
18
+ page: string[];
19
+ blocks: Block[];
20
+ }
2
21
  interface Presets {
3
22
  page: Preset;
4
23
  block: Preset;
24
+ vanillaPage?: (lines: string[]) => string[];
25
+ vanillaReader?: (preset: Preset) => Preset;
26
+ vanillaInline?: Preset;
5
27
  }
6
28
  /**
7
- * Converts a Logseq org page to Logseq Markdown, block by block.
29
+ * Converts a Logseq org page to Markdown, block by block: Logseq's,
30
+ * or Vanilla where the preset is on the input side only.
8
31
  * @param org The org page.
9
32
  * @param convert The core's fragment converter.
10
33
  * @param presets The presets for the page properties and for a block.
34
+ * @param context The side the preset is on.
11
35
  * @returns The Markdown page.
12
36
  */
13
- export declare function orgOutlineToMarkdown(org: string, convert: FragmentConverter, presets: Presets): string;
37
+ export declare function orgOutlineToMarkdown(org: string, convert: FragmentConverter, presets: Presets, context: ConversionContext): string;
14
38
  /**
15
39
  * Converts a Logseq Markdown page to Logseq org, block by block.
16
40
  * @param markdown The Markdown page.
17
41
  * @param convert The core's fragment converter.
18
42
  * @param presets The presets for the page properties and for a block.
43
+ * @param context The side the preset is on.
19
44
  * @returns The org page.
20
45
  */
21
- export declare function markdownOutlineToOrg(markdown: string, convert: FragmentConverter, presets: Presets): string;
46
+ export declare function markdownOutlineToOrg(markdown: string, convert: FragmentConverter, presets: Presets, context: ConversionContext): string;
22
47
  export {};