@remigius42/morg 0.9.1 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,7 +5,8 @@ import { fitsKeywordLine, FRONTMATTER_BLOCK_BEGIN, frontmatterBlock, isFrontmatt
5
5
  import { keyValueEntries } from "../core/keyValueLines.js";
6
6
  import { tryParse } from "../core/render.js";
7
7
  import { orgNodeToText } from "../core/uniorgToMdast/shared.js";
8
- import { markdownOutlineToOrg, orgOutlineToMarkdown } from "./logseqOutline.js";
8
+ import { FUZZY_LINK_RE } from "./links.js";
9
+ import { frontmatterLength, markdownOutlineToOrg, orgOutlineToMarkdown, translateMarkdownOutline, translateOrgOutline } from "./logseqOutline.js";
9
10
  /**
10
11
  * Logseq dialect preset: the outline of blocks, page properties, page
11
12
  * and block references, highlights and hiccup.
@@ -22,8 +23,21 @@ export function logseq() {
22
23
  };
23
24
  return {
24
25
  ...page,
26
+ markdown: {
27
+ ...page.markdown,
28
+ bullet: "-",
29
+ links: {
30
+ read: text => text.replace(LABELED_PAGE_REF_RE, "[[$2][$1]]"),
31
+ write: text => text.replace(FUZZY_LINK_RE, "[$2]([[$1]])")
32
+ }
33
+ },
25
34
  convertOrg: (org, convert, context) => orgOutlineToMarkdown(org, convert, presets, context),
26
- convertMarkdown: (markdown, convert, context) => markdownOutlineToOrg(markdown, convert, presets, context)
35
+ convertMarkdown: (markdown, convert, context) => markdownOutlineToOrg(markdown, convert, presets, context),
36
+ translateOrg: translateOrgOutline,
37
+ translateMarkdown: (markdown, context) => translateMarkdownOutline(markdown, context, {
38
+ vanilla: propertiesToFrontmatter,
39
+ logseq: frontmatterToProperties
40
+ })
27
41
  };
28
42
  }
29
43
  // the hooks for a page's properties
@@ -47,6 +61,8 @@ function pagePreset() {
47
61
  }
48
62
  };
49
63
  }
64
+ // a labeled page ref in Logseq Markdown
65
+ const LABELED_PAGE_REF_RE = /\[([^\][]+)\]\(\[\[([^\][]+)\]\]\)/g;
50
66
  // the hooks for a block's content
51
67
  function blockPreset() {
52
68
  const bareUrls = new Map();
@@ -333,6 +349,64 @@ function pagePropertiesToFrontmatter(lines) {
333
349
  ? [...block, ...rest]
334
350
  : [...rest.slice(0, at + 1), ...block.slice(1, -1), ...rest.slice(at + 1)];
335
351
  }
352
+ const MD_PAGE_PROPERTY_KEY_RE = /^[\w.-]+$/;
353
+ // a Logseq Markdown page's lines: a frontmatter, if any, and the
354
+ // properties below it, up to a blank line
355
+ function splitPageLines(lines) {
356
+ const frontmatter = frontmatterLength(lines);
357
+ const blank = lines.indexOf("", frontmatter);
358
+ const end = blank === -1 ? lines.length : blank;
359
+ return {
360
+ frontmatter: lines.slice(0, frontmatter),
361
+ properties: keyValueEntries(lines.slice(frontmatter, end).join("\n")),
362
+ rest: lines.slice(end)
363
+ };
364
+ }
365
+ // Logseq Markdown → Vanilla Markdown: page properties as plain
366
+ // frontmatter, as Markdown tools read metadata (ADR 0006); no org is
367
+ // in the way, so every key maps
368
+ function propertiesToFrontmatter(lines) {
369
+ const { frontmatter, properties, rest } = splitPageLines(lines);
370
+ if (!properties?.length) {
371
+ return lines;
372
+ }
373
+ const yaml = stringifyYaml(Object.fromEntries(properties), {
374
+ lineWidth: 0
375
+ }).replace(/\n$/, "");
376
+ return [
377
+ "---",
378
+ ...frontmatter.slice(1, -1),
379
+ ...yaml.split("\n"),
380
+ "---",
381
+ ...rest
382
+ ];
383
+ }
384
+ // the reverse: flat entries are page properties, written as Logseq
385
+ // writes them; what Logseq reads no property from stays frontmatter
386
+ function frontmatterToProperties(lines) {
387
+ if (lines[0] !== "---") {
388
+ return lines;
389
+ }
390
+ const end = lines.indexOf("---", 1);
391
+ const { keywords, yaml } = takeFrontmatterEntries(lines.slice(1, end).join("\n"), (key, value, text) => {
392
+ const flat = flatValue(value, text);
393
+ return flat !== null &&
394
+ MD_PAGE_PROPERTY_KEY_RE.test(text(key)) &&
395
+ !flat.includes("\n")
396
+ ? [[text(key), flat]]
397
+ : null;
398
+ });
399
+ if (!keywords.length) {
400
+ return lines;
401
+ }
402
+ const rest = yaml.replace(/\n+$/, "");
403
+ return [
404
+ ...(rest.trim() ? ["---", rest, "---"] : []),
405
+ ...keywords.map(([key, value]) => `${key}::${value ? ` ${value}` : ""}`),
406
+ "",
407
+ ...lines.slice(end + 1)
408
+ ];
409
+ }
336
410
  // the reverse: leading keywords become the first block, verbatim, keys
337
411
  // lower-cased (uniorg upper-cases them, Logseq reads only lower case)
338
412
  function pageProperties(uniorgAst) {
@@ -25,6 +25,16 @@ interface Presets {
25
25
  vanillaReader?: (preset: Preset) => Preset;
26
26
  vanillaInline?: Preset;
27
27
  }
28
+ interface MarkdownPages {
29
+ vanilla: (lines: string[]) => string[];
30
+ logseq: (lines: string[]) => string[];
31
+ }
32
+ /**
33
+ * How many lines a Markdown page's leading frontmatter takes, 0 for none.
34
+ * @param lines The page's lines.
35
+ * @returns The frontmatter's line count, its fences included.
36
+ */
37
+ export declare function frontmatterLength(lines: string[]): number;
28
38
  /**
29
39
  * Converts a Logseq org page to Markdown, block by block: Logseq's,
30
40
  * or Vanilla where the preset is on the input side only.
@@ -44,4 +54,24 @@ export declare function orgOutlineToMarkdown(org: string, convert: FragmentConve
44
54
  * @returns The org page.
45
55
  */
46
56
  export declare function markdownOutlineToOrg(markdown: string, convert: FragmentConverter, presets: Presets, context: ConversionContext): string;
57
+ /**
58
+ * Translates an org page between Logseq org and Vanilla org, which
59
+ * differ in their headlines only (ADR 0006): a block's lines are kept,
60
+ * but for its first line, on its stars' line in Logseq org and below an
61
+ * empty title in Vanilla org where it starts an element.
62
+ * @param org The org page.
63
+ * @param context The side Logseq is on.
64
+ * @returns The page in the other dialect.
65
+ */
66
+ export declare function translateOrgOutline(org: string, context: ConversionContext): string;
67
+ /**
68
+ * Translates a Markdown page between Logseq Markdown and Vanilla
69
+ * Markdown: blocks become list items and headings and back, their meta
70
+ * as Vanilla Markdown writes it (ADR 0006); a block's content is kept.
71
+ * @param markdown The Markdown page.
72
+ * @param context The side Logseq is on.
73
+ * @param pages What writes a page's lines in either dialect.
74
+ * @returns The page in the other dialect.
75
+ */
76
+ export declare function translateMarkdownOutline(markdown: string, context: ConversionContext, pages: MarkdownPages): string;
47
77
  export {};
@@ -2,6 +2,7 @@ import { consumesBracedScripts } from "../core/bracedScripts.js";
2
2
  import { mayBeLineSyntax, readsAsLineSyntax } from "../core/lineSyntax.js";
3
3
  import { positionParser, tryParse } from "../core/render.js";
4
4
  import { ZERO_WIDTH_SPACE } from "../core/markupBoundary.js";
5
+ import { mapOutsideCode } from "../core/outsideCode.js";
5
6
  import { isDrawerStart, isOrgBlockStart, orgElementEnd } from "../core/passthroughSource.js";
6
7
  import { readVanillaMarkdownOutline } from "./logseqVanillaMarkdown.js";
7
8
  // Logseq stores a page as an outline of blocks, each block a content
@@ -20,6 +21,7 @@ const STATE_LINE_RE = /^[-*] (?=State ")/;
20
21
  const MD_HEADING_RE = /^(#{1,6})(?: (.*))?$/;
21
22
  const HEADLINE_RE = /^\*+ /;
22
23
  const FENCE_RE = /^\s*(?:```|~~~)/;
24
+ const QUERY_LANGUAGE = "query";
23
25
  // md→org adds it for the text's bare `_` and `^`, which a block's
24
26
  // content holds as Logseq writes it
25
27
  const BRACED_SCRIPTS_LINE = "#+OPTIONS: ^:{}";
@@ -149,19 +151,26 @@ function readOrgBlock({ level, lines }, joined = false) {
149
151
  content: metaFirst ? body : [first, ...body]
150
152
  };
151
153
  }
152
- function readOrgOutline(org, vanilla) {
153
- const { page, blocks } = splitBlocks(org.replace(/\r?\n$/, "").split(/\r?\n/), line => {
154
+ function splitOrgBlocks(org) {
155
+ return splitBlocks(org.replace(/\r?\n$/, "").split(/\r?\n/), line => {
154
156
  const match = ORG_BLOCK_RE.exec(line);
155
157
  return match ? [match[1]?.length ?? 0, match[2] ?? ""] : null;
156
158
  });
159
+ }
160
+ // Vanilla org: a first line written below an empty title, as it starts
161
+ // an element, is the block's first line again
162
+ function joinTitle(lines) {
163
+ const [title, ...rest] = lines;
164
+ const joined = title === "" && startsElement(rest);
165
+ return { lines: joined ? rest : lines, joined };
166
+ }
167
+ function readOrgOutline(org, vanilla) {
168
+ const { page, blocks } = splitOrgBlocks(org);
157
169
  return {
158
170
  page,
159
171
  blocks: blocks.map(({ level, lines }) => {
160
- // Vanilla org: a first line written below an empty title, as it
161
- // starts an element, is the block's first line again
162
- const [title, ...rest] = lines;
163
- const joined = vanilla && title === "" && startsElement(rest);
164
- return readOrgBlock({ level, lines: joined ? rest : lines }, joined);
172
+ const read = vanilla ? joinTitle(lines) : { lines, joined: false };
173
+ return readOrgBlock({ level, lines: read.lines }, read.joined);
165
174
  })
166
175
  };
167
176
  }
@@ -199,10 +208,18 @@ function readMarkdownBlock({ level, lines: source }) {
199
208
  content: metaFirst ? body : [titleLine, ...body]
200
209
  };
201
210
  }
211
+ /**
212
+ * How many lines a Markdown page's leading frontmatter takes, 0 for none.
213
+ * @param lines The page's lines.
214
+ * @returns The frontmatter's line count, its fences included.
215
+ */
216
+ export function frontmatterLength(lines) {
217
+ return lines[0] === "---" ? lines.indexOf("---", 1) + 1 : 0;
218
+ }
202
219
  function readMarkdownOutline(markdown) {
203
220
  const lines = markdown.replace(/\r?\n$/, "").split(/\r?\n/);
204
221
  // a leading frontmatter is page content, its `- ` lines yaml items
205
- const frontmatter = lines.slice(0, lines[0] === "---" ? lines.indexOf("---", 1) + 1 : 0);
222
+ const frontmatter = lines.slice(0, frontmatterLength(lines));
206
223
  let fenced = false;
207
224
  const { page, blocks } = splitBlocks(lines.slice(frontmatter.length), line => {
208
225
  const match = MD_BLOCK_RE.exec(line);
@@ -285,23 +302,32 @@ function startsElement(lines) {
285
302
  // an empty title keeps the space Logseq org leaves out, and a first line
286
303
  // that starts an element goes below the stars (ADR 0006)
287
304
  function writeOrgBlock(block, vanilla = false) {
288
- const lines = arrange(block.metaFirst, orgMetaLines(block.meta, block.heading), block.content);
305
+ return writeOrgLines({
306
+ level: block.level,
307
+ lines: arrange(block.metaFirst, orgMetaLines(block.meta, block.heading), block.content)
308
+ }, vanilla);
309
+ }
310
+ function writeOrgLines({ level, lines }, vanilla) {
289
311
  const [title = "", ...more] = vanilla && startsElement(lines) ? ["", ...lines] : lines;
290
312
  return [
291
- `${"*".repeat(block.level)}${title || vanilla ? ` ${title}` : ""}`,
313
+ `${"*".repeat(level)}${title || vanilla ? ` ${title}` : ""}`,
292
314
  ...more
293
315
  ].join("\n");
294
316
  }
295
317
  function writeOutline({ page, blocks }, writeBlock) {
296
318
  return [...page, ...blocks.map(writeBlock)].join("\n").concat("\n");
297
319
  }
298
- // a block's title is a headline's, which Logseq reads as inline text
299
- // only; converted with the lines below it, it must not read as a list,
320
+ // a table or a rule, which Logseq reads on a headline's line
321
+ const TITLE_ELEMENT_RE = /^(\||-{5,}\s*$)/;
322
+ // a block's title is a headline's, inline text but for a table or a
323
+ // rule; converted with the lines below it, it must not read as a list,
300
324
  // a headline or a fixed-width line, so an escape the core drops keeps
301
325
  // it text (ADR 0006)
302
326
  function inlineTitle(block) {
303
327
  const [title = "", ...rest] = block.content;
304
- return !block.metaFirst && readsAsLineSyntax(title)
328
+ return !block.metaFirst &&
329
+ !TITLE_ELEMENT_RE.test(title) &&
330
+ readsAsLineSyntax(title)
305
331
  ? [`${ZERO_WIDTH_SPACE}${title}`, ...rest]
306
332
  : block.content;
307
333
  }
@@ -421,13 +447,18 @@ function writeVanillaMarkdownOutline({ page, blocks }, context) {
421
447
  export function orgOutlineToMarkdown(org, convert, presets, context) {
422
448
  const { page, blocks } = readOrgOutline(org, context.side === "output");
423
449
  const vanilla = context.side === "input";
450
+ if (context.side === "output") {
451
+ warnMisread(splitOrgBlocks(org).blocks, context, MISREAD_IN_MARKDOWN);
452
+ }
453
+ // read as Logseq org, a block's title is inline text (ADR 0006)
454
+ const inline = context.side !== "output";
424
455
  const pageLines = vanilla ? (presets.vanillaPage?.(page) ?? page) : page;
425
456
  const convertCarried = (fragment, preset) => convert(fragment, preset, vanilla ? presets.vanillaInline : undefined);
426
457
  const outline = {
427
458
  page: convertPage(pageLines, convertCarried, presets.page),
428
459
  blocks: blocks.map(block => ({
429
460
  ...block,
430
- content: convertContent(withBracedScripts(vanilla ? inlineTitle(block) : block.content), convertCarried, presets.block)
461
+ content: convertContent(withBracedScripts(inline ? inlineTitle(block) : block.content), convertCarried, presets.block)
431
462
  }))
432
463
  };
433
464
  return vanilla
@@ -461,7 +492,270 @@ export function markdownOutlineToOrg(markdown, convert, presets, context) {
461
492
  ...block,
462
493
  content: convertMarkdownContent(block.content, convertCarried, presets.block).filter(line => line !== BRACED_SCRIPTS_LINE)
463
494
  };
464
- return vanilla ? dropTitleEscape(converted) : converted;
495
+ return context.side === "input" ? converted : dropTitleEscape(converted);
465
496
  })
466
497
  }, block => writeOrgBlock(block, context.side === "input"));
467
498
  }
499
+ // Logseq's task markers (mldoc's), the done ones after the bar, for
500
+ // Emacs, which knows TODO and DONE only; Logseq reads no such line
501
+ const TODO_LINE = "#+TODO: TODO NOW LATER DOING WAIT WAITING IN-PROGRESS STARTED | DONE CANCELED CANCELLED";
502
+ const EMACS_UNKNOWN_MARKER_RE = /^(NOW|LATER|DOING|WAIT|WAITING|IN-PROGRESS|STARTED|CANCELED|CANCELLED)(?: |$)/;
503
+ const KEYWORD_LINE_RE = /^#\+\S+:/;
504
+ const OWN_TODO_LINE_RE = /^#\+(?:SEQ_|TYP_)?TODO:(.*)$/i;
505
+ // the markers a #+TODO: line declares, without their keys
506
+ function declaredMarkers(line) {
507
+ const [, markers = ""] = OWN_TODO_LINE_RE.exec(line) ?? [];
508
+ return markers
509
+ .split(/\s+/)
510
+ .map(marker => marker.replace(/\(.*\)$/, ""))
511
+ .filter(marker => marker && marker !== "|");
512
+ }
513
+ const LOGSEQ_MARKERS = new Set(declaredMarkers(TODO_LINE));
514
+ // Vanilla org: a page whose blocks use a marker Emacs does not know,
515
+ // nor the page's own #+TODO: lines, names them all after its leading
516
+ // keywords
517
+ function withTodoLine(page, blocks) {
518
+ const declared = new Set(page.flatMap(declaredMarkers));
519
+ const undeclared = blocks.some(({ lines }) => {
520
+ const [, marker = ""] = EMACS_UNKNOWN_MARKER_RE.exec(lines[0] ?? "") ?? [];
521
+ return marker && !declared.has(marker);
522
+ });
523
+ if (!undeclared) {
524
+ return page;
525
+ }
526
+ const end = page.findIndex(line => !KEYWORD_LINE_RE.test(line));
527
+ const at = end === -1 ? page.length : end;
528
+ return [...page.slice(0, at), TODO_LINE, ...page.slice(at)];
529
+ }
530
+ // Logseq org: the line is Vanilla org's alone; Logseq reads an own one
531
+ // as a page property, and its markers as text, unless they are its own
532
+ function withoutTodoLine(page, context) {
533
+ for (const line of page) {
534
+ const unknown = declaredMarkers(line).filter(marker => !LOGSEQ_MARKERS.has(marker));
535
+ if (unknown.length) {
536
+ context.onWarning?.(`Logseq reads no #+TODO: line; it shows ${unknown.join(", ")} as text`);
537
+ }
538
+ }
539
+ return page.filter(line => line !== TODO_LINE);
540
+ }
541
+ const PROPERTY_LINE_RE = /^(\s*):([^\s:]+):(?:\s+(.*?))?\s*$/;
542
+ // a block Logseq shows collapsed is one Emacs shows folded, a property
543
+ // in either's own terms
544
+ function foldedProperties(lines, vanilla, context) {
545
+ let drawer = false;
546
+ return lines.map(line => {
547
+ const match = PROPERTY_LINE_RE.exec(line);
548
+ const name = match?.[2]?.toUpperCase();
549
+ drawer = name === "PROPERTIES" || (drawer && name !== "END");
550
+ return drawer && match ? foldedProperty(match, vanilla, context) : line;
551
+ });
552
+ }
553
+ // a drawer's property in the other dialect; Logseq has none of
554
+ // Emacs's other visibilities
555
+ function foldedProperty([line, indent = "", key = "", value = ""], vanilla, context) {
556
+ const name = key.toUpperCase();
557
+ if (vanilla) {
558
+ return name === "COLLAPSED" && value === "true"
559
+ ? `${indent}:VISIBILITY: folded`
560
+ : line;
561
+ }
562
+ if (name !== "VISIBILITY") {
563
+ return line;
564
+ }
565
+ if (value === "folded") {
566
+ return `${indent}:collapsed: true`;
567
+ }
568
+ context.onWarning?.(`Logseq has no VISIBILITY ${value}; kept as a property`);
569
+ return line;
570
+ }
571
+ // what Logseq (mldoc) reads otherwise than Emacs does: a search link
572
+ // as a page ref, a radio target as a target and text; and every
573
+ // keyword line on the page as a page property
574
+ const MISREAD = [
575
+ [
576
+ /\[\[[*#][^\]]*\](?:\[[^\]]*\])?\]/g,
577
+ n => `Logseq reads ${count(n, "[[*heading]] or [[#custom-id]] link")} as refs to pages of that name`
578
+ ],
579
+ [
580
+ /\[\[id:[^\]]*\]\]/g,
581
+ n => `Logseq reads ${count(n, "[[id:…]] link")} without a label as a ref to a page of that name`
582
+ ],
583
+ [/<<<[^<>]+>>>/g, n => `Logseq misreads ${count(n, "<<<radio>>> target")}`],
584
+ [
585
+ /^\s*#\+[^\s:]+:/g,
586
+ n => `Logseq takes ${count(n, "#+KEY: line")} below the first headline for a page property`
587
+ ]
588
+ ];
589
+ // in Logseq md, where the others are Markdown links or text
590
+ const MISREAD_IN_MARKDOWN = [
591
+ [
592
+ /\[\[\*[^\]]*\](?:\[[^\]]*\])?\]/g,
593
+ n => `Logseq reads ${count(n, "[[*heading]] link")} as refs to pages of that name`
594
+ ]
595
+ ];
596
+ const ORG_BLOCK_BOUNDARY_RE = /^\s*#\+(BEGIN|END)_(\S+)/i;
597
+ function count(n, noun) {
598
+ return `${n} ${noun}${n === 1 ? "" : "s"}`;
599
+ }
600
+ // Logseq org or md from Vanilla org: what Emacs constructs Logseq
601
+ // misreads, outside org blocks, whose content is no markup
602
+ function warnMisread(blocks, context, misread) {
603
+ const counts = misread.map(() => 0);
604
+ let block = "";
605
+ for (const line of blocks.flatMap(({ lines }) => lines)) {
606
+ const [, boundary = "", name = ""] = ORG_BLOCK_BOUNDARY_RE.exec(line) ?? [];
607
+ if (block || boundary) {
608
+ const end = boundary.toUpperCase() === "END" && name.toUpperCase() === block;
609
+ block = end ? "" : block || name.toUpperCase();
610
+ continue;
611
+ }
612
+ misread.forEach(([pattern], i) => {
613
+ counts[i] = (counts[i] ?? 0) + (line.match(pattern)?.length ?? 0);
614
+ });
615
+ }
616
+ misread.forEach(([, message], i) => {
617
+ if (counts[i]) {
618
+ context.onWarning?.(message(counts[i]));
619
+ }
620
+ });
621
+ }
622
+ /**
623
+ * Translates an org page between Logseq org and Vanilla org, which
624
+ * differ in their headlines only (ADR 0006): a block's lines are kept,
625
+ * but for its first line, on its stars' line in Logseq org and below an
626
+ * empty title in Vanilla org where it starts an element.
627
+ * @param org The org page.
628
+ * @param context The side Logseq is on.
629
+ * @returns The page in the other dialect.
630
+ */
631
+ export function translateOrgOutline(org, context) {
632
+ if (!org) {
633
+ return org;
634
+ }
635
+ const vanilla = context.side === "input";
636
+ const { page, blocks } = splitOrgBlocks(org);
637
+ if (!vanilla) {
638
+ warnMisread(blocks, context, MISREAD);
639
+ }
640
+ return [
641
+ ...(vanilla ? withTodoLine(page, blocks) : withoutTodoLine(page, context)),
642
+ ...blocks.map(({ level, lines }) => writeOrgLines({
643
+ level,
644
+ lines: foldedProperties(vanilla ? lines : joinTitle(lines).lines, vanilla, context)
645
+ }, vanilla))
646
+ ]
647
+ .join("\n")
648
+ .concat("\n");
649
+ }
650
+ const ORG_BLOCK_START_RE = /^#\+begin_(\S+)(?:\s+(.*?))?\s*$/i;
651
+ // a fence the lines hold no run of backticks as long as
652
+ function fenceFor(lines) {
653
+ const longest = Math.max(2, ...lines.flatMap(line => line.match(/`+/g) ?? []).map(run => run.length));
654
+ return "`".repeat(longest + 1);
655
+ }
656
+ // Vanilla Markdown: a block's org blocks as Markdown writes them, a
657
+ // source or example block fenced, a quote quoted, a query in a `query`
658
+ // code block (ADR 0006); others have no Markdown form and stay
659
+ function orgBlockToMarkdown(lines) {
660
+ const [, type = "", parameters = ""] = ORG_BLOCK_START_RE.exec(lines[0] ?? "") ?? [];
661
+ const body = lines.slice(1, -1);
662
+ const info = {
663
+ SRC: parameters,
664
+ EXAMPLE: "",
665
+ QUERY: QUERY_LANGUAGE
666
+ }[type.toUpperCase()];
667
+ if (info !== undefined) {
668
+ const fence = fenceFor(body);
669
+ return [`${fence}${info}`, ...body, fence];
670
+ }
671
+ return type.toUpperCase() === "QUOTE"
672
+ ? body.map(line => (line ? `> ${line}` : ">"))
673
+ : lines;
674
+ }
675
+ function orgBlocksToMarkdown(content) {
676
+ const result = [];
677
+ let fenced = false;
678
+ for (let i = 0; i < content.length; i++) {
679
+ const line = content[i] ?? "";
680
+ fenced = FENCE_RE.test(line) ? !fenced : fenced;
681
+ const end = fenced || !isOrgBlockStart(line) ? -1 : orgElementEnd(content, i);
682
+ if (end === -1) {
683
+ result.push(line);
684
+ continue;
685
+ }
686
+ result.push(...orgBlockToMarkdown(content.slice(i, end + 1)));
687
+ i = end;
688
+ }
689
+ return result;
690
+ }
691
+ const QUERY_FENCE_RE = /^(`{3,}|~{3,})\s*query\s*$/;
692
+ // Logseq Markdown: a `query` code block is a query block, which Logseq
693
+ // Markdown writes as the block itself (ADR 0006)
694
+ function queryCodeToBlocks(content) {
695
+ const result = [];
696
+ let fence = "";
697
+ for (const line of content) {
698
+ const query = QUERY_FENCE_RE.exec(line)?.[1];
699
+ if (!fence && query) {
700
+ fence = query;
701
+ result.push("#+BEGIN_QUERY");
702
+ }
703
+ else if (fence && line.trim() === fence) {
704
+ fence = "";
705
+ result.push("#+END_QUERY");
706
+ }
707
+ else {
708
+ result.push(line);
709
+ }
710
+ }
711
+ return result;
712
+ }
713
+ // Logseq Markdown: spaces after a bullet the whole content shares, which
714
+ // a Vanilla list item would read as its content's column
715
+ function dedentCommon(content) {
716
+ const indent = Math.min(...content
717
+ .filter(line => line.trim())
718
+ .map(line => /^ */.exec(line)?.[0].length ?? 0));
719
+ return Number.isFinite(indent) && indent
720
+ ? content.map(line => line.slice(Math.min(indent, line.length)))
721
+ : content;
722
+ }
723
+ // a translation's page links, but in code
724
+ function relinkContent(content, relink) {
725
+ return relink
726
+ ? mapOutsideCode(content.join("\n"), relink).split("\n")
727
+ : content;
728
+ }
729
+ /**
730
+ * Translates a Markdown page between Logseq Markdown and Vanilla
731
+ * Markdown: blocks become list items and headings and back, their meta
732
+ * as Vanilla Markdown writes it (ADR 0006); a block's content is kept.
733
+ * @param markdown The Markdown page.
734
+ * @param context The side Logseq is on.
735
+ * @param pages What writes a page's lines in either dialect.
736
+ * @returns The page in the other dialect.
737
+ */
738
+ export function translateMarkdownOutline(markdown, context, pages) {
739
+ if (!markdown) {
740
+ return markdown;
741
+ }
742
+ if (context.side === "input") {
743
+ const { page, blocks } = readMarkdownOutline(markdown);
744
+ const text = pages.vanilla(page).join("\n").replace(/\n+$/, "");
745
+ return writeVanillaMarkdownOutline({
746
+ page: text ? [text] : [],
747
+ blocks: blocks.map(block => ({
748
+ ...block,
749
+ content: relinkContent(orgBlocksToMarkdown(dedentCommon(block.content)), context.relink)
750
+ }))
751
+ }, context);
752
+ }
753
+ const { page, blocks } = readVanillaMarkdownOutline(markdown, context);
754
+ return writeOutline({
755
+ page: pages.logseq(page),
756
+ blocks: blocks.map(block => ({
757
+ ...block,
758
+ content: relinkContent(queryCodeToBlocks(block.content), context.relink)
759
+ }))
760
+ }, writeMarkdownBlock);
761
+ }
@@ -22,7 +22,9 @@ function span(node) {
22
22
  (node.position?.end.line ?? 1) - 1
23
23
  ];
24
24
  }
25
- const BULLET_RE = /^\s*(?:[-*+]|\d+[.)])(?: |$)/;
25
+ // a bullet up to its content's column: one to four spaces after it, or
26
+ // one where more start indented code
27
+ const BULLET_RE = /^\s*(?:[-*+]|\d+[.)])(?: {1,4}(?=\S)| |$)/;
26
28
  const CHECKBOX_RE = /^\[([ xX])\](?: |$)/;
27
29
  // the markers Logseq shows unchecked other than TODO, which a task item
28
30
  // writes after its checkbox
@@ -275,7 +277,12 @@ export function readVanillaMarkdownOutline(markdown, context) {
275
277
  // text below a heading is its body; after a list, a block of its own
276
278
  function readText(node, level, body, reader) {
277
279
  const [start, end] = span(node);
278
- const source = reader.lines.slice(start, end + 1);
280
+ // a rule's source may be bulleted (`- ---`), a list in a block
281
+ // read from its own column, as an item's content is
282
+ const column = (node.position?.start.column ?? 1) - 1;
283
+ const source = node.type === "thematicBreak"
284
+ ? ["---"]
285
+ : reader.lines.slice(start, end + 1).map(line => dedent(line, column));
279
286
  if (body) {
280
287
  // the body's paragraphs keep the blank lines between them
281
288
  body.content.push(...(body.content.length > 1 ? [""] : []), ...source);
@@ -1,5 +1,7 @@
1
1
  import { visit } from "unist-util-visit";
2
2
  import { toString } from "orgast-util-to-string";
3
+ import { maskCode } from "../core/outsideCode.js";
4
+ import { FUZZY_LINK_RE } from "./links.js";
3
5
  /**
4
6
  * Obsidian dialect preset: `[[Page]]` / `[[Page|alias]]` wikilinks map
5
7
  * to org fuzzy links (`[[Page]]` / `[[Page][alias]]`).
@@ -9,16 +11,28 @@ export function obsidian() {
9
11
  name: "obsidian",
10
12
  markdown: {
11
13
  read: { org: rewriteAliasedWikilinks },
12
- write: fuzzyLinksToWikilinks
13
- }
14
+ write: fuzzyLinksToWikilinks,
15
+ links: {
16
+ read: text => text.replace(ALIASED_PAGE_LINK_RE, "[[$1][$2]]"),
17
+ // a bare `|` would split a table's cell
18
+ write: (text, inTable) => text.replace(FUZZY_LINK_RE, inTable ? "[[$1\\|$2]]" : "[[$1|$2]]")
19
+ }
20
+ },
21
+ // Vanilla Markdown reads Obsidian's own syntax but for these; the
22
+ // way back has nothing to do
23
+ translateMarkdown: (markdown, context) => context.side === "input" ? toVanilla(markdown) : markdown
14
24
  };
15
25
  }
26
+ const ALIASED_WIKILINK_RE = /\[\[([^\][|]+)\|([^\][]+)\]\]/g;
27
+ // in Markdown text: not an embed, whose `|300` is a size, and with a
28
+ // table cell's escaped pipe (`[[Page\|alias]]`)
29
+ const ALIASED_PAGE_LINK_RE = /(?<!!)\[\[([^\][|\\]+)\\?\|([^\][]+)\]\]/g;
16
30
  // md→org: a wikilink travels as plain text; org already reads `[[Page]]`
17
31
  // as a fuzzy link, only the `[[Page|alias]]` form needs rewriting to
18
32
  // org's `[[Page][alias]]` description syntax
19
33
  function rewriteAliasedWikilinks(uniorgAst) {
20
34
  visit(uniorgAst, "text", (node) => {
21
- node.value = node.value.replace(/\[\[([^\][|]+)\|([^\][]+)\]\]/g, "[[$1][$2]]");
35
+ node.value = node.value.replace(ALIASED_WIKILINK_RE, "[[$1][$2]]");
22
36
  });
23
37
  return uniorgAst;
24
38
  }
@@ -40,3 +54,53 @@ function fuzzyLinksToWikilinks(uniorgAst) {
40
54
  });
41
55
  return uniorgAst;
42
56
  }
57
+ const COMMENT_RE = /%%([\s\S]*?)%%/g;
58
+ const FOOTNOTE_LABEL_RE = /\[\^(\d+)\]/g;
59
+ // Obsidian Markdown → Vanilla Markdown: a comment is an HTML comment,
60
+ // an inline footnote a footnote, numbered on from the page's own, its
61
+ // definition at the end; their delimiters count outside code only, what
62
+ // they hold may be code
63
+ function toVanilla(markdown) {
64
+ let masked = maskCode(markdown);
65
+ const edits = [];
66
+ for (const { index, 0: comment } of masked.matchAll(COMMENT_RE)) {
67
+ const end = index + comment.length;
68
+ edits.push([index, end, `<!--${markdown.slice(index + 2, end - 2)}-->`]);
69
+ // a footnote in a comment is part of it
70
+ masked =
71
+ masked.slice(0, index) + "\0".repeat(comment.length) + masked.slice(end);
72
+ }
73
+ let next = Math.max(0, ...[...markdown.matchAll(FOOTNOTE_LABEL_RE)].map(([, n]) => Number(n))) + 1;
74
+ const definitions = [];
75
+ for (const [start, end] of inlineFootnotes(masked)) {
76
+ definitions.push(`[^${next}]: ${markdown.slice(start + 2, end)}`);
77
+ edits.push([start, end + 1, `[^${next++}]`]);
78
+ }
79
+ let result = markdown;
80
+ for (const [start, end, text] of edits.sort((a, b) => b[0] - a[0])) {
81
+ result = result.slice(0, start) + text + result.slice(end);
82
+ }
83
+ return definitions.length
84
+ ? `${result.replace(/\n*$/, "")}\n\n${definitions.join("\n")}\n`
85
+ : result;
86
+ }
87
+ // each `^[note]`, its brackets balanced: where it starts, and its `]`
88
+ function inlineFootnotes(text) {
89
+ const notes = [];
90
+ for (let start = text.indexOf("^["); start !== -1;) {
91
+ let depth = 0;
92
+ let end = start + 1;
93
+ for (; end < text.length; end++) {
94
+ depth += text[end] === "[" ? 1 : text[end] === "]" ? -1 : 0;
95
+ if (!depth) {
96
+ break;
97
+ }
98
+ }
99
+ if (end === text.length) {
100
+ break;
101
+ }
102
+ notes.push([start, end]);
103
+ start = text.indexOf("^[", end + 1);
104
+ }
105
+ return notes;
106
+ }