@remigius42/morg 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -17,8 +17,8 @@ Bidirectional **Markdown ↔ Org-mode** converter, built on the
17
17
 
18
18
  morg treats Org as a canonical plain-text format and Markdown (Obsidian,
19
19
  generic) as the interop surface. Dialect conventions, such as
20
- [Logseq](https://docs.logseq.com/)'s `heading::` properties and outline
21
- nesting, are supported via presets.
20
+ [Logseq](https://docs.logseq.com/)'s outline of blocks and page
21
+ properties, are supported via presets.
22
22
 
23
23
  ## Round-trip convergence
24
24
 
@@ -198,19 +198,20 @@ preset })`: `preserveOrgisms` default `true`; `useHtml` (default
198
198
  `- [x]`); headings become list items and do not restore on the
199
199
  return trip; anything with priority, tags or content keeps its
200
200
  heading and reports via `onWarning`
201
- - `logseq({ nestUnderHeadings })`: default `true`; content following a
202
- heading nests as child blocks of that heading: paragraphs become child
203
- headlines one level deeper (in Logseq org every outline block is a
204
- headline), other constructs stay in the preceding block's body. The
205
- reverse direction restores headings from `:heading:` properties and
206
- turns plain block headlines back into paragraphs. Hiccup blocks
207
- (`[:div …]`) pass through as plain text and are emitted unescaped in
208
- Markdown. Page properties map both directions: a first block of
201
+ - `logseq()`: a page is Logseq's outline of blocks, converted block by
202
+ block: a headline (stars, a space, the block's content, an empty
203
+ block as the bare stars) ↔ a `-` bullet indented one tab per level,
204
+ its lines below the first two spaces further in. A block's content
205
+ is one fragment, so a code block or table that starts on the
206
+ headline line converts as a whole. `:heading: N` ↔ `- ## …`, a
207
+ block's property drawer ↔ `key:: value` lines; planning lines and
208
+ other drawers (`:LOGBOOK:`) stay as written. A heading outside the
209
+ bullets is a top-level block, as Logseq writes a page's first one.
210
+ Page properties map both directions: a first block of
209
211
  `key:: value` lines and flat frontmatter entries ↔ leading
210
212
  `#+key: value` lines, which Logseq reads as page properties;
211
213
  frontmatter keys that act in Emacs (`todo`, `include`, …) stay inert.
212
- Logseq's own syntax maps both directions: `TODO`/`DONE`
213
- text markers and `[#A]` priorities ↔ org keywords/priorities, page
214
+ Task markers and `[#A]` priorities stay text, page
214
215
  references `[[page]]` and labeled forms `[label]([[page]])` ↔ org
215
216
  fuzzy links `[[page][label]]`, block refs `[label](((uuid)))` ↔
216
217
  `[[((uuid))][label]]`, and `^^highlight^^` markup survives verbatim
@@ -4,6 +4,13 @@ import type { OrgData } from "uniorg";
4
4
  * holds a bare underscore or caret org would read as a script.
5
5
  */
6
6
  export declare function requireBracedScripts(uniorgAst: OrgData): void;
7
+ /**
8
+ * Whether org→md would consume a `^:{}` setting at the head of `org`:
9
+ * its text reads a script org would otherwise take for one.
10
+ * @param org The org text, the setting included.
11
+ * @returns Whether the setting is taken as the one md→org adds.
12
+ */
13
+ export declare function consumesBracedScripts(org: string): boolean;
7
14
  /**
8
15
  * org→md: parses org, honoring its `^:` setting, and consuming `^:{}`
9
16
  * where the text needs it: md→org adds it only then, so anywhere else
@@ -135,6 +135,16 @@ function takeBracedScripts(uniorgAst) {
135
135
  return node.value !== "";
136
136
  });
137
137
  }
138
+ /**
139
+ * Whether org→md would consume a `^:{}` setting at the head of `org`:
140
+ * its text reads a script org would otherwise take for one.
141
+ * @param org The org text, the setting included.
142
+ * @returns Whether the setting is taken as the one md→org adds.
143
+ */
144
+ export function consumesBracedScripts(org) {
145
+ // text without a bare script candidate skips the parse
146
+ return (BARE_SCRIPT_RE.test(org) && readsBareScripts(bracedScriptsParser.parse(org)));
147
+ }
138
148
  /**
139
149
  * org→md: parses org, honoring its `^:` setting, and consuming `^:{}`
140
150
  * where the text needs it: md→org adds it only then, so anywhere else
@@ -6,6 +6,13 @@ import { type Node } from "./render.js";
6
6
  * space, or it would turn into a list item, headline, comment or table.
7
7
  */
8
8
  export declare function escapeLineSyntax(tree: Parent): void;
9
+ /**
10
+ * Whether org reads a text as just one passthrough element.
11
+ * @param text The org text.
12
+ * @returns Whether it is one block, drawer, keyword … org→md writes as
13
+ * its org text.
14
+ */
15
+ export declare function readsAsPassthrough(text: string): boolean;
9
16
  /**
10
17
  * md→org: whether a node is a paragraph org→md wrote as the org text of
11
18
  * one passthrough element, which goes back as it is.
@@ -135,8 +135,16 @@ function isPassthrough(starts) {
135
135
  if (!first || !MAY_BE_LINE_SYNTAX_RE.test(first.line)) {
136
136
  return false;
137
137
  }
138
- const lines = starts.map(({ line }) => line);
139
- const [only, ...rest] = tryParse(`${lines.join("\n")}\n`)?.children ?? [];
138
+ return readsAsPassthrough(starts.map(({ line }) => line).join("\n"));
139
+ }
140
+ /**
141
+ * Whether org reads a text as just one passthrough element.
142
+ * @param text The org text.
143
+ * @returns Whether it is one block, drawer, keyword … org→md writes as
144
+ * its org text.
145
+ */
146
+ export function readsAsPassthrough(text) {
147
+ const [only, ...rest] = tryParse(`${text}\n`)?.children ?? [];
140
148
  return !rest.length && PASSTHROUGH_TYPES.has(only?.type ?? "");
141
149
  }
142
150
  /**
@@ -125,8 +125,11 @@ export function transformMdastCode(node) {
125
125
  value: `${node.value}\n`
126
126
  };
127
127
  }
128
+ // the fence's meta is the block's switches and header arguments,
129
+ // which uniorg-stringify writes after the language
130
+ const language = [node.lang, node.meta].filter(Boolean).join(" ");
128
131
  return (node.lang
129
- ? { type: "src-block", language: node.lang, value: node.value }
132
+ ? { type: "src-block", language, value: node.value }
130
133
  : {
131
134
  type: "example-block",
132
135
  value: node.value
@@ -0,0 +1,21 @@
1
+ import type { Root } from "mdast";
2
+ /** Whether a line opens an org block (`#+begin_name`). */
3
+ export declare function isOrgBlockStart(line: string): boolean;
4
+ /** Whether a line opens an org drawer (`:NAME:`). */
5
+ export declare function isDrawerStart(line: string): boolean;
6
+ /**
7
+ * The line that ends the org block or drawer a line opens.
8
+ * @param lines The lines.
9
+ * @param start The index of the opening line.
10
+ * @returns The index of the `#+end_name` or `:END:` line, or -1.
11
+ */
12
+ export declare function orgElementEnd(lines: string[], start: number): number;
13
+ /**
14
+ * md→org: a passthrough element's org text may hold lines Markdown reads
15
+ * as syntax of its own (a blank line and an indented one, a `#`, `-`
16
+ * or `>` line); the top-level nodes it parsed into become one paragraph
17
+ * of the source text again, which goes back to org as it is.
18
+ * @param mdast The parsed Markdown.
19
+ * @param markdown Its source.
20
+ */
21
+ export declare function keepPassthroughSource(mdast: Root, markdown: string): void;
@@ -0,0 +1,94 @@
1
+ import { readsAsPassthrough } from "./lineSyntax.js";
2
+ // the first line of an org block or drawer
3
+ const BLOCK_START_RE = /^#\+begin_(\S+)/i;
4
+ const DRAWER_START_RE = /^:[\w-]+:$/;
5
+ /** Whether a line opens an org block (`#+begin_name`). */
6
+ export function isOrgBlockStart(line) {
7
+ return BLOCK_START_RE.test(line);
8
+ }
9
+ /** Whether a line opens an org drawer (`:NAME:`). */
10
+ export function isDrawerStart(line) {
11
+ return DRAWER_START_RE.test(line);
12
+ }
13
+ /**
14
+ * The line that ends the org block or drawer a line opens.
15
+ * @param lines The lines.
16
+ * @param start The index of the opening line.
17
+ * @returns The index of the `#+end_name` or `:END:` line, or -1.
18
+ */
19
+ export function orgElementEnd(lines, start) {
20
+ const end = endPattern(lines[start] ?? "");
21
+ return end ? lines.findIndex((other, i) => i > start && end.test(other)) : -1;
22
+ }
23
+ // the pattern of the line that ends the block or drawer `line` opens
24
+ function endPattern(line) {
25
+ const block = BLOCK_START_RE.exec(line)?.[1];
26
+ if (block) {
27
+ return new RegExp(`^#\\+end_${block.replace(/\W/g, "\\$&")}\\s*$`, "i");
28
+ }
29
+ return DRAWER_START_RE.test(line) ? /^:end:\s*$/i : null;
30
+ }
31
+ // the text from `offset` through the line that ends the block or
32
+ // drawer opening there, read line by line rather than splitting the
33
+ // rest of the document (a page may hold thousands of drawers)
34
+ function elementSource(markdown, offset) {
35
+ const lineEnd = (from) => {
36
+ const index = markdown.indexOf("\n", from);
37
+ return index === -1 ? markdown.length : index;
38
+ };
39
+ let end = lineEnd(offset);
40
+ const pattern = endPattern(markdown.slice(offset, end));
41
+ while (pattern && end < markdown.length) {
42
+ const next = lineEnd(end + 1);
43
+ if (pattern.test(markdown.slice(end + 1, next))) {
44
+ return markdown.slice(offset, next);
45
+ }
46
+ end = next;
47
+ }
48
+ return null;
49
+ }
50
+ // the source offset where the passthrough element starting at a node
51
+ // ends, or -1
52
+ function passthroughEnd(node, markdown) {
53
+ const start = node.position?.start;
54
+ if (start?.column !== 1 || start.offset === undefined) {
55
+ return -1;
56
+ }
57
+ const source = elementSource(markdown, start.offset);
58
+ return source !== null && readsAsPassthrough(source)
59
+ ? start.offset + source.length
60
+ : -1;
61
+ }
62
+ // the index of the last node ending at `end`, from `i` on, or -1
63
+ function lastNodeAt(nodes, i, end) {
64
+ let j = i;
65
+ while ((nodes[j]?.position?.end.offset ?? end) < end) {
66
+ j++;
67
+ }
68
+ return nodes[j]?.position?.end.offset === end ? j : -1;
69
+ }
70
+ /**
71
+ * md→org: a passthrough element's org text may hold lines Markdown reads
72
+ * as syntax of its own (a blank line and an indented one, a `#`, `-`
73
+ * or `>` line); the top-level nodes it parsed into become one paragraph
74
+ * of the source text again, which goes back to org as it is.
75
+ * @param mdast The parsed Markdown.
76
+ * @param markdown Its source.
77
+ */
78
+ export function keepPassthroughSource(mdast, markdown) {
79
+ const children = [];
80
+ const nodes = mdast.children;
81
+ for (let i = 0; i < nodes.length; i++) {
82
+ const node = nodes[i];
83
+ const end = passthroughEnd(node, markdown);
84
+ const last = end === -1 ? -1 : lastNodeAt(nodes, i, end);
85
+ if (last === -1) {
86
+ children.push(node);
87
+ continue;
88
+ }
89
+ const value = markdown.slice(node.position?.start.offset, end);
90
+ children.push({ type: "paragraph", children: [{ type: "text", value }] });
91
+ i = last;
92
+ }
93
+ mdast.children = children;
94
+ }
@@ -253,9 +253,14 @@ function transformQuoteBlock(ctx, node) {
253
253
  };
254
254
  }
255
255
  function transformSrcBlock(node) {
256
+ // switches and header arguments (`-n :results output`) are the
257
+ // fence's meta, after the language
258
+ const { switches, parameters } = node;
259
+ const meta = [switches, parameters].filter(Boolean).join(" ");
256
260
  return {
257
261
  type: "code",
258
262
  lang: node.language || null,
263
+ meta: node.language && meta ? meta : null,
259
264
  value: trimTrailingNewline(node.value)
260
265
  };
261
266
  }
@@ -24,7 +24,12 @@ export function keyValueParagraph(lines) {
24
24
  // trailing newline), for verbatim passthrough of org-only constructs
25
25
  export function orgNodeToText(node) {
26
26
  const orgText = unified()
27
- .use(uniorgStringify)
27
+ // a preset's verbatim-inline text is org text already
28
+ .use(uniorgStringify, {
29
+ handlers: {
30
+ "verbatim-inline": (inline) => inline.value
31
+ }
32
+ })
28
33
  .stringify({
29
34
  type: "org-data",
30
35
  children: [node],
package/dist/index.d.ts CHANGED
@@ -8,6 +8,5 @@ export { transformMdastToUniorgAst } from "./core/mdastToUniorg/index.js";
8
8
  export { transformUniorgAstToMdast } from "./core/uniorgToMdast/index.js";
9
9
  export { logseq } from "./presets/logseq.js";
10
10
  export { obsidian } from "./presets/obsidian.js";
11
- export type { Preset } from "./presets/types.js";
12
- export type { LogseqPresetOptions } from "./presets/logseq.js";
11
+ export type { FragmentConverter, Preset } from "./presets/types.js";
13
12
  export type { Toggle, MarkdownToOrgOptions, OrgToMarkdownOptions, MarkdownStyleOptions } from "./options.js";
@@ -14,6 +14,7 @@ import { escapeFootnoteReferences } from "./core/footnoteReferences.js";
14
14
  import { escapeBackslashCommands } from "./core/backslashCommands.js";
15
15
  import { escapeTablePipes } from "./core/tablePipes.js";
16
16
  import { requireBracedScripts } from "./core/bracedScripts.js";
17
+ import { keepPassthroughSource } from "./core/passthroughSource.js";
17
18
  import { keyValueEntries } from "./core/keyValueLines.js";
18
19
  /**
19
20
  * Converts a Markdown string to an Org-mode string.
@@ -22,6 +23,17 @@ import { keyValueEntries } from "./core/keyValueLines.js";
22
23
  * @returns The converted Org-mode string.
23
24
  */
24
25
  export function convertMarkdownToOrg(markdown, options = {}) {
26
+ const convertMarkdown = options.preset?.convertMarkdown;
27
+ return convertMarkdown
28
+ ? // a fragment's style is the preset's, not one to record
29
+ convertMarkdown(markdown, (fragment, preset) => convertMarkdownToOrg(fragment, {
30
+ ...options,
31
+ preset,
32
+ recordStyle: false
33
+ }))
34
+ : convertMarkdownDocument(markdown, options);
35
+ }
36
+ function convertMarkdownDocument(markdown, options) {
25
37
  // Phase 1: Parse Markdown to mdast
26
38
  const mdast = parseMarkdown(markdown, options.preset);
27
39
  // Phase 2: Generic mdast to uniorg-ast transformation
@@ -88,6 +100,7 @@ function parseMarkdown(markdown, preset) {
88
100
  .use(remarkFrontmatter)
89
101
  .use(remarkMath)
90
102
  .parse(markdown);
103
+ keepPassthroughSource(mdast, markdown);
91
104
  preset?.applyToMdast?.(mdast, markdown);
92
105
  return mdast;
93
106
  }
@@ -29,6 +29,12 @@ function separateTextAfterNestedList(left, right, parent) {
29
29
  * @returns The converted Markdown string.
30
30
  */
31
31
  export function convertOrgToMarkdown(org, options = {}) {
32
+ const convertOrg = options.preset?.convertOrg;
33
+ return convertOrg
34
+ ? convertOrg(org, (fragment, preset) => convertOrgToMarkdown(fragment, { ...options, preset }))
35
+ : convertOrgDocument(org, options);
36
+ }
37
+ function convertOrgDocument(org, options) {
32
38
  // Phase 1: Parse Org-mode to uniorg-ast
33
39
  // md text has no scripts, so ^:{} is implied there and consumed here
34
40
  // (see markdownToOrg); uniorg misreads `_.` lines (see underscoreBullets)
@@ -1,25 +1,6 @@
1
- import type { OrgData } from "uniorg";
2
1
  import type { Preset } from "./types.js";
3
- export interface LogseqPresetOptions {
4
- /**
5
- * Content following a heading becomes children of that heading's block
6
- * in Logseq's outline: paragraphs turn into child block headlines one
7
- * level deeper; other constructs stay in the preceding block's body.
8
- * Default: `true`.
9
- */
10
- nestUnderHeadings?: boolean;
11
- }
12
2
  /**
13
- * Logseq dialect preset: `heading::` properties and outline nesting.
3
+ * Logseq dialect preset: the outline of blocks, page properties, page
4
+ * and block references, highlights and hiccup.
14
5
  */
15
- export declare function logseq(options?: LogseqPresetOptions): Preset;
16
- /**
17
- * Applies Logseq-specific conventions to a uniorg AST: every heading gets
18
- * a `:heading:` property drawer, and with `nestUnderHeadings` following
19
- * paragraphs become child block headlines (in Logseq's org format every
20
- * outline block is a headline).
21
- * @param uniorgAst The uniorg AST to transform.
22
- * @param nestUnderHeadings Nest content as child blocks. Default: `true`.
23
- * @returns The Logseq-flavored uniorg AST.
24
- */
25
- export declare function applyLogseqSpecificsToUniorgAst(uniorgAst: OrgData, nestUnderHeadings?: boolean): OrgData;
6
+ export declare function logseq(): Preset;
@@ -3,43 +3,141 @@ import { toString } from "orgast-util-to-string";
3
3
  import { isScalar, isSeq } from "yaml";
4
4
  import { fitsKeywordLine, isFrontmatterNode, isModeLine, isModeLineComment, takeFrontmatterEntries } from "../core/frontmatterBlock.js";
5
5
  import { keyValueEntries } from "../core/keyValueLines.js";
6
+ import { tryParse } from "../core/render.js";
7
+ import { markdownOutlineToOrg, orgOutlineToMarkdown } from "./logseqOutline.js";
6
8
  /**
7
- * Logseq dialect preset: `heading::` properties and outline nesting.
9
+ * Logseq dialect preset: the outline of blocks, page properties, page
10
+ * and block references, highlights and hiccup.
8
11
  */
9
- export function logseq(options = {}) {
10
- const nestUnderHeadings = options.nestUnderHeadings ?? true;
11
- return {
12
+ export function logseq() {
13
+ const page = {
12
14
  name: "logseq",
13
15
  applyToMdast: keepPagePropertySource,
14
- applyToUniorg: uniorgAst => applyLogseqSpecificsToUniorgAst(uniorgAst, nestUnderHeadings),
15
- extractFromUniorg: extractLogseqSpecificsFromUniorgAst
16
+ applyToUniorg: uniorgAst => {
17
+ rewriteLabeledPageRefs(uniorgAst);
18
+ takePageProperties(uniorgAst);
19
+ return uniorgAst;
20
+ },
21
+ extractFromUniorg: uniorgAst => {
22
+ pageProperties(uniorgAst);
23
+ return extractInlineSpecifics(uniorgAst);
24
+ }
25
+ };
26
+ const bareUrls = new Map();
27
+ const block = {
28
+ name: "logseq",
29
+ applyToMdast: mdast => {
30
+ countBareUrls(mdast, bareUrls);
31
+ emailLinksToText(mdast);
32
+ },
33
+ applyToUniorg: uniorgAst => {
34
+ rewriteLabeledPageRefs(uniorgAst);
35
+ restoreBareUrls(uniorgAst, bareUrls);
36
+ return uniorgAst;
37
+ },
38
+ extractFromUniorg: uniorgAst => {
39
+ bareUrlsToText(uniorgAst);
40
+ return extractInlineSpecifics(uniorgAst);
41
+ }
16
42
  };
17
- }
18
- function headingProperty(level) {
19
- return { type: "node-property", key: "heading", value: String(level) };
20
- }
21
- function headingDrawer(level) {
22
43
  return {
23
- type: "property-drawer",
24
- children: [headingProperty(level)],
25
- contentsBegin: 0,
26
- contentsEnd: 0
44
+ ...page,
45
+ convertOrg: (org, convert) => orgOutlineToMarkdown(org, convert, { page, block }),
46
+ convertMarkdown: (markdown, convert) => markdownOutlineToOrg(markdown, convert, { page, block })
27
47
  };
28
48
  }
29
- // org fixes the order below a headline: the planning line first, then a
30
- // single property drawer. :heading: therefore has to let the planning
31
- // line pass and join an existing drawer rather than displace either.
32
- // Returns whether the property has been placed on `node`.
33
- function placeHeadingProperty(node, level) {
34
- if (node.type === "planning") {
35
- return false;
49
+ // Logseq md writes a url bare, as org writes a plain link; the core
50
+ // carries a md link as a [[url]] bracket link, so md→org counts the
51
+ // links that were bare in the source and turns as many back to plain.
52
+ // Where org's plain link ends short of md's autolink on trailing
53
+ // punctuation (a macro's `}}`), that tail splits off as text
54
+ function countBareUrls(mdast, counts) {
55
+ counts.clear();
56
+ visit(mdast, "link", (node, index, parent) => {
57
+ const url = bareUrl(node);
58
+ if (url === null) {
59
+ return undefined;
60
+ }
61
+ counts.set(url, (counts.get(url) ?? 0) + 1);
62
+ const tail = node.url.slice(url.length);
63
+ if (tail && parent && index !== undefined) {
64
+ node.url = url;
65
+ node.children = [{ type: "text", value: url }];
66
+ parent.children.splice(index + 1, 0, { type: "text", value: tail });
67
+ }
68
+ return undefined;
69
+ });
70
+ }
71
+ // the plain link org reads for a link written bare, if the rest is
72
+ // trailing punctuation
73
+ function bareUrl(node) {
74
+ const start = node.position?.start.offset;
75
+ const end = node.position?.end.offset;
76
+ if (start === undefined ||
77
+ end === undefined ||
78
+ end - start !== node.url.length) {
79
+ return null;
36
80
  }
37
- if (node.type !== "property-drawer") {
38
- return false;
81
+ const url = orgPlainLink(node.url);
82
+ return url !== null && /^[^\w(]*$/.test(node.url.slice(url.length))
83
+ ? url
84
+ : null;
85
+ }
86
+ // the url of the plain link org reads at the start of `url`, if any;
87
+ // it ends early at a `(` (`…/Bandwidth_(signal)`) or a macro's `}}`
88
+ function orgPlainLink(url) {
89
+ if (!/^https?:\/\//.test(url)) {
90
+ return null;
39
91
  }
40
- ;
41
- node.children.unshift(headingProperty(level));
42
- return true;
92
+ const [paragraph] = tryParse(`${url}\n`)?.children ?? [];
93
+ const [link] = paragraph?.children ?? [];
94
+ return link?.format === "plain"
95
+ ? link.rawLink
96
+ : null;
97
+ }
98
+ function restoreBareUrls(uniorgAst, counts) {
99
+ visit(uniorgAst, "link", (node) => {
100
+ const count = counts.get(node.rawLink) ?? 0;
101
+ if (node.format === "bracket" && !node.children.length && count) {
102
+ node.format = "plain";
103
+ counts.set(node.rawLink, count - 1);
104
+ }
105
+ });
106
+ }
107
+ // org→md: a plain http(s) link stays a bare url, unescaped
108
+ // an email address is text in Logseq org and written bare in Logseq md,
109
+ // where GFM links it: md→org takes such a link back to its text, and
110
+ // org→md keeps the address unescaped (keepVerbatimText)
111
+ const EMAIL_RE = /[\w.+-]+@[\w-]+(?:\.[\w-]+)+/;
112
+ function emailLinksToText(mdast) {
113
+ visit(mdast, "link", (node, index, parent) => {
114
+ const start = node.position?.start.offset;
115
+ const end = node.position?.end.offset;
116
+ const text = node.url.replace(/^mailto:/, "");
117
+ if (node.url.startsWith("mailto:") &&
118
+ start !== undefined &&
119
+ end !== undefined &&
120
+ end - start === text.length &&
121
+ parent &&
122
+ index !== undefined) {
123
+ parent.children[index] = { type: "text", value: text };
124
+ }
125
+ });
126
+ }
127
+ function bareUrlsToText(uniorgAst) {
128
+ visit(uniorgAst, "link", (node, index, parent) => {
129
+ if (node.format !== "plain" ||
130
+ !/^https?$/.test(node.linkType) ||
131
+ !parent ||
132
+ index === undefined) {
133
+ return undefined;
134
+ }
135
+ parent.children[index] = {
136
+ type: "verbatim-inline",
137
+ value: node.rawLink
138
+ };
139
+ return undefined;
140
+ });
43
141
  }
44
142
  // the page name of a [[page]] link destination, or null. The core
45
143
  // transform percent-encodes brackets in urls (an org link path cannot
@@ -66,80 +164,6 @@ function rewriteLabeledPageRefs(uniorgAst) {
66
164
  node.path = page;
67
165
  });
68
166
  }
69
- function toBlockHeadline(node, level) {
70
- const blockHeadline = {
71
- type: "headline",
72
- level,
73
- todoKeyword: null,
74
- priority: null,
75
- commented: false,
76
- rawValue: "",
77
- tags: [],
78
- children: node.children
79
- };
80
- takeTaskMarker(blockHeadline);
81
- return blockHeadline;
82
- }
83
- /**
84
- * Applies Logseq-specific conventions to a uniorg AST: every heading gets
85
- * a `:heading:` property drawer, and with `nestUnderHeadings` following
86
- * paragraphs become child block headlines (in Logseq's org format every
87
- * outline block is a headline).
88
- * @param uniorgAst The uniorg AST to transform.
89
- * @param nestUnderHeadings Nest content as child blocks. Default: `true`.
90
- * @returns The Logseq-flavored uniorg AST.
91
- */
92
- export function applyLogseqSpecificsToUniorgAst(uniorgAst, nestUnderHeadings = true) {
93
- rewriteLabeledPageRefs(uniorgAst);
94
- takePageProperties(uniorgAst);
95
- const children = uniorgAst.children;
96
- const result = [];
97
- let currentLevel = 0;
98
- // level of a headline whose :heading: property still needs a home
99
- let pending = null;
100
- const separate = () => {
101
- if (!nestUnderHeadings) {
102
- result.push({ type: "text", value: "\n" });
103
- }
104
- };
105
- // no drawer took the property: give it one of its own
106
- const settle = () => {
107
- if (pending === null) {
108
- return;
109
- }
110
- result.push(headingDrawer(pending));
111
- pending = null;
112
- separate();
113
- };
114
- for (const node of children) {
115
- if (node.type === "headline") {
116
- settle();
117
- const headline = node;
118
- currentLevel = headline.level;
119
- takeTaskMarker(headline);
120
- result.push(node);
121
- pending = headline.level;
122
- continue;
123
- }
124
- if (pending !== null) {
125
- if (placeHeadingProperty(node, pending)) {
126
- pending = null;
127
- result.push(node);
128
- separate();
129
- continue;
130
- }
131
- if (node.type !== "planning") {
132
- settle();
133
- }
134
- }
135
- result.push(nestUnderHeadings && node.type === "paragraph"
136
- ? toBlockHeadline(node, currentLevel + 1)
137
- : node);
138
- }
139
- settle();
140
- uniorgAst.children = result;
141
- return uniorgAst;
142
- }
143
167
  // Logseq reads a page's first block of `key:: value` lines (md) and its
144
168
  // leading `#+key: value` lines (org) as page properties (ADR 0005)
145
169
  function pagePropertyEntries(text) {
@@ -264,40 +288,35 @@ function pageProperties(uniorgAst) {
264
288
  contentsEnd: 0
265
289
  });
266
290
  }
267
- // Logseq md keeps TODO/DONE as leading text markers and priorities as
268
- // [#A] text; in org they are the headline's TODO keyword and priority
269
- // (other Logseq markers like DOING are not org keywords and simply
270
- // stay in the title text)
271
- function takeTaskMarker(headline) {
272
- const first = headline.children[0];
273
- if (first?.type !== "text") {
274
- return;
275
- }
276
- const marker = /^(TODO|DONE) (?:\[#([A-Z])\] )?/.exec(first.value);
277
- if (!marker) {
278
- return;
279
- }
280
- headline.todoKeyword = marker[1];
281
- if (marker[2]) {
282
- headline.priority = marker[2];
283
- }
284
- first.value = first.value.slice((marker[0] ?? "").length);
285
- }
286
- /**
287
- * Extracts Logseq-specific conventions from a uniorg AST: headlines with
288
- * a `:heading:` property become plain headings of that level, headlines
289
- * without one are outline blocks and become paragraphs.
290
- * @param uniorgAst The Logseq-flavored uniorg AST to transform.
291
- * @returns The generic uniorg AST.
292
- */
293
- function extractLogseqSpecificsFromUniorgAst(uniorgAst) {
294
- pageProperties(uniorgAst);
295
- extractInParent(uniorgAst);
291
+ function extractInlineSpecifics(uniorgAst) {
292
+ keepVerbatimText(uniorgAst);
296
293
  repairHighlights(uniorgAst);
297
294
  fuzzyLinksToPageRefs(uniorgAst);
298
295
  markHiccupParagraphs(uniorgAst);
299
296
  return uniorgAst;
300
297
  }
298
+ // text Logseq md writes as it is, which remark would escape: a task's
299
+ // priority ([#A]), an email address, and a tag's # (`#tag` starting a
300
+ // line, which mldoc reads as escaped plain text once written `\#`);
301
+ // verbatim-inline keeps it
302
+ const VERBATIM_TEXT_RE = new RegExp(`(\\[#[A-Z]\\]|${EMAIL_RE.source}|(?<!\\S)#(?=[^\\s#]|$))`);
303
+ function keepVerbatimText(uniorgAst) {
304
+ visit(uniorgAst, "text", (node, index, parent) => {
305
+ const parts = node.value.split(VERBATIM_TEXT_RE);
306
+ if (parts.length === 1 || !parent || index === undefined) {
307
+ return undefined;
308
+ }
309
+ // split's captures sit at the odd indices
310
+ const nodes = parts
311
+ .map((value, i) => ({
312
+ type: i % 2 ? "verbatim-inline" : "text",
313
+ value
314
+ }))
315
+ .filter(part => part.value);
316
+ parent.children.splice(index, 1, ...nodes);
317
+ return index + nodes.length;
318
+ });
319
+ }
301
320
  // Logseq highlight markup (^^words^^) re-parses as a caret plus a
302
321
  // superscript; merge the pieces back into literal text
303
322
  function repairHighlights(uniorgAst) {
@@ -360,69 +379,3 @@ function markHiccupParagraphs(uniorgAst) {
360
379
  }
361
380
  });
362
381
  }
363
- function extractInParent(parent) {
364
- const children = parent.children;
365
- for (let i = 0; i < children.length; i++) {
366
- const node = children[i];
367
- if (!node) {
368
- continue;
369
- }
370
- if (node.type === "section") {
371
- extractInParent(node);
372
- continue;
373
- }
374
- if (node.type !== "headline") {
375
- continue;
376
- }
377
- const headline = node;
378
- if (headline.todoKeyword) {
379
- // back to Logseq md's text conventions (TODO [#A] Ship it);
380
- // verbatim-inline keeps the [#A] brackets unescaped
381
- const priority = headline.priority ? `[#${headline.priority}] ` : "";
382
- headline.children.unshift({
383
- type: "verbatim-inline",
384
- value: `${headline.todoKeyword} ${priority}`
385
- });
386
- headline.todoKeyword = null;
387
- headline.priority = null;
388
- }
389
- // a planning line sits between the headline and its drawer
390
- const drawerIndex = children[i + 1]?.type === "planning" ? i + 2 : i + 1;
391
- const heading = takeHeadingProperty(children, drawerIndex);
392
- if (heading !== null) {
393
- headline.level = heading;
394
- }
395
- else {
396
- // a block headline (no :heading:) is outline structure only; its
397
- // title is the block's content
398
- children[i] = {
399
- type: "paragraph",
400
- children: headline.children,
401
- contentsBegin: 0,
402
- contentsEnd: 0
403
- };
404
- }
405
- }
406
- }
407
- // removes the heading property from a drawer at `index` (and the drawer
408
- // itself if that empties it); returns the heading level or null
409
- function takeHeadingProperty(children, index) {
410
- const drawer = children[index];
411
- if (drawer?.type !== "property-drawer") {
412
- return null;
413
- }
414
- const properties = drawer.children;
415
- const heading = properties.find(property => property.key === "heading");
416
- if (!heading) {
417
- return null;
418
- }
419
- const remaining = properties.filter(property => property !== heading);
420
- if (remaining.length) {
421
- ;
422
- drawer.children = remaining;
423
- }
424
- else {
425
- children.splice(index, 1);
426
- }
427
- return parseInt(heading.value, 10) || null;
428
- }
@@ -0,0 +1,22 @@
1
+ import type { FragmentConverter, Preset } from "./types.js";
2
+ interface Presets {
3
+ page: Preset;
4
+ block: Preset;
5
+ }
6
+ /**
7
+ * Converts a Logseq org page to Logseq Markdown, block by block.
8
+ * @param org The org page.
9
+ * @param convert The core's fragment converter.
10
+ * @param presets The presets for the page properties and for a block.
11
+ * @returns The Markdown page.
12
+ */
13
+ export declare function orgOutlineToMarkdown(org: string, convert: FragmentConverter, presets: Presets): string;
14
+ /**
15
+ * Converts a Logseq Markdown page to Logseq org, block by block.
16
+ * @param markdown The Markdown page.
17
+ * @param convert The core's fragment converter.
18
+ * @param presets The presets for the page properties and for a block.
19
+ * @returns The org page.
20
+ */
21
+ export declare function markdownOutlineToOrg(markdown: string, convert: FragmentConverter, presets: Presets): string;
22
+ export {};
@@ -0,0 +1,251 @@
1
+ import { consumesBracedScripts } from "../core/bracedScripts.js";
2
+ import { isDrawerStart, isOrgBlockStart, orgElementEnd } from "../core/passthroughSource.js";
3
+ // Logseq stores a page as an outline of blocks, each block a content
4
+ // string it parses on its own: org writes a block as its level's stars,
5
+ // a space and the content (an empty block as the bare stars), Markdown
6
+ // as a `- ` bullet indented one tab per level, continuation lines two
7
+ // spaces further in. Both are converted block by block, so a block's
8
+ // content (a code block, a table, lines that run on) is one fragment.
9
+ const ORG_BLOCK_RE = /^(\*+)(?: (.*))?$/;
10
+ const MD_BLOCK_RE = /^(\t*)-(?: (.*))?$/;
11
+ const PLANNING_RE = /^(?:SCHEDULED|DEADLINE): /;
12
+ const ORG_PROPERTY_RE = /^:([^\s:]+):(?: (.*))?$/;
13
+ const MD_PROPERTY_RE = /^([\w.-]+)::(?: (.*))?$/;
14
+ // a repeated task's log line, which Logseq bullets per format
15
+ const STATE_LINE_RE = /^[-*] (?=State ")/;
16
+ const MD_HEADING_RE = /^(#{1,6})(?: (.*))?$/;
17
+ const FENCE_RE = /^\s*(?:```|~~~)/;
18
+ // md→org adds it for the text's bare `_` and `^`, which a block's
19
+ // content holds as Logseq writes it
20
+ const BRACED_SCRIPTS_LINE = "#+OPTIONS: ^:{}";
21
+ function logbook(lines, bullet) {
22
+ return lines[0] === ":LOGBOOK:"
23
+ ? lines.map(line => line.replace(STATE_LINE_RE, `${bullet} `))
24
+ : lines;
25
+ }
26
+ function splitBlocks(lines, blockLevel) {
27
+ const page = [];
28
+ const blocks = [];
29
+ for (const line of lines) {
30
+ const start = blockLevel(line);
31
+ if (start) {
32
+ blocks.push({ level: start[0], lines: [start[1]] });
33
+ }
34
+ else if (blocks.length) {
35
+ blocks[blocks.length - 1]?.lines.push(line);
36
+ }
37
+ else {
38
+ page.push(line);
39
+ }
40
+ }
41
+ return { page, blocks };
42
+ }
43
+ // a block's leading planning lines and drawers, which Logseq writes
44
+ // below the content's first line in either format
45
+ function takeMeta(lines, isMeta) {
46
+ const meta = [];
47
+ let i = 0;
48
+ while (i < lines.length) {
49
+ const line = lines[i] ?? "";
50
+ if (isDrawerStart(line)) {
51
+ const end = orgElementEnd(lines, i);
52
+ if (end === -1) {
53
+ break;
54
+ }
55
+ meta.push(lines.slice(i, end + 1));
56
+ i = end + 1;
57
+ }
58
+ else if (PLANNING_RE.test(line) || isMeta(line)) {
59
+ meta.push([line]);
60
+ i++;
61
+ }
62
+ else {
63
+ break;
64
+ }
65
+ }
66
+ return { meta, body: lines.slice(i) };
67
+ }
68
+ // the one org block whose content Logseq md writes as Markdown markup
69
+ // (its <quote command), and org→md writes as a md quote
70
+ const QUOTE_BLOCK_RE = /^#\+begin_quote\b/i;
71
+ function convertOrgBlock(block, convert, preset) {
72
+ if (!QUOTE_BLOCK_RE.test(block[0] ?? "")) {
73
+ return block;
74
+ }
75
+ const content = convertContent(block.slice(1, -1), convert, preset);
76
+ return [block[0] ?? "", ...content, block.at(-1) ?? ""];
77
+ }
78
+ // md→org: a block's org blocks stay as written, as org→md writes them,
79
+ // but for a quote's content; the text around them converts
80
+ function convertMarkdownContent(lines, convert, preset) {
81
+ const result = [];
82
+ let text = [];
83
+ let fenced = false;
84
+ for (let i = 0; i < lines.length; i++) {
85
+ fenced = FENCE_RE.test(lines[i] ?? "") ? !fenced : fenced;
86
+ const end = fenced || !isOrgBlockStart(lines[i] ?? "") ? -1 : orgElementEnd(lines, i);
87
+ if (end === -1) {
88
+ text.push(lines[i] ?? "");
89
+ continue;
90
+ }
91
+ result.push(...convertContent(text, convert, preset), ...convertOrgBlock(lines.slice(i, end + 1), convert, preset));
92
+ text = [];
93
+ i = end;
94
+ }
95
+ return [...result, ...convertContent(text, convert, preset)];
96
+ }
97
+ function convertContent(lines, convert, preset) {
98
+ const content = lines.join("\n");
99
+ return content.trim()
100
+ ? convert(content, preset).replace(/\n+$/, "").split("\n")
101
+ : [];
102
+ }
103
+ function convertPage(lines, convert, preset) {
104
+ // Logseq ends the page properties with a blank line
105
+ return lines.some(line => line.trim())
106
+ ? [`${convert(lines.join("\n"), preset).replace(/\n+$/, "")}\n`]
107
+ : [];
108
+ }
109
+ /**
110
+ * Converts a Logseq org page to Logseq Markdown, block by block.
111
+ * @param org The org page.
112
+ * @param convert The core's fragment converter.
113
+ * @param presets The presets for the page properties and for a block.
114
+ * @returns The Markdown page.
115
+ */
116
+ export function orgOutlineToMarkdown(org, convert, presets) {
117
+ const { page, blocks } = splitBlocks(org.replace(/\r?\n$/, "").split(/\r?\n/), line => {
118
+ const match = ORG_BLOCK_RE.exec(line);
119
+ return match ? [match[1]?.length ?? 0, match[2] ?? ""] : null;
120
+ });
121
+ return [
122
+ ...convertPage(page, convert, presets.page),
123
+ ...blocks.map(block => orgBlockToMarkdown(block, convert, presets.block))
124
+ ]
125
+ .join("\n")
126
+ .concat("\n");
127
+ }
128
+ // org→md reads a block's bare `_` and `^` as text, as md→org writes
129
+ // them without the setting that would say so
130
+ function withBracedScripts(lines) {
131
+ const braced = [BRACED_SCRIPTS_LINE, ...lines];
132
+ return consumesBracedScripts(braced.join("\n")) ? braced : lines;
133
+ }
134
+ function orgBlockToMarkdown(block, convert, preset) {
135
+ const [first = "", ...rest] = block.lines;
136
+ // content that starts with the block's properties puts their drawer
137
+ // on the headline line; its heading level is then a property too
138
+ const metaFirst = isDrawerStart(first);
139
+ const { meta, body } = takeMeta(metaFirst ? block.lines : rest, () => false);
140
+ let heading = "";
141
+ const metaLines = meta.flatMap(lines => {
142
+ if (lines[0] !== ":PROPERTIES:") {
143
+ return logbook(lines, "*");
144
+ }
145
+ return lines.slice(1, -1).flatMap(line => {
146
+ const [, key = "", value = ""] = ORG_PROPERTY_RE.exec(line) ?? [];
147
+ if (!metaFirst && key === "heading" && /^[1-6]$/.test(value)) {
148
+ heading = "#".repeat(Number(value));
149
+ return [];
150
+ }
151
+ return [`${key}::${value ? ` ${value}` : ""}`];
152
+ });
153
+ });
154
+ const content = convertContent(withBracedScripts(metaFirst ? body : [first, ...body]), convert, preset);
155
+ const [title = "", ...more] = arrange(metaFirst, metaLines, content);
156
+ const head = [heading, title].filter(Boolean).join(" ");
157
+ const indent = "\t".repeat(block.level - 1);
158
+ return [
159
+ `${indent}-${head ? ` ${head}` : ""}`,
160
+ ...more.map(line => `${indent} ${line}`)
161
+ ].join("\n");
162
+ }
163
+ /**
164
+ * Converts a Logseq Markdown page to Logseq org, block by block.
165
+ * @param markdown The Markdown page.
166
+ * @param convert The core's fragment converter.
167
+ * @param presets The presets for the page properties and for a block.
168
+ * @returns The org page.
169
+ */
170
+ export function markdownOutlineToOrg(markdown, convert, presets) {
171
+ const lines = markdown.replace(/\r?\n$/, "").split(/\r?\n/);
172
+ // a leading frontmatter is page content, its `- ` lines yaml items
173
+ const frontmatter = lines.slice(0, lines[0] === "---" ? lines.indexOf("---", 1) + 1 : 0);
174
+ let fenced = false;
175
+ const { page, blocks } = splitBlocks(lines.slice(frontmatter.length), line => {
176
+ const match = MD_BLOCK_RE.exec(line);
177
+ // a bullet starts a block, which may open a fence of its own
178
+ fenced = (match ? false : fenced) !== FENCE_RE.test(match?.[2] ?? line);
179
+ if (match) {
180
+ return [(match[1]?.length ?? 0) + 1, match[2] ?? ""];
181
+ }
182
+ // a heading outside the bullets is a top-level block, as Logseq
183
+ // writes a page's first block if it is a heading
184
+ return !fenced && MD_HEADING_RE.test(line) ? [1, line] : null;
185
+ });
186
+ page.unshift(...frontmatter);
187
+ return [
188
+ ...convertPage(page, convert, presets.page),
189
+ ...blocks.map(block => mdBlockToOrg(block, convert, presets.block))
190
+ ]
191
+ .join("\n")
192
+ .concat("\n");
193
+ }
194
+ // a continuation line sits two spaces inside its bullet
195
+ function dedent(line, level) {
196
+ const indent = `${"\t".repeat(level - 1)} `;
197
+ return line.startsWith(indent)
198
+ ? line.slice(indent.length)
199
+ : line.trim()
200
+ ? line
201
+ : "";
202
+ }
203
+ // the md meta lines in org: `key::` lines become the property drawer,
204
+ // where the first of them was, or below the planning lines
205
+ function orgMetaLines(meta, heading) {
206
+ const properties = heading ? [`:heading: ${heading}`] : [];
207
+ const lines = [];
208
+ for (const group of meta) {
209
+ const property = MD_PROPERTY_RE.exec(group[0] ?? "");
210
+ if (!property) {
211
+ lines.push(...logbook(group, "-"));
212
+ continue;
213
+ }
214
+ if (!lines.includes(null)) {
215
+ lines.push(null);
216
+ }
217
+ properties.push(`:${property[1]}:${property[2] ? ` ${property[2]}` : ""}`);
218
+ }
219
+ if (properties.length && !lines.includes(null)) {
220
+ lines.splice(lines.filter(line => PLANNING_RE.test(line ?? "")).length, 0, null);
221
+ }
222
+ const drawer = [":PROPERTIES:", ...properties, ":END:"];
223
+ return lines.flatMap(line => (line === null ? drawer : [line]));
224
+ }
225
+ // a block's content's first line, its meta lines, then the rest of the
226
+ // content; content that starts with the meta lines keeps them first
227
+ function arrange(metaFirst, meta, content) {
228
+ return metaFirst
229
+ ? [...meta, ...content]
230
+ : [content[0] ?? "", ...meta, ...content.slice(1)];
231
+ }
232
+ // `## title`: the heading level and the title
233
+ function mdHeading(line) {
234
+ const match = MD_HEADING_RE.exec(line);
235
+ return match ? [match[1]?.length ?? 0, match[2] ?? ""] : [0, line];
236
+ }
237
+ function mdBlockToOrg(block, convert, preset) {
238
+ const [first = "", ...rest] = block.lines;
239
+ const lines = [first, ...rest.map(line => dedent(line, block.level))];
240
+ // content that starts with properties: their drawer opens on the
241
+ // headline line
242
+ const metaFirst = MD_PROPERTY_RE.test(first);
243
+ const [heading, titleLine] = metaFirst ? [0, ""] : mdHeading(first);
244
+ const { meta, body } = takeMeta(metaFirst ? lines : lines.slice(1), line => MD_PROPERTY_RE.test(line));
245
+ const content = convertMarkdownContent(metaFirst ? body : [titleLine, ...body], convert, preset).filter(line => line !== BRACED_SCRIPTS_LINE);
246
+ const [title = "", ...more] = arrange(metaFirst, orgMetaLines(meta, heading), content);
247
+ return [
248
+ `${"*".repeat(block.level)}${title ? ` ${title}` : ""}`,
249
+ ...more
250
+ ].join("\n");
251
+ }
@@ -6,10 +6,18 @@ import type { OrgData } from "uniorg";
6
6
  * transform, with the Markdown source at hand; `applyToUniorg` runs in the md→org pipeline after the generic transform;
7
7
  * `extractFromUniorg` runs in the org→md pipeline before the generic
8
8
  * transform, normalizing dialect conventions back to generic uniorg.
9
+ * `convertOrg` and `convertMarkdown` take a whole conversion over, for a
10
+ * dialect whose documents are no single org or Markdown document (an
11
+ * outline of blocks, each its own fragment); `convert` runs the core on
12
+ * a fragment, with the given preset's hooks.
9
13
  */
10
14
  export interface Preset {
11
15
  name: string;
12
16
  applyToMdast?: (mdast: MdastRoot, markdown: string) => void;
13
17
  applyToUniorg?: (uniorgAst: OrgData) => OrgData;
14
18
  extractFromUniorg?: (uniorgAst: OrgData) => OrgData;
19
+ convertOrg?: (org: string, convert: FragmentConverter) => string;
20
+ convertMarkdown?: (markdown: string, convert: FragmentConverter) => string;
15
21
  }
22
+ /** Converts a fragment with the core and the given preset's AST hooks. */
23
+ export type FragmentConverter = (fragment: string, preset?: Preset) => string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@remigius42/morg",
3
- "version": "0.7.0",
3
+ "version": "0.8.0",
4
4
  "description": "Bidirectional Markdown ↔ Org-mode converter with round-trip convergence",
5
5
  "keywords": [
6
6
  "org-mode",