@remigius42/morg 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,13 +5,16 @@ import remarkFrontmatter from "remark-frontmatter";
5
5
  import remarkMath from "remark-math";
6
6
  import { uniorgStringify } from "uniorg-stringify";
7
7
  import { visit } from "unist-util-visit";
8
- import { transformMdastToUniorgAst } from "./core/mdastToUniorg/index.js";
8
+ import { transformMdastToUniorgDraft } from "./core/mdastToUniorg/index.js";
9
9
  import { detectMarkdownStyle, STYLE_KEYWORD } from "./core/markdownStyle.js";
10
10
  import { escapeOrgMarkup } from "./core/markupBoundary.js";
11
+ import { renderFileHeader } from "./core/frontmatterBlock.js";
11
12
  import { escapeLineSyntax } from "./core/lineSyntax.js";
12
13
  import { escapeFootnoteReferences } from "./core/footnoteReferences.js";
14
+ import { escapeBackslashCommands } from "./core/backslashCommands.js";
13
15
  import { escapeTablePipes } from "./core/tablePipes.js";
14
16
  import { requireBracedScripts } from "./core/bracedScripts.js";
17
+ import { keyValueEntries } from "./core/keyValueLines.js";
15
18
  /**
16
19
  * Converts a Markdown string to an Org-mode string.
17
20
  * @param markdown The Markdown string to convert.
@@ -20,14 +23,9 @@ import { requireBracedScripts } from "./core/bracedScripts.js";
20
23
  */
21
24
  export function convertMarkdownToOrg(markdown, options = {}) {
22
25
  // Phase 1: Parse Markdown to mdast
23
- const mdast = unified()
24
- .use(remarkParse)
25
- .use(remarkGfm)
26
- .use(remarkFrontmatter)
27
- .use(remarkMath)
28
- .parse(markdown);
26
+ const mdast = parseMarkdown(markdown, options.preset);
29
27
  // Phase 2: Generic mdast to uniorg-ast transformation
30
- let uniorgAst = transformMdastToUniorgAst(mdast, {
28
+ let uniorgAst = transformMdastToUniorgDraft(mdast, {
31
29
  ...(options.preserveMdisms !== undefined && {
32
30
  preserveMdisms: options.preserveMdisms
33
31
  }),
@@ -60,8 +58,10 @@ export function convertMarkdownToOrg(markdown, options = {}) {
60
58
  // Phase 3b: literal footnote references, paragraph lines org would
61
59
  // read as line syntax, literal markers org would read as markup, and
62
60
  // markup touching a word character, need a zero-width space escape;
63
- // a pipe in a table cell an entity
64
- escapeTablePipes(uniorgAst);
61
+ // a pipe in a table cell an entity; a literal backslash before a
62
+ // letter first, the entity being none
63
+ escapeBackslashCommands(uniorgAst);
64
+ escapeTablePipes(uniorgAst, options.onWarning);
65
65
  escapeFootnoteReferences(uniorgAst);
66
66
  escapeLineSyntax(uniorgAst);
67
67
  escapeOrgMarkup(uniorgAst);
@@ -70,12 +70,28 @@ export function convertMarkdownToOrg(markdown, options = {}) {
70
70
  // rendered: after the preset, whose text rewrites (wikilink aliases)
71
71
  // org parses too, and after the escapes, next to which org reads them
72
72
  requireBracedScripts(uniorgAst);
73
+ // Phase 3d: the file's header (ADR 0005): mode line and file-level
74
+ // drawer first, keywords apart from what follows, the frontmatter as
75
+ // a marked comment block; raw text, past every pass that rewrites text
76
+ renderFileHeader(uniorgAst);
73
77
  // Phase 4: Render uniorg-ast to Org-mode string
74
78
  const processor = unified().use(uniorgStringify);
75
79
  const orgContent = processor.stringify(uniorgAst);
76
80
  return orgContent;
77
81
  }
78
- // leads the document so it survives the frontmatter keywords following it
82
+ // a preset's dialect conventions that need the Markdown source apply
83
+ // to the parse, before the generic transform
84
+ function parseMarkdown(markdown, preset) {
85
+ const mdast = unified()
86
+ .use(remarkParse)
87
+ .use(remarkGfm)
88
+ .use(remarkFrontmatter)
89
+ .use(remarkMath)
90
+ .parse(markdown);
91
+ preset?.applyToMdast?.(mdast, markdown);
92
+ return mdast;
93
+ }
94
+ // leads the document, ahead of restored keywords and the frontmatter block
79
95
  function recordStyleKeyword(uniorgAst, style) {
80
96
  if (!Object.keys(style).length) {
81
97
  return;
@@ -86,7 +102,6 @@ function recordStyleKeyword(uniorgAst, style) {
86
102
  value: JSON.stringify(style)
87
103
  });
88
104
  }
89
- const KEY_VALUE_LINE_RE = /^([\w-]+):: (.*)$/;
90
105
  function parseKeyValueParagraph(node) {
91
106
  if (node?.type !== "paragraph" || !node.children?.length) {
92
107
  return null;
@@ -94,19 +109,7 @@ function parseKeyValueParagraph(node) {
94
109
  if (!node.children.every(child => child.type === "text")) {
95
110
  return null;
96
111
  }
97
- const lines = node.children
98
- .map(child => child.value ?? "")
99
- .join("")
100
- .split("\n");
101
- const entries = [];
102
- for (const line of lines) {
103
- const match = KEY_VALUE_LINE_RE.exec(line);
104
- if (!match) {
105
- return null;
106
- }
107
- entries.push([match[1], match[2]]);
108
- }
109
- return entries;
112
+ return keyValueEntries(node.children.map(child => child.value ?? "").join(""));
110
113
  }
111
114
  function makeTimestamp(rawValue) {
112
115
  return { type: "timestamp", rawValue };
@@ -5,9 +5,12 @@ import remarkFrontmatter from "remark-frontmatter";
5
5
  import remarkMath from "remark-math";
6
6
  import { transformUniorgAstToMdast } from "./core/uniorgToMdast/index.js";
7
7
  import { takeRecordedStyle } from "./core/markdownStyle.js";
8
+ import { takeFileHeader } from "./core/frontmatterBlock.js";
8
9
  import { unescapeOrgMarkup } from "./core/markupBoundary.js";
9
10
  import { unescapeLineSyntax } from "./core/lineSyntax.js";
10
11
  import { unescapeFootnoteReferences } from "./core/footnoteReferences.js";
12
+ import { unescapeBackslashCommands } from "./core/backslashCommands.js";
13
+ import { unescapeTablePipes } from "./core/tablePipes.js";
11
14
  import { parseOrg } from "./core/bracedScripts.js";
12
15
  import { dropUnderscoreBulletGuards, guardUnderscoreBullets } from "./core/underscoreBullets.js";
13
16
  // a list item's paragraph after its nested list needs a blank line, or
@@ -29,15 +32,21 @@ export function convertOrgToMarkdown(org, options = {}) {
29
32
  // Phase 1: Parse Org-mode to uniorg-ast
30
33
  // md text has no scripts, so ^:{} is implied there and consumed here
31
34
  // (see markdownToOrg); uniorg misreads `_.` lines (see underscoreBullets)
32
- let uniorgAst = parseOrg(guardUnderscoreBullets(org));
35
+ const guarded = guardUnderscoreBullets(org);
36
+ let uniorgAst = parseOrg(guarded);
33
37
  dropUnderscoreBulletGuards(uniorgAst);
34
38
  // Phase 1b: a recorded style is morg's own (ADR 0004), so consume it so
35
39
  // it does not travel on as frontmatter; explicit options still win
36
40
  const recordedStyle = takeRecordedStyle(uniorgAst);
41
+ // frontmatter travels in a block marked as morg's (ADR 0005)
42
+ // and so does a file-level drawer (org-roam's :ID:)
43
+ const fileHeader = takeFileHeader(uniorgAst, guarded, options.onWarning);
37
44
  // Phase 1c: markdown needs no zero-width space escapes (inverse of md→org)
38
45
  unescapeOrgMarkup(uniorgAst);
39
46
  unescapeLineSyntax(uniorgAst);
40
47
  unescapeFootnoteReferences(uniorgAst);
48
+ unescapeBackslashCommands(uniorgAst);
49
+ unescapeTablePipes(uniorgAst);
41
50
  // Phase 2: Extract dialect preset conventions, if any
42
51
  if (options.preset?.extractFromUniorg) {
43
52
  uniorgAst = options.preset.extractFromUniorg(uniorgAst);
@@ -54,7 +63,8 @@ export function convertOrgToMarkdown(org, options = {}) {
54
63
  ...(options.orgismKeys !== undefined && {
55
64
  orgismKeys: options.orgismKeys
56
65
  }),
57
- ...(options.onWarning !== undefined && { onWarning: options.onWarning })
66
+ ...(options.onWarning !== undefined && { onWarning: options.onWarning }),
67
+ ...fileHeader
58
68
  });
59
69
  // Phase 4: Render mdast to Markdown string
60
70
  // bullet and rule "-" (not remark's default "*") are morg's canonical
@@ -1,5 +1,8 @@
1
1
  import { visit } from "unist-util-visit";
2
2
  import { toString } from "orgast-util-to-string";
3
+ import { isScalar, isSeq } from "yaml";
4
+ import { fitsKeywordLine, isFrontmatterNode, isModeLine, isModeLineComment, takeFrontmatterEntries } from "../core/frontmatterBlock.js";
5
+ import { keyValueEntries } from "../core/keyValueLines.js";
3
6
  /**
4
7
  * Logseq dialect preset: `heading::` properties and outline nesting.
5
8
  */
@@ -7,6 +10,7 @@ export function logseq(options = {}) {
7
10
  const nestUnderHeadings = options.nestUnderHeadings ?? true;
8
11
  return {
9
12
  name: "logseq",
13
+ applyToMdast: keepPagePropertySource,
10
14
  applyToUniorg: uniorgAst => applyLogseqSpecificsToUniorgAst(uniorgAst, nestUnderHeadings),
11
15
  extractFromUniorg: extractLogseqSpecificsFromUniorgAst
12
16
  };
@@ -87,6 +91,7 @@ function toBlockHeadline(node, level) {
87
91
  */
88
92
  export function applyLogseqSpecificsToUniorgAst(uniorgAst, nestUnderHeadings = true) {
89
93
  rewriteLabeledPageRefs(uniorgAst);
94
+ takePageProperties(uniorgAst);
90
95
  const children = uniorgAst.children;
91
96
  const result = [];
92
97
  let currentLevel = 0;
@@ -135,6 +140,130 @@ export function applyLogseqSpecificsToUniorgAst(uniorgAst, nestUnderHeadings = t
135
140
  uniorgAst.children = result;
136
141
  return uniorgAst;
137
142
  }
143
+ // Logseq reads a page's first block of `key:: value` lines (md) and its
144
+ // leading `#+key: value` lines (org) as page properties (ADR 0005)
145
+ function pagePropertyEntries(text) {
146
+ const entries = keyValueEntries(text);
147
+ // a key org reads as no keyword name (begin_src) would not come back
148
+ return entries?.every(entry => fitsKeywordLine(...entry)) ? entries : null;
149
+ }
150
+ // md→org: the page property block's values are Logseq's, not Markdown;
151
+ // its source text replaces the parse, so urls and markup stay verbatim
152
+ function keepPagePropertySource(mdast, markdown) {
153
+ // below the frontmatter and an Emacs mode line
154
+ const node = mdast.children.find(child => child.type !== "yaml" &&
155
+ !(child.type === "html" &&
156
+ child.value.startsWith("<!--") &&
157
+ isModeLine(child.value)));
158
+ const start = node?.position?.start.offset;
159
+ const end = node?.position?.end.offset;
160
+ if (node?.type !== "paragraph" || start === undefined || end === undefined) {
161
+ return;
162
+ }
163
+ const source = markdown.slice(start, end);
164
+ if (pagePropertyEntries(source)) {
165
+ node.children = [{ type: "text", value: source }];
166
+ }
167
+ }
168
+ function takePageProperties(uniorgAst) {
169
+ const children = uniorgAst.children;
170
+ const keywords = takeFrontmatterKeywords(children);
171
+ // below the keywords, the file-level drawer and a mode line
172
+ const index = children.findIndex(node => !["keyword", "property-drawer"].includes(node.type) &&
173
+ !isModeLineComment(node) &&
174
+ !isFrontmatterNode(node));
175
+ const lines = pagePropertyLines(children[index]);
176
+ if (lines) {
177
+ children.splice(index, 1);
178
+ keywords.push(...lines);
179
+ }
180
+ // ahead of the block, so all of them lead the page
181
+ const at = children.findIndex(node => node.type !== "keyword");
182
+ children.splice(at === -1 ? children.length : at, 0, ...keywords.map(([key, value]) => ({ type: "keyword", key, value })));
183
+ }
184
+ function takeFrontmatterKeywords(children) {
185
+ const frontmatter = children.find(isFrontmatterNode);
186
+ if (!frontmatter) {
187
+ return [];
188
+ }
189
+ const keywords = takeFlatEntries(frontmatter);
190
+ // an emptied block goes
191
+ if (!frontmatter.yaml && keywords.length) {
192
+ children.splice(children.indexOf(frontmatter), 1);
193
+ }
194
+ return keywords;
195
+ }
196
+ function pagePropertyLines(node) {
197
+ // plain text only: markup in a key means the source was no key:: line
198
+ const children = node?.children;
199
+ if (node?.type !== "paragraph" ||
200
+ !children?.every(child => child.type === "text")) {
201
+ return null;
202
+ }
203
+ return pagePropertyEntries(toString(node));
204
+ }
205
+ // keywords that act in Emacs or in export (TODO states, startup and
206
+ // export options, file inclusion, babel calls, a dynamic block's begin
207
+ // and end lines, citations, a table of
208
+ // contents, index entries, raw export lines and exporter options, of
209
+ // org's own exporters, org-info.js and KOMA letter class files among
210
+ // them, and of ox-hugo and org-re-reveal): as frontmatter they were
211
+ // passive data, so they stay in the block. A denylist: other
212
+ // third-party exporters' keys pass. Export metadata (`title`, `author`)
213
+ // passes too, and `tags`, Logseq's page tags property
214
+ const ACTING_KEYWORD_RE = /^(?:(?:SEQ_|TYP_)?TODO|STARTUP|OPTIONS|INCLUDE|SETUPFILE|BIND|MACRO|CALL|PROPERTY|LINK|CONSTANTS|PRIORITIES|BIBLIOGRAPHY|CITE_EXPORT|PRINT_BIBLIOGRAPHY|ARCHIVE|CATEGORY|COLUMNS|FILETAGS|EXCLUDE_TAGS|SELECT_TAGS|BEGIN|END|TOC|(?:C|F|K|P|T|V)?INDEX|INFOJS_OPT|LCO|EXPORT_\w+|(?:HTML|LATEX|BEAMER|ODT|TEXINFO|MAN|ASCII|MD|MARKDOWN|ICALENDAR|HUGO|REVEAL)(?:_\w+)?)$/i;
215
+ // flat frontmatter entries are page properties too, a sequence written
216
+ // as Logseq writes it (`a, b`); what a keyword line cannot hold stays
217
+ function takeFlatEntries(frontmatter) {
218
+ const { keywords, yaml } = takeFrontmatterEntries(frontmatter.yaml, (key, value, text) => {
219
+ const flat = flatValue(value, text);
220
+ // a key that comes back as a key:: line as written: uniorg
221
+ // upper-cases it, Logseq reads it lower-cased
222
+ return flat !== null &&
223
+ /^[a-z0-9_-]+$/.test(text(key)) &&
224
+ fitsKeywordLine(text(key), flat) &&
225
+ !ACTING_KEYWORD_RE.test(text(key))
226
+ ? [[text(key), flat]]
227
+ : null;
228
+ });
229
+ frontmatter.yaml = yaml;
230
+ return keywords;
231
+ }
232
+ function flatValue(value, text) {
233
+ const single = (item) => isScalar(item) && item.value !== null ? text(item) : null;
234
+ if (!isSeq(value)) {
235
+ return single(value);
236
+ }
237
+ const items = value.items.map(single);
238
+ return items.length &&
239
+ items.every(item => item !== null && !item.includes(","))
240
+ ? items.join(", ")
241
+ : null;
242
+ }
243
+ // the reverse: leading keywords become the first block, verbatim, keys
244
+ // lower-cased (uniorg upper-cases them, Logseq reads only lower case)
245
+ function pageProperties(uniorgAst) {
246
+ const children = uniorgAst.children;
247
+ // below a mode line, as on the way in
248
+ const first = children.findIndex(node => !isModeLineComment(node));
249
+ let count = 0;
250
+ while (first !== -1 && children[first + count]?.type === "keyword") {
251
+ count++;
252
+ }
253
+ const leading = children.slice(first, first + count);
254
+ // a key no `key::` line holds (`CAPTION[short]`) stays a keyword
255
+ const properties = leading.filter(keyword => /^[\w-]+$/.test(keyword.key));
256
+ if (!properties.length) {
257
+ return;
258
+ }
259
+ const lines = properties.map(keyword => `${keyword.key.toLowerCase()}::${keyword.value ? ` ${keyword.value}` : ""}`);
260
+ children.splice(first, count, ...leading.filter(keyword => !properties.includes(keyword)), {
261
+ type: "paragraph",
262
+ children: [{ type: "verbatim-inline", value: lines.join("\n") }],
263
+ contentsBegin: 0,
264
+ contentsEnd: 0
265
+ });
266
+ }
138
267
  // Logseq md keeps TODO/DONE as leading text markers and priorities as
139
268
  // [#A] text; in org they are the headline's TODO keyword and priority
140
269
  // (other Logseq markers like DOING are not org keywords and simply
@@ -162,6 +291,7 @@ function takeTaskMarker(headline) {
162
291
  * @returns The generic uniorg AST.
163
292
  */
164
293
  function extractLogseqSpecificsFromUniorgAst(uniorgAst) {
294
+ pageProperties(uniorgAst);
165
295
  extractInParent(uniorgAst);
166
296
  repairHighlights(uniorgAst);
167
297
  fuzzyLinksToPageRefs(uniorgAst);
@@ -1,12 +1,15 @@
1
+ import type { Root as MdastRoot } from "mdast";
1
2
  import type { OrgData } from "uniorg";
2
3
  /**
3
4
  * A dialect preset: transforms applied on top of the dialect-agnostic core.
4
- * `applyToUniorg` runs in the md→org pipeline after the generic transform;
5
+ * `applyToMdast` runs in the md→org pipeline before the generic
6
+ * transform, with the Markdown source at hand; `applyToUniorg` runs in the md→org pipeline after the generic transform;
5
7
  * `extractFromUniorg` runs in the org→md pipeline before the generic
6
8
  * transform, normalizing dialect conventions back to generic uniorg.
7
9
  */
8
10
  export interface Preset {
9
11
  name: string;
12
+ applyToMdast?: (mdast: MdastRoot, markdown: string) => void;
10
13
  applyToUniorg?: (uniorgAst: OrgData) => OrgData;
11
14
  extractFromUniorg?: (uniorgAst: OrgData) => OrgData;
12
15
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@remigius42/morg",
3
- "version": "0.6.0",
3
+ "version": "0.7.0",
4
4
  "description": "Bidirectional Markdown ↔ Org-mode converter with round-trip convergence",
5
5
  "keywords": [
6
6
  "org-mode",