@remigius42/morg 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +36 -30
  2. package/dist/cli/args.d.ts +2 -0
  3. package/dist/cli/args.js +22 -60
  4. package/dist/cli/error.js +1 -1
  5. package/dist/cli/flags.d.ts +27 -0
  6. package/dist/cli/flags.js +127 -0
  7. package/dist/cli/formats.js +18 -4
  8. package/dist/cli/help.d.ts +1 -0
  9. package/dist/cli/help.js +28 -0
  10. package/dist/cli.js +24 -1
  11. package/dist/conversionOptions.d.ts +2 -2
  12. package/dist/conversionOptions.js +2 -2
  13. package/dist/core/bracedScripts.d.ts +12 -0
  14. package/dist/core/bracedScripts.js +159 -0
  15. package/dist/core/footnoteReferences.d.ts +10 -0
  16. package/dist/core/footnoteReferences.js +24 -0
  17. package/dist/core/lineSyntax.d.ts +12 -0
  18. package/dist/core/lineSyntax.js +206 -0
  19. package/dist/core/markupBoundary.d.ts +12 -0
  20. package/dist/core/markupBoundary.js +195 -0
  21. package/dist/core/mdastToUniorg/blocks.js +1 -1
  22. package/dist/core/mdastToUniorg/index.js +10 -2
  23. package/dist/core/mdastToUniorg/lists.d.ts +1 -1
  24. package/dist/core/mdastToUniorg/lists.js +21 -2
  25. package/dist/core/mdastToUniorg/phrasing.js +145 -12
  26. package/dist/core/orgPath.d.ts +9 -0
  27. package/dist/core/orgPath.js +24 -0
  28. package/dist/core/render.d.ts +33 -0
  29. package/dist/core/render.js +101 -0
  30. package/dist/core/tablePipes.d.ts +8 -0
  31. package/dist/core/tablePipes.js +14 -0
  32. package/dist/core/underscoreBullets.d.ts +10 -0
  33. package/dist/core/underscoreBullets.js +32 -0
  34. package/dist/core/uniorgToMdast/elements.js +1 -1
  35. package/dist/core/uniorgToMdast/lists.js +12 -1
  36. package/dist/core/uniorgToMdast/objects.js +38 -6
  37. package/dist/core/uniorgToMdast/tables.js +12 -6
  38. package/dist/fileNames.d.ts +1 -1
  39. package/dist/fileNames.js +1 -1
  40. package/dist/markdownToOrg.js +19 -1
  41. package/dist/normalize.d.ts +2 -2
  42. package/dist/normalize.js +1 -1
  43. package/dist/options.d.ts +3 -3
  44. package/dist/orgToMarkdown.js +29 -5
  45. package/dist/presets/logseq.js +1 -1
  46. package/package.json +4 -1
@@ -0,0 +1,159 @@
1
+ import { unified } from "unified";
2
+ import uniorgParse from "uniorg-parse";
3
+ import { EXIT, visit } from "unist-util-visit";
4
+ import { isInline, orgParser, positionParser, renderChildren, tryParse } from "./render.js";
5
+ // org's `#+OPTIONS: ^:{}` limits sub/superscripts to the braced form
6
+ // (`H_{2}O`), so a bare underscore or caret (`a_b`, `x^y`) stays text.
7
+ // Markdown has no script syntax, so md text needs it; the braced
8
+ // scripts morg itself emits are unaffected
9
+ const BRACED_SCRIPTS = "^:{}";
10
+ // built once: constructing a processor per parse dominates the cost
11
+ const bracedScriptsParser = unified()
12
+ .use(uniorgParse, { useSubSuperscripts: "{}" })
13
+ .freeze();
14
+ function isScript(node) {
15
+ return node.type === "subscript" || node.type === "superscript";
16
+ }
17
+ // a tree uniorg fails to read counts as holding none
18
+ function countScripts(tree) {
19
+ let count = 0;
20
+ if (tree) {
21
+ visit(tree, (node) => {
22
+ if (isScript(node)) {
23
+ count++;
24
+ }
25
+ });
26
+ }
27
+ return count;
28
+ }
29
+ // a block's inline content as org renders it; a block element inside
30
+ // (a list item's nested list) only ends a line
31
+ function renderedContent(node) {
32
+ if (isInline(node) ||
33
+ !("children" in node) ||
34
+ !node.children.some(isInline)) {
35
+ return undefined;
36
+ }
37
+ return renderChildren(node.children).join("");
38
+ }
39
+ // a script org reads without `^:{}` only (uniorg's
40
+ // matchSubstringRegex without the braced form, loosened): a `_` or `^`
41
+ // after a non-blank, then `(`, `*`, or a word ending in a letter or
42
+ // digit. Text without one reads the same either way and skips the parses
43
+ const BARE_SCRIPT_RE = /\S[_^](?:[(*]|[+-]?[\p{L}\p{N}.,\\]*[\p{L}\p{N}])/u;
44
+ // whether `tree` (of `content`) holds a script not in the braced form.
45
+ // uniorg tries the braced form first, so without one both parsers read
46
+ // the same, and the braced parse can be skipped
47
+ function holdsBareScript(tree, content) {
48
+ let found = false;
49
+ visit(tree, (node) => {
50
+ const offset = node.position?.start.offset;
51
+ found =
52
+ isScript(node) && offset !== undefined && content[offset + 1] !== "{";
53
+ return found ? EXIT : undefined;
54
+ });
55
+ return found;
56
+ }
57
+ function readsBareScriptsIn(content) {
58
+ if (!BARE_SCRIPT_RE.test(content)) {
59
+ return false;
60
+ }
61
+ const tree = tryParse(content, positionParser);
62
+ return (tree !== undefined &&
63
+ holdsBareScript(tree, content) &&
64
+ countScripts(tree) > countScripts(tryParse(content, bracedScriptsParser)));
65
+ }
66
+ // whether org reads a script in the text that `^:{}` would keep text.
67
+ // Checked per block as rendered, not per text node: the char before a
68
+ // `_` or `^` may belong to a neighbor (a marker, a link's `]`, an
69
+ // escape); a script never spans blocks
70
+ function readsBareScripts(tree) {
71
+ let found = false;
72
+ visit(tree, (node) => {
73
+ const content = renderedContent(node);
74
+ found = content !== undefined && readsBareScriptsIn(content);
75
+ return found ? EXIT : undefined;
76
+ });
77
+ return found;
78
+ }
79
+ function isOptions(node) {
80
+ return (node.type === "keyword" && node.key.toUpperCase() === "OPTIONS");
81
+ }
82
+ /**
83
+ * md→org: adds `^:{}` to the document's `#+OPTIONS:` when its text
84
+ * holds a bare underscore or caret org would read as a script.
85
+ */
86
+ export function requireBracedScripts(uniorgAst) {
87
+ if (!readsBareScripts(uniorgAst)) {
88
+ return;
89
+ }
90
+ const options = uniorgAst.children.find(isOptions);
91
+ if (!options) {
92
+ uniorgAst.children.unshift({
93
+ type: "keyword",
94
+ key: "OPTIONS",
95
+ value: BRACED_SCRIPTS
96
+ });
97
+ }
98
+ else if (!/(^|\s)\^:/.test(options.value)) {
99
+ // an explicit ^: setting is the author's call
100
+ options.value = `${options.value} ${BRACED_SCRIPTS}`;
101
+ }
102
+ }
103
+ // the `^:` setting in an `#+OPTIONS:` value, if any
104
+ function scriptsSetting(options) {
105
+ const items = options.split(/\s+/).filter(item => item.startsWith("^:"));
106
+ return items.at(-1)?.slice(2);
107
+ }
108
+ // org→md honors every `^:` setting: `{}` limits scripts to the braced
109
+ // form, `nil` turns them off, `t` (org's default) keeps them on
110
+ const scriptsParsers = {
111
+ "{}": bracedScriptsParser,
112
+ nil: unified().use(uniorgParse, { useSubSuperscripts: false }).freeze()
113
+ };
114
+ // the settings the document may use, which the parser has to know up
115
+ // front; only a top-level keyword counts, which takes a parse to tell
116
+ function mayUseScripts(org) {
117
+ const settings = [...org.matchAll(/^[ \t]*#\+options:(.*)$/gim)].map(([, value]) => scriptsSetting(value ?? ""));
118
+ return [...new Set(settings)].filter((setting) => setting !== undefined);
119
+ }
120
+ function usesScripts(uniorgAst, setting) {
121
+ return uniorgAst.children.some(node => isOptions(node) && scriptsSetting(node.value) === setting);
122
+ }
123
+ // drops `^:{}` from the top-level `#+OPTIONS:`, the whole keyword if
124
+ // nothing else is left; md text has no scripts, so the return trip
125
+ // re-adds it wherever it is needed
126
+ function takeBracedScripts(uniorgAst) {
127
+ uniorgAst.children = uniorgAst.children.filter(node => {
128
+ if (!isOptions(node)) {
129
+ return true;
130
+ }
131
+ node.value = node.value
132
+ .split(/\s+/)
133
+ .filter(item => item && item !== BRACED_SCRIPTS)
134
+ .join(" ");
135
+ return node.value !== "";
136
+ });
137
+ }
138
+ /**
139
+ * org→md: parses org, honoring its `^:` setting, and consuming `^:{}`
140
+ * where the text needs it: md→org adds it only then, so anywhere else
141
+ * it is the author's own setting.
142
+ */
143
+ export function parseOrg(org) {
144
+ for (const setting of mayUseScripts(org)) {
145
+ const parser = scriptsParsers[setting];
146
+ if (!parser) {
147
+ continue;
148
+ }
149
+ // keywords parse the same either way
150
+ const uniorgAst = parser.parse(org);
151
+ if (usesScripts(uniorgAst, setting)) {
152
+ if (setting === "{}" && readsBareScripts(uniorgAst)) {
153
+ takeBracedScripts(uniorgAst);
154
+ }
155
+ return uniorgAst;
156
+ }
157
+ }
158
+ return orgParser.parse(org);
159
+ }
@@ -0,0 +1,10 @@
1
+ import type { Parent } from "unist";
2
+ /**
3
+ * md→org: escapes a literal `[fn:` in text.
4
+ */
5
+ export declare function escapeFootnoteReferences(tree: Parent): void;
6
+ /**
7
+ * org→md: drops the zero-width spaces `escapeFootnoteReferences`
8
+ * inserts.
9
+ */
10
+ export declare function unescapeFootnoteReferences(tree: Parent): void;
@@ -0,0 +1,24 @@
1
+ import { visit } from "unist-util-visit";
2
+ import { ZERO_WIDTH_SPACE } from "./markupBoundary.js";
3
+ // org reads `[fn:` in text as a footnote reference (`[fn:1]`, `[fn::x]`,
4
+ // `[fn:a:x]`), at a line start as a definition; a zero-width space after
5
+ // the `[` leaves it text
6
+ const REFERENCE = "[fn:";
7
+ const ESCAPED = `[${ZERO_WIDTH_SPACE}fn:`;
8
+ /**
9
+ * md→org: escapes a literal `[fn:` in text.
10
+ */
11
+ export function escapeFootnoteReferences(tree) {
12
+ visit(tree, "text", (node) => {
13
+ node.value = node.value?.replaceAll(REFERENCE, ESCAPED);
14
+ });
15
+ }
16
+ /**
17
+ * org→md: drops the zero-width spaces `escapeFootnoteReferences`
18
+ * inserts.
19
+ */
20
+ export function unescapeFootnoteReferences(tree) {
21
+ visit(tree, "text", (node) => {
22
+ node.value = node.value?.replaceAll(ESCAPED, REFERENCE);
23
+ });
24
+ }
@@ -0,0 +1,12 @@
1
+ import type { Parent } from "unist";
2
+ /**
3
+ * md→org: a paragraph line org would read as line syntax (an escaped
4
+ * `1\.` or `\*`, or a lazy continuation line) gets a leading zero-width
5
+ * space, or it would turn into a list item, headline, comment or table.
6
+ */
7
+ export declare function escapeLineSyntax(tree: Parent): void;
8
+ /**
9
+ * org→md: drops the line-start zero-width spaces `escapeLineSyntax`
10
+ * inserts.
11
+ */
12
+ export declare function unescapeLineSyntax(tree: Parent): void;
@@ -0,0 +1,206 @@
1
+ import { visit } from "unist-util-visit";
2
+ // a line starting with a zero-width space is no org line syntax (list
3
+ // item, headline, comment, keyword, table, ...), but renders as text
4
+ import { ZERO_WIDTH_SPACE } from "./markupBoundary.js";
5
+ import { delimiters, isInline, locate, positionParser, renderInline, tryParse } from "./render.js";
6
+ // org line syntax starts with a bullet or stars and a blank (`- `,
7
+ // `+ `, `** `), a rule (`-----`, table.el's `+-`), `#` and a blank or `+`,
8
+ // `|`, `:`, `[fn:`, `\begin{`, `%%(`, or a word followed by `.`, `)` or
9
+ // `:` (`1.`, `a)`, `CLOCK:`, `_.`); any other line (one starting with a
10
+ // link, markup or code, say) is text, and skips the parse
11
+ const MAY_BE_LINE_SYNTAX_RE = /^(?:[-+]|\*+)(?:\s|$)|^(?:-{5}|\+-|#(?:\s|$|\+)|[|:]|\[fn:|\\begin\{|%%\()|^[\p{L}\p{N}_]+[.):]/u;
12
+ // whether org reads `line` as anything but a plain paragraph
13
+ function readsAsLineSyntax(line) {
14
+ if (!MAY_BE_LINE_SYNTAX_RE.test(line)) {
15
+ return false;
16
+ }
17
+ // with its newline: a bullet ending the line (`1.`) needs one
18
+ const tree = tryParse(`${line}\n`);
19
+ if (!tree) {
20
+ // escaped, uniorg reads the line as text too
21
+ return true;
22
+ }
23
+ const [first, ...rest] = tree.children;
24
+ return first?.type !== "paragraph" || rest.length > 0;
25
+ }
26
+ // the rendered segments of inline content, descending into markup and
27
+ // link descriptions, since a line may start inside them (`*a\n# b*`)
28
+ function segments(children) {
29
+ return children.flatMap((node, index) => {
30
+ const at = { node, siblings: children, index };
31
+ if (!isInline(node)) {
32
+ // a block element (nested list, code block) only ends a line
33
+ return [{ ...at, text: "\n", opens: false }];
34
+ }
35
+ const inner = "children" in node ? node.children : [];
36
+ if (!inner.length) {
37
+ return [{ ...at, text: renderInline(node), opens: true }];
38
+ }
39
+ const [open, close] = delimiters(node);
40
+ return [
41
+ { ...at, text: open, opens: true },
42
+ ...segments(inner),
43
+ { ...at, text: close, opens: false }
44
+ ];
45
+ });
46
+ }
47
+ // the lines of the content as org renders it, not per text node: a
48
+ // line may start with another node (`[fn:1] a`) or its syntax span
49
+ // nodes (`* ~x~`). A line following a bullet or label has no start
50
+ function lineStarts(children, { afterBullet }) {
51
+ const parts = segments(children);
52
+ const rendered = parts.map(part => part.text);
53
+ const content = rendered.join("");
54
+ const breaks = [0];
55
+ for (let i = content.indexOf("\n"); i !== -1; i = content.indexOf("\n", i + 1)) {
56
+ breaks.push(i + 1);
57
+ }
58
+ const starts = breaks.flatMap((lineBreak, number) => {
59
+ // org keeps a continuation line's indentation in the text
60
+ const [, indent = "", line = ""] = /^([ \t]*)(.*)/.exec(content.slice(lineBreak)) ?? [];
61
+ const [index, offset] = locate(rendered, lineBreak + indent.length);
62
+ const segment = parts[index];
63
+ return line && segment && !(afterBullet && number === 0)
64
+ ? [{ segment, offset, line, number }]
65
+ : [];
66
+ });
67
+ return { starts, lines: content.split("\n") };
68
+ }
69
+ // calls `rewrite` on every paragraph, every list item whose inline
70
+ // content md→org flattens, and every headline (a logseq block's title
71
+ // may span lines), with its context;
72
+ // `applies` skips the rendering where there is nothing to rewrite
73
+ function rewriteLines(tree, rewrite, applies = () => true) {
74
+ visit(tree, (node, index, parent) => {
75
+ if (!["paragraph", "list-item", "headline"].includes(node.type)) {
76
+ return;
77
+ }
78
+ const children = node.children;
79
+ if (applies(children)) {
80
+ rewrite(children, contextOf(node, index ?? 0, parent));
81
+ }
82
+ });
83
+ }
84
+ function contextOf(node, index, parent) {
85
+ return {
86
+ afterBullet: node.type === "list-item" ||
87
+ node.type === "headline" ||
88
+ (index === 0 &&
89
+ (parent?.type === "list-item" ||
90
+ parent?.type === "footnote-definition")),
91
+ afterHeadline: parent?.children[index - 1]?.type === "headline",
92
+ headline: node.type === "headline"
93
+ };
94
+ }
95
+ /**
96
+ * md→org: a paragraph line org would read as line syntax (an escaped
97
+ * `1\.` or `\*`, or a lazy continuation line) gets a leading zero-width
98
+ * space, or it would turn into a list item, headline, comment or table.
99
+ */
100
+ export function escapeLineSyntax(tree) {
101
+ rewriteLines(tree, (children, context) => {
102
+ const { starts } = lineStarts(children, context);
103
+ // passthrough is a paragraph of its own
104
+ if (!context.afterBullet && isPassthrough(starts)) {
105
+ return;
106
+ }
107
+ // back to front, so earlier offsets and indices stay valid
108
+ for (const { segment, offset, line } of starts.reverse()) {
109
+ if (readsAsLineSyntax(line)) {
110
+ escapeLineStart(segment, offset);
111
+ }
112
+ }
113
+ escapeElementStarts(children, context);
114
+ });
115
+ }
116
+ // org→md writes these org elements as md paragraphs of their org text,
117
+ // to be read back as such (verbatim passthrough, see mappings.md)
118
+ const PASSTHROUGH_TYPES = new Set([
119
+ "fixed-width",
120
+ "drawer",
121
+ "clock",
122
+ "diary-sexp",
123
+ "keyword",
124
+ "babel-call",
125
+ "special-block",
126
+ "center-block",
127
+ "verse-block",
128
+ "comment-block",
129
+ "export-block"
130
+ ]);
131
+ // whether org reads the lines, all of them starting a line, as just
132
+ // one passthrough element
133
+ function isPassthrough(starts) {
134
+ const [first] = starts;
135
+ if (!first || !MAY_BE_LINE_SYNTAX_RE.test(first.line)) {
136
+ return false;
137
+ }
138
+ const lines = starts.map(({ line }) => line);
139
+ const [only, ...rest] = tryParse(`${lines.join("\n")}\n`)?.children ?? [];
140
+ return !rest.length && PASSTHROUGH_TYPES.has(only?.type ?? "");
141
+ }
142
+ // line syntax spanning lines (`#+begin_src`…`#+end_src`, a drawer) or
143
+ // depending on context (planning below a headline) shows only in the
144
+ // whole content as org reads it; each round escapes the first line of
145
+ // an element org reads there, as long as that exposes another
146
+ function escapeElementStarts(children, context) {
147
+ const escaped = new Set();
148
+ for (;;) {
149
+ const { starts, lines } = lineStarts(children, context);
150
+ if (!starts.some(({ line }) => MAY_BE_LINE_SYNTAX_RE.test(line))) {
151
+ return;
152
+ }
153
+ const number = elementLine(lines, context);
154
+ const start = starts.find(candidate => candidate.number === number);
155
+ if (number === undefined || !start || escaped.has(number)) {
156
+ return;
157
+ }
158
+ escaped.add(number);
159
+ escapeLineStart(start.segment, start.offset);
160
+ }
161
+ }
162
+ // the index of the first content line org reads as the start of an
163
+ // element other than a paragraph, if any
164
+ function elementLine(lines, context) {
165
+ // a line after a bullet only continues the item's text
166
+ const first = context.headline ? "* x" : "x";
167
+ const content = context.afterBullet ? [first, ...lines.slice(1)] : lines;
168
+ const prefix = context.afterHeadline ? ["* x"] : [];
169
+ const tree = tryParse([...prefix, ...content, ""].join("\n"), positionParser);
170
+ const elements = (tree?.children ?? []).flatMap(node => node.type === "section" && "children" in node
171
+ ? node.children
172
+ : [node]);
173
+ const element = elements.find(node => node.type !== "paragraph" && node.type !== "headline");
174
+ const line = element?.position?.start.line;
175
+ return line === undefined ? undefined : line - 1 - prefix.length;
176
+ }
177
+ function escapeLineStart({ node, siblings, index, opens }, offset) {
178
+ if (node.type === "text") {
179
+ const value = node.value ?? "";
180
+ node.value = value.slice(0, offset) + ZERO_WIDTH_SPACE + value.slice(offset);
181
+ }
182
+ else if (opens && offset === 0) {
183
+ siblings.splice(index, 0, { type: "text", value: ZERO_WIDTH_SPACE });
184
+ }
185
+ }
186
+ /**
187
+ * org→md: drops the line-start zero-width spaces `escapeLineSyntax`
188
+ * inserts.
189
+ */
190
+ export function unescapeLineSyntax(tree) {
191
+ rewriteLines(tree, (children, context) => {
192
+ const { starts } = lineStarts(children, context);
193
+ for (const { segment, offset } of starts.reverse()) {
194
+ const { node } = segment;
195
+ const value = node.value ?? "";
196
+ if (node.type === "text" && value[offset] === ZERO_WIDTH_SPACE) {
197
+ node.value = value.slice(0, offset) + value.slice(offset + 1);
198
+ }
199
+ }
200
+ }, holdsZeroWidthSpace);
201
+ }
202
+ function holdsZeroWidthSpace(children) {
203
+ return children.some(child => "children" in child
204
+ ? holdsZeroWidthSpace(child.children)
205
+ : child.type === "text" && child.value?.includes(ZERO_WIDTH_SPACE));
206
+ }
@@ -0,0 +1,12 @@
1
+ import type { Parent } from "unist";
2
+ export declare const ZERO_WIDTH_SPACE = "\u200B";
3
+ /**
4
+ * md→org: escapes with zero-width spaces what org would otherwise misread:
5
+ * literal markers in text that form markup (`/etc/`), and markup whose
6
+ * neighbor is no valid boundary (`a~x~s`), which would stay literal.
7
+ */
8
+ export declare function escapeOrgMarkup(tree: Parent): void;
9
+ /**
10
+ * org→md: drops the zero-width spaces `escapeOrgMarkup` inserts.
11
+ */
12
+ export declare function unescapeOrgMarkup(tree: Parent): void;
@@ -0,0 +1,195 @@
1
+ import { SKIP, visit } from "unist-util-visit";
2
+ import { locate, positionParser, renderChildren, tryParse } from "./render.js";
3
+ // the org manual's escape character: a zero-width space is a valid
4
+ // markup boundary (uniorg lists it in its emphasis regexp components)
5
+ // but renders as nothing
6
+ export const ZERO_WIDTH_SPACE = "\u200B";
7
+ const MARKUP_TYPES = new Set([
8
+ "bold",
9
+ "italic",
10
+ "underline",
11
+ "strike-through",
12
+ "code",
13
+ "verbatim"
14
+ ]);
15
+ // characters org accepts directly before an opening / after a closing
16
+ // marker (uniorg's emphasisRegexpComponents pre / post)
17
+ const PRE_RE = /[-–—\s\u200B('’"“”{]$/;
18
+ const POST_RE = /^[-–—\s\u200B.,:!?;'’"“”)}[]/;
19
+ function isMarkup(node) {
20
+ return MARKUP_TYPES.has(node?.type ?? "");
21
+ }
22
+ function allowsMarkupAfter(node) {
23
+ if (!node) {
24
+ return true;
25
+ }
26
+ if (node.type === "text") {
27
+ return !node.value || PRE_RE.test(node.value);
28
+ }
29
+ return false;
30
+ }
31
+ function allowsMarkupBefore(node) {
32
+ if (!node) {
33
+ return true;
34
+ }
35
+ if (node.type === "text") {
36
+ return !node.value || POST_RE.test(node.value);
37
+ }
38
+ // a link opens with `[`
39
+ return node.type === "link";
40
+ }
41
+ /**
42
+ * md→org: escapes with zero-width spaces what org would otherwise misread:
43
+ * literal markers in text that form markup (`/etc/`), and markup whose
44
+ * neighbor is no valid boundary (`a~x~s`), which would stay literal.
45
+ */
46
+ export function escapeOrgMarkup(tree) {
47
+ // the separators are valid markup boundaries, so literal markers are
48
+ // checked next to them
49
+ separateMarkupBoundaries(tree);
50
+ defuseLiteralMarkers(tree);
51
+ }
52
+ /**
53
+ * org→md: drops the zero-width spaces `escapeOrgMarkup` inserts.
54
+ */
55
+ export function unescapeOrgMarkup(tree) {
56
+ dropMarkupBoundaries(tree);
57
+ visit(tree, "text", (node) => {
58
+ node.value = node.value?.replace(DEFUSED_MARKER_RE, "$1");
59
+ });
60
+ }
61
+ const MARKER_RE = /[*/_=~+]/;
62
+ const MARKERS_RE = new RegExp(MARKER_RE.source, "g");
63
+ // org's emphasis rule, loosened, from an opening marker on: a non-blank
64
+ // after it, the same marker closing after a non-blank, then an allowed
65
+ // char, a table cell border or the end (the char before the marker is
66
+ // checked apart)
67
+ const MAY_OPEN_MARKUP_RE = /([*/_=~+])[^\s\u200B](?:[\s\S]*?[^\s\u200B])?\1(?:$|[-–—\s\u200B.,:!?;'’"“”)}[|])/y;
68
+ const DEFUSED_MARKER_RE = /([*/_=~+])\u200B/g;
69
+ // offsets of the opening markers of the outermost markup org reads in
70
+ // `text`; none where uniorg fails to read it
71
+ function markupOffsets(text) {
72
+ const tree = tryParse(text, positionParser);
73
+ if (!tree) {
74
+ return [];
75
+ }
76
+ const offsets = [];
77
+ visit(tree, (node) => {
78
+ if (!isMarkup(node)) {
79
+ return undefined;
80
+ }
81
+ const offset = node.position?.start.offset;
82
+ if (offset !== undefined) {
83
+ offsets.push(offset);
84
+ }
85
+ return SKIP;
86
+ });
87
+ return offsets;
88
+ }
89
+ function holdsMarker(node) {
90
+ return node.type === "text" && MARKER_RE.test(node.value ?? "");
91
+ }
92
+ // [child index, offset in its rendering] of each opening marker org
93
+ // reads as markup that lies in a text child, i.e. is a literal marker
94
+ function literalMarkers(children, rendered) {
95
+ return markupOffsets(rendered.join(""))
96
+ .map(offset => locate(rendered, offset))
97
+ .filter(([i]) => children[i]?.type === "text");
98
+ }
99
+ // whether a marker in a text child may open markup, by the rule above:
100
+ // a marker another node renders (bold's `*`) needs no escape, so
101
+ // neither it nor a marker org cannot read as markup takes a parse
102
+ function mayHoldLiteralMarkup(children, rendered) {
103
+ const line = rendered.join("");
104
+ let start = 0;
105
+ return rendered.some((part, i) => {
106
+ const offset = start;
107
+ start += part.length;
108
+ return (children[i]?.type === "text" &&
109
+ [...part.matchAll(MARKERS_RE)].some(({ index }) => {
110
+ MAY_OPEN_MARKUP_RE.lastIndex = offset + index;
111
+ return mayOpenAt(part, index) && MAY_OPEN_MARKUP_RE.test(line);
112
+ }));
113
+ });
114
+ }
115
+ // uniorg checks the char before an opening marker (`a/b/` is no
116
+ // markup), unless there is none in the text it parses: a char of another
117
+ // node's rendering counts as none. It reads a marker after one (`#**x*`)
118
+ // by backing off onto the first, which then opens without that check
119
+ function mayOpenAt(text, index) {
120
+ const before = text[index - 1];
121
+ return (before === undefined ||
122
+ PRE_RE.test(before) ||
123
+ // uniorg opens after a table cell border too, found by fuzzing
124
+ before === "|" ||
125
+ MARKER_RE.test(text[index + 1] ?? ""));
126
+ }
127
+ // a zero-width space after an opening marker leaves org nothing to read
128
+ // as markup (content may not start with one). Defusing an outer pair can
129
+ // expose an inner one, hence the rounds; each defuses at least one
130
+ // marker, so there are at most as many rounds as markers
131
+ function defuseRendered(children, rendered) {
132
+ let rounds = rendered.join("").match(MARKERS_RE)?.length ?? 0;
133
+ let markers;
134
+ while (rounds-- > 0 &&
135
+ mayHoldLiteralMarkup(children, rendered) &&
136
+ (markers = literalMarkers(children, rendered)).length) {
137
+ for (const [i, offset] of markers.reverse()) {
138
+ const part = rendered[i] ?? "";
139
+ rendered[i] =
140
+ part.slice(0, offset + 1) + ZERO_WIDTH_SPACE + part.slice(offset + 1);
141
+ }
142
+ }
143
+ }
144
+ // checked on the rendered line, not per text node: another inline node
145
+ // may split a literal pair (`*b ~x~ c*`)
146
+ function defuseLiteralMarkers(tree) {
147
+ visit(tree, (node) => {
148
+ if (!("children" in node) || !node.children.some(holdsMarker)) {
149
+ return;
150
+ }
151
+ const children = node.children;
152
+ const rendered = renderChildren(children);
153
+ defuseRendered(children, rendered);
154
+ for (const [i, child] of children.entries()) {
155
+ if (child.type === "text") {
156
+ child.value = rendered[i];
157
+ }
158
+ }
159
+ });
160
+ }
161
+ function separateMarkupBoundaries(tree) {
162
+ visit(tree, (node) => {
163
+ if (!("children" in node) || !node.children.some(isMarkup)) {
164
+ return;
165
+ }
166
+ const children = [];
167
+ for (const [i, child] of node.children.entries()) {
168
+ if (isMarkup(child) && !allowsMarkupAfter(children.at(-1))) {
169
+ children.push({ type: "text", value: ZERO_WIDTH_SPACE });
170
+ }
171
+ children.push(child);
172
+ if (isMarkup(child) && !allowsMarkupBefore(node.children[i + 1])) {
173
+ children.push({ type: "text", value: ZERO_WIDTH_SPACE });
174
+ }
175
+ }
176
+ node.children = children;
177
+ });
178
+ }
179
+ function dropMarkupBoundaries(tree) {
180
+ visit(tree, "text", (node, index, parent) => {
181
+ if (index === undefined || !parent) {
182
+ return;
183
+ }
184
+ let value = node.value ?? "";
185
+ if (isMarkup(parent.children[index + 1]) &&
186
+ value.endsWith(ZERO_WIDTH_SPACE)) {
187
+ value = value.slice(0, -1);
188
+ }
189
+ if (isMarkup(parent.children[index - 1]) &&
190
+ value.startsWith(ZERO_WIDTH_SPACE)) {
191
+ value = value.slice(1);
192
+ }
193
+ node.value = value;
194
+ });
195
+ }
@@ -98,7 +98,7 @@ export function transformMdastHtml(ctx, node) {
98
98
  : null;
99
99
  }
100
100
  // a bare <dl> whose body is nothing but attribute-less <dt>/<dd> pairs
101
- // (any whitespace between tags) becomes a ` :: ` list — the same
101
+ // (any whitespace between tags) becomes a ` :: ` list, the same
102
102
  // markdown convention descriptive lists use without useHtml, so the
103
103
  // org side re-parses it as a native descriptive list; anything richer
104
104
  // stays a preserved md-ism
@@ -45,7 +45,7 @@ export function transformMdastNodeToUniorgNode(ctx, node) {
45
45
  case "paragraph": {
46
46
  // a paragraph of only #+KEY: lines is affiliated keywords (or
47
47
  // mid-file keywords) traveling verbatim; emit as raw text so they
48
- // glue to the following element without a blank line — org only
48
+ // glue to the following element without a blank line: org only
49
49
  // attaches affiliated keywords when directly above their element
50
50
  const keywordLines = keywordOnlyLines(node);
51
51
  if (keywordLines) {
@@ -78,7 +78,7 @@ export function transformMdastNodeToUniorgNode(ctx, node) {
78
78
  case "blockquote":
79
79
  return {
80
80
  type: "quote-block",
81
- children: transformBlockChildren(ctx, node.children)
81
+ children: transformBlockChildren(ctx, node.children.map(child => quotedHeadingAsText(ctx, child)))
82
82
  };
83
83
  case "code":
84
84
  return transformMdastCode(node);
@@ -93,3 +93,11 @@ export function transformMdastNodeToUniorgNode(ctx, node) {
93
93
  return null;
94
94
  }
95
95
  }
96
+ // Emacs ends a quote block at a headline; the text stays
97
+ function quotedHeadingAsText(ctx, node) {
98
+ if (node.type !== "heading") {
99
+ return node;
100
+ }
101
+ warn(ctx, "heading inside a blockquote became text");
102
+ return { type: "paragraph", children: node.children };
103
+ }
@@ -1,4 +1,4 @@
1
1
  import type { List as MdastList } from "mdast";
2
2
  import type { List } from "uniorg";
3
- import type { TransformContext } from "./context.js";
3
+ import { type TransformContext } from "./context.js";
4
4
  export declare function transformMdastList(ctx: TransformContext, listNode: MdastList, indent: number): List;