@remigius42/morg 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -1
- package/dist/core/affiliated.d.ts +13 -0
- package/dist/core/affiliated.js +43 -0
- package/dist/core/backslashCommands.d.ts +10 -0
- package/dist/core/backslashCommands.js +43 -0
- package/dist/core/bracedScripts.d.ts +12 -0
- package/dist/core/bracedScripts.js +159 -0
- package/dist/core/footnoteReferences.d.ts +10 -0
- package/dist/core/footnoteReferences.js +24 -0
- package/dist/core/frontmatterBlock.d.ts +89 -0
- package/dist/core/frontmatterBlock.js +448 -0
- package/dist/core/keyValueLines.d.ts +6 -0
- package/dist/core/keyValueLines.js +18 -0
- package/dist/core/lineSyntax.d.ts +21 -0
- package/dist/core/lineSyntax.js +221 -0
- package/dist/core/markupBoundary.d.ts +12 -0
- package/dist/core/markupBoundary.js +195 -0
- package/dist/core/mdastToUniorg/blocks.d.ts +0 -1
- package/dist/core/mdastToUniorg/blocks.js +3 -31
- package/dist/core/mdastToUniorg/index.d.ts +6 -0
- package/dist/core/mdastToUniorg/index.js +81 -8
- package/dist/core/mdastToUniorg/lists.d.ts +1 -1
- package/dist/core/mdastToUniorg/lists.js +21 -2
- package/dist/core/mdastToUniorg/phrasing.js +145 -12
- package/dist/core/orgPath.d.ts +9 -0
- package/dist/core/orgPath.js +24 -0
- package/dist/core/render.d.ts +33 -0
- package/dist/core/render.js +101 -0
- package/dist/core/tablePipes.d.ts +17 -0
- package/dist/core/tablePipes.js +47 -0
- package/dist/core/underscoreBullets.d.ts +10 -0
- package/dist/core/underscoreBullets.js +32 -0
- package/dist/core/uniorgToMdast/elements.js +1 -1
- package/dist/core/uniorgToMdast/index.d.ts +3 -2
- package/dist/core/uniorgToMdast/index.js +100 -31
- package/dist/core/uniorgToMdast/lists.js +12 -1
- package/dist/core/uniorgToMdast/objects.js +44 -6
- package/dist/core/uniorgToMdast/shared.d.ts +5 -0
- package/dist/core/uniorgToMdast/shared.js +3 -12
- package/dist/core/uniorgToMdast/tables.js +12 -6
- package/dist/markdownToOrg.js +44 -23
- package/dist/orgToMarkdown.js +39 -5
- package/dist/presets/logseq.js +130 -0
- package/dist/presets/types.d.ts +4 -1
- package/package.json +4 -1
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
import { visit } from "unist-util-visit";
|
|
2
|
+
// a line starting with a zero-width space is no org line syntax (list
|
|
3
|
+
// item, headline, comment, keyword, table, ...), but renders as text
|
|
4
|
+
import { ZERO_WIDTH_SPACE } from "./markupBoundary.js";
|
|
5
|
+
import { delimiters, isInline, locate, positionParser, renderInline, tryParse } from "./render.js";
|
|
6
|
+
// org line syntax starts with a bullet or stars and a blank (`- `,
|
|
7
|
+
// `+ `, `** `), a rule (`-----`, table.el's `+-`), `#` and a blank or `+`,
|
|
8
|
+
// `|`, `:`, `[fn:`, `\begin{`, `%%(`, or a word followed by `.`, `)` or
|
|
9
|
+
// `:` (`1.`, `a)`, `CLOCK:`, `_.`); any other line (one starting with a
|
|
10
|
+
// link, markup or code, say) is text, and skips the parse
|
|
11
|
+
const MAY_BE_LINE_SYNTAX_RE = /^(?:[-+]|\*+)(?:\s|$)|^(?:-{5}|\+-|#(?:\s|$|\+)|[|:]|\[fn:|\\begin\{|%%\()|^[\p{L}\p{N}_]+[.):]/u;
|
|
12
|
+
// whether org reads `line` as anything but a plain paragraph
|
|
13
|
+
function readsAsLineSyntax(line) {
|
|
14
|
+
if (!MAY_BE_LINE_SYNTAX_RE.test(line)) {
|
|
15
|
+
return false;
|
|
16
|
+
}
|
|
17
|
+
// with its newline: a bullet ending the line (`1.`) needs one
|
|
18
|
+
const tree = tryParse(`${line}\n`);
|
|
19
|
+
if (!tree) {
|
|
20
|
+
// escaped, uniorg reads the line as text too
|
|
21
|
+
return true;
|
|
22
|
+
}
|
|
23
|
+
const [first, ...rest] = tree.children;
|
|
24
|
+
return first?.type !== "paragraph" || rest.length > 0;
|
|
25
|
+
}
|
|
26
|
+
// the rendered segments of inline content, descending into markup and
|
|
27
|
+
// link descriptions, since a line may start inside them (`*a\n# b*`)
|
|
28
|
+
function segments(children) {
|
|
29
|
+
return children.flatMap((node, index) => {
|
|
30
|
+
const at = { node, siblings: children, index };
|
|
31
|
+
if (!isInline(node)) {
|
|
32
|
+
// a block element (nested list, code block) only ends a line
|
|
33
|
+
return [{ ...at, text: "\n", opens: false }];
|
|
34
|
+
}
|
|
35
|
+
const inner = "children" in node ? node.children : [];
|
|
36
|
+
if (!inner.length) {
|
|
37
|
+
return [{ ...at, text: renderInline(node), opens: true }];
|
|
38
|
+
}
|
|
39
|
+
const [open, close] = delimiters(node);
|
|
40
|
+
return [
|
|
41
|
+
{ ...at, text: open, opens: true },
|
|
42
|
+
...segments(inner),
|
|
43
|
+
{ ...at, text: close, opens: false }
|
|
44
|
+
];
|
|
45
|
+
});
|
|
46
|
+
}
|
|
47
|
+
// the lines of the content as org renders it, not per text node: a
|
|
48
|
+
// line may start with another node (`[fn:1] a`) or its syntax span
|
|
49
|
+
// nodes (`* ~x~`). A line following a bullet or label has no start
|
|
50
|
+
function lineStarts(children, { afterBullet }) {
|
|
51
|
+
const parts = segments(children);
|
|
52
|
+
const rendered = parts.map(part => part.text);
|
|
53
|
+
const content = rendered.join("");
|
|
54
|
+
const breaks = [0];
|
|
55
|
+
for (let i = content.indexOf("\n"); i !== -1; i = content.indexOf("\n", i + 1)) {
|
|
56
|
+
breaks.push(i + 1);
|
|
57
|
+
}
|
|
58
|
+
const starts = breaks.flatMap((lineBreak, number) => {
|
|
59
|
+
// org keeps a continuation line's indentation in the text
|
|
60
|
+
const [, indent = "", line = ""] = /^([ \t]*)(.*)/.exec(content.slice(lineBreak)) ?? [];
|
|
61
|
+
const [index, offset] = locate(rendered, lineBreak + indent.length);
|
|
62
|
+
const segment = parts[index];
|
|
63
|
+
return line && segment && !(afterBullet && number === 0)
|
|
64
|
+
? [{ segment, offset, line, number }]
|
|
65
|
+
: [];
|
|
66
|
+
});
|
|
67
|
+
return { starts, lines: content.split("\n") };
|
|
68
|
+
}
|
|
69
|
+
// calls `rewrite` on every paragraph, every list item whose inline
|
|
70
|
+
// content md→org flattens, and every headline (a logseq block's title
|
|
71
|
+
// may span lines), with its context;
|
|
72
|
+
// `applies` skips the rendering where there is nothing to rewrite
|
|
73
|
+
function rewriteLines(tree, rewrite, applies = () => true) {
|
|
74
|
+
visit(tree, (node, index, parent) => {
|
|
75
|
+
if (!["paragraph", "list-item", "headline"].includes(node.type)) {
|
|
76
|
+
return;
|
|
77
|
+
}
|
|
78
|
+
const children = node.children;
|
|
79
|
+
if (applies(children)) {
|
|
80
|
+
rewrite(children, contextOf(node, index ?? 0, parent));
|
|
81
|
+
}
|
|
82
|
+
});
|
|
83
|
+
}
|
|
84
|
+
function contextOf(node, index, parent) {
|
|
85
|
+
return {
|
|
86
|
+
afterBullet: node.type === "list-item" ||
|
|
87
|
+
node.type === "headline" ||
|
|
88
|
+
(index === 0 &&
|
|
89
|
+
(parent?.type === "list-item" ||
|
|
90
|
+
parent?.type === "footnote-definition")),
|
|
91
|
+
afterHeadline: parent?.children[index - 1]?.type === "headline",
|
|
92
|
+
headline: node.type === "headline"
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* md→org: a paragraph line org would read as line syntax (an escaped
|
|
97
|
+
* `1\.` or `\*`, or a lazy continuation line) gets a leading zero-width
|
|
98
|
+
* space, or it would turn into a list item, headline, comment or table.
|
|
99
|
+
*/
|
|
100
|
+
export function escapeLineSyntax(tree) {
|
|
101
|
+
rewriteLines(tree, (children, context) => {
|
|
102
|
+
const { starts } = lineStarts(children, context);
|
|
103
|
+
// passthrough is a paragraph of its own
|
|
104
|
+
if (!context.afterBullet && isPassthrough(starts)) {
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
// back to front, so earlier offsets and indices stay valid
|
|
108
|
+
for (const { segment, offset, line } of starts.reverse()) {
|
|
109
|
+
if (readsAsLineSyntax(line)) {
|
|
110
|
+
escapeLineStart(segment, offset);
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
escapeElementStarts(children, context);
|
|
114
|
+
});
|
|
115
|
+
}
|
|
116
|
+
// org→md writes these org elements as md paragraphs of their org text,
|
|
117
|
+
// to be read back as such (verbatim passthrough, see mappings.md)
|
|
118
|
+
const PASSTHROUGH_TYPES = new Set([
|
|
119
|
+
"fixed-width",
|
|
120
|
+
"drawer",
|
|
121
|
+
"clock",
|
|
122
|
+
"diary-sexp",
|
|
123
|
+
"keyword",
|
|
124
|
+
"babel-call",
|
|
125
|
+
"special-block",
|
|
126
|
+
"center-block",
|
|
127
|
+
"verse-block",
|
|
128
|
+
"comment-block",
|
|
129
|
+
"export-block"
|
|
130
|
+
]);
|
|
131
|
+
// whether org reads the lines, all of them starting a line, as just
|
|
132
|
+
// one passthrough element
|
|
133
|
+
function isPassthrough(starts) {
|
|
134
|
+
const [first] = starts;
|
|
135
|
+
if (!first || !MAY_BE_LINE_SYNTAX_RE.test(first.line)) {
|
|
136
|
+
return false;
|
|
137
|
+
}
|
|
138
|
+
const lines = starts.map(({ line }) => line);
|
|
139
|
+
const [only, ...rest] = tryParse(`${lines.join("\n")}\n`)?.children ?? [];
|
|
140
|
+
return !rest.length && PASSTHROUGH_TYPES.has(only?.type ?? "");
|
|
141
|
+
}
|
|
142
|
+
/**
|
|
143
|
+
* md→org: whether a node is a paragraph org→md wrote as the org text of
|
|
144
|
+
* one passthrough element, which goes back as it is.
|
|
145
|
+
* @param node The node.
|
|
146
|
+
* @param index Its index in its parent.
|
|
147
|
+
* @param parent Its parent.
|
|
148
|
+
*/
|
|
149
|
+
export function isPassthroughParagraph(node, index, parent) {
|
|
150
|
+
if (node.type !== "paragraph") {
|
|
151
|
+
return false;
|
|
152
|
+
}
|
|
153
|
+
const context = contextOf(node, index, parent);
|
|
154
|
+
const children = node.children;
|
|
155
|
+
return (!context.afterBullet && isPassthrough(lineStarts(children, context).starts));
|
|
156
|
+
}
|
|
157
|
+
// line syntax spanning lines (`#+begin_src`…`#+end_src`, a drawer) or
|
|
158
|
+
// depending on context (planning below a headline) shows only in the
|
|
159
|
+
// whole content as org reads it; each round escapes the first line of
|
|
160
|
+
// an element org reads there, as long as that exposes another
|
|
161
|
+
function escapeElementStarts(children, context) {
|
|
162
|
+
const escaped = new Set();
|
|
163
|
+
for (;;) {
|
|
164
|
+
const { starts, lines } = lineStarts(children, context);
|
|
165
|
+
if (!starts.some(({ line }) => MAY_BE_LINE_SYNTAX_RE.test(line))) {
|
|
166
|
+
return;
|
|
167
|
+
}
|
|
168
|
+
const number = elementLine(lines, context);
|
|
169
|
+
const start = starts.find(candidate => candidate.number === number);
|
|
170
|
+
if (number === undefined || !start || escaped.has(number)) {
|
|
171
|
+
return;
|
|
172
|
+
}
|
|
173
|
+
escaped.add(number);
|
|
174
|
+
escapeLineStart(start.segment, start.offset);
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
// the index of the first content line org reads as the start of an
|
|
178
|
+
// element other than a paragraph, if any
|
|
179
|
+
function elementLine(lines, context) {
|
|
180
|
+
// a line after a bullet only continues the item's text
|
|
181
|
+
const first = context.headline ? "* x" : "x";
|
|
182
|
+
const content = context.afterBullet ? [first, ...lines.slice(1)] : lines;
|
|
183
|
+
const prefix = context.afterHeadline ? ["* x"] : [];
|
|
184
|
+
const tree = tryParse([...prefix, ...content, ""].join("\n"), positionParser);
|
|
185
|
+
const elements = (tree?.children ?? []).flatMap(node => node.type === "section" && "children" in node
|
|
186
|
+
? node.children
|
|
187
|
+
: [node]);
|
|
188
|
+
const element = elements.find(node => node.type !== "paragraph" && node.type !== "headline");
|
|
189
|
+
const line = element?.position?.start.line;
|
|
190
|
+
return line === undefined ? undefined : line - 1 - prefix.length;
|
|
191
|
+
}
|
|
192
|
+
function escapeLineStart({ node, siblings, index, opens }, offset) {
|
|
193
|
+
if (node.type === "text") {
|
|
194
|
+
const value = node.value ?? "";
|
|
195
|
+
node.value = value.slice(0, offset) + ZERO_WIDTH_SPACE + value.slice(offset);
|
|
196
|
+
}
|
|
197
|
+
else if (opens && offset === 0) {
|
|
198
|
+
siblings.splice(index, 0, { type: "text", value: ZERO_WIDTH_SPACE });
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
/**
|
|
202
|
+
* org→md: drops the line-start zero-width spaces `escapeLineSyntax`
|
|
203
|
+
* inserts.
|
|
204
|
+
*/
|
|
205
|
+
export function unescapeLineSyntax(tree) {
|
|
206
|
+
rewriteLines(tree, (children, context) => {
|
|
207
|
+
const { starts } = lineStarts(children, context);
|
|
208
|
+
for (const { segment, offset } of starts.reverse()) {
|
|
209
|
+
const { node } = segment;
|
|
210
|
+
const value = node.value ?? "";
|
|
211
|
+
if (node.type === "text" && value[offset] === ZERO_WIDTH_SPACE) {
|
|
212
|
+
node.value = value.slice(0, offset) + value.slice(offset + 1);
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
}, holdsZeroWidthSpace);
|
|
216
|
+
}
|
|
217
|
+
function holdsZeroWidthSpace(children) {
|
|
218
|
+
return children.some(child => "children" in child
|
|
219
|
+
? holdsZeroWidthSpace(child.children)
|
|
220
|
+
: child.type === "text" && child.value?.includes(ZERO_WIDTH_SPACE));
|
|
221
|
+
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { Parent } from "unist";
|
|
2
|
+
export declare const ZERO_WIDTH_SPACE = "\u200B";
|
|
3
|
+
/**
|
|
4
|
+
* md→org: escapes with zero-width spaces what org would otherwise misread:
|
|
5
|
+
* literal markers in text that form markup (`/etc/`), and markup whose
|
|
6
|
+
* neighbor is no valid boundary (`a~x~s`), which would stay literal.
|
|
7
|
+
*/
|
|
8
|
+
export declare function escapeOrgMarkup(tree: Parent): void;
|
|
9
|
+
/**
|
|
10
|
+
* org→md: drops the zero-width spaces `escapeOrgMarkup` inserts.
|
|
11
|
+
*/
|
|
12
|
+
export declare function unescapeOrgMarkup(tree: Parent): void;
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
import { SKIP, visit } from "unist-util-visit";
|
|
2
|
+
import { locate, positionParser, renderChildren, tryParse } from "./render.js";
|
|
3
|
+
// the org manual's escape character: a zero-width space is a valid
|
|
4
|
+
// markup boundary (uniorg lists it in its emphasis regexp components)
|
|
5
|
+
// but renders as nothing
|
|
6
|
+
export const ZERO_WIDTH_SPACE = "\u200B";
|
|
7
|
+
const MARKUP_TYPES = new Set([
|
|
8
|
+
"bold",
|
|
9
|
+
"italic",
|
|
10
|
+
"underline",
|
|
11
|
+
"strike-through",
|
|
12
|
+
"code",
|
|
13
|
+
"verbatim"
|
|
14
|
+
]);
|
|
15
|
+
// characters org accepts directly before an opening / after a closing
|
|
16
|
+
// marker (uniorg's emphasisRegexpComponents pre / post)
|
|
17
|
+
const PRE_RE = /[-–—\s\u200B('’"“”{]$/;
|
|
18
|
+
const POST_RE = /^[-–—\s\u200B.,:!?;'’"“”)}[]/;
|
|
19
|
+
function isMarkup(node) {
|
|
20
|
+
return MARKUP_TYPES.has(node?.type ?? "");
|
|
21
|
+
}
|
|
22
|
+
function allowsMarkupAfter(node) {
|
|
23
|
+
if (!node) {
|
|
24
|
+
return true;
|
|
25
|
+
}
|
|
26
|
+
if (node.type === "text") {
|
|
27
|
+
return !node.value || PRE_RE.test(node.value);
|
|
28
|
+
}
|
|
29
|
+
return false;
|
|
30
|
+
}
|
|
31
|
+
function allowsMarkupBefore(node) {
|
|
32
|
+
if (!node) {
|
|
33
|
+
return true;
|
|
34
|
+
}
|
|
35
|
+
if (node.type === "text") {
|
|
36
|
+
return !node.value || POST_RE.test(node.value);
|
|
37
|
+
}
|
|
38
|
+
// a link opens with `[`
|
|
39
|
+
return node.type === "link";
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* md→org: escapes with zero-width spaces what org would otherwise misread:
|
|
43
|
+
* literal markers in text that form markup (`/etc/`), and markup whose
|
|
44
|
+
* neighbor is no valid boundary (`a~x~s`), which would stay literal.
|
|
45
|
+
*/
|
|
46
|
+
export function escapeOrgMarkup(tree) {
|
|
47
|
+
// the separators are valid markup boundaries, so literal markers are
|
|
48
|
+
// checked next to them
|
|
49
|
+
separateMarkupBoundaries(tree);
|
|
50
|
+
defuseLiteralMarkers(tree);
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* org→md: drops the zero-width spaces `escapeOrgMarkup` inserts.
|
|
54
|
+
*/
|
|
55
|
+
export function unescapeOrgMarkup(tree) {
|
|
56
|
+
dropMarkupBoundaries(tree);
|
|
57
|
+
visit(tree, "text", (node) => {
|
|
58
|
+
node.value = node.value?.replace(DEFUSED_MARKER_RE, "$1");
|
|
59
|
+
});
|
|
60
|
+
}
|
|
61
|
+
const MARKER_RE = /[*/_=~+]/;
|
|
62
|
+
const MARKERS_RE = new RegExp(MARKER_RE.source, "g");
|
|
63
|
+
// org's emphasis rule, loosened, from an opening marker on: a non-blank
|
|
64
|
+
// after it, the same marker closing after a non-blank, then an allowed
|
|
65
|
+
// char, a table cell border or the end (the char before the marker is
|
|
66
|
+
// checked apart)
|
|
67
|
+
const MAY_OPEN_MARKUP_RE = /([*/_=~+])[^\s\u200B](?:[\s\S]*?[^\s\u200B])?\1(?:$|[-–—\s\u200B.,:!?;'’"“”)}[|])/y;
|
|
68
|
+
const DEFUSED_MARKER_RE = /([*/_=~+])\u200B/g;
|
|
69
|
+
// offsets of the opening markers of the outermost markup org reads in
|
|
70
|
+
// `text`; none where uniorg fails to read it
|
|
71
|
+
function markupOffsets(text) {
|
|
72
|
+
const tree = tryParse(text, positionParser);
|
|
73
|
+
if (!tree) {
|
|
74
|
+
return [];
|
|
75
|
+
}
|
|
76
|
+
const offsets = [];
|
|
77
|
+
visit(tree, (node) => {
|
|
78
|
+
if (!isMarkup(node)) {
|
|
79
|
+
return undefined;
|
|
80
|
+
}
|
|
81
|
+
const offset = node.position?.start.offset;
|
|
82
|
+
if (offset !== undefined) {
|
|
83
|
+
offsets.push(offset);
|
|
84
|
+
}
|
|
85
|
+
return SKIP;
|
|
86
|
+
});
|
|
87
|
+
return offsets;
|
|
88
|
+
}
|
|
89
|
+
function holdsMarker(node) {
|
|
90
|
+
return node.type === "text" && MARKER_RE.test(node.value ?? "");
|
|
91
|
+
}
|
|
92
|
+
// [child index, offset in its rendering] of each opening marker org
|
|
93
|
+
// reads as markup that lies in a text child, i.e. is a literal marker
|
|
94
|
+
function literalMarkers(children, rendered) {
|
|
95
|
+
return markupOffsets(rendered.join(""))
|
|
96
|
+
.map(offset => locate(rendered, offset))
|
|
97
|
+
.filter(([i]) => children[i]?.type === "text");
|
|
98
|
+
}
|
|
99
|
+
// whether a marker in a text child may open markup, by the rule above:
|
|
100
|
+
// a marker another node renders (bold's `*`) needs no escape, so
|
|
101
|
+
// neither it nor a marker org cannot read as markup takes a parse
|
|
102
|
+
function mayHoldLiteralMarkup(children, rendered) {
|
|
103
|
+
const line = rendered.join("");
|
|
104
|
+
let start = 0;
|
|
105
|
+
return rendered.some((part, i) => {
|
|
106
|
+
const offset = start;
|
|
107
|
+
start += part.length;
|
|
108
|
+
return (children[i]?.type === "text" &&
|
|
109
|
+
[...part.matchAll(MARKERS_RE)].some(({ index }) => {
|
|
110
|
+
MAY_OPEN_MARKUP_RE.lastIndex = offset + index;
|
|
111
|
+
return mayOpenAt(part, index) && MAY_OPEN_MARKUP_RE.test(line);
|
|
112
|
+
}));
|
|
113
|
+
});
|
|
114
|
+
}
|
|
115
|
+
// uniorg checks the char before an opening marker (`a/b/` is no
|
|
116
|
+
// markup), unless there is none in the text it parses: a char of another
|
|
117
|
+
// node's rendering counts as none. It reads a marker after one (`#**x*`)
|
|
118
|
+
// by backing off onto the first, which then opens without that check
|
|
119
|
+
function mayOpenAt(text, index) {
|
|
120
|
+
const before = text[index - 1];
|
|
121
|
+
return (before === undefined ||
|
|
122
|
+
PRE_RE.test(before) ||
|
|
123
|
+
// uniorg opens after a table cell border too, found by fuzzing
|
|
124
|
+
before === "|" ||
|
|
125
|
+
MARKER_RE.test(text[index + 1] ?? ""));
|
|
126
|
+
}
|
|
127
|
+
// a zero-width space after an opening marker leaves org nothing to read
|
|
128
|
+
// as markup (content may not start with one). Defusing an outer pair can
|
|
129
|
+
// expose an inner one, hence the rounds; each defuses at least one
|
|
130
|
+
// marker, so there are at most as many rounds as markers
|
|
131
|
+
function defuseRendered(children, rendered) {
|
|
132
|
+
let rounds = rendered.join("").match(MARKERS_RE)?.length ?? 0;
|
|
133
|
+
let markers;
|
|
134
|
+
while (rounds-- > 0 &&
|
|
135
|
+
mayHoldLiteralMarkup(children, rendered) &&
|
|
136
|
+
(markers = literalMarkers(children, rendered)).length) {
|
|
137
|
+
for (const [i, offset] of markers.reverse()) {
|
|
138
|
+
const part = rendered[i] ?? "";
|
|
139
|
+
rendered[i] =
|
|
140
|
+
part.slice(0, offset + 1) + ZERO_WIDTH_SPACE + part.slice(offset + 1);
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
// checked on the rendered line, not per text node: another inline node
|
|
145
|
+
// may split a literal pair (`*b ~x~ c*`)
|
|
146
|
+
function defuseLiteralMarkers(tree) {
|
|
147
|
+
visit(tree, (node) => {
|
|
148
|
+
if (!("children" in node) || !node.children.some(holdsMarker)) {
|
|
149
|
+
return;
|
|
150
|
+
}
|
|
151
|
+
const children = node.children;
|
|
152
|
+
const rendered = renderChildren(children);
|
|
153
|
+
defuseRendered(children, rendered);
|
|
154
|
+
for (const [i, child] of children.entries()) {
|
|
155
|
+
if (child.type === "text") {
|
|
156
|
+
child.value = rendered[i];
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
});
|
|
160
|
+
}
|
|
161
|
+
function separateMarkupBoundaries(tree) {
|
|
162
|
+
visit(tree, (node) => {
|
|
163
|
+
if (!("children" in node) || !node.children.some(isMarkup)) {
|
|
164
|
+
return;
|
|
165
|
+
}
|
|
166
|
+
const children = [];
|
|
167
|
+
for (const [i, child] of node.children.entries()) {
|
|
168
|
+
if (isMarkup(child) && !allowsMarkupAfter(children.at(-1))) {
|
|
169
|
+
children.push({ type: "text", value: ZERO_WIDTH_SPACE });
|
|
170
|
+
}
|
|
171
|
+
children.push(child);
|
|
172
|
+
if (isMarkup(child) && !allowsMarkupBefore(node.children[i + 1])) {
|
|
173
|
+
children.push({ type: "text", value: ZERO_WIDTH_SPACE });
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
node.children = children;
|
|
177
|
+
});
|
|
178
|
+
}
|
|
179
|
+
function dropMarkupBoundaries(tree) {
|
|
180
|
+
visit(tree, "text", (node, index, parent) => {
|
|
181
|
+
if (index === undefined || !parent) {
|
|
182
|
+
return;
|
|
183
|
+
}
|
|
184
|
+
let value = node.value ?? "";
|
|
185
|
+
if (isMarkup(parent.children[index + 1]) &&
|
|
186
|
+
value.endsWith(ZERO_WIDTH_SPACE)) {
|
|
187
|
+
value = value.slice(0, -1);
|
|
188
|
+
}
|
|
189
|
+
if (isMarkup(parent.children[index - 1]) &&
|
|
190
|
+
value.startsWith(ZERO_WIDTH_SPACE)) {
|
|
191
|
+
value = value.slice(1);
|
|
192
|
+
}
|
|
193
|
+
node.value = value;
|
|
194
|
+
});
|
|
195
|
+
}
|
|
@@ -1,7 +1,6 @@
|
|
|
1
1
|
import type { RootContent, PhrasingContent } from "mdast";
|
|
2
2
|
import type { ElementType } from "uniorg";
|
|
3
3
|
import { type TransformContext } from "./context.js";
|
|
4
|
-
export declare function frontmatterToKeywords(yamlValue: string): ElementType[];
|
|
5
4
|
export declare function transformMdastTable(ctx: TransformContext, node: Extract<RootContent, {
|
|
6
5
|
type: "table";
|
|
7
6
|
}>): ElementType;
|
|
@@ -1,35 +1,7 @@
|
|
|
1
1
|
import { toString } from "orgast-util-to-string";
|
|
2
|
-
import { parse as parseYaml } from "yaml";
|
|
3
2
|
import { mdismEnabled } from "./context.js";
|
|
4
3
|
import { transformPhrasingChildren } from "./phrasing.js";
|
|
5
|
-
|
|
6
|
-
// scalar is JSON-encoded and restored by JSON.parse on the way back
|
|
7
|
-
function keywordValue(value) {
|
|
8
|
-
if (value !== null && typeof value === "object") {
|
|
9
|
-
return JSON.stringify(value);
|
|
10
|
-
}
|
|
11
|
-
const text = String(value);
|
|
12
|
-
return text.includes("\n") ? JSON.stringify(text) : text;
|
|
13
|
-
}
|
|
14
|
-
// frontmatter entries become #+KEY: value keywords; scalar values as-is,
|
|
15
|
-
// structured values JSON-encoded on a single line (see ADR 0002)
|
|
16
|
-
export function frontmatterToKeywords(yamlValue) {
|
|
17
|
-
const data = parseYaml(yamlValue);
|
|
18
|
-
if (!data || typeof data !== "object" || Array.isArray(data)) {
|
|
19
|
-
return [];
|
|
20
|
-
}
|
|
21
|
-
return Object.entries(data).flatMap(([key, value]) => {
|
|
22
|
-
// a sequence becomes repeated keywords -- org's own way of carrying
|
|
23
|
-
// several values for one key, and what they read back as
|
|
24
|
-
const values = Array.isArray(value) && value.length ? value : [value];
|
|
25
|
-
return values.map(item => ({
|
|
26
|
-
type: "keyword",
|
|
27
|
-
affiliated: {},
|
|
28
|
-
key: key.toUpperCase(),
|
|
29
|
-
value: keywordValue(item)
|
|
30
|
-
}));
|
|
31
|
-
});
|
|
32
|
-
}
|
|
4
|
+
import { KEYWORD_NAME } from "../frontmatterBlock.js";
|
|
33
5
|
export function transformMdastTable(ctx, node) {
|
|
34
6
|
const [headerRow, ...bodyRows] = node.children;
|
|
35
7
|
// GFM column alignment maps to an org alignment cookie row
|
|
@@ -196,7 +168,7 @@ export function transformMdastHeading(ctx, node) {
|
|
|
196
168
|
children: transformPhrasingChildren(ctx, node.children)
|
|
197
169
|
};
|
|
198
170
|
}
|
|
199
|
-
const KEYWORD_LINE_RE =
|
|
171
|
+
const KEYWORD_LINE_RE = new RegExp(String.raw `^#\+${KEYWORD_NAME}: `);
|
|
200
172
|
export function keywordOnlyLines(node) {
|
|
201
173
|
if (!node.children.every(child => child.type === "text")) {
|
|
202
174
|
return null;
|
|
@@ -204,7 +176,7 @@ export function keywordOnlyLines(node) {
|
|
|
204
176
|
const lines = node.children
|
|
205
177
|
.map(child => child.value)
|
|
206
178
|
.join("")
|
|
207
|
-
.split(
|
|
179
|
+
.split(/\r?\n/);
|
|
208
180
|
return lines.length && lines.every(line => KEYWORD_LINE_RE.test(line))
|
|
209
181
|
? lines
|
|
210
182
|
: null;
|
|
@@ -10,4 +10,10 @@ import type { MdastToUniorgOptions } from "./context.js";
|
|
|
10
10
|
* @returns The transformed uniorg AST.
|
|
11
11
|
*/
|
|
12
12
|
export declare function transformMdastToUniorgAst(mdast: MdastRoot, options?: MdastToUniorgOptions): OrgData;
|
|
13
|
+
/**
|
|
14
|
+
* md→org pipeline: like `transformMdastToUniorgAst`, but the frontmatter
|
|
15
|
+
* stays a `morg-frontmatter` node a preset can still take entries out
|
|
16
|
+
* of; `renderFileHeader` turns it into the block.
|
|
17
|
+
*/
|
|
18
|
+
export declare function transformMdastToUniorgDraft(mdast: MdastRoot, options?: MdastToUniorgOptions): OrgData;
|
|
13
19
|
export declare function transformMdastNodeToUniorgNode(ctx: TransformContext, node: RootContent | PhrasingContent): GreaterElementType | ElementType | Text | null;
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { visit } from "unist-util-visit";
|
|
2
2
|
import { warn } from "./context.js";
|
|
3
|
-
import {
|
|
3
|
+
import { fitsKeywordLine, fitsPropertyLine, markModeLine, renderFileHeader, takeMorgEntry, takesMorgEntries } from "../frontmatterBlock.js";
|
|
4
|
+
import { keywordOnlyLines, transformMdastCode, transformMdastHeading, transformMdastHtml, transformMdastMath, transformMdastTable } from "./blocks.js";
|
|
4
5
|
import { transformMdastList } from "./lists.js";
|
|
5
6
|
import { transformPhrasingChildren } from "./phrasing.js";
|
|
6
7
|
/**
|
|
@@ -10,6 +11,16 @@ import { transformPhrasingChildren } from "./phrasing.js";
|
|
|
10
11
|
* @returns The transformed uniorg AST.
|
|
11
12
|
*/
|
|
12
13
|
export function transformMdastToUniorgAst(mdast, options = {}) {
|
|
14
|
+
const orgAst = transformMdastToUniorgDraft(mdast, options);
|
|
15
|
+
renderFileHeader(orgAst);
|
|
16
|
+
return orgAst;
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* md→org pipeline: like `transformMdastToUniorgAst`, but the frontmatter
|
|
20
|
+
* stays a `morg-frontmatter` node a preset can still take entries out
|
|
21
|
+
* of; `renderFileHeader` turns it into the block.
|
|
22
|
+
*/
|
|
23
|
+
export function transformMdastToUniorgDraft(mdast, options = {}) {
|
|
13
24
|
const ctx = { options, definitions: new Map() };
|
|
14
25
|
visit(mdast, "definition", (definition) => {
|
|
15
26
|
ctx.definitions.set(definition.identifier, {
|
|
@@ -17,13 +28,20 @@ export function transformMdastToUniorgAst(mdast, options = {}) {
|
|
|
17
28
|
...(definition.title != null && { title: definition.title })
|
|
18
29
|
});
|
|
19
30
|
});
|
|
31
|
+
const firstBody = mdast.children.find(child => child.type !== "yaml");
|
|
20
32
|
const children = mdast.children
|
|
21
|
-
.flatMap(child =>
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
33
|
+
.flatMap(child => {
|
|
34
|
+
// frontmatter is passive data, org keywords can act; it travels
|
|
35
|
+
// inert in a marked comment block (ADR 0005)
|
|
36
|
+
if (child.type === "yaml") {
|
|
37
|
+
return transformFrontmatter(child.value);
|
|
38
|
+
}
|
|
39
|
+
const node = transformMdastNodeToUniorgNode(ctx, child);
|
|
40
|
+
if (child === firstBody) {
|
|
41
|
+
markModeLine(node);
|
|
42
|
+
}
|
|
43
|
+
return [node];
|
|
44
|
+
})
|
|
27
45
|
.filter(Boolean);
|
|
28
46
|
const orgAst = {
|
|
29
47
|
type: "org-data",
|
|
@@ -78,7 +96,7 @@ export function transformMdastNodeToUniorgNode(ctx, node) {
|
|
|
78
96
|
case "blockquote":
|
|
79
97
|
return {
|
|
80
98
|
type: "quote-block",
|
|
81
|
-
children: transformBlockChildren(ctx, node.children)
|
|
99
|
+
children: transformBlockChildren(ctx, node.children.map(child => quotedHeadingAsText(ctx, child)))
|
|
82
100
|
};
|
|
83
101
|
case "code":
|
|
84
102
|
return transformMdastCode(node);
|
|
@@ -93,3 +111,58 @@ export function transformMdastNodeToUniorgNode(ctx, node) {
|
|
|
93
111
|
return null;
|
|
94
112
|
}
|
|
95
113
|
}
|
|
114
|
+
// Emacs ends a quote block at a headline; the text stays
|
|
115
|
+
function quotedHeadingAsText(ctx, node) {
|
|
116
|
+
if (node.type !== "heading") {
|
|
117
|
+
return node;
|
|
118
|
+
}
|
|
119
|
+
warn(ctx, "heading inside a blockquote became text");
|
|
120
|
+
return { type: "paragraph", children: node.children };
|
|
121
|
+
}
|
|
122
|
+
// morg's own entries, both or neither: one left in the YAML would keep
|
|
123
|
+
// the other from joining it again on the way back
|
|
124
|
+
function takeMorgEntries(yaml) {
|
|
125
|
+
// most frontmatter holds neither: no need to parse it
|
|
126
|
+
if (!yaml.includes("morg_")) {
|
|
127
|
+
return { keywords: [], properties: [], yaml };
|
|
128
|
+
}
|
|
129
|
+
const taken = takeMorgEntry(yaml, "morg_keywords", fitsKeywordLine);
|
|
130
|
+
const rest = takeMorgEntry(taken.yaml, "morg_properties", fitsPropertyLine);
|
|
131
|
+
return takesMorgEntries(rest.yaml)
|
|
132
|
+
? { keywords: taken.keywords, properties: rest.keywords, yaml: rest.yaml }
|
|
133
|
+
: { keywords: [], properties: [], yaml };
|
|
134
|
+
}
|
|
135
|
+
function keywordNodes(keywords) {
|
|
136
|
+
return keywords.map(([key, value]) => ({
|
|
137
|
+
type: "keyword",
|
|
138
|
+
affiliated: {},
|
|
139
|
+
key,
|
|
140
|
+
value
|
|
141
|
+
}));
|
|
142
|
+
}
|
|
143
|
+
// org's own keywords go back to keywords; the rest of the frontmatter
|
|
144
|
+
// stays inert data in the block
|
|
145
|
+
function transformFrontmatter(value) {
|
|
146
|
+
// in the org file's line endings, which are LF
|
|
147
|
+
const lf = value.replaceAll("\r\n", "\n");
|
|
148
|
+
const { keywords, properties, yaml } = takeMorgEntries(lf);
|
|
149
|
+
const drawer = properties.length
|
|
150
|
+
? [
|
|
151
|
+
{
|
|
152
|
+
type: "property-drawer",
|
|
153
|
+
children: properties.map(([key, propertyValue]) => ({
|
|
154
|
+
type: "node-property",
|
|
155
|
+
key,
|
|
156
|
+
value: propertyValue
|
|
157
|
+
}))
|
|
158
|
+
}
|
|
159
|
+
]
|
|
160
|
+
: [];
|
|
161
|
+
const nodes = keywordNodes(keywords);
|
|
162
|
+
// an empty frontmatter stays a block
|
|
163
|
+
if (yaml || !(keywords.length || properties.length)) {
|
|
164
|
+
const node = { type: "morg-frontmatter", yaml };
|
|
165
|
+
nodes.push(node);
|
|
166
|
+
}
|
|
167
|
+
return [...drawer, ...nodes];
|
|
168
|
+
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
import type { List as MdastList } from "mdast";
|
|
2
2
|
import type { List } from "uniorg";
|
|
3
|
-
import type
|
|
3
|
+
import { type TransformContext } from "./context.js";
|
|
4
4
|
export declare function transformMdastList(ctx: TransformContext, listNode: MdastList, indent: number): List;
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { warn } from "./context.js";
|
|
1
2
|
import { transformPhrasingChildren } from "./phrasing.js";
|
|
2
3
|
// circular import with index.js is fine in ESM: both sides only export
|
|
3
4
|
// hoisted function declarations called after module initialization
|
|
@@ -17,19 +18,37 @@ export function transformMdastList(ctx, listNode, indent) {
|
|
|
17
18
|
contentsEnd: 0
|
|
18
19
|
};
|
|
19
20
|
}
|
|
21
|
+
// uniorg-stringify re-indents a list item's block by stripping up to
|
|
22
|
+
// the item's indentation from each line first: a code block's value
|
|
23
|
+
// has to carry that indentation (as uniorg's parser reads it), or its
|
|
24
|
+
// own indentation shrinks
|
|
25
|
+
function indentCode(node, level) {
|
|
26
|
+
const block = node;
|
|
27
|
+
if ((block?.type === "src-block" || block?.type === "example-block") &&
|
|
28
|
+
block.value !== undefined) {
|
|
29
|
+
block.value = block.value.replace(/^(?=.)/gm, " ".repeat(level));
|
|
30
|
+
}
|
|
31
|
+
return node;
|
|
32
|
+
}
|
|
20
33
|
function transformMdastListItem(ctx, item, indent, bullet) {
|
|
21
34
|
const children = item.children
|
|
22
35
|
.flatMap((child) => {
|
|
23
36
|
if (child.type === "list") {
|
|
24
37
|
return [transformMdastList(ctx, child, indent + bullet.length)];
|
|
25
38
|
}
|
|
26
|
-
if (child.type === "
|
|
39
|
+
if (child.type === "heading") {
|
|
40
|
+
// org headlines cannot live inside a list item; the text stays
|
|
41
|
+
warn(ctx, "heading inside a list item became text");
|
|
42
|
+
}
|
|
43
|
+
if (child.type === "paragraph" || child.type === "heading") {
|
|
27
44
|
return [
|
|
28
45
|
...transformPhrasingChildren(ctx, child.children),
|
|
29
46
|
{ type: "text", value: "\n" }
|
|
30
47
|
];
|
|
31
48
|
}
|
|
32
|
-
return [
|
|
49
|
+
return [
|
|
50
|
+
indentCode(transformMdastNodeToUniorgNode(ctx, child), indent + bullet.length)
|
|
51
|
+
];
|
|
33
52
|
})
|
|
34
53
|
.filter(Boolean);
|
|
35
54
|
return {
|