@remigius42/morg 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +36 -30
- package/dist/cli/args.d.ts +2 -0
- package/dist/cli/args.js +22 -60
- package/dist/cli/error.js +1 -1
- package/dist/cli/flags.d.ts +27 -0
- package/dist/cli/flags.js +127 -0
- package/dist/cli/formats.js +18 -4
- package/dist/cli/help.d.ts +1 -0
- package/dist/cli/help.js +28 -0
- package/dist/cli.js +24 -1
- package/dist/conversionOptions.d.ts +2 -2
- package/dist/conversionOptions.js +2 -2
- package/dist/core/bracedScripts.d.ts +12 -0
- package/dist/core/bracedScripts.js +159 -0
- package/dist/core/footnoteReferences.d.ts +10 -0
- package/dist/core/footnoteReferences.js +24 -0
- package/dist/core/lineSyntax.d.ts +12 -0
- package/dist/core/lineSyntax.js +206 -0
- package/dist/core/markupBoundary.d.ts +12 -0
- package/dist/core/markupBoundary.js +195 -0
- package/dist/core/mdastToUniorg/blocks.js +1 -1
- package/dist/core/mdastToUniorg/index.js +10 -2
- package/dist/core/mdastToUniorg/lists.d.ts +1 -1
- package/dist/core/mdastToUniorg/lists.js +21 -2
- package/dist/core/mdastToUniorg/phrasing.js +145 -12
- package/dist/core/orgPath.d.ts +9 -0
- package/dist/core/orgPath.js +24 -0
- package/dist/core/render.d.ts +33 -0
- package/dist/core/render.js +101 -0
- package/dist/core/tablePipes.d.ts +8 -0
- package/dist/core/tablePipes.js +14 -0
- package/dist/core/underscoreBullets.d.ts +10 -0
- package/dist/core/underscoreBullets.js +32 -0
- package/dist/core/uniorgToMdast/elements.js +1 -1
- package/dist/core/uniorgToMdast/lists.js +12 -1
- package/dist/core/uniorgToMdast/objects.js +38 -6
- package/dist/core/uniorgToMdast/tables.js +12 -6
- package/dist/fileNames.d.ts +1 -1
- package/dist/fileNames.js +1 -1
- package/dist/markdownToOrg.js +19 -1
- package/dist/normalize.d.ts +2 -2
- package/dist/normalize.js +1 -1
- package/dist/options.d.ts +3 -3
- package/dist/orgToMarkdown.js +29 -5
- package/dist/presets/logseq.js +1 -1
- package/package.json +4 -1
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { warn } from "./context.js";
|
|
1
2
|
import { transformPhrasingChildren } from "./phrasing.js";
|
|
2
3
|
// circular import with index.js is fine in ESM: both sides only export
|
|
3
4
|
// hoisted function declarations called after module initialization
|
|
@@ -17,19 +18,37 @@ export function transformMdastList(ctx, listNode, indent) {
|
|
|
17
18
|
contentsEnd: 0
|
|
18
19
|
};
|
|
19
20
|
}
|
|
21
|
+
// uniorg-stringify re-indents a list item's block by stripping up to
|
|
22
|
+
// the item's indentation from each line first: a code block's value
|
|
23
|
+
// has to carry that indentation (as uniorg's parser reads it), or its
|
|
24
|
+
// own indentation shrinks
|
|
25
|
+
function indentCode(node, level) {
|
|
26
|
+
const block = node;
|
|
27
|
+
if ((block?.type === "src-block" || block?.type === "example-block") &&
|
|
28
|
+
block.value !== undefined) {
|
|
29
|
+
block.value = block.value.replace(/^(?=.)/gm, " ".repeat(level));
|
|
30
|
+
}
|
|
31
|
+
return node;
|
|
32
|
+
}
|
|
20
33
|
function transformMdastListItem(ctx, item, indent, bullet) {
|
|
21
34
|
const children = item.children
|
|
22
35
|
.flatMap((child) => {
|
|
23
36
|
if (child.type === "list") {
|
|
24
37
|
return [transformMdastList(ctx, child, indent + bullet.length)];
|
|
25
38
|
}
|
|
26
|
-
if (child.type === "
|
|
39
|
+
if (child.type === "heading") {
|
|
40
|
+
// org headlines cannot live inside a list item; the text stays
|
|
41
|
+
warn(ctx, "heading inside a list item became text");
|
|
42
|
+
}
|
|
43
|
+
if (child.type === "paragraph" || child.type === "heading") {
|
|
27
44
|
return [
|
|
28
45
|
...transformPhrasingChildren(ctx, child.children),
|
|
29
46
|
{ type: "text", value: "\n" }
|
|
30
47
|
];
|
|
31
48
|
}
|
|
32
|
-
return [
|
|
49
|
+
return [
|
|
50
|
+
indentCode(transformMdastNodeToUniorgNode(ctx, child), indent + bullet.length)
|
|
51
|
+
];
|
|
33
52
|
})
|
|
34
53
|
.filter(Boolean);
|
|
35
54
|
return {
|
|
@@ -1,4 +1,7 @@
|
|
|
1
1
|
import { mdismEnabled, warn } from "./context.js";
|
|
2
|
+
import { visit } from "unist-util-visit";
|
|
3
|
+
import { escapeOrgPath } from "../orgPath.js";
|
|
4
|
+
import { orgParser, renderInline } from "../render.js";
|
|
2
5
|
// html tags morg itself emits under useHtml; with interpretHtml a bare
|
|
3
6
|
// open/close pair becomes the corresponding native org object
|
|
4
7
|
const INLINE_HTML_ORG_TYPES = {
|
|
@@ -30,20 +33,87 @@ export function transformPhrasingChildren(ctx, children) {
|
|
|
30
33
|
? matchInlineHtmlPair(children, i)
|
|
31
34
|
: null;
|
|
32
35
|
if (pair) {
|
|
33
|
-
result.push({
|
|
36
|
+
result.push(...hoistEdgeWhitespace(joinLines({
|
|
34
37
|
type: pair.orgType,
|
|
35
38
|
children: transformPhrasingChildren(ctx, children.slice(i + 1, pair.end))
|
|
36
|
-
});
|
|
39
|
+
})));
|
|
37
40
|
i = pair.end;
|
|
38
41
|
continue;
|
|
39
42
|
}
|
|
43
|
+
if (node.type === "inlineCode") {
|
|
44
|
+
result.push(...transformMdastInlineCode(ctx, node.value));
|
|
45
|
+
continue;
|
|
46
|
+
}
|
|
40
47
|
const transformed = transformMdastPhrasingContentToUniorgObject(ctx, node);
|
|
41
48
|
if (transformed) {
|
|
42
|
-
result.push(transformed);
|
|
49
|
+
result.push(...hoistEdgeWhitespace(joinLines(transformed)));
|
|
43
50
|
}
|
|
44
51
|
}
|
|
45
52
|
return result;
|
|
46
53
|
}
|
|
54
|
+
const CONTAINER_MARKUP = new Set([
|
|
55
|
+
"bold",
|
|
56
|
+
"italic",
|
|
57
|
+
"strike-through",
|
|
58
|
+
"underline"
|
|
59
|
+
]);
|
|
60
|
+
// org markup spans at most two lines; beyond that its line endings
|
|
61
|
+
// become spaces, as CommonMark renders them anyway (see
|
|
62
|
+
// transformMdastInlineCode)
|
|
63
|
+
function joinLines(node) {
|
|
64
|
+
if (!CONTAINER_MARKUP.has(node.type)) {
|
|
65
|
+
return node;
|
|
66
|
+
}
|
|
67
|
+
const texts = [];
|
|
68
|
+
visit(node, "text", (text) => {
|
|
69
|
+
texts.push(text);
|
|
70
|
+
});
|
|
71
|
+
const lineEndings = texts
|
|
72
|
+
.map(text => text.value)
|
|
73
|
+
.join("")
|
|
74
|
+
.split("\n").length - 1;
|
|
75
|
+
if (lineEndings > 1) {
|
|
76
|
+
for (const text of texts) {
|
|
77
|
+
text.value = text.value.replaceAll("\n", " ");
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
return node;
|
|
81
|
+
}
|
|
82
|
+
// strips the whitespace at the edges of `children`, returning it
|
|
83
|
+
function takeEdgeWhitespace(children) {
|
|
84
|
+
const first = children[0];
|
|
85
|
+
const last = children.at(-1);
|
|
86
|
+
let before = "";
|
|
87
|
+
let after = "";
|
|
88
|
+
if (first?.type === "text") {
|
|
89
|
+
before = /^\s*/.exec(first.value)?.[0] ?? "";
|
|
90
|
+
first.value = first.value.slice(before.length);
|
|
91
|
+
}
|
|
92
|
+
if (last?.type === "text") {
|
|
93
|
+
after = /\s*$/.exec(last.value)?.[0] ?? "";
|
|
94
|
+
last.value = last.value.slice(0, last.value.length - after.length);
|
|
95
|
+
}
|
|
96
|
+
return { before, after };
|
|
97
|
+
}
|
|
98
|
+
// org markup may not start or end with whitespace, which a code span's
|
|
99
|
+
// moved-out edge whitespace (see transformMdastInlineCode) can leave
|
|
100
|
+
// inside it; it moves out further, to the markup's siblings
|
|
101
|
+
function hoistEdgeWhitespace(node) {
|
|
102
|
+
if (!CONTAINER_MARKUP.has(node.type) || !("children" in node)) {
|
|
103
|
+
return [node];
|
|
104
|
+
}
|
|
105
|
+
const { before, after } = takeEdgeWhitespace(node.children);
|
|
106
|
+
if (!before && !after) {
|
|
107
|
+
return [node];
|
|
108
|
+
}
|
|
109
|
+
const content = node.children.filter(child => child.type !== "text" || child.value !== "");
|
|
110
|
+
const text = (value) => value ? [{ type: "text", value }] : [];
|
|
111
|
+
return [
|
|
112
|
+
...text(before),
|
|
113
|
+
...(content.length ? [{ ...node, children: content }] : []),
|
|
114
|
+
...text(after)
|
|
115
|
+
];
|
|
116
|
+
}
|
|
47
117
|
// an org bracket-link path cannot contain [ or ]; percent-encoding keeps
|
|
48
118
|
// the url equivalent and, unlike org's backslash escaping, survives being
|
|
49
119
|
// parsed back (uniorg does not decode `\[`). A silent normalization, not
|
|
@@ -51,6 +121,36 @@ export function transformPhrasingChildren(ctx, children) {
|
|
|
51
121
|
function orgSafeUrl(url) {
|
|
52
122
|
return url.replaceAll("[", "%5B").replaceAll("]", "%5D");
|
|
53
123
|
}
|
|
124
|
+
// a url without a scheme is a relative path in markdown, but a bare org
|
|
125
|
+
// path is a fuzzy link (a heading search); org's file: type keeps it a
|
|
126
|
+
// file. A #anchor becomes org's search option, which finds a
|
|
127
|
+
// <<target>> or a headline of that name. `[[page]]` / `((uuid))` urls
|
|
128
|
+
// are dialect references (logseq), not paths
|
|
129
|
+
const SCHEME_RE = /^[a-z][a-z0-9+.-]*:/i;
|
|
130
|
+
function isRelativePath(url) {
|
|
131
|
+
return url !== "" && !SCHEME_RE.test(url) && !/^(#|\/\/|\[|\()/.test(url);
|
|
132
|
+
}
|
|
133
|
+
// md urls are percent-encoded, org paths are not; a malformed escape
|
|
134
|
+
// stays as written
|
|
135
|
+
function decodeUrlPart(part) {
|
|
136
|
+
try {
|
|
137
|
+
return decodeURIComponent(part);
|
|
138
|
+
}
|
|
139
|
+
catch {
|
|
140
|
+
return part;
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
function orgLinkTarget(url) {
|
|
144
|
+
if (!isRelativePath(url)) {
|
|
145
|
+
return { rawLink: orgSafeUrl(url), linkType: "url" };
|
|
146
|
+
}
|
|
147
|
+
const hash = url.indexOf("#");
|
|
148
|
+
const path = escapeOrgPath(decodeUrlPart(hash === -1 ? url : url.slice(0, hash)));
|
|
149
|
+
const search = hash === -1
|
|
150
|
+
? ""
|
|
151
|
+
: `::${escapeOrgPath(decodeUrlPart(url.slice(hash + 1)), true)}`;
|
|
152
|
+
return { rawLink: `file:${path}${search}`, linkType: "file" };
|
|
153
|
+
}
|
|
54
154
|
function transformMdastLink(ctx, linkNode) {
|
|
55
155
|
const [only] = linkNode.children;
|
|
56
156
|
// text equal to the url (autolinks) is no description; a plain
|
|
@@ -61,13 +161,13 @@ function transformMdastLink(ctx, linkNode) {
|
|
|
61
161
|
? []
|
|
62
162
|
: transformPhrasingChildren(ctx, linkNode.children);
|
|
63
163
|
// rawLink should just be the URL, uniorg-stringify adds the brackets
|
|
64
|
-
const
|
|
164
|
+
const { rawLink, linkType } = orgLinkTarget(linkNode.url);
|
|
65
165
|
return {
|
|
66
166
|
type: "link",
|
|
67
167
|
format: "bracket", // Assuming bracket format for Markdown links
|
|
68
|
-
linkType
|
|
69
|
-
rawLink
|
|
70
|
-
path:
|
|
168
|
+
linkType,
|
|
169
|
+
rawLink,
|
|
170
|
+
path: rawLink,
|
|
71
171
|
children: linkChildren
|
|
72
172
|
};
|
|
73
173
|
}
|
|
@@ -101,16 +201,51 @@ function transformMdastImage(ctx, node) {
|
|
|
101
201
|
if (node.title) {
|
|
102
202
|
warn(ctx, `dropped image title "${node.title}" (${node.url})`);
|
|
103
203
|
}
|
|
104
|
-
const
|
|
204
|
+
const { rawLink } = orgLinkTarget(node.url);
|
|
105
205
|
return {
|
|
106
206
|
type: "link",
|
|
107
207
|
format: "bracket",
|
|
108
208
|
linkType: "file",
|
|
109
|
-
rawLink
|
|
110
|
-
path:
|
|
209
|
+
rawLink,
|
|
210
|
+
path: rawLink,
|
|
111
211
|
children: node.alt ? [{ type: "text", value: node.alt }] : []
|
|
112
212
|
};
|
|
113
213
|
}
|
|
214
|
+
// org markup spans at most two lines; CommonMark renders a line ending
|
|
215
|
+
// inside a code span as a space, so nothing is lost. Org markup may not
|
|
216
|
+
// start or end with whitespace either: edge whitespace moves outside
|
|
217
|
+
// the markers, a one-space shift in the rendered output
|
|
218
|
+
function transformMdastInlineCode(ctx, value) {
|
|
219
|
+
const [, before = "", code = "", after = ""] = /^(\s*)(.*?)(\s*)$/s.exec(value.replace(/\r\n?|\n/g, " ")) ?? [];
|
|
220
|
+
if (!code) {
|
|
221
|
+
// org has no empty code markup
|
|
222
|
+
warn(ctx, "whitespace-only inline code kept as text");
|
|
223
|
+
return [{ type: "text", value: before }];
|
|
224
|
+
}
|
|
225
|
+
const type = codeType(code);
|
|
226
|
+
if (!type) {
|
|
227
|
+
warn(ctx, "inline code holding both ~ and = kept as text");
|
|
228
|
+
return [{ type: "text", value: before + code + after }];
|
|
229
|
+
}
|
|
230
|
+
return [
|
|
231
|
+
...(before ? [{ type: "text", value: before }] : []),
|
|
232
|
+
{ type, value: code },
|
|
233
|
+
...(after ? [{ type: "text", value: after }] : [])
|
|
234
|
+
];
|
|
235
|
+
}
|
|
236
|
+
// org ends code at the first `~` it may close on (`~a~ b~`); verbatim
|
|
237
|
+
// keeps such code whole, and comes back as md code too, unless a `=`
|
|
238
|
+
// ends it early in turn
|
|
239
|
+
function codeType(code) {
|
|
240
|
+
return ["code", "verbatim"].find(type => !code.includes(type === "code" ? "~" : "=") ||
|
|
241
|
+
readsWhole({ type, value: code }));
|
|
242
|
+
}
|
|
243
|
+
// whether org reads `node`, rendered, back as the same node
|
|
244
|
+
function readsWhole(node) {
|
|
245
|
+
const [paragraph] = orgParser.parse(renderInline(node)).children;
|
|
246
|
+
const [first, ...rest] = paragraph && "children" in paragraph ? paragraph.children : [];
|
|
247
|
+
return !rest.length && first?.type === node.type && first.value === node.value;
|
|
248
|
+
}
|
|
114
249
|
function transformMdastInlineMath(node) {
|
|
115
250
|
const value = node.value;
|
|
116
251
|
return {
|
|
@@ -144,8 +279,6 @@ function transformMdastPhrasingContentToUniorgObject(ctx, node) {
|
|
|
144
279
|
return transformMdastLinkReference(ctx, node);
|
|
145
280
|
case "imageReference":
|
|
146
281
|
return transformMdastImageReference(ctx, node);
|
|
147
|
-
case "inlineCode":
|
|
148
|
-
return { type: "code", value: node.value };
|
|
149
282
|
case "inlineMath":
|
|
150
283
|
return transformMdastInlineMath(node);
|
|
151
284
|
case "break":
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* md→org: a decoded file link path, or its search option, as org link
|
|
3
|
+
* text.
|
|
4
|
+
*/
|
|
5
|
+
export declare function escapeOrgPath(part: string, isSearch?: boolean): string;
|
|
6
|
+
/**
|
|
7
|
+
* org→md: the inverse of `escapeOrgPath`.
|
|
8
|
+
*/
|
|
9
|
+
export declare function unescapeOrgPath(part: string): string;
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
// an org bracket-link path cannot contain [ or ], and `::` starts its
|
|
2
|
+
// search option. morg percent-encodes those (uniorg does not decode org's
|
|
3
|
+
// own backslash escaping); a literal % in front of such an escape gets
|
|
4
|
+
// one more `25`, so the encoding stays reversible (`%5B` ↔ `%255B`)
|
|
5
|
+
const ESCAPE_RE = /%((?:25)*)(5B|5D|3A)/g;
|
|
6
|
+
const ESCAPED = { "5B": "[", "5D": "]", "3A": ":" };
|
|
7
|
+
/**
|
|
8
|
+
* md→org: a decoded file link path, or its search option, as org link
|
|
9
|
+
* text.
|
|
10
|
+
*/
|
|
11
|
+
export function escapeOrgPath(part, isSearch = false) {
|
|
12
|
+
const escaped = part
|
|
13
|
+
.replace(ESCAPE_RE, "%25$1$2")
|
|
14
|
+
.replaceAll("[", "%5B")
|
|
15
|
+
.replaceAll("]", "%5D");
|
|
16
|
+
// a colon next to another, or before the `::` separator
|
|
17
|
+
return isSearch ? escaped : escaped.replace(/:(?=:|$)/g, "%3A");
|
|
18
|
+
}
|
|
19
|
+
/**
|
|
20
|
+
* org→md: the inverse of `escapeOrgPath`.
|
|
21
|
+
*/
|
|
22
|
+
export function unescapeOrgPath(part) {
|
|
23
|
+
return part.replace(ESCAPE_RE, (_, more, code) => more ? `%${more.slice(2)}${code}` : (ESCAPED[code] ?? code));
|
|
24
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import type { Parent } from "unist";
|
|
2
|
+
export type Node = Parent["children"][number] & {
|
|
3
|
+
value?: string;
|
|
4
|
+
};
|
|
5
|
+
export declare const orgParser: import("unified").Processor<import("uniorg").OrgData, undefined, undefined, undefined, undefined>;
|
|
6
|
+
export declare const positionParser: import("unified").Processor<import("uniorg").OrgData, undefined, undefined, undefined, undefined>;
|
|
7
|
+
/**
|
|
8
|
+
* Parses `text` as an org document, or returns undefined where uniorg
|
|
9
|
+
* throws: it takes a line starting `_.` or `_)` for a bullet, then
|
|
10
|
+
* fails to read it (org has no such bullet; see underscoreBullets).
|
|
11
|
+
*/
|
|
12
|
+
export declare function tryParse(text: string, parser?: {
|
|
13
|
+
parse(text: string): unknown;
|
|
14
|
+
}): Parent | undefined;
|
|
15
|
+
export declare function isInline(node: Node | Parent): boolean;
|
|
16
|
+
/**
|
|
17
|
+
* How org renders an inline node within its line.
|
|
18
|
+
*/
|
|
19
|
+
export declare function renderInline(node: Node): string;
|
|
20
|
+
/**
|
|
21
|
+
* An inline node's delimiters around its children, as org renders them
|
|
22
|
+
* (`*` and `*`, `[[url][` and `]]`).
|
|
23
|
+
*/
|
|
24
|
+
export declare function delimiters(node: Node): [string, string];
|
|
25
|
+
/**
|
|
26
|
+
* The org rendering of each child; a block element only ends a line.
|
|
27
|
+
*/
|
|
28
|
+
export declare function renderChildren(children: Node[]): string[];
|
|
29
|
+
/**
|
|
30
|
+
* The index of the rendered child `offset` (into the joined rendering)
|
|
31
|
+
* lies in, and the offset within that child.
|
|
32
|
+
*/
|
|
33
|
+
export declare function locate(rendered: string[], offset: number): [number, number];
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
import { unified } from "unified";
|
|
2
|
+
import uniorgParse from "uniorg-parse";
|
|
3
|
+
import { uniorgStringify } from "uniorg-stringify";
|
|
4
|
+
// built once: constructing a processor per parse dominates the cost
|
|
5
|
+
export const orgParser = unified().use(uniorgParse).freeze();
|
|
6
|
+
export const positionParser = unified()
|
|
7
|
+
.use(uniorgParse, { trackPosition: true })
|
|
8
|
+
.freeze();
|
|
9
|
+
const stringifier = unified().use(uniorgStringify).freeze();
|
|
10
|
+
/**
|
|
11
|
+
* Parses `text` as an org document, or returns undefined where uniorg
|
|
12
|
+
* throws: it takes a line starting `_.` or `_)` for a bullet, then
|
|
13
|
+
* fails to read it (org has no such bullet; see underscoreBullets).
|
|
14
|
+
*/
|
|
15
|
+
export function tryParse(text, parser = orgParser) {
|
|
16
|
+
try {
|
|
17
|
+
return parser.parse(text);
|
|
18
|
+
}
|
|
19
|
+
catch {
|
|
20
|
+
return undefined;
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
// uniorg's inline node types; anything else is a block element (in a
|
|
24
|
+
// list item's flattened content: a nested list or code block)
|
|
25
|
+
const INLINE_TYPES = new Set([
|
|
26
|
+
"text",
|
|
27
|
+
"bold",
|
|
28
|
+
"italic",
|
|
29
|
+
"underline",
|
|
30
|
+
"strike-through",
|
|
31
|
+
"code",
|
|
32
|
+
"verbatim",
|
|
33
|
+
"link",
|
|
34
|
+
"footnote-reference",
|
|
35
|
+
"latex-fragment",
|
|
36
|
+
"entity",
|
|
37
|
+
"timestamp",
|
|
38
|
+
"subscript",
|
|
39
|
+
"superscript",
|
|
40
|
+
"export-snippet",
|
|
41
|
+
"statistics-cookie",
|
|
42
|
+
"citation",
|
|
43
|
+
"line-break"
|
|
44
|
+
]);
|
|
45
|
+
export function isInline(node) {
|
|
46
|
+
return INLINE_TYPES.has(node.type);
|
|
47
|
+
}
|
|
48
|
+
// ends the rendering, so the paragraph's own trailing newline and
|
|
49
|
+
// whitespace trimming stay out of it
|
|
50
|
+
const SENTINEL = "\u0000";
|
|
51
|
+
/**
|
|
52
|
+
* How org renders an inline node within its line.
|
|
53
|
+
*/
|
|
54
|
+
export function renderInline(node) {
|
|
55
|
+
if (node.type === "text") {
|
|
56
|
+
return node.value ?? "";
|
|
57
|
+
}
|
|
58
|
+
const rendered = String(stringifier.stringify({
|
|
59
|
+
type: "org-data",
|
|
60
|
+
children: [
|
|
61
|
+
{
|
|
62
|
+
type: "paragraph",
|
|
63
|
+
children: [node, { type: "text", value: SENTINEL }]
|
|
64
|
+
}
|
|
65
|
+
]
|
|
66
|
+
}));
|
|
67
|
+
return rendered.slice(0, rendered.lastIndexOf(SENTINEL));
|
|
68
|
+
}
|
|
69
|
+
// stands in for an inline node's content
|
|
70
|
+
const CONTENT = "\u0001";
|
|
71
|
+
/**
|
|
72
|
+
* An inline node's delimiters around its children, as org renders them
|
|
73
|
+
* (`*` and `*`, `[[url][` and `]]`).
|
|
74
|
+
*/
|
|
75
|
+
export function delimiters(node) {
|
|
76
|
+
const [open = "", close = ""] = renderInline({
|
|
77
|
+
...node,
|
|
78
|
+
children: [{ type: "text", value: CONTENT }]
|
|
79
|
+
}).split(CONTENT);
|
|
80
|
+
return [open, close];
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* The org rendering of each child; a block element only ends a line.
|
|
84
|
+
*/
|
|
85
|
+
export function renderChildren(children) {
|
|
86
|
+
return children.map(child => (isInline(child) ? renderInline(child) : "\n"));
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* The index of the rendered child `offset` (into the joined rendering)
|
|
90
|
+
* lies in, and the offset within that child.
|
|
91
|
+
*/
|
|
92
|
+
export function locate(rendered, offset) {
|
|
93
|
+
let start = 0;
|
|
94
|
+
for (const [index, part] of rendered.entries()) {
|
|
95
|
+
if (offset < start + part.length) {
|
|
96
|
+
return [index, offset - start];
|
|
97
|
+
}
|
|
98
|
+
start += part.length;
|
|
99
|
+
}
|
|
100
|
+
return [-1, 0];
|
|
101
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
import type { Parent } from "unist";
|
|
2
|
+
/**
|
|
3
|
+
* md→org: a `|` in a table cell's text (md `\|`) would end the cell; org
|
|
4
|
+
* has no escaped `|`, but renders the `\vert` entity as one, `{}` ending
|
|
5
|
+
* its name before any letter. Runs after the preset, whose wikilink
|
|
6
|
+
* aliases (`[[Page|alias]]`) are no text.
|
|
7
|
+
*/
|
|
8
|
+
export declare function escapeTablePipes(tree: Parent): void;
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import { visit } from "unist-util-visit";
|
|
2
|
+
/**
|
|
3
|
+
* md→org: a `|` in a table cell's text (md `\|`) would end the cell; org
|
|
4
|
+
* has no escaped `|`, but renders the `\vert` entity as one, `{}` ending
|
|
5
|
+
* its name before any letter. Runs after the preset, whose wikilink
|
|
6
|
+
* aliases (`[[Page|alias]]`) are no text.
|
|
7
|
+
*/
|
|
8
|
+
export function escapeTablePipes(tree) {
|
|
9
|
+
visit(tree, "table-cell", (cell) => {
|
|
10
|
+
visit(cell, "text", (node) => {
|
|
11
|
+
node.value = node.value?.replaceAll("|", "\\vert{}");
|
|
12
|
+
});
|
|
13
|
+
});
|
|
14
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { Parent } from "unist";
|
|
2
|
+
/**
|
|
3
|
+
* org→md: guards the lines uniorg misreads before parsing.
|
|
4
|
+
*/
|
|
5
|
+
export declare function guardUnderscoreBullets(org: string): string;
|
|
6
|
+
/**
|
|
7
|
+
* org→md: drops the guards again, also where uniorg reads no text (a
|
|
8
|
+
* src block's code). A zero-width space the author put there goes too.
|
|
9
|
+
*/
|
|
10
|
+
export declare function dropUnderscoreBulletGuards(tree: Parent): void;
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { visit } from "unist-util-visit";
|
|
2
|
+
import { ZERO_WIDTH_SPACE } from "./markupBoundary.js";
|
|
3
|
+
// ====================================================================
|
|
4
|
+
// WORKAROUND for a bug in uniorg-parse 3.2.2 (upstream issue: not filed
|
|
5
|
+
// yet, see TODO.md). Drop this module once a fixed version is in; the
|
|
6
|
+
// canary in tests/uniorgWorkarounds.spec.ts fails then.
|
|
7
|
+
// ====================================================================
|
|
8
|
+
// Its `listItemRe` takes a line starting `_.` or `_)` for a list item
|
|
9
|
+
// (`\w` includes `_`), but its `fullListItemRe`, like org itself, has
|
|
10
|
+
// no such bullet. `parseListStructure` then throws (`match error`), or,
|
|
11
|
+
// when a real bullet line follows, reads that one instead and silently
|
|
12
|
+
// drops the line. Org reads the line as text, and so does uniorg with a
|
|
13
|
+
// zero-width space in front (morg's line-start escape, see lineSyntax)
|
|
14
|
+
const UNDERSCORE_BULLET_RE = /^([ \t]*)(?=_[.)](?:[ \t]|$))/gm;
|
|
15
|
+
/**
|
|
16
|
+
* org→md: guards the lines uniorg misreads before parsing.
|
|
17
|
+
*/
|
|
18
|
+
export function guardUnderscoreBullets(org) {
|
|
19
|
+
return org.replace(UNDERSCORE_BULLET_RE, `$1${ZERO_WIDTH_SPACE}`);
|
|
20
|
+
}
|
|
21
|
+
const GUARD_RE = new RegExp(`(^|\\n)([ \\t]*)${ZERO_WIDTH_SPACE}(?=_[.)](?:[ \\t\\n]|$))`, "g");
|
|
22
|
+
/**
|
|
23
|
+
* org→md: drops the guards again, also where uniorg reads no text (a
|
|
24
|
+
* src block's code). A zero-width space the author put there goes too.
|
|
25
|
+
*/
|
|
26
|
+
export function dropUnderscoreBulletGuards(tree) {
|
|
27
|
+
visit(tree, (node) => {
|
|
28
|
+
if (typeof node.value === "string") {
|
|
29
|
+
node.value = node.value.replace(GUARD_RE, "$1$2");
|
|
30
|
+
}
|
|
31
|
+
});
|
|
32
|
+
}
|
|
@@ -191,7 +191,7 @@ function transformParagraph(ctx, node) {
|
|
|
191
191
|
const children = transformUniorgObjects(ctx, node.children);
|
|
192
192
|
// md gives leading whitespace structural meaning (list
|
|
193
193
|
// continuation, code); collapse per-line indentation inside
|
|
194
|
-
// paragraphs
|
|
194
|
+
// paragraphs, insignificant in org and in rendered md alike
|
|
195
195
|
children.forEach((child, index) => {
|
|
196
196
|
if (child.type === "text") {
|
|
197
197
|
child.value = child.value.replace(/\n[ \t]+/g, "\n");
|
|
@@ -67,11 +67,22 @@ function transformUniorgList(ctx, node) {
|
|
|
67
67
|
};
|
|
68
68
|
});
|
|
69
69
|
}
|
|
70
|
+
// uniorg keeps a list item's indentation in its code blocks' values;
|
|
71
|
+
// md indents them itself. Only that much is dropped, as uniorg-stringify
|
|
72
|
+
// does, so the code keeps its own
|
|
73
|
+
function outdent(value, level) {
|
|
74
|
+
return value.replace(new RegExp(`^ {0,${level}}`, "gm"), "");
|
|
75
|
+
}
|
|
70
76
|
function transformUniorgListItem(ctx, item) {
|
|
71
77
|
// md has no descriptive lists, so keep the ` :: ` syntax literally in the
|
|
72
|
-
// item text
|
|
78
|
+
// item text; the return trip re-parses it as a descriptive list
|
|
73
79
|
const tag = listItemTag(item);
|
|
74
80
|
const children = transformNodes(ctx, (item.children || []).filter(child => child !== tag));
|
|
81
|
+
for (const child of children) {
|
|
82
|
+
if (child.type === "code") {
|
|
83
|
+
child.value = outdent(child.value, item.indent + item.bullet.length);
|
|
84
|
+
}
|
|
85
|
+
}
|
|
75
86
|
if (tag) {
|
|
76
87
|
const term = {
|
|
77
88
|
type: "text",
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { toString as orgastToString } from "orgast-util-to-string";
|
|
2
2
|
import { htmlEnabled, orgNodeToText, warn } from "./shared.js";
|
|
3
3
|
import { transformFootnoteReference } from "./footnotes.js";
|
|
4
|
+
import { unescapeOrgPath } from "../orgPath.js";
|
|
4
5
|
const IMAGE_EXTENSION_RE = /\.(png|jpe?g|gif|svg|webp|avif|bmp|ico)$/i;
|
|
5
6
|
export function transformUniorgObjects(ctx, children) {
|
|
6
7
|
return (children || [])
|
|
@@ -74,12 +75,43 @@ function transformUniorgObjectToMdastPhrasingContent(ctx, node) {
|
|
|
74
75
|
return null;
|
|
75
76
|
}
|
|
76
77
|
}
|
|
78
|
+
// the inverse of md→org's decoding: % and # would be read as an escape
|
|
79
|
+
// or the anchor, and a markdown url cannot hold a bare space. Org path
|
|
80
|
+
// escapes (`[`, `]`, `:`) are undone first rather than escaped twice
|
|
81
|
+
function encodeUrlPart(part) {
|
|
82
|
+
return unescapeOrgPath(part)
|
|
83
|
+
.replaceAll("%", "%25")
|
|
84
|
+
.replaceAll("#", "%23")
|
|
85
|
+
.replaceAll(" ", "%20");
|
|
86
|
+
}
|
|
87
|
+
// a path starting like `a:` would read as a url scheme in markdown
|
|
88
|
+
function encodeUrlPath(path) {
|
|
89
|
+
const encoded = encodeUrlPart(path);
|
|
90
|
+
return /^[a-z][a-z0-9+.-]*:/i.test(encoded)
|
|
91
|
+
? encoded.replaceAll(":", "%3A")
|
|
92
|
+
: encoded;
|
|
93
|
+
}
|
|
94
|
+
// an org file: link (or a ./ path) is a relative markdown link, the
|
|
95
|
+
// search option becoming the #anchor
|
|
96
|
+
function markdownUrl(node) {
|
|
97
|
+
if (node.linkType !== "file") {
|
|
98
|
+
return node.rawLink;
|
|
99
|
+
}
|
|
100
|
+
const target = node.rawLink.replace(/^file:/, "");
|
|
101
|
+
const search = target.indexOf("::");
|
|
102
|
+
return search === -1
|
|
103
|
+
? encodeUrlPath(target)
|
|
104
|
+
: `${encodeUrlPath(target.slice(0, search))}#${encodeUrlPart(target.slice(search + 2))}`;
|
|
105
|
+
}
|
|
77
106
|
function transformUniorgLink(ctx, node) {
|
|
78
107
|
const descriptionText = orgastToString(node);
|
|
108
|
+
const url = markdownUrl(node);
|
|
79
109
|
// org has no dedicated image syntax; the common convention is a
|
|
80
|
-
// link to an image file, so map those to markdown images
|
|
81
|
-
|
|
82
|
-
|
|
110
|
+
// link to an image file, so map those to markdown images; a file
|
|
111
|
+
// link's ::search option is no part of the file name
|
|
112
|
+
const path = node.linkType === "file" ? node.rawLink.replace(/::.*$/s, "") : node.rawLink;
|
|
113
|
+
if (IMAGE_EXTENSION_RE.test(path)) {
|
|
114
|
+
return { type: "image", url, alt: descriptionText };
|
|
83
115
|
}
|
|
84
116
|
// a description equal to the url (a common Logseq pattern) is no
|
|
85
117
|
// description: text === url makes remark-stringify emit an
|
|
@@ -100,8 +132,8 @@ function transformUniorgLink(ctx, node) {
|
|
|
100
132
|
withoutMarkers(serializedDescription) === withoutMarkers(node.rawLink)) {
|
|
101
133
|
return {
|
|
102
134
|
type: "link",
|
|
103
|
-
url
|
|
104
|
-
children: [{ type: "text", value:
|
|
135
|
+
url,
|
|
136
|
+
children: [{ type: "text", value: url }]
|
|
105
137
|
};
|
|
106
138
|
}
|
|
107
139
|
const children = transformUniorgObjects(ctx, node.children);
|
|
@@ -110,7 +142,7 @@ function transformUniorgLink(ctx, node) {
|
|
|
110
142
|
const flattened = children.some(child => child.type === "link" || child.type === "image")
|
|
111
143
|
? [{ type: "text", value: descriptionText }]
|
|
112
144
|
: children;
|
|
113
|
-
return { type: "link", url
|
|
145
|
+
return { type: "link", url, children: flattened };
|
|
114
146
|
}
|
|
115
147
|
function transformScriptMarkup(ctx, node) {
|
|
116
148
|
// markdown has no equivalents; with useHtml render as raw html
|
|
@@ -16,12 +16,18 @@ function isAlignmentCookieRow(row) {
|
|
|
16
16
|
cells.some(cell => cellText(cell) !== ""));
|
|
17
17
|
}
|
|
18
18
|
// org cell content keeps the aligning whitespace padding; markdown
|
|
19
|
-
// cells are re-padded by the stringifier
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
19
|
+
// cells are re-padded by the stringifier. Only the cell's edges are
|
|
20
|
+
// padding: a space between text and markup is content
|
|
21
|
+
function trimCellPadding(children) {
|
|
22
|
+
const first = children[0];
|
|
23
|
+
if (first?.type === "text") {
|
|
24
|
+
first.value = first.value.trimStart();
|
|
23
25
|
}
|
|
24
|
-
|
|
26
|
+
const last = children.at(-1);
|
|
27
|
+
if (last?.type === "text") {
|
|
28
|
+
last.value = last.value.trimEnd();
|
|
29
|
+
}
|
|
30
|
+
return children;
|
|
25
31
|
}
|
|
26
32
|
export function transformTable(ctx, node) {
|
|
27
33
|
if (node.tableType === "table.el") {
|
|
@@ -59,7 +65,7 @@ export function transformTable(ctx, node) {
|
|
|
59
65
|
type: "tableRow",
|
|
60
66
|
children: (row.children || []).map(cell => ({
|
|
61
67
|
type: "tableCell",
|
|
62
|
-
children: transformUniorgObjects(ctx, cell.children)
|
|
68
|
+
children: trimCellPadding(transformUniorgObjects(ctx, cell.children))
|
|
63
69
|
}))
|
|
64
70
|
}))
|
|
65
71
|
};
|
package/dist/fileNames.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Splits a file name into the part a derived name keeps and the extension
|
|
3
3
|
* that carries its format. A dotless name and a dotfile both have no
|
|
4
|
-
* extension
|
|
4
|
+
* extension: `org` is a file called org, and `.org` is a hidden file whose
|
|
5
5
|
* name happens to start with a dot.
|
|
6
6
|
*
|
|
7
7
|
* The extension comes from the last path segment, since the CLI is handed
|
package/dist/fileNames.js
CHANGED
|
@@ -6,7 +6,7 @@ const DOCUMENT_FORMATS = new Map([
|
|
|
6
6
|
/**
|
|
7
7
|
* Splits a file name into the part a derived name keeps and the extension
|
|
8
8
|
* that carries its format. A dotless name and a dotfile both have no
|
|
9
|
-
* extension
|
|
9
|
+
* extension: `org` is a file called org, and `.org` is a hidden file whose
|
|
10
10
|
* name happens to start with a dot.
|
|
11
11
|
*
|
|
12
12
|
* The extension comes from the last path segment, since the CLI is handed
|