@remigius42/morg 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +36 -30
  2. package/dist/cli/args.d.ts +2 -0
  3. package/dist/cli/args.js +22 -60
  4. package/dist/cli/error.js +1 -1
  5. package/dist/cli/flags.d.ts +27 -0
  6. package/dist/cli/flags.js +127 -0
  7. package/dist/cli/formats.js +18 -4
  8. package/dist/cli/help.d.ts +1 -0
  9. package/dist/cli/help.js +28 -0
  10. package/dist/cli.js +24 -1
  11. package/dist/conversionOptions.d.ts +2 -2
  12. package/dist/conversionOptions.js +2 -2
  13. package/dist/core/bracedScripts.d.ts +12 -0
  14. package/dist/core/bracedScripts.js +159 -0
  15. package/dist/core/footnoteReferences.d.ts +10 -0
  16. package/dist/core/footnoteReferences.js +24 -0
  17. package/dist/core/lineSyntax.d.ts +12 -0
  18. package/dist/core/lineSyntax.js +206 -0
  19. package/dist/core/markupBoundary.d.ts +12 -0
  20. package/dist/core/markupBoundary.js +195 -0
  21. package/dist/core/mdastToUniorg/blocks.js +1 -1
  22. package/dist/core/mdastToUniorg/index.js +10 -2
  23. package/dist/core/mdastToUniorg/lists.d.ts +1 -1
  24. package/dist/core/mdastToUniorg/lists.js +21 -2
  25. package/dist/core/mdastToUniorg/phrasing.js +145 -12
  26. package/dist/core/orgPath.d.ts +9 -0
  27. package/dist/core/orgPath.js +24 -0
  28. package/dist/core/render.d.ts +33 -0
  29. package/dist/core/render.js +101 -0
  30. package/dist/core/tablePipes.d.ts +8 -0
  31. package/dist/core/tablePipes.js +14 -0
  32. package/dist/core/underscoreBullets.d.ts +10 -0
  33. package/dist/core/underscoreBullets.js +32 -0
  34. package/dist/core/uniorgToMdast/elements.js +1 -1
  35. package/dist/core/uniorgToMdast/lists.js +12 -1
  36. package/dist/core/uniorgToMdast/objects.js +38 -6
  37. package/dist/core/uniorgToMdast/tables.js +12 -6
  38. package/dist/fileNames.d.ts +1 -1
  39. package/dist/fileNames.js +1 -1
  40. package/dist/markdownToOrg.js +19 -1
  41. package/dist/normalize.d.ts +2 -2
  42. package/dist/normalize.js +1 -1
  43. package/dist/options.d.ts +3 -3
  44. package/dist/orgToMarkdown.js +29 -5
  45. package/dist/presets/logseq.js +1 -1
  46. package/package.json +4 -1
@@ -1,3 +1,4 @@
1
+ import { warn } from "./context.js";
1
2
  import { transformPhrasingChildren } from "./phrasing.js";
2
3
  // circular import with index.js is fine in ESM: both sides only export
3
4
  // hoisted function declarations called after module initialization
@@ -17,19 +18,37 @@ export function transformMdastList(ctx, listNode, indent) {
17
18
  contentsEnd: 0
18
19
  };
19
20
  }
21
+ // uniorg-stringify re-indents a list item's block by stripping up to
22
+ // the item's indentation from each line first: a code block's value
23
+ // has to carry that indentation (as uniorg's parser reads it), or its
24
+ // own indentation shrinks
25
+ function indentCode(node, level) {
26
+ const block = node;
27
+ if ((block?.type === "src-block" || block?.type === "example-block") &&
28
+ block.value !== undefined) {
29
+ block.value = block.value.replace(/^(?=.)/gm, " ".repeat(level));
30
+ }
31
+ return node;
32
+ }
20
33
  function transformMdastListItem(ctx, item, indent, bullet) {
21
34
  const children = item.children
22
35
  .flatMap((child) => {
23
36
  if (child.type === "list") {
24
37
  return [transformMdastList(ctx, child, indent + bullet.length)];
25
38
  }
26
- if (child.type === "paragraph") {
39
+ if (child.type === "heading") {
40
+ // org headlines cannot live inside a list item; the text stays
41
+ warn(ctx, "heading inside a list item became text");
42
+ }
43
+ if (child.type === "paragraph" || child.type === "heading") {
27
44
  return [
28
45
  ...transformPhrasingChildren(ctx, child.children),
29
46
  { type: "text", value: "\n" }
30
47
  ];
31
48
  }
32
- return [transformMdastNodeToUniorgNode(ctx, child)];
49
+ return [
50
+ indentCode(transformMdastNodeToUniorgNode(ctx, child), indent + bullet.length)
51
+ ];
33
52
  })
34
53
  .filter(Boolean);
35
54
  return {
@@ -1,4 +1,7 @@
1
1
  import { mdismEnabled, warn } from "./context.js";
2
+ import { visit } from "unist-util-visit";
3
+ import { escapeOrgPath } from "../orgPath.js";
4
+ import { orgParser, renderInline } from "../render.js";
2
5
  // html tags morg itself emits under useHtml; with interpretHtml a bare
3
6
  // open/close pair becomes the corresponding native org object
4
7
  const INLINE_HTML_ORG_TYPES = {
@@ -30,20 +33,87 @@ export function transformPhrasingChildren(ctx, children) {
30
33
  ? matchInlineHtmlPair(children, i)
31
34
  : null;
32
35
  if (pair) {
33
- result.push({
36
+ result.push(...hoistEdgeWhitespace(joinLines({
34
37
  type: pair.orgType,
35
38
  children: transformPhrasingChildren(ctx, children.slice(i + 1, pair.end))
36
- });
39
+ })));
37
40
  i = pair.end;
38
41
  continue;
39
42
  }
43
+ if (node.type === "inlineCode") {
44
+ result.push(...transformMdastInlineCode(ctx, node.value));
45
+ continue;
46
+ }
40
47
  const transformed = transformMdastPhrasingContentToUniorgObject(ctx, node);
41
48
  if (transformed) {
42
- result.push(transformed);
49
+ result.push(...hoistEdgeWhitespace(joinLines(transformed)));
43
50
  }
44
51
  }
45
52
  return result;
46
53
  }
54
+ const CONTAINER_MARKUP = new Set([
55
+ "bold",
56
+ "italic",
57
+ "strike-through",
58
+ "underline"
59
+ ]);
60
+ // org markup spans at most two lines; beyond that its line endings
61
+ // become spaces, as CommonMark renders them anyway (see
62
+ // transformMdastInlineCode)
63
+ function joinLines(node) {
64
+ if (!CONTAINER_MARKUP.has(node.type)) {
65
+ return node;
66
+ }
67
+ const texts = [];
68
+ visit(node, "text", (text) => {
69
+ texts.push(text);
70
+ });
71
+ const lineEndings = texts
72
+ .map(text => text.value)
73
+ .join("")
74
+ .split("\n").length - 1;
75
+ if (lineEndings > 1) {
76
+ for (const text of texts) {
77
+ text.value = text.value.replaceAll("\n", " ");
78
+ }
79
+ }
80
+ return node;
81
+ }
82
+ // strips the whitespace at the edges of `children`, returning it
83
+ function takeEdgeWhitespace(children) {
84
+ const first = children[0];
85
+ const last = children.at(-1);
86
+ let before = "";
87
+ let after = "";
88
+ if (first?.type === "text") {
89
+ before = /^\s*/.exec(first.value)?.[0] ?? "";
90
+ first.value = first.value.slice(before.length);
91
+ }
92
+ if (last?.type === "text") {
93
+ after = /\s*$/.exec(last.value)?.[0] ?? "";
94
+ last.value = last.value.slice(0, last.value.length - after.length);
95
+ }
96
+ return { before, after };
97
+ }
98
+ // org markup may not start or end with whitespace, which a code span's
99
+ // moved-out edge whitespace (see transformMdastInlineCode) can leave
100
+ // inside it; it moves out further, to the markup's siblings
101
+ function hoistEdgeWhitespace(node) {
102
+ if (!CONTAINER_MARKUP.has(node.type) || !("children" in node)) {
103
+ return [node];
104
+ }
105
+ const { before, after } = takeEdgeWhitespace(node.children);
106
+ if (!before && !after) {
107
+ return [node];
108
+ }
109
+ const content = node.children.filter(child => child.type !== "text" || child.value !== "");
110
+ const text = (value) => value ? [{ type: "text", value }] : [];
111
+ return [
112
+ ...text(before),
113
+ ...(content.length ? [{ ...node, children: content }] : []),
114
+ ...text(after)
115
+ ];
116
+ }
47
117
  // an org bracket-link path cannot contain [ or ]; percent-encoding keeps
48
118
  // the url equivalent and, unlike org's backslash escaping, survives being
49
119
  // parsed back (uniorg does not decode `\[`). A silent normalization, not
@@ -51,6 +121,36 @@ export function transformPhrasingChildren(ctx, children) {
51
121
  function orgSafeUrl(url) {
52
122
  return url.replaceAll("[", "%5B").replaceAll("]", "%5D");
53
123
  }
124
+ // a url without a scheme is a relative path in markdown, but a bare org
125
+ // path is a fuzzy link (a heading search); org's file: type keeps it a
126
+ // file. A #anchor becomes org's search option, which finds a
127
+ // <<target>> or a headline of that name. `[[page]]` / `((uuid))` urls
128
+ // are dialect references (logseq), not paths
129
+ const SCHEME_RE = /^[a-z][a-z0-9+.-]*:/i;
130
+ function isRelativePath(url) {
131
+ return url !== "" && !SCHEME_RE.test(url) && !/^(#|\/\/|\[|\()/.test(url);
132
+ }
133
+ // md urls are percent-encoded, org paths are not; a malformed escape
134
+ // stays as written
135
+ function decodeUrlPart(part) {
136
+ try {
137
+ return decodeURIComponent(part);
138
+ }
139
+ catch {
140
+ return part;
141
+ }
142
+ }
143
+ function orgLinkTarget(url) {
144
+ if (!isRelativePath(url)) {
145
+ return { rawLink: orgSafeUrl(url), linkType: "url" };
146
+ }
147
+ const hash = url.indexOf("#");
148
+ const path = escapeOrgPath(decodeUrlPart(hash === -1 ? url : url.slice(0, hash)));
149
+ const search = hash === -1
150
+ ? ""
151
+ : `::${escapeOrgPath(decodeUrlPart(url.slice(hash + 1)), true)}`;
152
+ return { rawLink: `file:${path}${search}`, linkType: "file" };
153
+ }
54
154
  function transformMdastLink(ctx, linkNode) {
55
155
  const [only] = linkNode.children;
56
156
  // text equal to the url (autolinks) is no description; a plain
@@ -61,13 +161,13 @@ function transformMdastLink(ctx, linkNode) {
61
161
  ? []
62
162
  : transformPhrasingChildren(ctx, linkNode.children);
63
163
  // rawLink should just be the URL, uniorg-stringify adds the brackets
64
- const url = orgSafeUrl(linkNode.url);
164
+ const { rawLink, linkType } = orgLinkTarget(linkNode.url);
65
165
  return {
66
166
  type: "link",
67
167
  format: "bracket", // Assuming bracket format for Markdown links
68
- linkType: "url",
69
- rawLink: url,
70
- path: url,
168
+ linkType,
169
+ rawLink,
170
+ path: rawLink,
71
171
  children: linkChildren
72
172
  };
73
173
  }
@@ -101,16 +201,51 @@ function transformMdastImage(ctx, node) {
101
201
  if (node.title) {
102
202
  warn(ctx, `dropped image title "${node.title}" (${node.url})`);
103
203
  }
104
- const url = orgSafeUrl(node.url);
204
+ const { rawLink } = orgLinkTarget(node.url);
105
205
  return {
106
206
  type: "link",
107
207
  format: "bracket",
108
208
  linkType: "file",
109
- rawLink: url,
110
- path: url,
209
+ rawLink,
210
+ path: rawLink,
111
211
  children: node.alt ? [{ type: "text", value: node.alt }] : []
112
212
  };
113
213
  }
214
+ // org markup spans at most two lines; CommonMark renders a line ending
215
+ // inside a code span as a space, so nothing is lost. Org markup may not
216
+ // start or end with whitespace either: edge whitespace moves outside
217
+ // the markers, a one-space shift in the rendered output
218
+ function transformMdastInlineCode(ctx, value) {
219
+ const [, before = "", code = "", after = ""] = /^(\s*)(.*?)(\s*)$/s.exec(value.replace(/\r\n?|\n/g, " ")) ?? [];
220
+ if (!code) {
221
+ // org has no empty code markup
222
+ warn(ctx, "whitespace-only inline code kept as text");
223
+ return [{ type: "text", value: before }];
224
+ }
225
+ const type = codeType(code);
226
+ if (!type) {
227
+ warn(ctx, "inline code holding both ~ and = kept as text");
228
+ return [{ type: "text", value: before + code + after }];
229
+ }
230
+ return [
231
+ ...(before ? [{ type: "text", value: before }] : []),
232
+ { type, value: code },
233
+ ...(after ? [{ type: "text", value: after }] : [])
234
+ ];
235
+ }
236
+ // org ends code at the first `~` it may close on (`~a~ b~`); verbatim
237
+ // keeps such code whole, and comes back as md code too, unless a `=`
238
+ // ends it early in turn
239
+ function codeType(code) {
240
+ return ["code", "verbatim"].find(type => !code.includes(type === "code" ? "~" : "=") ||
241
+ readsWhole({ type, value: code }));
242
+ }
243
+ // whether org reads `node`, rendered, back as the same node
244
+ function readsWhole(node) {
245
+ const [paragraph] = orgParser.parse(renderInline(node)).children;
246
+ const [first, ...rest] = paragraph && "children" in paragraph ? paragraph.children : [];
247
+ return !rest.length && first?.type === node.type && first.value === node.value;
248
+ }
114
249
  function transformMdastInlineMath(node) {
115
250
  const value = node.value;
116
251
  return {
@@ -144,8 +279,6 @@ function transformMdastPhrasingContentToUniorgObject(ctx, node) {
144
279
  return transformMdastLinkReference(ctx, node);
145
280
  case "imageReference":
146
281
  return transformMdastImageReference(ctx, node);
147
- case "inlineCode":
148
- return { type: "code", value: node.value };
149
282
  case "inlineMath":
150
283
  return transformMdastInlineMath(node);
151
284
  case "break":
@@ -0,0 +1,9 @@
1
+ /**
2
+ * md→org: a decoded file link path, or its search option, as org link
3
+ * text.
4
+ */
5
+ export declare function escapeOrgPath(part: string, isSearch?: boolean): string;
6
+ /**
7
+ * org→md: the inverse of `escapeOrgPath`.
8
+ */
9
+ export declare function unescapeOrgPath(part: string): string;
@@ -0,0 +1,24 @@
1
+ // an org bracket-link path cannot contain [ or ], and `::` starts its
2
+ // search option. morg percent-encodes those (uniorg does not decode org's
3
+ // own backslash escaping); a literal % in front of such an escape gets
4
+ // one more `25`, so the encoding stays reversible (`%5B` ↔ `%255B`)
5
+ const ESCAPE_RE = /%((?:25)*)(5B|5D|3A)/g;
6
+ const ESCAPED = { "5B": "[", "5D": "]", "3A": ":" };
7
+ /**
8
+ * md→org: a decoded file link path, or its search option, as org link
9
+ * text.
10
+ */
11
+ export function escapeOrgPath(part, isSearch = false) {
12
+ const escaped = part
13
+ .replace(ESCAPE_RE, "%25$1$2")
14
+ .replaceAll("[", "%5B")
15
+ .replaceAll("]", "%5D");
16
+ // a colon next to another, or before the `::` separator
17
+ return isSearch ? escaped : escaped.replace(/:(?=:|$)/g, "%3A");
18
+ }
19
+ /**
20
+ * org→md: the inverse of `escapeOrgPath`.
21
+ */
22
+ export function unescapeOrgPath(part) {
23
+ return part.replace(ESCAPE_RE, (_, more, code) => more ? `%${more.slice(2)}${code}` : (ESCAPED[code] ?? code));
24
+ }
@@ -0,0 +1,33 @@
1
+ import type { Parent } from "unist";
2
+ export type Node = Parent["children"][number] & {
3
+ value?: string;
4
+ };
5
+ export declare const orgParser: import("unified").Processor<import("uniorg").OrgData, undefined, undefined, undefined, undefined>;
6
+ export declare const positionParser: import("unified").Processor<import("uniorg").OrgData, undefined, undefined, undefined, undefined>;
7
+ /**
8
+ * Parses `text` as an org document, or returns undefined where uniorg
9
+ * throws: it takes a line starting `_.` or `_)` for a bullet, then
10
+ * fails to read it (org has no such bullet; see underscoreBullets).
11
+ */
12
+ export declare function tryParse(text: string, parser?: {
13
+ parse(text: string): unknown;
14
+ }): Parent | undefined;
15
+ export declare function isInline(node: Node | Parent): boolean;
16
+ /**
17
+ * How org renders an inline node within its line.
18
+ */
19
+ export declare function renderInline(node: Node): string;
20
+ /**
21
+ * An inline node's delimiters around its children, as org renders them
22
+ * (`*` and `*`, `[[url][` and `]]`).
23
+ */
24
+ export declare function delimiters(node: Node): [string, string];
25
+ /**
26
+ * The org rendering of each child; a block element only ends a line.
27
+ */
28
+ export declare function renderChildren(children: Node[]): string[];
29
+ /**
30
+ * The index of the rendered child `offset` (into the joined rendering)
31
+ * lies in, and the offset within that child.
32
+ */
33
+ export declare function locate(rendered: string[], offset: number): [number, number];
@@ -0,0 +1,101 @@
1
+ import { unified } from "unified";
2
+ import uniorgParse from "uniorg-parse";
3
+ import { uniorgStringify } from "uniorg-stringify";
4
+ // built once: constructing a processor per parse dominates the cost
5
+ export const orgParser = unified().use(uniorgParse).freeze();
6
+ export const positionParser = unified()
7
+ .use(uniorgParse, { trackPosition: true })
8
+ .freeze();
9
+ const stringifier = unified().use(uniorgStringify).freeze();
10
+ /**
11
+ * Parses `text` as an org document, or returns undefined where uniorg
12
+ * throws: it takes a line starting `_.` or `_)` for a bullet, then
13
+ * fails to read it (org has no such bullet; see underscoreBullets).
14
+ */
15
+ export function tryParse(text, parser = orgParser) {
16
+ try {
17
+ return parser.parse(text);
18
+ }
19
+ catch {
20
+ return undefined;
21
+ }
22
+ }
23
+ // uniorg's inline node types; anything else is a block element (in a
24
+ // list item's flattened content: a nested list or code block)
25
+ const INLINE_TYPES = new Set([
26
+ "text",
27
+ "bold",
28
+ "italic",
29
+ "underline",
30
+ "strike-through",
31
+ "code",
32
+ "verbatim",
33
+ "link",
34
+ "footnote-reference",
35
+ "latex-fragment",
36
+ "entity",
37
+ "timestamp",
38
+ "subscript",
39
+ "superscript",
40
+ "export-snippet",
41
+ "statistics-cookie",
42
+ "citation",
43
+ "line-break"
44
+ ]);
45
+ export function isInline(node) {
46
+ return INLINE_TYPES.has(node.type);
47
+ }
48
+ // ends the rendering, so the paragraph's own trailing newline and
49
+ // whitespace trimming stay out of it
50
+ const SENTINEL = "\u0000";
51
+ /**
52
+ * How org renders an inline node within its line.
53
+ */
54
+ export function renderInline(node) {
55
+ if (node.type === "text") {
56
+ return node.value ?? "";
57
+ }
58
+ const rendered = String(stringifier.stringify({
59
+ type: "org-data",
60
+ children: [
61
+ {
62
+ type: "paragraph",
63
+ children: [node, { type: "text", value: SENTINEL }]
64
+ }
65
+ ]
66
+ }));
67
+ return rendered.slice(0, rendered.lastIndexOf(SENTINEL));
68
+ }
69
+ // stands in for an inline node's content
70
+ const CONTENT = "\u0001";
71
+ /**
72
+ * An inline node's delimiters around its children, as org renders them
73
+ * (`*` and `*`, `[[url][` and `]]`).
74
+ */
75
+ export function delimiters(node) {
76
+ const [open = "", close = ""] = renderInline({
77
+ ...node,
78
+ children: [{ type: "text", value: CONTENT }]
79
+ }).split(CONTENT);
80
+ return [open, close];
81
+ }
82
+ /**
83
+ * The org rendering of each child; a block element only ends a line.
84
+ */
85
+ export function renderChildren(children) {
86
+ return children.map(child => (isInline(child) ? renderInline(child) : "\n"));
87
+ }
88
+ /**
89
+ * The index of the rendered child `offset` (into the joined rendering)
90
+ * lies in, and the offset within that child.
91
+ */
92
+ export function locate(rendered, offset) {
93
+ let start = 0;
94
+ for (const [index, part] of rendered.entries()) {
95
+ if (offset < start + part.length) {
96
+ return [index, offset - start];
97
+ }
98
+ start += part.length;
99
+ }
100
+ return [-1, 0];
101
+ }
@@ -0,0 +1,8 @@
1
+ import type { Parent } from "unist";
2
+ /**
3
+ * md→org: a `|` in a table cell's text (md `\|`) would end the cell; org
4
+ * has no escaped `|`, but renders the `\vert` entity as one, `{}` ending
5
+ * its name before any letter. Runs after the preset, whose wikilink
6
+ * aliases (`[[Page|alias]]`) are no text.
7
+ */
8
+ export declare function escapeTablePipes(tree: Parent): void;
@@ -0,0 +1,14 @@
1
+ import { visit } from "unist-util-visit";
2
+ /**
3
+ * md→org: a `|` in a table cell's text (md `\|`) would end the cell; org
4
+ * has no escaped `|`, but renders the `\vert` entity as one, `{}` ending
5
+ * its name before any letter. Runs after the preset, whose wikilink
6
+ * aliases (`[[Page|alias]]`) are no text.
7
+ */
8
+ export function escapeTablePipes(tree) {
9
+ visit(tree, "table-cell", (cell) => {
10
+ visit(cell, "text", (node) => {
11
+ node.value = node.value?.replaceAll("|", "\\vert{}");
12
+ });
13
+ });
14
+ }
@@ -0,0 +1,10 @@
1
+ import type { Parent } from "unist";
2
+ /**
3
+ * org→md: guards the lines uniorg misreads before parsing.
4
+ */
5
+ export declare function guardUnderscoreBullets(org: string): string;
6
+ /**
7
+ * org→md: drops the guards again, also where uniorg reads no text (a
8
+ * src block's code). A zero-width space the author put there goes too.
9
+ */
10
+ export declare function dropUnderscoreBulletGuards(tree: Parent): void;
@@ -0,0 +1,32 @@
1
+ import { visit } from "unist-util-visit";
2
+ import { ZERO_WIDTH_SPACE } from "./markupBoundary.js";
3
+ // ====================================================================
4
+ // WORKAROUND for a bug in uniorg-parse 3.2.2 (upstream issue: not filed
5
+ // yet, see TODO.md). Drop this module once a fixed version is in; the
6
+ // canary in tests/uniorgWorkarounds.spec.ts fails then.
7
+ // ====================================================================
8
+ // Its `listItemRe` takes a line starting `_.` or `_)` for a list item
9
+ // (`\w` includes `_`), but its `fullListItemRe`, like org itself, has
10
+ // no such bullet. `parseListStructure` then throws (`match error`), or,
11
+ // when a real bullet line follows, reads that one instead and silently
12
+ // drops the line. Org reads the line as text, and so does uniorg with a
13
+ // zero-width space in front (morg's line-start escape, see lineSyntax)
14
+ const UNDERSCORE_BULLET_RE = /^([ \t]*)(?=_[.)](?:[ \t]|$))/gm;
15
+ /**
16
+ * org→md: guards the lines uniorg misreads before parsing.
17
+ */
18
+ export function guardUnderscoreBullets(org) {
19
+ return org.replace(UNDERSCORE_BULLET_RE, `$1${ZERO_WIDTH_SPACE}`);
20
+ }
21
+ const GUARD_RE = new RegExp(`(^|\\n)([ \\t]*)${ZERO_WIDTH_SPACE}(?=_[.)](?:[ \\t\\n]|$))`, "g");
22
+ /**
23
+ * org→md: drops the guards again, also where uniorg reads no text (a
24
+ * src block's code). A zero-width space the author put there goes too.
25
+ */
26
+ export function dropUnderscoreBulletGuards(tree) {
27
+ visit(tree, (node) => {
28
+ if (typeof node.value === "string") {
29
+ node.value = node.value.replace(GUARD_RE, "$1$2");
30
+ }
31
+ });
32
+ }
@@ -191,7 +191,7 @@ function transformParagraph(ctx, node) {
191
191
  const children = transformUniorgObjects(ctx, node.children);
192
192
  // md gives leading whitespace structural meaning (list
193
193
  // continuation, code); collapse per-line indentation inside
194
- // paragraphs — insignificant in org and in rendered md alike
194
+ // paragraphs, insignificant in org and in rendered md alike
195
195
  children.forEach((child, index) => {
196
196
  if (child.type === "text") {
197
197
  child.value = child.value.replace(/\n[ \t]+/g, "\n");
@@ -67,11 +67,22 @@ function transformUniorgList(ctx, node) {
67
67
  };
68
68
  });
69
69
  }
70
+ // uniorg keeps a list item's indentation in its code blocks' values;
71
+ // md indents them itself. Only that much is dropped, as uniorg-stringify
72
+ // does, so the code keeps its own
73
+ function outdent(value, level) {
74
+ return value.replace(new RegExp(`^ {0,${level}}`, "gm"), "");
75
+ }
70
76
  function transformUniorgListItem(ctx, item) {
71
77
  // md has no descriptive lists, so keep the ` :: ` syntax literally in the
72
- // item text — the return trip re-parses it as a descriptive list
78
+ // item text; the return trip re-parses it as a descriptive list
73
79
  const tag = listItemTag(item);
74
80
  const children = transformNodes(ctx, (item.children || []).filter(child => child !== tag));
81
+ for (const child of children) {
82
+ if (child.type === "code") {
83
+ child.value = outdent(child.value, item.indent + item.bullet.length);
84
+ }
85
+ }
75
86
  if (tag) {
76
87
  const term = {
77
88
  type: "text",
@@ -1,6 +1,7 @@
1
1
  import { toString as orgastToString } from "orgast-util-to-string";
2
2
  import { htmlEnabled, orgNodeToText, warn } from "./shared.js";
3
3
  import { transformFootnoteReference } from "./footnotes.js";
4
+ import { unescapeOrgPath } from "../orgPath.js";
4
5
  const IMAGE_EXTENSION_RE = /\.(png|jpe?g|gif|svg|webp|avif|bmp|ico)$/i;
5
6
  export function transformUniorgObjects(ctx, children) {
6
7
  return (children || [])
@@ -74,12 +75,43 @@ function transformUniorgObjectToMdastPhrasingContent(ctx, node) {
74
75
  return null;
75
76
  }
76
77
  }
78
+ // the inverse of md→org's decoding: % and # would be read as an escape
79
+ // or the anchor, and a markdown url cannot hold a bare space. Org path
80
+ // escapes (`[`, `]`, `:`) are undone first rather than escaped twice
81
+ function encodeUrlPart(part) {
82
+ return unescapeOrgPath(part)
83
+ .replaceAll("%", "%25")
84
+ .replaceAll("#", "%23")
85
+ .replaceAll(" ", "%20");
86
+ }
87
+ // a path starting like `a:` would read as a url scheme in markdown
88
+ function encodeUrlPath(path) {
89
+ const encoded = encodeUrlPart(path);
90
+ return /^[a-z][a-z0-9+.-]*:/i.test(encoded)
91
+ ? encoded.replaceAll(":", "%3A")
92
+ : encoded;
93
+ }
94
+ // an org file: link (or a ./ path) is a relative markdown link, the
95
+ // search option becoming the #anchor
96
+ function markdownUrl(node) {
97
+ if (node.linkType !== "file") {
98
+ return node.rawLink;
99
+ }
100
+ const target = node.rawLink.replace(/^file:/, "");
101
+ const search = target.indexOf("::");
102
+ return search === -1
103
+ ? encodeUrlPath(target)
104
+ : `${encodeUrlPath(target.slice(0, search))}#${encodeUrlPart(target.slice(search + 2))}`;
105
+ }
77
106
  function transformUniorgLink(ctx, node) {
78
107
  const descriptionText = orgastToString(node);
108
+ const url = markdownUrl(node);
79
109
  // org has no dedicated image syntax; the common convention is a
80
- // link to an image file, so map those to markdown images
81
- if (IMAGE_EXTENSION_RE.test(node.rawLink)) {
82
- return { type: "image", url: node.rawLink, alt: descriptionText };
110
+ // link to an image file, so map those to markdown images; a file
111
+ // link's ::search option is no part of the file name
112
+ const path = node.linkType === "file" ? node.rawLink.replace(/::.*$/s, "") : node.rawLink;
113
+ if (IMAGE_EXTENSION_RE.test(path)) {
114
+ return { type: "image", url, alt: descriptionText };
83
115
  }
84
116
  // a description equal to the url (a common Logseq pattern) is no
85
117
  // description: text === url makes remark-stringify emit an
@@ -100,8 +132,8 @@ function transformUniorgLink(ctx, node) {
100
132
  withoutMarkers(serializedDescription) === withoutMarkers(node.rawLink)) {
101
133
  return {
102
134
  type: "link",
103
- url: node.rawLink,
104
- children: [{ type: "text", value: node.rawLink }]
135
+ url,
136
+ children: [{ type: "text", value: url }]
105
137
  };
106
138
  }
107
139
  const children = transformUniorgObjects(ctx, node.children);
@@ -110,7 +142,7 @@ function transformUniorgLink(ctx, node) {
110
142
  const flattened = children.some(child => child.type === "link" || child.type === "image")
111
143
  ? [{ type: "text", value: descriptionText }]
112
144
  : children;
113
- return { type: "link", url: node.rawLink, children: flattened };
145
+ return { type: "link", url, children: flattened };
114
146
  }
115
147
  function transformScriptMarkup(ctx, node) {
116
148
  // markdown has no equivalents; with useHtml render as raw html
@@ -16,12 +16,18 @@ function isAlignmentCookieRow(row) {
16
16
  cells.some(cell => cellText(cell) !== ""));
17
17
  }
18
18
  // org cell content keeps the aligning whitespace padding; markdown
19
- // cells are re-padded by the stringifier
20
- function trimCellPadding(node) {
21
- if (node.type === "text") {
22
- node.value = node.value.trim();
19
+ // cells are re-padded by the stringifier. Only the cell's edges are
20
+ // padding: a space between text and markup is content
21
+ function trimCellPadding(children) {
22
+ const first = children[0];
23
+ if (first?.type === "text") {
24
+ first.value = first.value.trimStart();
23
25
  }
24
- return node;
26
+ const last = children.at(-1);
27
+ if (last?.type === "text") {
28
+ last.value = last.value.trimEnd();
29
+ }
30
+ return children;
25
31
  }
26
32
  export function transformTable(ctx, node) {
27
33
  if (node.tableType === "table.el") {
@@ -59,7 +65,7 @@ export function transformTable(ctx, node) {
59
65
  type: "tableRow",
60
66
  children: (row.children || []).map(cell => ({
61
67
  type: "tableCell",
62
- children: transformUniorgObjects(ctx, cell.children).map(trimCellPadding)
68
+ children: trimCellPadding(transformUniorgObjects(ctx, cell.children))
63
69
  }))
64
70
  }))
65
71
  };
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Splits a file name into the part a derived name keeps and the extension
3
3
  * that carries its format. A dotless name and a dotfile both have no
4
- * extension — `org` is a file called org, and `.org` is a hidden file whose
4
+ * extension: `org` is a file called org, and `.org` is a hidden file whose
5
5
  * name happens to start with a dot.
6
6
  *
7
7
  * The extension comes from the last path segment, since the CLI is handed
package/dist/fileNames.js CHANGED
@@ -6,7 +6,7 @@ const DOCUMENT_FORMATS = new Map([
6
6
  /**
7
7
  * Splits a file name into the part a derived name keeps and the extension
8
8
  * that carries its format. A dotless name and a dotfile both have no
9
- * extension — `org` is a file called org, and `.org` is a hidden file whose
9
+ * extension: `org` is a file called org, and `.org` is a hidden file whose
10
10
  * name happens to start with a dot.
11
11
  *
12
12
  * The extension comes from the last path segment, since the CLI is handed