@rohal12/spindle 0.52.8 → 0.52.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/parser.ts CHANGED
@@ -1,3 +1,6 @@
1
+ import { tokenize } from './markup/tokenizer';
2
+ import { isCodeAttribute, splitSigilTemplate } from './markup/code-attributes';
3
+
1
4
  export interface Passage {
2
5
  pid: number;
3
6
  name: string;
@@ -18,25 +21,128 @@ export interface StoryData {
18
21
  userScript: string;
19
22
  }
20
23
 
24
+ const HTML_NAMESPACE = 'http://www.w3.org/1999/xhtml';
25
+
26
+ /** HTML elements written without content or end tag, as innerHTML does. */
27
+ const VOID_ELEMENTS = new Set([
28
+ 'area',
29
+ 'base',
30
+ 'basefont',
31
+ 'bgsound',
32
+ 'br',
33
+ 'col',
34
+ 'embed',
35
+ 'frame',
36
+ 'hr',
37
+ 'img',
38
+ 'input',
39
+ 'keygen',
40
+ 'link',
41
+ 'meta',
42
+ 'param',
43
+ 'source',
44
+ 'track',
45
+ 'wbr',
46
+ ]);
47
+
21
48
  /**
22
- * Decode the HTML entities that browsers use when serializing innerHTML.
23
- * &amp; is decoded last to avoid double-decoding (e.g. &amp;lt; → &lt;, not <).
49
+ * The passage markup in the nodes of passage data.
50
+ *
51
+ * A compiler that HTML-escapes passage text (Twine, Tweego) leaves a text
52
+ * node whose text is the markup; a no-break space in it is U+00A0, which
53
+ * innerHTML writes as `&nbsp;`. One that doesn't escape it leaves the
54
+ * browser's parse of it, whose elements are written back as markup here.
55
+ * innerHTML did that too, but its escaping can't be undone: a `"` in an
56
+ * attribute value is written `&quot;`, and decoding that ends the value
57
+ * early, while text it leaves unescaped (in `<style>`, comments) would be
58
+ * decoded wrongly.
24
59
  */
25
- function decodeHtmlEntities(html: string): string {
26
- return html
27
- .replace(/&lt;/g, '<')
28
- .replace(/&gt;/g, '>')
29
- .replace(/&quot;/g, '"')
30
- .replace(/&#39;/g, "'")
31
- .replace(/&amp;/g, '&');
60
+ function passageMarkup(parent: ParentNode): string {
61
+ let markup = '';
62
+ for (const node of Array.from(parent.childNodes)) {
63
+ if (node.nodeType === Node.TEXT_NODE) {
64
+ markup += (node as Text).data;
65
+ } else if (node.nodeType === Node.COMMENT_NODE) {
66
+ markup += `<!--${(node as Comment).data}-->`;
67
+ } else if (node.nodeType === Node.ELEMENT_NODE) {
68
+ markup += elementMarkup(node as Element);
69
+ }
70
+ }
71
+ return markup;
72
+ }
73
+
74
+ /** An element of passage data as markup (see passageMarkup). */
75
+ function elementMarkup(el: Element): string {
76
+ const name = el.localName;
77
+ let markup = `<${name}`;
78
+ for (const attr of Array.from(el.attributes)) {
79
+ markup += ` ${attr.name}=${quoteAttribute(attr.name, attr.value)}`;
80
+ }
81
+ markup += '>';
82
+ if (el.namespaceURI === HTML_NAMESPACE && VOID_ELEMENTS.has(name)) {
83
+ return markup;
84
+ }
85
+ const content =
86
+ name === 'template' ? (el as HTMLTemplateElement).content : el;
87
+ return `${markup}${passageMarkup(content)}</${name}>`;
88
+ }
89
+
90
+ /**
91
+ * An attribute value quoted so that the passage tokenizer reads it back:
92
+ * in double quotes, else single ones, else with the double quotes in its
93
+ * text written as `&quot;`, which rendering decodes. Quotes inside markup in
94
+ * the value (`{print "a"}`) are code, and don't end it.
95
+ */
96
+ function quoteAttribute(name: string, value: string): string {
97
+ const code = isCodeAttribute(name);
98
+ const probe = code ? 'on' : 'a';
99
+ const readsBack = (quoted: string) => {
100
+ const tokens = tokenize(`<i ${probe}=${quoted}>`);
101
+ const tag = tokens[0];
102
+ return (
103
+ tokens.length === 1 &&
104
+ tag?.type === 'html' &&
105
+ tag.attributes[probe] === quoted.slice(1, -1)
106
+ );
107
+ };
108
+ return (
109
+ [`"${value}"`, `'${value}'`].find(readsBack) ??
110
+ `"${escapeTextQuotes(value, code)}"`
111
+ );
112
+ }
113
+
114
+ /**
115
+ * `value` with the double quotes in its text written as `&quot;`, but not
116
+ * those in markup (`{print "a"}`) or, in a code attribute, in sigil
117
+ * references (see splitSigilTemplate), which are code.
118
+ */
119
+ function escapeTextQuotes(value: string, code: boolean): string {
120
+ const escape = (text: string) => text.replace(/"/g, '&quot;');
121
+ if (code) {
122
+ return splitSigilTemplate(value)
123
+ .map((part) =>
124
+ 'text' in part
125
+ ? escape(part.text)
126
+ : 'expr' in part
127
+ ? `{${part.expr}}`
128
+ : part.verbatim,
129
+ )
130
+ .join('');
131
+ }
132
+ return tokenize(value, { text: true })
133
+ .map((token) => {
134
+ const source = value.slice(token.start, token.end);
135
+ return token.type === 'text' ? escape(source) : source;
136
+ })
137
+ .join('');
32
138
  }
33
139
 
34
140
  /**
35
141
  * Parse <tw-storydata> and all <tw-passagedata> elements from the DOM.
36
142
  *
37
- * Uses innerHTML + entity decoding instead of textContent so that HTML
38
- * tags inside passage markup (e.g. <div> inside {button}) are preserved
39
- * regardless of whether the compiler HTML-encoded them.
143
+ * Passage content is read with passageMarkup instead of textContent, so
144
+ * that HTML tags inside passage markup (e.g. <div> inside {button}) are
145
+ * preserved whether or not the compiler HTML-encoded them.
40
146
  */
41
147
  export function parseStoryData(): StoryData {
42
148
  const storyEl = document.querySelector('tw-storydata');
@@ -67,7 +173,7 @@ export function parseStoryData(): StoryData {
67
173
  const tags = (el.getAttribute('tags') || '')
68
174
  .split(/\s+/)
69
175
  .filter((t) => t.length > 0);
70
- const content = decodeHtmlEntities(el.innerHTML);
176
+ const content = passageMarkup(el);
71
177
 
72
178
  const metadata: Record<string, string> = {};
73
179
  const skipAttrs = new Set(['pid', 'name', 'tags']);