@panphora/clayjs 1.2.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +7 -3
  2. package/THIRD-PARTY-NOTICES.md +10 -0
  3. package/dist/clay.standalone.js +22539 -14260
  4. package/entries/clay-data.js +1 -1
  5. package/entries/sap.js +1 -1
  6. package/package.json +7 -2
  7. package/packed-contract.json +17 -0
  8. package/src/attrs/save-freeze.js +10 -18
  9. package/src/core/admin-contenteditable.js +8 -6
  10. package/src/core/admin-inputs.js +23 -13
  11. package/src/core/admin-onclick.js +5 -0
  12. package/src/core/is-edit-mode.js +7 -2
  13. package/src/core/persist.js +5 -10
  14. package/src/core/save-core.js +35 -0
  15. package/src/core/save.js +13 -1
  16. package/src/core/snapshot.js +186 -14
  17. package/src/core/source-map.js +1025 -0
  18. package/src/core/unsaved-warning.js +3 -0
  19. package/src/dom/dom-helpers.js +5 -1
  20. package/src/lib/content-dom.js +108 -0
  21. package/src/lib/mutation.js +26 -3
  22. package/src/lib/region-capabilities.js +69 -0
  23. package/src/lib/region-policy.js +18 -13
  24. package/src/loader-logic.js +27 -5
  25. package/src/loader.js +6 -0
  26. package/src/plugins/ai-edit.js +625 -0
  27. package/src/plugins/demo.js +3 -0
  28. package/src/plugins/sortable.js +6 -1
  29. package/src/plugins/source.js +410 -0
  30. package/src/plugins/wire.js +248 -47
  31. package/src/sync/live-sync.js +106 -42
  32. package/src/sync/presence.js +303 -0
  33. package/src/sync/section-notice.js +230 -0
  34. package/src/sync/splice-merge.js +7 -10
  35. package/src/sync/stream.js +190 -0
  36. package/src/vendor/control-serialize.vendor.js +10 -6
  37. package/src/vendor/hyper-morph.vendor.js +2 -2
  38. package/src/vendor/hyper-undo.vendor.js +1 -1
  39. package/src/vendor/hypercms.vendor.js +438 -45
  40. package/src/vendor/parse5.vendor.js +3 -0
  41. package/src/vendor/quickcrop.vendor.js +1 -1
  42. package/src/vendor/richclay.vendor.js +22 -15
@@ -0,0 +1,1025 @@
1
+ /**
2
+ * source-map.js — save the file, not a serialization of it.
3
+ *
4
+ * Every program that edits a malleable HTML file rewrites the whole file, because
5
+ * each one parses to a tree and serializes the tree back out, and a serializer does
6
+ * not reproduce its input. A no-edit save through `clone.outerHTML` rewrites about
7
+ * 88% of an authored document's lines: attributes reorder, quoting normalises, `&`
8
+ * becomes `&`, every tag reprints canonically. Nothing is lost, and the file is
9
+ * no longer the file anybody wrote.
10
+ *
11
+ * This module keeps the bytes the document was loaded from, pairs each live node to
12
+ * a byte range in them, and on save copies source bytes for everything unchanged and
13
+ * prints only what changed.
14
+ *
15
+ * FIVE OPERATIONS
16
+ *
17
+ * model(src) parse5 tree of byte locations; implied tags get a content range
18
+ * pair(root, m) live node -> byte range, by tiered signature alignment
19
+ * render(clone, ...) the save clone, emitted exactly: every child, whitespace
20
+ * included, no gaps invented and none dropped
21
+ * verify(out, today) reparse and compare trees before anything is sent
22
+ * adopt(bytes) re-model and re-pair against what the host accepted
23
+ *
24
+ * THE FLOOR IS TODAY'S BEHAVIOUR. Anything that does not verify is sent as the full
25
+ * serialization instead, and counted. A fetch that does not return this document, a
26
+ * source whose shape outside <html> disagrees with the live document, a render that
27
+ * throws: each one means this module does nothing at all and the save is exactly the
28
+ * save it would have been.
29
+ *
30
+ * WHAT IS DELIBERATELY NOT HERE
31
+ *
32
+ * No ids in the file. Identity is the live node object, held in a WeakMap.
33
+ * No sidecar, no server change, no second format on disk.
34
+ * No tolerance in the verifier. A reprint is fixed in the renderer or in the page,
35
+ * never by widening what counts as equal — a verifier that shares a predicate with
36
+ * the renderer cannot see the renderer's mistakes, which is how an earlier version
37
+ * of this code wrote `by <a>Ana</a> <a>Bo</a>` back as `AnaBo` with a green check.
38
+ */
39
+
40
+ import { parse } from '../vendor/parse5.vendor.js';
41
+
42
+ const RAW_TEXT = new Set(['script', 'style', 'xmp', 'iframe', 'noembed', 'noframes', 'plaintext', 'noscript']);
43
+ const RCDATA = new Set(['textarea', 'title']);
44
+ const textContext = (tag) => (RAW_TEXT.has(tag) ? 'raw' : RCDATA.has(tag) ? 'rcdata' : 'normal');
45
+ const VOID = new Set(['area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr']);
46
+ const XHTML = 'http://www.w3.org/1999/xhtml';
47
+
48
+ const escText = (s) => s.replace(/&/g, '&amp;').replace(/ /g, '&nbsp;').replace(/</g, '&lt;').replace(/>/g, '&gt;');
49
+ const escAttr = (s) => s.replace(/&/g, '&amp;').replace(/ /g, '&nbsp;').replace(/"/g, '&quot;');
50
+
51
+ // =============================================================================
52
+ // MODEL
53
+ // =============================================================================
54
+
55
+ export function model(src) {
56
+ // The error counts come out of the SAME parse, so they cost nothing here. They are
57
+ // the only evidence of input a parser throws away, which a tree comparison cannot
58
+ // see by construction: a duplicate attribute is discarded during parsing, so a render
59
+ // that wrote one twice reparsed to exactly the right tree. That is how the same
60
+ // `xmlns:xlink` was added to a file on every save with the verifier green each time.
61
+ const parseErrors = new Map();
62
+ const doc = parse(src, {
63
+ sourceCodeLocationInfo: true,
64
+ onParseError: (e) => parseErrors.set(e.code, (parseErrors.get(e.code) || 0) + 1),
65
+ });
66
+ const html = doc.childNodes.find((n) => n.nodeName === 'html');
67
+ const outside = doc.childNodes.filter((n) => n !== html).map((n) => n.nodeName === '#documentType'
68
+ ? { kind: 'doctype', name: n.name || '', publicId: n.publicId || '', systemId: n.systemId || '' }
69
+ : { kind: n.nodeName === '#comment' ? 'comment' : n.nodeName, value: n.data });
70
+ const root = html ? locOf(html, null, src) : null;
71
+ if (root) settle(root, src);
72
+ return {
73
+ src, doc, outside, root, parseErrors,
74
+ relocated: root ? firstOutOfOrder(root) : null,
75
+ movedIn: root ? firstOutOfOrder(root, 'movedIn') : null,
76
+ tailOwner: root ? tailOwner(root) : null,
77
+ };
78
+ }
79
+
80
+ function locOf(n, parent, src) {
81
+ const l = n.sourceCodeLocation || null;
82
+ const kind = n.nodeName === '#text' ? 'text' : n.nodeName === '#comment' ? 'comment' : 'element';
83
+ const loc = { node: n, parent, live: null, kind, children: [], from: l ? l.startOffset : -1, to: l ? l.endOffset : -1 };
84
+ if (kind === 'text') { loc.value = n.value; return loc; }
85
+ if (kind === 'comment') { loc.value = n.data; return loc; }
86
+ loc.tag = n.tagName;
87
+ loc.located = !!(l && l.startTag);
88
+ loc.openFrom = loc.located ? l.startTag.startOffset : -1;
89
+ loc.openTo = loc.located ? l.startTag.endOffset : -1;
90
+ loc.closeFrom = l && l.endTag ? l.endTag.startOffset : (loc.located ? loc.to : -1);
91
+ loc.closeTo = l && l.endTag ? l.endTag.endOffset : (loc.located ? loc.to : -1);
92
+ loc.endTagged = !!(l && l.endTag);
93
+ // Nothing can end before its own start tag does, but parse5 reports exactly that for
94
+ // an element the parser inserts and immediately pops. A <form> written directly
95
+ // inside a <table> is the case that ships: the form is inserted, the form pointer is
96
+ // set, and the element is popped off the stack at once, so nothing ever closes it and
97
+ // its endOffset stays at the `<` of its start tag. The parent then rewound its copy
98
+ // cursor to that offset after emitting the tag and copied the whole `<form ...>`
99
+ // again, so every save added one more copy and the tree never changed, which is why
100
+ // verify passed each time. Both ends move up to the end of the start tag, which is
101
+ // the least this element can be said to occupy.
102
+ if (loc.located) {
103
+ if (loc.to < loc.openTo) loc.to = loc.openTo;
104
+ if (loc.closeFrom < loc.openTo) loc.closeFrom = loc.closeTo = loc.openTo;
105
+ }
106
+ loc.attrs = [];
107
+ const attrLocs = (loc.located && l.startTag.attrs) || (l && l.attrs) || {};
108
+ // Keyed by the name as WRITTEN, which is the only name both sides agree on. parse5
109
+ // adjusts a foreign-content attribute in the tree, so `xmlns:xlink` arrives as
110
+ // `{ name: 'xlink', prefix: 'xmlns' }`, while it keys the location map and the live
111
+ // DOM's `attr.name` by `xmlns:xlink`. Keyed by the adjusted name the lookup missed,
112
+ // the attribute got no source range, its bytes rode along inside a copied run, and
113
+ // the live copy was appended as a new attribute: every save added another
114
+ // `xmlns:xlink="..."` and the tree stayed identical, so verify passed each time.
115
+ for (const a of n.attrs) {
116
+ const written = a.prefix ? a.prefix + ':' + a.name : a.name;
117
+ const al = attrLocs[written.toLowerCase()];
118
+ loc.attrs.push({ name: written, key: written.toLowerCase(), value: a.value, from: al ? al.startOffset : -1, to: al ? al.endOffset : -1 });
119
+ }
120
+ const kids = n.nodeName === 'template' && n.content ? n.content.childNodes : n.childNodes;
121
+ for (const c of kids) {
122
+ if (c.nodeName === '#documentType') continue;
123
+ loc.children.push(locOf(c, loc, src));
124
+ }
125
+ // An implied tag (<html>, <head>, <body> the author never wrote) has no tag bytes. Its
126
+ // content range is its children's span, and its open and close tags are empty ranges at
127
+ // either end, so emitting it copies the children and writes no tag.
128
+ if (!loc.located) {
129
+ const located = loc.children.filter((c) => c.from >= 0);
130
+ const first = located.length ? Math.min(...located.map((c) => c.from)) : -1;
131
+ const last = located.length ? Math.max(...located.map((c) => c.to)) : -1;
132
+ loc.openFrom = loc.openTo = first;
133
+ loc.closeFrom = loc.closeTo = last;
134
+ loc.from = first; loc.to = last;
135
+ }
136
+ return loc;
137
+ }
138
+
139
+ /**
140
+ * Clamp every child's ranges into its parent's, top-down.
141
+ *
142
+ * Top-down because a parent's close range can itself be corrected here, and its
143
+ * children have to be clamped against the corrected one. An element the parser closed
144
+ * implicitly (a missing </div>, an omitted </p> at </body>) has no end tag, and parse5
145
+ * gives it the end of the file as its end offset. Its close gap then spanned its
146
+ * ancestors' end tags, which it copied, and they copied them again: the file grew by
147
+ * `</body></html>` on every save, and the extra tags reparse to nothing, so verify
148
+ * passed each time.
149
+ */
150
+ function settle(loc, src, limit = Infinity) {
151
+ if (loc.kind !== 'element') return;
152
+ for (const cl of loc.children) {
153
+ // Content written after an end tag that the parser carried back inside the element:
154
+ // anything after </body> or </html> other than whitespace lands in <body>. Its
155
+ // bytes sit past the close that every copy of the file emits verbatim, so a render
156
+ // writes them twice. Refused at install, like foster parenting, for the same reason.
157
+ if (cl.from >= limit && !(cl.kind === 'text' && !/\S/.test(cl.value))) loc.movedIn = true;
158
+ // Clamp every child into its parent's content range, at BOTH ends.
159
+ //
160
+ // The spec (and parse5, and every browser) appends whitespace that follows an end
161
+ // tag to the last text node before it, so such a node's range runs past the tag,
162
+ // and when nothing precedes the tag the range starts past it as well. Left alone,
163
+ // the first case emits the same bytes twice and the second inverts the range:
164
+ // `</script></body>\n</html>` gave body a last text node beginning AFTER `</body>`,
165
+ // and the copy of the gap before it wrote `</body>` a second time. Chromium found
166
+ // that one; jsdom did not, only because every hand-written fixture happened to have
167
+ // a newline before `</body>`.
168
+ if (cl.from >= 0 && loc.located) {
169
+ // Both ends land inside [openTo, closeFrom], and `from` never passes `to`: a
170
+ // range that begins after the end tag collapses to an empty one AT the end tag,
171
+ // not after it. Putting it after is what wrote `</body>` twice, because the copy
172
+ // of the gap up to that range then spanned the tag.
173
+ const to = Math.max(loc.openTo, Math.min(cl.to, loc.closeFrom));
174
+ const from = Math.min(Math.max(cl.from, loc.openTo), to);
175
+ if (from !== cl.from || to !== cl.to) {
176
+ // What the bytes inside the parent account for stays with this node; the rest
177
+ // was relocated from outside, belongs to an ancestor, and is emitted there. A
178
+ // node that gets PRINTED rather than copied has to leave it out or it appears
179
+ // twice, which is a tree difference and a fallback on the next save.
180
+ //
181
+ // Neither end of the span can be measured directly. Counting back by the
182
+ // overhang's byte length is wrong because the span covers the end tag too, and
183
+ // slicing the source after the end tag is wrong because the relocation can
184
+ // cross two of them: `</p>\n</body>\n</html>\n` puts three newlines from three
185
+ // different places into one text node.
186
+ //
187
+ // Whitespace only, which is all a parser relocates.
188
+ if (cl.kind === 'text' && !RAW_TEXT.has(loc.tag) && !RCDATA.has(loc.tag)) {
189
+ // Read from the bytes PAST the clamped end, with the end tags and comments the
190
+ // span crosses taken out, never by matching the bytes inside against the
191
+ // node's value: those can hold a character reference, `&amp;` in the file and
192
+ // `&` in the value, so a prefix match found no overhang at all and an edit to
193
+ // the text printed the relocated newlines on top of the ones still copied
194
+ // after the end tags. Normalized, because the node's value is: the parser
195
+ // rewrites every CRLF and lone CR to LF before a text node sees them. Only the
196
+ // tail's text is normalized; the copy path still uses the raw bytes, which is
197
+ // why an unedited CRLF file round trips exactly. Never inside a raw-text or
198
+ // RCDATA element, whose "tags" are text: an unclosed <textarea> at the end of a
199
+ // file holds `</html>\n` as characters, not as an end tag and a relocation.
200
+ const outside = src.slice(to, cl.to).replace(/<!--[\s\S]*?-->|<[^>]*>/g, '').replace(/\r\n?/g, '\n');
201
+ if (outside && !/\S/.test(outside) && cl.value.endsWith(outside)) cl.outsideTail = outside;
202
+ }
203
+ cl.clamped = true;
204
+ cl.from = from;
205
+ cl.to = to;
206
+ }
207
+ }
208
+ if (cl.kind === 'element' && cl.located && !cl.endTagged && cl.closeFrom > cl.to) {
209
+ cl.closeFrom = cl.closeTo = cl.to;
210
+ }
211
+ }
212
+ // Every copy below walks the source forward, so the children have to be in source
213
+ // order. Foster parenting is where they are not: content written inside a <table>
214
+ // that does not belong there is moved OUT, to just before the table, so the tree
215
+ // order and the byte order disagree and a forward walk emits those bytes at the new
216
+ // position and again inside the table. The bytes belong to one element and the node
217
+ // to another, which this model has no way to say, so the document is refused instead
218
+ // and saves it with today's serializer. That is what such a page already got: the
219
+ // render was wrong, verify caught it, and every save fell back.
220
+ let floor = loc.located ? loc.openTo : -1;
221
+ for (const cl of loc.children) {
222
+ if (cl.from < 0) continue;
223
+ if (cl.from < floor) { loc.outOfOrder = true; break; }
224
+ floor = cl.to;
225
+ }
226
+ const inner = loc.located && loc.endTagged ? Math.min(limit, loc.closeTo) : limit;
227
+ for (const cl of loc.children) settle(cl, src, inner);
228
+ }
229
+
230
+ function firstOutOfOrder(loc, flag = 'outOfOrder') {
231
+ if (loc[flag]) return loc.tag;
232
+ for (const c of loc.children) {
233
+ if (c.kind !== 'element') continue;
234
+ const found = firstOutOfOrder(c, flag);
235
+ if (found) return found;
236
+ }
237
+ return null;
238
+ }
239
+
240
+ /** The one text node holding whitespace the parser carried in from past </body>, if any. */
241
+ function tailOwner(loc) {
242
+ if (loc.outsideTail) return loc;
243
+ for (const c of loc.children || []) {
244
+ const found = tailOwner(c);
245
+ if (found) return found;
246
+ }
247
+ return null;
248
+ }
249
+
250
+ /**
251
+ * Refuse a model that is not this document.
252
+ *
253
+ * The live document parsed the bytes that were actually served, so its doctype and
254
+ * its document-level comments are an oracle for everything render copies from
255
+ * outside <html> — the one region no tree comparison can check, because a tree
256
+ * comparison starts at documentElement. A boot fetch that returned a login page or
257
+ * an error page disagrees here, and that is the difference between doing nothing and
258
+ * writing that page's doctype into somebody's file.
259
+ *
260
+ * @returns {?string} the disagreement, or null when the source is this document
261
+ */
262
+ export function checkSource(m, liveDocument) {
263
+ if (!m.root) return 'source has no <html> element';
264
+ // Nothing inside <html> carries a byte range, which is what a document with no
265
+ // elements, text or comments at all looks like. render brackets the whole output
266
+ // with the root's range, so there is nothing here to preserve and nothing to
267
+ // bracket with.
268
+ if (m.root.openFrom < 0) return 'source has no content inside <html>';
269
+ if (m.relocated) return 'the parser moved content out of <' + m.relocated + '>, so source order and tree order disagree';
270
+ if (m.movedIn) return 'the parser moved content written after the end of <' + m.movedIn + '> back inside it, so source order and tree order disagree';
271
+ const want = outsideOf(liveDocument);
272
+ const got = m.outside.map(outsideKey);
273
+ if (want.length !== got.length) return 'outside <html>: ' + JSON.stringify(got) + ' vs live ' + JSON.stringify(want);
274
+ for (let i = 0; i < want.length; i++) if (want[i] !== got[i]) return 'outside <html>: ' + got[i] + ' vs live ' + want[i];
275
+ return null;
276
+ }
277
+ function outsideKey(o) { return o.kind === 'doctype' ? 'doctype:' + o.name + '|' + o.publicId + '|' + o.systemId : o.kind + ':' + o.value; }
278
+ function outsideOf(doc) {
279
+ const out = [];
280
+ for (const n of doc.childNodes) {
281
+ if (n.nodeType === 10) out.push('doctype:' + n.name + '|' + n.publicId + '|' + n.systemId);
282
+ else if (n.nodeType === 8) out.push('comment:' + n.data);
283
+ else if (n.nodeType !== 1) out.push('#' + n.nodeType + ':' + (n.data || ''));
284
+ }
285
+ return out;
286
+ }
287
+
288
+ // =============================================================================
289
+ // SIGNATURES
290
+ // =============================================================================
291
+ // Four keys per node, computed once, bottom-up. `deep` is the whole subtree (tag,
292
+ // attributes, text), `content` ignores attributes at every level, `shallow` is this
293
+ // node's tag and attributes only, `tag` is the tag. Alignment tries them strongest
294
+ // first, so a runtime attribute change deep inside a card still lets the card pair by
295
+ // its text, and two genuinely identical rows pair by position.
296
+
297
+ const UNIT_SEP = '\u0000';
298
+
299
+ function mix(h, s) {
300
+ for (let i = 0; i < s.length; i++) { h ^= s.charCodeAt(i); h = Math.imul(h, 16777619); }
301
+ return h >>> 0;
302
+ }
303
+ function hashKey(parts) {
304
+ let a = 2166136261, b = 5381;
305
+ for (const p of parts) { a = mix(a, p); b = (Math.imul(b, 33) ^ mix(a ^ 0x9e3779b9, p)) >>> 0; }
306
+ return a.toString(36) + '.' + b.toString(36);
307
+ }
308
+ function liveKids(el) {
309
+ const list = el.nodeType === 1 && el.tagName === 'TEMPLATE' && el.content ? el.content.childNodes : el.childNodes;
310
+ const out = [];
311
+ for (const n of list) if (n.nodeType === 1 || n.nodeType === 3 || n.nodeType === 8) out.push(n);
312
+ return out;
313
+ }
314
+ const liveAttrsKey = (el) => Array.from(el.attributes, (a) => a.name.toLowerCase() + '=' + a.value).sort().join(UNIT_SEP);
315
+ const srcAttrsKey = (loc) => loc.attrs.map((a) => a.key + '=' + a.value).sort().join(UNIT_SEP);
316
+ function keysOfLive(n, keyMap) {
317
+ let k;
318
+ if (n.nodeType === 3) k = { tag: '#text', shallow: 't:' + n.data, content: 't:' + n.data, deep: 't:' + n.data };
319
+ else if (n.nodeType === 8) k = { tag: '#comment', shallow: 'c:' + n.data, content: 'c:' + n.data, deep: 'c:' + n.data };
320
+ else {
321
+ const kids = liveKids(n).map((c) => keysOfLive(c, keyMap));
322
+ const shallow = 'e:' + n.localName + '|' + liveAttrsKey(n);
323
+ k = { tag: n.localName, shallow, content: 'e:' + n.localName + '|' + hashKey(kids.map((c) => c.content)), deep: shallow + '|' + hashKey(kids.map((c) => c.deep)) };
324
+ }
325
+ keyMap.set(n, k);
326
+ return k;
327
+ }
328
+ function keysOfLoc(loc) {
329
+ if (loc.kind === 'text') return (loc.keys = { tag: '#text', shallow: 't:' + loc.value, content: 't:' + loc.value, deep: 't:' + loc.value });
330
+ if (loc.kind === 'comment') return (loc.keys = { tag: '#comment', shallow: 'c:' + loc.value, content: 'c:' + loc.value, deep: 'c:' + loc.value });
331
+ const kids = loc.children.map(keysOfLoc);
332
+ const shallow = 'e:' + loc.tag + '|' + srcAttrsKey(loc);
333
+ return (loc.keys = { tag: loc.tag, shallow, content: 'e:' + loc.tag + '|' + hashKey(kids.map((c) => c.content)), deep: shallow + '|' + hashKey(kids.map((c) => c.deep)) });
334
+ }
335
+
336
+ // =============================================================================
337
+ // ALIGNMENT
338
+ // =============================================================================
339
+
340
+ const TIERS = ['deep', 'content', 'shallow', 'tag'];
341
+ const DP_CELLS = 4096;
342
+
343
+ /**
344
+ * Order-preserving matching of two child lists.
345
+ *
346
+ * Per tier: trim the common prefix and suffix, anchor on keys unique to both sides
347
+ * (patience), recurse into the gaps at the same tier, and when a gap has no anchors
348
+ * run an exact LCS only if it is small, otherwise hand the gap to the next weaker
349
+ * tier. Every step is linear in the gap except the LIS (a log a) and the bounded DP,
350
+ * so this is O(n log n) per parent and the DP table never exceeds DP_CELLS cells.
351
+ *
352
+ * The prefix and suffix trim is what makes it linear in practice, not the anchors:
353
+ * half of every child list in a formatted document is indentation text nodes, which
354
+ * are never unique and so never anchor anything. An earlier exact-DP version took 84
355
+ * seconds on 4,000 identical siblings and allocated a 256 MB table; this one takes
356
+ * 4.7 ms, and 32,000 siblings in 15.5 ms.
357
+ */
358
+ export function align(LK, SK) {
359
+ const pairs = [];
360
+ const LI = LK.map((_, i) => i), SI = SK.map((_, j) => j);
361
+ stage(LI, SI, 0);
362
+ // An anchor a boot script moved across a run of identical siblings strands that run
363
+ // on opposite sides of it. One more pass over what is still unpaired lets the run
364
+ // pair, crossing the anchor; render treats a crossing pair as a move, which is still
365
+ // a byte copy rather than a reprint.
366
+ const usedL = new Set(pairs.map((p) => p[0])), usedS = new Set(pairs.map((p) => p[1]));
367
+ const restL = LI.filter((i) => !usedL.has(i)), restS = SI.filter((j) => !usedS.has(j));
368
+ if (restL.length && restS.length) stage(restL, restS, 0);
369
+ pairs.sort((a, b) => a[0] - b[0]);
370
+ return pairs;
371
+
372
+ function stage(li, si, t) {
373
+ let l0 = 0, l1 = li.length, s0 = 0, s1 = si.length;
374
+ if (l0 >= l1 || s0 >= s1) return;
375
+ if (t >= TIERS.length) return positional(li, si);
376
+ const k = TIERS[t];
377
+ while (l0 < l1 && s0 < s1 && LK[li[l0]][k] === SK[si[s0]][k]) { pairs.push([li[l0], si[s0]]); l0++; s0++; }
378
+ while (l0 < l1 && s0 < s1 && LK[li[l1 - 1]][k] === SK[si[s1 - 1]][k]) { l1--; s1--; pairs.push([li[l1], si[s1]]); }
379
+ if (l0 >= l1 || s0 >= s1) return;
380
+ const seenL = new Map(), seenS = new Map();
381
+ for (let x = l0; x < l1; x++) { const key = LK[li[x]][k]; seenL.set(key, seenL.has(key) ? -1 : x); }
382
+ for (let y = s0; y < s1; y++) { const key = SK[si[y]][k]; seenS.set(key, seenS.has(key) ? -1 : y); }
383
+ const cands = [];
384
+ for (let x = l0; x < l1; x++) { const key = LK[li[x]][k]; if (seenL.get(key) === x) { const y = seenS.get(key); if (y !== undefined && y >= 0) cands.push([x, y]); } }
385
+ const anchors = lisPairs(cands);
386
+ if (!anchors.length) {
387
+ if ((l1 - l0) * (s1 - s0) <= DP_CELLS) {
388
+ const got = lcs(LK, SK, li, si, l0, l1, s0, s1, k);
389
+ if (got.length) { let px = l0, py = s0; for (const [x, y] of got) { pairs.push([li[x], si[y]]); stage(li.slice(px, x), si.slice(py, y), t + 1); px = x + 1; py = y + 1; } stage(li.slice(px, l1), si.slice(py, s1), t + 1); return; }
390
+ }
391
+ return stage(li.slice(l0, l1), si.slice(s0, s1), t + 1);
392
+ }
393
+ let px = l0, py = s0;
394
+ for (const [x, y] of anchors) { pairs.push([li[x], si[y]]); stage(li.slice(px, x), si.slice(py, y), t); px = x + 1; py = y + 1; }
395
+ stage(li.slice(px, l1), si.slice(py, s1), t);
396
+ }
397
+ // Last resort for a large gap nothing else resolved: same-tag runs of equal length
398
+ // pair by position.
399
+ function positional(li, si) {
400
+ const byTagL = new Map(), byTagS = new Map();
401
+ for (const i of li) { const tg = LK[i].tag; if (!byTagL.has(tg)) byTagL.set(tg, []); byTagL.get(tg).push(i); }
402
+ for (const j of si) { const tg = SK[j].tag; if (!byTagS.has(tg)) byTagS.set(tg, []); byTagS.get(tg).push(j); }
403
+ for (const [tg, l] of byTagL) { const s = byTagS.get(tg); if (s && s.length === l.length && tg !== '#text' && tg !== '#comment') l.forEach((i, x) => pairs.push([i, s[x]])); }
404
+ }
405
+ }
406
+
407
+ function lisPairs(cands) { // cands increasing in i; keep the longest subsequence increasing in j
408
+ const tails = [], tailAt = [], prev = new Array(cands.length).fill(-1);
409
+ for (let x = 0; x < cands.length; x++) {
410
+ const v = cands[x][1];
411
+ let lo = 0, hi = tails.length;
412
+ while (lo < hi) { const mid = (lo + hi) >> 1; if (tails[mid] < v) lo = mid + 1; else hi = mid; }
413
+ tails[lo] = v; tailAt[lo] = x; prev[x] = lo > 0 ? tailAt[lo - 1] : -1;
414
+ }
415
+ const out = []; let x = tailAt.length ? tailAt[tailAt.length - 1] : -1;
416
+ while (x >= 0) { out.push(cands[x]); x = prev[x]; }
417
+ return out.reverse();
418
+ }
419
+
420
+ function lcs(LK, SK, li, si, l0, l1, s0, s1, k) {
421
+ const n = l1 - l0, m = s1 - s0;
422
+ const eq = (i, j) => LK[li[l0 + i]][k] === SK[si[s0 + j]][k];
423
+ const dp = Array.from({ length: n + 1 }, () => new Uint16Array(m + 1));
424
+ for (let i = n - 1; i >= 0; i--) for (let j = m - 1; j >= 0; j--)
425
+ dp[i][j] = eq(i, j) ? dp[i + 1][j + 1] + 1 : Math.max(dp[i + 1][j], dp[i][j + 1]);
426
+ const out = []; let i = 0, j = 0;
427
+ while (i < n && j < m) {
428
+ if (eq(i, j)) { out.push([l0 + i, s0 + j]); i++; j++; }
429
+ else if (dp[i + 1][j] >= dp[i][j + 1]) i++; else j++;
430
+ }
431
+ return out;
432
+ }
433
+
434
+ // =============================================================================
435
+ // PAIRING
436
+ // =============================================================================
437
+
438
+ function describe(n) {
439
+ if (n.nodeType === 3) return '#text ' + JSON.stringify(n.data.slice(0, 40));
440
+ if (n.nodeType === 8) return '#comment';
441
+ return '<' + n.localName + Array.from(n.attributes, (a) => ' ' + a.name + '="' + a.value.slice(0, 30) + '"').join('') + '>';
442
+ }
443
+ function describeLoc(l) {
444
+ if (l.kind === 'text') return '#text ' + JSON.stringify(l.value.slice(0, 40));
445
+ if (l.kind === 'comment') return '#comment';
446
+ return '<' + l.tag + l.attrs.map((a) => ' ' + a.name + '="' + a.value.slice(0, 30) + '"').join('') + '>';
447
+ }
448
+
449
+ /**
450
+ * Pair a tree against the model, and key the result by whatever `resolve` returns.
451
+ *
452
+ * The tree walked here is the SAVE CLONE, not the live DOM, because the clone is in
453
+ * the same domain as the file: edit mode has been deactivated back to the inert
454
+ * attribute forms, [no-save] regions are gone, and every document transform has run.
455
+ * The live DOM is in the activated domain, where `contenteditable="true"` stands
456
+ * where the file says `inert-contenteditable="true"`, and pairing there would fail
457
+ * the two strongest signature tiers on every activated node and on every ancestor of
458
+ * one.
459
+ *
460
+ * The MAP is keyed by the LIVE node (`resolve` is snapshot provenance), because the
461
+ * clone is rebuilt from scratch on every save and its nodes are new objects each
462
+ * time. A clone node a transform created has no live original and simply goes
463
+ * unpaired, which means it is printed, the same thing that happens to it today.
464
+ *
465
+ * @param {Node} root - the tree to walk (the save clone's document element)
466
+ * @param {Object} m - a model()
467
+ * @param {Function} [resolve] - node -> the identity to key the map by (default: itself)
468
+ */
469
+ export function pair(root, m, resolve = (n) => n) {
470
+ const map = new WeakMap();
471
+ const keyMap = new Map();
472
+ keysOfLive(root, keyMap);
473
+ keysOfLoc(m.root);
474
+ const stats = { paired: 0, unmatchedLive: [], unmatchedSource: [], attrDiffs: [], textDiffs: 0, unresolved: 0 };
475
+ // A token identifying THIS pairing pass, stamped on the model and on every loc that
476
+ // binds. render trusts a loc only when the two still agree, which rules out a loc
477
+ // left over from an earlier pass and a map paired against a different model. The
478
+ // token is an object rather than the live node on purpose: a loc lives as long as the
479
+ // model, so holding the node here would keep every node the page has since deleted.
480
+ const gen = {};
481
+ m.gen = gen;
482
+ bind(root, m.root, map, keyMap, stats, 'html', resolve, gen);
483
+ return { map, stats };
484
+ }
485
+
486
+ function bind(node, loc, map, keyMap, stats, path, resolve, gen) {
487
+ const key = resolve(node);
488
+ if (key) { map.set(key, loc); loc.gen = gen; stats.paired++; }
489
+ else stats.unresolved++;
490
+ if (loc.kind !== 'element') { if (node.data !== loc.value) stats.textDiffs++; return; }
491
+ if (liveAttrsKey(node) !== srcAttrsKey(loc)) stats.attrDiffs.push({ path, live: describe(node), source: describeLoc(loc) });
492
+ const L = liveKids(node), S = loc.children;
493
+ const pairs = align(L.map((n) => keyMap.get(n)), S.map((l) => l.keys));
494
+ const usedL = new Set(), usedS = new Set();
495
+ for (const [i, j] of pairs) {
496
+ usedL.add(i); usedS.add(j);
497
+ bind(L[i], S[j], map, keyMap, stats, path + '>' + (L[i].nodeType === 1 ? L[i].localName : '#') + '[' + i + ']', resolve, gen);
498
+ }
499
+ for (let i = 0; i < L.length; i++) if (!usedL.has(i)) stats.unmatchedLive.push({ path, node: describe(L[i]) });
500
+ for (let j = 0; j < S.length; j++) if (!usedS.has(j)) stats.unmatchedSource.push({ path, node: describeLoc(S[j]) });
501
+ }
502
+
503
+ // =============================================================================
504
+ // RENDER
505
+ // =============================================================================
506
+
507
+ /**
508
+ * Emit the save clone as bytes, copying source for everything unchanged.
509
+ *
510
+ * EXACT. Every clone child is emitted, whitespace text nodes included, and nothing
511
+ * here knows or cares whether a gap is indentation or a rendered space between two
512
+ * inline elements. An earlier version had a "layout parent" rule that dropped
513
+ * whitespace-only children of block elements, which is correct for indentation and
514
+ * wrong for `by <a>Ana</a> <a>Bo</a>`, and it wrote that back as `AnaBo`. CSS
515
+ * collapses a newline and a space identically between inline boxes, so there is no
516
+ * predicate that separates the two cases and the rule had to go, not be refined.
517
+ *
518
+ * @param {HTMLElement} clone - the prepared save clone
519
+ * @param {WeakMap} map - live node -> loc, from pair()
520
+ * @param {Object} m - the model
521
+ * @param {Function} provenance - clone node -> live node
522
+ * @param {Object} [opts] - break switches, for tests that must exercise the fallback
523
+ */
524
+ /**
525
+ * Where in the file this live element is.
526
+ *
527
+ * The whole of what an agent editing loop needs beyond the save itself: point at an element in the
528
+ * page, get its range in the bytes that will be written, and express the edit as a source range
529
+ * rather than as a DOM mutation. No ids in the file, no map surviving a reload, no patch lists.
530
+ *
531
+ * OFFSETS ARE UTF-16 CODE UNITS into the same string `text()` returns, because that is the string
532
+ * the model was built from. They are NOT byte offsets, and on a document with any non-ASCII content
533
+ * the two differ: one line of `café 🎉 naïve` puts the same position at 51 code units and 55 UTF-8
534
+ * bytes. Slice the text this module hands you and the answer is exact; feed these numbers to a
535
+ * byte-oriented tool and it edits the wrong place, silently, and only on some documents. `column` is
536
+ * in the same units for the same reason.
537
+ *
538
+ * `null` rather than a guess, and the three reasons are different: a node the page created after
539
+ * boot is not in the file yet, an implied <html>, <head> or <body> the author never wrote has no
540
+ * bytes to point at, and a stale generation means the model was replaced since this map was built.
541
+ * A caller that treats all three as "not found" will mistake the third for the first.
542
+ */
543
+ export function locate(node, map, m) {
544
+ const loc = map.get(node);
545
+ if (!loc) return null; // never paired, so not in the file
546
+ if (loc.gen !== m.gen) return null; // a map and a model that were not paired together
547
+ if (loc.from < 0 || loc.to < 0) return null; // an implied tag has no bytes
548
+ // The bytes at this offset still have to BE this element. A map pointing into the wrong model
549
+ // slices one file's offsets out of another file's bytes, which is silent and lands in whatever
550
+ // the agent writes next, with no verifier between it and the file. Cheap, and it turns the one
551
+ // failure this function can have into a null instead of a wrong answer.
552
+ if (loc.kind === 'element' && loc.located) {
553
+ const head = m.src.slice(loc.openFrom, loc.openFrom + loc.tag.length + 1);
554
+ if (head.toLowerCase() !== '<' + loc.tag.toLowerCase()) return null;
555
+ }
556
+ const starts = lineStarts(m);
557
+ const line = upperBound(starts, loc.from);
558
+ return { from: loc.from, to: loc.to, line: line + 1, column: loc.from - starts[line] + 1 };
559
+ }
560
+
561
+ function lineStarts(m) {
562
+ if (!m.lineStarts) {
563
+ const starts = [0];
564
+ for (let i = 0; i < m.src.length; i++) if (m.src.charCodeAt(i) === 10) starts.push(i + 1);
565
+ m.lineStarts = starts;
566
+ }
567
+ return m.lineStarts;
568
+ }
569
+
570
+ /** The index of the last start at or before `at`. */
571
+ function upperBound(starts, at) {
572
+ let lo = 0;
573
+ let hi = starts.length - 1;
574
+ while (lo < hi) {
575
+ const mid = (lo + hi + 1) >> 1;
576
+ if (starts[mid] <= at) lo = mid;
577
+ else hi = mid - 1;
578
+ }
579
+ return lo;
580
+ }
581
+
582
+ export function render(clone, map, m, provenance, opts = {}) {
583
+ const now = () => (typeof performance !== 'undefined' ? performance.now() : Date.now());
584
+ const t0 = now();
585
+ const src = m.src;
586
+ const pieces = [];
587
+ // `from >= 0` is not defensive padding: `src.slice(-1, n)` is the LAST BYTE of the
588
+ // source, so an unset offset reaching here does not produce nothing, it produces one
589
+ // wrong byte and looks like a successful render.
590
+ const keep = (from, to) => { if (from >= 0 && to > from) pieces.push({ from, to }); };
591
+ const text = (t) => { if (t) pieces.push({ text: t }); };
592
+ // Elements in `opts.print`, and everything inside them, are printed rather than
593
+ // copied: the answer to a region whose copy did not verify.
594
+ let printing = 0;
595
+ const locOfClone = (n) => { if (printing) return null; const live = provenance(n); const l = live ? map.get(live) : null; return l && l.gen === m.gen ? l : null; };
596
+ // Whitespace written after </body> or </html> is not where its bytes are. The parser
597
+ // appends it to the last text node in <body> (the model's `tailOwner`), and every
598
+ // copy of the file writes it back after the end tags, which reparses to the same
599
+ // node. That only holds while the owner is still the last thing in <body>. Once
600
+ // something follows it (an appended element, a removed or printed owner), the same
601
+ // bytes reparse as a new text node at the end of <body>, so the render has to put the
602
+ // tail back on the owner and write nothing but tags and comments after </body>.
603
+ let tailMark = null; // where the owner's copied text ends, and the tail it left out
604
+ let afterBody = -1; // the first piece after body's end
605
+ let tailClean = false;
606
+ const rootLoc = m.root;
607
+ keep(0, rootLoc.openFrom); // the authored doctype and anything before <html>, verbatim
608
+ emitElement(clone, rootLoc);
609
+ keep(rootLoc.closeTo, src.length); // the trailing bytes, verbatim
610
+ if (m.tailOwner && afterBody >= 0 && !tailClean) {
611
+ for (let i = afterBody; i < pieces.length; i++) pieces[i] = { text: withoutWhitespace(piece(pieces[i])) };
612
+ if (tailMark && tailMark.tail) pieces.splice(tailMark.at, 0, { text: tailMark.tail });
613
+ }
614
+
615
+ const out = pieces.map(piece).join('');
616
+ return { text: out, ms: now() - t0 };
617
+
618
+ function piece(p) { return p.text !== undefined ? p.text : src.slice(p.from, p.to); }
619
+ function withoutWhitespace(s) {
620
+ return s.split(/(<!--[\s\S]*?-->|<[^>]*>)/).map((seg, i) => (i % 2 ? seg : seg.replace(/[\t\n\f\r ]+/g, ''))).join('');
621
+ }
622
+ // Called once <body>'s end tag, if any, has been emitted. The owner is still last when
623
+ // nothing but end tags was emitted after its text.
624
+ function endOfBody() {
625
+ afterBody = pieces.length;
626
+ tailClean = !!tailMark && tailMark.whole
627
+ && /^(?:<\/[^>]*>)*$/.test(pieces.slice(tailMark.at).map(piece).join(''));
628
+ }
629
+
630
+ function emitNode(n, parentTag) {
631
+ if (n.nodeType === 3) return emitText(n, parentTag);
632
+ if (n.nodeType === 8) return emitComment(n);
633
+ if (n.nodeType === 1) {
634
+ if (opts.print && opts.print.has(n)) {
635
+ printing++;
636
+ try { emitElement(n, null); } finally { printing--; }
637
+ } else {
638
+ emitElement(n, locOfClone(n));
639
+ }
640
+ if (n.localName === 'body' && n.parentNode === clone) endOfBody();
641
+ }
642
+ }
643
+ function emitComment(n) {
644
+ const loc = locOfClone(n);
645
+ if (loc && loc.kind === 'comment' && loc.value === n.data && loc.from >= 0 && !opts.breakText) keep(loc.from, loc.to);
646
+ else text('<!--' + n.data + '-->');
647
+ }
648
+ function emitText(n, parentTag) {
649
+ const loc = locOfClone(n);
650
+ let data = n.data;
651
+ if (opts.corrupt === 'drop-first-text' && !opts.corrupted && data.trim()) { opts.corrupted = true; data = data.slice(1); }
652
+ // The same bytes mean different things in different parents. A text node whose data
653
+ // is `<img>` is written `&lt;img&gt;` in a <div>, and raw in a <noscript>, whose
654
+ // content a scripting browser reads as raw text; copied into one from the other it
655
+ // says something else. A text node moved across that boundary is printed for its
656
+ // new parent, never copied from its old one.
657
+ const sameContext = loc && loc.parent && textContext(loc.parent.tag) === textContext(parentTag);
658
+ if (loc && loc.kind === 'text' && loc.value === data && loc.from >= 0 && !opts.breakText && sameContext) {
659
+ keep(loc.from, loc.to);
660
+ // Test switch: copy one named text node's bytes twice, a corruption only the
661
+ // copy path makes, so printing the element around it is the fix.
662
+ if (opts.doubleCopied !== undefined && data === opts.doubleCopied) keep(loc.from, loc.to);
663
+ if (loc.outsideTail) tailMark = { at: pieces.length, tail: loc.outsideTail, whole: true };
664
+ return;
665
+ }
666
+ let owner = null;
667
+ if (loc && loc.kind === 'text' && loc.outsideTail) {
668
+ const whole = data.endsWith(loc.outsideTail);
669
+ if (whole) data = data.slice(0, data.length - loc.outsideTail.length);
670
+ owner = { tail: whole ? loc.outsideTail : '', whole };
671
+ }
672
+ text(RAW_TEXT.has(parentTag) ? data : escText(data));
673
+ if (owner) tailMark = { at: pieces.length, ...owner };
674
+ }
675
+ function emitElement(n, loc) {
676
+ const tag = n.localName;
677
+ const isVoid = VOID.has(tag) && n.namespaceURI === XHTML;
678
+ const paired = loc && loc.kind === 'element' && !opts.breakTags ? loc : null;
679
+ if (paired && !paired.located) {
680
+ // An implied tag stays implied when the clone element carries nothing a tag would
681
+ // have to say. Otherwise it is printed, which is what the parser would have to
682
+ // imply anyway plus the attributes.
683
+ //
684
+ // `openFrom < 0` means the source has no bytes here at all, which is what an
685
+ // empty implied <head> looks like in a file that goes straight from <html> to
686
+ // <body>. Printing `<head></head>` there would be this module's own addition to
687
+ // a file nobody asked it to change, so an empty one emits nothing; one that has
688
+ // gained children emits the children and still no tag, and the parser implies it
689
+ // back on the next load.
690
+ if (n.attributes.length === 0) {
691
+ if (paired.openFrom >= 0) { emitChildren(n, paired, tag); return; }
692
+ if (liveKids(n).length) emitChildren(n, null, tag);
693
+ return;
694
+ }
695
+ text(printOpenTag(n));
696
+ if (isVoid) return;
697
+ emitChildren(n, paired.openFrom >= 0 ? paired : null, tag);
698
+ text('</' + tag + '>');
699
+ return;
700
+ }
701
+ if (!paired) {
702
+ text(printOpenTag(n));
703
+ if (isVoid) return;
704
+ emitChildren(n, null, tag);
705
+ text('</' + tag + '>');
706
+ return;
707
+ }
708
+ emitOpenTag(n, paired);
709
+ if (isVoid) return;
710
+ emitChildren(n, paired, tag);
711
+ keep(paired.closeFrom, paired.closeTo);
712
+ }
713
+ function printOpenTag(n) {
714
+ let s = '<' + n.localName;
715
+ for (const a of n.attributes) s += ' ' + a.name + (a.value === '' ? '' : '="' + escAttr(a.value) + '"');
716
+ return s + '>';
717
+ }
718
+ // Attribute by attribute, in the source's own order and its own spelling. One the
719
+ // clone still has with the same value copies its bytes; a changed one keeps the
720
+ // author's quoting; a removed one takes its leading whitespace with it; a new one is
721
+ // appended just before the `>`.
722
+ function emitOpenTag(n, loc) {
723
+ const live = new Map();
724
+ for (const a of n.attributes) live.set(a.name.toLowerCase(), a);
725
+ const seen = new Set();
726
+ let cursor = loc.openFrom;
727
+ for (const a of loc.attrs) {
728
+ if (a.from < 0) {
729
+ // No bytes in this tag: parse5 merged it in from a second <body> or <html> tag,
730
+ // where its bytes still are. Appending it here as well wrote it twice.
731
+ const cur = live.get(a.key);
732
+ if (cur && cur.value === a.value) seen.add(a.key);
733
+ continue;
734
+ }
735
+ const cur = live.get(a.key);
736
+ seen.add(a.key);
737
+ if (!cur) { keep(cursor, gapStart(cursor, a.from)); cursor = a.to; continue; }
738
+ if (cur.value === a.value) { keep(cursor, a.to); cursor = a.to; continue; }
739
+ keep(cursor, a.from);
740
+ text(respellAttr(src.slice(a.from, a.to), cur.value));
741
+ cursor = a.to;
742
+ }
743
+ // The `/` before `>` is the tag's own self-closing slash only when it sits outside
744
+ // every attribute. An unquoted value ends at whitespace or `>`, never at `/`, so
745
+ // `<a href=x/>` has the value `x/` and that slash is the value's last byte. Taken
746
+ // for the tag's, it was cut off the copied value and written again before `>`,
747
+ // giving `<a href=x//>`.
748
+ let closeAt = loc.openTo - 1;
749
+ const attrsEnd = loc.attrs.reduce((max, a) => (a.to > max ? a.to : max), loc.openFrom);
750
+ if (src[closeAt - 1] === '/' && closeAt - 1 >= attrsEnd) closeAt--;
751
+ keep(cursor, closeAt);
752
+ for (const a of n.attributes) {
753
+ if (seen.has(a.name.toLowerCase())) continue;
754
+ text(' ' + a.name + (a.value === '' ? '' : '="' + escAttr(a.value) + '"'));
755
+ }
756
+ keep(closeAt, loc.openTo);
757
+ }
758
+ function gapStart(cursor, attrFrom) {
759
+ let i = attrFrom;
760
+ while (i > cursor && /[ \t\n\r\f]/.test(src[i - 1])) i--;
761
+ return i;
762
+ }
763
+ function respellAttr(raw, value) {
764
+ const eq = raw.indexOf('=');
765
+ const name = (eq < 0 ? raw : raw.slice(0, eq)).trimEnd();
766
+ const rest = eq < 0 ? '' : raw.slice(eq + 1).trim();
767
+ const q = rest[0] === "'" ? "'" : rest[0] === '"' ? '"' : '';
768
+ const esc = value.replace(/&/g, '&amp;').replace(/ /g, '&nbsp;');
769
+ if (q === "'" && !value.includes("'")) return name + "='" + esc + "'";
770
+ if (q === '' && rest !== '' && value !== '' && !/[\s"'=<>`]/.test(value)) return name + '=' + esc;
771
+ return name + '="' + esc.replace(/"/g, '&quot;') + '"';
772
+ }
773
+ // Children, exact. Identity decides order: clone children whose source node belongs to
774
+ // this parent and form the longest in-order run stay in place and copy their bytes;
775
+ // every other clone child is emitted where the clone has it (printed, or copied out of
776
+ // place as a move); every source child the clone no longer has is dropped.
777
+ // An element the author left open (`<li>a`, a `<div>` missing its `</div>`) is closed
778
+ // by whatever the parser met next in the file. Copied without an end tag, it stays
779
+ // closed only while that same node still follows it. Anything else emitted after it
780
+ // (an appended element, a text node, a moved sibling) would parse INSIDE it, so it
781
+ // gets the end tag the file never had, written just before that next node.
782
+ function copiedOpen(c) {
783
+ if (c.nodeType !== 1 || printing || (opts.print && opts.print.has(c))) return null;
784
+ const l = locOfClone(c);
785
+ return l && l.kind === 'element' && l.located && !l.endTagged && !VOID.has(l.tag) && !opts.breakTags ? l : null;
786
+ }
787
+ // The end tags a copied-open element needs, innermost first: its own, and those of the
788
+ // open elements it ends with, since one end tag does not close them all. `</b>` inside
789
+ // `<b><b>` closes only the inner one.
790
+ function openChain(c) {
791
+ const l = copiedOpen(c);
792
+ if (!l) return null;
793
+ const kids = liveKids(c);
794
+ const inner = kids.length ? openChain(kids[kids.length - 1]) : null;
795
+ return (inner || []).concat(l.tag);
796
+ }
797
+ function emitChildren(el, loc, parentTag) {
798
+ const C = liveKids(el);
799
+ let open = null; // { tags, j }: the last child's end tags if it was copied open, and its source index
800
+ const closeOpen = (i, j) => {
801
+ if (open && !(j >= 0 && open.j >= 0 && j === open.j + 1)) text(open.tags.map((t) => '</' + t + '>').join(''));
802
+ open = null;
803
+ };
804
+ if (!loc) {
805
+ for (const c of C) { closeOpen(-1, -1); emitNode(c, parentTag); const tags = openChain(c); if (tags) open = { tags, j: -1 }; }
806
+ return;
807
+ }
808
+ const S = loc.children;
809
+ const sIndex = new Map(S.map((s, j) => [s, j]));
810
+ const owned = C.map((c) => { const l = locOfClone(c); return l && l.parent === loc && l.from >= 0 ? sIndex.get(l) : -1; });
811
+ // A child with no identity at all is usually not new. A [freeze] restore, a
812
+ // [persist] textarea, an innerHTML rebuild after pairing: each hands the save clone
813
+ // fresh nodes for content the file already holds. Matched here, by exact subtree
814
+ // signature against this parent's source children nothing else claimed, in order,
815
+ // and copied whole. Exact means the bytes say what the node says; verify still checks.
816
+ const whole = new Set();
817
+ if (!opts.breakText && !opts.breakTags) {
818
+ const taken = new Set(owned);
819
+ const free = new Map();
820
+ S.forEach((s, j) => {
821
+ if (taken.has(j) || s.from < 0 || !s.keys) return;
822
+ if (!free.has(s.keys.deep)) free.set(s.keys.deep, []);
823
+ free.get(s.keys.deep).push(j);
824
+ });
825
+ if (free.size) C.forEach((c, i) => {
826
+ if (owned[i] >= 0 || locOfClone(c)) return;
827
+ const q = free.get(keysOfLive(c, new Map()).deep);
828
+ if (q && q.length) { owned[i] = q.shift(); whole.add(i); }
829
+ });
830
+ }
831
+ const emitChild = (c, i) => {
832
+ if (!whole.has(i)) return emitNode(c, parentTag);
833
+ const s = S[owned[i]];
834
+ keep(s.from, s.to);
835
+ if (s.outsideTail) tailMark = { at: pieces.length, tail: s.outsideTail, whole: true };
836
+ };
837
+ const openOf = (c, i) => {
838
+ if (!whole.has(i)) return openChain(c);
839
+ const s = S[owned[i]];
840
+ return s.kind === 'element' && s.located && !s.endTagged && !VOID.has(s.tag) ? [s.tag] : null;
841
+ };
842
+ const inPlace = new Set(lis(owned));
843
+ let cursor = loc.openTo;
844
+ let lastS = -1;
845
+ const dropTo = (j) => { for (let k = lastS + 1; k < j; k++) { const sk = S[k]; if (sk.from < 0) continue; keep(cursor, sk.from); cursor = Math.max(cursor, sk.to); } };
846
+ for (let i = 0; i < C.length; i++) {
847
+ const c = C[i];
848
+ const tags = openOf(c, i);
849
+ if (inPlace.has(i)) {
850
+ const j = owned[i], s = S[j];
851
+ closeOpen(i, j);
852
+ dropTo(j);
853
+ keep(cursor, s.from);
854
+ emitChild(c, i);
855
+ cursor = Math.max(cursor, s.to); lastS = j;
856
+ open = tags ? { tags, j } : null;
857
+ continue;
858
+ }
859
+ // An original follower the parser implied rather than read (a <p> made by a stray
860
+ // </p>) has no bytes of its own, so it is not "in place", but it still follows.
861
+ const lc = locOfClone(c);
862
+ closeOpen(i, lc && lc.parent === loc ? sIndex.get(lc) : -1);
863
+ emitChild(c, i);
864
+ open = tags ? { tags, j: -1 } : null;
865
+ }
866
+ dropTo(S.length);
867
+ keep(cursor, loc.closeFrom);
868
+ }
869
+ }
870
+
871
+ function lis(vals) {
872
+ const tails = [], tailIdx = [], prev = new Array(vals.length).fill(-1);
873
+ for (let i = 0; i < vals.length; i++) {
874
+ const v = vals[i]; if (v < 0) continue;
875
+ let lo = 0, hi = tails.length;
876
+ while (lo < hi) { const mid = (lo + hi) >> 1; if (tails[mid] < v) lo = mid + 1; else hi = mid; }
877
+ tails[lo] = v; tailIdx[lo] = i; prev[i] = lo > 0 ? tailIdx[lo - 1] : -1;
878
+ }
879
+ const out = []; let i = tailIdx.length ? tailIdx[tailIdx.length - 1] : -1;
880
+ while (i >= 0) { out.push(i); i = prev[i]; }
881
+ return out.reverse();
882
+ }
883
+
884
+ // =============================================================================
885
+ // VERIFY
886
+ // =============================================================================
887
+ // Independent of render on purpose: its own child enumeration, its own attribute key,
888
+ // no notion of layout or whitespace, no shared helper. A check built out of the thing
889
+ // it is checking cannot fail in the cases it exists to catch.
890
+ //
891
+ // Three checks, all exact:
892
+ // 1. the rendered document's nodes outside <html> are the live document's
893
+ // 2. the rendered <html> subtree equals today's serialization reparsed, node for node
894
+ // 3. or, failing that, equals the save clone itself
895
+ //
896
+ // Either oracle suffices, and they are not the same claim. "Equal to today's bytes" is
897
+ // the floor: no worse than the save that would otherwise have gone out. "Equal to the
898
+ // clone" is stronger, and is sometimes the only one available, because today's
899
+ // serializer is not itself exact: it drops the newline that starts a <pre>, so on a
900
+ // document with one, today's bytes reload as a different document and these reload as
901
+ // the right one.
902
+
903
+ export function verify(rendered, today, liveDocument, clone = null, sourceErrors = null) {
904
+ const P = new DOMParser();
905
+ const A = P.parseFromString(rendered, 'text/html');
906
+ const B = P.parseFromString(today, 'text/html');
907
+ const outsideA = vOutside(A), outsideLive = vOutside(liveDocument);
908
+ if (outsideA.join('\n') !== outsideLive.join('\n')) return { ok: false, diff: 'document: outside <html> ' + JSON.stringify(outsideA) + ' vs live ' + JSON.stringify(outsideLive) };
909
+ const diff = vDiff(A.documentElement, B.documentElement, 'html');
910
+ const worse = diff ? null : vParseErrors(rendered, sourceErrors);
911
+ if (!diff) return worse ? { ok: false, diff: worse } : { ok: true, diff: null, oracle: 'today' };
912
+ if (clone) {
913
+ // `at` is the index path, in the clone, of the node where the render first differs,
914
+ // so a caller can print that element and try again instead of discarding the whole
915
+ // render. It is valid in the clone because every earlier sibling at every level on
916
+ // the way down compared equal.
917
+ const at = [];
918
+ if (!vDiff(A.documentElement, clone, 'html', at)) {
919
+ const w = vParseErrors(rendered, sourceErrors);
920
+ return w ? { ok: false, diff: w } : { ok: true, diff: null, oracle: 'clone', todayDiff: diff };
921
+ }
922
+ return { ok: false, diff, at };
923
+ }
924
+ return { ok: false, diff };
925
+ }
926
+
927
+ /** The node at an index path from verify's `at`, walking the same children verify walked. */
928
+ export function nodeAt(root, at) {
929
+ let n = root;
930
+ for (const i of at) {
931
+ const kids = vKids(n);
932
+ if (i >= kids.length) return null;
933
+ n = kids[i];
934
+ }
935
+ return n;
936
+ }
937
+
938
+ /**
939
+ * The check the tree comparison cannot make.
940
+ *
941
+ * A parser DISCARDS some input rather than representing it, so the tree is the same
942
+ * whether or not it was there. A second copy of an attribute is the case that bit:
943
+ * the render wrote `xmlns:xlink` twice, the parser kept the first and dropped the
944
+ * second, and the trees matched. Counting, not presence, because 4 of 198 real user
945
+ * documents already parse with errors: only an INCREASE is the render's doing, and a
946
+ * full serialization cannot produce one, because the DOM it comes from cannot hold a
947
+ * duplicate attribute in the first place.
948
+ *
949
+ * This is narrower than it sounds and worth saying plainly. It sees what parse5
950
+ * reports, which is tokenizer errors such as `duplicate-attribute`. Input the tree
951
+ * builder discards is not reported at all: a stray end tag, a second `<body>` tag's
952
+ * attributes, a stray doctype. A render that wrote one of those would reparse to the
953
+ * same tree with no new error, and nothing here would see it. Bytes the parser DOES
954
+ * represent, such as a duplicated element, change the tree, and the comparison above
955
+ * is what catches those.
956
+ */
957
+ function vParseErrors(rendered, sourceErrors) {
958
+ if (!sourceErrors) return null;
959
+ const seen = new Map();
960
+ parse(rendered, { onParseError: (e) => seen.set(e.code, (seen.get(e.code) || 0) + 1) });
961
+ for (const [code, n] of seen) {
962
+ const was = sourceErrors.get(code) || 0;
963
+ if (n > was) return `the render introduced parse errors the source does not have: ${code} ${was} -> ${n}`;
964
+ }
965
+ return null;
966
+ }
967
+ function vOutside(doc) {
968
+ const out = [];
969
+ for (const n of doc.childNodes) {
970
+ if (n.nodeType === 10) out.push('doctype ' + n.name + '|' + n.publicId + '|' + n.systemId);
971
+ else if (n.nodeType === 8) out.push('comment ' + n.data);
972
+ else if (n.nodeType !== 1) out.push('node' + n.nodeType + ' ' + (n.data || ''));
973
+ }
974
+ return out;
975
+ }
976
+ /**
977
+ * <noscript> is the one element whose children depend on a flag neither side of this
978
+ * comparison controls. A page the browser loaded was parsed with scripting ENABLED, so
979
+ * the live tree holds the block's markup as a single text node. DOMParser always parses
980
+ * with scripting DISABLED, so re-reading the rendered bytes turns that same markup back
981
+ * into elements. The two trees can never agree, and without this a document containing
982
+ * one <noscript> would fail verification on every save, forever, which is worse than a
983
+ * wrong answer: it is a permanent false alarm on the counter that stage 2 reads.
984
+ *
985
+ * So both sides hand their content to ONE parser and the resulting trees are compared.
986
+ * That is exact rather than tolerant: anything the renderer corrupted inside the block
987
+ * still changes the tree its markup parses to. Recursion terminates because each step
988
+ * compares strictly less markup than the one above it.
989
+ */
990
+ function vNoscript(a, b, path) {
991
+ const P = new DOMParser();
992
+ // Read each side the way it was parsed, which verify fixes, not the node. `a` is always
993
+ // the render reparsed by DOMParser, with scripting off, so its <noscript> text is TEXT,
994
+ // and reading its `.data` as markup would turn `&lt;img&gt;` into an <img> and pass a
995
+ // render that wrote escaped text where the page has an image. `b` always comes from the
996
+ // page, parsed with scripting on, so a <noscript> holding only text holds its markup
997
+ // raw. Asking the node's own document gets `b` wrong: Chrome builds the save clone in a
998
+ // document with no window.
999
+ const raw = (el) => {
1000
+ const kids = Array.from(el.childNodes);
1001
+ return kids.length && kids.every((n) => n.nodeType === 3) ? kids.map((n) => n.data).join('') : el.innerHTML;
1002
+ };
1003
+ return vDiff(P.parseFromString(a.innerHTML, 'text/html').body, P.parseFromString(raw(b), 'text/html').body, path + '>#noscript');
1004
+ }
1005
+ function vKids(el) {
1006
+ const list = el.localName === 'template' && el.content ? el.content.childNodes : el.childNodes;
1007
+ return Array.from(list).filter((n) => n.nodeType === 1 || n.nodeType === 3 || n.nodeType === 8);
1008
+ }
1009
+ const vAttrs = (el) => Array.from(el.attributes, (a) => (a.namespaceURI || '') + ' ' + a.name + '=' + JSON.stringify(a.value)).sort().join('; ');
1010
+ function vDiff(a, b, path, at = []) {
1011
+ if (a.nodeType !== b.nodeType) return path + ': node type ' + a.nodeType + ' vs ' + b.nodeType;
1012
+ if (a.nodeType === 3 || a.nodeType === 8) return a.data === b.data ? null : path + ': text ' + JSON.stringify(a.data.slice(0, 60)) + ' vs ' + JSON.stringify(b.data.slice(0, 60));
1013
+ if (a.localName !== b.localName || a.namespaceURI !== b.namespaceURI) return path + ': tag ' + a.localName + ' vs ' + b.localName;
1014
+ if (vAttrs(a) !== vAttrs(b)) return path + ': attrs ' + vAttrs(a) + ' vs ' + vAttrs(b);
1015
+ if (a.localName === 'noscript' && a.namespaceURI === XHTML) return vNoscript(a, b, path);
1016
+ const ka = vKids(a), kb = vKids(b);
1017
+ if (ka.length !== kb.length) return path + ': ' + ka.length + ' vs ' + kb.length + ' children';
1018
+ for (let i = 0; i < ka.length; i++) {
1019
+ at.push(i);
1020
+ const d = vDiff(ka[i], kb[i], path + '>' + (ka[i].localName || '#') + '[' + i + ']', at);
1021
+ if (d) return d;
1022
+ at.pop();
1023
+ }
1024
+ return null;
1025
+ }