@panphora/clayjs 1.2.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -1
- package/THIRD-PARTY-NOTICES.md +10 -0
- package/dist/clay.standalone.js +21437 -13925
- package/entries/clay-data.js +1 -1
- package/package.json +7 -2
- package/packed-contract.json +17 -0
- package/src/attrs/save-freeze.js +10 -18
- package/src/core/admin-contenteditable.js +5 -6
- package/src/core/persist.js +5 -10
- package/src/core/save-core.js +35 -0
- package/src/core/snapshot.js +143 -14
- package/src/core/source-map.js +817 -0
- package/src/core/unsaved-warning.js +3 -0
- package/src/dom/dom-helpers.js +5 -1
- package/src/lib/content-dom.js +108 -0
- package/src/lib/mutation.js +26 -3
- package/src/lib/region-capabilities.js +69 -0
- package/src/lib/region-policy.js +18 -13
- package/src/loader-logic.js +20 -4
- package/src/loader.js +4 -0
- package/src/plugins/demo.js +3 -0
- package/src/plugins/sortable.js +6 -1
- package/src/plugins/source.js +326 -0
- package/src/plugins/wire.js +248 -47
- package/src/sync/live-sync.js +103 -42
- package/src/sync/presence.js +303 -0
- package/src/sync/section-notice.js +230 -0
- package/src/sync/splice-merge.js +7 -10
- package/src/sync/stream.js +190 -0
- package/src/vendor/hyper-morph.vendor.js +2 -2
- package/src/vendor/hyper-undo.vendor.js +1 -1
- package/src/vendor/hypercms.vendor.js +438 -45
- package/src/vendor/parse5.vendor.js +3 -0
- package/src/vendor/quickcrop.vendor.js +1 -1
- package/src/vendor/richclay.vendor.js +22 -15
|
@@ -0,0 +1,817 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* source-map.js — save the file, not a serialization of it.
|
|
3
|
+
*
|
|
4
|
+
* Every program that edits a malleable HTML file rewrites the whole file, because
|
|
5
|
+
* each one parses to a tree and serializes the tree back out, and a serializer does
|
|
6
|
+
* not reproduce its input. A no-edit save through `clone.outerHTML` rewrites about
|
|
7
|
+
* 88% of an authored document's lines: attributes reorder, quoting normalises, `&`
|
|
8
|
+
* becomes `&`, every tag reprints canonically. Nothing is lost, and the file is
|
|
9
|
+
* no longer the file anybody wrote.
|
|
10
|
+
*
|
|
11
|
+
* This module keeps the bytes the document was loaded from, pairs each live node to
|
|
12
|
+
* a byte range in them, and on save copies source bytes for everything unchanged and
|
|
13
|
+
* prints only what changed.
|
|
14
|
+
*
|
|
15
|
+
* FIVE OPERATIONS
|
|
16
|
+
*
|
|
17
|
+
* model(src) parse5 tree of byte locations; implied tags get a content range
|
|
18
|
+
* pair(root, m) live node -> byte range, by tiered signature alignment
|
|
19
|
+
* render(clone, ...) the save clone, emitted exactly: every child, whitespace
|
|
20
|
+
* included, no gaps invented and none dropped
|
|
21
|
+
* verify(out, today) reparse and compare trees before anything is sent
|
|
22
|
+
* adopt(bytes) re-model and re-pair against what the host accepted
|
|
23
|
+
*
|
|
24
|
+
* THE FLOOR IS TODAY'S BEHAVIOUR. Anything that does not verify is sent as the full
|
|
25
|
+
* serialization instead, and counted. A fetch that does not return this document, a
|
|
26
|
+
* source whose shape outside <html> disagrees with the live document, a render that
|
|
27
|
+
* throws: each one means this module does nothing at all and the save is exactly the
|
|
28
|
+
* save it would have been.
|
|
29
|
+
*
|
|
30
|
+
* WHAT IS DELIBERATELY NOT HERE
|
|
31
|
+
*
|
|
32
|
+
* No ids in the file. Identity is the live node object, held in a WeakMap.
|
|
33
|
+
* No sidecar, no server change, no second format on disk.
|
|
34
|
+
* No tolerance in the verifier. A reprint is fixed in the renderer or in the page,
|
|
35
|
+
* never by widening what counts as equal — a verifier that shares a predicate with
|
|
36
|
+
* the renderer cannot see the renderer's mistakes, which is how an earlier version
|
|
37
|
+
* of this code wrote `by <a>Ana</a> <a>Bo</a>` back as `AnaBo` with a green check.
|
|
38
|
+
*/
|
|
39
|
+
|
|
40
|
+
import { parse } from '../vendor/parse5.vendor.js';
|
|
41
|
+
|
|
42
|
+
const RAW_TEXT = new Set(['script', 'style', 'xmp', 'iframe', 'noembed', 'noframes', 'plaintext', 'noscript']);
|
|
43
|
+
const VOID = new Set(['area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr']);
|
|
44
|
+
const XHTML = 'http://www.w3.org/1999/xhtml';
|
|
45
|
+
|
|
46
|
+
const escText = (s) => s.replace(/&/g, '&').replace(/ /g, ' ').replace(/</g, '<').replace(/>/g, '>');
|
|
47
|
+
const escAttr = (s) => s.replace(/&/g, '&').replace(/ /g, ' ').replace(/"/g, '"');
|
|
48
|
+
|
|
49
|
+
// =============================================================================
|
|
50
|
+
// MODEL
|
|
51
|
+
// =============================================================================
|
|
52
|
+
|
|
53
|
+
export function model(src) {
|
|
54
|
+
// The error counts come out of the SAME parse, so they cost nothing here. They are
|
|
55
|
+
// the only evidence of input a parser throws away, which a tree comparison cannot
|
|
56
|
+
// see by construction: a duplicate attribute is discarded during parsing, so a render
|
|
57
|
+
// that wrote one twice reparsed to exactly the right tree. That is how the same
|
|
58
|
+
// `xmlns:xlink` was added to a file on every save with the verifier green each time.
|
|
59
|
+
const parseErrors = new Map();
|
|
60
|
+
const doc = parse(src, {
|
|
61
|
+
sourceCodeLocationInfo: true,
|
|
62
|
+
onParseError: (e) => parseErrors.set(e.code, (parseErrors.get(e.code) || 0) + 1),
|
|
63
|
+
});
|
|
64
|
+
const html = doc.childNodes.find((n) => n.nodeName === 'html');
|
|
65
|
+
const outside = doc.childNodes.filter((n) => n !== html).map((n) => n.nodeName === '#documentType'
|
|
66
|
+
? { kind: 'doctype', name: n.name || '', publicId: n.publicId || '', systemId: n.systemId || '' }
|
|
67
|
+
: { kind: n.nodeName === '#comment' ? 'comment' : n.nodeName, value: n.data });
|
|
68
|
+
const root = html ? locOf(html, null, src) : null;
|
|
69
|
+
return { src, doc, outside, root, parseErrors, relocated: root ? firstOutOfOrder(root) : null };
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function locOf(n, parent, src) {
|
|
73
|
+
const l = n.sourceCodeLocation || null;
|
|
74
|
+
const kind = n.nodeName === '#text' ? 'text' : n.nodeName === '#comment' ? 'comment' : 'element';
|
|
75
|
+
const loc = { node: n, parent, live: null, kind, children: [], from: l ? l.startOffset : -1, to: l ? l.endOffset : -1 };
|
|
76
|
+
if (kind === 'text') { loc.value = n.value; return loc; }
|
|
77
|
+
if (kind === 'comment') { loc.value = n.data; return loc; }
|
|
78
|
+
loc.tag = n.tagName;
|
|
79
|
+
loc.located = !!(l && l.startTag);
|
|
80
|
+
loc.openFrom = loc.located ? l.startTag.startOffset : -1;
|
|
81
|
+
loc.openTo = loc.located ? l.startTag.endOffset : -1;
|
|
82
|
+
loc.closeFrom = l && l.endTag ? l.endTag.startOffset : (loc.located ? loc.to : -1);
|
|
83
|
+
loc.closeTo = l && l.endTag ? l.endTag.endOffset : (loc.located ? loc.to : -1);
|
|
84
|
+
// Nothing can end before its own start tag does, but parse5 reports exactly that for
|
|
85
|
+
// an element the parser inserts and immediately pops. A <form> written directly
|
|
86
|
+
// inside a <table> is the case that ships: the form is inserted, the form pointer is
|
|
87
|
+
// set, and the element is popped off the stack at once, so nothing ever closes it and
|
|
88
|
+
// its endOffset stays at the `<` of its start tag. The parent then rewound its copy
|
|
89
|
+
// cursor to that offset after emitting the tag and copied the whole `<form ...>`
|
|
90
|
+
// again, so every save added one more copy and the tree never changed, which is why
|
|
91
|
+
// verify passed each time. Both ends move up to the end of the start tag, which is
|
|
92
|
+
// the least this element can be said to occupy.
|
|
93
|
+
if (loc.located) {
|
|
94
|
+
if (loc.to < loc.openTo) loc.to = loc.openTo;
|
|
95
|
+
if (loc.closeFrom < loc.openTo) loc.closeFrom = loc.closeTo = loc.openTo;
|
|
96
|
+
}
|
|
97
|
+
loc.attrs = [];
|
|
98
|
+
const attrLocs = (loc.located && l.startTag.attrs) || (l && l.attrs) || {};
|
|
99
|
+
// Keyed by the name as WRITTEN, which is the only name both sides agree on. parse5
|
|
100
|
+
// adjusts a foreign-content attribute in the tree, so `xmlns:xlink` arrives as
|
|
101
|
+
// `{ name: 'xlink', prefix: 'xmlns' }`, while it keys the location map and the live
|
|
102
|
+
// DOM's `attr.name` by `xmlns:xlink`. Keyed by the adjusted name the lookup missed,
|
|
103
|
+
// the attribute got no source range, its bytes rode along inside a copied run, and
|
|
104
|
+
// the live copy was appended as a new attribute: every save added another
|
|
105
|
+
// `xmlns:xlink="..."` and the tree stayed identical, so verify passed each time.
|
|
106
|
+
for (const a of n.attrs) {
|
|
107
|
+
const written = a.prefix ? a.prefix + ':' + a.name : a.name;
|
|
108
|
+
const al = attrLocs[written.toLowerCase()];
|
|
109
|
+
loc.attrs.push({ name: written, key: written.toLowerCase(), value: a.value, from: al ? al.startOffset : -1, to: al ? al.endOffset : -1 });
|
|
110
|
+
}
|
|
111
|
+
const kids = n.nodeName === 'template' && n.content ? n.content.childNodes : n.childNodes;
|
|
112
|
+
for (const c of kids) {
|
|
113
|
+
if (c.nodeName === '#documentType') continue;
|
|
114
|
+
loc.children.push(locOf(c, loc, src));
|
|
115
|
+
}
|
|
116
|
+
// An implied tag (<html>, <head>, <body> the author never wrote) has no tag bytes. Its
|
|
117
|
+
// content range is its children's span, and its open and close tags are empty ranges at
|
|
118
|
+
// either end, so emitting it copies the children and writes no tag.
|
|
119
|
+
if (!loc.located) {
|
|
120
|
+
const located = loc.children.filter((c) => c.from >= 0);
|
|
121
|
+
const first = located.length ? Math.min(...located.map((c) => c.from)) : -1;
|
|
122
|
+
const last = located.length ? Math.max(...located.map((c) => c.to)) : -1;
|
|
123
|
+
loc.openFrom = loc.openTo = first;
|
|
124
|
+
loc.closeFrom = loc.closeTo = last;
|
|
125
|
+
loc.from = first; loc.to = last;
|
|
126
|
+
}
|
|
127
|
+
for (const cl of loc.children) {
|
|
128
|
+
// Clamp every child into its parent's content range, at BOTH ends.
|
|
129
|
+
//
|
|
130
|
+
// The spec (and parse5, and every browser) appends whitespace that follows an end
|
|
131
|
+
// tag to the last text node before it, so such a node's range runs past the tag,
|
|
132
|
+
// and when nothing precedes the tag the range starts past it as well. Left alone,
|
|
133
|
+
// the first case emits the same bytes twice and the second inverts the range:
|
|
134
|
+
// `</script></body>\n</html>` gave body a last text node beginning AFTER `</body>`,
|
|
135
|
+
// and the copy of the gap before it wrote `</body>` a second time. Chromium found
|
|
136
|
+
// that one; jsdom did not, only because every hand-written fixture happened to have
|
|
137
|
+
// a newline before `</body>`.
|
|
138
|
+
if (cl.from >= 0 && loc.located) {
|
|
139
|
+
// Both ends land inside [openTo, closeFrom], and `from` never passes `to`: a
|
|
140
|
+
// range that begins after the end tag collapses to an empty one AT the end tag,
|
|
141
|
+
// not after it. Putting it after is what wrote `</body>` twice, because the copy
|
|
142
|
+
// of the gap up to that range then spanned the tag.
|
|
143
|
+
const to = Math.max(loc.openTo, Math.min(cl.to, loc.closeFrom));
|
|
144
|
+
const from = Math.min(Math.max(cl.from, loc.openTo), to);
|
|
145
|
+
if (from !== cl.from || to !== cl.to) {
|
|
146
|
+
// What the bytes inside the parent account for stays with this node; the rest
|
|
147
|
+
// was relocated from outside, belongs to an ancestor, and is emitted there. A
|
|
148
|
+
// node that gets PRINTED rather than copied has to leave it out or it appears
|
|
149
|
+
// twice, which is a tree difference and a fallback on the next save.
|
|
150
|
+
//
|
|
151
|
+
// Neither end of the span can be measured directly. Counting back by the
|
|
152
|
+
// overhang's byte length is wrong because the span covers the end tag too, and
|
|
153
|
+
// slicing the source after the end tag is wrong because the relocation can
|
|
154
|
+
// cross two of them: `</p>\n</body>\n</html>\n` puts three newlines from three
|
|
155
|
+
// different places into one text node.
|
|
156
|
+
//
|
|
157
|
+
// Whitespace only, which is all a parser relocates.
|
|
158
|
+
if (cl.kind === 'text') {
|
|
159
|
+
// Normalized, because the node's value is. The parser rewrites every CRLF and
|
|
160
|
+
// every lone CR in the input stream to LF before a text node ever sees them,
|
|
161
|
+
// so on a CRLF file the raw bytes and the value can never match and this
|
|
162
|
+
// comparison silently found no overhang at all. The tail then printed on top
|
|
163
|
+
// of the bytes it had been relocated from and the file grew a blank line per
|
|
164
|
+
// save. Only the LENGTH taken from here is normalized; the copy path still
|
|
165
|
+
// uses the raw bytes, which is why an unedited CRLF file round trips exactly.
|
|
166
|
+
const inside = src.slice(from, to).replace(/\r\n?/g, '\n');
|
|
167
|
+
if (cl.value.startsWith(inside)) {
|
|
168
|
+
const outside = cl.value.slice(inside.length);
|
|
169
|
+
if (outside && !/\S/.test(outside)) cl.outsideTail = outside;
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
cl.clamped = true;
|
|
173
|
+
cl.from = from;
|
|
174
|
+
cl.to = to;
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
// Every copy below walks the source forward, so the children have to be in source
|
|
179
|
+
// order. Foster parenting is where they are not: content written inside a <table>
|
|
180
|
+
// that does not belong there is moved OUT, to just before the table, so the tree
|
|
181
|
+
// order and the byte order disagree and a forward walk emits those bytes at the new
|
|
182
|
+
// position and again inside the table. The bytes belong to one element and the node
|
|
183
|
+
// to another, which this model has no way to say, so the document is refused instead
|
|
184
|
+
// and saves it with today's serializer. That is what such a page already got: the
|
|
185
|
+
// render was wrong, verify caught it, and every save fell back.
|
|
186
|
+
let floor = loc.located ? loc.openTo : -1;
|
|
187
|
+
for (const cl of loc.children) {
|
|
188
|
+
if (cl.from < 0) continue;
|
|
189
|
+
if (cl.from < floor) { loc.outOfOrder = true; break; }
|
|
190
|
+
floor = cl.to;
|
|
191
|
+
}
|
|
192
|
+
return loc;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
function firstOutOfOrder(loc) {
|
|
196
|
+
if (loc.outOfOrder) return loc.tag;
|
|
197
|
+
for (const c of loc.children) {
|
|
198
|
+
if (c.kind !== 'element') continue;
|
|
199
|
+
const found = firstOutOfOrder(c);
|
|
200
|
+
if (found) return found;
|
|
201
|
+
}
|
|
202
|
+
return null;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/**
|
|
206
|
+
* Refuse a model that is not this document.
|
|
207
|
+
*
|
|
208
|
+
* The live document parsed the bytes that were actually served, so its doctype and
|
|
209
|
+
* its document-level comments are an oracle for everything render copies from
|
|
210
|
+
* outside <html> — the one region no tree comparison can check, because a tree
|
|
211
|
+
* comparison starts at documentElement. A boot fetch that returned a login page or
|
|
212
|
+
* an error page disagrees here, and that is the difference between doing nothing and
|
|
213
|
+
* writing that page's doctype into somebody's file.
|
|
214
|
+
*
|
|
215
|
+
* @returns {?string} the disagreement, or null when the source is this document
|
|
216
|
+
*/
|
|
217
|
+
export function checkSource(m, liveDocument) {
|
|
218
|
+
if (!m.root) return 'source has no <html> element';
|
|
219
|
+
// Nothing inside <html> carries a byte range, which is what a document with no
|
|
220
|
+
// elements, text or comments at all looks like. render brackets the whole output
|
|
221
|
+
// with the root's range, so there is nothing here to preserve and nothing to
|
|
222
|
+
// bracket with.
|
|
223
|
+
if (m.root.openFrom < 0) return 'source has no content inside <html>';
|
|
224
|
+
if (m.relocated) return 'the parser moved content out of <' + m.relocated + '>, so source order and tree order disagree';
|
|
225
|
+
const want = outsideOf(liveDocument);
|
|
226
|
+
const got = m.outside.map(outsideKey);
|
|
227
|
+
if (want.length !== got.length) return 'outside <html>: ' + JSON.stringify(got) + ' vs live ' + JSON.stringify(want);
|
|
228
|
+
for (let i = 0; i < want.length; i++) if (want[i] !== got[i]) return 'outside <html>: ' + got[i] + ' vs live ' + want[i];
|
|
229
|
+
return null;
|
|
230
|
+
}
|
|
231
|
+
function outsideKey(o) { return o.kind === 'doctype' ? 'doctype:' + o.name + '|' + o.publicId + '|' + o.systemId : o.kind + ':' + o.value; }
|
|
232
|
+
function outsideOf(doc) {
|
|
233
|
+
const out = [];
|
|
234
|
+
for (const n of doc.childNodes) {
|
|
235
|
+
if (n.nodeType === 10) out.push('doctype:' + n.name + '|' + n.publicId + '|' + n.systemId);
|
|
236
|
+
else if (n.nodeType === 8) out.push('comment:' + n.data);
|
|
237
|
+
else if (n.nodeType !== 1) out.push('#' + n.nodeType + ':' + (n.data || ''));
|
|
238
|
+
}
|
|
239
|
+
return out;
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
// =============================================================================
|
|
243
|
+
// SIGNATURES
|
|
244
|
+
// =============================================================================
|
|
245
|
+
// Four keys per node, computed once, bottom-up. `deep` is the whole subtree (tag,
|
|
246
|
+
// attributes, text), `content` ignores attributes at every level, `shallow` is this
|
|
247
|
+
// node's tag and attributes only, `tag` is the tag. Alignment tries them strongest
|
|
248
|
+
// first, so a runtime attribute change deep inside a card still lets the card pair by
|
|
249
|
+
// its text, and two genuinely identical rows pair by position.
|
|
250
|
+
|
|
251
|
+
const UNIT_SEP = '\u0000';
|
|
252
|
+
|
|
253
|
+
function mix(h, s) {
|
|
254
|
+
for (let i = 0; i < s.length; i++) { h ^= s.charCodeAt(i); h = Math.imul(h, 16777619); }
|
|
255
|
+
return h >>> 0;
|
|
256
|
+
}
|
|
257
|
+
function hashKey(parts) {
|
|
258
|
+
let a = 2166136261, b = 5381;
|
|
259
|
+
for (const p of parts) { a = mix(a, p); b = (Math.imul(b, 33) ^ mix(a ^ 0x9e3779b9, p)) >>> 0; }
|
|
260
|
+
return a.toString(36) + '.' + b.toString(36);
|
|
261
|
+
}
|
|
262
|
+
function liveKids(el) {
|
|
263
|
+
const list = el.nodeType === 1 && el.tagName === 'TEMPLATE' && el.content ? el.content.childNodes : el.childNodes;
|
|
264
|
+
const out = [];
|
|
265
|
+
for (const n of list) if (n.nodeType === 1 || n.nodeType === 3 || n.nodeType === 8) out.push(n);
|
|
266
|
+
return out;
|
|
267
|
+
}
|
|
268
|
+
const liveAttrsKey = (el) => Array.from(el.attributes, (a) => a.name.toLowerCase() + '=' + a.value).sort().join(UNIT_SEP);
|
|
269
|
+
const srcAttrsKey = (loc) => loc.attrs.map((a) => a.key + '=' + a.value).sort().join(UNIT_SEP);
|
|
270
|
+
function keysOfLive(n, keyMap) {
|
|
271
|
+
let k;
|
|
272
|
+
if (n.nodeType === 3) k = { tag: '#text', shallow: 't:' + n.data, content: 't:' + n.data, deep: 't:' + n.data };
|
|
273
|
+
else if (n.nodeType === 8) k = { tag: '#comment', shallow: 'c:' + n.data, content: 'c:' + n.data, deep: 'c:' + n.data };
|
|
274
|
+
else {
|
|
275
|
+
const kids = liveKids(n).map((c) => keysOfLive(c, keyMap));
|
|
276
|
+
const shallow = 'e:' + n.localName + '|' + liveAttrsKey(n);
|
|
277
|
+
k = { tag: n.localName, shallow, content: 'e:' + n.localName + '|' + hashKey(kids.map((c) => c.content)), deep: shallow + '|' + hashKey(kids.map((c) => c.deep)) };
|
|
278
|
+
}
|
|
279
|
+
keyMap.set(n, k);
|
|
280
|
+
return k;
|
|
281
|
+
}
|
|
282
|
+
function keysOfLoc(loc) {
|
|
283
|
+
if (loc.kind === 'text') return (loc.keys = { tag: '#text', shallow: 't:' + loc.value, content: 't:' + loc.value, deep: 't:' + loc.value });
|
|
284
|
+
if (loc.kind === 'comment') return (loc.keys = { tag: '#comment', shallow: 'c:' + loc.value, content: 'c:' + loc.value, deep: 'c:' + loc.value });
|
|
285
|
+
const kids = loc.children.map(keysOfLoc);
|
|
286
|
+
const shallow = 'e:' + loc.tag + '|' + srcAttrsKey(loc);
|
|
287
|
+
return (loc.keys = { tag: loc.tag, shallow, content: 'e:' + loc.tag + '|' + hashKey(kids.map((c) => c.content)), deep: shallow + '|' + hashKey(kids.map((c) => c.deep)) });
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
// =============================================================================
|
|
291
|
+
// ALIGNMENT
|
|
292
|
+
// =============================================================================
|
|
293
|
+
|
|
294
|
+
const TIERS = ['deep', 'content', 'shallow', 'tag'];
|
|
295
|
+
const DP_CELLS = 4096;
|
|
296
|
+
|
|
297
|
+
/**
|
|
298
|
+
* Order-preserving matching of two child lists.
|
|
299
|
+
*
|
|
300
|
+
* Per tier: trim the common prefix and suffix, anchor on keys unique to both sides
|
|
301
|
+
* (patience), recurse into the gaps at the same tier, and when a gap has no anchors
|
|
302
|
+
* run an exact LCS only if it is small, otherwise hand the gap to the next weaker
|
|
303
|
+
* tier. Every step is linear in the gap except the LIS (a log a) and the bounded DP,
|
|
304
|
+
* so this is O(n log n) per parent and the DP table never exceeds DP_CELLS cells.
|
|
305
|
+
*
|
|
306
|
+
* The prefix and suffix trim is what makes it linear in practice, not the anchors:
|
|
307
|
+
* half of every child list in a formatted document is indentation text nodes, which
|
|
308
|
+
* are never unique and so never anchor anything. An earlier exact-DP version took 84
|
|
309
|
+
* seconds on 4,000 identical siblings and allocated a 256 MB table; this one takes
|
|
310
|
+
* 4.7 ms, and 32,000 siblings in 15.5 ms.
|
|
311
|
+
*/
|
|
312
|
+
export function align(LK, SK) {
|
|
313
|
+
const pairs = [];
|
|
314
|
+
const LI = LK.map((_, i) => i), SI = SK.map((_, j) => j);
|
|
315
|
+
stage(LI, SI, 0);
|
|
316
|
+
// An anchor a boot script moved across a run of identical siblings strands that run
|
|
317
|
+
// on opposite sides of it. One more pass over what is still unpaired lets the run
|
|
318
|
+
// pair, crossing the anchor; render treats a crossing pair as a move, which is still
|
|
319
|
+
// a byte copy rather than a reprint.
|
|
320
|
+
const usedL = new Set(pairs.map((p) => p[0])), usedS = new Set(pairs.map((p) => p[1]));
|
|
321
|
+
const restL = LI.filter((i) => !usedL.has(i)), restS = SI.filter((j) => !usedS.has(j));
|
|
322
|
+
if (restL.length && restS.length) stage(restL, restS, 0);
|
|
323
|
+
pairs.sort((a, b) => a[0] - b[0]);
|
|
324
|
+
return pairs;
|
|
325
|
+
|
|
326
|
+
function stage(li, si, t) {
|
|
327
|
+
let l0 = 0, l1 = li.length, s0 = 0, s1 = si.length;
|
|
328
|
+
if (l0 >= l1 || s0 >= s1) return;
|
|
329
|
+
if (t >= TIERS.length) return positional(li, si);
|
|
330
|
+
const k = TIERS[t];
|
|
331
|
+
while (l0 < l1 && s0 < s1 && LK[li[l0]][k] === SK[si[s0]][k]) { pairs.push([li[l0], si[s0]]); l0++; s0++; }
|
|
332
|
+
while (l0 < l1 && s0 < s1 && LK[li[l1 - 1]][k] === SK[si[s1 - 1]][k]) { l1--; s1--; pairs.push([li[l1], si[s1]]); }
|
|
333
|
+
if (l0 >= l1 || s0 >= s1) return;
|
|
334
|
+
const seenL = new Map(), seenS = new Map();
|
|
335
|
+
for (let x = l0; x < l1; x++) { const key = LK[li[x]][k]; seenL.set(key, seenL.has(key) ? -1 : x); }
|
|
336
|
+
for (let y = s0; y < s1; y++) { const key = SK[si[y]][k]; seenS.set(key, seenS.has(key) ? -1 : y); }
|
|
337
|
+
const cands = [];
|
|
338
|
+
for (let x = l0; x < l1; x++) { const key = LK[li[x]][k]; if (seenL.get(key) === x) { const y = seenS.get(key); if (y !== undefined && y >= 0) cands.push([x, y]); } }
|
|
339
|
+
const anchors = lisPairs(cands);
|
|
340
|
+
if (!anchors.length) {
|
|
341
|
+
if ((l1 - l0) * (s1 - s0) <= DP_CELLS) {
|
|
342
|
+
const got = lcs(LK, SK, li, si, l0, l1, s0, s1, k);
|
|
343
|
+
if (got.length) { let px = l0, py = s0; for (const [x, y] of got) { pairs.push([li[x], si[y]]); stage(li.slice(px, x), si.slice(py, y), t + 1); px = x + 1; py = y + 1; } stage(li.slice(px, l1), si.slice(py, s1), t + 1); return; }
|
|
344
|
+
}
|
|
345
|
+
return stage(li.slice(l0, l1), si.slice(s0, s1), t + 1);
|
|
346
|
+
}
|
|
347
|
+
let px = l0, py = s0;
|
|
348
|
+
for (const [x, y] of anchors) { pairs.push([li[x], si[y]]); stage(li.slice(px, x), si.slice(py, y), t); px = x + 1; py = y + 1; }
|
|
349
|
+
stage(li.slice(px, l1), si.slice(py, s1), t);
|
|
350
|
+
}
|
|
351
|
+
// Last resort for a large gap nothing else resolved: same-tag runs of equal length
|
|
352
|
+
// pair by position.
|
|
353
|
+
function positional(li, si) {
|
|
354
|
+
const byTagL = new Map(), byTagS = new Map();
|
|
355
|
+
for (const i of li) { const tg = LK[i].tag; if (!byTagL.has(tg)) byTagL.set(tg, []); byTagL.get(tg).push(i); }
|
|
356
|
+
for (const j of si) { const tg = SK[j].tag; if (!byTagS.has(tg)) byTagS.set(tg, []); byTagS.get(tg).push(j); }
|
|
357
|
+
for (const [tg, l] of byTagL) { const s = byTagS.get(tg); if (s && s.length === l.length && tg !== '#text' && tg !== '#comment') l.forEach((i, x) => pairs.push([i, s[x]])); }
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
function lisPairs(cands) { // cands increasing in i; keep the longest subsequence increasing in j
|
|
362
|
+
const tails = [], tailAt = [], prev = new Array(cands.length).fill(-1);
|
|
363
|
+
for (let x = 0; x < cands.length; x++) {
|
|
364
|
+
const v = cands[x][1];
|
|
365
|
+
let lo = 0, hi = tails.length;
|
|
366
|
+
while (lo < hi) { const mid = (lo + hi) >> 1; if (tails[mid] < v) lo = mid + 1; else hi = mid; }
|
|
367
|
+
tails[lo] = v; tailAt[lo] = x; prev[x] = lo > 0 ? tailAt[lo - 1] : -1;
|
|
368
|
+
}
|
|
369
|
+
const out = []; let x = tailAt.length ? tailAt[tailAt.length - 1] : -1;
|
|
370
|
+
while (x >= 0) { out.push(cands[x]); x = prev[x]; }
|
|
371
|
+
return out.reverse();
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
function lcs(LK, SK, li, si, l0, l1, s0, s1, k) {
|
|
375
|
+
const n = l1 - l0, m = s1 - s0;
|
|
376
|
+
const eq = (i, j) => LK[li[l0 + i]][k] === SK[si[s0 + j]][k];
|
|
377
|
+
const dp = Array.from({ length: n + 1 }, () => new Uint16Array(m + 1));
|
|
378
|
+
for (let i = n - 1; i >= 0; i--) for (let j = m - 1; j >= 0; j--)
|
|
379
|
+
dp[i][j] = eq(i, j) ? dp[i + 1][j + 1] + 1 : Math.max(dp[i + 1][j], dp[i][j + 1]);
|
|
380
|
+
const out = []; let i = 0, j = 0;
|
|
381
|
+
while (i < n && j < m) {
|
|
382
|
+
if (eq(i, j)) { out.push([l0 + i, s0 + j]); i++; j++; }
|
|
383
|
+
else if (dp[i + 1][j] >= dp[i][j + 1]) i++; else j++;
|
|
384
|
+
}
|
|
385
|
+
return out;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
// =============================================================================
|
|
389
|
+
// PAIRING
|
|
390
|
+
// =============================================================================
|
|
391
|
+
|
|
392
|
+
function describe(n) {
|
|
393
|
+
if (n.nodeType === 3) return '#text ' + JSON.stringify(n.data.slice(0, 40));
|
|
394
|
+
if (n.nodeType === 8) return '#comment';
|
|
395
|
+
return '<' + n.localName + Array.from(n.attributes, (a) => ' ' + a.name + '="' + a.value.slice(0, 30) + '"').join('') + '>';
|
|
396
|
+
}
|
|
397
|
+
function describeLoc(l) {
|
|
398
|
+
if (l.kind === 'text') return '#text ' + JSON.stringify(l.value.slice(0, 40));
|
|
399
|
+
if (l.kind === 'comment') return '#comment';
|
|
400
|
+
return '<' + l.tag + l.attrs.map((a) => ' ' + a.name + '="' + a.value.slice(0, 30) + '"').join('') + '>';
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/**
|
|
404
|
+
* Pair a tree against the model, and key the result by whatever `resolve` returns.
|
|
405
|
+
*
|
|
406
|
+
* The tree walked here is the SAVE CLONE, not the live DOM, because the clone is in
|
|
407
|
+
* the same domain as the file: edit mode has been deactivated back to the inert
|
|
408
|
+
* attribute forms, [no-save] regions are gone, and every document transform has run.
|
|
409
|
+
* The live DOM is in the activated domain, where `contenteditable="true"` stands
|
|
410
|
+
* where the file says `inert-contenteditable="true"`, and pairing there would fail
|
|
411
|
+
* the two strongest signature tiers on every activated node and on every ancestor of
|
|
412
|
+
* one.
|
|
413
|
+
*
|
|
414
|
+
* The MAP is keyed by the LIVE node (`resolve` is snapshot provenance), because the
|
|
415
|
+
* clone is rebuilt from scratch on every save and its nodes are new objects each
|
|
416
|
+
* time. A clone node a transform created has no live original and simply goes
|
|
417
|
+
* unpaired, which means it is printed, the same thing that happens to it today.
|
|
418
|
+
*
|
|
419
|
+
* @param {Node} root - the tree to walk (the save clone's document element)
|
|
420
|
+
* @param {Object} m - a model()
|
|
421
|
+
* @param {Function} [resolve] - node -> the identity to key the map by (default: itself)
|
|
422
|
+
*/
|
|
423
|
+
export function pair(root, m, resolve = (n) => n) {
|
|
424
|
+
const map = new WeakMap();
|
|
425
|
+
const keyMap = new Map();
|
|
426
|
+
keysOfLive(root, keyMap);
|
|
427
|
+
keysOfLoc(m.root);
|
|
428
|
+
const stats = { paired: 0, unmatchedLive: [], unmatchedSource: [], attrDiffs: [], textDiffs: 0, unresolved: 0 };
|
|
429
|
+
// A token identifying THIS pairing pass, stamped on the model and on every loc that
|
|
430
|
+
// binds. render trusts a loc only when the two still agree, which rules out a loc
|
|
431
|
+
// left over from an earlier pass and a map paired against a different model. The
|
|
432
|
+
// token is an object rather than the live node on purpose: a loc lives as long as the
|
|
433
|
+
// model, so holding the node here would keep every node the page has since deleted.
|
|
434
|
+
const gen = {};
|
|
435
|
+
m.gen = gen;
|
|
436
|
+
bind(root, m.root, map, keyMap, stats, 'html', resolve, gen);
|
|
437
|
+
return { map, stats };
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
function bind(node, loc, map, keyMap, stats, path, resolve, gen) {
|
|
441
|
+
const key = resolve(node);
|
|
442
|
+
if (key) { map.set(key, loc); loc.gen = gen; stats.paired++; }
|
|
443
|
+
else stats.unresolved++;
|
|
444
|
+
if (loc.kind !== 'element') { if (node.data !== loc.value) stats.textDiffs++; return; }
|
|
445
|
+
if (liveAttrsKey(node) !== srcAttrsKey(loc)) stats.attrDiffs.push({ path, live: describe(node), source: describeLoc(loc) });
|
|
446
|
+
const L = liveKids(node), S = loc.children;
|
|
447
|
+
const pairs = align(L.map((n) => keyMap.get(n)), S.map((l) => l.keys));
|
|
448
|
+
const usedL = new Set(), usedS = new Set();
|
|
449
|
+
for (const [i, j] of pairs) {
|
|
450
|
+
usedL.add(i); usedS.add(j);
|
|
451
|
+
bind(L[i], S[j], map, keyMap, stats, path + '>' + (L[i].nodeType === 1 ? L[i].localName : '#') + '[' + i + ']', resolve, gen);
|
|
452
|
+
}
|
|
453
|
+
for (let i = 0; i < L.length; i++) if (!usedL.has(i)) stats.unmatchedLive.push({ path, node: describe(L[i]) });
|
|
454
|
+
for (let j = 0; j < S.length; j++) if (!usedS.has(j)) stats.unmatchedSource.push({ path, node: describeLoc(S[j]) });
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
// =============================================================================
|
|
458
|
+
// RENDER
|
|
459
|
+
// =============================================================================
|
|
460
|
+
|
|
461
|
+
/**
|
|
462
|
+
* Emit the save clone as bytes, copying source for everything unchanged.
|
|
463
|
+
*
|
|
464
|
+
* EXACT. Every clone child is emitted, whitespace text nodes included, and nothing
|
|
465
|
+
* here knows or cares whether a gap is indentation or a rendered space between two
|
|
466
|
+
* inline elements. An earlier version had a "layout parent" rule that dropped
|
|
467
|
+
* whitespace-only children of block elements, which is correct for indentation and
|
|
468
|
+
* wrong for `by <a>Ana</a> <a>Bo</a>`, and it wrote that back as `AnaBo`. CSS
|
|
469
|
+
* collapses a newline and a space identically between inline boxes, so there is no
|
|
470
|
+
* predicate that separates the two cases and the rule had to go, not be refined.
|
|
471
|
+
*
|
|
472
|
+
* @param {HTMLElement} clone - the prepared save clone
|
|
473
|
+
* @param {WeakMap} map - live node -> loc, from pair()
|
|
474
|
+
* @param {Object} m - the model
|
|
475
|
+
* @param {Function} provenance - clone node -> live node
|
|
476
|
+
* @param {Object} [opts] - break switches, for tests that must exercise the fallback
|
|
477
|
+
*/
|
|
478
|
+
/**
|
|
479
|
+
* Where in the file this live element is.
|
|
480
|
+
*
|
|
481
|
+
* The whole of what an agent editing loop needs beyond the save itself: point at an element in the
|
|
482
|
+
* page, get its range in the bytes that will be written, and express the edit as a source range
|
|
483
|
+
* rather than as a DOM mutation. No ids in the file, no map surviving a reload, no patch lists.
|
|
484
|
+
*
|
|
485
|
+
* OFFSETS ARE UTF-16 CODE UNITS into the same string `text()` returns, because that is the string
|
|
486
|
+
* the model was built from. They are NOT byte offsets, and on a document with any non-ASCII content
|
|
487
|
+
* the two differ: one line of `café 🎉 naïve` puts the same position at 51 code units and 55 UTF-8
|
|
488
|
+
* bytes. Slice the text this module hands you and the answer is exact; feed these numbers to a
|
|
489
|
+
* byte-oriented tool and it edits the wrong place, silently, and only on some documents. `column` is
|
|
490
|
+
* in the same units for the same reason.
|
|
491
|
+
*
|
|
492
|
+
* `null` rather than a guess, and the three reasons are different: a node the page created after
|
|
493
|
+
* boot is not in the file yet, an implied <html>, <head> or <body> the author never wrote has no
|
|
494
|
+
* bytes to point at, and a stale generation means the model was replaced since this map was built.
|
|
495
|
+
* A caller that treats all three as "not found" will mistake the third for the first.
|
|
496
|
+
*/
|
|
497
|
+
export function locate(node, map, m) {
|
|
498
|
+
const loc = map.get(node);
|
|
499
|
+
if (!loc) return null; // never paired, so not in the file
|
|
500
|
+
if (loc.gen !== m.gen) return null; // a map and a model that were not paired together
|
|
501
|
+
if (loc.from < 0 || loc.to < 0) return null; // an implied tag has no bytes
|
|
502
|
+
// The bytes at this offset still have to BE this element. A map pointing into the wrong model
|
|
503
|
+
// slices one file's offsets out of another file's bytes, which is silent and lands in whatever
|
|
504
|
+
// the agent writes next, with no verifier between it and the file. Cheap, and it turns the one
|
|
505
|
+
// failure this function can have into a null instead of a wrong answer.
|
|
506
|
+
if (loc.kind === 'element' && loc.located) {
|
|
507
|
+
const head = m.src.slice(loc.openFrom, loc.openFrom + loc.tag.length + 1);
|
|
508
|
+
if (head.toLowerCase() !== '<' + loc.tag.toLowerCase()) return null;
|
|
509
|
+
}
|
|
510
|
+
const starts = lineStarts(m);
|
|
511
|
+
const line = upperBound(starts, loc.from);
|
|
512
|
+
return { from: loc.from, to: loc.to, line: line + 1, column: loc.from - starts[line] + 1 };
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
function lineStarts(m) {
|
|
516
|
+
if (!m.lineStarts) {
|
|
517
|
+
const starts = [0];
|
|
518
|
+
for (let i = 0; i < m.src.length; i++) if (m.src.charCodeAt(i) === 10) starts.push(i + 1);
|
|
519
|
+
m.lineStarts = starts;
|
|
520
|
+
}
|
|
521
|
+
return m.lineStarts;
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
/** The index of the last start at or before `at`. */
|
|
525
|
+
function upperBound(starts, at) {
|
|
526
|
+
let lo = 0;
|
|
527
|
+
let hi = starts.length - 1;
|
|
528
|
+
while (lo < hi) {
|
|
529
|
+
const mid = (lo + hi + 1) >> 1;
|
|
530
|
+
if (starts[mid] <= at) lo = mid;
|
|
531
|
+
else hi = mid - 1;
|
|
532
|
+
}
|
|
533
|
+
return lo;
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
export function render(clone, map, m, provenance, opts = {}) {
|
|
537
|
+
const now = () => (typeof performance !== 'undefined' ? performance.now() : Date.now());
|
|
538
|
+
const t0 = now();
|
|
539
|
+
const src = m.src;
|
|
540
|
+
const pieces = [];
|
|
541
|
+
// `from >= 0` is not defensive padding: `src.slice(-1, n)` is the LAST BYTE of the
|
|
542
|
+
// source, so an unset offset reaching here does not produce nothing, it produces one
|
|
543
|
+
// wrong byte and looks like a successful render.
|
|
544
|
+
const keep = (from, to) => { if (from >= 0 && to > from) pieces.push({ from, to }); };
|
|
545
|
+
const text = (t) => { if (t) pieces.push({ text: t }); };
|
|
546
|
+
const locOfClone = (n) => { const live = provenance(n); const l = live ? map.get(live) : null; return l && l.gen === m.gen ? l : null; };
|
|
547
|
+
const rootLoc = m.root;
|
|
548
|
+
keep(0, rootLoc.openFrom); // the authored doctype and anything before <html>, verbatim
|
|
549
|
+
emitElement(clone, rootLoc);
|
|
550
|
+
keep(rootLoc.closeTo, src.length); // the trailing bytes, verbatim
|
|
551
|
+
|
|
552
|
+
const out = pieces.map((p) => p.text !== undefined ? p.text : src.slice(p.from, p.to)).join('');
|
|
553
|
+
return { text: out, ms: now() - t0 };
|
|
554
|
+
|
|
555
|
+
function emitNode(n, parentTag) {
|
|
556
|
+
if (n.nodeType === 3) return emitText(n, parentTag);
|
|
557
|
+
if (n.nodeType === 8) return emitComment(n);
|
|
558
|
+
if (n.nodeType === 1) return emitElement(n, locOfClone(n));
|
|
559
|
+
}
|
|
560
|
+
function emitComment(n) {
|
|
561
|
+
const loc = locOfClone(n);
|
|
562
|
+
if (loc && loc.kind === 'comment' && loc.value === n.data && loc.from >= 0 && !opts.breakText) keep(loc.from, loc.to);
|
|
563
|
+
else text('<!--' + n.data + '-->');
|
|
564
|
+
}
|
|
565
|
+
function emitText(n, parentTag) {
|
|
566
|
+
const loc = locOfClone(n);
|
|
567
|
+
let data = n.data;
|
|
568
|
+
if (opts.corrupt === 'drop-first-text' && !opts.corrupted && data.trim()) { opts.corrupted = true; data = data.slice(1); }
|
|
569
|
+
if (loc && loc.kind === 'text' && loc.value === data && loc.from >= 0 && !opts.breakText) { keep(loc.from, loc.to); return; }
|
|
570
|
+
if (loc && loc.kind === 'text' && loc.outsideTail && data.endsWith(loc.outsideTail)) {
|
|
571
|
+
data = data.slice(0, data.length - loc.outsideTail.length);
|
|
572
|
+
}
|
|
573
|
+
text(RAW_TEXT.has(parentTag) ? data : escText(data));
|
|
574
|
+
}
|
|
575
|
+
function emitElement(n, loc) {
|
|
576
|
+
const tag = n.localName;
|
|
577
|
+
const isVoid = VOID.has(tag) && n.namespaceURI === XHTML;
|
|
578
|
+
const paired = loc && loc.kind === 'element' && !opts.breakTags ? loc : null;
|
|
579
|
+
if (paired && !paired.located) {
|
|
580
|
+
// An implied tag stays implied when the clone element carries nothing a tag would
|
|
581
|
+
// have to say. Otherwise it is printed, which is what the parser would have to
|
|
582
|
+
// imply anyway plus the attributes.
|
|
583
|
+
//
|
|
584
|
+
// `openFrom < 0` means the source has no bytes here at all, which is what an
|
|
585
|
+
// empty implied <head> looks like in a file that goes straight from <html> to
|
|
586
|
+
// <body>. Printing `<head></head>` there would be this module's own addition to
|
|
587
|
+
// a file nobody asked it to change, so an empty one emits nothing; one that has
|
|
588
|
+
// gained children emits the children and still no tag, and the parser implies it
|
|
589
|
+
// back on the next load.
|
|
590
|
+
if (n.attributes.length === 0) {
|
|
591
|
+
if (paired.openFrom >= 0) { emitChildren(n, paired, tag); return; }
|
|
592
|
+
if (liveKids(n).length) emitChildren(n, null, tag);
|
|
593
|
+
return;
|
|
594
|
+
}
|
|
595
|
+
text(printOpenTag(n));
|
|
596
|
+
if (isVoid) return;
|
|
597
|
+
emitChildren(n, paired.openFrom >= 0 ? paired : null, tag);
|
|
598
|
+
text('</' + tag + '>');
|
|
599
|
+
return;
|
|
600
|
+
}
|
|
601
|
+
if (!paired) {
|
|
602
|
+
text(printOpenTag(n));
|
|
603
|
+
if (isVoid) return;
|
|
604
|
+
emitChildren(n, null, tag);
|
|
605
|
+
text('</' + tag + '>');
|
|
606
|
+
return;
|
|
607
|
+
}
|
|
608
|
+
emitOpenTag(n, paired);
|
|
609
|
+
if (isVoid) return;
|
|
610
|
+
emitChildren(n, paired, tag);
|
|
611
|
+
keep(paired.closeFrom, paired.closeTo);
|
|
612
|
+
}
|
|
613
|
+
function printOpenTag(n) {
|
|
614
|
+
let s = '<' + n.localName;
|
|
615
|
+
for (const a of n.attributes) s += ' ' + a.name + (a.value === '' ? '' : '="' + escAttr(a.value) + '"');
|
|
616
|
+
return s + '>';
|
|
617
|
+
}
|
|
618
|
+
// Attribute by attribute, in the source's own order and its own spelling. One the
|
|
619
|
+
// clone still has with the same value copies its bytes; a changed one keeps the
|
|
620
|
+
// author's quoting; a removed one takes its leading whitespace with it; a new one is
|
|
621
|
+
// appended just before the `>`.
|
|
622
|
+
function emitOpenTag(n, loc) {
|
|
623
|
+
const live = new Map();
|
|
624
|
+
for (const a of n.attributes) live.set(a.name.toLowerCase(), a);
|
|
625
|
+
const seen = new Set();
|
|
626
|
+
let cursor = loc.openFrom;
|
|
627
|
+
for (const a of loc.attrs) {
|
|
628
|
+
if (a.from < 0) continue;
|
|
629
|
+
const cur = live.get(a.key);
|
|
630
|
+
seen.add(a.key);
|
|
631
|
+
if (!cur) { keep(cursor, gapStart(cursor, a.from)); cursor = a.to; continue; }
|
|
632
|
+
if (cur.value === a.value) { keep(cursor, a.to); cursor = a.to; continue; }
|
|
633
|
+
keep(cursor, a.from);
|
|
634
|
+
text(respellAttr(src.slice(a.from, a.to), cur.value));
|
|
635
|
+
cursor = a.to;
|
|
636
|
+
}
|
|
637
|
+
// The `/` before `>` is the tag's own self-closing slash only when it sits outside
|
|
638
|
+
// every attribute. An unquoted value ends at whitespace or `>`, never at `/`, so
|
|
639
|
+
// `<a href=x/>` has the value `x/` and that slash is the value's last byte. Taken
|
|
640
|
+
// for the tag's, it was cut off the copied value and written again before `>`,
|
|
641
|
+
// giving `<a href=x//>`.
|
|
642
|
+
let closeAt = loc.openTo - 1;
|
|
643
|
+
const attrsEnd = loc.attrs.reduce((max, a) => (a.to > max ? a.to : max), loc.openFrom);
|
|
644
|
+
if (src[closeAt - 1] === '/' && closeAt - 1 >= attrsEnd) closeAt--;
|
|
645
|
+
keep(cursor, closeAt);
|
|
646
|
+
for (const a of n.attributes) {
|
|
647
|
+
if (seen.has(a.name.toLowerCase())) continue;
|
|
648
|
+
text(' ' + a.name + (a.value === '' ? '' : '="' + escAttr(a.value) + '"'));
|
|
649
|
+
}
|
|
650
|
+
keep(closeAt, loc.openTo);
|
|
651
|
+
}
|
|
652
|
+
function gapStart(cursor, attrFrom) {
|
|
653
|
+
let i = attrFrom;
|
|
654
|
+
while (i > cursor && /[ \t\n\r\f]/.test(src[i - 1])) i--;
|
|
655
|
+
return i;
|
|
656
|
+
}
|
|
657
|
+
function respellAttr(raw, value) {
|
|
658
|
+
const eq = raw.indexOf('=');
|
|
659
|
+
const name = (eq < 0 ? raw : raw.slice(0, eq)).trimEnd();
|
|
660
|
+
const rest = eq < 0 ? '' : raw.slice(eq + 1).trim();
|
|
661
|
+
const q = rest[0] === "'" ? "'" : rest[0] === '"' ? '"' : '';
|
|
662
|
+
const esc = value.replace(/&/g, '&').replace(/ /g, ' ');
|
|
663
|
+
if (q === "'" && !value.includes("'")) return name + "='" + esc + "'";
|
|
664
|
+
if (q === '' && rest !== '' && value !== '' && !/[\s"'=<>`]/.test(value)) return name + '=' + esc;
|
|
665
|
+
return name + '="' + esc.replace(/"/g, '"') + '"';
|
|
666
|
+
}
|
|
667
|
+
// Children, exact. Identity decides order: clone children whose source node belongs to
|
|
668
|
+
// this parent and form the longest in-order run stay in place and copy their bytes;
|
|
669
|
+
// every other clone child is emitted where the clone has it (printed, or copied out of
|
|
670
|
+
// place as a move); every source child the clone no longer has is dropped.
|
|
671
|
+
function emitChildren(el, loc, parentTag) {
|
|
672
|
+
const C = liveKids(el);
|
|
673
|
+
if (!loc) { for (const c of C) emitNode(c, parentTag); return; }
|
|
674
|
+
const S = loc.children;
|
|
675
|
+
const sIndex = new Map(S.map((s, j) => [s, j]));
|
|
676
|
+
const owned = C.map((c) => { const l = locOfClone(c); return l && l.parent === loc && l.from >= 0 ? sIndex.get(l) : -1; });
|
|
677
|
+
const inPlace = new Set(lis(owned));
|
|
678
|
+
let cursor = loc.openTo;
|
|
679
|
+
let lastS = -1;
|
|
680
|
+
const dropTo = (j) => { for (let k = lastS + 1; k < j; k++) { const sk = S[k]; if (sk.from < 0) continue; keep(cursor, sk.from); cursor = Math.max(cursor, sk.to); } };
|
|
681
|
+
for (let i = 0; i < C.length; i++) {
|
|
682
|
+
const c = C[i];
|
|
683
|
+
if (inPlace.has(i)) {
|
|
684
|
+
const j = owned[i], s = S[j];
|
|
685
|
+
dropTo(j);
|
|
686
|
+
keep(cursor, s.from);
|
|
687
|
+
emitNode(c, parentTag);
|
|
688
|
+
cursor = Math.max(cursor, s.to); lastS = j;
|
|
689
|
+
continue;
|
|
690
|
+
}
|
|
691
|
+
emitNode(c, parentTag);
|
|
692
|
+
}
|
|
693
|
+
dropTo(S.length);
|
|
694
|
+
keep(cursor, loc.closeFrom);
|
|
695
|
+
}
|
|
696
|
+
}
|
|
697
|
+
|
|
698
|
+
function lis(vals) {
|
|
699
|
+
const tails = [], tailIdx = [], prev = new Array(vals.length).fill(-1);
|
|
700
|
+
for (let i = 0; i < vals.length; i++) {
|
|
701
|
+
const v = vals[i]; if (v < 0) continue;
|
|
702
|
+
let lo = 0, hi = tails.length;
|
|
703
|
+
while (lo < hi) { const mid = (lo + hi) >> 1; if (tails[mid] < v) lo = mid + 1; else hi = mid; }
|
|
704
|
+
tails[lo] = v; tailIdx[lo] = i; prev[i] = lo > 0 ? tailIdx[lo - 1] : -1;
|
|
705
|
+
}
|
|
706
|
+
const out = []; let i = tailIdx.length ? tailIdx[tailIdx.length - 1] : -1;
|
|
707
|
+
while (i >= 0) { out.push(i); i = prev[i]; }
|
|
708
|
+
return out.reverse();
|
|
709
|
+
}
|
|
710
|
+
|
|
711
|
+
// =============================================================================
|
|
712
|
+
// VERIFY
|
|
713
|
+
// =============================================================================
|
|
714
|
+
// Independent of render on purpose: its own child enumeration, its own attribute key,
|
|
715
|
+
// no notion of layout or whitespace, no shared helper. A check built out of the thing
|
|
716
|
+
// it is checking cannot fail in the cases it exists to catch.
|
|
717
|
+
//
|
|
718
|
+
// Three checks, all exact:
|
|
719
|
+
// 1. the rendered document's nodes outside <html> are the live document's
|
|
720
|
+
// 2. the rendered <html> subtree equals today's serialization reparsed, node for node
|
|
721
|
+
// 3. or, failing that, equals the save clone itself
|
|
722
|
+
//
|
|
723
|
+
// Either oracle suffices, and they are not the same claim. "Equal to today's bytes" is
|
|
724
|
+
// the floor: no worse than the save that would otherwise have gone out. "Equal to the
|
|
725
|
+
// clone" is stronger, and is sometimes the only one available, because today's
|
|
726
|
+
// serializer is not itself exact: it drops the newline that starts a <pre>, so on a
|
|
727
|
+
// document with one, today's bytes reload as a different document and these reload as
|
|
728
|
+
// the right one.
|
|
729
|
+
|
|
730
|
+
export function verify(rendered, today, liveDocument, clone = null, sourceErrors = null) {
|
|
731
|
+
const P = new DOMParser();
|
|
732
|
+
const A = P.parseFromString(rendered, 'text/html');
|
|
733
|
+
const B = P.parseFromString(today, 'text/html');
|
|
734
|
+
const outsideA = vOutside(A), outsideLive = vOutside(liveDocument);
|
|
735
|
+
if (outsideA.join('\n') !== outsideLive.join('\n')) return { ok: false, diff: 'document: outside <html> ' + JSON.stringify(outsideA) + ' vs live ' + JSON.stringify(outsideLive) };
|
|
736
|
+
const diff = vDiff(A.documentElement, B.documentElement, 'html');
|
|
737
|
+
const worse = diff ? null : vParseErrors(rendered, sourceErrors);
|
|
738
|
+
if (!diff) return worse ? { ok: false, diff: worse } : { ok: true, diff: null, oracle: 'today' };
|
|
739
|
+
if (clone && !vDiff(A.documentElement, clone, 'html')) {
|
|
740
|
+
const w = vParseErrors(rendered, sourceErrors);
|
|
741
|
+
return w ? { ok: false, diff: w } : { ok: true, diff: null, oracle: 'clone', todayDiff: diff };
|
|
742
|
+
}
|
|
743
|
+
return { ok: false, diff };
|
|
744
|
+
}
|
|
745
|
+
|
|
746
|
+
/**
|
|
747
|
+
* The check the tree comparison cannot make.
|
|
748
|
+
*
|
|
749
|
+
* A parser DISCARDS some input rather than representing it, so the tree is the same
|
|
750
|
+
* whether or not it was there. A second copy of an attribute is the case that bit:
|
|
751
|
+
* the render wrote `xmlns:xlink` twice, the parser kept the first and dropped the
|
|
752
|
+
* second, and the trees matched. Counting, not presence, because 4 of 198 real user
|
|
753
|
+
* documents already parse with errors: only an INCREASE is the render's doing, and a
|
|
754
|
+
* full serialization cannot produce one, because the DOM it comes from cannot hold a
|
|
755
|
+
* duplicate attribute in the first place.
|
|
756
|
+
*
|
|
757
|
+
* This is narrower than it sounds and worth saying plainly: it catches the discarded
|
|
758
|
+
* class only. Bytes the parser DOES represent, such as a duplicated element, change
|
|
759
|
+
* the tree, and the comparison above is what catches those.
|
|
760
|
+
*/
|
|
761
|
+
function vParseErrors(rendered, sourceErrors) {
|
|
762
|
+
if (!sourceErrors) return null;
|
|
763
|
+
const seen = new Map();
|
|
764
|
+
parse(rendered, { onParseError: (e) => seen.set(e.code, (seen.get(e.code) || 0) + 1) });
|
|
765
|
+
for (const [code, n] of seen) {
|
|
766
|
+
const was = sourceErrors.get(code) || 0;
|
|
767
|
+
if (n > was) return `the render introduced parse errors the source does not have: ${code} ${was} -> ${n}`;
|
|
768
|
+
}
|
|
769
|
+
return null;
|
|
770
|
+
}
|
|
771
|
+
function vOutside(doc) {
|
|
772
|
+
const out = [];
|
|
773
|
+
for (const n of doc.childNodes) {
|
|
774
|
+
if (n.nodeType === 10) out.push('doctype ' + n.name + '|' + n.publicId + '|' + n.systemId);
|
|
775
|
+
else if (n.nodeType === 8) out.push('comment ' + n.data);
|
|
776
|
+
else if (n.nodeType !== 1) out.push('node' + n.nodeType + ' ' + (n.data || ''));
|
|
777
|
+
}
|
|
778
|
+
return out;
|
|
779
|
+
}
|
|
780
|
+
/**
|
|
781
|
+
* <noscript> is the one element whose children depend on a flag neither side of this
|
|
782
|
+
* comparison controls. A page the browser loaded was parsed with scripting ENABLED, so
|
|
783
|
+
* the live tree holds the block's markup as a single text node. DOMParser always parses
|
|
784
|
+
* with scripting DISABLED, so re-reading the rendered bytes turns that same markup back
|
|
785
|
+
* into elements. The two trees can never agree, and without this a document containing
|
|
786
|
+
* one <noscript> would fail verification on every save, forever, which is worse than a
|
|
787
|
+
* wrong answer: it is a permanent false alarm on the counter that stage 2 reads.
|
|
788
|
+
*
|
|
789
|
+
* So both sides hand their content to ONE parser and the resulting trees are compared.
|
|
790
|
+
* That is exact rather than tolerant: anything the renderer corrupted inside the block
|
|
791
|
+
* still changes the tree its markup parses to. Recursion terminates because each step
|
|
792
|
+
* compares strictly less markup than the one above it.
|
|
793
|
+
*/
|
|
794
|
+
function vNoscript(a, b, path) {
|
|
795
|
+
const P = new DOMParser();
|
|
796
|
+
const inner = (el) => {
|
|
797
|
+
const kids = Array.from(el.childNodes);
|
|
798
|
+
return kids.length && kids.every((n) => n.nodeType === 3) ? kids.map((n) => n.data).join('') : el.innerHTML;
|
|
799
|
+
};
|
|
800
|
+
return vDiff(P.parseFromString(inner(a), 'text/html').body, P.parseFromString(inner(b), 'text/html').body, path + '>#noscript');
|
|
801
|
+
}
|
|
802
|
+
function vKids(el) {
|
|
803
|
+
const list = el.localName === 'template' && el.content ? el.content.childNodes : el.childNodes;
|
|
804
|
+
return Array.from(list).filter((n) => n.nodeType === 1 || n.nodeType === 3 || n.nodeType === 8);
|
|
805
|
+
}
|
|
806
|
+
const vAttrs = (el) => Array.from(el.attributes, (a) => (a.namespaceURI || '') + ' ' + a.name + '=' + JSON.stringify(a.value)).sort().join('; ');
|
|
807
|
+
function vDiff(a, b, path) {
|
|
808
|
+
if (a.nodeType !== b.nodeType) return path + ': node type ' + a.nodeType + ' vs ' + b.nodeType;
|
|
809
|
+
if (a.nodeType === 3 || a.nodeType === 8) return a.data === b.data ? null : path + ': text ' + JSON.stringify(a.data.slice(0, 60)) + ' vs ' + JSON.stringify(b.data.slice(0, 60));
|
|
810
|
+
if (a.localName !== b.localName || a.namespaceURI !== b.namespaceURI) return path + ': tag ' + a.localName + ' vs ' + b.localName;
|
|
811
|
+
if (vAttrs(a) !== vAttrs(b)) return path + ': attrs ' + vAttrs(a) + ' vs ' + vAttrs(b);
|
|
812
|
+
if (a.localName === 'noscript' && a.namespaceURI === XHTML) return vNoscript(a, b, path);
|
|
813
|
+
const ka = vKids(a), kb = vKids(b);
|
|
814
|
+
if (ka.length !== kb.length) return path + ': ' + ka.length + ' vs ' + kb.length + ' children';
|
|
815
|
+
for (let i = 0; i < ka.length; i++) { const d = vDiff(ka[i], kb[i], path + '>' + (ka[i].localName || '#') + '[' + i + ']'); if (d) return d; }
|
|
816
|
+
return null;
|
|
817
|
+
}
|