@erdemtuna/doc-review 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,7 +5,7 @@ export function normalizeReviewMode(mode) {
5
5
  }
6
6
 
7
7
  export function savePolicyForPage(page) {
8
- return page && (page.kind === "url" || page.markdown || page.feedbackOnly)
8
+ return page && (page.savePolicy === "feedback-only" || page.kind === "url" || page.markdown || page.feedbackOnly)
9
9
  ? "feedback-only"
10
10
  : "writable";
11
11
  }
@@ -0,0 +1,440 @@
1
+ import { diffArrays, diffLines, diffWordsWithSpace } from "diff";
2
+
3
+ export const DIFF_VERSION = 2;
4
+ export const DIFF_LIMITS = Object.freeze({
5
+ maxBlocks: 2000,
6
+ maxCharacters: 1000000,
7
+ maxBlockCharacters: 100000,
8
+ maxRuns: 1000,
9
+ maxTokens: 50000,
10
+ maxChanges: 4000,
11
+ maxEditLength: 2000,
12
+ maxRows: 50000,
13
+ maxResponseCharacters: 16000000,
14
+ timeoutMs: 250,
15
+ });
16
+
17
+ class ComparisonLimit extends Error {}
18
+ const emptyCounts = () => ({ added: 0, modified: 0, removed: 0, total: 0 });
19
+ const attributes = new Set([
20
+ "id", "href", "src", "alt", "title", "start", "reversed", "value",
21
+ "rowspan", "colspan", "scope", "headers", "type",
22
+ ]);
23
+ const stable = (value) => {
24
+ if (Array.isArray(value)) return `[${value.map(stable).join(",")}]`;
25
+ if (value && typeof value === "object") {
26
+ return `{${Object.keys(value).sort().map((key) => `${JSON.stringify(key)}:${stable(value[key])}`).join(",")}}`;
27
+ }
28
+ return JSON.stringify(value);
29
+ };
30
+ const content = (block) => stable({
31
+ tag: block.tag, text: block.text, attributes: block.attributes,
32
+ runs: block.runs, path: block.path,
33
+ });
34
+ const hash = (value) => {
35
+ let n = 2166136261;
36
+ for (let i = 0; i < value.length; i++) n = Math.imul(n ^ value.charCodeAt(i), 16777619);
37
+ return (n >>> 0).toString(36);
38
+ };
39
+
40
+ function context(options) {
41
+ const limits = { ...DIFF_LIMITS };
42
+ for (const key of Object.keys(limits)) {
43
+ if (options?.[key] !== undefined) {
44
+ if (!Number.isSafeInteger(options[key]) || options[key] < 1) throw new TypeError(`Invalid ${key}`);
45
+ limits[key] = Math.min(limits[key], options[key]);
46
+ }
47
+ }
48
+ const deadline = Date.now() + limits.timeoutMs;
49
+ return {
50
+ limits,
51
+ check() {
52
+ if (Date.now() > deadline) throw new ComparisonLimit("time");
53
+ },
54
+ primitives() {
55
+ this.check();
56
+ return { timeout: Math.max(1, deadline - Date.now()), maxEditLength: limits.maxEditLength };
57
+ },
58
+ words(before, after) {
59
+ if ((before.match(/\s+|[^\s]+/g)?.length || 0) +
60
+ (after.match(/\s+|[^\s]+/g)?.length || 0) > limits.maxTokens) {
61
+ throw new ComparisonLimit("tokens");
62
+ }
63
+ const parts = diffWordsWithSpace(before, after, this.primitives());
64
+ if (!parts) throw new ComparisonLimit("word-diff");
65
+ this.check();
66
+ return parts.map(({ value, added, removed }) => ({
67
+ value, ...(added ? { added: true } : {}), ...(removed ? { removed: true } : {}),
68
+ }));
69
+ },
70
+ };
71
+ }
72
+
73
+ function validateSnapshot(snapshot, ctx) {
74
+ if (!snapshot || snapshot.version !== 1 || !Array.isArray(snapshot.blocks) ||
75
+ !Array.isArray(snapshot.limitations) || snapshot.limitations.length > 64 ||
76
+ snapshot.limitations.some((item) => typeof item !== "string" || item.length > 256)) {
77
+ throw new TypeError("Invalid semantic snapshot");
78
+ }
79
+ if (snapshot.blocks.length > ctx.limits.maxBlocks) throw new ComparisonLimit("blocks");
80
+ let characters = 0;
81
+ const ids = new Set();
82
+ for (const block of snapshot.blocks) {
83
+ ctx.check();
84
+ if (!block || typeof block.id !== "string" || !block.id || block.id.length > 4096 ||
85
+ ids.has(block.id) || typeof block.tag !== "string" || !/^[a-z][a-z0-9-]{0,63}$/.test(block.tag) ||
86
+ typeof block.selector !== "string" || block.selector.length > 4096 ||
87
+ typeof block.text !== "string" || !Array.isArray(block.runs) ||
88
+ !Array.isArray(block.path) || block.path.length > 64 ||
89
+ block.path.some((item) => typeof item !== "string" || item.length > 4096) ||
90
+ !block.attributes || typeof block.attributes !== "object" || Array.isArray(block.attributes)) {
91
+ throw new TypeError("Invalid semantic block");
92
+ }
93
+ ids.add(block.id);
94
+ if (block.text.length > ctx.limits.maxBlockCharacters) throw new ComparisonLimit("block-text");
95
+ if (block.runs.length > ctx.limits.maxRuns) throw new ComparisonLimit("runs");
96
+ characters += block.text.length + block.id.length + block.selector.length +
97
+ block.path.reduce((sum, item) => sum + item.length, 0);
98
+ for (const [key, value] of Object.entries(block.attributes)) {
99
+ if (!attributes.has(key) || typeof value !== "string" || value.length > 4096 ||
100
+ (key === "value" && block.tag !== "li") ||
101
+ (key === "type" && block.tag !== "button")) throw new TypeError("Invalid semantic attributes");
102
+ characters += value.length;
103
+ }
104
+ for (const run of block.runs) {
105
+ if (!run || typeof run.text !== "string" || !Array.isArray(run.marks) ||
106
+ run.marks.length > 16 || run.marks.some((mark) => typeof mark !== "string" || mark.length > 64) ||
107
+ (run.href !== undefined && (typeof run.href !== "string" || run.href.length > 4096))) {
108
+ throw new TypeError("Invalid semantic run");
109
+ }
110
+ characters += run.text.length + (run.href?.length || 0) +
111
+ run.marks.reduce((sum, mark) => sum + mark.length, 0);
112
+ }
113
+ if (characters > ctx.limits.maxCharacters * 2) throw new ComparisonLimit("characters");
114
+ if (block.runs.map((run) => run.text).join("") !== block.text) {
115
+ throw new TypeError("Semantic text does not match runs");
116
+ }
117
+ }
118
+ }
119
+
120
+ function result(mode, ctx, limitations) {
121
+ return {
122
+ version: DIFF_VERSION, mode, status: "complete", changes: [], rows: [], hunks: [],
123
+ counts: emptyCounts(), limits: { ...ctx.limits }, limitations: [...new Set(limitations)].sort(),
124
+ };
125
+ }
126
+
127
+ function failed(mode, ctx, error) {
128
+ if (!(error instanceof ComparisonLimit) && !(error instanceof TypeError)) throw error;
129
+ return {
130
+ version: DIFF_VERSION, mode,
131
+ status: error instanceof ComparisonLimit ? "limited" : "unavailable",
132
+ changes: [], rows: [], hunks: [], counts: null, limits: { ...(ctx?.limits || DIFF_LIMITS) },
133
+ limitations: [error instanceof ComparisonLimit ? `comparison-limit:${error.message}` : "invalid-input"],
134
+ };
135
+ }
136
+
137
+ function append(output, ctx, change, beforeIndex, afterIndex) {
138
+ ctx.check();
139
+ if (output.changes.length >= ctx.limits.maxChanges) throw new ComparisonLimit("changes");
140
+ change.id = `change-${beforeIndex ?? "none"}-${afterIndex ?? "none"}-${hash(stable(change))}`;
141
+ output.changes.push(change);
142
+ output.counts[change.kind]++;
143
+ output.counts.total++;
144
+ return change;
145
+ }
146
+
147
+ function appendRow(output, ctx, row) {
148
+ ctx.check();
149
+ if (output.rows.length >= ctx.limits.maxRows) throw new ComparisonLimit("rows");
150
+ output.rows.push({
151
+ id: `row-${row.beforeIndex ?? "none"}-${row.afterIndex ?? "none"}`,
152
+ ...row,
153
+ });
154
+ }
155
+
156
+ function finish(output, ctx) {
157
+ const order = new Map();
158
+ for (const [index, row] of output.rows.entries()) {
159
+ if (row.changeId && !order.has(row.changeId)) order.set(row.changeId, index);
160
+ const kind = row.kind === "unchanged" ? "context" : "changes";
161
+ const previous = output.hunks.at(-1);
162
+ if (previous?.kind === kind) previous.count++;
163
+ else output.hunks.push({ id: `hunk-${row.id}`, kind, start: index, count: 1 });
164
+ }
165
+ output.changes.sort((a, b) => order.get(a.id) - order.get(b.id));
166
+ if (JSON.stringify(output).length > ctx.limits.maxResponseCharacters) {
167
+ throw new ComparisonLimit("response");
168
+ }
169
+ ctx.check();
170
+ return output;
171
+ }
172
+
173
+ function groups(blocks, key) {
174
+ const map = new Map();
175
+ blocks.forEach((block, index) => {
176
+ const value = key(block);
177
+ if (!value) return;
178
+ if (!map.has(value)) map.set(value, []);
179
+ map.get(value).push(index);
180
+ });
181
+ return map;
182
+ }
183
+
184
+ /**
185
+ * Compare equivalent semantic representations. Local block IDs are never
186
+ * matching evidence. Repeated, moved and contextual matches cannot navigate.
187
+ * Optional third-argument budgets can only lower the documented hard ceilings.
188
+ *
189
+ * Version 2 adds monotonic rows covering each endpoint exactly once. Row
190
+ * indices are zero-based; null denotes a gap. changeId refers to changes,
191
+ * while moveId links two positional rows to a single counted relocation.
192
+ * Context/changes hunks partition all rows with zero-based start and count.
193
+ * Neither context collapsing nor row grouping can change the actual counts.
194
+ */
195
+ export function compareSemanticSnapshots(before, after, options = {}) {
196
+ let ctx;
197
+ try {
198
+ ctx = context(options);
199
+ validateSnapshot(before, ctx);
200
+ validateSnapshot(after, ctx);
201
+ const output = result("semantic", ctx, [...before.limitations, ...after.limitations]);
202
+ const oldBlocks = before.blocks;
203
+ const newBlocks = after.blocks;
204
+ const oldKeys = oldBlocks.map(content);
205
+ const newKeys = newBlocks.map(content);
206
+ const oldContent = groups(oldBlocks, (block) => block.text);
207
+ const newContent = groups(newBlocks, (block) => block.text);
208
+ const oldIds = groups(oldBlocks, (block) => block.attributes.id);
209
+ const newIds = groups(newBlocks, (block) => block.attributes.id);
210
+ const oldSignatures = groups(oldBlocks, content);
211
+ const newSignatures = groups(newBlocks, content);
212
+ const pairs = new Map();
213
+ const used = new Set();
214
+ const pair = (i, j, confidence, evidence) => {
215
+ if (pairs.has(i) || used.has(j)) return;
216
+ pairs.set(i, {
217
+ i, j, confidence, evidence,
218
+ moved: stable(oldBlocks[i].path) !== stable(newBlocks[j].path),
219
+ });
220
+ used.add(j);
221
+ };
222
+ for (const [id, indices] of oldIds) {
223
+ const candidates = newIds.get(id);
224
+ if (indices.length === 1 && candidates?.length === 1) {
225
+ pair(indices[0], candidates[0], "exact", "authored-id");
226
+ }
227
+ }
228
+ for (const [text, indices] of oldContent) {
229
+ const candidates = newContent.get(text);
230
+ if (indices.length === 1 && candidates?.length === 1) {
231
+ const oldId = oldBlocks[indices[0]].attributes.id;
232
+ const newId = newBlocks[candidates[0]].attributes.id;
233
+ if (!oldId || !newId || oldId === newId) pair(indices[0], candidates[0], "exact", "unique-text");
234
+ }
235
+ }
236
+ const oldRest = oldKeys.flatMap((key, i) => pairs.has(i) ? [] : [{ key, i }]);
237
+ const newRest = newKeys.flatMap((key, j) => used.has(j) ? [] : [{ key, j }]);
238
+ const sequence = diffArrays(oldRest, newRest, {
239
+ ...ctx.primitives(), comparator: (a, b) => a.key === b.key,
240
+ });
241
+ if (!sequence) throw new ComparisonLimit("sequence-diff");
242
+ let oldAt = 0;
243
+ let newAt = 0;
244
+ for (const part of sequence) {
245
+ if (!part.added && !part.removed) {
246
+ for (let k = 0; k < part.count; k++) {
247
+ const { i, key } = oldRest[oldAt + k];
248
+ const { j } = newRest[newAt + k];
249
+ const repeated = oldSignatures.get(key)?.length > 1 || newSignatures.get(key)?.length > 1;
250
+ pair(i, j, repeated ? "ambiguous" : "context", repeated ? "repeated-content" : "sequence");
251
+ }
252
+ }
253
+ if (!part.added) oldAt += part.count;
254
+ if (!part.removed) newAt += part.count;
255
+ }
256
+ // Crossing matches identify relocation without treating inserted siblings as moves.
257
+ const ordered = [...pairs.values()].sort((a, b) => a.i - b.i);
258
+ let maximum = -1;
259
+ for (const item of ordered) {
260
+ if (item.j < maximum) item.moved = true;
261
+ maximum = Math.max(maximum, item.j);
262
+ }
263
+ let minimum = Infinity;
264
+ for (let i = ordered.length - 1; i >= 0; i--) {
265
+ if (ordered[i].j > minimum) ordered[i].moved = true;
266
+ minimum = Math.min(minimum, ordered[i].j);
267
+ }
268
+ const anchors = [
269
+ { i: -1, j: -1 },
270
+ ...ordered.filter((item) => !item.moved),
271
+ { i: oldBlocks.length, j: newBlocks.length },
272
+ ];
273
+ for (let a = 1; a < anchors.length; a++) {
274
+ ctx.check();
275
+ const left = anchors[a - 1];
276
+ const right = anchors[a];
277
+ const oldGap = [];
278
+ const newGap = [];
279
+ for (let i = left.i + 1; i < right.i; i++) if (!pairs.has(i)) oldGap.push(i);
280
+ for (let j = left.j + 1; j < right.j; j++) if (!used.has(j)) newGap.push(j);
281
+ if (oldGap.length !== newGap.length) continue;
282
+ for (let k = 0; k < oldGap.length; k++) {
283
+ const i = oldGap[k];
284
+ const j = newGap[k];
285
+ const oldBlock = oldBlocks[i];
286
+ const newBlock = newBlocks[j];
287
+ const oldId = oldBlock.attributes.id;
288
+ const newId = newBlock.attributes.id;
289
+ if (oldBlock.tag !== newBlock.tag || (oldId && newId && oldId !== newId) ||
290
+ oldContent.get(oldBlock.text)?.length > 1 ||
291
+ newContent.get(newBlock.text)?.length > 1) continue;
292
+ pair(i, j, "context", "neighbor-context");
293
+ }
294
+ }
295
+ const makeChange = (i, j, matched) => {
296
+ const oldBlock = i === null ? null : oldBlocks[i];
297
+ const newBlock = j === null ? null : newBlocks[j];
298
+ const kind = !oldBlock ? "added" : !newBlock ? "removed" : "modified";
299
+ const fields = [];
300
+ if (!oldBlock || !newBlock) fields.push("block");
301
+ else {
302
+ if (oldBlock.text !== newBlock.text) fields.push("text");
303
+ if (oldBlock.tag !== newBlock.tag || stable(oldBlock.path) !== stable(newBlock.path)) fields.push("structure");
304
+ if (stable(oldBlock.attributes) !== stable(newBlock.attributes)) fields.push("attributes");
305
+ if (stable(oldBlock.runs) !== stable(newBlock.runs)) fields.push("runs");
306
+ if (matched.moved) fields.push("order");
307
+ }
308
+ const confidence = matched?.moved ? "ambiguous" : (matched?.confidence || "context");
309
+ return append(output, ctx, {
310
+ kind, before: oldBlock?.text || "", after: newBlock?.text || "",
311
+ beforeBlock: oldBlock, afterBlock: newBlock, confidence,
312
+ evidence: matched?.evidence || "unmatched",
313
+ navigation: confidence === "exact" && newBlock?.selector
314
+ ? { selector: newBlock.selector, blockId: newBlock.id } : null,
315
+ fields, segments: ctx.words(oldBlock?.text || "", newBlock?.text || ""),
316
+ }, i, j);
317
+ };
318
+ const oldChanges = new Map();
319
+ const newChanges = new Map();
320
+ for (let i = 0; i < oldBlocks.length; i++) {
321
+ const matched = pairs.get(i);
322
+ let change;
323
+ if (!matched) change = makeChange(i, null, null);
324
+ else if (oldKeys[i] !== newKeys[matched.j] || matched.moved) change = makeChange(i, matched.j, matched);
325
+ if (change) {
326
+ oldChanges.set(i, change);
327
+ if (matched) newChanges.set(matched.j, change);
328
+ }
329
+ }
330
+ for (let j = 0; j < newBlocks.length; j++) {
331
+ if (!used.has(j)) newChanges.set(j, makeChange(null, j, null));
332
+ }
333
+ const add = (i, j) => {
334
+ const match = i === null ? null : pairs.get(i);
335
+ const change = oldChanges.get(i) || newChanges.get(j);
336
+ appendRow(output, ctx, {
337
+ kind: i === null ? "added" : j === null ? "removed" : change ? "modified" : "unchanged",
338
+ beforeIndex: i, afterIndex: j,
339
+ beforeBlock: i === null ? null : oldBlocks[i],
340
+ afterBlock: j === null ? null : newBlocks[j],
341
+ changeId: change?.id || null,
342
+ confidence: change?.confidence || match?.confidence || "context",
343
+ evidence: change?.evidence || match?.evidence || "unmatched",
344
+ segments: change?.segments || [],
345
+ ...(change?.fields.includes("order") ? { moveId: change.id } : {}),
346
+ });
347
+ };
348
+ // Only monotonic matches share a row. Relocations occupy both original
349
+ // positions, linked to the same counted change, never crossing columns.
350
+ let i = 0;
351
+ let j = 0;
352
+ for (const match of [...pairs.values()].filter((item) => !item.moved).sort((a, b) => a.i - b.i)) {
353
+ while (i < match.i) add(i++, null);
354
+ while (j < match.j) add(null, j++);
355
+ add(i++, j++);
356
+ }
357
+ while (i < oldBlocks.length) add(i++, null);
358
+ while (j < newBlocks.length) add(null, j++);
359
+ return finish(output, ctx);
360
+ } catch (error) {
361
+ return failed("semantic", ctx, error);
362
+ }
363
+ }
364
+
365
+ /** Raw-source hunks are inert text. They are not DOM navigation targets. */
366
+ export function compareSources(before, after, options = {}) {
367
+ let ctx;
368
+ try {
369
+ ctx = context(options);
370
+ if (typeof before !== "string" || typeof after !== "string") throw new TypeError("Sources must be strings");
371
+ if (before.length > ctx.limits.maxCharacters || after.length > ctx.limits.maxCharacters) {
372
+ throw new ComparisonLimit("characters");
373
+ }
374
+ if ((before.match(/\n/g)?.length || 0) + (after.match(/\n/g)?.length || 0) > ctx.limits.maxTokens) {
375
+ throw new ComparisonLimit("lines");
376
+ }
377
+ const output = result("source", ctx, ["source-is-not-rendered-dom"]);
378
+ output.endOfFile = {
379
+ before: { empty: !before.length, newline: before.endsWith("\n") },
380
+ after: { empty: !after.length, newline: after.endsWith("\n") },
381
+ };
382
+ const lines = (text) => text.match(/[^\n]*\n|[^\n]+$/g) || [];
383
+ const addLine = (oldText, newText, i, j, change = null) => {
384
+ appendRow(output, ctx, {
385
+ kind: oldText === null ? "added" : newText === null ? "removed" : change ? "modified" : "unchanged",
386
+ beforeIndex: i === null ? null : i - 1, afterIndex: j === null ? null : j - 1,
387
+ beforeBlock: oldText === null ? null : { text: oldText, startLine: i, endLine: i },
388
+ afterBlock: newText === null ? null : { text: newText, startLine: j, endLine: j },
389
+ changeId: change?.id || null, confidence: "exact", evidence: "source-lines",
390
+ segments: change ? ctx.words(oldText || "", newText || "") : [],
391
+ });
392
+ };
393
+ const parts = diffLines(before, after, ctx.primitives());
394
+ if (!parts) throw new ComparisonLimit("line-diff");
395
+ let oldLine = 1;
396
+ let newLine = 1;
397
+ for (let p = 0; p < parts.length; p++) {
398
+ const part = parts[p];
399
+ if (!part.added && !part.removed) {
400
+ for (const [offset, text] of lines(part.value).entries()) {
401
+ addLine(text, text, oldLine + offset, newLine + offset);
402
+ }
403
+ oldLine += part.count;
404
+ newLine += part.count;
405
+ continue;
406
+ }
407
+ let oldValue = "";
408
+ let newValue = "";
409
+ let oldCount = 0;
410
+ let newCount = 0;
411
+ do {
412
+ const current = parts[p];
413
+ if (current.removed) { oldValue += current.value; oldCount += current.count; }
414
+ if (current.added) { newValue += current.value; newCount += current.count; }
415
+ p++;
416
+ } while (p < parts.length && (parts[p].added || parts[p].removed));
417
+ p--;
418
+ const kind = oldCount && newCount ? "modified" : oldCount ? "removed" : "added";
419
+ const change = append(output, ctx, {
420
+ kind, before: oldValue, after: newValue,
421
+ beforeBlock: oldCount ? { startLine: oldLine, endLine: oldLine + oldCount - 1, text: oldValue } : null,
422
+ afterBlock: newCount ? { startLine: newLine, endLine: newLine + newCount - 1, text: newValue } : null,
423
+ confidence: "exact", evidence: "source-lines", navigation: null,
424
+ fields: ["source"], segments: ctx.words(oldValue, newValue),
425
+ }, oldLine, newLine);
426
+ const oldLines = lines(oldValue);
427
+ const newLines = lines(newValue);
428
+ for (let offset = 0; offset < Math.max(oldLines.length, newLines.length); offset++) {
429
+ addLine(oldLines[offset] ?? null, newLines[offset] ?? null,
430
+ offset < oldLines.length ? oldLine + offset : null,
431
+ offset < newLines.length ? newLine + offset : null, change);
432
+ }
433
+ oldLine += oldCount;
434
+ newLine += newCount;
435
+ }
436
+ return finish(output, ctx);
437
+ } catch (error) {
438
+ return failed("source", ctx, error);
439
+ }
440
+ }
@@ -0,0 +1,219 @@
1
+ export const REVISION_SCHEMA_VERSION = 1;
2
+ export const HISTORY_SCHEMA_VERSION = 1;
3
+ export const COMPLETED_ROUNDS_TO_KEEP = 5;
4
+ export const CAPTURE_LEASE_MS = 60_000;
5
+ export const ROUND_FEEDBACK_STATUSES = Object.freeze(["queued", "delivered", "acknowledged", "superseded"]);
6
+ export const ROUND_CAPTURE_STATUSES = Object.freeze(["pending", "ready", "partial", "failed", "cancelled"]);
7
+ export const REVISION_LIMITS = Object.freeze({
8
+ sourceBytes: 8 * 1024 * 1024,
9
+ semanticBytes: 4 * 1024 * 1024,
10
+ totalBytes: 12 * 1024 * 1024,
11
+ blocks: 2_000,
12
+ textCharacters: 100_000,
13
+ totalTextCharacters: 1_000_000,
14
+ runsPerBlock: 1_000,
15
+ depth: 64,
16
+ attributeCharacters: 4_096,
17
+ selectorCharacters: 4_096,
18
+ });
19
+
20
+ const encoder = new TextEncoder();
21
+ const TAGS = new Set([
22
+ "article", "section", "main", "header", "footer", "aside", "nav", "div",
23
+ "h1", "h2", "h3", "h4", "h5", "h6", "p", "blockquote", "pre", "code",
24
+ "ul", "ol", "li", "dl", "dt", "dd", "table", "caption", "thead", "tbody",
25
+ "tfoot", "tr", "th", "td", "figure", "figcaption", "img", "a", "span",
26
+ "strong", "em", "b", "i", "u", "s", "del", "ins", "sub", "sup", "br", "hr",
27
+ "details", "summary", "body", "address", "button", "dialog", "fieldset",
28
+ ]);
29
+ const ATTRIBUTES = new Set([
30
+ "id", "href", "src", "alt", "title", "start", "reversed", "value",
31
+ "colspan", "rowspan", "scope", "headers", "type",
32
+ ]);
33
+ const MARKS = new Set([
34
+ "strong", "em", "underline", "strike", "delete", "insert",
35
+ "code", "kbd", "samp", "sub", "sup", "mark",
36
+ ]);
37
+
38
+ export function revisionError(message, code = "INVALID_REVISION") {
39
+ const status = code === "SNAPSHOT_TOO_LARGE" ? 413
40
+ : code === "CAPTURE_CONFLICT" || code === "CAPTURE_FINALIZED" ? 409
41
+ : code === "SNAPSHOT_CORRUPT" ? 500 : 400;
42
+ return Object.assign(new Error(message), { code, status });
43
+ }
44
+
45
+ export function utf8Bytes(value) {
46
+ return encoder.encode(value).byteLength;
47
+ }
48
+
49
+ export function normalizeRevisionLimits(value = {}) {
50
+ object(value, "snapshot limits");
51
+ const limits = { ...REVISION_LIMITS };
52
+ for (const [key, limit] of Object.entries(value)) {
53
+ if (!(key in limits) || !Number.isSafeInteger(limit) || limit < 0) throw revisionError("Invalid snapshot limit.");
54
+ limits[key] = limit;
55
+ }
56
+ return limits;
57
+ }
58
+
59
+ function object(value, name) {
60
+ if (!value || typeof value !== "object" || Array.isArray(value)) throw revisionError(`Invalid ${name}.`);
61
+ }
62
+
63
+ function boundedString(value, name, limit, { empty = true } = {}) {
64
+ if (typeof value !== "string" || (!empty && !value)) {
65
+ throw revisionError(`Invalid ${name}.`);
66
+ }
67
+ if (value.length > limit) throw revisionError(`${name} exceeds the size limit.`, "SNAPSHOT_TOO_LARGE");
68
+ return value;
69
+ }
70
+
71
+ function referenceUrl(value) {
72
+ if (/^(?:javascript|vbscript|data|blob):/i.test(value.replace(/[\u0000-\u0020\u007f]+/g, ""))) {
73
+ throw revisionError("Executable or embedded URLs cannot be captured.");
74
+ }
75
+ try {
76
+ const url = new URL(value, "https://doc-review.invalid/");
77
+ if (url.username || url.password) throw revisionError("URL credentials cannot be captured.");
78
+ } catch (error) {
79
+ if (error.code === "INVALID_REVISION") throw error;
80
+ throw revisionError("Invalid semantic URL.");
81
+ }
82
+ return value;
83
+ }
84
+
85
+ export function isRevisionId(value) {
86
+ return typeof value === "string" && /^v_[a-f0-9]{64}$/.test(value);
87
+ }
88
+
89
+ export function normalizeSemanticSnapshot(value, limits = REVISION_LIMITS) {
90
+ limits = normalizeRevisionLimits(limits);
91
+ object(value, "semantic snapshot");
92
+ if (value.version !== REVISION_SCHEMA_VERSION || !Array.isArray(value.blocks)) {
93
+ throw revisionError("Unsupported semantic snapshot schema.");
94
+ }
95
+ if (value.blocks.length > limits.blocks) throw revisionError("Semantic snapshot has too many blocks.", "SNAPSHOT_TOO_LARGE");
96
+ const ids = new Set();
97
+ let totalText = 0;
98
+ const blocks = value.blocks.map((block) => {
99
+ object(block, "semantic block");
100
+ if ("html" in block || "children" in block) throw revisionError("Semantic blocks must be flat and cannot contain HTML.");
101
+ const id = boundedString(block.id, "block id", 200, { empty: false });
102
+ if (ids.has(id)) throw revisionError("Duplicate semantic block id.");
103
+ ids.add(id);
104
+ if (!TAGS.has(block.tag)) throw revisionError("Unsupported semantic block tag.");
105
+ if (block.path !== undefined && typeof block.path !== "string" && !Array.isArray(block.path)) {
106
+ throw revisionError("Invalid block path.");
107
+ }
108
+ const normalized = {
109
+ id,
110
+ tag: block.tag,
111
+ text: boundedString(block.text, "block text", limits.textCharacters),
112
+ path: Array.isArray(block.path)
113
+ ? block.path.map((part) => boundedString(part, "block path entry", limits.selectorCharacters))
114
+ : block.path ? [boundedString(block.path, "block path", limits.selectorCharacters)] : [],
115
+ selector: boundedString(block.selector ?? "", "block selector", limits.selectorCharacters),
116
+ attributes: {},
117
+ runs: [],
118
+ };
119
+ if (Array.isArray(normalized.path) && normalized.path.length > limits.depth) {
120
+ throw revisionError("Semantic path exceeds the depth limit.", "SNAPSHOT_TOO_LARGE");
121
+ }
122
+ totalText += normalized.text.length;
123
+ if (totalText > limits.totalTextCharacters) throw revisionError("Semantic text exceeds the size limit.", "SNAPSHOT_TOO_LARGE");
124
+ if (block.parentId !== undefined && block.parentId !== null) {
125
+ normalized.parentId = boundedString(block.parentId, "parent id", 200, { empty: false });
126
+ if (normalized.parentId === id || !ids.has(normalized.parentId)) {
127
+ throw revisionError("Semantic parents must precede their children.");
128
+ }
129
+ }
130
+ if (block.attrs !== undefined && block.attributes !== undefined) throw revisionError("Ambiguous semantic attributes.");
131
+ const attributes = block.attributes !== undefined ? block.attributes : block.attrs;
132
+ if (attributes !== undefined) {
133
+ object(attributes, "block attributes");
134
+ normalized.attributes = {};
135
+ for (const name of Object.keys(attributes).sort()) {
136
+ if (!ATTRIBUTES.has(name) || (name === "value" && block.tag !== "li") ||
137
+ (name === "type" && block.tag !== "button")) throw revisionError("Unsupported semantic attribute.");
138
+ const attribute = boundedString(attributes[name], "semantic attribute", limits.attributeCharacters);
139
+ normalized.attributes[name] = name === "src" || name === "href" ? referenceUrl(attribute) : attribute;
140
+ }
141
+ }
142
+ if (block.runs !== undefined) {
143
+ if (!Array.isArray(block.runs)) throw revisionError("Invalid inline runs.");
144
+ if (block.runs.length > limits.runsPerBlock) throw revisionError("Too many inline runs.", "SNAPSHOT_TOO_LARGE");
145
+ let runText = 0;
146
+ normalized.runs = block.runs.map((run) => {
147
+ object(run, "inline run");
148
+ if ("html" in run || !Array.isArray(run.marks) || run.marks.length > MARKS.size ||
149
+ run.marks.some((mark) => !MARKS.has(mark))) throw revisionError("Invalid inline formatting.");
150
+ const text = boundedString(run.text, "inline text", limits.textCharacters);
151
+ runText += text.length;
152
+ if (runText > limits.textCharacters) throw revisionError("Inline text exceeds the size limit.", "SNAPSHOT_TOO_LARGE");
153
+ return {
154
+ text,
155
+ marks: [...new Set(run.marks)].sort(),
156
+ ...(run.href !== undefined ? { href: referenceUrl(boundedString(run.href, "inline link", limits.attributeCharacters)) } : {}),
157
+ };
158
+ });
159
+ if (normalized.runs.map((run) => run.text).join("") !== normalized.text) {
160
+ throw revisionError("Inline runs do not match the block text.");
161
+ }
162
+ } else if (normalized.text) normalized.runs = [{ text: normalized.text, marks: [] }];
163
+ return normalized;
164
+ });
165
+ const snapshot = { version: REVISION_SCHEMA_VERSION, blocks, limitations: [] };
166
+ if (value.limitations !== undefined) {
167
+ if (!Array.isArray(value.limitations) || value.limitations.length > 100) throw revisionError("Invalid semantic limitations.");
168
+ snapshot.limitations = [...new Set(value.limitations.map((code) => boundedString(code, "limitation", 500)))];
169
+ }
170
+ if (utf8Bytes(JSON.stringify(snapshot)) > limits.semanticBytes) {
171
+ throw revisionError("Semantic snapshot exceeds the size limit.", "SNAPSHOT_TOO_LARGE");
172
+ }
173
+ return snapshot;
174
+ }
175
+
176
+ export function normalizeCaptureProvenance(value = {}) {
177
+ object(value, "capture provenance");
178
+ const result = {};
179
+ for (const name of ["sessionId", "pageKey", "sourceHash", "sourceRevisionId", "captureId"]) {
180
+ if (value[name] !== undefined) result[name] = boundedString(value[name], name, 200);
181
+ }
182
+ if (value.generation !== undefined) {
183
+ if (!Number.isSafeInteger(value.generation) || value.generation < 0) throw revisionError("Invalid capture generation.");
184
+ result.generation = value.generation;
185
+ }
186
+ if (value.feedbackOnlyEdits !== undefined) result.feedbackOnlyEdits = value.feedbackOnlyEdits === true;
187
+ if (value.trustedInteractive !== undefined) {
188
+ if (typeof value.trustedInteractive !== "boolean") throw revisionError("Invalid trusted capture provenance.");
189
+ result.trustedInteractive = value.trustedInteractive;
190
+ }
191
+ return result;
192
+ }
193
+
194
+ export function normalizeHistoryTargets(targets) {
195
+ if (!Array.isArray(targets) || !targets.length || targets.length > 100) throw revisionError("Invalid history targets.");
196
+ const keys = new Set();
197
+ return targets.map((target) => {
198
+ object(target, "history target");
199
+ const key = boundedString(target.key, "document key", 200, { empty: false });
200
+ if (keys.has(key)) throw revisionError("Duplicate history target.");
201
+ keys.add(key);
202
+ if (target.baselineRevisionId !== undefined && !isRevisionId(target.baselineRevisionId)) {
203
+ throw revisionError("Invalid baseline revision ID.");
204
+ }
205
+ if (!target.baselineRevisionId && !target.baselineUnavailable) {
206
+ throw revisionError("A target needs a baseline or an explicit unavailable reason.");
207
+ }
208
+ if (target.baselineRevisionId && target.baselineUnavailable) throw revisionError("Baseline availability is ambiguous.");
209
+ return {
210
+ key,
211
+ ...(target.ownerSessionId !== undefined ? {
212
+ ownerSessionId: boundedString(target.ownerSessionId, "baseline owner session", 200, { empty: false }),
213
+ } : {}),
214
+ ...(target.baselineRevisionId ? { baselineRevisionId: target.baselineRevisionId } : {
215
+ baselineUnavailable: boundedString(target.baselineUnavailable, "baseline unavailable reason", 500, { empty: false }),
216
+ }),
217
+ };
218
+ });
219
+ }