@hydranium/core 1.0.0-next.95 → 1.0.0-next.96
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/index.d.ts +1 -0
- package/lib/index.d.ts.map +1 -1
- package/lib/index.js +1 -0
- package/lib/index.js.map +1 -1
- package/lib/langium/integrity/integrity-service.d.ts.map +1 -1
- package/lib/langium/integrity/integrity-service.js +8 -1
- package/lib/langium/integrity/integrity-service.js.map +1 -1
- package/lib/langium/language-module.d.ts +25 -0
- package/lib/langium/language-module.d.ts.map +1 -1
- package/lib/langium/language-module.js +14 -0
- package/lib/langium/language-module.js.map +1 -1
- package/lib/langium/model-service/model-service.d.ts +30 -18
- package/lib/langium/model-service/model-service.d.ts.map +1 -1
- package/lib/langium/model-service/model-service.js +41 -1
- package/lib/langium/model-service/model-service.js.map +1 -1
- package/lib/langium/residency/cst-residency-service.d.ts +4 -6
- package/lib/langium/residency/cst-residency-service.d.ts.map +1 -1
- package/lib/langium/residency/cst-residency-service.js.map +1 -1
- package/lib/langium/serialization/abstract-serializer.d.ts +11 -20
- package/lib/langium/serialization/abstract-serializer.d.ts.map +1 -1
- package/lib/langium/serialization/abstract-serializer.js +12 -21
- package/lib/langium/serialization/abstract-serializer.js.map +1 -1
- package/lib/langium/trivia/comment-preserver.d.ts +423 -0
- package/lib/langium/trivia/comment-preserver.d.ts.map +1 -0
- package/lib/langium/trivia/comment-preserver.js +906 -0
- package/lib/langium/trivia/comment-preserver.js.map +1 -0
- package/lib/langium/trivia/document-ending-preserver.d.ts +43 -0
- package/lib/langium/trivia/document-ending-preserver.d.ts.map +1 -0
- package/lib/langium/trivia/document-ending-preserver.js +48 -0
- package/lib/langium/trivia/document-ending-preserver.js.map +1 -0
- package/lib/langium/trivia/index.d.ts +14 -0
- package/lib/langium/trivia/index.d.ts.map +1 -0
- package/lib/langium/trivia/index.js +14 -0
- package/lib/langium/trivia/index.js.map +1 -0
- package/lib/langium/trivia/trivia-contribution.d.ts +37 -0
- package/lib/langium/trivia/trivia-contribution.d.ts.map +1 -0
- package/lib/langium/trivia/trivia-contribution.js +10 -0
- package/lib/langium/trivia/trivia-contribution.js.map +1 -0
- package/lib/langium/trivia/trivia-preserver.d.ts +50 -0
- package/lib/langium/trivia/trivia-preserver.d.ts.map +1 -0
- package/lib/langium/trivia/trivia-preserver.js +10 -0
- package/lib/langium/trivia/trivia-preserver.js.map +1 -0
- package/lib/langium/trivia/trivia-service.d.ts +70 -0
- package/lib/langium/trivia/trivia-service.d.ts.map +1 -0
- package/lib/langium/trivia/trivia-service.js +56 -0
- package/lib/langium/trivia/trivia-service.js.map +1 -0
- package/lib/testing/node/scratch-workspace.js +2 -2
- package/package.json +5 -5
- package/src/index.ts +1 -0
- package/src/langium/integrity/integrity-service.ts +9 -1
- package/src/langium/language-module.ts +38 -0
- package/src/langium/model-service/model-service.ts +58 -19
- package/src/langium/residency/cst-residency-service.ts +4 -6
- package/src/langium/serialization/abstract-serializer.ts +13 -32
- package/src/langium/trivia/comment-preserver.ts +1100 -0
- package/src/langium/trivia/document-ending-preserver.ts +55 -0
- package/src/langium/trivia/index.ts +14 -0
- package/src/langium/trivia/trivia-contribution.ts +39 -0
- package/src/langium/trivia/trivia-preserver.ts +54 -0
- package/src/langium/trivia/trivia-service.ts +99 -0
- package/src/testing/node/scratch-workspace.ts +2 -2
|
@@ -0,0 +1,906 @@
|
|
|
1
|
+
/********************************************************************************
|
|
2
|
+
* Copyright (c) 2026 CrossBreeze, EclipseSource and others.
|
|
3
|
+
*
|
|
4
|
+
* This program and the accompanying materials are made available under the
|
|
5
|
+
* terms of the MIT License which is available in the project root.
|
|
6
|
+
*
|
|
7
|
+
* SPDX-License-Identifier: MIT
|
|
8
|
+
********************************************************************************/
|
|
9
|
+
import { AstUtils, GrammarAST, GrammarUtils, isAstNode } from '@hydranium/langium';
|
|
10
|
+
function isComposite(node) {
|
|
11
|
+
return 'content' in node;
|
|
12
|
+
}
|
|
13
|
+
function tokenNameOf(node) {
|
|
14
|
+
return 'tokenType' in node ? node.tokenType.name : undefined;
|
|
15
|
+
}
|
|
16
|
+
/**
|
|
17
|
+
* Carries a document's comments across a write that re-serializes it from the
|
|
18
|
+
* AST: extract against the node each comment hangs off, then splice into the
|
|
19
|
+
* serializer's output, located by re-parsing it.
|
|
20
|
+
*
|
|
21
|
+
* **Emitting comments from inside a serializer instead drops them silently for
|
|
22
|
+
* most nodes.** A hand-written concrete-syntax serializer reaches its children
|
|
23
|
+
* through its own per-`$type` emitters, not through
|
|
24
|
+
* `AbstractSerializer.serializeNode`, so a hook on that seam is reached by some
|
|
25
|
+
* nodes and bypassed by the rest — and the ones it misses fail as absence,
|
|
26
|
+
* which no test notices unless it asserts on exact text.
|
|
27
|
+
*/
|
|
28
|
+
export class CommentPreserver {
|
|
29
|
+
services;
|
|
30
|
+
/**
|
|
31
|
+
* **Registering a second preserver under this id throws.** A subclass added
|
|
32
|
+
* ALONGSIDE the framework's — rather than replacing it by rebinding the
|
|
33
|
+
* `comments` contribution sub-key — collides here, and the throw surfaces
|
|
34
|
+
* from the first write rather than from server start, because the service is
|
|
35
|
+
* constructed lazily. Pass an `id` to run both.
|
|
36
|
+
*/
|
|
37
|
+
id;
|
|
38
|
+
label = 'Comments';
|
|
39
|
+
/** Default {@link CommentPreserverOptions.maxIsolatedEdits}. */
|
|
40
|
+
static DEFAULT_ISOLATED_EDITS = 24;
|
|
41
|
+
tracer;
|
|
42
|
+
maxIsolatedEdits;
|
|
43
|
+
commentTokenNames;
|
|
44
|
+
/** Set for the duration of one {@link apply}; see {@link cachedAnchorKey}. */
|
|
45
|
+
keyMemo;
|
|
46
|
+
/** Set for the duration of one {@link apply}; see {@link ambiguousRetainedKey}. */
|
|
47
|
+
listVerdicts;
|
|
48
|
+
constructor(services, options = {}) {
|
|
49
|
+
this.services = services;
|
|
50
|
+
this.id = options.id ?? 'comments';
|
|
51
|
+
this.maxIsolatedEdits = options.maxIsolatedEdits ?? CommentPreserver.DEFAULT_ISOLATED_EDITS;
|
|
52
|
+
this.tracer = services.shared.Tracer.for(options.logName ?? 'CommentPreserver');
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* Every comment terminal the grammar declares, by token name.
|
|
56
|
+
*
|
|
57
|
+
* Deliberately NOT `GrammarConfig.multilineCommentRules`, which Langium
|
|
58
|
+
* populates only with terminals whose regex spans lines — so a grammar that
|
|
59
|
+
* comments with `//` answers an empty list there and every one of its
|
|
60
|
+
* comments would be invisible to a capture keyed on it.
|
|
61
|
+
*/
|
|
62
|
+
getCommentTokenNames() {
|
|
63
|
+
if (!this.commentTokenNames) {
|
|
64
|
+
this.commentTokenNames = this.services.Grammar.rules
|
|
65
|
+
.filter(GrammarAST.isTerminalRule)
|
|
66
|
+
.filter(rule => GrammarUtils.isCommentTerminal(rule))
|
|
67
|
+
.map(rule => rule.name);
|
|
68
|
+
}
|
|
69
|
+
return this.commentTokenNames;
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* Stable identity for one AST node, used to find it again in the
|
|
73
|
+
* re-serialized text — or `undefined` when no stable identity exists.
|
|
74
|
+
*
|
|
75
|
+
* **Read through the `NameProvider`, so the anchor is what the grammar treats
|
|
76
|
+
* as IDENTITY rather than whatever sits on `name`.** A grammar whose `name`
|
|
77
|
+
* is a display label, carrying its identifier on another property, would
|
|
78
|
+
* otherwise anchor comments to a value users retitle freely. A simple
|
|
79
|
+
* retitle can be matched, but the real identifier remains the safer key.
|
|
80
|
+
* `NameProviderOptions.nameProperties` names that property.
|
|
81
|
+
*
|
|
82
|
+
* **Whatever an override returns must survive a sibling being inserted into
|
|
83
|
+
* the same list.** A positional key does not: every comment after the
|
|
84
|
+
* insertion point is then written against the following declaration — a move,
|
|
85
|
+
* which reads as text the author wrote there and which nothing downstream can
|
|
86
|
+
* detect. That is why an unidentified node answers `undefined` here rather
|
|
87
|
+
* than falling back to a container index, and why this cannot delegate to
|
|
88
|
+
* `ElementKeyProvider`, whose name-based strategy makes exactly that fallback:
|
|
89
|
+
* its keys are derived and consumed within one snapshot, where no position can
|
|
90
|
+
* have shifted underneath them.
|
|
91
|
+
*
|
|
92
|
+
* Overriding is the supported route for a grammar that identifies some nodes
|
|
93
|
+
* outside the naming surface altogether — by a cross-reference that is unique
|
|
94
|
+
* among its siblings, say. Capture and lookup both read this one definition,
|
|
95
|
+
* so an override cannot make those two disagree.
|
|
96
|
+
*
|
|
97
|
+
* **Matching a rename does not read it.** That comparison asks the
|
|
98
|
+
* `NameProvider` what changed, so a node identified only by an override is
|
|
99
|
+
* carried across a write that leaves its identity alone and dropped by one
|
|
100
|
+
* that rewrites it — never misplaced, but never followed either.
|
|
101
|
+
*/
|
|
102
|
+
anchorKey(node) {
|
|
103
|
+
const nameProvider = this.services.references.NameProvider;
|
|
104
|
+
const segments = [];
|
|
105
|
+
let current = node;
|
|
106
|
+
while (current?.$container) {
|
|
107
|
+
const name = nameProvider.getOwnName(current);
|
|
108
|
+
if (name === undefined || name.length === 0) {
|
|
109
|
+
return undefined;
|
|
110
|
+
}
|
|
111
|
+
segments.unshift(`${current.$type}#${this.escapeKeySegment(name)}`);
|
|
112
|
+
current = current.$container;
|
|
113
|
+
}
|
|
114
|
+
return segments.length === 0 ? `@root:${node.$type}` : segments.join('/');
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* Escape the characters {@link anchorKey} builds its keys out of, so a name
|
|
118
|
+
* containing one cannot spell a key another node also answers to — which
|
|
119
|
+
* would make two distinct nodes indistinguishable to the lookup.
|
|
120
|
+
*/
|
|
121
|
+
escapeKeySegment(name) {
|
|
122
|
+
return name.replace(/[\\#/]/g, character => `\\${character}`);
|
|
123
|
+
}
|
|
124
|
+
/**
|
|
125
|
+
* Collect every comment in `document`, each bound to the node it hangs off.
|
|
126
|
+
*
|
|
127
|
+
* Call this BEFORE the mutation that prompts the write. The anchors are AST
|
|
128
|
+
* objects, so a rule that renames one in place is reflected automatically;
|
|
129
|
+
* capturing afterwards would work too, but capturing first is what keeps the
|
|
130
|
+
* CST and the comments describing the same state.
|
|
131
|
+
*/
|
|
132
|
+
extract(document) {
|
|
133
|
+
// The capture reads the CST, which a residency policy may have shed.
|
|
134
|
+
// Restored rather than skipped: skipping would lose the whole file's
|
|
135
|
+
// comments for a document that had gone idle, silently and only under a
|
|
136
|
+
// shedding strategy.
|
|
137
|
+
this.services.shared.workspace.CstResidencyService.rehydrate(document);
|
|
138
|
+
const root = document.parseResult.value.$cstNode;
|
|
139
|
+
const commentTokens = this.getCommentTokenNames();
|
|
140
|
+
if (!root || commentTokens.length === 0) {
|
|
141
|
+
return { comments: [], sourceRoot: document.parseResult.value };
|
|
142
|
+
}
|
|
143
|
+
const source = document.textDocument.getText();
|
|
144
|
+
const comments = [];
|
|
145
|
+
this.captureFrom(root, source, commentTokens, comments);
|
|
146
|
+
return { comments, sourceRoot: document.parseResult.value };
|
|
147
|
+
}
|
|
148
|
+
/** Recursive half of {@link extract}: one composite node's content, then its children. */
|
|
149
|
+
captureFrom(node, source, commentTokens, into) {
|
|
150
|
+
if (!isComposite(node)) {
|
|
151
|
+
return;
|
|
152
|
+
}
|
|
153
|
+
const content = node.content;
|
|
154
|
+
content.forEach((child, index) => {
|
|
155
|
+
const tokenName = tokenNameOf(child);
|
|
156
|
+
if (child.hidden && tokenName !== undefined && commentTokens.includes(tokenName)) {
|
|
157
|
+
into.push(this.classify(child, index, node, source));
|
|
158
|
+
}
|
|
159
|
+
this.captureFrom(child, source, commentTokens, into);
|
|
160
|
+
});
|
|
161
|
+
}
|
|
162
|
+
/**
|
|
163
|
+
* Decide which node a comment belongs to, and how it sat against it.
|
|
164
|
+
*
|
|
165
|
+
* A comment on the same line the previous node ENDED on is that node's
|
|
166
|
+
* trailing comment — checked first, because by source order it also precedes
|
|
167
|
+
* whatever comes next, and reading it as the NEXT node's leading comment is
|
|
168
|
+
* what relocates an end-of-line note onto the following declaration.
|
|
169
|
+
*/
|
|
170
|
+
classify(comment, index, container, source) {
|
|
171
|
+
const content = container.content;
|
|
172
|
+
const ownsOwnNode = (candidate) => candidate?.astNode !== undefined && candidate.astNode !== container.astNode;
|
|
173
|
+
const realSibling = (from, step) => {
|
|
174
|
+
for (let scan = from; scan >= 0 && scan < content.length; scan += step) {
|
|
175
|
+
if (!content[scan].hidden) {
|
|
176
|
+
return content[scan];
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
return undefined;
|
|
180
|
+
};
|
|
181
|
+
const previous = realSibling(index - 1, -1);
|
|
182
|
+
const next = realSibling(index + 1, 1);
|
|
183
|
+
// Measured to whatever comes NEXT IN SOURCE, comment or not — never to the
|
|
184
|
+
// next real node, which would skip the rest of a comment block and make
|
|
185
|
+
// every line of it re-emit the whole gap that follows the block.
|
|
186
|
+
const following = content[index + 1];
|
|
187
|
+
const blankLinesAfter = following === undefined ? 0 : this.countBlankLines(source.slice(comment.end, following.offset));
|
|
188
|
+
const preceding = index > 0 ? content[index - 1] : undefined;
|
|
189
|
+
const blankLinesBefore = preceding === undefined ? 0 : this.countBlankLines(source.slice(preceding.end, comment.offset));
|
|
190
|
+
// Only a comment that OWNS its line can be reindented — one sharing a line
|
|
191
|
+
// with code keeps whatever spacing that line gives it.
|
|
192
|
+
const lineStart = source.lastIndexOf('\n', comment.offset - 1) + 1;
|
|
193
|
+
const beforeOnLine = source.slice(lineStart, comment.offset);
|
|
194
|
+
const sourceIndent = beforeOnLine.trim().length === 0 ? beforeOnLine : undefined;
|
|
195
|
+
const text = this.normalizeLineEndings(comment.text);
|
|
196
|
+
const sharesLineWithPrevious = previous !== undefined && !source.slice(previous.end, comment.offset).includes('\n');
|
|
197
|
+
if (sharesLineWithPrevious) {
|
|
198
|
+
// On the same line as whatever precedes it, so it annotates that line.
|
|
199
|
+
// When the preceding sibling is the CONTAINER's own syntax — an opening
|
|
200
|
+
// brace, a name token, a keyword — the line is the container's header
|
|
201
|
+
// and the comment belongs to the container, NOT to the first member
|
|
202
|
+
// that happens to follow. Reading it as that member's leading comment
|
|
203
|
+
// moves a note about the declaration onto its first child.
|
|
204
|
+
return ownsOwnNode(previous)
|
|
205
|
+
? { text, anchor: previous.astNode, placement: 'trailing', blankLinesAfter: 0, blankLinesBefore: 0 }
|
|
206
|
+
: { text, anchor: container.astNode, placement: 'trailingOnContainer', blankLinesAfter: 0, blankLinesBefore: 0 };
|
|
207
|
+
}
|
|
208
|
+
if (ownsOwnNode(next)) {
|
|
209
|
+
return { text, anchor: next.astNode, placement: 'leading', blankLinesAfter, blankLinesBefore: 0, sourceIndent };
|
|
210
|
+
}
|
|
211
|
+
if (ownsOwnNode(previous)) {
|
|
212
|
+
return { text, anchor: previous.astNode, placement: 'afterNode', blankLinesAfter: 0, blankLinesBefore, sourceIndent };
|
|
213
|
+
}
|
|
214
|
+
if (previous === undefined) {
|
|
215
|
+
return { text, anchor: container.astNode, placement: 'atContainerStart', blankLinesAfter, blankLinesBefore: 0, sourceIndent };
|
|
216
|
+
}
|
|
217
|
+
// Surrounded by the container's own syntax on both sides. A further
|
|
218
|
+
// sibling after it means the comment sits INSIDE the construct — between
|
|
219
|
+
// the braces of an empty body, or partway through a header — so it stays
|
|
220
|
+
// with the container rather than being pushed past its closing syntax,
|
|
221
|
+
// which would move it out to the enclosing scope. Only a comment with
|
|
222
|
+
// nothing after it at all actually trails the container.
|
|
223
|
+
const placement = next === undefined ? 'atContainerEnd' : 'trailingOnContainer';
|
|
224
|
+
return { text, anchor: container.astNode, placement, blankLinesAfter, blankLinesBefore, sourceIndent };
|
|
225
|
+
}
|
|
226
|
+
/**
|
|
227
|
+
* Comment text with CRLF reduced to LF.
|
|
228
|
+
*
|
|
229
|
+
* Serializers emit LF, so splicing a comment captured from a CRLF document
|
|
230
|
+
* verbatim puts lone CR bytes in the middle of LF-terminated lines.
|
|
231
|
+
*/
|
|
232
|
+
normalizeLineEndings(text) {
|
|
233
|
+
return text.includes('\r') ? text.replace(/\r\n/g, '\n') : text;
|
|
234
|
+
}
|
|
235
|
+
/** Blank lines in a run of whitespace — one fewer than its newlines. */
|
|
236
|
+
countBlankLines(gap) {
|
|
237
|
+
return Math.max(0, (gap.match(/\n/g)?.length ?? 0) - 1);
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* Splice the captured comments back into `serialized`.
|
|
241
|
+
*
|
|
242
|
+
* A comment whose anchor the write deleted, or whose anchor has no stable
|
|
243
|
+
* key, is dropped — counted at `debug`, because a silent drop is the failure
|
|
244
|
+
* this preserver exists to remove and an unexplained one just moves it.
|
|
245
|
+
*/
|
|
246
|
+
apply(serialized, trivia, uri) {
|
|
247
|
+
if (trivia.comments.length === 0) {
|
|
248
|
+
return serialized;
|
|
249
|
+
}
|
|
250
|
+
// Both memos are scoped to this call and not to the instance, because an
|
|
251
|
+
// in-place write changes what a node's key IS: a repair that renames a
|
|
252
|
+
// node between two writes would otherwise be answered from the first.
|
|
253
|
+
this.keyMemo = new Map();
|
|
254
|
+
this.listVerdicts = new Map();
|
|
255
|
+
try {
|
|
256
|
+
return this.applyComments(serialized, trivia, uri);
|
|
257
|
+
}
|
|
258
|
+
finally {
|
|
259
|
+
this.keyMemo = undefined;
|
|
260
|
+
this.listVerdicts = undefined;
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
/** {@link apply}'s body, inside the per-write memos it sets up. */
|
|
264
|
+
applyComments(serialized, trivia, uri) {
|
|
265
|
+
const spliced = this.spliceComments(serialized, trivia, uri);
|
|
266
|
+
// **Never hand back text the grammar cannot read.** A splice edits syntax
|
|
267
|
+
// it did not produce, and a line comment in particular ends whatever
|
|
268
|
+
// shares its line — so a placement that is merely wrong becomes a file
|
|
269
|
+
// that no longer parses, written to disk by an integrity repair with no
|
|
270
|
+
// user gesture. Dropping the comments is recoverable; corrupting the
|
|
271
|
+
// document is not.
|
|
272
|
+
if (spliced !== serialized && this.parse(spliced, uri) === undefined) {
|
|
273
|
+
this.tracer.withUri(uri.toString()).warn('Reattaching comments produced text that does not parse; writing without them');
|
|
274
|
+
return serialized;
|
|
275
|
+
}
|
|
276
|
+
return spliced;
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* {@link anchorKey}, answered once per node per write.
|
|
280
|
+
*
|
|
281
|
+
* The key walks `$container` to the root building a segment each step, and
|
|
282
|
+
* every stage of a write asks about the same nodes — locating spans, matching
|
|
283
|
+
* a rename, judging a key the output kept. Recomputing makes each of those
|
|
284
|
+
* stages cost nodes × depth, and the stages that scan a sibling list do it
|
|
285
|
+
* once per comment.
|
|
286
|
+
*/
|
|
287
|
+
cachedAnchorKey(node) {
|
|
288
|
+
const memo = this.keyMemo;
|
|
289
|
+
if (memo === undefined) {
|
|
290
|
+
return this.anchorKey(node);
|
|
291
|
+
}
|
|
292
|
+
if (!memo.has(node)) {
|
|
293
|
+
memo.set(node, this.anchorKey(node));
|
|
294
|
+
}
|
|
295
|
+
return memo.get(node);
|
|
296
|
+
}
|
|
297
|
+
/**
|
|
298
|
+
* Parse `text` as a throwaway document, or `undefined` when it is not
|
|
299
|
+
* readable — by reported error, or by a throw from a URI that routes to no
|
|
300
|
+
* services. Does NOT register the result, so `LangiumDocuments` is untouched.
|
|
301
|
+
*/
|
|
302
|
+
parse(text, uri) {
|
|
303
|
+
let document;
|
|
304
|
+
try {
|
|
305
|
+
document = this.services.shared.workspace.LangiumDocumentFactory.fromString(text, uri);
|
|
306
|
+
}
|
|
307
|
+
catch {
|
|
308
|
+
return undefined;
|
|
309
|
+
}
|
|
310
|
+
return document.parseResult.parserErrors.length > 0 || document.parseResult.lexerErrors.length > 0 ? undefined : document;
|
|
311
|
+
}
|
|
312
|
+
/** The splice itself, run before {@link apply}'s parse check. */
|
|
313
|
+
spliceComments(serialized, trivia, uri) {
|
|
314
|
+
const located = this.locate(serialized, uri);
|
|
315
|
+
if (located === undefined) {
|
|
316
|
+
return serialized;
|
|
317
|
+
}
|
|
318
|
+
const sourceOwners = new Map();
|
|
319
|
+
const sourceCollisions = new Set();
|
|
320
|
+
const liveAnchors = new Set();
|
|
321
|
+
for (const node of [trivia.sourceRoot, ...AstUtils.streamAllContents(trivia.sourceRoot)]) {
|
|
322
|
+
liveAnchors.add(node);
|
|
323
|
+
const key = this.cachedAnchorKey(node);
|
|
324
|
+
if (key === undefined) {
|
|
325
|
+
continue;
|
|
326
|
+
}
|
|
327
|
+
const owner = sourceOwners.get(key);
|
|
328
|
+
if (owner !== undefined && owner !== node) {
|
|
329
|
+
sourceCollisions.add(key);
|
|
330
|
+
}
|
|
331
|
+
else {
|
|
332
|
+
sourceOwners.set(key, node);
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
const needsRename = trivia.comments.some(comment => {
|
|
336
|
+
const key = this.cachedAnchorKey(comment.anchor);
|
|
337
|
+
return key !== undefined && !located.has(key);
|
|
338
|
+
});
|
|
339
|
+
const renamed = needsRename
|
|
340
|
+
? this.matchRenamedAnchors(trivia.sourceRoot, serialized, uri, located, sourceOwners, sourceCollisions)
|
|
341
|
+
: new Map();
|
|
342
|
+
const edits = [];
|
|
343
|
+
let dropped = 0;
|
|
344
|
+
trivia.comments.forEach((comment, order) => {
|
|
345
|
+
const key = this.cachedAnchorKey(comment.anchor);
|
|
346
|
+
const destination = key === undefined ? undefined : located.has(key) ? key : this.renamedKey(comment.anchor, key, renamed);
|
|
347
|
+
const retained = destination === key ? located.get(key) : undefined;
|
|
348
|
+
const span = key === undefined ||
|
|
349
|
+
!liveAnchors.has(comment.anchor) ||
|
|
350
|
+
sourceCollisions.has(key) ||
|
|
351
|
+
destination === undefined ||
|
|
352
|
+
(retained !== undefined && this.ambiguousRetainedKey(comment.anchor, retained.owner))
|
|
353
|
+
? undefined
|
|
354
|
+
: located.get(destination);
|
|
355
|
+
if (span === undefined) {
|
|
356
|
+
dropped++;
|
|
357
|
+
return;
|
|
358
|
+
}
|
|
359
|
+
const edit = this.editFor(comment, span, serialized);
|
|
360
|
+
edits.push({
|
|
361
|
+
...edit,
|
|
362
|
+
order,
|
|
363
|
+
inline: this.needsReadBack(comment, span, serialized) ? { text: comment.text, key: destination } : undefined
|
|
364
|
+
});
|
|
365
|
+
});
|
|
366
|
+
if (dropped > 0) {
|
|
367
|
+
this.tracer.withUri(uri.toString()).debug(`Dropped ${dropped} comment(s): anchor deleted by the write, or carrying no stable key`);
|
|
368
|
+
}
|
|
369
|
+
// Stable by offset, then by capture order, so several comments landing on
|
|
370
|
+
// one anchor keep the order the author wrote them in.
|
|
371
|
+
edits.sort((left, right) => left.at - right.at || left.order - right.order);
|
|
372
|
+
// **This is what licenses the mid-line split** `editFor` makes for a
|
|
373
|
+
// `leading` comment whose anchor does not start its line. Parsing alone
|
|
374
|
+
// cannot license it: text can parse perfectly well having handed the
|
|
375
|
+
// comment to the following declaration instead. Re-extracting and
|
|
376
|
+
// requiring the same anchor back is the only evidence the comment still
|
|
377
|
+
// belongs where its author put it.
|
|
378
|
+
//
|
|
379
|
+
// A failed edit is dropped and the rest reassembled, since removing one
|
|
380
|
+
// changes what the others land in; two rounds, then the inline edits go
|
|
381
|
+
// as a set rather than iterating towards a text nothing has verified.
|
|
382
|
+
let remaining = edits;
|
|
383
|
+
let result = this.assembleComments(serialized, remaining);
|
|
384
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
385
|
+
const failed = this.unverifiedInlineEdits(result, remaining, uri, serialized);
|
|
386
|
+
if (failed.size === 0) {
|
|
387
|
+
return result;
|
|
388
|
+
}
|
|
389
|
+
remaining = remaining.filter(edit => !failed.has(edit));
|
|
390
|
+
result = this.assembleComments(serialized, remaining);
|
|
391
|
+
}
|
|
392
|
+
return this.assembleComments(serialized, remaining.filter(edit => edit.inline === undefined));
|
|
393
|
+
}
|
|
394
|
+
/**
|
|
395
|
+
* Whether this comment's edit has to be read back before it can be trusted.
|
|
396
|
+
*
|
|
397
|
+
* Both cases are the serializer having put the anchor on a line it shares.
|
|
398
|
+
* A comment that wants the line above has to open one mid-construct; a
|
|
399
|
+
* comment that wants the end of the anchor's line gets the end of a line that
|
|
400
|
+
* may belong to a later sibling, and a second such comment lands inside the
|
|
401
|
+
* first one's text and stops being a comment at all. Neither can be judged
|
|
402
|
+
* from the offsets alone — only from what the text says once re-read.
|
|
403
|
+
*/
|
|
404
|
+
needsReadBack(comment, span, serialized) {
|
|
405
|
+
if (comment.placement === 'leading' || comment.placement === 'atContainerStart') {
|
|
406
|
+
return !this.startOfLineIsBlank(serialized, span.offset);
|
|
407
|
+
}
|
|
408
|
+
return comment.placement === 'trailing' && this.endOfLineAt(serialized, span.end) !== span.end;
|
|
409
|
+
}
|
|
410
|
+
/**
|
|
411
|
+
* Splice `edits` into `serialized`, in the order given.
|
|
412
|
+
*
|
|
413
|
+
* Two inline edits landing on one offset collapse into a single opened line,
|
|
414
|
+
* so a stack of comments above one inline member comes back as a stack rather
|
|
415
|
+
* than with a blank line between each pair.
|
|
416
|
+
*/
|
|
417
|
+
assembleComments(serialized, edits) {
|
|
418
|
+
let result = '';
|
|
419
|
+
let cursor = 0;
|
|
420
|
+
let previousInlineAt;
|
|
421
|
+
for (const edit of edits) {
|
|
422
|
+
const text = edit.inline !== undefined && edit.at === previousInlineAt ? edit.text.replace(/^\n[ \t]*/, '') : edit.text;
|
|
423
|
+
result += serialized.slice(cursor, edit.at) + text;
|
|
424
|
+
cursor = edit.at;
|
|
425
|
+
previousInlineAt = edit.inline !== undefined ? edit.at : undefined;
|
|
426
|
+
}
|
|
427
|
+
return result + serialized.slice(cursor);
|
|
428
|
+
}
|
|
429
|
+
/**
|
|
430
|
+
* Whether the output node now answering to `source`'s key is too likely to be
|
|
431
|
+
* a DIFFERENT node for the comment to follow it.
|
|
432
|
+
*
|
|
433
|
+
* A key surviving the write normally means the node did. It can also mean a
|
|
434
|
+
* sibling took the name over, which reads as the author having written the
|
|
435
|
+
* comment about a declaration they never saw.
|
|
436
|
+
*
|
|
437
|
+
* **What is detectable is a list whose members changed places; what is not is
|
|
438
|
+
* a payload that renames a node and gives its old name to another.** Those
|
|
439
|
+
* two writes produce the same text and the same keys, and the node carrying
|
|
440
|
+
* the old name afterwards is byte-identical under both readings, so no rule
|
|
441
|
+
* over this evidence separates them. A caller that knows which node it
|
|
442
|
+
* renamed is the only thing that could, and the transfer model carries no
|
|
443
|
+
* such record.
|
|
444
|
+
*
|
|
445
|
+
* **The evidence is the ORDER of the keys the two lists share.** Adding or
|
|
446
|
+
* removing siblings shifts the rest but leaves them in the same relative
|
|
447
|
+
* order; only names changing places can put two shared keys out of order. A
|
|
448
|
+
* list whose shared keys invert is therefore read as exchanged identities and
|
|
449
|
+
* those siblings lose their comments — which costs a deliberate reorder its
|
|
450
|
+
* comments, the conservative half of a trade whose other half is that a bare
|
|
451
|
+
* swap cannot move one.
|
|
452
|
+
*
|
|
453
|
+
* **Comparing counts or key SETS instead misses cases each way.** A swap
|
|
454
|
+
* alongside an insertion leaves the counts differing and the key set whole,
|
|
455
|
+
* so neither test sees it; a deletion alongside an insertion leaves the counts
|
|
456
|
+
* equal while every surviving sibling is still itself, so a count test drops
|
|
457
|
+
* comments that were never in doubt.
|
|
458
|
+
*/
|
|
459
|
+
ambiguousRetainedKey(source, output) {
|
|
460
|
+
const sourceParent = source.$container;
|
|
461
|
+
const outputParent = output.$container;
|
|
462
|
+
if (!sourceParent || !outputParent) {
|
|
463
|
+
return false;
|
|
464
|
+
}
|
|
465
|
+
const parentKey = this.cachedAnchorKey(sourceParent);
|
|
466
|
+
if (parentKey !== this.cachedAnchorKey(outputParent) || source.$containerProperty !== output.$containerProperty) {
|
|
467
|
+
return true;
|
|
468
|
+
}
|
|
469
|
+
if (source.$containerIndex === output.$containerIndex) {
|
|
470
|
+
return false;
|
|
471
|
+
}
|
|
472
|
+
const property = source.$containerProperty;
|
|
473
|
+
if (property === undefined) {
|
|
474
|
+
return true;
|
|
475
|
+
}
|
|
476
|
+
// Answered once per list rather than once per comment. One insertion moves
|
|
477
|
+
// every later sibling, so each of their comments asks the same question of
|
|
478
|
+
// the same two lists — and scanning both per comment costs the list's
|
|
479
|
+
// length squared.
|
|
480
|
+
//
|
|
481
|
+
// Memoised against the parent NODE rather than its key, because two
|
|
482
|
+
// same-named parents answer to one key and would otherwise share a verdict
|
|
483
|
+
// about lists that have nothing to do with each other.
|
|
484
|
+
const cached = this.listVerdicts?.get(sourceParent)?.get(property);
|
|
485
|
+
if (cached !== undefined) {
|
|
486
|
+
return cached;
|
|
487
|
+
}
|
|
488
|
+
const verdict = this.reidentifiedList(sourceParent, outputParent, property);
|
|
489
|
+
if (this.listVerdicts !== undefined) {
|
|
490
|
+
const byProperty = this.listVerdicts.get(sourceParent) ?? new Map();
|
|
491
|
+
byProperty.set(property, verdict);
|
|
492
|
+
this.listVerdicts.set(sourceParent, byProperty);
|
|
493
|
+
}
|
|
494
|
+
return verdict;
|
|
495
|
+
}
|
|
496
|
+
/**
|
|
497
|
+
* Whether the two lists differ in a way a plain insertion or deletion cannot
|
|
498
|
+
* explain — the expensive half of {@link ambiguousRetainedKey}, split out so
|
|
499
|
+
* it can be answered once per list.
|
|
500
|
+
*
|
|
501
|
+
* What says the members may have swapped names rather than moved:
|
|
502
|
+
*
|
|
503
|
+
* - **The lists are the same length.** Every rename chain over a fixed set of
|
|
504
|
+
* slots looks exactly like the shift a deletion plus an insertion produces,
|
|
505
|
+
* and for members carrying nothing but a name the properties match under
|
|
506
|
+
* both readings. Neither this nor anything downstream can separate them.
|
|
507
|
+
* - **Keys present on both sides have changed places.** Adding or removing
|
|
508
|
+
* siblings shifts the rest but never reorders them, so an inversion is
|
|
509
|
+
* evidence no insertion or deletion can account for — and it is evidence
|
|
510
|
+
* the lengths alone miss, since a swap alongside an insertion leaves the
|
|
511
|
+
* counts differing and the key set whole.
|
|
512
|
+
*/
|
|
513
|
+
reidentifiedList(sourceParent, outputParent, property) {
|
|
514
|
+
const before = sourceParent[property];
|
|
515
|
+
const after = outputParent[property];
|
|
516
|
+
if (!Array.isArray(before) || !Array.isArray(after)) {
|
|
517
|
+
return true;
|
|
518
|
+
}
|
|
519
|
+
if (before.length === after.length) {
|
|
520
|
+
return true;
|
|
521
|
+
}
|
|
522
|
+
const positions = new Map();
|
|
523
|
+
after.filter(isAstNode).forEach((node, index) => {
|
|
524
|
+
const key = this.cachedAnchorKey(node);
|
|
525
|
+
if (key !== undefined && !positions.has(key)) {
|
|
526
|
+
positions.set(key, index);
|
|
527
|
+
}
|
|
528
|
+
});
|
|
529
|
+
let furthest = -1;
|
|
530
|
+
for (const node of before.filter(isAstNode)) {
|
|
531
|
+
const key = this.cachedAnchorKey(node);
|
|
532
|
+
const position = key === undefined ? undefined : positions.get(key);
|
|
533
|
+
if (position === undefined) {
|
|
534
|
+
continue;
|
|
535
|
+
}
|
|
536
|
+
if (position < furthest) {
|
|
537
|
+
return true;
|
|
538
|
+
}
|
|
539
|
+
furthest = position;
|
|
540
|
+
}
|
|
541
|
+
return false;
|
|
542
|
+
}
|
|
543
|
+
/**
|
|
544
|
+
* The inline edits `candidate` did not round-trip: each one whose comment came
|
|
545
|
+
* back on a different anchor, and — when `candidate` does not parse at all —
|
|
546
|
+
* each one that cannot be put back beside the others without breaking it
|
|
547
|
+
* again. `serialized` is the unspliced text those trials are rebuilt from,
|
|
548
|
+
* since finding the single edit at fault means assembling the rest without it.
|
|
549
|
+
*
|
|
550
|
+
* Costs a parse and an extract, so it answers empty without either when
|
|
551
|
+
* nothing was spliced inline — which is every write into a document the
|
|
552
|
+
* serializer lays out one declaration per line.
|
|
553
|
+
*/
|
|
554
|
+
unverifiedInlineEdits(candidate, edits, uri, serialized) {
|
|
555
|
+
const inlineEdits = edits.filter(edit => edit.inline !== undefined);
|
|
556
|
+
if (inlineEdits.length === 0) {
|
|
557
|
+
return new Set();
|
|
558
|
+
}
|
|
559
|
+
const reparsed = this.parse(candidate, uri);
|
|
560
|
+
if (reparsed === undefined) {
|
|
561
|
+
if (inlineEdits.length > this.maxIsolatedEdits) {
|
|
562
|
+
this.tracer
|
|
563
|
+
.withUri(uri.toString())
|
|
564
|
+
.debug(`Spliced text did not parse; dropping ${inlineEdits.length} shared-line comments without isolating`);
|
|
565
|
+
return new Set(inlineEdits);
|
|
566
|
+
}
|
|
567
|
+
const retained = edits.filter(edit => edit.inline === undefined);
|
|
568
|
+
const failed = new Set();
|
|
569
|
+
for (const edit of inlineEdits) {
|
|
570
|
+
const trial = [...retained, edit].sort((left, right) => left.at - right.at || left.order - right.order);
|
|
571
|
+
const text = this.assembleComments(serialized, trial);
|
|
572
|
+
if (this.parse(text, uri) !== undefined && this.unverifiedInlineEdits(text, trial, uri, serialized).size === 0) {
|
|
573
|
+
retained.push(edit);
|
|
574
|
+
}
|
|
575
|
+
else {
|
|
576
|
+
failed.add(edit);
|
|
577
|
+
}
|
|
578
|
+
}
|
|
579
|
+
return failed;
|
|
580
|
+
}
|
|
581
|
+
// Counted rather than scanned per edit: several comments can share an
|
|
582
|
+
// anchor and a text, and each edit must consume one of them.
|
|
583
|
+
const captured = new Map();
|
|
584
|
+
for (const comment of this.extract(reparsed).comments) {
|
|
585
|
+
const seen = `${this.cachedAnchorKey(comment.anchor) ?? ''}\u0000${comment.text}`;
|
|
586
|
+
captured.set(seen, (captured.get(seen) ?? 0) + 1);
|
|
587
|
+
}
|
|
588
|
+
const failed = new Set();
|
|
589
|
+
for (const edit of inlineEdits) {
|
|
590
|
+
const expected = edit.inline;
|
|
591
|
+
const wanted = `${expected.key}\u0000${expected.text}`;
|
|
592
|
+
const remaining = captured.get(wanted) ?? 0;
|
|
593
|
+
if (remaining === 0) {
|
|
594
|
+
failed.add(edit);
|
|
595
|
+
}
|
|
596
|
+
else {
|
|
597
|
+
captured.set(wanted, remaining - 1);
|
|
598
|
+
}
|
|
599
|
+
}
|
|
600
|
+
return failed;
|
|
601
|
+
}
|
|
602
|
+
renamedKey(anchor, oldKey, renamed) {
|
|
603
|
+
let current = anchor;
|
|
604
|
+
while (current !== undefined) {
|
|
605
|
+
const newPrefix = renamed.get(current);
|
|
606
|
+
const oldPrefix = this.cachedAnchorKey(current);
|
|
607
|
+
if (newPrefix !== undefined && oldPrefix !== undefined && (oldKey === oldPrefix || oldKey.startsWith(`${oldPrefix}/`))) {
|
|
608
|
+
return newPrefix + oldKey.slice(oldPrefix.length);
|
|
609
|
+
}
|
|
610
|
+
current = current.$container;
|
|
611
|
+
}
|
|
612
|
+
return undefined;
|
|
613
|
+
}
|
|
614
|
+
matchRenamedAnchors(sourceRoot, serialized, uri, located, sourceOwners, sourceCollisions) {
|
|
615
|
+
const matches = new Map();
|
|
616
|
+
const output = this.parse(serialized, uri);
|
|
617
|
+
if (output === undefined) {
|
|
618
|
+
return matches;
|
|
619
|
+
}
|
|
620
|
+
const outputNodes = [output.parseResult.value, ...AstUtils.streamAllContents(output.parseResult.value)];
|
|
621
|
+
const unmatchedBySlot = new Map();
|
|
622
|
+
for (const node of outputNodes) {
|
|
623
|
+
const key = this.cachedAnchorKey(node);
|
|
624
|
+
const parent = node.$container && this.cachedAnchorKey(node.$container);
|
|
625
|
+
if (key === undefined || !located.has(key) || sourceOwners.has(key) || parent === undefined) {
|
|
626
|
+
continue;
|
|
627
|
+
}
|
|
628
|
+
const slot = JSON.stringify([parent, node.$containerProperty, node.$containerIndex, node.$type]);
|
|
629
|
+
const candidates = unmatchedBySlot.get(slot) ?? [];
|
|
630
|
+
candidates.push(node);
|
|
631
|
+
unmatchedBySlot.set(slot, candidates);
|
|
632
|
+
}
|
|
633
|
+
const claimed = new Map();
|
|
634
|
+
const ambiguous = new Set();
|
|
635
|
+
for (const source of [sourceRoot, ...AstUtils.streamAllContents(sourceRoot)]) {
|
|
636
|
+
const key = this.cachedAnchorKey(source);
|
|
637
|
+
if (key === undefined || sourceCollisions.has(key) || located.has(key) || !source.$container) {
|
|
638
|
+
continue;
|
|
639
|
+
}
|
|
640
|
+
const parentKey = this.cachedAnchorKey(source.$container);
|
|
641
|
+
if (parentKey === undefined) {
|
|
642
|
+
continue;
|
|
643
|
+
}
|
|
644
|
+
const slot = JSON.stringify([parentKey, source.$containerProperty, source.$containerIndex, source.$type]);
|
|
645
|
+
const candidates = (unmatchedBySlot.get(slot) ?? []).filter(candidate => this.sameAsideFromName(source, candidate));
|
|
646
|
+
if (candidates.length === 1) {
|
|
647
|
+
const candidate = candidates[0];
|
|
648
|
+
if (ambiguous.has(candidate)) {
|
|
649
|
+
continue;
|
|
650
|
+
}
|
|
651
|
+
if (claimed.has(candidate)) {
|
|
652
|
+
matches.delete(claimed.get(candidate));
|
|
653
|
+
ambiguous.add(candidate);
|
|
654
|
+
}
|
|
655
|
+
else {
|
|
656
|
+
matches.set(source, this.cachedAnchorKey(candidate));
|
|
657
|
+
}
|
|
658
|
+
claimed.set(candidate, source);
|
|
659
|
+
}
|
|
660
|
+
}
|
|
661
|
+
return matches;
|
|
662
|
+
}
|
|
663
|
+
sameAsideFromName(source, target) {
|
|
664
|
+
const nameProvider = this.services.references.NameProvider;
|
|
665
|
+
const before = nameProvider.getOwnName(source);
|
|
666
|
+
const after = nameProvider.getOwnName(target);
|
|
667
|
+
if (before === undefined || after === undefined || before === after) {
|
|
668
|
+
return false;
|
|
669
|
+
}
|
|
670
|
+
let renamed = false;
|
|
671
|
+
for (const key of this.comparableProperties(source.$type)) {
|
|
672
|
+
const oldValue = source[key];
|
|
673
|
+
const newValue = target[key];
|
|
674
|
+
if (oldValue === before && newValue === after) {
|
|
675
|
+
renamed = true;
|
|
676
|
+
}
|
|
677
|
+
else if (!this.sameValue(oldValue, newValue)) {
|
|
678
|
+
return false;
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
return renamed;
|
|
682
|
+
}
|
|
683
|
+
/**
|
|
684
|
+
* The properties `type` declares in the grammar — everything a rename match
|
|
685
|
+
* may compare, and nothing else.
|
|
686
|
+
*
|
|
687
|
+
* **Comparing own keys instead compares derived state, and matches nothing.**
|
|
688
|
+
* A built document carries whatever `ast.extensions` computed onto its nodes;
|
|
689
|
+
* the re-parsed serializer output is never built and carries none of it. Every
|
|
690
|
+
* candidate then differs on a property the grammar never mentioned, so no
|
|
691
|
+
* rename is ever matched — on the real write path only, because a document a
|
|
692
|
+
* test parses from a string has no computed state to disagree about.
|
|
693
|
+
*/
|
|
694
|
+
comparableProperties(type) {
|
|
695
|
+
return Object.keys(this.services.shared.AstReflection.getTypeMetaData(type).properties);
|
|
696
|
+
}
|
|
697
|
+
sameValue(left, right) {
|
|
698
|
+
if (left === right) {
|
|
699
|
+
return true;
|
|
700
|
+
}
|
|
701
|
+
if (Array.isArray(left) && Array.isArray(right)) {
|
|
702
|
+
return left.length === right.length && left.every((value, index) => this.sameValue(value, right[index]));
|
|
703
|
+
}
|
|
704
|
+
if (isAstNode(left) && isAstNode(right)) {
|
|
705
|
+
if (left.$type !== right.$type) {
|
|
706
|
+
return false;
|
|
707
|
+
}
|
|
708
|
+
return this.comparableProperties(left.$type).every(key => this.sameValue(left[key], right[key]));
|
|
709
|
+
}
|
|
710
|
+
if (left !== null && right !== null && typeof left === 'object' && typeof right === 'object') {
|
|
711
|
+
if ('$refText' in left && '$refText' in right) {
|
|
712
|
+
return left.$refText === right.$refText;
|
|
713
|
+
}
|
|
714
|
+
}
|
|
715
|
+
return false;
|
|
716
|
+
}
|
|
717
|
+
/** The insertion this comment's placement implies, against its anchor's span. */
|
|
718
|
+
editFor(comment, span, serialized) {
|
|
719
|
+
const blanks = '\n'.repeat(comment.blankLinesAfter);
|
|
720
|
+
const blanksBefore = '\n'.repeat(comment.blankLinesBefore);
|
|
721
|
+
switch (comment.placement) {
|
|
722
|
+
case 'trailing':
|
|
723
|
+
return { at: this.endOfLineAt(serialized, span.end), text: ` ${comment.text}` };
|
|
724
|
+
// Appended to the container's FIRST line, which is its header. The
|
|
725
|
+
// comment may have sat further along that line in the source — mid
|
|
726
|
+
// header, or between braces the serializer has since collapsed to `{}`
|
|
727
|
+
// — and the exact column is not recoverable from a re-emitted
|
|
728
|
+
// document. The line is, and that is what keeps it on its own
|
|
729
|
+
// declaration instead of on a neighbour.
|
|
730
|
+
case 'trailingOnContainer':
|
|
731
|
+
return { at: this.endOfLineAt(serialized, span.offset), text: ` ${comment.text}` };
|
|
732
|
+
// Both of these open a NEW line after the anchor, which is only safe
|
|
733
|
+
// while nothing else shares the anchor's line. When the serializer put
|
|
734
|
+
// the anchor and its container's closing syntax together — an inline
|
|
735
|
+
// enum body, say — splitting there pushes that syntax onto the comment's
|
|
736
|
+
// line, and a line comment then ends the construct. Appending to the
|
|
737
|
+
// line instead keeps the comment on the same declaration and the
|
|
738
|
+
// document readable.
|
|
739
|
+
case 'afterNode': {
|
|
740
|
+
if (!this.restOfLineIsBlank(serialized, span.end)) {
|
|
741
|
+
return { at: this.endOfLineAt(serialized, span.end), text: ` ${comment.text}` };
|
|
742
|
+
}
|
|
743
|
+
const indent = this.indentAt(serialized, span.offset);
|
|
744
|
+
return { at: span.end, text: `\n${blanksBefore}${indent}${this.reindent(comment, indent)}` };
|
|
745
|
+
}
|
|
746
|
+
case 'atContainerEnd':
|
|
747
|
+
return this.restOfLineIsBlank(serialized, span.end)
|
|
748
|
+
? { at: span.end, text: `\n${blanksBefore}${comment.text}` }
|
|
749
|
+
: { at: this.endOfLineAt(serialized, span.end), text: ` ${comment.text}` };
|
|
750
|
+
case 'atContainerStart':
|
|
751
|
+
case 'leading':
|
|
752
|
+
default: {
|
|
753
|
+
// `leading` means "on the line above", and the anchor may sit
|
|
754
|
+
// mid-line — a member of a body the serializer emitted inline.
|
|
755
|
+
// **Opening a line there is sound only because the caller reads the
|
|
756
|
+
// result back and requires this comment on this same anchor**,
|
|
757
|
+
// dropping the edit when it is not. Unguarded, the split strands the
|
|
758
|
+
// rest of the construct at column zero and lands somewhere different
|
|
759
|
+
// again on the next write.
|
|
760
|
+
if (!this.startOfLineIsBlank(serialized, span.offset)) {
|
|
761
|
+
const indent = comment.sourceIndent ?? this.indentAt(serialized, span.offset);
|
|
762
|
+
return { at: span.offset, text: `\n${indent}${this.reindent(comment, indent)}\n${blanks}${indent}` };
|
|
763
|
+
}
|
|
764
|
+
const indent = this.indentAt(serialized, span.offset);
|
|
765
|
+
return { at: span.offset, text: `${this.reindent(comment, indent)}\n${blanks}${indent}` };
|
|
766
|
+
}
|
|
767
|
+
}
|
|
768
|
+
}
|
|
769
|
+
/**
|
|
770
|
+
* A multi-line comment's continuation lines, shifted by the same delta its
|
|
771
|
+
* first line moves — so a block emitted at a new indentation stays square
|
|
772
|
+
* instead of trailing its original column.
|
|
773
|
+
*
|
|
774
|
+
* **Only the leading whitespace run is touched, and an outdent removes at
|
|
775
|
+
* most what is there** — shifting further would eat the comment's own text.
|
|
776
|
+
*
|
|
777
|
+
* Single-line comments and ones sharing a line with code are returned
|
|
778
|
+
* unchanged.
|
|
779
|
+
*/
|
|
780
|
+
reindent(comment, targetIndent) {
|
|
781
|
+
if (comment.sourceIndent === undefined || !comment.text.includes('\n')) {
|
|
782
|
+
return comment.text;
|
|
783
|
+
}
|
|
784
|
+
// Indentation is compared as characters, which only means anything while
|
|
785
|
+
// both sides use the SAME whitespace character. A tab-indented source
|
|
786
|
+
// re-emitted with spaces has no meaningful delta, and shifting by one
|
|
787
|
+
// anyway prepends spaces in front of tabs.
|
|
788
|
+
const sourceUnit = /^\t*$/.test(comment.sourceIndent) ? '\t' : ' ';
|
|
789
|
+
const targetUnit = /^\t*$/.test(targetIndent) ? '\t' : ' ';
|
|
790
|
+
const delta = targetIndent.length - comment.sourceIndent.length;
|
|
791
|
+
if (delta === 0 || sourceUnit !== targetUnit) {
|
|
792
|
+
return comment.text;
|
|
793
|
+
}
|
|
794
|
+
const [first, ...rest] = comment.text.split('\n');
|
|
795
|
+
const shifted = rest.map(line => {
|
|
796
|
+
if (delta > 0) {
|
|
797
|
+
return targetUnit.repeat(delta) + line;
|
|
798
|
+
}
|
|
799
|
+
const removable = /^[ \t]*/.exec(line)?.[0].length ?? 0;
|
|
800
|
+
return line.slice(Math.min(-delta, removable));
|
|
801
|
+
});
|
|
802
|
+
return [first, ...shifted].join('\n');
|
|
803
|
+
}
|
|
804
|
+
/** Whether everything from `offset` to the end of its line is whitespace. */
|
|
805
|
+
restOfLineIsBlank(text, offset) {
|
|
806
|
+
return text.slice(offset, this.endOfLineAt(text, offset)).trim().length === 0;
|
|
807
|
+
}
|
|
808
|
+
/** Whether `offset` is preceded on its own line by whitespace alone. */
|
|
809
|
+
startOfLineIsBlank(text, offset) {
|
|
810
|
+
return text.slice(text.lastIndexOf('\n', offset - 1) + 1, offset).trim().length === 0;
|
|
811
|
+
}
|
|
812
|
+
/**
|
|
813
|
+
* End of the line `offset` sits on, as an insertion point for a trailing
|
|
814
|
+
* comment. Stops before a CR so an insertion into CRLF text lands inside the
|
|
815
|
+
* line rather than between its two terminator bytes.
|
|
816
|
+
*/
|
|
817
|
+
endOfLineAt(text, offset) {
|
|
818
|
+
const newline = text.indexOf('\n', offset);
|
|
819
|
+
const lineEnd = newline < 0 ? text.length : newline;
|
|
820
|
+
return lineEnd > 0 && text[lineEnd - 1] === '\r' ? lineEnd - 1 : lineEnd;
|
|
821
|
+
}
|
|
822
|
+
/** Leading whitespace of the line `offset` sits on, so an insertion lines up with it. */
|
|
823
|
+
indentAt(text, offset) {
|
|
824
|
+
const lineStart = text.lastIndexOf('\n', offset - 1) + 1;
|
|
825
|
+
return /^\s*/.exec(text.slice(lineStart, offset))?.[0] ?? '';
|
|
826
|
+
}
|
|
827
|
+
/**
|
|
828
|
+
* Where every node landed in the serializer's output, keyed by
|
|
829
|
+
* {@link anchorKey}.
|
|
830
|
+
*
|
|
831
|
+
* Re-parses `serialized` through the document factory, which does NOT
|
|
832
|
+
* register the result, so this leaves `LangiumDocuments` alone. Output the
|
|
833
|
+
* grammar cannot read, by reported error or by a throw from a URI that
|
|
834
|
+
* routes to no services, yields `undefined`: the caller then writes the
|
|
835
|
+
* serializer's text unchanged rather than splicing into text already wrong.
|
|
836
|
+
*
|
|
837
|
+
* **A key claimed by two DIFFERENT nodes is removed, not merged.** A repeated
|
|
838
|
+
* identity is exactly what the integrity tier exists to repair, so collisions
|
|
839
|
+
* reach this method routinely; merging their spans puts every one of their
|
|
840
|
+
* comments on whichever came first. Removing the key drops those instead,
|
|
841
|
+
* which is the only outcome here that keeps a comment off a declaration its
|
|
842
|
+
* author did not write it on.
|
|
843
|
+
*/
|
|
844
|
+
locate(serialized, uri) {
|
|
845
|
+
const document = this.parse(serialized, uri);
|
|
846
|
+
if (document === undefined) {
|
|
847
|
+
this.tracer.withUri(uri.toString()).warn('Serialized output did not re-parse; writing it without comments');
|
|
848
|
+
return undefined;
|
|
849
|
+
}
|
|
850
|
+
const root = document.parseResult.value.$cstNode;
|
|
851
|
+
if (!root) {
|
|
852
|
+
return undefined;
|
|
853
|
+
}
|
|
854
|
+
const found = new Map();
|
|
855
|
+
// One AST node owns many CST nodes — its composite plus every token under
|
|
856
|
+
// it — so a repeated key is only a collision when a DIFFERENT node claims
|
|
857
|
+
// it. Tracking the owner is what separates the two.
|
|
858
|
+
const owners = new Map();
|
|
859
|
+
const collided = new Set();
|
|
860
|
+
// One AST node owns every CST node beneath it, so this asks for the same
|
|
861
|
+
// key once per token; `cachedAnchorKey` is what keeps the pass from
|
|
862
|
+
// costing nodes x depth.
|
|
863
|
+
const visit = (node) => {
|
|
864
|
+
const key = node.astNode !== undefined && !node.hidden ? this.cachedAnchorKey(node.astNode) : undefined;
|
|
865
|
+
if (key !== undefined) {
|
|
866
|
+
const owner = owners.get(key);
|
|
867
|
+
if (owner === undefined) {
|
|
868
|
+
owners.set(key, node.astNode);
|
|
869
|
+
found.set(key, { offset: node.offset, end: node.end, owner: node.astNode });
|
|
870
|
+
}
|
|
871
|
+
else if (owner === node.astNode) {
|
|
872
|
+
// Widest span per node: its first CST node gives the start, a
|
|
873
|
+
// later token contributing to the same node extends the end.
|
|
874
|
+
const existing = found.get(key);
|
|
875
|
+
found.set(key, {
|
|
876
|
+
offset: Math.min(existing.offset, node.offset),
|
|
877
|
+
end: Math.max(existing.end, node.end),
|
|
878
|
+
owner: existing.owner
|
|
879
|
+
});
|
|
880
|
+
}
|
|
881
|
+
else {
|
|
882
|
+
collided.add(key);
|
|
883
|
+
}
|
|
884
|
+
}
|
|
885
|
+
if (isComposite(node)) {
|
|
886
|
+
node.content.forEach(visit);
|
|
887
|
+
}
|
|
888
|
+
};
|
|
889
|
+
visit(root);
|
|
890
|
+
for (const key of collided) {
|
|
891
|
+
found.delete(key);
|
|
892
|
+
}
|
|
893
|
+
return found;
|
|
894
|
+
}
|
|
895
|
+
}
|
|
896
|
+
/** Registers {@link CommentPreserver}; bound at `trivia.preservers.comments`. */
|
|
897
|
+
export class CommentPreserverContribution {
|
|
898
|
+
services;
|
|
899
|
+
constructor(services) {
|
|
900
|
+
this.services = services;
|
|
901
|
+
}
|
|
902
|
+
registerTriviaPreservers(registry) {
|
|
903
|
+
registry.register(new CommentPreserver(this.services));
|
|
904
|
+
}
|
|
905
|
+
}
|
|
906
|
+
//# sourceMappingURL=comment-preserver.js.map
|