@zombie-mermaid/mermaid-parser 4.0.0 → 4.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1327 @@
1
+ import type {
2
+ MermaidGraph,
3
+ MermaidNode,
4
+ MermaidSubgraph,
5
+ NodeShape,
6
+ EdgeStyle,
7
+ Statement,
8
+ } from '@zombie-mermaid/core'
9
+ import {
10
+ toDirection,
11
+ normalizeBrTags,
12
+ splitStatementsByLine,
13
+ extractInitConfig,
14
+ applyClickStatement as applyClickStatementShared,
15
+ parseStyleProps,
16
+ tryApplyClassDef,
17
+ tryApplyClassAssignment,
18
+ tryApplyStyleStatement,
19
+ } from '@zombie-mermaid/core'
20
+
21
+ import {
22
+ matchExpandedBlock,
23
+ parseExpandedMeta,
24
+ resolveShapeName,
25
+ } from './expanded-shapes.ts'
26
+ import type { ExpandedNodeMeta } from './expanded-shapes.ts'
27
+ /** Remove a single layer of matching wrapping quotes (`"…"` or `'…'`). */
28
+ function stripWrappingQuotes(s: string): string {
29
+ const t = s.trim()
30
+ if (
31
+ t.length >= 2 &&
32
+ ((t[0] === '"' && t[t.length - 1] === '"') ||
33
+ (t[0] === "'" && t[t.length - 1] === "'"))
34
+ ) {
35
+ return t.slice(1, -1)
36
+ }
37
+ return t
38
+ }
39
+
40
+ // ============================================================================
41
+ // Mermaid parser — flowcharts and state diagrams
42
+ //
43
+ // Supports:
44
+ // Flowcharts: graph TD / flowchart LR
45
+ // State diagrams: stateDiagram-v2
46
+ //
47
+ // Line-by-line regex approach — the grammar is regular enough
48
+ // that we don't need a grammar generator or full parser combinator.
49
+ // ============================================================================
50
+
51
+ // `toDirection` used to be defined here, with this file as its only caller.
52
+ // #624 gave it a second caller — `packages/mermaid-parser/src/er/parser.ts`,
53
+ // which cannot import this umbrella file without creating a cycle
54
+ // (mermaid-parser -> umbrella -> mermaid-parser) — so it moved to
55
+ // `packages/core/src/direction.ts` alongside `isDirection`, for the same
56
+ // reason that one moved under #625. Re-exported here so existing importers
57
+ // of `toDirection` from `./parser.ts` (this file's own tests included) keep
58
+ // working unchanged.
59
+ export { toDirection } from '@zombie-mermaid/core'
60
+
61
+ /**
62
+ * All diagram-type headers this library recognizes, for the "supported
63
+ * headers" list in the invalid-header error below. Kept in one place so a
64
+ * newly-added diagram type doesn't get forgotten in the error message the
65
+ * way sequence/class/ER/xychart were before issue #541.
66
+ */
67
+ const SUPPORTED_HEADERS =
68
+ '"graph <dir>"/"flowchart <dir>" (dir: TD, TB, LR, BT, RL), "stateDiagram-v2", "sequenceDiagram", "classDiagram", "erDiagram", "xychart-beta", "C4Context"/"C4Container"/"C4Component"/"C4Dynamic"/"C4Deployment", "architecture-beta", "pie"'
69
+
70
+ /**
71
+ * Best-guess canonical header for a header line that looks like an attempt
72
+ * at one of the *other* diagram types this library supports, but didn't
73
+ * match `detectDiagramType`'s exact-match regex (a typo, extra trailing
74
+ * text, wrong case pattern, etc.) and so fell through to this flowchart
75
+ * parser as the default. Returns undefined when `header` doesn't look like
76
+ * any of those — in which case the generic "supported headers" list in the
77
+ * thrown error is the best we can do.
78
+ *
79
+ * This only recognizes the *other* headers
80
+ * (sequenceDiagram/classDiagram/erDiagram/xychart-beta/stateDiagram-v2/pie);
81
+ * "graph"/"flowchart" typos are handled by the direction-specific message
82
+ * below instead, since those already matched the diagram-type keyword and
83
+ * just have a bad or missing direction token.
84
+ */
85
+ function suggestedHeaderFor(header: string): string | undefined {
86
+ const lower = header.trim().toLowerCase()
87
+ if (/^sequence/.test(lower)) return 'sequenceDiagram'
88
+ if (/^class/.test(lower)) return 'classDiagram'
89
+ if (/^er/.test(lower)) return 'erDiagram'
90
+ if (/^xychart/.test(lower)) return 'xychart-beta'
91
+ if (/^state/.test(lower)) return 'stateDiagram-v2'
92
+ // Mermaid's `pie` keyword is case-sensitive, so `PieChart` lands here.
93
+ if (/^pie/.test(lower)) return 'pie'
94
+ return undefined
95
+ }
96
+
97
+ /**
98
+ * Matches the start of a statement that continues the *previous* one's edge
99
+ * chain rather than beginning a new one: Mermaid lets a vertex-chain
100
+ * statement break across lines, with the link operator leading the next
101
+ * line — e.g.
102
+ *
103
+ * start([Start])
104
+ * ==> green([Change some code])
105
+ * ==> finish([Finish])
106
+ *
107
+ * is one chained statement (`start ==> green ==> finish`), identical to
108
+ * writing it on one line. `splitStatementsByLine` only knows about newlines
109
+ * and `;` as separators, so without this it hands `parseFlowchart` three
110
+ * unrelated-looking statements — the continuation lines start with a bare
111
+ * arrow and no node group, so `parseEdgeLine` can't find a source node and
112
+ * drops them entirely (see issue mermaid-js/mermaid#6049's repro, reported
113
+ * against this parser).
114
+ *
115
+ * Covers every arrow opener `parseEdgeLine` itself recognizes (ARROW_REGEX /
116
+ * TEXT_ARROW_REGEX below): a solid/thick run (`--`, `===`), a dotted run
117
+ * (`-.`, `-.-`), a `~~~` run, each optionally preceded by a `<` marker and/or
118
+ * an edge id (`e1@-->`).
119
+ *
120
+ * Deliberately excludes the `o`/`x` start markers ARROW_REGEX also accepts
121
+ * (`o--o`, `x--x`): unlike `<`, both are also valid bare node ids, so a
122
+ * first-body-line statement like `x-->Y` would otherwise be misread as an
123
+ * `x`-marked continuation of the header and merged into it, corrupting the
124
+ * header line (`flowchart TD x-->Y`) instead of parsing `x` as its own node.
125
+ * A genuine marked-start continuation line is rare enough that losing it is
126
+ * the safer tradeoff.
127
+ */
128
+ const CONTINUATION_START_REGEX = /^(?:[\w-]+@)?<?(?:-{2,}|={2,}|-\.+-?|~{3,})/
129
+
130
+ /**
131
+ * Rejoin a continuation statement onto the one it continues.
132
+ *
133
+ * `splitStatementsByLine` groups statements by their originating physical
134
+ * line, which is what makes this safe: only the *first* statement in a
135
+ * group is eligible to merge backward into the previous group's last
136
+ * statement (a genuine multi-line continuation, see
137
+ * `CONTINUATION_START_REGEX`). A later statement in the same group got
138
+ * there via an explicit `;` on that line (`A --> B; --> C`) — that boundary
139
+ * must stay a boundary, or `--> C` would wrongly become part of the chain
140
+ * instead of the source-less fragment it actually is.
141
+ */
142
+ function mergeContinuationLines(groups: Statement[][]): Statement[] {
143
+ const merged: Statement[] = []
144
+ for (const group of groups) {
145
+ group.forEach((statement, index) => {
146
+ if (
147
+ index === 0 &&
148
+ merged.length > 0 &&
149
+ CONTINUATION_START_REGEX.test(statement.text)
150
+ ) {
151
+ // Keep the *first* line's number: a merged statement is reported at
152
+ // the line its logical statement began, not the continuation line.
153
+ const prev = merged[merged.length - 1]!
154
+ merged[merged.length - 1] = {
155
+ text: `${prev.text} ${statement.text}`,
156
+ line: prev.line,
157
+ }
158
+ } else {
159
+ merged.push(statement)
160
+ }
161
+ })
162
+ }
163
+ return merged
164
+ }
165
+
166
+ /**
167
+ * Parse Mermaid text into a logical graph structure.
168
+ * Auto-detects diagram type (flowchart or state diagram).
169
+ * Throws on invalid/unsupported input.
170
+ */
171
+ export function parseMermaid(text: string): MermaidGraph {
172
+ /*
173
+ * Init directives must be read from the raw lines: `%%{init: ...}%%` begins
174
+ * with `%%`, so splitStatementsByLine — which treats `%%` as a comment and
175
+ * truncates the line there — would otherwise discard it along with real
176
+ * comments before it could be seen.
177
+ */
178
+ const initConfig = extractInitConfig(text.split('\n').map((l) => l.trim()))
179
+
180
+ const lines = mergeContinuationLines(splitStatementsByLine(text))
181
+
182
+ if (lines.length === 0) {
183
+ throw new Error('Empty mermaid diagram')
184
+ }
185
+
186
+ // Detect diagram type from header
187
+ const headerStmt = lines[0]!
188
+
189
+ // State diagram: "stateDiagram-v2" or "stateDiagram"
190
+ const graph = /^stateDiagram(-v2)?\s*$/i.test(headerStmt.text)
191
+ ? parseStateDiagram(lines)
192
+ : parseFlowchart(lines)
193
+
194
+ graph.initConfig = initConfig
195
+ return graph
196
+ }
197
+
198
+ // ============================================================================
199
+ // Flowchart parser
200
+ // ============================================================================
201
+
202
+ function parseFlowchart(lines: Statement[]): MermaidGraph {
203
+ // parseFlowchart is only ever invoked by parseMermaid, which has already
204
+ // verified `lines` is non-empty — but that invariant isn't visible to the
205
+ // type checker across the function boundary, so validate it here instead
206
+ // of asserting past it.
207
+ const headerStmt = lines[0]
208
+ if (headerStmt === undefined) {
209
+ /* v8 ignore next */
210
+ throw new Error('parseFlowchart called with no lines')
211
+ }
212
+ const header = headerStmt.text
213
+
214
+ const headerMatch = header.match(
215
+ /^(?:graph|flowchart)\s+(TD|TB|LR|BT|RL)\s*$/i,
216
+ )
217
+ if (!headerMatch) {
218
+ // A header that starts with "graph"/"flowchart" but has a bad or
219
+ // missing direction gets a targeted message about the direction token
220
+ // specifically — it already picked the right diagram type.
221
+ const partialFlowchartMatch = header.match(
222
+ /^(?:graph|flowchart)\b\s*(.*)$/i,
223
+ )
224
+ if (partialFlowchartMatch) {
225
+ const rest = partialFlowchartMatch[1]!.trim()
226
+ throw new Error(
227
+ rest.length > 0
228
+ ? `Line ${headerStmt.line}: Invalid direction "${rest}" in header "${header}". Expected one of: TD, TB, LR, BT, RL.`
229
+ : `Line ${headerStmt.line}: Missing direction in header "${header}". Expected e.g. "graph TD" or "flowchart LR" — one of: TD, TB, LR, BT, RL.`,
230
+ )
231
+ }
232
+
233
+ // Otherwise this header didn't match any recognized diagram-type
234
+ // keyword at all (typo, extra text, or genuinely unsupported syntax) —
235
+ // point at the closest known diagram type when the text resembles one.
236
+ const suggestion = suggestedHeaderFor(header)
237
+ const hint =
238
+ suggestion && suggestion.toLowerCase() !== header.trim().toLowerCase()
239
+ ? ` Did you mean "${suggestion}"?`
240
+ : ''
241
+ throw new Error(
242
+ `Line ${headerStmt.line}: Invalid mermaid header: "${header}".${hint} Supported headers: ${SUPPORTED_HEADERS}.`,
243
+ )
244
+ }
245
+
246
+ const direction = toDirection(headerMatch[1])
247
+
248
+ const graph: MermaidGraph = {
249
+ direction,
250
+ nodes: new Map(),
251
+ edges: [],
252
+ subgraphs: [],
253
+ classDefs: new Map(),
254
+ classAssignments: new Map(),
255
+ nodeStyles: new Map(),
256
+ linkStyles: new Map(),
257
+ interactions: new Map(),
258
+ }
259
+
260
+ // Subgraph stack for nested subgraphs.
261
+ const subgraphStack: MermaidSubgraph[] = []
262
+
263
+ for (let i = 1; i < lines.length; i++) {
264
+ const stmt = lines[i]!
265
+ const line = stmt.text
266
+
267
+ // --- classDef / class assignment / style — shared with the class-diagram
268
+ // parser, see packages/core/src/style-directives.ts ---
269
+ if (tryApplyClassDef(line, graph)) continue
270
+ if (tryApplyClassAssignment(line, graph)) continue
271
+ if (tryApplyStyleStatement(line, graph)) continue
272
+
273
+ // --- click interaction: `click A "url" "tooltip" _blank` / `click A call fn()` ---
274
+ if (/^click\s+/i.test(line)) {
275
+ applyClickStatement(line, graph)
276
+ continue
277
+ }
278
+
279
+ // --- edge metadata: `e1@{ animate: true }` — shared with parseStateDiagram ---
280
+ if (tryApplyEdgeMetaLine(line, graph)) {
281
+ continue
282
+ }
283
+
284
+ // --- linkStyle: `linkStyle 0 stroke:#f00` or `linkStyle default stroke:#f00` ---
285
+ const linkStyleMatch = line.match(/^linkStyle\s+(default|[\d,\s]+)\s+(.+)$/)
286
+ if (linkStyleMatch) {
287
+ const target = linkStyleMatch[1]!.trim()
288
+ const props = parseStyleProps(linkStyleMatch[2]!)
289
+ if (target === 'default') {
290
+ graph.linkStyles.set('default', {
291
+ ...graph.linkStyles.get('default'),
292
+ ...props,
293
+ })
294
+ } else {
295
+ const indices = target.split(',').map((s) => parseInt(s.trim(), 10))
296
+ for (const idx of indices) {
297
+ if (!isNaN(idx)) {
298
+ graph.linkStyles.set(idx, {
299
+ ...graph.linkStyles.get(idx),
300
+ ...props,
301
+ })
302
+ }
303
+ }
304
+ }
305
+ continue
306
+ }
307
+
308
+ // --- direction override inside subgraph: `direction LR` ---
309
+ const dirMatch = line.match(/^direction\s+(TD|TB|LR|BT|RL)\s*$/i)
310
+ if (dirMatch && subgraphStack.length > 0) {
311
+ subgraphStack[subgraphStack.length - 1]!.direction = toDirection(
312
+ dirMatch[1],
313
+ )
314
+ continue
315
+ }
316
+
317
+ // --- subgraph start: `subgraph Label` or `subgraph id [Label]` ---
318
+ const subgraphMatch = line.match(/^subgraph\s+(.+)$/)
319
+ if (subgraphMatch) {
320
+ const rest = subgraphMatch[1]!.trim()
321
+ // Check for `subgraph id [Label]` / `subgraph id ["Label"]` form. The id
322
+ // may contain non-ASCII chars (e.g. CJK), so match any run of non-space,
323
+ // non-`[` chars rather than ASCII [\w-]; strip optional quotes wrapping
324
+ // the title.
325
+ const bracketMatch = rest.match(/^([^\s[]+)\s*\[(.+)\]$/)
326
+ let id: string
327
+ let label: string
328
+ if (bracketMatch) {
329
+ id = bracketMatch[1]!
330
+ label = normalizeBrTags(stripWrappingQuotes(bracketMatch[2]!))
331
+ } else {
332
+ // `subgraph Label` (optionally quoted): the label doubles as the id,
333
+ // slugified — but preserve unicode letters/numbers so a CJK-only title
334
+ // doesn't collapse to an empty id (which broke nested layout).
335
+ label = normalizeBrTags(stripWrappingQuotes(rest))
336
+ id = label.replace(/\s+/g, '_').replace(/[^\p{L}\p{N}_-]/gu, '')
337
+ }
338
+ const sg: MermaidSubgraph = { id, label, nodeIds: [], children: [] }
339
+ subgraphStack.push(sg)
340
+ continue
341
+ }
342
+
343
+ // --- subgraph end ---
344
+ if (line === 'end') {
345
+ const completed = subgraphStack.pop()
346
+ if (completed) {
347
+ if (subgraphStack.length > 0) {
348
+ subgraphStack[subgraphStack.length - 1]!.children.push(completed)
349
+ } else {
350
+ graph.subgraphs.push(completed)
351
+ }
352
+ }
353
+ continue
354
+ }
355
+
356
+ // --- Edge/node definitions ---
357
+ parseEdgeLine(line, graph, subgraphStack)
358
+ }
359
+
360
+ return graph
361
+ }
362
+
363
+ // ============================================================================
364
+ // State diagram parser
365
+ //
366
+ // Supported syntax:
367
+ // stateDiagram-v2
368
+ // s1 : Description
369
+ // state "Description" as s1
370
+ // s1 --> s2 : label
371
+ // [*] --> s1 (start pseudostate)
372
+ // s1 --> [*] (end pseudostate)
373
+ // state CompositeState {
374
+ // inner1 --> inner2
375
+ // }
376
+ // classDef name fill:#f00 (style class — shared helpers)
377
+ // class s1,s2 name
378
+ // s1:::name --> s2 (inline class shorthand)
379
+ // ============================================================================
380
+
381
+ function parseStateDiagram(lines: Statement[]): MermaidGraph {
382
+ const graph: MermaidGraph = {
383
+ direction: 'TD',
384
+ nodes: new Map(),
385
+ edges: [],
386
+ subgraphs: [],
387
+ classDefs: new Map(),
388
+ classAssignments: new Map(),
389
+ nodeStyles: new Map(),
390
+ linkStyles: new Map(),
391
+ interactions: new Map(),
392
+ }
393
+
394
+ // Track composite state nesting (like subgraphs)
395
+ const compositeStack: MermaidSubgraph[] = []
396
+ // Track all composite state IDs to avoid creating duplicate nodes
397
+ const compositeStateIds = new Set<string>()
398
+ // Counter for unique [*] pseudostate IDs
399
+ let startCount = 0
400
+ let endCount = 0
401
+
402
+ for (let i = 1; i < lines.length; i++) {
403
+ const stmt = lines[i]!
404
+ const line = stmt.text
405
+
406
+ // --- classDef / class assignment / style — shared with the flowchart and
407
+ // class-diagram parsers, see packages/core/src/style-directives.ts.
408
+ // Mermaid can't style `[*]` or composite states yet, so an assignment to
409
+ // one of those ids is stored but never matches a rendered node. ---
410
+ if (tryApplyClassDef(line, graph)) continue
411
+ if (tryApplyClassAssignment(line, graph)) continue
412
+ if (tryApplyStyleStatement(line, graph)) continue
413
+
414
+ // --- direction override ---
415
+ const dirMatch = line.match(/^direction\s+(TD|TB|LR|BT|RL)\s*$/i)
416
+ if (dirMatch) {
417
+ if (compositeStack.length > 0) {
418
+ compositeStack[compositeStack.length - 1]!.direction = toDirection(
419
+ dirMatch[1],
420
+ )
421
+ } else {
422
+ graph.direction = toDirection(dirMatch[1])
423
+ }
424
+ continue
425
+ }
426
+
427
+ // --- linkStyle: `linkStyle 0 stroke:#f00` or `linkStyle default stroke:#f00` ---
428
+ const linkStyleMatch = line.match(/^linkStyle\s+(default|[\d,\s]+)\s+(.+)$/)
429
+ if (linkStyleMatch) {
430
+ const target = linkStyleMatch[1]!.trim()
431
+ const props = parseStyleProps(linkStyleMatch[2]!)
432
+ if (target === 'default') {
433
+ graph.linkStyles.set('default', {
434
+ ...graph.linkStyles.get('default'),
435
+ ...props,
436
+ })
437
+ } else {
438
+ const indices = target.split(',').map((s) => parseInt(s.trim(), 10))
439
+ for (const idx of indices) {
440
+ if (!isNaN(idx)) {
441
+ graph.linkStyles.set(idx, {
442
+ ...graph.linkStyles.get(idx),
443
+ ...props,
444
+ })
445
+ }
446
+ }
447
+ }
448
+ continue
449
+ }
450
+
451
+ // --- edge metadata: `e1@{ animate: true }` — shared with parseFlowchart ---
452
+ if (tryApplyEdgeMetaLine(line, graph)) {
453
+ continue
454
+ }
455
+
456
+ // --- composite state start: `state CompositeState {` ---
457
+ const compositeMatch = line.match(
458
+ /^state\s+(?:"([^"]+)"\s+as\s+)?([\w\p{L}]+)\s*\{$/u,
459
+ )
460
+ if (compositeMatch) {
461
+ const label = compositeMatch[1] ?? compositeMatch[2]!
462
+ const id = compositeMatch[2]!
463
+ const sg: MermaidSubgraph = { id, label, nodeIds: [], children: [] }
464
+ compositeStack.push(sg)
465
+ // Track this ID to avoid creating a duplicate node for the composite state
466
+ compositeStateIds.add(id)
467
+ // Remove any existing node that was created when parsing transitions before
468
+ // this composite state definition (e.g., "A --> Processing" before "state Processing {")
469
+ graph.nodes.delete(id)
470
+ continue
471
+ }
472
+
473
+ // --- composite state end ---
474
+ if (line === '}') {
475
+ const completed = compositeStack.pop()
476
+ if (completed) {
477
+ if (compositeStack.length > 0) {
478
+ compositeStack[compositeStack.length - 1]!.children.push(completed)
479
+ } else {
480
+ graph.subgraphs.push(completed)
481
+ }
482
+ }
483
+ continue
484
+ }
485
+
486
+ // --- state alias: `state "Description" as s1` (without brace) ---
487
+ const stateAliasMatch = line.match(
488
+ /^state\s+"([^"]+)"\s+as\s+([\w\p{L}]+)\s*$/u,
489
+ )
490
+ if (stateAliasMatch) {
491
+ const label = normalizeBrTags(stateAliasMatch[1]!)
492
+ const id = stateAliasMatch[2]!
493
+ registerStateDescription(graph, compositeStack, id, label)
494
+ continue
495
+ }
496
+
497
+ /*
498
+ * --- transition: `s1 --> s2`, `s1 --> s2 : label`, `[*] --> s1`, or
499
+ * with an edge id (Mermaid v11.10.0+): `s1 e1@--> s2` ---
500
+ *
501
+ * State-diagram transitions only ever use `-->` (unlike flowchart's
502
+ * many arrow variants), so the edge id — same `id@` prefix syntax as
503
+ * flowchart's `A e1@--> B` — is captured inline in this one regex
504
+ * rather than needing flowchart's separate pre-arrow scan.
505
+ */
506
+ const transitionMatch = line.match(
507
+ /^(\[\*\]|[\w\p{L}-]+)(?::::([\w][\w-]*))?\s*(?:([\w-]+)@)?-->\s*(\[\*\]|[\w\p{L}-]+)(?::::([\w][\w-]*))?(?:\s*:\s*(.+))?$/u,
508
+ )
509
+ if (transitionMatch) {
510
+ let sourceId = transitionMatch[1]!
511
+ const sourceClass = transitionMatch[2]
512
+ const edgeId = transitionMatch[3]
513
+ let targetId = transitionMatch[4]!
514
+ const targetClass = transitionMatch[5]
515
+ const rawTransitionLabel = transitionMatch[6]?.trim()
516
+ const edgeLabel = rawTransitionLabel
517
+ ? normalizeBrTags(rawTransitionLabel)
518
+ : undefined
519
+
520
+ // Handle [*] pseudostates — each occurrence gets a unique ID
521
+ if (sourceId === '[*]') {
522
+ startCount++
523
+ sourceId = `_start${startCount > 1 ? startCount : ''}`
524
+ registerStateNode(graph, compositeStack, {
525
+ id: sourceId,
526
+ label: '',
527
+ shape: 'state-start',
528
+ })
529
+ } else if (!compositeStateIds.has(sourceId)) {
530
+ // Only create a node if this isn't a composite state
531
+ ensureStateNode(graph, compositeStack, sourceId)
532
+ }
533
+
534
+ if (targetId === '[*]') {
535
+ endCount++
536
+ targetId = `_end${endCount > 1 ? endCount : ''}`
537
+ registerStateNode(graph, compositeStack, {
538
+ id: targetId,
539
+ label: '',
540
+ shape: 'state-end',
541
+ })
542
+ } else if (!compositeStateIds.has(targetId)) {
543
+ // Only create a node if this isn't a composite state
544
+ ensureStateNode(graph, compositeStack, targetId)
545
+ }
546
+
547
+ // `S1:::name --> S2:::name` — inline class shorthand on either end
548
+ if (sourceClass !== undefined) {
549
+ graph.classAssignments.set(sourceId, sourceClass)
550
+ }
551
+ if (targetClass !== undefined) {
552
+ graph.classAssignments.set(targetId, targetClass)
553
+ }
554
+
555
+ graph.edges.push({
556
+ source: sourceId,
557
+ target: targetId,
558
+ label: edgeLabel,
559
+ style: 'solid',
560
+ hasArrowStart: false,
561
+ hasArrowEnd: true,
562
+ ...(edgeId !== undefined ? { id: edgeId } : {}),
563
+ })
564
+ continue
565
+ }
566
+
567
+ // --- bare class shorthand: `S2:::name`. Must precede the description
568
+ // rule below, which would otherwise read it as `S2 : ::name`. ---
569
+ const stateClassMatch = line.match(/^([\w\p{L}-]+):::([\w][\w-]*)\s*$/u)
570
+ if (stateClassMatch) {
571
+ const id = stateClassMatch[1]!
572
+ if (!compositeStateIds.has(id)) {
573
+ ensureStateNode(graph, compositeStack, id)
574
+ }
575
+ graph.classAssignments.set(id, stateClassMatch[2]!)
576
+ continue
577
+ }
578
+
579
+ // --- state description: `s1 : Description` ---
580
+ const stateDescMatch = line.match(/^([\w\p{L}-]+)\s*:\s*(.+)$/u)
581
+ if (stateDescMatch) {
582
+ const id = stateDescMatch[1]!
583
+ const label = normalizeBrTags(stateDescMatch[2]!.trim())
584
+ registerStateDescription(graph, compositeStack, id, label)
585
+ continue
586
+ }
587
+ }
588
+
589
+ // `class S2 foo` on a state that no transition mentions still declares it
590
+ // (Mermaid creates the state on first reference). Composite ids are skipped:
591
+ // they render as clusters, not nodes.
592
+ for (const id of graph.classAssignments.keys()) {
593
+ if (!graph.nodes.has(id) && !compositeStateIds.has(id)) {
594
+ ensureStateNode(graph, [], id)
595
+ }
596
+ }
597
+
598
+ return graph
599
+ }
600
+
601
+ /**
602
+ * Register a state with an explicit description (`s1 : text` or
603
+ * `state "text" as s1`). Unlike a bare reference, this replaces the
604
+ * placeholder label an earlier `s1 --> s2` or `s1:::name` gave the node.
605
+ */
606
+ function registerStateDescription(
607
+ graph: MermaidGraph,
608
+ compositeStack: MermaidSubgraph[],
609
+ id: string,
610
+ label: string,
611
+ ): void {
612
+ const existing = graph.nodes.get(id)
613
+ if (existing) existing.label = label
614
+ registerStateNode(graph, compositeStack, { id, label, shape: 'rounded' })
615
+ }
616
+
617
+ /** Register a state node and track in composite state if applicable */
618
+ function registerStateNode(
619
+ graph: MermaidGraph,
620
+ compositeStack: MermaidSubgraph[],
621
+ node: MermaidNode,
622
+ ): void {
623
+ const isNew = !graph.nodes.has(node.id)
624
+ if (isNew) {
625
+ graph.nodes.set(node.id, node)
626
+ }
627
+ if (compositeStack.length > 0) {
628
+ const current = compositeStack[compositeStack.length - 1]!
629
+ if (!current.nodeIds.includes(node.id)) {
630
+ current.nodeIds.push(node.id)
631
+ }
632
+ }
633
+ }
634
+
635
+ /** Ensure a state node exists with default rounded shape */
636
+ function ensureStateNode(
637
+ graph: MermaidGraph,
638
+ compositeStack: MermaidSubgraph[],
639
+ id: string,
640
+ ): void {
641
+ if (!graph.nodes.has(id)) {
642
+ registerStateNode(graph, compositeStack, {
643
+ id,
644
+ label: id,
645
+ shape: 'rounded',
646
+ })
647
+ } else {
648
+ // Track in composite if applicable
649
+ if (compositeStack.length > 0) {
650
+ const current = compositeStack[compositeStack.length - 1]!
651
+ if (!current.nodeIds.includes(id)) {
652
+ current.nodeIds.push(id)
653
+ }
654
+ }
655
+ }
656
+ }
657
+
658
+ // ============================================================================
659
+ // Flowchart edge line parser
660
+ //
661
+ // Handles chained edges like: A[Label] --> B(Label) -.-> C{Label}
662
+ // Also handles & parallel links: A & B --> C & D
663
+ // ============================================================================
664
+
665
+ /**
666
+ * Arrow regex — matches all arrow operators with optional labels.
667
+ *
668
+ * Supported operators:
669
+ * --> --- solid arrow / solid line
670
+ * -.-> -.- dotted arrow / dotted line
671
+ * ==> === thick arrow / thick line
672
+ * <--> <-.-> <==> bidirectional variants
673
+ * --o --x circle-end / cross-end arrow (see issue #65)
674
+ * o-- x-- circle-start / cross-start arrow
675
+ * o--o x--x circle/cross at both ends
676
+ *
677
+ * The doubled `o--o`/`x--x` forms must be tried before the single-sided
678
+ * `o--`/`x--` forms — regex alternation picks the first alternative that
679
+ * matches, not the longest, so `o--` listed first would consume just the
680
+ * start marker and strand the trailing `o` as a bogus token. Capturing the
681
+ * markers as their own groups (rather than baking them into an alternation
682
+ * of whole tokens) sidesteps that ordering problem entirely.
683
+ *
684
+ * The body is a variable-length run: Mermaid uses extra characters as a
685
+ * layout-rank hint (`A ---- B` pushes B one rank further than `A --> B`).
686
+ * A fixed alternation of `-->|---|==>` mis-tokenized anything longer,
687
+ * stranding the surplus characters and corrupting the following token.
688
+ * The run length is parsed so the edge survives; the rank hint it encodes
689
+ * is not yet modelled by the layout engine.
690
+ *
691
+ * `~~~` is Mermaid's invisible link: it participates in layout but draws
692
+ * nothing.
693
+ *
694
+ * Optional label: -->|label text|
695
+ */
696
+ const ARROW_REGEX =
697
+ /^(<|o|x)?(-{2,}|={2,}|-\.+-|~{3,})(>|o|x)?(?:\s*\|([^|]*)\|)?/
698
+
699
+ /**
700
+ * Link bodies that are only a link when a start or end marker accompanies
701
+ * them.
702
+ *
703
+ * A bare `--` (or `==`) opens Mermaid's text-embedded label syntax —
704
+ * `A -- Yes --> B` — which TEXT_ARROW_REGEX handles. Treating it as an
705
+ * unmarked open link here would consume the opener and strand the label.
706
+ * Mermaid itself requires three characters for an unmarked open link
707
+ * (`A --- B`), so this only rejects what Mermaid also rejects.
708
+ */
709
+ const AMBIGUOUS_UNMARKED_BODIES = new Set(['--', '=='])
710
+
711
+ /**
712
+ * Map a raw start/end marker character (`<`, `>`, `o`, `x`, or undefined) to
713
+ * the distinct terminator shape it draws, if any.
714
+ *
715
+ * `<`/`>` are plain arrowheads — already handled by `hasArrowStart`/
716
+ * `hasArrowEnd` — so they map to `undefined` here, same as no marker at all.
717
+ * Only `o`/`x` need their own shape (circle/cross) recorded, so the ASCII
718
+ * renderer can draw something other than the default arrowhead glyph — see
719
+ * issue #330.
720
+ */
721
+ function markerKind(
722
+ marker: string | undefined,
723
+ ): 'circle' | 'cross' | undefined {
724
+ if (marker === 'o') return 'circle'
725
+ if (marker === 'x') return 'cross'
726
+ return undefined
727
+ }
728
+
729
+ /**
730
+ * Text-embedded label regex — matches "-- label -->", "-. label .->", "== label ==>" syntax.
731
+ * Tried as fallback when ARROW_REGEX doesn't match.
732
+ *
733
+ * The closing operator is a variable-length run, matching ARROW_REGEX. While
734
+ * it was a fixed alternation, `A -- label ----> B` consumed only `---` from
735
+ * `---->`, leaving `-> B`, which forms no node group — so the edge and its
736
+ * target node were both dropped silently. Exactly the failure mode the
737
+ * variable-length work exists to remove.
738
+ *
739
+ * Based on PR #36 by @liuxiaopai-ai (https://github.com/lukilabs/beautiful-mermaid/pull/36)
740
+ */
741
+ const TEXT_ARROW_REGEX =
742
+ /^(<|o|x)?(--|-\.|==)\s+(.+?)\s+(-{2,}[>ox]|={2,}[>ox]|\.+-[>ox]|-{3,}|={3,}|-\.+-)/
743
+
744
+ /**
745
+ * Node shape patterns — ordered from most specific delimiters to least.
746
+ * Multi-char delimiters must be tried before single-char to avoid false matches.
747
+ *
748
+ * The label-content group for each shape is quote-aware: it matches either a
749
+ * complete `"..."` quoted span (any character, including this shape's own
750
+ * closing delimiter chars, is fine inside quotes) or a single character that
751
+ * isn't the start of the closing delimiter. This stops the scanner from
752
+ * treating a `]`/`)`/`}` etc. *inside* a quoted label as the node's real
753
+ * closing bracket (see issue #61) — without this, `A["test [] brackets"]`
754
+ * would stop at the first `]`, which lives inside the quoted string.
755
+ */
756
+ /**
757
+ * A node shape pattern.
758
+ *
759
+ * `shape` is usually a fixed `NodeShape`. For the slash-bracket family it is
760
+ * a function instead, because those four shapes are distinguished by which
761
+ * delimiter *closes* them — which the regex has to capture rather than
762
+ * hard-code. See SLASH_BRACKET note below.
763
+ */
764
+ interface NodePattern {
765
+ regex: RegExp
766
+ shape: NodeShape | ((closingDelimiter: string) => NodeShape)
767
+ }
768
+
769
+ const NODE_PATTERNS: NodePattern[] = [
770
+ // Triple delimiters (must be first)
771
+ {
772
+ regex: /^([\w\p{L}-]+)\(\(\(((?:"[^"]*"|(?!\)\)\)).)+)\)\)\)/u,
773
+ shape: 'doublecircle',
774
+ }, // A(((text)))
775
+
776
+ // Double delimiters with mixed brackets
777
+ {
778
+ regex: /^([\w\p{L}-]+)\(\[((?:"[^"]*"|(?!\]\)).)+)\]\)/u,
779
+ shape: 'stadium',
780
+ }, // A([text])
781
+ { regex: /^([\w\p{L}-]+)\(\(((?:"[^"]*"|(?!\)\)).)+)\)\)/u, shape: 'circle' }, // A((text))
782
+ {
783
+ regex: /^([\w\p{L}-]+)\[\[((?:"[^"]*"|(?!\]\]).)+)\]\]/u,
784
+ shape: 'subroutine',
785
+ }, // A[[text]]
786
+ {
787
+ regex: /^([\w\p{L}-]+)\[\(((?:"[^"]*"|(?!\)\]).)+)\)\]/u,
788
+ shape: 'cylinder',
789
+ }, // A[(text)]
790
+
791
+ /*
792
+ * SLASH_BRACKET family — must come before plain [text].
793
+ *
794
+ * Four shapes share a `[` + slash opener and differ only in which
795
+ * delimiter closes them:
796
+ *
797
+ * A[/text\] trapezoid A[/text/] parallelogram
798
+ * A[\text/] trapezoid-alt A[\text\] parallelogram-alt
799
+ *
800
+ * They CANNOT be four separate patterns each negated against its own
801
+ * closing pair. A pattern for `[/…\]` whose content merely excludes `\]`
802
+ * will happily run past a `/]` to reach a `\]` later on the same line, so
803
+ *
804
+ * A[/parallelogram/] --> B[\alt\]
805
+ *
806
+ * matched as a single trapezoid whose label was the entire statement —
807
+ * silently swallowing the edge and the second node. Ordering the patterns
808
+ * differently only moves which input breaks.
809
+ *
810
+ * Instead, one pattern per opener stops at whichever slash-close comes
811
+ * first and captures it, and the captured delimiter selects the shape.
812
+ */
813
+ {
814
+ regex: /^([\w\p{L}-]+)\[\/((?:"[^"]*"|(?![\\/]\]).)+)([\\/])\]/u,
815
+ shape: (close) => (close === '\\' ? 'trapezoid' : 'parallelogram'),
816
+ }, // A[/text\] or A[/text/]
817
+ {
818
+ regex: /^([\w\p{L}-]+)\[\\((?:"[^"]*"|(?![\\/]\]).)+)([\\/])\]/u,
819
+ shape: (close) => (close === '/' ? 'trapezoid-alt' : 'parallelogram-alt'),
820
+ }, // A[\text/] or A[\text\]
821
+
822
+ // Asymmetric flag shape
823
+ { regex: /^([\w\p{L}-]+)>((?:"[^"]*"|(?!\]).)+)\]/u, shape: 'asymmetric' }, // A>text]
824
+
825
+ // Double curly braces (hexagon) — must come before single {text}
826
+ {
827
+ regex: /^([\w\p{L}-]+)\{\{((?:"[^"]*"|(?!\}\}).)+)\}\}/u,
828
+ shape: 'hexagon',
829
+ }, // A{{text}}
830
+
831
+ // Single-char delimiters (last — most common, least specific)
832
+ { regex: /^([\w\p{L}-]+)\[((?:"[^"]*"|(?!\]).)+)\]/u, shape: 'rectangle' }, // A[text]
833
+ { regex: /^([\w\p{L}-]+)\(((?:"[^"]*"|(?!\)).)+)\)/u, shape: 'rounded' }, // A(text)
834
+ { regex: /^([\w\p{L}-]+)\{((?:"[^"]*"|(?!\}).)+)\}/u, shape: 'diamond' }, // A{text}
835
+ ]
836
+
837
+ /**
838
+ * Regex for a bare node reference (just an ID, no shape brackets).
839
+ *
840
+ * Only allows a hyphen when it's sandwiched between word characters
841
+ * (`foo-bar`, `step-1-b`), never a bare/trailing/doubled hyphen. This keeps
842
+ * legitimately hyphenated ids intact while stopping the id from swallowing
843
+ * the leading dashes of an immediately-following arrow when there's no
844
+ * whitespace before it — e.g. `A-->B` (see issue #61): the naive `[\w-]+`
845
+ * greedily consumed `A--`, leaving a bogus node and no edge. Arrow tokens
846
+ * always start with `-`/`=`/`<` followed by another non-word character
847
+ * (`-`, `.`, `=`, `>`), which this pattern never matches into, so it now
848
+ * stops cleanly at `A` and lets the arrow regex take over.
849
+ *
850
+ * `\p{L}` (any Unicode letter, alongside plain `\w`) is included so a bare
851
+ * non-ASCII name (`Lasaña`, `日本`) matches in full instead of truncating at
852
+ * the first non-ASCII character and silently stranding the rest of the line
853
+ * as unparsed text — see issue #328. This mirrors the state-diagram
854
+ * transition regex below, which already allows `\p{L}` for the same reason.
855
+ */
856
+ const BARE_NODE_REGEX = /^([\w\p{L}]+(?:-[\w\p{L}]+)*)/u
857
+
858
+ /**
859
+ * Node id immediately followed by the expanded-syntax opener: `A@{`.
860
+ *
861
+ * Only the id is captured — the block itself needs depth- and quote-aware
862
+ * scanning (a label may contain `}`), which a regex would do badly, so
863
+ * `matchExpandedBlock` takes over from here.
864
+ */
865
+ const EXPANDED_NODE_ID_REGEX = /^([\w\p{L}-]+)(?=@\{)/u
866
+
867
+ /**
868
+ * Resolve the geometry for an `A@{ ... }` node.
869
+ *
870
+ * An `icon:` or `img:` node has no `shape:` of its own — its outline comes
871
+ * from `form:` instead (Mermaid defaults to a square). An unrecognized shape
872
+ * name falls back to a rectangle rather than throwing: Mermaid adds shape
873
+ * names regularly, and rendering a plain box beats failing the whole diagram
874
+ * over one unknown name.
875
+ */
876
+ function expandedNodeShape(meta: ExpandedNodeMeta): NodeShape {
877
+ if (meta.shape) {
878
+ return resolveShapeName(meta.shape) ?? 'rectangle'
879
+ }
880
+
881
+ if (meta.icon !== undefined || meta.img !== undefined) {
882
+ switch (meta.form?.toLowerCase()) {
883
+ case 'circle':
884
+ return 'circle'
885
+ case 'rounded':
886
+ return 'rounded'
887
+ default:
888
+ return 'rectangle'
889
+ }
890
+ }
891
+
892
+ return 'rectangle'
893
+ }
894
+
895
+ /**
896
+ * Resolve the display label for an `A@{ ... }` node.
897
+ *
898
+ * Falls back to the node id when no `label:` is given, matching how the
899
+ * bracket syntax treats a bare `A`. For an icon or image node with no label,
900
+ * the icon/image reference itself is shown: this renderer draws neither
901
+ * FontAwesome glyphs nor remote images, so showing the reference is more
902
+ * useful than an empty box, and it keeps the node identifiable.
903
+ */
904
+ function expandedNodeLabel(id: string, meta: ExpandedNodeMeta): string {
905
+ if (meta.label !== undefined && meta.label.length > 0) return meta.label
906
+ if (meta.icon) return meta.icon
907
+ if (meta.img) return meta.img
908
+ return id
909
+ }
910
+
911
+ /**
912
+ * Apply a `click` statement to the graph.
913
+ *
914
+ * Mermaid's forms:
915
+ * click A "https://example.com"
916
+ * click A "https://example.com" "Tooltip"
917
+ * click A "https://example.com" _blank
918
+ * click A href "https://example.com" "Tooltip" _blank
919
+ * click A call myCallback()
920
+ * click A callback "Tooltip"
921
+ *
922
+ * A callback is recorded but never invoked — this renderer emits static SVG
923
+ * and does not execute script supplied by a diagram. An href, by contrast, is
924
+ * genuinely actionable: the node is wrapped in an SVG <a>, which works in any
925
+ * browser without script.
926
+ *
927
+ * The actual grammar and href-safety rules live in packages/core/src/click-directive.ts,
928
+ * shared with the class diagram parser
929
+ * (packages/mermaid-parser/src/class/parser.ts) — this is a thin wrapper
930
+ * binding it to this parser's `graph.interactions` map.
931
+ */
932
+ function applyClickStatement(line: string, graph: MermaidGraph): void {
933
+ applyClickStatementShared(line, graph.interactions)
934
+ }
935
+
936
+ /**
937
+ * Id immediately followed by the expanded-syntax opener on a standalone
938
+ * metadata line: `e1@{ ... }`.
939
+ *
940
+ * Shared by both parsers (`tryApplyEdgeMetaLine`) since the syntax and its
941
+ * "must already be a known edge id" disambiguation rule are identical for
942
+ * flowcharts and state diagrams — only how edge ids get declared differs
943
+ * (`A e1@--> B` vs. `A e1@--> B` with a simpler arrow set).
944
+ */
945
+ const EDGE_META_ID_REGEX = /^([\w-]+)(?=@\{)/
946
+
947
+ /**
948
+ * Try to parse and apply a standalone `e1@{ animate: true }` edge-metadata
949
+ * line.
950
+ *
951
+ * Told apart from a node's `A@{ shape: ... }` by whether the id was already
952
+ * declared as an edge id (`A e1@--> B`) — checked before node parsing so an
953
+ * edge id is never registered as a stray node. Shared between
954
+ * `parseFlowchart` and `parseStateDiagram`; returns `true` if the line was
955
+ * consumed as edge metadata (caller should `continue` its loop).
956
+ */
957
+ function tryApplyEdgeMetaLine(line: string, graph: MermaidGraph): boolean {
958
+ const edgeMetaMatch = line.match(EDGE_META_ID_REGEX)
959
+ if (!edgeMetaMatch || !graph.edges.some((e) => e.id === edgeMetaMatch[1])) {
960
+ return false
961
+ }
962
+ const block = matchExpandedBlock(line.slice(edgeMetaMatch[1]!.length))
963
+ if (!block) return false
964
+ applyEdgeMeta(graph, edgeMetaMatch[1]!, parseExpandedMeta(block.body))
965
+ return true
966
+ }
967
+
968
+ /**
969
+ * Apply an `e1@{ ... }` metadata block to the edge with that id.
970
+ *
971
+ * Only `animate` is acted on; Mermaid's other edge keys (`animation`,
972
+ * `curve`) are recorded as no-ops rather than errors, matching how an
973
+ * unknown node shape degrades.
974
+ */
975
+ function applyEdgeMeta(
976
+ graph: MermaidGraph,
977
+ edgeId: string,
978
+ meta: Record<string, string | undefined>,
979
+ ): void {
980
+ for (const edge of graph.edges) {
981
+ if (edge.id !== edgeId) continue
982
+ if (meta.animate !== undefined) {
983
+ edge.animate = meta.animate.toLowerCase() !== 'false'
984
+ }
985
+ // Mermaid's `animation: fast|slow` is a speed hint on top of animate.
986
+ if (meta.animation !== undefined) edge.animate = true
987
+ }
988
+ }
989
+
990
+ /** Regex for ::: class shorthand suffix — matches :::className immediately after a node */
991
+ const CLASS_SHORTHAND_REGEX = /^:::([\w][\w-]*)/
992
+
993
+ /**
994
+ * Regex for ::: class shorthand appearing BEFORE the shape brackets,
995
+ * e.g. A:::external[Label]. Captures the bare id and the class name so
996
+ * the shorthand can be stripped out before shape-pattern matching runs.
997
+ *
998
+ * The id group allows `\p{L}` (see BARE_NODE_REGEX above) so a non-ASCII
999
+ * id keeps working with this shorthand instead of falling through to
1000
+ * BARE_NODE_REGEX with the `:::className` suffix left dangling as
1001
+ * unparsed text. The class-name group stays ASCII-only — Mermaid class
1002
+ * names follow CSS identifier conventions, not node-name conventions.
1003
+ */
1004
+ const PRE_CLASS_SHORTHAND_REGEX = /^([\w\p{L}-]+):::([\w][\w-]*)/u
1005
+
1006
+ /**
1007
+ * Parse a line that contains node definitions and edges.
1008
+ * Handles chaining: A --> B --> C produces edges A→B and B→C.
1009
+ * Handles parallel links: A & B --> C & D produces 4 edges.
1010
+ */
1011
+ function parseEdgeLine(
1012
+ line: string,
1013
+ graph: MermaidGraph,
1014
+ subgraphStack: MermaidSubgraph[],
1015
+ ): void {
1016
+ let remaining = line.trim()
1017
+
1018
+ // Parse the first node group (possibly with & separators)
1019
+ const firstGroup = consumeNodeGroup(remaining, graph, subgraphStack)
1020
+ if (!firstGroup || firstGroup.ids.length === 0) return
1021
+
1022
+ remaining = firstGroup.remaining.trim()
1023
+ let prevGroupIds = firstGroup.ids
1024
+
1025
+ // Parse arrow + node-group pairs until the line is exhausted
1026
+ while (remaining.length > 0) {
1027
+ let hasArrowStart: boolean
1028
+ let style: EdgeStyle
1029
+ let hasArrowEnd: boolean
1030
+ let edgeLabel: string | undefined
1031
+ let startMarkerKind: 'circle' | 'cross' | undefined
1032
+ let endMarkerKind: 'circle' | 'cross' | undefined
1033
+
1034
+ /*
1035
+ * Optional edge id prefix: `A e1@--> B` (Mermaid v11.10.0+). Consumed
1036
+ * before the arrow so ARROW_REGEX still sees the link token at position
1037
+ * zero. `@` is not otherwise valid ahead of a link, so this cannot
1038
+ * shadow existing syntax.
1039
+ */
1040
+ let edgeId: string | undefined
1041
+ const edgeIdMatch = remaining.match(/^([\w-]+)@(?=[-=<~ox])/)
1042
+ if (edgeIdMatch) {
1043
+ edgeId = edgeIdMatch[1]
1044
+ remaining = remaining.slice(edgeIdMatch[0].length)
1045
+ }
1046
+
1047
+ const arrowMatch = remaining.match(ARROW_REGEX)
1048
+ const arrowBody = arrowMatch?.[2]
1049
+ const startMarker = arrowMatch?.[1]
1050
+ const endMarker = arrowMatch?.[3]
1051
+
1052
+ if (
1053
+ arrowMatch &&
1054
+ arrowBody !== undefined &&
1055
+ // An unmarked `--`/`==` is the text-label opener, not a link.
1056
+ !(
1057
+ startMarker === undefined &&
1058
+ endMarker === undefined &&
1059
+ AMBIGUOUS_UNMARKED_BODIES.has(arrowBody)
1060
+ )
1061
+ ) {
1062
+ // `o`/`x` mark a circle/cross terminator (alongside `<` for a reversed
1063
+ // arrow) — `markerKind` records which, so the ASCII renderer can draw
1064
+ // a distinct glyph instead of the default arrowhead (see issue #330,
1065
+ // a follow-up to issue #65 which first stopped these from being
1066
+ // dropped entirely).
1067
+ hasArrowStart = startMarker !== undefined
1068
+ startMarkerKind = markerKind(startMarker)
1069
+ const rawEdgeLabel = arrowMatch[4]?.trim()
1070
+ edgeLabel = rawEdgeLabel ? normalizeBrTags(rawEdgeLabel) : undefined
1071
+ remaining = remaining.slice(arrowMatch[0].length).trim()
1072
+ style = arrowStyleFromBody(arrowBody)
1073
+ hasArrowEnd = endMarker !== undefined
1074
+ endMarkerKind = markerKind(endMarker)
1075
+ } else {
1076
+ // Fallback: text-embedded label syntax (-- Yes -->, -. Maybe .->, == Sure ==>)
1077
+ const textMatch = remaining.match(TEXT_ARROW_REGEX)
1078
+ if (!textMatch) break
1079
+ hasArrowStart = Boolean(textMatch[1])
1080
+ startMarkerKind = markerKind(textMatch[1])
1081
+ const rawLabel = textMatch[3]!.trim()
1082
+ edgeLabel = rawLabel ? normalizeBrTags(rawLabel) : undefined
1083
+ const openOp = textMatch[2]!
1084
+ const closeOp = textMatch[4]!
1085
+ remaining = remaining.slice(textMatch[0].length).trim()
1086
+ style = textArrowStyleFromOps(openOp, closeOp)
1087
+ // closeOp is a variable-length run (e.g. "---o", "==x", "-.-.->"); only
1088
+ // its final character can be a circle/cross marker. A circle/cross
1089
+ // terminator is its own kind of arrow ending — like `>` — so it counts
1090
+ // toward hasArrowEnd too; an unmarked run (`---`, `===`, `-.-`) is the
1091
+ // only case with no arrowhead of any kind at the close.
1092
+ endMarkerKind = markerKind(closeOp.slice(-1))
1093
+ hasArrowEnd = closeOp.endsWith('>') || endMarkerKind !== undefined
1094
+ }
1095
+
1096
+ // Parse the next node group
1097
+ const nextGroup = consumeNodeGroup(remaining, graph, subgraphStack)
1098
+ if (!nextGroup || nextGroup.ids.length === 0) break
1099
+
1100
+ remaining = nextGroup.remaining.trim()
1101
+
1102
+ // Emit Cartesian product of edges: every source × every target
1103
+ for (const sourceId of prevGroupIds) {
1104
+ for (const targetId of nextGroup.ids) {
1105
+ graph.edges.push({
1106
+ source: sourceId,
1107
+ target: targetId,
1108
+ label: edgeLabel,
1109
+ style,
1110
+ hasArrowStart,
1111
+ hasArrowEnd,
1112
+ ...(startMarkerKind !== undefined
1113
+ ? { startMarker: startMarkerKind }
1114
+ : {}),
1115
+ ...(endMarkerKind !== undefined ? { endMarker: endMarkerKind } : {}),
1116
+ ...(edgeId !== undefined ? { id: edgeId } : {}),
1117
+ })
1118
+ }
1119
+ }
1120
+
1121
+ prevGroupIds = nextGroup.ids
1122
+ }
1123
+ }
1124
+
1125
+ interface ConsumedNodeGroup {
1126
+ ids: string[]
1127
+ remaining: string
1128
+ }
1129
+
1130
+ /**
1131
+ * Consume one or more nodes separated by `&`.
1132
+ * E.g. "A & B & C --> ..." returns ids: ['A', 'B', 'C']
1133
+ */
1134
+ function consumeNodeGroup(
1135
+ text: string,
1136
+ graph: MermaidGraph,
1137
+ subgraphStack: MermaidSubgraph[],
1138
+ ): ConsumedNodeGroup | null {
1139
+ const first = consumeNode(text, graph, subgraphStack)
1140
+ if (!first) return null
1141
+
1142
+ const ids = [first.id]
1143
+ let remaining = first.remaining.trim()
1144
+
1145
+ // Check for & separators
1146
+ while (remaining.startsWith('&')) {
1147
+ remaining = remaining.slice(1).trim()
1148
+ const next = consumeNode(remaining, graph, subgraphStack)
1149
+ if (!next) break
1150
+ ids.push(next.id)
1151
+ remaining = next.remaining.trim()
1152
+ }
1153
+
1154
+ return { ids, remaining }
1155
+ }
1156
+
1157
+ interface ConsumedNode {
1158
+ id: string
1159
+ remaining: string
1160
+ }
1161
+
1162
+ /**
1163
+ * Try to consume a node definition from the start of `text`.
1164
+ * If the node has a shape+label (e.g. A[Text]), it's registered in the graph.
1165
+ * If it's a bare reference (e.g. A), we look it up or create a default.
1166
+ * Also handles ::: class shorthand suffix.
1167
+ */
1168
+ function consumeNode(
1169
+ text: string,
1170
+ graph: MermaidGraph,
1171
+ subgraphStack: MermaidSubgraph[],
1172
+ ): ConsumedNode | null {
1173
+ let id: string | null = null
1174
+ let remaining: string = text
1175
+
1176
+ // Check for ::: class shorthand appearing BEFORE the shape brackets
1177
+ // (e.g. A:::external[Label]). Strip it out so shape-pattern matching
1178
+ // below still sees the id directly adjacent to its brackets.
1179
+ let preClassName: string | undefined
1180
+ const preClassMatch = remaining.match(PRE_CLASS_SHORTHAND_REGEX)
1181
+ if (preClassMatch) {
1182
+ preClassName = preClassMatch[2]!
1183
+ remaining = preClassMatch[1]! + remaining.slice(preClassMatch[0].length)
1184
+ }
1185
+
1186
+ /*
1187
+ * Expanded syntax: `A@{ shape: doc, label: "Report" }` (Mermaid v11.3.0+).
1188
+ *
1189
+ * Tried before the bracket patterns because the id is followed by `@{`,
1190
+ * which none of them match — without this the id would fall through to
1191
+ * BARE_NODE_REGEX and the whole metadata block would be stranded as
1192
+ * unparsed text.
1193
+ */
1194
+ const expandedIdMatch = remaining.match(EXPANDED_NODE_ID_REGEX)
1195
+ if (expandedIdMatch) {
1196
+ const expandedId = expandedIdMatch[1]!
1197
+ const block = matchExpandedBlock(remaining.slice(expandedId.length))
1198
+ if (block) {
1199
+ const meta = parseExpandedMeta(block.body)
1200
+ registerNode(graph, subgraphStack, {
1201
+ id: expandedId,
1202
+ label: normalizeBrTags(expandedNodeLabel(expandedId, meta)),
1203
+ shape: expandedNodeShape(meta),
1204
+ })
1205
+ id = expandedId
1206
+ remaining = remaining.slice(expandedId.length + block.length)
1207
+ }
1208
+ }
1209
+
1210
+ // Try each node pattern (shape-qualified)
1211
+ if (id === null) {
1212
+ for (const { regex, shape } of NODE_PATTERNS) {
1213
+ const match = remaining.match(regex)
1214
+ if (match) {
1215
+ id = match[1]!
1216
+ const label = normalizeBrTags(match[2]!)
1217
+ // The slash-bracket family resolves its shape from the closing
1218
+ // delimiter it captured in group 3; every other pattern has a fixed
1219
+ // shape. See the SLASH_BRACKET note in NODE_PATTERNS.
1220
+ const resolvedShape =
1221
+ typeof shape === 'function' ? shape(match[3] ?? '') : shape
1222
+ registerNode(graph, subgraphStack, { id, label, shape: resolvedShape })
1223
+ remaining = remaining.slice(match[0].length)
1224
+ break
1225
+ }
1226
+ }
1227
+ }
1228
+
1229
+ // Bare node reference — register it if the node doesn't exist yet.
1230
+ // If it already exists, it stays in the subgraph that claimed it first
1231
+ // (nodes belong to the subgraph where they're first defined). A node that
1232
+ // no subgraph has claimed yet (declared earlier at the top level, e.g.
1233
+ // `X --> A` before `subgraph SA; A; end`) is adopted by the current one,
1234
+ // as Mermaid does; otherwise the frame would come out empty (#1289).
1235
+ if (id === null) {
1236
+ const bareMatch = remaining.match(BARE_NODE_REGEX)
1237
+ if (bareMatch) {
1238
+ id = bareMatch[1]!
1239
+ if (!graph.nodes.has(id)) {
1240
+ registerNode(graph, subgraphStack, {
1241
+ id,
1242
+ label: id,
1243
+ shape: 'rectangle',
1244
+ })
1245
+ } else if (!isClaimedBySubgraph(graph, subgraphStack, id)) {
1246
+ trackInSubgraph(subgraphStack, id)
1247
+ }
1248
+ remaining = remaining.slice(bareMatch[0].length)
1249
+ }
1250
+ }
1251
+
1252
+ if (id === null) return null
1253
+
1254
+ if (preClassName) {
1255
+ graph.classAssignments.set(id, preClassName)
1256
+ }
1257
+
1258
+ // Check for ::: class shorthand suffix immediately after the node
1259
+ const classMatch = remaining.match(CLASS_SHORTHAND_REGEX)
1260
+ if (classMatch) {
1261
+ graph.classAssignments.set(id, classMatch[1]!)
1262
+ remaining = remaining.slice(classMatch[0].length)
1263
+ }
1264
+
1265
+ return { id, remaining }
1266
+ }
1267
+
1268
+ /** Register a node in the graph and track it in the current subgraph */
1269
+ function registerNode(
1270
+ graph: MermaidGraph,
1271
+ subgraphStack: MermaidSubgraph[],
1272
+ node: MermaidNode,
1273
+ ): void {
1274
+ const isNew = !graph.nodes.has(node.id)
1275
+ if (isNew) {
1276
+ graph.nodes.set(node.id, node)
1277
+ }
1278
+ trackInSubgraph(subgraphStack, node.id)
1279
+ }
1280
+
1281
+ /** True when any finished or still-open subgraph already lists `nodeId`. */
1282
+ function isClaimedBySubgraph(
1283
+ graph: MermaidGraph,
1284
+ subgraphStack: MermaidSubgraph[],
1285
+ nodeId: string,
1286
+ ): boolean {
1287
+ const claims = (sg: MermaidSubgraph): boolean =>
1288
+ sg.nodeIds.includes(nodeId) || sg.children.some(claims)
1289
+ return graph.subgraphs.some(claims) || subgraphStack.some(claims)
1290
+ }
1291
+
1292
+ /** Add node ID to the innermost subgraph if we're inside one */
1293
+ function trackInSubgraph(
1294
+ subgraphStack: MermaidSubgraph[],
1295
+ nodeId: string,
1296
+ ): void {
1297
+ if (subgraphStack.length > 0) {
1298
+ const current = subgraphStack[subgraphStack.length - 1]!
1299
+ if (!current.nodeIds.includes(nodeId)) {
1300
+ current.nodeIds.push(nodeId)
1301
+ }
1302
+ }
1303
+ }
1304
+
1305
+ /**
1306
+ * Map a link body to its edge style, ignoring direction and run length.
1307
+ *
1308
+ * The body is the run between the optional start/end markers: `--`, `----`,
1309
+ * `==`, `-.-`, `-..-`, `~~~`, etc. Classification is by the characters used,
1310
+ * not the length, so every run length of a given style behaves identically.
1311
+ */
1312
+ function arrowStyleFromBody(body: string): EdgeStyle {
1313
+ if (body.startsWith('~')) return 'invisible'
1314
+ if (body.includes('.')) return 'dotted'
1315
+ if (body.startsWith('=')) return 'thick'
1316
+ return 'solid'
1317
+ }
1318
+
1319
+ /** Map text-embedded arrow open/close operators to edge style */
1320
+ function textArrowStyleFromOps(openOp: string, closeOp: string): EdgeStyle {
1321
+ // Classify by the characters used, not by exact token, so every run length
1322
+ // of a given style resolves identically — the same rule arrowStyleFromBody
1323
+ // applies to the plain arrow forms.
1324
+ if (openOp.includes('.') || closeOp.includes('.')) return 'dotted'
1325
+ if (openOp.startsWith('=') || closeOp.startsWith('=')) return 'thick'
1326
+ return 'solid'
1327
+ }