@zombie-mermaid/mermaid-parser 3.2.0 → 4.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1324 @@
1
+ import type {
2
+ MermaidGraph,
3
+ MermaidNode,
4
+ MermaidSubgraph,
5
+ NodeShape,
6
+ EdgeStyle,
7
+ Statement,
8
+ } from '@zombie-mermaid/core'
9
+ import {
10
+ toDirection,
11
+ normalizeBrTags,
12
+ splitStatementsByLine,
13
+ extractInitConfig,
14
+ applyClickStatement as applyClickStatementShared,
15
+ parseStyleProps,
16
+ tryApplyClassDef,
17
+ tryApplyClassAssignment,
18
+ tryApplyStyleStatement,
19
+ } from '@zombie-mermaid/core'
20
+
21
+ import {
22
+ matchExpandedBlock,
23
+ parseExpandedMeta,
24
+ resolveShapeName,
25
+ } from './expanded-shapes.ts'
26
+ import type { ExpandedNodeMeta } from './expanded-shapes.ts'
27
+ /** Remove a single layer of matching wrapping quotes (`"…"` or `'…'`). */
28
+ function stripWrappingQuotes(s: string): string {
29
+ const t = s.trim()
30
+ if (
31
+ t.length >= 2 &&
32
+ ((t[0] === '"' && t[t.length - 1] === '"') ||
33
+ (t[0] === "'" && t[t.length - 1] === "'"))
34
+ ) {
35
+ return t.slice(1, -1)
36
+ }
37
+ return t
38
+ }
39
+
40
+ // ============================================================================
41
+ // Mermaid parser — flowcharts and state diagrams
42
+ //
43
+ // Supports:
44
+ // Flowcharts: graph TD / flowchart LR
45
+ // State diagrams: stateDiagram-v2
46
+ //
47
+ // Line-by-line regex approach — the grammar is regular enough
48
+ // that we don't need a grammar generator or full parser combinator.
49
+ // ============================================================================
50
+
51
+ // `toDirection` used to be defined here, with this file as its only caller.
52
+ // #624 gave it a second caller — `packages/mermaid-parser/src/er/parser.ts`,
53
+ // which cannot import this umbrella file without creating a cycle
54
+ // (mermaid-parser -> umbrella -> mermaid-parser) — so it moved to
55
+ // `packages/core/src/direction.ts` alongside `isDirection`, for the same
56
+ // reason that one moved under #625. Re-exported here so existing importers
57
+ // of `toDirection` from `./parser.ts` (this file's own tests included) keep
58
+ // working unchanged.
59
+ export { toDirection } from '@zombie-mermaid/core'
60
+
61
+ /**
62
+ * All diagram-type headers this library recognizes, for the "supported
63
+ * headers" list in the invalid-header error below. Kept in one place so a
64
+ * newly-added diagram type doesn't get forgotten in the error message the
65
+ * way sequence/class/ER/xychart were before issue #541.
66
+ */
67
+ const SUPPORTED_HEADERS =
68
+ '"graph <dir>"/"flowchart <dir>" (dir: TD, TB, LR, BT, RL), "stateDiagram-v2", "sequenceDiagram", "classDiagram", "erDiagram", "xychart-beta", "C4Context"/"C4Container"/"C4Component"/"C4Dynamic"/"C4Deployment", "architecture-beta"'
69
+
70
+ /**
71
+ * Best-guess canonical header for a header line that looks like an attempt
72
+ * at one of the *other* diagram types this library supports, but didn't
73
+ * match `detectDiagramType`'s exact-match regex (a typo, extra trailing
74
+ * text, wrong case pattern, etc.) and so fell through to this flowchart
75
+ * parser as the default. Returns undefined when `header` doesn't look like
76
+ * any of those — in which case the generic "supported headers" list in the
77
+ * thrown error is the best we can do.
78
+ *
79
+ * This only recognizes the four *other* multi-word headers
80
+ * (sequenceDiagram/classDiagram/erDiagram/xychart-beta/stateDiagram-v2);
81
+ * "graph"/"flowchart" typos are handled by the direction-specific message
82
+ * below instead, since those already matched the diagram-type keyword and
83
+ * just have a bad or missing direction token.
84
+ */
85
+ function suggestedHeaderFor(header: string): string | undefined {
86
+ const lower = header.trim().toLowerCase()
87
+ if (/^sequence/.test(lower)) return 'sequenceDiagram'
88
+ if (/^class/.test(lower)) return 'classDiagram'
89
+ if (/^er/.test(lower)) return 'erDiagram'
90
+ if (/^xychart/.test(lower)) return 'xychart-beta'
91
+ if (/^state/.test(lower)) return 'stateDiagram-v2'
92
+ return undefined
93
+ }
94
+
95
+ /**
96
+ * Matches the start of a statement that continues the *previous* one's edge
97
+ * chain rather than beginning a new one: Mermaid lets a vertex-chain
98
+ * statement break across lines, with the link operator leading the next
99
+ * line — e.g.
100
+ *
101
+ * start([Start])
102
+ * ==> green([Change some code])
103
+ * ==> finish([Finish])
104
+ *
105
+ * is one chained statement (`start ==> green ==> finish`), identical to
106
+ * writing it on one line. `splitStatementsByLine` only knows about newlines
107
+ * and `;` as separators, so without this it hands `parseFlowchart` three
108
+ * unrelated-looking statements — the continuation lines start with a bare
109
+ * arrow and no node group, so `parseEdgeLine` can't find a source node and
110
+ * drops them entirely (see issue mermaid-js/mermaid#6049's repro, reported
111
+ * against this parser).
112
+ *
113
+ * Covers every arrow opener `parseEdgeLine` itself recognizes (ARROW_REGEX /
114
+ * TEXT_ARROW_REGEX below): a solid/thick run (`--`, `===`), a dotted run
115
+ * (`-.`, `-.-`), a `~~~` run, each optionally preceded by a `<` marker and/or
116
+ * an edge id (`e1@-->`).
117
+ *
118
+ * Deliberately excludes the `o`/`x` start markers ARROW_REGEX also accepts
119
+ * (`o--o`, `x--x`): unlike `<`, both are also valid bare node ids, so a
120
+ * first-body-line statement like `x-->Y` would otherwise be misread as an
121
+ * `x`-marked continuation of the header and merged into it, corrupting the
122
+ * header line (`flowchart TD x-->Y`) instead of parsing `x` as its own node.
123
+ * A genuine marked-start continuation line is rare enough that losing it is
124
+ * the safer tradeoff.
125
+ */
126
+ const CONTINUATION_START_REGEX = /^(?:[\w-]+@)?<?(?:-{2,}|={2,}|-\.+-?|~{3,})/
127
+
128
+ /**
129
+ * Rejoin a continuation statement onto the one it continues.
130
+ *
131
+ * `splitStatementsByLine` groups statements by their originating physical
132
+ * line, which is what makes this safe: only the *first* statement in a
133
+ * group is eligible to merge backward into the previous group's last
134
+ * statement (a genuine multi-line continuation, see
135
+ * `CONTINUATION_START_REGEX`). A later statement in the same group got
136
+ * there via an explicit `;` on that line (`A --> B; --> C`) — that boundary
137
+ * must stay a boundary, or `--> C` would wrongly become part of the chain
138
+ * instead of the source-less fragment it actually is.
139
+ */
140
+ function mergeContinuationLines(groups: Statement[][]): Statement[] {
141
+ const merged: Statement[] = []
142
+ for (const group of groups) {
143
+ group.forEach((statement, index) => {
144
+ if (
145
+ index === 0 &&
146
+ merged.length > 0 &&
147
+ CONTINUATION_START_REGEX.test(statement.text)
148
+ ) {
149
+ // Keep the *first* line's number: a merged statement is reported at
150
+ // the line its logical statement began, not the continuation line.
151
+ const prev = merged[merged.length - 1]!
152
+ merged[merged.length - 1] = {
153
+ text: `${prev.text} ${statement.text}`,
154
+ line: prev.line,
155
+ }
156
+ } else {
157
+ merged.push(statement)
158
+ }
159
+ })
160
+ }
161
+ return merged
162
+ }
163
+
164
+ /**
165
+ * Parse Mermaid text into a logical graph structure.
166
+ * Auto-detects diagram type (flowchart or state diagram).
167
+ * Throws on invalid/unsupported input.
168
+ */
169
+ export function parseMermaid(text: string): MermaidGraph {
170
+ /*
171
+ * Init directives must be read from the raw lines: `%%{init: ...}%%` begins
172
+ * with `%%`, so splitStatementsByLine — which treats `%%` as a comment and
173
+ * truncates the line there — would otherwise discard it along with real
174
+ * comments before it could be seen.
175
+ */
176
+ const initConfig = extractInitConfig(text.split('\n').map((l) => l.trim()))
177
+
178
+ const lines = mergeContinuationLines(splitStatementsByLine(text))
179
+
180
+ if (lines.length === 0) {
181
+ throw new Error('Empty mermaid diagram')
182
+ }
183
+
184
+ // Detect diagram type from header
185
+ const headerStmt = lines[0]!
186
+
187
+ // State diagram: "stateDiagram-v2" or "stateDiagram"
188
+ const graph = /^stateDiagram(-v2)?\s*$/i.test(headerStmt.text)
189
+ ? parseStateDiagram(lines)
190
+ : parseFlowchart(lines)
191
+
192
+ graph.initConfig = initConfig
193
+ return graph
194
+ }
195
+
196
+ // ============================================================================
197
+ // Flowchart parser
198
+ // ============================================================================
199
+
200
+ function parseFlowchart(lines: Statement[]): MermaidGraph {
201
+ // parseFlowchart is only ever invoked by parseMermaid, which has already
202
+ // verified `lines` is non-empty — but that invariant isn't visible to the
203
+ // type checker across the function boundary, so validate it here instead
204
+ // of asserting past it.
205
+ const headerStmt = lines[0]
206
+ if (headerStmt === undefined) {
207
+ /* v8 ignore next */
208
+ throw new Error('parseFlowchart called with no lines')
209
+ }
210
+ const header = headerStmt.text
211
+
212
+ const headerMatch = header.match(
213
+ /^(?:graph|flowchart)\s+(TD|TB|LR|BT|RL)\s*$/i,
214
+ )
215
+ if (!headerMatch) {
216
+ // A header that starts with "graph"/"flowchart" but has a bad or
217
+ // missing direction gets a targeted message about the direction token
218
+ // specifically — it already picked the right diagram type.
219
+ const partialFlowchartMatch = header.match(
220
+ /^(?:graph|flowchart)\b\s*(.*)$/i,
221
+ )
222
+ if (partialFlowchartMatch) {
223
+ const rest = partialFlowchartMatch[1]!.trim()
224
+ throw new Error(
225
+ rest.length > 0
226
+ ? `Line ${headerStmt.line}: Invalid direction "${rest}" in header "${header}". Expected one of: TD, TB, LR, BT, RL.`
227
+ : `Line ${headerStmt.line}: Missing direction in header "${header}". Expected e.g. "graph TD" or "flowchart LR" — one of: TD, TB, LR, BT, RL.`,
228
+ )
229
+ }
230
+
231
+ // Otherwise this header didn't match any recognized diagram-type
232
+ // keyword at all (typo, extra text, or genuinely unsupported syntax) —
233
+ // point at the closest known diagram type when the text resembles one.
234
+ const suggestion = suggestedHeaderFor(header)
235
+ const hint =
236
+ suggestion && suggestion.toLowerCase() !== header.trim().toLowerCase()
237
+ ? ` Did you mean "${suggestion}"?`
238
+ : ''
239
+ throw new Error(
240
+ `Line ${headerStmt.line}: Invalid mermaid header: "${header}".${hint} Supported headers: ${SUPPORTED_HEADERS}.`,
241
+ )
242
+ }
243
+
244
+ const direction = toDirection(headerMatch[1])
245
+
246
+ const graph: MermaidGraph = {
247
+ direction,
248
+ nodes: new Map(),
249
+ edges: [],
250
+ subgraphs: [],
251
+ classDefs: new Map(),
252
+ classAssignments: new Map(),
253
+ nodeStyles: new Map(),
254
+ linkStyles: new Map(),
255
+ interactions: new Map(),
256
+ }
257
+
258
+ // Subgraph stack for nested subgraphs.
259
+ const subgraphStack: MermaidSubgraph[] = []
260
+
261
+ for (let i = 1; i < lines.length; i++) {
262
+ const stmt = lines[i]!
263
+ const line = stmt.text
264
+
265
+ // --- classDef / class assignment / style — shared with the class-diagram
266
+ // parser, see packages/core/src/style-directives.ts ---
267
+ if (tryApplyClassDef(line, graph)) continue
268
+ if (tryApplyClassAssignment(line, graph)) continue
269
+ if (tryApplyStyleStatement(line, graph)) continue
270
+
271
+ // --- click interaction: `click A "url" "tooltip" _blank` / `click A call fn()` ---
272
+ if (/^click\s+/i.test(line)) {
273
+ applyClickStatement(line, graph)
274
+ continue
275
+ }
276
+
277
+ // --- edge metadata: `e1@{ animate: true }` — shared with parseStateDiagram ---
278
+ if (tryApplyEdgeMetaLine(line, graph)) {
279
+ continue
280
+ }
281
+
282
+ // --- linkStyle: `linkStyle 0 stroke:#f00` or `linkStyle default stroke:#f00` ---
283
+ const linkStyleMatch = line.match(/^linkStyle\s+(default|[\d,\s]+)\s+(.+)$/)
284
+ if (linkStyleMatch) {
285
+ const target = linkStyleMatch[1]!.trim()
286
+ const props = parseStyleProps(linkStyleMatch[2]!)
287
+ if (target === 'default') {
288
+ graph.linkStyles.set('default', {
289
+ ...graph.linkStyles.get('default'),
290
+ ...props,
291
+ })
292
+ } else {
293
+ const indices = target.split(',').map((s) => parseInt(s.trim(), 10))
294
+ for (const idx of indices) {
295
+ if (!isNaN(idx)) {
296
+ graph.linkStyles.set(idx, {
297
+ ...graph.linkStyles.get(idx),
298
+ ...props,
299
+ })
300
+ }
301
+ }
302
+ }
303
+ continue
304
+ }
305
+
306
+ // --- direction override inside subgraph: `direction LR` ---
307
+ const dirMatch = line.match(/^direction\s+(TD|TB|LR|BT|RL)\s*$/i)
308
+ if (dirMatch && subgraphStack.length > 0) {
309
+ subgraphStack[subgraphStack.length - 1]!.direction = toDirection(
310
+ dirMatch[1],
311
+ )
312
+ continue
313
+ }
314
+
315
+ // --- subgraph start: `subgraph Label` or `subgraph id [Label]` ---
316
+ const subgraphMatch = line.match(/^subgraph\s+(.+)$/)
317
+ if (subgraphMatch) {
318
+ const rest = subgraphMatch[1]!.trim()
319
+ // Check for `subgraph id [Label]` / `subgraph id ["Label"]` form. The id
320
+ // may contain non-ASCII chars (e.g. CJK), so match any run of non-space,
321
+ // non-`[` chars rather than ASCII [\w-]; strip optional quotes wrapping
322
+ // the title.
323
+ const bracketMatch = rest.match(/^([^\s[]+)\s*\[(.+)\]$/)
324
+ let id: string
325
+ let label: string
326
+ if (bracketMatch) {
327
+ id = bracketMatch[1]!
328
+ label = normalizeBrTags(stripWrappingQuotes(bracketMatch[2]!))
329
+ } else {
330
+ // `subgraph Label` (optionally quoted): the label doubles as the id,
331
+ // slugified — but preserve unicode letters/numbers so a CJK-only title
332
+ // doesn't collapse to an empty id (which broke nested layout).
333
+ label = normalizeBrTags(stripWrappingQuotes(rest))
334
+ id = label.replace(/\s+/g, '_').replace(/[^\p{L}\p{N}_-]/gu, '')
335
+ }
336
+ const sg: MermaidSubgraph = { id, label, nodeIds: [], children: [] }
337
+ subgraphStack.push(sg)
338
+ continue
339
+ }
340
+
341
+ // --- subgraph end ---
342
+ if (line === 'end') {
343
+ const completed = subgraphStack.pop()
344
+ if (completed) {
345
+ if (subgraphStack.length > 0) {
346
+ subgraphStack[subgraphStack.length - 1]!.children.push(completed)
347
+ } else {
348
+ graph.subgraphs.push(completed)
349
+ }
350
+ }
351
+ continue
352
+ }
353
+
354
+ // --- Edge/node definitions ---
355
+ parseEdgeLine(line, graph, subgraphStack)
356
+ }
357
+
358
+ return graph
359
+ }
360
+
361
+ // ============================================================================
362
+ // State diagram parser
363
+ //
364
+ // Supported syntax:
365
+ // stateDiagram-v2
366
+ // s1 : Description
367
+ // state "Description" as s1
368
+ // s1 --> s2 : label
369
+ // [*] --> s1 (start pseudostate)
370
+ // s1 --> [*] (end pseudostate)
371
+ // state CompositeState {
372
+ // inner1 --> inner2
373
+ // }
374
+ // classDef name fill:#f00 (style class — shared helpers)
375
+ // class s1,s2 name
376
+ // s1:::name --> s2 (inline class shorthand)
377
+ // ============================================================================
378
+
379
+ function parseStateDiagram(lines: Statement[]): MermaidGraph {
380
+ const graph: MermaidGraph = {
381
+ direction: 'TD',
382
+ nodes: new Map(),
383
+ edges: [],
384
+ subgraphs: [],
385
+ classDefs: new Map(),
386
+ classAssignments: new Map(),
387
+ nodeStyles: new Map(),
388
+ linkStyles: new Map(),
389
+ interactions: new Map(),
390
+ }
391
+
392
+ // Track composite state nesting (like subgraphs)
393
+ const compositeStack: MermaidSubgraph[] = []
394
+ // Track all composite state IDs to avoid creating duplicate nodes
395
+ const compositeStateIds = new Set<string>()
396
+ // Counter for unique [*] pseudostate IDs
397
+ let startCount = 0
398
+ let endCount = 0
399
+
400
+ for (let i = 1; i < lines.length; i++) {
401
+ const stmt = lines[i]!
402
+ const line = stmt.text
403
+
404
+ // --- classDef / class assignment / style — shared with the flowchart and
405
+ // class-diagram parsers, see packages/core/src/style-directives.ts.
406
+ // Mermaid can't style `[*]` or composite states yet, so an assignment to
407
+ // one of those ids is stored but never matches a rendered node. ---
408
+ if (tryApplyClassDef(line, graph)) continue
409
+ if (tryApplyClassAssignment(line, graph)) continue
410
+ if (tryApplyStyleStatement(line, graph)) continue
411
+
412
+ // --- direction override ---
413
+ const dirMatch = line.match(/^direction\s+(TD|TB|LR|BT|RL)\s*$/i)
414
+ if (dirMatch) {
415
+ if (compositeStack.length > 0) {
416
+ compositeStack[compositeStack.length - 1]!.direction = toDirection(
417
+ dirMatch[1],
418
+ )
419
+ } else {
420
+ graph.direction = toDirection(dirMatch[1])
421
+ }
422
+ continue
423
+ }
424
+
425
+ // --- linkStyle: `linkStyle 0 stroke:#f00` or `linkStyle default stroke:#f00` ---
426
+ const linkStyleMatch = line.match(/^linkStyle\s+(default|[\d,\s]+)\s+(.+)$/)
427
+ if (linkStyleMatch) {
428
+ const target = linkStyleMatch[1]!.trim()
429
+ const props = parseStyleProps(linkStyleMatch[2]!)
430
+ if (target === 'default') {
431
+ graph.linkStyles.set('default', {
432
+ ...graph.linkStyles.get('default'),
433
+ ...props,
434
+ })
435
+ } else {
436
+ const indices = target.split(',').map((s) => parseInt(s.trim(), 10))
437
+ for (const idx of indices) {
438
+ if (!isNaN(idx)) {
439
+ graph.linkStyles.set(idx, {
440
+ ...graph.linkStyles.get(idx),
441
+ ...props,
442
+ })
443
+ }
444
+ }
445
+ }
446
+ continue
447
+ }
448
+
449
+ // --- edge metadata: `e1@{ animate: true }` — shared with parseFlowchart ---
450
+ if (tryApplyEdgeMetaLine(line, graph)) {
451
+ continue
452
+ }
453
+
454
+ // --- composite state start: `state CompositeState {` ---
455
+ const compositeMatch = line.match(
456
+ /^state\s+(?:"([^"]+)"\s+as\s+)?([\w\p{L}]+)\s*\{$/u,
457
+ )
458
+ if (compositeMatch) {
459
+ const label = compositeMatch[1] ?? compositeMatch[2]!
460
+ const id = compositeMatch[2]!
461
+ const sg: MermaidSubgraph = { id, label, nodeIds: [], children: [] }
462
+ compositeStack.push(sg)
463
+ // Track this ID to avoid creating a duplicate node for the composite state
464
+ compositeStateIds.add(id)
465
+ // Remove any existing node that was created when parsing transitions before
466
+ // this composite state definition (e.g., "A --> Processing" before "state Processing {")
467
+ graph.nodes.delete(id)
468
+ continue
469
+ }
470
+
471
+ // --- composite state end ---
472
+ if (line === '}') {
473
+ const completed = compositeStack.pop()
474
+ if (completed) {
475
+ if (compositeStack.length > 0) {
476
+ compositeStack[compositeStack.length - 1]!.children.push(completed)
477
+ } else {
478
+ graph.subgraphs.push(completed)
479
+ }
480
+ }
481
+ continue
482
+ }
483
+
484
+ // --- state alias: `state "Description" as s1` (without brace) ---
485
+ const stateAliasMatch = line.match(
486
+ /^state\s+"([^"]+)"\s+as\s+([\w\p{L}]+)\s*$/u,
487
+ )
488
+ if (stateAliasMatch) {
489
+ const label = normalizeBrTags(stateAliasMatch[1]!)
490
+ const id = stateAliasMatch[2]!
491
+ registerStateDescription(graph, compositeStack, id, label)
492
+ continue
493
+ }
494
+
495
+ /*
496
+ * --- transition: `s1 --> s2`, `s1 --> s2 : label`, `[*] --> s1`, or
497
+ * with an edge id (Mermaid v11.10.0+): `s1 e1@--> s2` ---
498
+ *
499
+ * State-diagram transitions only ever use `-->` (unlike flowchart's
500
+ * many arrow variants), so the edge id — same `id@` prefix syntax as
501
+ * flowchart's `A e1@--> B` — is captured inline in this one regex
502
+ * rather than needing flowchart's separate pre-arrow scan.
503
+ */
504
+ const transitionMatch = line.match(
505
+ /^(\[\*\]|[\w\p{L}-]+)(?::::([\w][\w-]*))?\s*(?:([\w-]+)@)?-->\s*(\[\*\]|[\w\p{L}-]+)(?::::([\w][\w-]*))?(?:\s*:\s*(.+))?$/u,
506
+ )
507
+ if (transitionMatch) {
508
+ let sourceId = transitionMatch[1]!
509
+ const sourceClass = transitionMatch[2]
510
+ const edgeId = transitionMatch[3]
511
+ let targetId = transitionMatch[4]!
512
+ const targetClass = transitionMatch[5]
513
+ const rawTransitionLabel = transitionMatch[6]?.trim()
514
+ const edgeLabel = rawTransitionLabel
515
+ ? normalizeBrTags(rawTransitionLabel)
516
+ : undefined
517
+
518
+ // Handle [*] pseudostates — each occurrence gets a unique ID
519
+ if (sourceId === '[*]') {
520
+ startCount++
521
+ sourceId = `_start${startCount > 1 ? startCount : ''}`
522
+ registerStateNode(graph, compositeStack, {
523
+ id: sourceId,
524
+ label: '',
525
+ shape: 'state-start',
526
+ })
527
+ } else if (!compositeStateIds.has(sourceId)) {
528
+ // Only create a node if this isn't a composite state
529
+ ensureStateNode(graph, compositeStack, sourceId)
530
+ }
531
+
532
+ if (targetId === '[*]') {
533
+ endCount++
534
+ targetId = `_end${endCount > 1 ? endCount : ''}`
535
+ registerStateNode(graph, compositeStack, {
536
+ id: targetId,
537
+ label: '',
538
+ shape: 'state-end',
539
+ })
540
+ } else if (!compositeStateIds.has(targetId)) {
541
+ // Only create a node if this isn't a composite state
542
+ ensureStateNode(graph, compositeStack, targetId)
543
+ }
544
+
545
+ // `S1:::name --> S2:::name` — inline class shorthand on either end
546
+ if (sourceClass !== undefined) {
547
+ graph.classAssignments.set(sourceId, sourceClass)
548
+ }
549
+ if (targetClass !== undefined) {
550
+ graph.classAssignments.set(targetId, targetClass)
551
+ }
552
+
553
+ graph.edges.push({
554
+ source: sourceId,
555
+ target: targetId,
556
+ label: edgeLabel,
557
+ style: 'solid',
558
+ hasArrowStart: false,
559
+ hasArrowEnd: true,
560
+ ...(edgeId !== undefined ? { id: edgeId } : {}),
561
+ })
562
+ continue
563
+ }
564
+
565
+ // --- bare class shorthand: `S2:::name`. Must precede the description
566
+ // rule below, which would otherwise read it as `S2 : ::name`. ---
567
+ const stateClassMatch = line.match(/^([\w\p{L}-]+):::([\w][\w-]*)\s*$/u)
568
+ if (stateClassMatch) {
569
+ const id = stateClassMatch[1]!
570
+ if (!compositeStateIds.has(id)) {
571
+ ensureStateNode(graph, compositeStack, id)
572
+ }
573
+ graph.classAssignments.set(id, stateClassMatch[2]!)
574
+ continue
575
+ }
576
+
577
+ // --- state description: `s1 : Description` ---
578
+ const stateDescMatch = line.match(/^([\w\p{L}-]+)\s*:\s*(.+)$/u)
579
+ if (stateDescMatch) {
580
+ const id = stateDescMatch[1]!
581
+ const label = normalizeBrTags(stateDescMatch[2]!.trim())
582
+ registerStateDescription(graph, compositeStack, id, label)
583
+ continue
584
+ }
585
+ }
586
+
587
+ // `class S2 foo` on a state that no transition mentions still declares it
588
+ // (Mermaid creates the state on first reference). Composite ids are skipped:
589
+ // they render as clusters, not nodes.
590
+ for (const id of graph.classAssignments.keys()) {
591
+ if (!graph.nodes.has(id) && !compositeStateIds.has(id)) {
592
+ ensureStateNode(graph, [], id)
593
+ }
594
+ }
595
+
596
+ return graph
597
+ }
598
+
599
+ /**
600
+ * Register a state with an explicit description (`s1 : text` or
601
+ * `state "text" as s1`). Unlike a bare reference, this replaces the
602
+ * placeholder label an earlier `s1 --> s2` or `s1:::name` gave the node.
603
+ */
604
+ function registerStateDescription(
605
+ graph: MermaidGraph,
606
+ compositeStack: MermaidSubgraph[],
607
+ id: string,
608
+ label: string,
609
+ ): void {
610
+ const existing = graph.nodes.get(id)
611
+ if (existing) existing.label = label
612
+ registerStateNode(graph, compositeStack, { id, label, shape: 'rounded' })
613
+ }
614
+
615
+ /** Register a state node and track in composite state if applicable */
616
+ function registerStateNode(
617
+ graph: MermaidGraph,
618
+ compositeStack: MermaidSubgraph[],
619
+ node: MermaidNode,
620
+ ): void {
621
+ const isNew = !graph.nodes.has(node.id)
622
+ if (isNew) {
623
+ graph.nodes.set(node.id, node)
624
+ }
625
+ if (compositeStack.length > 0) {
626
+ const current = compositeStack[compositeStack.length - 1]!
627
+ if (!current.nodeIds.includes(node.id)) {
628
+ current.nodeIds.push(node.id)
629
+ }
630
+ }
631
+ }
632
+
633
+ /** Ensure a state node exists with default rounded shape */
634
+ function ensureStateNode(
635
+ graph: MermaidGraph,
636
+ compositeStack: MermaidSubgraph[],
637
+ id: string,
638
+ ): void {
639
+ if (!graph.nodes.has(id)) {
640
+ registerStateNode(graph, compositeStack, {
641
+ id,
642
+ label: id,
643
+ shape: 'rounded',
644
+ })
645
+ } else {
646
+ // Track in composite if applicable
647
+ if (compositeStack.length > 0) {
648
+ const current = compositeStack[compositeStack.length - 1]!
649
+ if (!current.nodeIds.includes(id)) {
650
+ current.nodeIds.push(id)
651
+ }
652
+ }
653
+ }
654
+ }
655
+
656
+ // ============================================================================
657
+ // Flowchart edge line parser
658
+ //
659
+ // Handles chained edges like: A[Label] --> B(Label) -.-> C{Label}
660
+ // Also handles & parallel links: A & B --> C & D
661
+ // ============================================================================
662
+
663
+ /**
664
+ * Arrow regex — matches all arrow operators with optional labels.
665
+ *
666
+ * Supported operators:
667
+ * --> --- solid arrow / solid line
668
+ * -.-> -.- dotted arrow / dotted line
669
+ * ==> === thick arrow / thick line
670
+ * <--> <-.-> <==> bidirectional variants
671
+ * --o --x circle-end / cross-end arrow (see issue #65)
672
+ * o-- x-- circle-start / cross-start arrow
673
+ * o--o x--x circle/cross at both ends
674
+ *
675
+ * The doubled `o--o`/`x--x` forms must be tried before the single-sided
676
+ * `o--`/`x--` forms — regex alternation picks the first alternative that
677
+ * matches, not the longest, so `o--` listed first would consume just the
678
+ * start marker and strand the trailing `o` as a bogus token. Capturing the
679
+ * markers as their own groups (rather than baking them into an alternation
680
+ * of whole tokens) sidesteps that ordering problem entirely.
681
+ *
682
+ * The body is a variable-length run: Mermaid uses extra characters as a
683
+ * layout-rank hint (`A ---- B` pushes B one rank further than `A --> B`).
684
+ * A fixed alternation of `-->|---|==>` mis-tokenized anything longer,
685
+ * stranding the surplus characters and corrupting the following token.
686
+ * The run length is parsed so the edge survives; the rank hint it encodes
687
+ * is not yet modelled by the layout engine.
688
+ *
689
+ * `~~~` is Mermaid's invisible link: it participates in layout but draws
690
+ * nothing.
691
+ *
692
+ * Optional label: -->|label text|
693
+ */
694
+ const ARROW_REGEX = /^(<|o|x)?(-{2,}|={2,}|-\.+-|~{3,})(>|o|x)?(?:\|([^|]*)\|)?/
695
+
696
+ /**
697
+ * Link bodies that are only a link when a start or end marker accompanies
698
+ * them.
699
+ *
700
+ * A bare `--` (or `==`) opens Mermaid's text-embedded label syntax —
701
+ * `A -- Yes --> B` — which TEXT_ARROW_REGEX handles. Treating it as an
702
+ * unmarked open link here would consume the opener and strand the label.
703
+ * Mermaid itself requires three characters for an unmarked open link
704
+ * (`A --- B`), so this only rejects what Mermaid also rejects.
705
+ */
706
+ const AMBIGUOUS_UNMARKED_BODIES = new Set(['--', '=='])
707
+
708
+ /**
709
+ * Map a raw start/end marker character (`<`, `>`, `o`, `x`, or undefined) to
710
+ * the distinct terminator shape it draws, if any.
711
+ *
712
+ * `<`/`>` are plain arrowheads — already handled by `hasArrowStart`/
713
+ * `hasArrowEnd` — so they map to `undefined` here, same as no marker at all.
714
+ * Only `o`/`x` need their own shape (circle/cross) recorded, so the ASCII
715
+ * renderer can draw something other than the default arrowhead glyph — see
716
+ * issue #330.
717
+ */
718
+ function markerKind(
719
+ marker: string | undefined,
720
+ ): 'circle' | 'cross' | undefined {
721
+ if (marker === 'o') return 'circle'
722
+ if (marker === 'x') return 'cross'
723
+ return undefined
724
+ }
725
+
726
+ /**
727
+ * Text-embedded label regex — matches "-- label -->", "-. label .->", "== label ==>" syntax.
728
+ * Tried as fallback when ARROW_REGEX doesn't match.
729
+ *
730
+ * The closing operator is a variable-length run, matching ARROW_REGEX. While
731
+ * it was a fixed alternation, `A -- label ----> B` consumed only `---` from
732
+ * `---->`, leaving `-> B`, which forms no node group — so the edge and its
733
+ * target node were both dropped silently. Exactly the failure mode the
734
+ * variable-length work exists to remove.
735
+ *
736
+ * Based on PR #36 by @liuxiaopai-ai (https://github.com/lukilabs/beautiful-mermaid/pull/36)
737
+ */
738
+ const TEXT_ARROW_REGEX =
739
+ /^(<|o|x)?(--|-\.|==)\s+(.+?)\s+(-{2,}[>ox]|={2,}[>ox]|\.+-[>ox]|-{3,}|={3,}|-\.+-)/
740
+
741
+ /**
742
+ * Node shape patterns — ordered from most specific delimiters to least.
743
+ * Multi-char delimiters must be tried before single-char to avoid false matches.
744
+ *
745
+ * The label-content group for each shape is quote-aware: it matches either a
746
+ * complete `"..."` quoted span (any character, including this shape's own
747
+ * closing delimiter chars, is fine inside quotes) or a single character that
748
+ * isn't the start of the closing delimiter. This stops the scanner from
749
+ * treating a `]`/`)`/`}` etc. *inside* a quoted label as the node's real
750
+ * closing bracket (see issue #61) — without this, `A["test [] brackets"]`
751
+ * would stop at the first `]`, which lives inside the quoted string.
752
+ */
753
+ /**
754
+ * A node shape pattern.
755
+ *
756
+ * `shape` is usually a fixed `NodeShape`. For the slash-bracket family it is
757
+ * a function instead, because those four shapes are distinguished by which
758
+ * delimiter *closes* them — which the regex has to capture rather than
759
+ * hard-code. See SLASH_BRACKET note below.
760
+ */
761
+ interface NodePattern {
762
+ regex: RegExp
763
+ shape: NodeShape | ((closingDelimiter: string) => NodeShape)
764
+ }
765
+
766
+ const NODE_PATTERNS: NodePattern[] = [
767
+ // Triple delimiters (must be first)
768
+ {
769
+ regex: /^([\w\p{L}-]+)\(\(\(((?:"[^"]*"|(?!\)\)\)).)+)\)\)\)/u,
770
+ shape: 'doublecircle',
771
+ }, // A(((text)))
772
+
773
+ // Double delimiters with mixed brackets
774
+ {
775
+ regex: /^([\w\p{L}-]+)\(\[((?:"[^"]*"|(?!\]\)).)+)\]\)/u,
776
+ shape: 'stadium',
777
+ }, // A([text])
778
+ { regex: /^([\w\p{L}-]+)\(\(((?:"[^"]*"|(?!\)\)).)+)\)\)/u, shape: 'circle' }, // A((text))
779
+ {
780
+ regex: /^([\w\p{L}-]+)\[\[((?:"[^"]*"|(?!\]\]).)+)\]\]/u,
781
+ shape: 'subroutine',
782
+ }, // A[[text]]
783
+ {
784
+ regex: /^([\w\p{L}-]+)\[\(((?:"[^"]*"|(?!\)\]).)+)\)\]/u,
785
+ shape: 'cylinder',
786
+ }, // A[(text)]
787
+
788
+ /*
789
+ * SLASH_BRACKET family — must come before plain [text].
790
+ *
791
+ * Four shapes share a `[` + slash opener and differ only in which
792
+ * delimiter closes them:
793
+ *
794
+ * A[/text\] trapezoid A[/text/] parallelogram
795
+ * A[\text/] trapezoid-alt A[\text\] parallelogram-alt
796
+ *
797
+ * They CANNOT be four separate patterns each negated against its own
798
+ * closing pair. A pattern for `[/…\]` whose content merely excludes `\]`
799
+ * will happily run past a `/]` to reach a `\]` later on the same line, so
800
+ *
801
+ * A[/parallelogram/] --> B[\alt\]
802
+ *
803
+ * matched as a single trapezoid whose label was the entire statement —
804
+ * silently swallowing the edge and the second node. Ordering the patterns
805
+ * differently only moves which input breaks.
806
+ *
807
+ * Instead, one pattern per opener stops at whichever slash-close comes
808
+ * first and captures it, and the captured delimiter selects the shape.
809
+ */
810
+ {
811
+ regex: /^([\w\p{L}-]+)\[\/((?:"[^"]*"|(?![\\/]\]).)+)([\\/])\]/u,
812
+ shape: (close) => (close === '\\' ? 'trapezoid' : 'parallelogram'),
813
+ }, // A[/text\] or A[/text/]
814
+ {
815
+ regex: /^([\w\p{L}-]+)\[\\((?:"[^"]*"|(?![\\/]\]).)+)([\\/])\]/u,
816
+ shape: (close) => (close === '/' ? 'trapezoid-alt' : 'parallelogram-alt'),
817
+ }, // A[\text/] or A[\text\]
818
+
819
+ // Asymmetric flag shape
820
+ { regex: /^([\w\p{L}-]+)>((?:"[^"]*"|(?!\]).)+)\]/u, shape: 'asymmetric' }, // A>text]
821
+
822
+ // Double curly braces (hexagon) — must come before single {text}
823
+ {
824
+ regex: /^([\w\p{L}-]+)\{\{((?:"[^"]*"|(?!\}\}).)+)\}\}/u,
825
+ shape: 'hexagon',
826
+ }, // A{{text}}
827
+
828
+ // Single-char delimiters (last — most common, least specific)
829
+ { regex: /^([\w\p{L}-]+)\[((?:"[^"]*"|(?!\]).)+)\]/u, shape: 'rectangle' }, // A[text]
830
+ { regex: /^([\w\p{L}-]+)\(((?:"[^"]*"|(?!\)).)+)\)/u, shape: 'rounded' }, // A(text)
831
+ { regex: /^([\w\p{L}-]+)\{((?:"[^"]*"|(?!\}).)+)\}/u, shape: 'diamond' }, // A{text}
832
+ ]
833
+
834
+ /**
835
+ * Regex for a bare node reference (just an ID, no shape brackets).
836
+ *
837
+ * Only allows a hyphen when it's sandwiched between word characters
838
+ * (`foo-bar`, `step-1-b`), never a bare/trailing/doubled hyphen. This keeps
839
+ * legitimately hyphenated ids intact while stopping the id from swallowing
840
+ * the leading dashes of an immediately-following arrow when there's no
841
+ * whitespace before it — e.g. `A-->B` (see issue #61): the naive `[\w-]+`
842
+ * greedily consumed `A--`, leaving a bogus node and no edge. Arrow tokens
843
+ * always start with `-`/`=`/`<` followed by another non-word character
844
+ * (`-`, `.`, `=`, `>`), which this pattern never matches into, so it now
845
+ * stops cleanly at `A` and lets the arrow regex take over.
846
+ *
847
+ * `\p{L}` (any Unicode letter, alongside plain `\w`) is included so a bare
848
+ * non-ASCII name (`Lasaña`, `日本`) matches in full instead of truncating at
849
+ * the first non-ASCII character and silently stranding the rest of the line
850
+ * as unparsed text — see issue #328. This mirrors the state-diagram
851
+ * transition regex below, which already allows `\p{L}` for the same reason.
852
+ */
853
+ const BARE_NODE_REGEX = /^([\w\p{L}]+(?:-[\w\p{L}]+)*)/u
854
+
855
+ /**
856
+ * Node id immediately followed by the expanded-syntax opener: `A@{`.
857
+ *
858
+ * Only the id is captured — the block itself needs depth- and quote-aware
859
+ * scanning (a label may contain `}`), which a regex would do badly, so
860
+ * `matchExpandedBlock` takes over from here.
861
+ */
862
+ const EXPANDED_NODE_ID_REGEX = /^([\w\p{L}-]+)(?=@\{)/u
863
+
864
+ /**
865
+ * Resolve the geometry for an `A@{ ... }` node.
866
+ *
867
+ * An `icon:` or `img:` node has no `shape:` of its own — its outline comes
868
+ * from `form:` instead (Mermaid defaults to a square). An unrecognized shape
869
+ * name falls back to a rectangle rather than throwing: Mermaid adds shape
870
+ * names regularly, and rendering a plain box beats failing the whole diagram
871
+ * over one unknown name.
872
+ */
873
+ function expandedNodeShape(meta: ExpandedNodeMeta): NodeShape {
874
+ if (meta.shape) {
875
+ return resolveShapeName(meta.shape) ?? 'rectangle'
876
+ }
877
+
878
+ if (meta.icon !== undefined || meta.img !== undefined) {
879
+ switch (meta.form?.toLowerCase()) {
880
+ case 'circle':
881
+ return 'circle'
882
+ case 'rounded':
883
+ return 'rounded'
884
+ default:
885
+ return 'rectangle'
886
+ }
887
+ }
888
+
889
+ return 'rectangle'
890
+ }
891
+
892
+ /**
893
+ * Resolve the display label for an `A@{ ... }` node.
894
+ *
895
+ * Falls back to the node id when no `label:` is given, matching how the
896
+ * bracket syntax treats a bare `A`. For an icon or image node with no label,
897
+ * the icon/image reference itself is shown: this renderer draws neither
898
+ * FontAwesome glyphs nor remote images, so showing the reference is more
899
+ * useful than an empty box, and it keeps the node identifiable.
900
+ */
901
+ function expandedNodeLabel(id: string, meta: ExpandedNodeMeta): string {
902
+ if (meta.label !== undefined && meta.label.length > 0) return meta.label
903
+ if (meta.icon) return meta.icon
904
+ if (meta.img) return meta.img
905
+ return id
906
+ }
907
+
908
+ /**
909
+ * Apply a `click` statement to the graph.
910
+ *
911
+ * Mermaid's forms:
912
+ * click A "https://example.com"
913
+ * click A "https://example.com" "Tooltip"
914
+ * click A "https://example.com" _blank
915
+ * click A href "https://example.com" "Tooltip" _blank
916
+ * click A call myCallback()
917
+ * click A callback "Tooltip"
918
+ *
919
+ * A callback is recorded but never invoked — this renderer emits static SVG
920
+ * and does not execute script supplied by a diagram. An href, by contrast, is
921
+ * genuinely actionable: the node is wrapped in an SVG <a>, which works in any
922
+ * browser without script.
923
+ *
924
+ * The actual grammar and href-safety rules live in packages/core/src/click-directive.ts,
925
+ * shared with the class diagram parser
926
+ * (packages/mermaid-parser/src/class/parser.ts) — this is a thin wrapper
927
+ * binding it to this parser's `graph.interactions` map.
928
+ */
929
+ function applyClickStatement(line: string, graph: MermaidGraph): void {
930
+ applyClickStatementShared(line, graph.interactions)
931
+ }
932
+
933
+ /**
934
+ * Id immediately followed by the expanded-syntax opener on a standalone
935
+ * metadata line: `e1@{ ... }`.
936
+ *
937
+ * Shared by both parsers (`tryApplyEdgeMetaLine`) since the syntax and its
938
+ * "must already be a known edge id" disambiguation rule are identical for
939
+ * flowcharts and state diagrams — only how edge ids get declared differs
940
+ * (`A e1@--> B` vs. `A e1@--> B` with a simpler arrow set).
941
+ */
942
+ const EDGE_META_ID_REGEX = /^([\w-]+)(?=@\{)/
943
+
944
+ /**
945
+ * Try to parse and apply a standalone `e1@{ animate: true }` edge-metadata
946
+ * line.
947
+ *
948
+ * Told apart from a node's `A@{ shape: ... }` by whether the id was already
949
+ * declared as an edge id (`A e1@--> B`) — checked before node parsing so an
950
+ * edge id is never registered as a stray node. Shared between
951
+ * `parseFlowchart` and `parseStateDiagram`; returns `true` if the line was
952
+ * consumed as edge metadata (caller should `continue` its loop).
953
+ */
954
+ function tryApplyEdgeMetaLine(line: string, graph: MermaidGraph): boolean {
955
+ const edgeMetaMatch = line.match(EDGE_META_ID_REGEX)
956
+ if (!edgeMetaMatch || !graph.edges.some((e) => e.id === edgeMetaMatch[1])) {
957
+ return false
958
+ }
959
+ const block = matchExpandedBlock(line.slice(edgeMetaMatch[1]!.length))
960
+ if (!block) return false
961
+ applyEdgeMeta(graph, edgeMetaMatch[1]!, parseExpandedMeta(block.body))
962
+ return true
963
+ }
964
+
965
+ /**
966
+ * Apply an `e1@{ ... }` metadata block to the edge with that id.
967
+ *
968
+ * Only `animate` is acted on; Mermaid's other edge keys (`animation`,
969
+ * `curve`) are recorded as no-ops rather than errors, matching how an
970
+ * unknown node shape degrades.
971
+ */
972
+ function applyEdgeMeta(
973
+ graph: MermaidGraph,
974
+ edgeId: string,
975
+ meta: Record<string, string | undefined>,
976
+ ): void {
977
+ for (const edge of graph.edges) {
978
+ if (edge.id !== edgeId) continue
979
+ if (meta.animate !== undefined) {
980
+ edge.animate = meta.animate.toLowerCase() !== 'false'
981
+ }
982
+ // Mermaid's `animation: fast|slow` is a speed hint on top of animate.
983
+ if (meta.animation !== undefined) edge.animate = true
984
+ }
985
+ }
986
+
987
+ /** Regex for ::: class shorthand suffix — matches :::className immediately after a node */
988
+ const CLASS_SHORTHAND_REGEX = /^:::([\w][\w-]*)/
989
+
990
+ /**
991
+ * Regex for ::: class shorthand appearing BEFORE the shape brackets,
992
+ * e.g. A:::external[Label]. Captures the bare id and the class name so
993
+ * the shorthand can be stripped out before shape-pattern matching runs.
994
+ *
995
+ * The id group allows `\p{L}` (see BARE_NODE_REGEX above) so a non-ASCII
996
+ * id keeps working with this shorthand instead of falling through to
997
+ * BARE_NODE_REGEX with the `:::className` suffix left dangling as
998
+ * unparsed text. The class-name group stays ASCII-only — Mermaid class
999
+ * names follow CSS identifier conventions, not node-name conventions.
1000
+ */
1001
+ const PRE_CLASS_SHORTHAND_REGEX = /^([\w\p{L}-]+):::([\w][\w-]*)/u
1002
+
1003
+ /**
1004
+ * Parse a line that contains node definitions and edges.
1005
+ * Handles chaining: A --> B --> C produces edges A→B and B→C.
1006
+ * Handles parallel links: A & B --> C & D produces 4 edges.
1007
+ */
1008
+ function parseEdgeLine(
1009
+ line: string,
1010
+ graph: MermaidGraph,
1011
+ subgraphStack: MermaidSubgraph[],
1012
+ ): void {
1013
+ let remaining = line.trim()
1014
+
1015
+ // Parse the first node group (possibly with & separators)
1016
+ const firstGroup = consumeNodeGroup(remaining, graph, subgraphStack)
1017
+ if (!firstGroup || firstGroup.ids.length === 0) return
1018
+
1019
+ remaining = firstGroup.remaining.trim()
1020
+ let prevGroupIds = firstGroup.ids
1021
+
1022
+ // Parse arrow + node-group pairs until the line is exhausted
1023
+ while (remaining.length > 0) {
1024
+ let hasArrowStart: boolean
1025
+ let style: EdgeStyle
1026
+ let hasArrowEnd: boolean
1027
+ let edgeLabel: string | undefined
1028
+ let startMarkerKind: 'circle' | 'cross' | undefined
1029
+ let endMarkerKind: 'circle' | 'cross' | undefined
1030
+
1031
+ /*
1032
+ * Optional edge id prefix: `A e1@--> B` (Mermaid v11.10.0+). Consumed
1033
+ * before the arrow so ARROW_REGEX still sees the link token at position
1034
+ * zero. `@` is not otherwise valid ahead of a link, so this cannot
1035
+ * shadow existing syntax.
1036
+ */
1037
+ let edgeId: string | undefined
1038
+ const edgeIdMatch = remaining.match(/^([\w-]+)@(?=[-=<~ox])/)
1039
+ if (edgeIdMatch) {
1040
+ edgeId = edgeIdMatch[1]
1041
+ remaining = remaining.slice(edgeIdMatch[0].length)
1042
+ }
1043
+
1044
+ const arrowMatch = remaining.match(ARROW_REGEX)
1045
+ const arrowBody = arrowMatch?.[2]
1046
+ const startMarker = arrowMatch?.[1]
1047
+ const endMarker = arrowMatch?.[3]
1048
+
1049
+ if (
1050
+ arrowMatch &&
1051
+ arrowBody !== undefined &&
1052
+ // An unmarked `--`/`==` is the text-label opener, not a link.
1053
+ !(
1054
+ startMarker === undefined &&
1055
+ endMarker === undefined &&
1056
+ AMBIGUOUS_UNMARKED_BODIES.has(arrowBody)
1057
+ )
1058
+ ) {
1059
+ // `o`/`x` mark a circle/cross terminator (alongside `<` for a reversed
1060
+ // arrow) — `markerKind` records which, so the ASCII renderer can draw
1061
+ // a distinct glyph instead of the default arrowhead (see issue #330,
1062
+ // a follow-up to issue #65 which first stopped these from being
1063
+ // dropped entirely).
1064
+ hasArrowStart = startMarker !== undefined
1065
+ startMarkerKind = markerKind(startMarker)
1066
+ const rawEdgeLabel = arrowMatch[4]?.trim()
1067
+ edgeLabel = rawEdgeLabel ? normalizeBrTags(rawEdgeLabel) : undefined
1068
+ remaining = remaining.slice(arrowMatch[0].length).trim()
1069
+ style = arrowStyleFromBody(arrowBody)
1070
+ hasArrowEnd = endMarker !== undefined
1071
+ endMarkerKind = markerKind(endMarker)
1072
+ } else {
1073
+ // Fallback: text-embedded label syntax (-- Yes -->, -. Maybe .->, == Sure ==>)
1074
+ const textMatch = remaining.match(TEXT_ARROW_REGEX)
1075
+ if (!textMatch) break
1076
+ hasArrowStart = Boolean(textMatch[1])
1077
+ startMarkerKind = markerKind(textMatch[1])
1078
+ const rawLabel = textMatch[3]!.trim()
1079
+ edgeLabel = rawLabel ? normalizeBrTags(rawLabel) : undefined
1080
+ const openOp = textMatch[2]!
1081
+ const closeOp = textMatch[4]!
1082
+ remaining = remaining.slice(textMatch[0].length).trim()
1083
+ style = textArrowStyleFromOps(openOp, closeOp)
1084
+ // closeOp is a variable-length run (e.g. "---o", "==x", "-.-.->"); only
1085
+ // its final character can be a circle/cross marker. A circle/cross
1086
+ // terminator is its own kind of arrow ending — like `>` — so it counts
1087
+ // toward hasArrowEnd too; an unmarked run (`---`, `===`, `-.-`) is the
1088
+ // only case with no arrowhead of any kind at the close.
1089
+ endMarkerKind = markerKind(closeOp.slice(-1))
1090
+ hasArrowEnd = closeOp.endsWith('>') || endMarkerKind !== undefined
1091
+ }
1092
+
1093
+ // Parse the next node group
1094
+ const nextGroup = consumeNodeGroup(remaining, graph, subgraphStack)
1095
+ if (!nextGroup || nextGroup.ids.length === 0) break
1096
+
1097
+ remaining = nextGroup.remaining.trim()
1098
+
1099
+ // Emit Cartesian product of edges: every source × every target
1100
+ for (const sourceId of prevGroupIds) {
1101
+ for (const targetId of nextGroup.ids) {
1102
+ graph.edges.push({
1103
+ source: sourceId,
1104
+ target: targetId,
1105
+ label: edgeLabel,
1106
+ style,
1107
+ hasArrowStart,
1108
+ hasArrowEnd,
1109
+ ...(startMarkerKind !== undefined
1110
+ ? { startMarker: startMarkerKind }
1111
+ : {}),
1112
+ ...(endMarkerKind !== undefined ? { endMarker: endMarkerKind } : {}),
1113
+ ...(edgeId !== undefined ? { id: edgeId } : {}),
1114
+ })
1115
+ }
1116
+ }
1117
+
1118
+ prevGroupIds = nextGroup.ids
1119
+ }
1120
+ }
1121
+
1122
+ interface ConsumedNodeGroup {
1123
+ ids: string[]
1124
+ remaining: string
1125
+ }
1126
+
1127
+ /**
1128
+ * Consume one or more nodes separated by `&`.
1129
+ * E.g. "A & B & C --> ..." returns ids: ['A', 'B', 'C']
1130
+ */
1131
+ function consumeNodeGroup(
1132
+ text: string,
1133
+ graph: MermaidGraph,
1134
+ subgraphStack: MermaidSubgraph[],
1135
+ ): ConsumedNodeGroup | null {
1136
+ const first = consumeNode(text, graph, subgraphStack)
1137
+ if (!first) return null
1138
+
1139
+ const ids = [first.id]
1140
+ let remaining = first.remaining.trim()
1141
+
1142
+ // Check for & separators
1143
+ while (remaining.startsWith('&')) {
1144
+ remaining = remaining.slice(1).trim()
1145
+ const next = consumeNode(remaining, graph, subgraphStack)
1146
+ if (!next) break
1147
+ ids.push(next.id)
1148
+ remaining = next.remaining.trim()
1149
+ }
1150
+
1151
+ return { ids, remaining }
1152
+ }
1153
+
1154
+ interface ConsumedNode {
1155
+ id: string
1156
+ remaining: string
1157
+ }
1158
+
1159
+ /**
1160
+ * Try to consume a node definition from the start of `text`.
1161
+ * If the node has a shape+label (e.g. A[Text]), it's registered in the graph.
1162
+ * If it's a bare reference (e.g. A), we look it up or create a default.
1163
+ * Also handles ::: class shorthand suffix.
1164
+ */
1165
+ function consumeNode(
1166
+ text: string,
1167
+ graph: MermaidGraph,
1168
+ subgraphStack: MermaidSubgraph[],
1169
+ ): ConsumedNode | null {
1170
+ let id: string | null = null
1171
+ let remaining: string = text
1172
+
1173
+ // Check for ::: class shorthand appearing BEFORE the shape brackets
1174
+ // (e.g. A:::external[Label]). Strip it out so shape-pattern matching
1175
+ // below still sees the id directly adjacent to its brackets.
1176
+ let preClassName: string | undefined
1177
+ const preClassMatch = remaining.match(PRE_CLASS_SHORTHAND_REGEX)
1178
+ if (preClassMatch) {
1179
+ preClassName = preClassMatch[2]!
1180
+ remaining = preClassMatch[1]! + remaining.slice(preClassMatch[0].length)
1181
+ }
1182
+
1183
+ /*
1184
+ * Expanded syntax: `A@{ shape: doc, label: "Report" }` (Mermaid v11.3.0+).
1185
+ *
1186
+ * Tried before the bracket patterns because the id is followed by `@{`,
1187
+ * which none of them match — without this the id would fall through to
1188
+ * BARE_NODE_REGEX and the whole metadata block would be stranded as
1189
+ * unparsed text.
1190
+ */
1191
+ const expandedIdMatch = remaining.match(EXPANDED_NODE_ID_REGEX)
1192
+ if (expandedIdMatch) {
1193
+ const expandedId = expandedIdMatch[1]!
1194
+ const block = matchExpandedBlock(remaining.slice(expandedId.length))
1195
+ if (block) {
1196
+ const meta = parseExpandedMeta(block.body)
1197
+ registerNode(graph, subgraphStack, {
1198
+ id: expandedId,
1199
+ label: normalizeBrTags(expandedNodeLabel(expandedId, meta)),
1200
+ shape: expandedNodeShape(meta),
1201
+ })
1202
+ id = expandedId
1203
+ remaining = remaining.slice(expandedId.length + block.length)
1204
+ }
1205
+ }
1206
+
1207
+ // Try each node pattern (shape-qualified)
1208
+ if (id === null) {
1209
+ for (const { regex, shape } of NODE_PATTERNS) {
1210
+ const match = remaining.match(regex)
1211
+ if (match) {
1212
+ id = match[1]!
1213
+ const label = normalizeBrTags(match[2]!)
1214
+ // The slash-bracket family resolves its shape from the closing
1215
+ // delimiter it captured in group 3; every other pattern has a fixed
1216
+ // shape. See the SLASH_BRACKET note in NODE_PATTERNS.
1217
+ const resolvedShape =
1218
+ typeof shape === 'function' ? shape(match[3] ?? '') : shape
1219
+ registerNode(graph, subgraphStack, { id, label, shape: resolvedShape })
1220
+ remaining = remaining.slice(match[0].length)
1221
+ break
1222
+ }
1223
+ }
1224
+ }
1225
+
1226
+ // Bare node reference — register it if the node doesn't exist yet.
1227
+ // If it already exists, it stays in the subgraph that claimed it first
1228
+ // (nodes belong to the subgraph where they're first defined). A node that
1229
+ // no subgraph has claimed yet (declared earlier at the top level, e.g.
1230
+ // `X --> A` before `subgraph SA; A; end`) is adopted by the current one,
1231
+ // as Mermaid does; otherwise the frame would come out empty (#1289).
1232
+ if (id === null) {
1233
+ const bareMatch = remaining.match(BARE_NODE_REGEX)
1234
+ if (bareMatch) {
1235
+ id = bareMatch[1]!
1236
+ if (!graph.nodes.has(id)) {
1237
+ registerNode(graph, subgraphStack, {
1238
+ id,
1239
+ label: id,
1240
+ shape: 'rectangle',
1241
+ })
1242
+ } else if (!isClaimedBySubgraph(graph, subgraphStack, id)) {
1243
+ trackInSubgraph(subgraphStack, id)
1244
+ }
1245
+ remaining = remaining.slice(bareMatch[0].length)
1246
+ }
1247
+ }
1248
+
1249
+ if (id === null) return null
1250
+
1251
+ if (preClassName) {
1252
+ graph.classAssignments.set(id, preClassName)
1253
+ }
1254
+
1255
+ // Check for ::: class shorthand suffix immediately after the node
1256
+ const classMatch = remaining.match(CLASS_SHORTHAND_REGEX)
1257
+ if (classMatch) {
1258
+ graph.classAssignments.set(id, classMatch[1]!)
1259
+ remaining = remaining.slice(classMatch[0].length)
1260
+ }
1261
+
1262
+ return { id, remaining }
1263
+ }
1264
+
1265
+ /** Register a node in the graph and track it in the current subgraph */
1266
+ function registerNode(
1267
+ graph: MermaidGraph,
1268
+ subgraphStack: MermaidSubgraph[],
1269
+ node: MermaidNode,
1270
+ ): void {
1271
+ const isNew = !graph.nodes.has(node.id)
1272
+ if (isNew) {
1273
+ graph.nodes.set(node.id, node)
1274
+ }
1275
+ trackInSubgraph(subgraphStack, node.id)
1276
+ }
1277
+
1278
+ /** True when any finished or still-open subgraph already lists `nodeId`. */
1279
+ function isClaimedBySubgraph(
1280
+ graph: MermaidGraph,
1281
+ subgraphStack: MermaidSubgraph[],
1282
+ nodeId: string,
1283
+ ): boolean {
1284
+ const claims = (sg: MermaidSubgraph): boolean =>
1285
+ sg.nodeIds.includes(nodeId) || sg.children.some(claims)
1286
+ return graph.subgraphs.some(claims) || subgraphStack.some(claims)
1287
+ }
1288
+
1289
+ /** Add node ID to the innermost subgraph if we're inside one */
1290
+ function trackInSubgraph(
1291
+ subgraphStack: MermaidSubgraph[],
1292
+ nodeId: string,
1293
+ ): void {
1294
+ if (subgraphStack.length > 0) {
1295
+ const current = subgraphStack[subgraphStack.length - 1]!
1296
+ if (!current.nodeIds.includes(nodeId)) {
1297
+ current.nodeIds.push(nodeId)
1298
+ }
1299
+ }
1300
+ }
1301
+
1302
+ /**
1303
+ * Map a link body to its edge style, ignoring direction and run length.
1304
+ *
1305
+ * The body is the run between the optional start/end markers: `--`, `----`,
1306
+ * `==`, `-.-`, `-..-`, `~~~`, etc. Classification is by the characters used,
1307
+ * not the length, so every run length of a given style behaves identically.
1308
+ */
1309
+ function arrowStyleFromBody(body: string): EdgeStyle {
1310
+ if (body.startsWith('~')) return 'invisible'
1311
+ if (body.includes('.')) return 'dotted'
1312
+ if (body.startsWith('=')) return 'thick'
1313
+ return 'solid'
1314
+ }
1315
+
1316
+ /** Map text-embedded arrow open/close operators to edge style */
1317
+ function textArrowStyleFromOps(openOp: string, closeOp: string): EdgeStyle {
1318
+ // Classify by the characters used, not by exact token, so every run length
1319
+ // of a given style resolves identically — the same rule arrowStyleFromBody
1320
+ // applies to the plain arrow forms.
1321
+ if (openOp.includes('.') || closeOp.includes('.')) return 'dotted'
1322
+ if (openOp.startsWith('=') || closeOp.startsWith('=')) return 'thick'
1323
+ return 'solid'
1324
+ }