@zombie-mermaid/core 2.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,520 @@
1
+ // ============================================================================
2
+ // Text Metrics — Variable-width character measurement for SVG layout
3
+ // ============================================================================
4
+ //
5
+ // Provides font-agnostic text width estimation using character class buckets.
6
+ // More accurate than uniform character width for proportional fonts.
7
+ //
8
+ // Width ratios are normalized where 1.0 = average lowercase letter.
9
+ // Final pixel width = sum(charWidths) * fontSize * baseRatio
10
+ // ============================================================================
11
+
12
+ /**
13
+ * Exact advance widths for every printable ASCII character (U+0020–U+007E),
14
+ * measured from the real Inter font (canvas measureText, headless Chrome,
15
+ * weight 400) and normalized so that ratio × fontSize × baseRatio reproduces
16
+ * the measured pixel advance. Completeness over this range is asserted by
17
+ * the `getCharWidth` "table completeness" test in text-metrics.test.ts.
18
+ *
19
+ * The coarse character-class buckets below remain as fallback for
20
+ * non-ASCII characters not in this table. The buckets systematically
21
+ * underestimated Inter widths (spaces, m/w, digits, punctuation), and the
22
+ * error grew with text length — wide edge labels overflowed their
23
+ * 8px-padded background rects.
24
+ */
25
+ const INTER_ADVANCES: Record<string, number> = {
26
+ ' ': 0.521,
27
+ '!': 0.533,
28
+ '"': 0.863,
29
+ '#': 1.173,
30
+ $: 1.188,
31
+ '%': 1.818,
32
+ '&': 1.193,
33
+ "'": 0.555,
34
+ '(': 0.675,
35
+ ')': 0.675,
36
+ '*': 0.928,
37
+ '+': 1.225,
38
+ ',': 0.534,
39
+ '-': 0.852,
40
+ '.': 0.534,
41
+ '/': 0.667,
42
+ '0': 1.168,
43
+ '1': 0.753,
44
+ '2': 1.129,
45
+ '3': 1.144,
46
+ '4': 1.196,
47
+ '5': 1.099,
48
+ '6': 1.148,
49
+ '7': 1.048,
50
+ '8': 1.146,
51
+ '9': 1.148,
52
+ ':': 0.534,
53
+ ';': 0.559,
54
+ '<': 1.225,
55
+ '=': 1.225,
56
+ '>': 1.225,
57
+ '?': 0.947,
58
+ '@': 1.789,
59
+ A: 1.278,
60
+ B: 1.212,
61
+ C: 1.353,
62
+ D: 1.336,
63
+ E: 1.113,
64
+ F: 1.093,
65
+ G: 1.382,
66
+ H: 1.376,
67
+ I: 0.497,
68
+ J: 1.057,
69
+ K: 1.244,
70
+ L: 1.047,
71
+ M: 1.673,
72
+ N: 1.395,
73
+ O: 1.416,
74
+ P: 1.183,
75
+ Q: 1.416,
76
+ R: 1.192,
77
+ S: 1.188,
78
+ T: 1.195,
79
+ U: 1.378,
80
+ V: 1.278,
81
+ W: 1.825,
82
+ X: 1.263,
83
+ Y: 1.257,
84
+ Z: 1.165,
85
+ '[': 0.675,
86
+ '\\': 0.667,
87
+ ']': 0.675,
88
+ '^': 0.873,
89
+ _: 0.845,
90
+ '`': 0.598,
91
+ a: 1.04,
92
+ b: 1.134,
93
+ c: 1.058,
94
+ d: 1.134,
95
+ e: 1.08,
96
+ f: 0.685,
97
+ g: 1.136,
98
+ h: 1.095,
99
+ i: 0.448,
100
+ j: 0.448,
101
+ k: 1.016,
102
+ l: 0.448,
103
+ m: 1.622,
104
+ n: 1.094,
105
+ o: 1.11,
106
+ p: 1.134,
107
+ q: 1.134,
108
+ r: 0.697,
109
+ s: 0.977,
110
+ t: 0.606,
111
+ u: 1.095,
112
+ v: 1.041,
113
+ w: 1.515,
114
+ x: 1.011,
115
+ y: 1.041,
116
+ z: 1.023,
117
+ '{': 0.789,
118
+ '|': 0.616,
119
+ '}': 0.789,
120
+ '~': 1.225,
121
+ }
122
+
123
+ /**
124
+ * Whether `char` has an exact measured Inter advance in `INTER_ADVANCES`,
125
+ * as opposed to falling through to a coarse character-class bucket or the
126
+ * default width in `getCharWidth`. Exported only so tests can assert the
127
+ * table covers every printable ASCII character (U+0020–U+007E) — a gap
128
+ * here means a real character silently degrades to an approximate width.
129
+ */
130
+ export function hasMeasuredAdvance(char: string): boolean {
131
+ return Object.hasOwn(INTER_ADVANCES, char)
132
+ }
133
+
134
+ /**
135
+ * Narrow characters - visually thin glyphs.
136
+ * Note: '1' is included because in proportional fonts (like Inter), it's
137
+ * significantly narrower than other digits which use tabular/uniform width.
138
+ */
139
+ const NARROW_CHARS = new Set([
140
+ 'i',
141
+ 'l',
142
+ 't',
143
+ 'f',
144
+ 'j',
145
+ 'I',
146
+ '1',
147
+ '!',
148
+ '|',
149
+ '.',
150
+ ',',
151
+ ':',
152
+ ';',
153
+ "'",
154
+ ])
155
+
156
+ /**
157
+ * Wide characters - visually wide glyphs
158
+ */
159
+ const WIDE_CHARS = new Set(['W', 'M', 'w', 'm', '@', '%'])
160
+
161
+ /**
162
+ * Very wide characters - widest Latin glyphs
163
+ */
164
+ const VERY_WIDE_CHARS = new Set(['W', 'M'])
165
+
166
+ /**
167
+ * Semi-narrow punctuation - brackets and slashes are narrower than letters
168
+ * but wider than narrow chars like dots/commas
169
+ */
170
+ const SEMI_NARROW_PUNCT = new Set([
171
+ '(',
172
+ ')',
173
+ '[',
174
+ ']',
175
+ '{',
176
+ '}',
177
+ '/',
178
+ '\\',
179
+ '-',
180
+ '"',
181
+ '`',
182
+ ])
183
+
184
+ /**
185
+ * Check if a code point is a combining diacritical mark (zero-width overlay)
186
+ */
187
+ function isCombiningMark(code: number): boolean {
188
+ // Combining Diacritical Marks: U+0300–U+036F
189
+ // Combining Diacritical Marks Extended: U+1AB0–U+1AFF
190
+ // Combining Diacritical Marks Supplement: U+1DC0–U+1DFF
191
+ // Combining Diacritical Marks for Symbols: U+20D0–U+20FF
192
+ // Combining Half Marks: U+FE20–U+FE2F
193
+ return (
194
+ (code >= 0x0300 && code <= 0x036f) ||
195
+ (code >= 0x1ab0 && code <= 0x1aff) ||
196
+ (code >= 0x1dc0 && code <= 0x1dff) ||
197
+ (code >= 0x20d0 && code <= 0x20ff) ||
198
+ (code >= 0xfe20 && code <= 0xfe2f)
199
+ )
200
+ }
201
+
202
+ /**
203
+ * Check if a code point is fullwidth (CJK, emoji, etc.)
204
+ * These characters occupy approximately 2x the width of Latin letters.
205
+ */
206
+ function isFullwidth(code: number): boolean {
207
+ // CJK Radicals Supplement: U+2E80–U+2EFF
208
+ // Kangxi Radicals: U+2F00–U+2FDF
209
+ // CJK Symbols and Punctuation: U+3000–U+303F
210
+ // Hiragana: U+3040–U+309F
211
+ // Katakana: U+30A0–U+30FF
212
+ // Bopomofo: U+3100–U+312F
213
+ // Hangul Compatibility Jamo: U+3130–U+318F
214
+ // Kanbun: U+3190–U+319F
215
+ // Bopomofo Extended: U+31A0–U+31BF
216
+ // CJK Strokes: U+31C0–U+31EF
217
+ // Katakana Phonetic Extensions: U+31F0–U+31FF
218
+ // Enclosed CJK Letters and Months: U+3200–U+32FF
219
+ // CJK Compatibility: U+3300–U+33FF
220
+ // CJK Unified Ideographs Extension A: U+3400–U+4DBF
221
+ // CJK Unified Ideographs: U+4E00–U+9FFF
222
+ // Hangul Syllables: U+AC00–U+D7AF
223
+ // CJK Compatibility Ideographs: U+F900–U+FAFF
224
+ // Halfwidth and Fullwidth Forms (fullwidth part): U+FF00–U+FF60, U+FFE0–U+FFE6
225
+ // CJK Unified Ideographs Extension B+: U+20000–U+2A6DF (and beyond)
226
+
227
+ return (
228
+ (code >= 0x1100 && code <= 0x115f) || // Hangul Jamo
229
+ (code >= 0x2e80 && code <= 0x2eff) || // CJK Radicals Supplement
230
+ (code >= 0x2f00 && code <= 0x2fdf) || // Kangxi Radicals
231
+ (code >= 0x3000 && code <= 0x303f) || // CJK Symbols and Punctuation
232
+ (code >= 0x3040 && code <= 0x309f) || // Hiragana
233
+ (code >= 0x30a0 && code <= 0x30ff) || // Katakana
234
+ (code >= 0x3100 && code <= 0x312f) || // Bopomofo
235
+ (code >= 0x3130 && code <= 0x318f) || // Hangul Compatibility Jamo
236
+ (code >= 0x3190 && code <= 0x31ff) || // Kanbun + extensions
237
+ (code >= 0x3200 && code <= 0x33ff) || // Enclosed CJK + Compatibility
238
+ (code >= 0x3400 && code <= 0x4dbf) || // CJK Extension A
239
+ (code >= 0x4e00 && code <= 0x9fff) || // CJK Unified Ideographs
240
+ (code >= 0xac00 && code <= 0xd7af) || // Hangul Syllables
241
+ (code >= 0xf900 && code <= 0xfaff) || // CJK Compatibility Ideographs
242
+ (code >= 0xff00 && code <= 0xff60) || // Fullwidth ASCII
243
+ (code >= 0xffe0 && code <= 0xffe6) || // Fullwidth symbols
244
+ code >= 0x20000 // CJK Extension B and beyond
245
+ )
246
+ }
247
+
248
+ /**
249
+ * Regex for emoji detection using Unicode property escapes.
250
+ * Uses Emoji_Presentation and Extended_Pictographic (not just Emoji)
251
+ * because \p{Emoji} includes digits and # which we don't want as fullwidth.
252
+ *
253
+ * \p{Extended_Pictographic} on its own is broader than "renders wide" — it
254
+ * flags characters that are merely emoji-*capable* (could take a VS16
255
+ * selector to become emoji), not characters that actually render wide by
256
+ * default. Most such characters (e.g. ❤ U+2764) are still conventionally
257
+ * treated as wide, so the OR is kept — except for the Geometric Shapes
258
+ * block (U+25A0–U+25FF, see `isGeometricShapesTextDefault` below), which
259
+ * this renderer itself uses for narrow box-drawing/arrowhead glyphs like
260
+ * ▶ U+25B6 and ◀ U+25C0 (see src/ascii/sequence.ts); those render as a
261
+ * single narrow column in every real terminal despite being flagged
262
+ * Extended_Pictographic.
263
+ */
264
+ const EMOJI_PRESENTATION_REGEX = /\p{Emoji_Presentation}/u
265
+ const EXTENDED_PICTOGRAPHIC_REGEX = /\p{Extended_Pictographic}/u
266
+
267
+ /**
268
+ * Geometric Shapes block (U+25A0–U+25FF). Characters here that also carry
269
+ * Emoji_Presentation (e.g. ◽ ◾) are unaffected by this — they're already
270
+ * caught by the Emoji_Presentation check in `isEmoji` before this applies.
271
+ */
272
+ function isGeometricShapesTextDefault(code: number): boolean {
273
+ return code >= 0x25a0 && code <= 0x25ff
274
+ }
275
+
276
+ /**
277
+ * Arrows-block glyphs that carry Extended_Pictographic (emoji-capable) but
278
+ * no default Emoji_Presentation, and render as a single narrow column in
279
+ * every real terminal — the same misclassification the Geometric Shapes
280
+ * block above hits. Covers:
281
+ *
282
+ * - ↖ U+2196, ↗ U+2197, ↘ U+2198, ↙ U+2199 — drawn by the ASCII renderer
283
+ * itself for diagonal edge-routing arrowheads
284
+ * (packages/ascii-renderer/src/draw-arrows.ts's `drawArrowHead`) in place
285
+ * of the filled triangles (◢◣◤◥) JetBrains Mono NL has no glyph for at
286
+ * all — see issue #1062.
287
+ * - ↔ U+2194, ↕ U+2195, ↩ U+21A9, ↪ U+21AA — not drawn by the renderer
288
+ * itself, but reachable if a diagram author types one into node/edge/note
289
+ * label text, which passes through to the ASCII grid verbatim.
290
+ */
291
+ function isArrowsBlockTextDefault(code: number): boolean {
292
+ return (
293
+ (code >= 0x2194 && code <= 0x2199) || code === 0x21a9 || code === 0x21aa
294
+ )
295
+ }
296
+
297
+ /**
298
+ * Check if a character is an emoji (fullwidth)
299
+ */
300
+ function isEmoji(char: string): boolean {
301
+ if (EMOJI_PRESENTATION_REGEX.test(char)) return true
302
+ const code = char.codePointAt(0)
303
+ if (code !== undefined && isGeometricShapesTextDefault(code)) return false
304
+ if (code !== undefined && isArrowsBlockTextDefault(code)) return false
305
+ return EXTENDED_PICTOGRAPHIC_REGEX.test(char)
306
+ }
307
+
308
+ /**
309
+ * Check if a single character (code point) is "wide" — i.e. renders as two
310
+ * columns in a monospace terminal / two grid cells wide instead of one.
311
+ * Covers CJK/kana/hangul/fullwidth-form characters (via `isFullwidth`) as
312
+ * well as emoji. Shared by the SVG text-measurement path (`getCharWidth`
313
+ * above) and the ASCII grid's display-width helpers
314
+ * (`src/ascii/display-width.ts`), so both agree on what counts as wide.
315
+ */
316
+ export function isWideChar(char: string): boolean {
317
+ const code = char.codePointAt(0)
318
+ if (code === undefined) return false
319
+ return isFullwidth(code) || isEmoji(char)
320
+ }
321
+
322
+ /**
323
+ * Get the relative width of a single character.
324
+ *
325
+ * Returns a normalized width ratio where:
326
+ * - 0.0 = zero-width (combining marks)
327
+ * - 0.3 = space
328
+ * - 0.4 = narrow (i, l, t, f, j, I, 1)
329
+ * - 0.8 = semi-narrow (r)
330
+ * - 1.0 = average lowercase
331
+ * - 1.2 = wide lowercase / uppercase
332
+ * - 1.5 = very wide (W, M)
333
+ * - 2.0 = fullwidth (CJK, emoji)
334
+ */
335
+ export function getCharWidth(char: string): number {
336
+ const code = char.codePointAt(0)
337
+ if (code === undefined) return 0
338
+
339
+ // Exact measured advance for printable ASCII (Inter). The own-property
340
+ // check must come first: without it, bracket access on a plain object
341
+ // literal also resolves inherited Object.prototype properties (e.g.
342
+ // `INTER_ADVANCES['toString']` is a function, not undefined).
343
+ if (Object.hasOwn(INTER_ADVANCES, char)) {
344
+ const measured = INTER_ADVANCES[char]
345
+ // The `undefined` case is unreachable: INTER_ADVANCES is a `Record<string,
346
+ // number>` object literal whose every value is a numeric literal, and
347
+ // `noUncheckedIndexedAccess` is what forces this check's type, not any
348
+ // real possibility that an own property here holds `undefined`. Given
349
+ // `Object.hasOwn` is already true, `measured` is always a defined
350
+ // number — kept (not simplified to a bare `return measured`) so the
351
+ // check still holds if a future entry's value type ever widens.
352
+ /* v8 ignore else */
353
+ if (measured !== undefined) return measured
354
+ }
355
+
356
+ // Zero-width: combining diacritical marks
357
+ if (isCombiningMark(code)) return 0
358
+
359
+ // Fullwidth: CJK, emoji
360
+ if (isFullwidth(code) || isEmoji(char)) return 2.0
361
+
362
+ // Accented Latin: use the base letter's measured advance (é → e, Ü → U)
363
+ const base = char.normalize('NFD')[0]
364
+ if (base !== undefined && base !== char) {
365
+ const baseMeasured = INTER_ADVANCES[base]
366
+ if (baseMeasured !== undefined) return baseMeasured
367
+ }
368
+
369
+ // Space
370
+ if (char === ' ') return 0.3
371
+
372
+ // Very wide Latin
373
+ if (VERY_WIDE_CHARS.has(char)) return 1.5
374
+
375
+ // Wide Latin
376
+ if (WIDE_CHARS.has(char)) return 1.2
377
+
378
+ // Narrow Latin
379
+ if (NARROW_CHARS.has(char)) return 0.4
380
+
381
+ // Semi-narrow punctuation (brackets, slashes, hyphens)
382
+ if (SEMI_NARROW_PUNCT.has(char)) return 0.5
383
+
384
+ // Semi-narrow letter
385
+ if (char === 'r') return 0.8
386
+
387
+ // Uppercase (slightly wider than lowercase on average)
388
+ if (code >= 65 && code <= 90) return 1.2
389
+
390
+ // Digits (uniform width in most fonts)
391
+ if (code >= 48 && code <= 57) return 1.0
392
+
393
+ // Default: average lowercase width
394
+ return 1.0
395
+ }
396
+
397
+ /**
398
+ * Measure the pixel width of a text string.
399
+ *
400
+ * Uses character class buckets for more accurate width estimation
401
+ * than uniform character width assumptions.
402
+ *
403
+ * @param text - The text to measure
404
+ * @param fontSize - Font size in pixels
405
+ * @param fontWeight - Font weight (affects width slightly)
406
+ * @returns Estimated width in pixels
407
+ */
408
+ /** Advance width of a monospace glyph, as a fraction of font size. */
409
+ const MONO_ADVANCE = 0.6
410
+
411
+ /**
412
+ * Whether text is measured with monospace metrics.
413
+ *
414
+ * The character buckets above model a proportional face (they are calibrated for Inter), but
415
+ * `RenderOptions.font` accepts any family. Measuring a monospace font with them sizes every box
416
+ * for the wrong glyph widths: narrow strings under-measure by ~60% and overflow, wide ones
417
+ * over-measure by ~37%.
418
+ *
419
+ * The switch lives here rather than in `styles.ts` because this is the single choke point every
420
+ * diagram type measures through — `estimateTextWidth` and `measureMultilineText` both land here,
421
+ * and flowchart layout only reaches text metrics via the latter.
422
+ */
423
+ let monospaceMetrics = false
424
+
425
+ /** Fonts whose glyphs all share one advance width. */
426
+ export function isMonospaceFont(font: string): boolean {
427
+ // `mono` unanchored so it also catches `ui-monospace`; `code` bounded so it does not fire on
428
+ // unrelated names. "Mona Sans" is deliberately not a match.
429
+ return /mono|consol|menlo|courier|\bcode\b/i.test(font)
430
+ }
431
+
432
+ /** Select the metrics model for subsequent measurements. Called once per render. */
433
+ export function setMonospaceMetrics(monospace: boolean): void {
434
+ monospaceMetrics = monospace
435
+ }
436
+
437
+ export function measureTextWidth(
438
+ text: string,
439
+ fontSize: number,
440
+ fontWeight: number,
441
+ ): number {
442
+ // Add minimum padding to prevent truncation at text boundaries
443
+ // Increased from 0.1 to 0.15 for better label separation and collision prevention
444
+ const minPadding = fontSize * 0.15
445
+
446
+ if (monospaceMetrics) {
447
+ let count = 0
448
+ // Count code points, so surrogate pairs stay one cell wide.
449
+ for (const _ of text) count += 1
450
+ return count * fontSize * MONO_ADVANCE + minPadding
451
+ }
452
+
453
+ // Base ratio calibrated for Inter font family
454
+ // Heavier weights are slightly wider
455
+ // Added +0.02 buffer to prevent edge truncation of characters like 's' at line ends
456
+ const baseRatio = fontWeight >= 600 ? 0.6 : fontWeight >= 500 ? 0.57 : 0.54
457
+
458
+ let totalWidth = 0
459
+
460
+ // Iterate over code points (handles surrogate pairs for emoji/CJK)
461
+ for (const char of text) {
462
+ totalWidth += getCharWidth(char)
463
+ }
464
+
465
+ return totalWidth * fontSize * baseRatio + minPadding
466
+ }
467
+
468
+ // ============================================================================
469
+ // Multi-line Text Measurement
470
+ // ============================================================================
471
+
472
+ /** Standard line height ratio for multi-line text (1.3 = 130% of font size) */
473
+ export const LINE_HEIGHT_RATIO = 1.3
474
+
475
+ /** Metrics for multi-line text measurement */
476
+ export interface MultilineMetrics {
477
+ /** Maximum line width in pixels */
478
+ width: number
479
+ /** Total height in pixels (lines × lineHeight) */
480
+ height: number
481
+ /** Individual lines after splitting */
482
+ lines: string[]
483
+ /** Computed line height in pixels */
484
+ lineHeight: number
485
+ }
486
+
487
+ /**
488
+ * Measure multi-line text dimensions.
489
+ *
490
+ * Splits text on newlines and returns the maximum width across all lines,
491
+ * total height based on line count, and the split lines for rendering.
492
+ *
493
+ * @param text - The text to measure (may contain \n)
494
+ * @param fontSize - Font size in pixels
495
+ * @param fontWeight - Font weight (affects width slightly)
496
+ * @returns Metrics including width, height, lines array, and lineHeight
497
+ */
498
+ export function measureMultilineText(
499
+ text: string,
500
+ fontSize: number,
501
+ fontWeight: number,
502
+ ): MultilineMetrics {
503
+ const lines = text.split('\n')
504
+ const lineHeight = fontSize * LINE_HEIGHT_RATIO
505
+
506
+ // Width = max of all line widths
507
+ let maxWidth = 0
508
+ for (const line of lines) {
509
+ const plain = line.replace(/<\/?(?:b|strong|i|em|u|s|del)\s*>/gi, '')
510
+ const w = measureTextWidth(plain, fontSize, fontWeight)
511
+ if (w > maxWidth) maxWidth = w
512
+ }
513
+
514
+ return {
515
+ width: maxWidth,
516
+ height: lines.length * lineHeight,
517
+ lines,
518
+ lineHeight,
519
+ }
520
+ }