ntk 8.13.1 → 8.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,81 @@
1
+ // fontkit's mark attachment, for a face whose tables leave an anchor out.
2
+ //
3
+ // A mark-to-base, mark-to-ligature or mark-to-mark subtable holds an anchor
4
+ // for each base (ligature component, earlier mark) and mark class, and the
5
+ // OpenType spec lets one be NULL: that subtable attaches no mark of that
6
+ // class to that glyph. HarfBuzz answers "not applied", and the lookup's next
7
+ // subtable gets its turn. fontkit reads the NULL's coordinates and throws
8
+ // `Cannot read properties of null (reading 'xCoordinate')` out of a layout,
9
+ // and faces ship them: Noto Sans Bold holds 2,822, DejaVu Sans Mono 198 (a
10
+ // Lithuanian Į̃ reaches one), Amiri, Noto Naskh Arabic and FreeSerif
11
+ // thousands. Upstream that is foliojs/fontkit#367; #374 returns early
12
+ // instead, which ends the crash but still counts the subtable as applied.
13
+ //
14
+ // Answering "not applied" means wrapping the processor's `applyLookup`,
15
+ // which runs for every glyph at every lookup: 2% of shaping a word the
16
+ // first time. So a face pays it only from the first NULL its text reaches.
17
+ // Until then `applyAnchor` alone is watched — it runs when a mark attaches —
18
+ // and a NULL there abandons the layout (`NO_ANCHOR`), which `Font#_layout`
19
+ // shapes again with the wrapper in. Abandoning loses nothing: fontkit keeps
20
+ // no state from a layout but the tables it decoded, which a second one
21
+ // reads the same.
22
+
23
+ /**
24
+ * Thrown through fontkit's layout from a NULL anchor, and caught here or in
25
+ * `Font#_layout`. One error, made once: a face can reach NULLs many times a
26
+ * word, and a fresh error would take a stack trace each time. It says what
27
+ * happened to anything that calls a watched face's `layout` directly.
28
+ */
29
+ export const NO_ANCHOR = new Error(
30
+ 'a mark attachment subtable has a NULL anchor for this glyph and mark class: ' +
31
+ "shaped through ntk's Font, that is a subtable that did not apply"
32
+ );
33
+
34
+ /** The face's GPOS processor, where fontkit shapes it through GPOS; else null. */
35
+ function processorOf(fk) {
36
+ try {
37
+ return fk._layoutEngine?.engine?.GPOSProcessor ?? null;
38
+ } catch {
39
+ return null;
40
+ }
41
+ }
42
+
43
+ /**
44
+ * Watch a face's mark attachment for a NULL anchor, which then abandons the
45
+ * layout it is reached in with `NO_ANCHOR`.
46
+ *
47
+ * @param {object} fk a fontkit font
48
+ * @returns {boolean} whether there is anything to watch: false for a face
49
+ * fontkit does not position through GPOS
50
+ */
51
+ export function watchAnchors(fk) {
52
+ const gpos = processorOf(fk);
53
+ if (!gpos || typeof gpos.applyAnchor !== 'function') return false;
54
+ if (Object.hasOwn(gpos, 'applyAnchor')) return true;
55
+ const applyAnchor = gpos.applyAnchor;
56
+ gpos.applyAnchor = function (markRecord, baseAnchor, baseGlyphIndex) {
57
+ if (baseAnchor == null || markRecord.markAnchor == null) throw NO_ANCHOR;
58
+ return applyAnchor.call(this, markRecord, baseAnchor, baseGlyphIndex);
59
+ };
60
+ return true;
61
+ }
62
+
63
+ /**
64
+ * From here on a subtable that reaches a NULL anchor is one that did not
65
+ * apply, so the lookup's next subtable is tried, as HarfBuzz tries it.
66
+ *
67
+ * @param {object} fk a fontkit font `watchAnchors` has been handed
68
+ */
69
+ export function tolerateAnchors(fk) {
70
+ const gpos = processorOf(fk);
71
+ if (!gpos || Object.hasOwn(gpos, 'applyLookup')) return;
72
+ const applyLookup = gpos.applyLookup;
73
+ gpos.applyLookup = function (lookupType, table) {
74
+ try {
75
+ return applyLookup.call(this, lookupType, table);
76
+ } catch (err) {
77
+ if (err === NO_ANCHOR) return false;
78
+ throw err;
79
+ }
80
+ };
81
+ }
package/lib/text/font.js CHANGED
@@ -1,6 +1,8 @@
1
1
  import * as fontkit from 'fontkit';
2
2
 
3
3
  import { flatten, rasterizePath } from '../rasterize.js';
4
+ import { NO_ANCHOR, tolerateAnchors, watchAnchors } from './anchors.js';
5
+ import { marksCover, standInMarks } from './marks.js';
4
6
 
5
7
  // Axis coordinates are rounded to this many decimals before anything is
6
8
  // instantiated or cached. Every distinct coordinate is a font in its own
@@ -122,6 +124,14 @@ export default class Font {
122
124
  this._upem = 0;
123
125
  this._vertical = null;
124
126
  this._space = -1;
127
+ // the face's mark lookups while they are stood in for (./marks.js), set
128
+ // at its first shaping; null once the real ones are back, or never were
129
+ // stood in for
130
+ this._marks = undefined;
131
+ this._drawable = undefined;
132
+ // whether the face's mark attachment is watched for a NULL anchor
133
+ // (./anchors.js): set at its first shaping, false once one has been met
134
+ this._anchors = false;
125
135
  }
126
136
 
127
137
  static loadSync(path, postscriptName) {
@@ -303,8 +313,29 @@ export default class Font {
303
313
  return this._space * this.scale(size);
304
314
  }
305
315
 
316
+ /**
317
+ * Whether fontkit can make a glyph of this face at all. It makes one from
318
+ * `glyf`, `CFF ` or `CFF2` outlines, or an `sbix` or `COLR`/`CPAL` colour
319
+ * glyph, and answers null for anything else — and its shaper throws on the
320
+ * null. A bitmap-only colour font (`CBDT`/`CBLC`) is that face: Noto Color
321
+ * Emoji and EmojiOne as most Linux desktops ship them, which fontconfig
322
+ * answers first for an emoji. Its cmap says it has the character and
323
+ * nothing can be shaped or drawn from it, so it covers nothing here
324
+ * (`hasGlyph`), and `FontManager` hands it out neither as a match nor as a
325
+ * fallback.
326
+ */
327
+ get drawable() {
328
+ if (this._drawable === undefined) {
329
+ const t = this.fk.directory?.tables ?? {};
330
+ this._drawable = Boolean(
331
+ t.glyf || t['CFF '] || t.CFF2 || t.sbix || (t.COLR && t.CPAL)
332
+ );
333
+ }
334
+ return this._drawable;
335
+ }
336
+
306
337
  hasGlyph(codepoint) {
307
- return this.fk.hasGlyphForCodePoint(codepoint);
338
+ return this.drawable && this.fk.hasGlyphForCodePoint(codepoint);
308
339
  }
309
340
 
310
341
  /**
@@ -324,7 +355,7 @@ export default class Font {
324
355
  * @returns {number|null} font glyph id, as `shape()` would report in `glyphs[].id`
325
356
  */
326
357
  glyphIdFor(codepoint) {
327
- if (!this.fk.hasGlyphForCodePoint(codepoint)) return null;
358
+ if (!this.hasGlyph(codepoint)) return null;
328
359
  return this.fk.glyphForCodePoint(codepoint).id;
329
360
  }
330
361
 
@@ -339,7 +370,7 @@ export default class Font {
339
370
  * ax = advance, dx/dy = drawing offset from pen position (y up = positive dy)
340
371
  */
341
372
  shape(text, size, opts = {}) {
342
- const run = this.fk.layout(text, opts.features, opts.script, opts.language, opts.direction);
373
+ const run = this._layout(text, opts);
343
374
  const s = this.scale(size);
344
375
  const glyphs = new Array(run.glyphs.length);
345
376
  let width = 0;
@@ -358,6 +389,53 @@ export default class Font {
358
389
  return { font: this, size, direction: run.direction, width, glyphs };
359
390
  }
360
391
 
392
+ /**
393
+ * fontkit's layout, with the face's mark lookups stood in for until a run
394
+ * holds a glyph they act at.
395
+ *
396
+ * fontkit decodes a feature's lookups the first time a run asks for it,
397
+ * and a face's mark attachment can be most of its tables: 14 ms of Noto
398
+ * Sans' first shaping, in the first frame. Until a run needs them the
399
+ * lookups are empty stand-ins (./marks.js), which a run holding none of
400
+ * their glyphs comes out the same with. A run that holds one puts the
401
+ * real ones back and is shaped again, and so is every run after it.
402
+ */
403
+ _layout(text, opts) {
404
+ const { features, script, language, direction } = opts;
405
+ if (this._marks === undefined) {
406
+ this._marks = standInMarks(this.fk);
407
+ this._anchors = watchAnchors(this.fk);
408
+ }
409
+ const marks = this._marks;
410
+ // fontkit adds to the features object it is handed: a second shaping
411
+ // gets what the caller asked for, not what the first left there
412
+ const again = marks === null || features == null ? features : Array.isArray(features) ? [...features] : { ...features };
413
+ const run = this._fkLayout(text, features, script, language, direction);
414
+ if (marks === null || !run.glyphs.some((glyph) => marksCover(marks.bits, glyph.id))) return run;
415
+ marks.restore();
416
+ this._marks = null;
417
+ return this._fkLayout(text, again, script, language, direction);
418
+ }
419
+
420
+ /**
421
+ * fontkit's layout, shaped once more where a NULL anchor abandoned it: a
422
+ * face's mark attachment is watched for one until its text reaches the
423
+ * first, and takes one as a subtable that did not apply from then on
424
+ * (./anchors.js).
425
+ */
426
+ _fkLayout(text, features, script, language, direction) {
427
+ if (!this._anchors) return this.fk.layout(text, features, script, language, direction);
428
+ const again = features == null ? features : Array.isArray(features) ? [...features] : { ...features };
429
+ try {
430
+ return this.fk.layout(text, features, script, language, direction);
431
+ } catch (err) {
432
+ if (err !== NO_ANCHOR) throw err;
433
+ tolerateAnchors(this.fk);
434
+ this._anchors = false;
435
+ return this.fk.layout(text, again, script, language, direction);
436
+ }
437
+ }
438
+
361
439
  /** nominal (unshaped) advance of a glyph id, in pixels */
362
440
  advanceOf(glyphId, size) {
363
441
  return this.fk.getGlyph(glyphId).advanceWidth * this.scale(size);
@@ -111,8 +111,12 @@ function shapePrefixOf(style) {
111
111
  // plain answered every later request for it with the plain glyphs, and a
112
112
  // `tnum` asked for after that point was silently ignored.
113
113
  const shaping = shapingKeyOf(style);
114
+ // The family list too where a face is given, since the letters that face
115
+ // lacks are set in the families after it (`fallbackFor`): a word shaped
116
+ // under `Ahem, Times New Roman` answered the same word under `Ahem,
117
+ // Arial`, in Times.
114
118
  return font
115
- ? `${font.key}|${style.size}|${style.weight}|${style.style}|${shaping}`
119
+ ? `${font.key}|${style.family ?? ''}|${style.size}|${style.weight}|${style.style}|${shaping}`
116
120
  : `${style.family}|${style.size}|${style.weight}|${style.style}|${variationsKeyOf(
117
121
  style.variations
118
122
  )}|${opticalKeyOf(style)}|${shaping}`;
@@ -227,6 +231,54 @@ export default class FontManager {
227
231
  return this._source ?? defaultFontSource();
228
232
  }
229
233
 
234
+ /**
235
+ * Have `family` ready before a layout asks for it: a source that resolves
236
+ * families by asking the system (fontconfig) starts on its faces now, off
237
+ * the event loop, so the first layout in that family takes the answer
238
+ * instead of stalling for it. For a family the caller knows is coming — a
239
+ * code editor's monospace, set before it renders. A source with nothing to
240
+ * look up ignores it, and it never throws.
241
+ *
242
+ * `faces`, each `{ weight, style }`, names the faces to start on instead
243
+ * of the family's regular, bold, italic and bold italic: a face outside
244
+ * those that the caller knows text will be set in, such as a menu's
245
+ * medium. The family is spelled the way `match` spells it, so a list or a
246
+ * quoted name warms the pattern a layout will ask for.
247
+ */
248
+ prewarm(family, faces) {
249
+ const spelled = familiesOf(family).join(',');
250
+ if (!spelled) return;
251
+ this.source.prewarm?.(
252
+ spelled,
253
+ faces?.map(({ weight, style }) => ({
254
+ weight: numWeight(weight),
255
+ style: style?.includes('italic') ? 'italic' : 'normal'
256
+ }))
257
+ );
258
+ }
259
+
260
+ /**
261
+ * The best of `candidates` a glyph can be made from (`Font.drawable`):
262
+ * `emoji`, or a stylesheet's `"Noto Color Emoji"`, is a bitmap-only face
263
+ * on most Linux desktops, and a base face nothing can be shaped in threw
264
+ * on the first character of its text. The first candidate still, where
265
+ * none is — there is nothing better to hand out.
266
+ */
267
+ _firstDrawable(candidates) {
268
+ let first = null;
269
+ for (const c of candidates) {
270
+ let font;
271
+ try {
272
+ font = this._open(c);
273
+ } catch {
274
+ continue; // unparseable candidate — try the next one
275
+ }
276
+ if (font.drawable) return font;
277
+ first ??= font;
278
+ }
279
+ return first ?? this._open(candidates[0]);
280
+ }
281
+
230
282
  /** open (and cache) a match candidate — see fontsource.js for the shape */
231
283
  _open(candidate) {
232
284
  const key = candidate.key ?? `${candidate.path}#${candidate.postscriptName || ''}`;
@@ -283,7 +335,7 @@ export default class FontManager {
283
335
  let bestScore = Infinity;
284
336
  for (const family of families) {
285
337
  for (const r of this._registered) {
286
- if (r.family !== family) continue;
338
+ if (r.family !== family || !r.font.drawable) continue;
287
339
  const score = Math.abs(r.weight - weight) + (r.italic !== italic ? 1000 : 0);
288
340
  if (score < bestScore) {
289
341
  bestScore = score;
@@ -333,9 +385,21 @@ export default class FontManager {
333
385
  italic
334
386
  );
335
387
  if (!face) {
336
- // sources understand comma-separated family lists natively
337
- const candidates = this.source.matchSorted(patternOf(family, weight, italic));
338
- face = this._open(candidates[0]);
388
+ // sources understand comma-separated family lists natively; one that
389
+ // can answer the best face alone saves reading the fallback chain the
390
+ // face may never need — unless no glyph can be made from that face
391
+ const pattern = patternOf(family, weight, italic);
392
+ const source = this.source;
393
+ let head = null;
394
+ if (typeof source.matchFirst === 'function') {
395
+ const candidate = source.matchFirst(pattern);
396
+ try {
397
+ head = this._open(candidate);
398
+ } catch {
399
+ // unparseable: the chain has the next candidate
400
+ }
401
+ }
402
+ face = head?.drawable ? head : this._firstDrawable(source.matchSorted(pattern));
339
403
  }
340
404
  this._matches.set(cacheKey, face);
341
405
  // A long-lived app can name a lot of families. Same sweep as the shaping
@@ -352,9 +416,17 @@ export default class FontManager {
352
416
 
353
417
  /**
354
418
  * Find a font that has a glyph for `codepoint`, for use when the primary
355
- * font doesn't. Registered fonts first, then the source's fallback chain
356
- * (filtered by the source's coverage data — font files are only opened to
357
- * confirm). Returns null when nothing on the system covers the codepoint.
419
+ * font doesn't. The registered faces of the families the style names
420
+ * first, in the order it names them, then any registered face, then the
421
+ * source's fallback chain (filtered by the source's coverage data — font
422
+ * files are only opened to confirm). Returns null when nothing on the
423
+ * system covers the codepoint.
424
+ *
425
+ * The families first because that is what a list of them is for: CSS's
426
+ * `font-family: Icons, Arial` sets a letter the icon face lacks in Arial,
427
+ * as a browser does. Walking every registered face instead gave it to
428
+ * whichever was registered first — an app's serif — however far down the
429
+ * list, or off it, that face was.
358
430
  *
359
431
  * "Nothing covers it" includes "this environment has no system fonts at
360
432
  * all". An app that loaded its own faces still reaches here for the first
@@ -373,10 +445,23 @@ export default class FontManager {
373
445
  if (perCp.has(codepoint)) return perCp.get(codepoint);
374
446
 
375
447
  let found = null;
376
- for (const r of this._registered) {
377
- if (r.font.hasGlyph(codepoint)) {
378
- found = r.font;
379
- break;
448
+ if (this._registered.length) {
449
+ const weight = numWeight(opts.weight);
450
+ const italic = !!opts.style?.includes('italic');
451
+ for (const name of familiesOf(family)) {
452
+ const font = this._matchRegistered([name.toLowerCase()], weight, italic);
453
+ if (font && font.hasGlyph(codepoint)) {
454
+ found = font;
455
+ break;
456
+ }
457
+ }
458
+ }
459
+ if (!found) {
460
+ for (const r of this._registered) {
461
+ if (r.font.hasGlyph(codepoint)) {
462
+ found = r.font;
463
+ break;
464
+ }
380
465
  }
381
466
  }
382
467
  if (!found) {
@@ -48,10 +48,12 @@
48
48
  import { builtin } from '../builtin.js';
49
49
  import {
50
50
  charsetHas,
51
+ matchFirstSync,
51
52
  matchSorted,
52
53
  matchSortedSync,
53
54
  noFontsError,
54
55
  prewarmFaces,
56
+ prewarmPatterns,
55
57
  supported
56
58
  } from '../fontconfig.js';
57
59
  import Font from './font.js';
@@ -129,6 +131,29 @@ export class FontconfigFontSource {
129
131
  return matchSortedSync(pattern);
130
132
  }
131
133
 
134
+ /**
135
+ * `matchSorted(pattern)[0]`, reading only the head of a prewarm's answer
136
+ * where one is waiting (`matchFirstSync` in fontconfig.js): a face's
137
+ * match needs its first candidate, and the fallback chain behind it is
138
+ * read when a character first falls back.
139
+ */
140
+ matchFirst(pattern) {
141
+ return matchFirstSync(pattern);
142
+ }
143
+
144
+ /**
145
+ * The family's faces — regular, bold, and both in italic, or the `faces`
146
+ * named, each `{ weight, style }` — matched off the event loop, as the
147
+ * constructor does for sans-serif, and answerable by a layout that asks
148
+ * before the loop runs again (see `prewarm` in fontconfig.js). A pattern
149
+ * already cached or running is not asked again; never rejects and never
150
+ * reports.
151
+ */
152
+ prewarm(family, faces) {
153
+ if (faces) prewarmPatterns(faces.map(({ weight, style }) => ({ family, weight, style })));
154
+ else prewarmFaces({ family });
155
+ }
156
+
132
157
  /**
133
158
  * Non-blocking sibling: the fc-match spawn runs off the event loop and
134
159
  * seeds the same cache, so a later synchronous layout for the pattern is
@@ -407,7 +407,7 @@ export class TextLayout {
407
407
  const shaped = this.fonts._shapeCached(fragText, shaping, fragLevels);
408
408
  // the levels ride along so a later split can re-shape a piece at the
409
409
  // level it actually has, rather than assuming ltr
410
- fragments.push({ text: fragText, span, shaped, start: pos, levels: fragLevels });
410
+ fragments.push({ text: fragText, span, shaping, shaped, start: pos, levels: fragLevels });
411
411
  width += shaped.width;
412
412
  }
413
413
  pos = fragEnd;
@@ -497,33 +497,47 @@ export class TextLayout {
497
497
  used += frag.shaped.width;
498
498
  continue;
499
499
  }
500
- // binary search the longest grapheme prefix of this fragment that fits.
500
+ // shaped at the bidi level this text actually has: assuming level 0
501
+ // here re-shapes an rtl word as ltr, which lays its glyphs out
502
+ // backwards and hands reorderRuns an even level that stops it from
503
+ // being reordered at all
504
+ const shaping = frag.shaping ?? frag.span;
505
+ const shape = (text, from, to) =>
506
+ this.fonts._shapeCached(text, shaping, sliceLevels(frag.levels, from, to));
507
+ let best = null;
501
508
  // Graphemes rather than code points: cutting between a base character
502
509
  // and its combining mark, or inside an emoji ZWJ sequence, leaves a
503
510
  // dotted circle or a pair of half-emoji on the two sides of the break.
504
- const cps = graphemes(frag.text);
505
- let lo = 0;
506
- let hi = cps.length - 1;
507
- let best = null;
508
- while (lo <= hi) {
509
- const mid = (lo + hi) >> 1;
510
- const prefix = cps.slice(0, mid + 1).join('');
511
- // shaped at the bidi level this text actually has: assuming level 0
512
- // here re-shapes an rtl word as ltr, which lays its glyphs out
513
- // backwards and hands reorderRuns an even level that stops it from
514
- // being reordered at all
515
- const shaped = this.fonts._shapeCached(prefix, frag.span, sliceLevels(frag.levels, 0, prefix.length));
516
- if (used + shaped.width <= maxWidth) {
517
- best = { len: prefix.length, shaped, text: prefix };
518
- lo = mid + 1;
519
- } else {
520
- hi = mid - 1;
511
+ const lead = firstGrapheme(frag.text);
512
+ if (used + shape(lead, 0, lead.length).width <= maxWidth) {
513
+ // binary search the longest grapheme prefix of this fragment that fits
514
+ const cps = graphemes(frag.text);
515
+ let lo = 0;
516
+ let hi = cps.length - 1;
517
+ while (lo <= hi) {
518
+ const mid = (lo + hi) >> 1;
519
+ const prefix = cps.slice(0, mid + 1).join('');
520
+ const shaped = shape(prefix, 0, prefix.length);
521
+ if (used + shaped.width <= maxWidth) {
522
+ best = { len: prefix.length, shaped, text: prefix };
523
+ lo = mid + 1;
524
+ } else {
525
+ hi = mid - 1;
526
+ }
521
527
  }
528
+ } else if (!headFrags.length) {
529
+ // Not even its first cluster fits, so nothing of the token does: the
530
+ // caller lets it overflow whole. That is every word at width 0, where
531
+ // a layout is asked for the narrowest it can be, and the search found
532
+ // it out by segmenting the whole word and shaping a dozen prefixes of
533
+ // it, each a word the memo had never seen.
534
+ return [null, token];
522
535
  }
523
536
  if (best) {
524
537
  headFrags.push({
525
538
  text: best.text,
526
539
  span: frag.span,
540
+ shaping,
527
541
  shaped: best.shaped,
528
542
  start: frag.start,
529
543
  levels: sliceLevels(frag.levels, 0, best.len)
@@ -538,7 +552,8 @@ export class TextLayout {
538
552
  restFrags.push({
539
553
  text: restText,
540
554
  span: frag.span,
541
- shaped: this.fonts._shapeCached(restText, frag.span, restLevels),
555
+ shaping,
556
+ shaped: this.fonts._shapeCached(restText, shaping, restLevels),
542
557
  start: frag.start + cut,
543
558
  levels: restLevels
544
559
  });
@@ -958,19 +973,48 @@ export class TextLayout {
958
973
  // somehow absent, code points are the old behaviour and still safe for the
959
974
  // scripts that reach a force-break most often.
960
975
  let segmenter;
961
- function graphemes(text) {
976
+ function graphemeSegmenter() {
962
977
  if (segmenter === undefined) {
963
978
  segmenter =
964
979
  typeof Intl !== 'undefined' && Intl.Segmenter
965
980
  ? new Intl.Segmenter(undefined, { granularity: 'grapheme' })
966
981
  : null;
967
982
  }
968
- if (!segmenter) return Array.from(text);
983
+ return segmenter;
984
+ }
985
+
986
+ // Two ASCII characters are never one cluster, bar a CR before a LF (UAX#29
987
+ // GB3): whatever extends a cluster or joins into one — marks, joiners,
988
+ // spacing marks, prepends — sits above U+02FF. So text of nothing else,
989
+ // which is most of what a document cuts, needs no segmenter: it costs a
990
+ // microsecond a call, and the width floors ask once a word.
991
+ function plainAscii(text) {
992
+ for (let i = 0; i < text.length; i++) {
993
+ const c = text.charCodeAt(i);
994
+ if (c >= 0x80 || (c === 13 && text.charCodeAt(i + 1) === 10)) return false;
995
+ }
996
+ return true;
997
+ }
998
+
999
+ function graphemes(text) {
1000
+ if (plainAscii(text)) return text.split('');
1001
+ const segments = graphemeSegmenter();
1002
+ if (!segments) return Array.from(text);
969
1003
  const out = [];
970
- for (const { segment } of segmenter.segment(text)) out.push(segment);
1004
+ for (const { segment } of segments.segment(text)) out.push(segment);
971
1005
  return out;
972
1006
  }
973
1007
 
1008
+ /** The first grapheme cluster of `text`, without segmenting the rest. */
1009
+ function firstGrapheme(text) {
1010
+ const c0 = text.charCodeAt(0);
1011
+ if (c0 < 0x80 && c0 !== 13 && !(text.charCodeAt(1) >= 0x80)) return text.slice(0, 1);
1012
+ const segments = graphemeSegmenter();
1013
+ if (!segments) return String.fromCodePoint(text.codePointAt(0));
1014
+ for (const { segment } of segments.segment(text)) return segment;
1015
+ return '';
1016
+ }
1017
+
974
1018
  // UTF-16 length of a glyph cluster's codePoints array
975
1019
  function cuLength(codePoints) {
976
1020
  let len = 0;