@lovrozagar/flare 0.2.1 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1603 @@
1
+ /**
2
+ * MIT License
3
+ *
4
+ * Copyright (c) 2026 shadcn
5
+ *
6
+ * Vendored from https://github.com/shadcn-ui/cn
7
+ * tag cn@0.4.0
8
+ * commit 84db83298f69a229d6f1ffc5d8c8d99ef451fb49
9
+ * Source: packages/cn/src/engine.ts
10
+ */
11
+
12
+ // cn — a build-time-compiled, single-pass, allocation-free Tailwind class
13
+ // merger. Architecture:
14
+ //
15
+ // scan: one pass finds token bounds + a fused full FNV hash (the scan
16
+ // reads every char anyway, so hashing rides along); all
17
+ // structural parsing (variants, '!', '/') is deferred to the
18
+ // memo-miss path, where the base feeds a compiled radix automaton
19
+ // that classifies while scanning — no split() arrays, no per-part
20
+ // Map hashing, no parse objects.
21
+ // context: variant prefixes intern to dense integer context ids via span
22
+ // hashing (substring materialized once per unique prefix);
23
+ // canonicalization (segment-sorted variants, important bit) runs
24
+ // once per unique raw prefix, ever.
25
+ // conflict: no-variant static claims stamp an epoch array indexed by group
26
+ // id; variant contexts and dynamic groups share an epoch-stamped
27
+ // hash set keyed (ctxId, gid). A kept token walks its compiled
28
+ // conflict adjacency row to claim the groups it overrides; a
29
+ // token is dropped iff its own (ctx, group) is already claimed.
30
+ // emit: span-based — survivors are sliced from the input exactly once;
31
+ // a no-op merge returns the input string itself.
32
+ //
33
+ // Comments are free (stripped by minifiers); code is sized deliberately.
34
+
35
+ import type {
36
+ ClassNameValue,
37
+ ClassValue,
38
+ CnFunction,
39
+ Engine,
40
+ EngineOptions,
41
+ FreshMerge,
42
+ Tables,
43
+ ValidatorImpls,
44
+ } from "./types.ts"
45
+
46
+ // JavaScriptCore (Bun, Safari) puts `line` on Error instances; V8 does not
47
+ const IS_JSC = "line" in new Error()
48
+
49
+ const EXTERNAL = -1
50
+ const DEAD = -1
51
+
52
+ const fnv = (str: string, s: number, e: number): number => {
53
+ let h = 0x811c9dc5
54
+ for (let p = s; p < e; p++) h = Math.imul(h ^ str.charCodeAt(p), 0x01000193)
55
+ return h
56
+ }
57
+
58
+ // positional span hash: O(1) regardless of length; samples head, quarter
59
+ // points, and tail. Used only by the doorkeeper, whose full-hash tag makes
60
+ // a collision cost one wasted cache insert, never a wrong result.
61
+ const spanHash = (str: string, s: number, e: number): number => {
62
+ const len = e - s
63
+ let h = Math.imul(len, 0x9e3779b1) ^ str.charCodeAt(s)
64
+ if (len > 3) {
65
+ const q = len >> 2
66
+ const m = len >> 1
67
+ h = Math.imul(
68
+ h ^
69
+ (str.charCodeAt(s + 1) << 8) ^
70
+ (str.charCodeAt(s + 2) << 16) ^
71
+ str.charCodeAt(s + q),
72
+ 0x85ebca6b
73
+ )
74
+ h = Math.imul(
75
+ h ^
76
+ (str.charCodeAt(s + m) << 8) ^
77
+ (str.charCodeAt(s + m + q) << 16) ^
78
+ str.charCodeAt(e - 3),
79
+ 0xc2b2ae35
80
+ )
81
+ h ^= (str.charCodeAt(e - 2) << 8) ^ (str.charCodeAt(e - 1) << 16)
82
+ // arbitrary values keep their digits a few chars from an end
83
+ // (`w-[123px]`, `bg-[#a1b2c3]`), between the samples above: fold five
84
+ // more chars from each end, walking inwards, so those strings stop
85
+ // colliding (one loop, two reads: the fold has to stay small enough
86
+ // for mergeCached to keep inlining into its callers)
87
+ for (let p = s + 3, q = e - 4; p < s + 8 && p < q; p++, q--)
88
+ h = Math.imul(
89
+ h ^ str.charCodeAt(p) ^ (str.charCodeAt(q) << 8),
90
+ 0x01000193
91
+ )
92
+ }
93
+ return (h ^ (h >>> 15)) | 0
94
+ }
95
+
96
+ export const createEngine = (
97
+ T: Tables,
98
+ validatorImpls?: ValidatorImpls,
99
+ options: EngineOptions = {}
100
+ ): Engine => {
101
+ const {
102
+ GROUP_COUNT,
103
+ edgeStart,
104
+ labelStart,
105
+ labelText,
106
+ edgeTarget,
107
+ nodeGroup,
108
+ nodeVlist,
109
+ vlistPat,
110
+ vlistOps,
111
+ vlistRef,
112
+ vlistGroup,
113
+ litAnchor,
114
+ litGroup,
115
+ litPool,
116
+ poolOffsets,
117
+ poolText,
118
+ adjGid,
119
+ adjStart,
120
+ adjTgt,
121
+ patGid,
122
+ patTgt,
123
+ postfixLookupGroups,
124
+ customValidatorNames,
125
+ orderSensitiveModifiers,
126
+ } = T
127
+
128
+ // ---- conflict adjacency row index ---------------------------------------
129
+ // adjRow[g] / patRow[g] point at each group's target list (or -1). Claims
130
+ // are tracked in one epoch-stamped hash set keyed (slot, gid) — covering
131
+ // static and dynamic groups alike — and a kept token walks its adjacency
132
+ // row to claim the groups it overrides.
133
+ const adjRow = new Int32Array(GROUP_COUNT).fill(-1)
134
+ for (let i = 0; i < adjGid.length; i++) adjRow[adjGid[i]] = i
135
+
136
+ // claims per kept token: itself + its adjacency row + postfix pairs;
137
+ // sizes the claim table so it can never fill under any config
138
+ let maxAdj = 0
139
+ for (let r = 0; r + 1 < adjStart.length; r++) {
140
+ const n = adjStart[r + 1] - adjStart[r]
141
+ if (n > maxAdj) maxAdj = n
142
+ }
143
+ let CLAIM_PER_TOKEN = 32
144
+ while (CLAIM_PER_TOKEN < 2 * (1 + maxAdj + patGid.length))
145
+ CLAIM_PER_TOKEN <<= 1
146
+
147
+ // per-list gid offsets (lists share op patterns; gids are flat per list)
148
+ const vgStart = new Int32Array(vlistRef.length + 1)
149
+ for (let l = 0; l < vlistRef.length; l++) {
150
+ vgStart[l + 1] =
151
+ vgStart[l] + vlistPat[vlistRef[l] + 1] - vlistPat[vlistRef[l]]
152
+ }
153
+
154
+ const postfixLookupSet = new Uint8Array(GROUP_COUNT)
155
+ for (let i = 0; i < postfixLookupGroups.length; i++)
156
+ postfixLookupSet[postfixLookupGroups[i]] = 1
157
+
158
+ // ---- literal maps (lifted trie subtrees) -------------------------------
159
+ // (anchorNode, tailString) → group as one open-addressed table; tails
160
+ // live in the compiled text pool. Probes hash the input span in place —
161
+ // a tail substring is never materialized.
162
+ const nodeCount = edgeStart.length - 1
163
+ const nodeHasLit = new Uint8Array(nodeCount)
164
+ let litMaxLen = 0
165
+ let litNoArb = true // no literal tail starts with '[' or '('
166
+ for (let i = 0; i < litAnchor.length; i++) {
167
+ nodeHasLit[litAnchor[i]] = 1
168
+ const len = poolOffsets[litPool[i] * 2 + 1]
169
+ if (len > litMaxLen) litMaxLen = len
170
+ const c0 = poolText.charCodeAt(poolOffsets[litPool[i] * 2])
171
+ if (c0 === 91 || c0 === 40) litNoArb = false
172
+ }
173
+ let LIT_SIZE = 1
174
+ while (LIT_SIZE < litAnchor.length * 2) LIT_SIZE <<= 1
175
+ const litTable = new Int32Array(LIT_SIZE).fill(-1)
176
+ for (let i = 0; i < litAnchor.length; i++) {
177
+ const off = poolOffsets[litPool[i] * 2]
178
+ let idx =
179
+ ((fnv(poolText, off, off + poolOffsets[litPool[i] * 2 + 1]) ^
180
+ Math.imul(litAnchor[i], 0x9e3779b1)) |
181
+ 0) &
182
+ (LIT_SIZE - 1)
183
+ while (litTable[idx] !== -1) idx = (idx + 1) & (LIT_SIZE - 1)
184
+ litTable[idx] = i
185
+ }
186
+ const litProbe = (
187
+ anchor: number,
188
+ input: string,
189
+ s: number,
190
+ e: number
191
+ ): number => {
192
+ let idx =
193
+ ((fnv(input, s, e) ^ Math.imul(anchor, 0x9e3779b1)) | 0) & (LIT_SIZE - 1)
194
+ const len = e - s
195
+ for (;;) {
196
+ const entry = litTable[idx]
197
+ if (entry === -1) return -1
198
+ if (
199
+ litAnchor[entry] === anchor &&
200
+ poolOffsets[litPool[entry] * 2 + 1] === len
201
+ ) {
202
+ const off = poolOffsets[litPool[entry] * 2]
203
+ let ok = true
204
+ for (let k = 0; k < len; k++) {
205
+ if (poolText.charCodeAt(off + k) !== input.charCodeAt(s + k)) {
206
+ ok = false
207
+ break
208
+ }
209
+ }
210
+ if (ok) return litGroup[entry]
211
+ }
212
+ idx = (idx + 1) & (LIT_SIZE - 1)
213
+ }
214
+ }
215
+
216
+ const cacheSize = options.cacheSize ?? 8192
217
+
218
+ // Tailwind v4 prefix (written like a leading variant: `tw:hover:p-4`).
219
+ // Tokens not starting with `${prefix}:` pass through as external.
220
+ const RAW_PREFIX = options.prefix ?? T.prefix ?? ""
221
+ const FULL_PREFIX = RAW_PREFIX === "" ? "" : RAW_PREFIX + ":"
222
+ const FPL = FULL_PREFIX.length
223
+
224
+ // ---- span validators ----------------------------------------------------
225
+ // Opcodes over (input, start, end) spans — no substring, no regex on the
226
+ // hot path; exact tailwind-merge semantics incl. the arbitrary-value
227
+ // regex's label backtracking. Rare shapes fall back to lazy slice + regex.
228
+ const vCustom = (customValidatorNames ?? []).map((name) => {
229
+ const fn = validatorImpls && validatorImpls[name]
230
+ if (!fn) throw new Error("cn: missing validator " + name)
231
+ return fn
232
+ })
233
+
234
+ const lengthUnitRegex =
235
+ /\d+(%|px|r?em|[sdl]?v([hwib]|min|max)|pt|pc|in|cm|mm|cap|ch|ex|r?lh|cq(w|h|i|b|min|max))|\b(calc|min|max|clamp)\(.+\)|^0$/
236
+ const colorFunctionRegex =
237
+ /^(rgba?|hsla?|hwb|(ok)?(lab|lch)|color-mix)\(.+\)$/
238
+ const shadowRegex =
239
+ /^(inset_)?-?((\d+)?\.?(\d+)[a-z]+|0)_-?((\d+)?\.?(\d+)[a-z]+|0)/
240
+ const imageRegex =
241
+ /^(url|image|image-set|cross-fade|element|(repeating-)?(linear|radial|conic)-gradient)\(.+\)$/
242
+
243
+ // scratch filled by analyzeArb for the current tail span
244
+ let aKind = 0 // 0 none, 1 [..], 2 (..)
245
+ let aLabelS = -1
246
+ let aLabelE = -1
247
+ let aValS = -1
248
+ let aValE = -1
249
+
250
+ const isWordCode = (c: number) =>
251
+ (c >= 97 && c <= 122) ||
252
+ (c >= 65 && c <= 90) ||
253
+ (c >= 48 && c <= 57) ||
254
+ c === 95
255
+
256
+ // non-ascii members of js \s (u00a0, u1680, u2000-u200a, u2028/9, u202f,
257
+ // u205f, u3000, ufeff); only reached for code units >= 0xa0
258
+ const isUniWS = (c: number): boolean => /\s/.test(String.fromCharCode(c))
259
+
260
+ const analyzeArb = (input: string, s: number, e: number): void => {
261
+ aKind = 0
262
+ aLabelS = -1
263
+ if (e - s < 3) return
264
+ const c0 = input.charCodeAt(s)
265
+ const cl = input.charCodeAt(e - 1)
266
+ if (c0 === 91 && cl === 93) aKind = 1
267
+ else if (c0 === 40 && cl === 41) aKind = 2
268
+ else return
269
+ aValS = s + 1
270
+ aValE = e - 1
271
+ // label: \w[\w-]* directly followed by ':' with non-empty value
272
+ let p = s + 1
273
+ if (isWordCode(input.charCodeAt(p))) {
274
+ p++
275
+ while (p < e - 1) {
276
+ const c = input.charCodeAt(p)
277
+ if (!isWordCode(c) && c !== 45) break
278
+ p++
279
+ }
280
+ if (p < e - 2 && input.charCodeAt(p) === 58) {
281
+ aLabelS = s + 1
282
+ aLabelE = p
283
+ aValS = p + 1
284
+ }
285
+ }
286
+ }
287
+
288
+ const spanEq = (
289
+ input: string,
290
+ s: number,
291
+ e: number,
292
+ str: string
293
+ ): boolean => {
294
+ if (e - s !== str.length) return false
295
+ for (let i = 0; i < str.length; i++) {
296
+ if (input.charCodeAt(s + i) !== str.charCodeAt(i)) return false
297
+ }
298
+ return true
299
+ }
300
+
301
+ // simple value shapes as regexes on lazy slices (memo-miss path only;
302
+ // the hot arbitrary-value analysis stays span-based in analyzeArb)
303
+ const fractionRegex = /^\d+(?:\.\d+)?\/\d+(?:\.\d+)?$/
304
+ const tshirtRegex = /^(\d+(\.\d+)?)?(xs|sm|md|lg|xl)$/
305
+ const isNumStr = (v: string) => !!v && !Number.isNaN(Number(v))
306
+ const spanIsNamedContainerQuery = (
307
+ input: string,
308
+ s: number,
309
+ e: number
310
+ ): boolean => {
311
+ if (e - s < 11 || !spanEq(input, s, s + 10, "@container")) return false
312
+ if (input.charCodeAt(s + 10) === 47) return e - s >= 12
313
+ const c11 = input.charCodeAt(s + 11)
314
+ return (
315
+ (c11 === 115 && e - s >= 17 && spanEq(input, s + 10, s + 16, "-size/")) ||
316
+ (c11 === 110 && e - s >= 19 && spanEq(input, s + 10, s + 18, "-normal/"))
317
+ )
318
+ }
319
+
320
+ // ops 10-24: required kind, allowed labels, unlabeled behavior
321
+ // (0 false, 1 true, 2 length-shape, 3 number, 4 image, 5 shadow)
322
+ const VKIND = [1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 2]
323
+ const VLABELS =
324
+ "length|number|number weight|family-name|position percentage|length size bg-size|image url|shadow|length|family-name|position percentage|length size bg-size|image url|shadow|number weight"
325
+ .split("|")
326
+ .map((s) => s.split(" "))
327
+ const VFALL = [2, 3, 1, 0, 0, 0, 4, 5, 0, 0, 0, 0, 0, 1, 1]
328
+
329
+ const runValidator = (
330
+ op: number,
331
+ input: string,
332
+ s: number,
333
+ e: number
334
+ ): boolean => {
335
+ if (op >= 10) {
336
+ if (op >= 25) return vCustom[op - 25](input.slice(s, e))
337
+ const i = op - 10
338
+ if (aKind !== VKIND[i]) return false
339
+ if (aLabelS >= 0) {
340
+ for (const L of VLABELS[i])
341
+ if (spanEq(input, aLabelS, aLabelE, L)) return true
342
+ return false
343
+ }
344
+ switch (VFALL[i]) {
345
+ case 0:
346
+ return false
347
+ case 1:
348
+ return true
349
+ case 2: {
350
+ const v = input.slice(aValS, aValE)
351
+ return lengthUnitRegex.test(v) && !colorFunctionRegex.test(v)
352
+ }
353
+ case 3:
354
+ return isNumStr(input.slice(aValS, aValE))
355
+ case 4:
356
+ return imageRegex.test(input.slice(aValS, aValE))
357
+ default:
358
+ return shadowRegex.test(input.slice(aValS, aValE))
359
+ }
360
+ }
361
+ switch (op) {
362
+ case 0:
363
+ return true
364
+ case 1:
365
+ return aKind === 0
366
+ case 2:
367
+ return aKind === 1
368
+ case 3:
369
+ return aKind === 2
370
+ case 4:
371
+ return fractionRegex.test(input.slice(s, e))
372
+ case 5:
373
+ return isNumStr(input.slice(s, e))
374
+ case 6: {
375
+ const v = input.slice(s, e)
376
+ return !!v && Number.isInteger(Number(v))
377
+ }
378
+ case 7:
379
+ return (
380
+ e > s &&
381
+ input.charCodeAt(e - 1) === 37 &&
382
+ isNumStr(input.slice(s, e - 1))
383
+ )
384
+ case 8:
385
+ return tshirtRegex.test(input.slice(s, e))
386
+ default:
387
+ return spanIsNamedContainerQuery(input, s, e)
388
+ }
389
+ }
390
+
391
+ const orderSensitive = new Set(
392
+ typeof orderSensitiveModifiers === "string"
393
+ ? orderSensitiveModifiers.split(" ")
394
+ : orderSensitiveModifiers
395
+ )
396
+
397
+ // ---- span interning (contexts + dynamic groups), process lifetime ------
398
+ // hash buckets hold materialized strings (allocated once per unique span);
399
+ // resets happen only between merges so ids stay consistent within a pass.
400
+ interface InternEntry {
401
+ k: string
402
+ imp: number
403
+ id: number
404
+ }
405
+ const internSpan = (
406
+ map: Map<number, InternEntry[]>,
407
+ input: string,
408
+ s: number,
409
+ e: number,
410
+ imp: number,
411
+ make: (raw: string) => number
412
+ ): number => {
413
+ const h = (fnv(input, s, e) ^ (imp ? 0x9e3779b9 : 0)) | 0
414
+ let bucket = map.get(h)
415
+ if (bucket !== undefined) {
416
+ outer: for (let b = 0; b < bucket.length; b++) {
417
+ const en = bucket[b]
418
+ if (en.imp !== imp || en.k.length !== e - s) continue
419
+ for (let i = 0; i < en.k.length; i++) {
420
+ if (en.k.charCodeAt(i) !== input.charCodeAt(s + i)) continue outer
421
+ }
422
+ return en.id
423
+ }
424
+ } else map.set(h, (bucket = []))
425
+ const k = input.slice(s, e)
426
+ const id = make(k)
427
+ bucket.push({ k, imp, id })
428
+ return id
429
+ }
430
+
431
+ let ctxByHash = new Map()
432
+ let ctxByCanon = new Map()
433
+ let nextCtxId = 2 // 0 = no variants, 1 = no variants + important
434
+ const MAX_CTX = 4096
435
+
436
+ const canonicalizeContext = (raw: string, important: boolean): number => {
437
+ // split raw prefix at top-level ':' (depth-guarded), then segment-sort
438
+ // exactly like tailwind-merge's sortModifiers
439
+ const mods = []
440
+ let dB = 0,
441
+ dP = 0,
442
+ start = 0
443
+ for (let i = 0; i < raw.length; i++) {
444
+ const c = raw.charCodeAt(i)
445
+ if (dB === 0 && dP === 0 && c === 58) {
446
+ mods.push(raw.slice(start, i))
447
+ start = i + 1
448
+ } else if (c === 91) dB++
449
+ else if (c === 93) dB--
450
+ else if (c === 40) dP++
451
+ else if (c === 41) dP--
452
+ }
453
+ mods.push(raw.slice(start))
454
+
455
+ let canonical = mods[0]
456
+ if (mods.length > 1) {
457
+ const result = []
458
+ let segment = []
459
+ for (const mod of mods) {
460
+ if (mod.charCodeAt(0) === 91 || orderSensitive.has(mod)) {
461
+ if (segment.length) {
462
+ result.push(...segment.sort())
463
+ segment = []
464
+ }
465
+ result.push(mod)
466
+ } else segment.push(mod)
467
+ }
468
+ if (segment.length) result.push(...segment.sort())
469
+ canonical = result.join(":")
470
+ }
471
+ const key = important ? canonical + " !" : canonical
472
+ let id = ctxByCanon.get(key)
473
+ if (id === undefined) ctxByCanon.set(key, (id = nextCtxId++))
474
+ return id
475
+ }
476
+
477
+ let dynByHash = new Map()
478
+ let nextDynId = GROUP_COUNT
479
+ const MAX_DYN = GROUP_COUNT + 4096
480
+ const newDynId = () => nextDynId++
481
+ const ID_LIMIT = 2097152 // 2^21: keeps ctx * 2^21 + gid exact in a double
482
+
483
+ // ---- token memo (process lifetime, 2-way set-associative) --------------
484
+ // full token span → (gid, ctxId, flags); hit verifies chars in place, so
485
+ // it allocates nothing — unlike a string-keyed cache, which must
486
+ // materialize the token substring before it can even look it up.
487
+ const TOKEN_TABLE = 8192
488
+ const memoHash = new Int32Array(TOKEN_TABLE)
489
+ const memoStr = new Array(TOKEN_TABLE).fill(null)
490
+ const memoGid = new Int32Array(TOKEN_TABLE)
491
+ const memoCtx = new Int32Array(TOKEN_TABLE)
492
+ const memoFlags = new Uint8Array(TOKEN_TABLE)
493
+ // second-chance insertion: an occupied way is overwritten only every 4th
494
+ // colliding miss, so one-shot tokens can't thrash out hot entries
495
+ let memoTick = 0
496
+
497
+ const memoPut = (
498
+ way0: number,
499
+ input: string,
500
+ ts: number,
501
+ te: number,
502
+ h: number,
503
+ gid: number,
504
+ ctxId: number,
505
+ flags: number
506
+ ): void => {
507
+ let slot = way0
508
+ if (memoStr[way0] !== null) {
509
+ if (memoStr[way0 | 1] === null) slot = way0 | 1
510
+ else if ((memoTick++ & 3) === 0) slot = way0 | ((memoTick >> 2) & 1)
511
+ else return
512
+ }
513
+ memoStr[slot] = input.slice(ts, te)
514
+ memoHash[slot] = h
515
+ memoGid[slot] = gid
516
+ memoCtx[slot] = ctxId
517
+ memoFlags[slot] = flags
518
+ }
519
+ const memoReset = () => memoStr.fill(null)
520
+
521
+ // ---- per-merge reusable state -------------------------------------------
522
+ let cap = 256
523
+ let tokI32 = [
524
+ new Int32Array(cap),
525
+ new Int32Array(cap),
526
+ new Int32Array(cap),
527
+ new Int32Array(cap),
528
+ ]
529
+ let [tokStart, tokEnd, tokGid, tokCtx] = tokI32
530
+ let tokFlags = new Uint8Array(cap)
531
+ let keep = new Uint8Array(cap)
532
+ const growTokens = () => {
533
+ cap *= 2
534
+ tokI32 = tokI32.map((a) => {
535
+ const n = new Int32Array(cap)
536
+ n.set(a)
537
+ return n
538
+ })
539
+ ;[tokStart, tokEnd, tokGid, tokCtx] = tokI32
540
+ const nf = new Uint8Array(cap)
541
+ nf.set(tokFlags)
542
+ tokFlags = nf
543
+ keep = new Uint8Array(cap)
544
+ }
545
+
546
+ let ckptCap = 64
547
+ let ckptNode = new Int32Array(ckptCap)
548
+ let ckptTail = new Int32Array(ckptCap)
549
+
550
+ // no-variant static claims (the dominant case) index an epoch-stamped
551
+ // array directly by gid — one load to test, one store to claim
552
+ const claim0 = new Int32Array(GROUP_COUNT)
553
+
554
+ // unified claim set for the rest (variant contexts, dynamic groups):
555
+ // epoch-stamped open-addressed (ctxId, gid) keys stored as exact doubles —
556
+ // ids are bounded per merge by ID_LIMIT
557
+ let CLAIM_TABLE = 2048
558
+ let claimShift = 21 // 32 - log2(CLAIM_TABLE)
559
+ let claimKeys = new Float64Array(CLAIM_TABLE)
560
+ let claimEpochs = new Int32Array(CLAIM_TABLE)
561
+ let epoch = 0
562
+ // test-and-claim in one probe: returns 1 if (ctx, gid) was already
563
+ // claimed this merge, else claims it and returns 0
564
+ const claimTest = (ctx: number, gid: number): number => {
565
+ if (ctx === 0 && gid < GROUP_COUNT) {
566
+ if (claim0[gid] === epoch) return 1
567
+ claim0[gid] = epoch
568
+ return 0
569
+ }
570
+ const key = ctx * 2097152 + gid + 1
571
+ let idx = Math.imul(key, 0x9e3779b1) >>> claimShift
572
+ for (;;) {
573
+ if (claimEpochs[idx] !== epoch) break // free slot
574
+ if (claimKeys[idx] === key) return 1
575
+ idx = (idx + 1) & (CLAIM_TABLE - 1)
576
+ }
577
+ claimKeys[idx] = key
578
+ claimEpochs[idx] = epoch
579
+ return 0
580
+ }
581
+
582
+ // ---- cold resolver (mirrors upstream mergeClassList) -------------------
583
+ const resolveAt = (
584
+ input: string,
585
+ bs: number,
586
+ endPos: number,
587
+ nodeAt: number,
588
+ ckptAt: number
589
+ ): number => {
590
+ // arbitrary property: '[prop:...]' → dynamic per-property group
591
+ if (
592
+ endPos - bs >= 2 &&
593
+ input.charCodeAt(bs) === 91 &&
594
+ input.charCodeAt(endPos - 1) === 93
595
+ ) {
596
+ let colon = -1
597
+ for (let p = bs + 1; p < endPos - 1; p++) {
598
+ if (input.charCodeAt(p) === 58) {
599
+ colon = p
600
+ break
601
+ }
602
+ }
603
+ if (colon === -1 || colon === bs + 1) return EXTERNAL
604
+ return internSpan(dynByHash, input, bs + 1, colon, 0, newDynId)
605
+ }
606
+ // exact match: automaton at a node with a group id
607
+ if (nodeAt >= 0 && nodeGroup[nodeAt] >= 0) return nodeGroup[nodeAt]
608
+ // backtrack levels, deepest first: literal-map probe (lifted exact
609
+ // matches beat validators, exactly as deeper trie paths beat
610
+ // validators upstream), then the level's validator opcodes
611
+ for (let k = ckptAt - 1; k >= 0; k--) {
612
+ const tailStart = ckptTail[k]
613
+ if (tailStart > endPos) continue
614
+ const nodeId = ckptNode[k]
615
+ const tlen = endPos - tailStart
616
+ if (nodeHasLit[nodeId] === 1 && tlen > 0 && tlen <= litMaxLen) {
617
+ // arbitrary-value tails ('[…]', '(…)') can't match a literal
618
+ // unless the compiled pool actually contains one
619
+ const c0 = input.charCodeAt(tailStart)
620
+ if (litNoArb === false || (c0 !== 91 && c0 !== 40)) {
621
+ const g = litProbe(nodeId, input, tailStart, endPos)
622
+ if (g >= 0) return g
623
+ }
624
+ }
625
+ const vl = nodeVlist[nodeId]
626
+ if (vl < 0) continue
627
+ const pat = vlistRef[vl]
628
+ const vs = vlistPat[pat]
629
+ const ve = vlistPat[pat + 1]
630
+ if (vs === ve) continue
631
+ analyzeArb(input, tailStart, endPos)
632
+ const g0 = vgStart[vl] - vs
633
+ for (let v = vs; v < ve; v++) {
634
+ if (runValidator(vlistOps[v], input, tailStart, endPos)) {
635
+ return vlistGroup[g0 + v]
636
+ }
637
+ }
638
+ }
639
+ return EXTERNAL
640
+ }
641
+
642
+ // ---- the merge -----------------------------------------------------------
643
+ const mergeClassList = (input: string): string => {
644
+ const n = input.length
645
+ let tokenCount = 0
646
+ let totalTokenChars = 0
647
+ let sawNonSpaceWS = false
648
+
649
+ // bounded-growth resets, between merges only
650
+ if (nextCtxId > MAX_CTX || ctxByHash.size > MAX_CTX) {
651
+ ctxByHash = new Map()
652
+ ctxByCanon = new Map()
653
+ nextCtxId = 2
654
+ memoReset()
655
+ }
656
+ if (nextDynId > MAX_DYN) {
657
+ dynByHash = new Map()
658
+ nextDynId = GROUP_COUNT
659
+ memoReset()
660
+ }
661
+
662
+ let i = 0
663
+ while (i < n) {
664
+ let c = input.charCodeAt(i)
665
+ if (c === 32 || (c >= 9 && c <= 13) || (c >= 0xa0 && isUniWS(c))) {
666
+ if (c !== 32) sawNonSpaceWS = true
667
+ i++
668
+ continue
669
+ }
670
+ const ts = i
671
+ // the scan already reads every token char, so the memo hash rides
672
+ // along as a fused FNV accumulator — no second pass per token
673
+ let th = 0
674
+ while (i < n) {
675
+ c = input.charCodeAt(i)
676
+ if (c <= 32) {
677
+ if (c === 32) break
678
+ if (c >= 9 && c <= 13) {
679
+ sawNonSpaceWS = true
680
+ break
681
+ }
682
+ // control chars 0-8/14-31 are token chars (parity)
683
+ } else if (c >= 0xa0 && isUniWS(c)) {
684
+ sawNonSpaceWS = true
685
+ break
686
+ }
687
+ th = Math.imul(th ^ c, 0x01000193)
688
+ i++
689
+ }
690
+ const te = i
691
+ const len = te - ts
692
+ if (tokenCount === cap) growTokens()
693
+ const t = tokenCount++
694
+ tokStart[t] = ts
695
+ tokEnd[t] = te
696
+ totalTokenChars += len
697
+
698
+ th ^= Math.imul(len, 0x9e3779b1)
699
+ const h = (th ^ (th >>> 15)) | 0
700
+
701
+ // 2-way set-associative memo probe: hit → done, zero allocation
702
+ const way0 = h & (TOKEN_TABLE - 1) & ~1
703
+ {
704
+ let hitAt = -1
705
+ if (
706
+ memoHash[way0] === h &&
707
+ memoStr[way0] !== null &&
708
+ memoStr[way0].length === len
709
+ )
710
+ hitAt = way0
711
+ else if (
712
+ memoHash[way0 | 1] === h &&
713
+ memoStr[way0 | 1] !== null &&
714
+ memoStr[way0 | 1].length === len
715
+ )
716
+ hitAt = way0 | 1
717
+ if (hitAt >= 0) {
718
+ const s = memoStr[hitAt]
719
+ // in-place byte verify at every length: a slice + '==='
720
+ // would allocate the substring the memo exists to avoid
721
+ let ok = true
722
+ for (let k = 0; k < len; k++) {
723
+ if (s.charCodeAt(k) !== input.charCodeAt(ts + k)) {
724
+ ok = false
725
+ break
726
+ }
727
+ }
728
+ if (ok) {
729
+ tokGid[t] = memoGid[hitAt]
730
+ tokCtx[t] = memoCtx[hitAt]
731
+ tokFlags[t] = memoFlags[hitAt]
732
+ continue
733
+ }
734
+ }
735
+ }
736
+
737
+ // ===== memo miss: optional prefix gate, then structural parse ===
738
+ let pts = ts
739
+ if (FPL !== 0) {
740
+ if (te - ts <= FPL || !input.startsWith(FULL_PREFIX, ts)) {
741
+ tokGid[t] = EXTERNAL
742
+ memoPut(way0, input, ts, te, h, EXTERNAL, 0, 0)
743
+ continue
744
+ }
745
+ pts = ts + FPL
746
+ }
747
+ let depthB = 0,
748
+ depthP = 0
749
+ let lastColon = -1,
750
+ lastSlash = -1
751
+ for (let p = pts; p < te; p++) {
752
+ const pc = input.charCodeAt(p)
753
+ if (depthB === 0 && depthP === 0) {
754
+ if (pc === 58) {
755
+ lastColon = p
756
+ continue
757
+ }
758
+ if (pc === 47) {
759
+ lastSlash = p
760
+ continue
761
+ }
762
+ }
763
+ if (pc === 91) depthB++
764
+ else if (pc === 93) depthB--
765
+ else if (pc === 40) depthP++
766
+ else if (pc === 41) depthP--
767
+ }
768
+
769
+ const modStart = lastColon >= pts ? lastColon + 1 : pts
770
+
771
+ // important modifier: suffix '!' first (v4), else legacy prefix
772
+ let bs = modStart
773
+ let be = te
774
+ let important = false
775
+ let prefixShift = 0
776
+ if (be > bs && input.charCodeAt(be - 1) === 33) {
777
+ important = true
778
+ be--
779
+ } else if (be > bs && input.charCodeAt(bs) === 33) {
780
+ important = true
781
+ bs++
782
+ prefixShift = 1
783
+ }
784
+
785
+ // postfix candidate — replicates upstream exactly, including the
786
+ // prefix-'!' index-shift quirk (end includes '/' when shifted)
787
+ let postfixEnd = -1
788
+ if (lastSlash > modStart) {
789
+ postfixEnd = lastSlash + prefixShift
790
+ if (postfixEnd >= be) postfixEnd = -1
791
+ }
792
+
793
+ // ===== pass B: feed base through the radix automaton ============
794
+ let feedStart = bs
795
+ if (be - bs > 1 && input.charCodeAt(bs) === 45) feedStart = bs + 1 // negative values
796
+
797
+ let node = 0
798
+ let lp = 0 // label window: lp < le → mid-edge
799
+ let le = 0
800
+ let pending = -1
801
+ let ckptTop = 0
802
+ if (nodeVlist[0] >= 0 || nodeHasLit[0] === 1) {
803
+ ckptNode[0] = 0
804
+ ckptTail[0] = feedStart
805
+ ckptTop = 1
806
+ }
807
+ let slashNode = DEAD
808
+ let slashCkpt = 0
809
+
810
+ for (let p = feedStart; p < be; p++) {
811
+ if (p === postfixEnd) {
812
+ slashNode = lp < le ? DEAD : node
813
+ slashCkpt = ckptTop
814
+ }
815
+ if (node !== DEAD) {
816
+ const cc = input.charCodeAt(p)
817
+ let arrived = -1
818
+ if (lp < le) {
819
+ if (labelText.charCodeAt(lp) === cc) {
820
+ lp++
821
+ if (lp === le) arrived = node = pending
822
+ } else node = DEAD
823
+ } else {
824
+ const es = edgeStart[node]
825
+ const ee = edgeStart[node + 1]
826
+ let next = DEAD
827
+ for (let e = es; e < ee; e++) {
828
+ const ls = labelStart[e]
829
+ if (labelText.charCodeAt(ls) === cc) {
830
+ if (labelStart[e + 1] - ls === 1) arrived = next = edgeTarget[e]
831
+ else {
832
+ lp = ls + 1
833
+ le = labelStart[e + 1]
834
+ pending = edgeTarget[e]
835
+ next = node
836
+ }
837
+ break
838
+ }
839
+ }
840
+ node = next
841
+ }
842
+ if (
843
+ arrived >= 0 &&
844
+ (nodeVlist[arrived] >= 0 || nodeHasLit[arrived] === 1) &&
845
+ p + 1 < be &&
846
+ input.charCodeAt(p + 1) === 45
847
+ ) {
848
+ if (ckptTop === ckptCap) {
849
+ ckptCap *= 2
850
+ const nv = new Int32Array(ckptCap)
851
+ nv.set(ckptNode)
852
+ ckptNode = nv
853
+ const nt = new Int32Array(ckptCap)
854
+ nt.set(ckptTail)
855
+ ckptTail = nt
856
+ }
857
+ ckptNode[ckptTop] = arrived
858
+ ckptTail[ckptTop] = p + 2
859
+ ckptTop++
860
+ }
861
+ }
862
+ }
863
+ if (postfixEnd === be) {
864
+ slashNode = lp < le ? DEAD : node
865
+ slashCkpt = ckptTop
866
+ }
867
+ const endNode = lp < le ? DEAD : node
868
+
869
+ let gid
870
+ let hasPostfix = false
871
+ if (postfixEnd >= 0) {
872
+ hasPostfix = true
873
+ gid = resolveAt(input, bs, postfixEnd, slashNode, slashCkpt)
874
+ if (gid !== EXTERNAL && gid < GROUP_COUNT && postfixLookupSet[gid]) {
875
+ const gidFull = resolveAt(input, bs, be, endNode, ckptTop)
876
+ if (gidFull !== EXTERNAL && gidFull !== gid) {
877
+ gid = gidFull
878
+ hasPostfix = false
879
+ }
880
+ } else if (gid === EXTERNAL) {
881
+ gid = resolveAt(input, bs, be, endNode, ckptTop)
882
+ hasPostfix = false
883
+ }
884
+ } else {
885
+ gid = resolveAt(input, bs, be, endNode, ckptTop)
886
+ }
887
+
888
+ let ctxId = 0
889
+ let flags = 0
890
+ if (gid === EXTERNAL) {
891
+ tokGid[t] = EXTERNAL
892
+ } else {
893
+ flags = hasPostfix ? 1 : 0
894
+ ctxId =
895
+ pts >= lastColon
896
+ ? important
897
+ ? 1
898
+ : 0
899
+ : internSpan(
900
+ ctxByHash,
901
+ input,
902
+ pts,
903
+ lastColon,
904
+ important ? 1 : 0,
905
+ (k: string) => canonicalizeContext(k, important)
906
+ )
907
+ tokGid[t] = gid
908
+ tokFlags[t] = flags
909
+ tokCtx[t] = ctxId
910
+ }
911
+
912
+ memoPut(way0, input, ts, te, h, gid, ctxId, flags)
913
+ }
914
+
915
+ // ===== fast paths =====================================================
916
+ if (tokenCount === 0) return ""
917
+ if (tokenCount === 1) {
918
+ return tokStart[0] === 0 && tokEnd[0] === n
919
+ ? input
920
+ : input.slice(tokStart[0], tokEnd[0])
921
+ }
922
+
923
+ // ===== backward claim pass ============================================
924
+ // worst-case claims = tokens x CLAIM_PER_TOKEN, where CLAIM_PER_TOKEN is
925
+ // derived from the tables' real max fan-out; keep load factor under 50%
926
+ // so probes stay short and the table can never fill
927
+ if (tokenCount * CLAIM_PER_TOKEN > CLAIM_TABLE) {
928
+ while (tokenCount * CLAIM_PER_TOKEN > CLAIM_TABLE) {
929
+ CLAIM_TABLE <<= 1
930
+ claimShift--
931
+ }
932
+ claimKeys = new Float64Array(CLAIM_TABLE)
933
+ claimEpochs = new Int32Array(CLAIM_TABLE)
934
+ }
935
+ if (nextCtxId >= ID_LIMIT || nextDynId >= ID_LIMIT)
936
+ throw new Error("cn: too many distinct classes in one merge")
937
+ // epoch is stored in Int32Arrays, so it has to truncate the same way; on
938
+ // the wrap through 0 the tables must be cleared or unclaimed slots read
939
+ // as claimed
940
+ epoch = (epoch + 1) | 0
941
+ if (epoch === 0) {
942
+ claim0.fill(0)
943
+ claimEpochs.fill(0)
944
+ epoch = 1
945
+ }
946
+ let didDrop = false
947
+ for (let t = tokenCount - 1; t >= 0; t--) {
948
+ const gid = tokGid[t]
949
+ if (gid === EXTERNAL) {
950
+ keep[t] = 1
951
+ continue
952
+ }
953
+ const ctxId = tokCtx[t]
954
+ if (claimTest(ctxId, gid) === 1) {
955
+ keep[t] = 0
956
+ didDrop = true
957
+ continue
958
+ }
959
+ keep[t] = 1
960
+ if (gid < GROUP_COUNT) {
961
+ // claim overridden groups: base adjacency, plus the flat
962
+ // postfix-extra pairs (conflictingClassGroupModifiers)
963
+ const r = adjRow[gid]
964
+ if (r >= 0) {
965
+ for (let k = adjStart[r]; k < adjStart[r + 1]; k++)
966
+ claimTest(ctxId, adjTgt[k])
967
+ }
968
+ if (tokFlags[t] & 1) {
969
+ for (let k = 0; k < patGid.length; k++) {
970
+ if (patGid[k] === gid) claimTest(ctxId, patTgt[k])
971
+ }
972
+ }
973
+ }
974
+ }
975
+
976
+ // ===== emission =======================================================
977
+ if (!didDrop && !sawNonSpaceWS && n === totalTokenChars + tokenCount - 1) {
978
+ return input // already normalized, nothing dropped
979
+ }
980
+ // emit contiguous runs of kept tokens as single slices: fewer
981
+ // allocations, and the result is a flat string (cheap to hash when it
982
+ // lands in a downstream cache) instead of a cons-string chain
983
+ let out = ""
984
+ let t = 0
985
+ while (t < tokenCount) {
986
+ if (!keep[t]) {
987
+ t++
988
+ continue
989
+ }
990
+ const runStart = tokStart[t]
991
+ let runEnd = tokEnd[t]
992
+ let u = t + 1
993
+ while (
994
+ u < tokenCount &&
995
+ keep[u] &&
996
+ tokStart[u] === runEnd + 1 &&
997
+ input.charCodeAt(runEnd) === 32
998
+ ) {
999
+ runEnd = tokEnd[u]
1000
+ u++
1001
+ }
1002
+ if (out.length > 0) out += " "
1003
+ out += input.slice(runStart, runEnd)
1004
+ t = u
1005
+ }
1006
+ return out
1007
+ }
1008
+
1009
+ // ---- whole-string cache (2-generation, doorkeeper-admitted) -------------
1010
+ // A string enters the cache only on its second sighting, tracked by an
1011
+ // epoch-stamped filter keyed with an O(1) positional hash. One-shot
1012
+ // strings (SSR streams) skip both the insert *and* the cache lookup, so
1013
+ // cache-hostile traffic pays a few sampled chars instead of an O(n) hash
1014
+ // per call, and large recurring working sets still warm fully.
1015
+ // 16384 slots × two generations = 128 KB; sized so real-repo working
1016
+ // sets (~10k distinct strings at the corpus p95) fit without exact-tag
1017
+ // slot conflicts evicting each other's sightings
1018
+ const DOOR_SIZE = 16384
1019
+ const door = new Int32Array(DOOR_SIZE * 2) // two generations, base-flipped
1020
+ let doorBase = 0
1021
+ let doorEpoch = 1
1022
+ let cache = Object.create(null)
1023
+ let prevCache = Object.create(null)
1024
+ let cacheMap = new Map<string, string>()
1025
+ let prevCacheMap = new Map<string, string>()
1026
+ let cacheCount = 0
1027
+ let doorMarks = 0
1028
+ // two-generation rotation: sightings survive mark pressure instead of
1029
+ // being wiped, so recurring working sets larger than the filter still
1030
+ // accumulate the two sightings admission needs (a full wipe starves them
1031
+ // forever — measured 6-15x slower on real-repo corpus replays)
1032
+ // the previous generation's epoch is always doorEpoch - 1: rotation
1033
+ // advances both together, so it needs no variable of its own
1034
+ // no wrap guard on the epoch: a wrap needs 2^32 rotations, and even then
1035
+ // a stale match costs one wasted insert, never a wrong result
1036
+ const rotateDoor = () => {
1037
+ doorBase ^= DOOR_SIZE
1038
+ doorEpoch = (doorEpoch + 1) | 0
1039
+ doorMarks = 0
1040
+ }
1041
+ // doorkeeper: a string seen once in the current or previous generation
1042
+ // admits on this sighting. Slots store the full 32-bit hash (xor epoch),
1043
+ // so a slot collision must match all hash bits to count as a sighting —
1044
+ // unique streams (SSR) almost never false-admit, which would cost a
1045
+ // dictionary insert plus generation churn per call. Stale slots from two
1046
+ // generations back self-invalidate via the epoch xor. Slot bits never
1047
+ // overlap the base bit, so the sibling generation's slot is one xor away.
1048
+ const mergeCached = (input: string): string => {
1049
+ // hit path first: warm, identity-stable strings stay at one
1050
+ // object-property read with a V8-cached hash
1051
+ let merged = cache[input]
1052
+ if (merged !== undefined) return merged
1053
+ const hash = spanHash(input, 0, input.length)
1054
+ const slot = (hash & (DOOR_SIZE - 1)) + doorBase
1055
+ const wasSeen =
1056
+ door[slot] === (hash ^ doorEpoch) ||
1057
+ door[slot ^ DOOR_SIZE] === (hash ^ (doorEpoch - 1))
1058
+ if (wasSeen) {
1059
+ merged = prevCache[input]
1060
+ if (merged !== undefined) {
1061
+ cache[input] = merged // promote
1062
+ return merged
1063
+ }
1064
+ }
1065
+ merged = mergeClassList(input)
1066
+ if (wasSeen) {
1067
+ cache[input] = merged
1068
+ if (++cacheCount > cacheSize) {
1069
+ cacheCount = 0
1070
+ prevCache = cache
1071
+ cache = Object.create(null)
1072
+ rotateDoor()
1073
+ }
1074
+ } else {
1075
+ door[slot] = hash ^ doorEpoch
1076
+ if (++doorMarks > DOOR_SIZE) rotateDoor()
1077
+ }
1078
+ return merged
1079
+ }
1080
+ // same as mergeCached over Maps: JSC looks up a string key in a
1081
+ // dictionary-mode object in ~21 ns and a fresh key in ~220 ns, where a Map
1082
+ // takes 6 and 75 (V8 is the reverse, 5 vs 18 on a hit, so it keeps the
1083
+ // dictionary above). Kept as a copy rather than an accessor layer, which
1084
+ // cost V8 5% on every hit.
1085
+ const mergeCachedMap = (input: string): string => {
1086
+ let merged = cacheMap.get(input)
1087
+ if (merged !== undefined) return merged
1088
+ const hash = spanHash(input, 0, input.length)
1089
+ const slot = (hash & (DOOR_SIZE - 1)) + doorBase
1090
+ const wasSeen =
1091
+ door[slot] === (hash ^ doorEpoch) ||
1092
+ door[slot ^ DOOR_SIZE] === (hash ^ (doorEpoch - 1))
1093
+ if (wasSeen) {
1094
+ merged = prevCacheMap.get(input)
1095
+ if (merged !== undefined) {
1096
+ cacheMap.set(input, merged) // promote
1097
+ return merged
1098
+ }
1099
+ }
1100
+ merged = mergeClassList(input)
1101
+ if (wasSeen) {
1102
+ cacheMap.set(input, merged)
1103
+ if (++cacheCount > cacheSize) {
1104
+ cacheCount = 0
1105
+ prevCacheMap = cacheMap
1106
+ cacheMap = new Map()
1107
+ rotateDoor()
1108
+ }
1109
+ } else {
1110
+ door[slot] = hash ^ doorEpoch
1111
+ if (++doorMarks > DOOR_SIZE) rotateDoor()
1112
+ }
1113
+ return merged
1114
+ }
1115
+ // a string that was just built cannot be cached by identity, and a
1116
+ // never-seen key is the expensive dictionary case (V8 hashes and
1117
+ // internalizes it: ~200 ns at 17 chars, ~1.1 µs at 360). The doorkeeper
1118
+ // answers "never seen" in O(1) instead, so the caller can merge a
1119
+ // one-shot string uncached and skip caching it anywhere
1120
+ const seenBefore = (input: string): boolean => {
1121
+ const hash = spanHash(input, 0, input.length)
1122
+ const slot = (hash & (DOOR_SIZE - 1)) + doorBase
1123
+ if (
1124
+ door[slot] === (hash ^ doorEpoch) ||
1125
+ door[slot ^ DOOR_SIZE] === (hash ^ (doorEpoch - 1))
1126
+ )
1127
+ return true
1128
+ door[slot] = hash ^ doorEpoch
1129
+ if (++doorMarks > DOOR_SIZE) rotateDoor()
1130
+ return false
1131
+ }
1132
+ // JSC will not inline mergeCached with the doorkeeper body in it, so it
1133
+ // gets a two-line hit front that only falls through to the full function
1134
+ // on a miss (the repeated lookup there rides the hash the front just
1135
+ // cached). V8 inlines the full closure and loses ~15% on long strings
1136
+ // with the body outlined, so it uses mergeCached directly.
1137
+ const mergeString =
1138
+ cacheSize === 0
1139
+ ? mergeClassList
1140
+ : IS_JSC
1141
+ ? (input: string): string => {
1142
+ const merged = cacheMap.get(input)
1143
+ return merged !== undefined ? merged : mergeCachedMap(input)
1144
+ }
1145
+ : mergeCached
1146
+
1147
+ const merge = function (): string {
1148
+ return arguments.length === 1 && typeof arguments[0] === "string"
1149
+ ? mergeString(arguments[0])
1150
+ : mergeString(twJoin.apply(null, arguments as never))
1151
+ } as Engine["merge"]
1152
+
1153
+ return {
1154
+ merge,
1155
+ mergeString,
1156
+ seenBefore: cacheSize === 0 ? () => false : seenBefore,
1157
+ mergeUncached: mergeClassList,
1158
+ }
1159
+ }
1160
+
1161
+ // shared value resolution. clsxMode adds clsx's extras (numbers, object
1162
+ // syntax); twJoin mode ignores them, matching tailwind-merge's twJoin.
1163
+ const resolveValue = (v: ClassValue, clsxMode: boolean): string => {
1164
+ if (!v) return ""
1165
+ if (typeof v === "string") return v
1166
+ let out = ""
1167
+ if (
1168
+ typeof (v as { length?: unknown }).length === "number" &&
1169
+ (clsxMode ? Array.isArray(v) : true)
1170
+ ) {
1171
+ const arr = v as ArrayLike<ClassValue>
1172
+ for (let i = 0; i < arr.length; i++) {
1173
+ const item = arr[i]
1174
+ if (!item) continue
1175
+ const r = typeof item === "string" ? item : resolveValue(item, clsxMode)
1176
+ if (r) {
1177
+ if (out) out += " "
1178
+ out += r
1179
+ }
1180
+ }
1181
+ return out
1182
+ }
1183
+ if (clsxMode) {
1184
+ if (typeof v === "number") return "" + v
1185
+ if (typeof v === "object") {
1186
+ for (const k in v)
1187
+ if ((v as Record<string, unknown>)[k]) {
1188
+ if (out) out += " "
1189
+ out += k
1190
+ }
1191
+ }
1192
+ }
1193
+ return out
1194
+ }
1195
+
1196
+ const joinArgs = (args: IArguments, clsxMode: boolean): string => {
1197
+ let s = ""
1198
+ for (let i = 0; i < args.length; i++) {
1199
+ const a = args[i]
1200
+ if (!a) continue
1201
+ const r =
1202
+ typeof a === "string" ? a : resolveValue(a as ClassValue, clsxMode)
1203
+ if (r) {
1204
+ if (s) s += " "
1205
+ s += r
1206
+ }
1207
+ }
1208
+ return s
1209
+ }
1210
+
1211
+ /** join-only, `twJoin`-compatible (strings + nested arrays, falsy skipped) */
1212
+ export const twJoin = function (): string {
1213
+ return joinArgs(arguments, false)
1214
+ } as (...inputs: ClassNameValue[]) => string
1215
+
1216
+ /** join-only, `clsx`-compatible (no merging) */
1217
+ export const clsx = function (): string {
1218
+ return joinArgs(arguments, true)
1219
+ } as (...inputs: ClassValue[]) => string
1220
+
1221
+ // clsx-parity join over any mergeString. Separate export so merge-only
1222
+ // consumers tree-shake it.
1223
+ interface ArgEntry {
1224
+ /** merged result */
1225
+ r: string
1226
+ /** truthy arg count (=== a.length, denormalized for the unrolled probes) */
1227
+ t: number
1228
+ /** first three truthy args, '' padded — monomorphic fields so the arity
1229
+ * fronts verify without an array indirection */
1230
+ a0: string
1231
+ a1: string
1232
+ a2: string
1233
+ /** the truthy string args, in order (identity-compared; generic paths) */
1234
+ a: string[]
1235
+ /** the entry that followed this one last time (sequence prediction) */
1236
+ n: ArgEntry | null
1237
+ }
1238
+
1239
+ // One base string's tuples. `miss` counts bucket walks that found nothing
1240
+ // since the last hit; past BUCKET_MISS_CAP the bucket is churning (a fresh
1241
+ // string instance per call, e.g. an interpolated arbitrary value), so walk
1242
+ // and insert are skipped for BUCKET_SKIP calls before probing again.
1243
+ interface ArgBucket {
1244
+ e: ArgEntry[]
1245
+ miss: number
1246
+ skip: number
1247
+ /** next overwrite slot once the bucket is full (ring, no shift) */
1248
+ at: number
1249
+ }
1250
+ const BUCKET_CAP = 256
1251
+ const BUCKET_MISS_CAP = 16
1252
+ const BUCKET_SKIP = 1024
1253
+ // churn front: a joined string built from a fresh arg can't be found by
1254
+ // identity, and looking it up in the dictionary makes V8 hash every char
1255
+ // (~110 ns at 45 chars). This 2-way table keys on the O(1) positional hash
1256
+ // and verifies with one string compare instead. Allocated on first churn,
1257
+ // so apps without interpolated args never carry it.
1258
+ const CHURN_SIZE = 4096
1259
+
1260
+ export const wrapClsx = (
1261
+ mergeString: (input: string) => string,
1262
+ fresh?: FreshMerge
1263
+ ): CnFunction => {
1264
+ // without an engine's doorkeeper every join counts as seen and is cached
1265
+ const seenBefore = fresh === undefined ? () => true : fresh.seenBefore
1266
+ const mergeUncached = fresh === undefined ? mergeString : fresh.mergeUncached
1267
+ // arg-identity cache: repeated calls whose truthy args are the same string
1268
+ // *instances* (stable JSX literals — the dominant component shape) skip
1269
+ // the re-join and the O(n) hash of the fresh joined string. Only engages
1270
+ // when every truthy arg is a string: objects/arrays are mutable at the
1271
+ // same identity, so they always take the full resolve path.
1272
+ //
1273
+ // Render loops replay call *sequences*, not just calls, so each entry also
1274
+ // remembers which entry came next last time. When the prediction verifies
1275
+ // (pure identity compares), the call skips even the bucket lookup.
1276
+ let argCache = new Map<string, ArgBucket>()
1277
+ let prevArgCache = new Map<string, ArgBucket>()
1278
+ let argCount = 0
1279
+ let lastHit: ArgEntry | null = null
1280
+ // churn front slots: `key` is the string hashed and value-compared (the
1281
+ // fresh arg on the arity-2 path, the whole joined string otherwise),
1282
+ // `own` the identity-compared base that goes with it ('' for joined).
1283
+ let churnKey: string[] | null = null
1284
+ let churnOwn: string[] = []
1285
+ let churnVal: string[] = []
1286
+ let churnTick = 0
1287
+
1288
+ const churnLookup = (key: string, own: string, join: boolean): string => {
1289
+ if (churnKey === null) {
1290
+ churnKey = new Array<string>(CHURN_SIZE).fill("")
1291
+ churnOwn = new Array<string>(CHURN_SIZE).fill("")
1292
+ churnVal = new Array<string>(CHURN_SIZE).fill("")
1293
+ }
1294
+ const way0 = spanHash(key, 0, key.length) & (CHURN_SIZE - 2)
1295
+ if (churnKey[way0] === key && churnOwn[way0] === own) return churnVal[way0]!
1296
+ if (churnKey[way0 | 1] === key && churnOwn[way0 | 1] === own)
1297
+ return churnVal[way0 | 1]!
1298
+ const merged = mergeString(join ? own + " " + key : key)
1299
+ // fill the empty way first, then round-robin
1300
+ const slot =
1301
+ churnKey[way0] === ""
1302
+ ? way0
1303
+ : way0 | (churnKey[way0 | 1] === "" ? 1 : churnTick++ & 1)
1304
+ churnKey[slot] = key
1305
+ churnOwn[slot] = own
1306
+ churnVal[slot] = merged
1307
+ return merged
1308
+ }
1309
+
1310
+ // unrolled truthy-sequence verify for arity ≤ 3, against the entry's
1311
+ // monomorphic fields. Arity-2 calls pass '' as v2: a falsy pad skips the
1312
+ // slot, so the same code serves both arities. Non-string truthy args can
1313
+ // never strict-equal a string field, so they fail here and take the
1314
+ // resolve path below.
1315
+ const match3 = (
1316
+ e: ArgEntry,
1317
+ v0: ClassValue,
1318
+ v1: ClassValue,
1319
+ v2: ClassValue
1320
+ ): boolean => {
1321
+ let k = 0
1322
+ if (v0) {
1323
+ if (v0 !== e.a0) return false
1324
+ k = 1
1325
+ }
1326
+ if (v1) {
1327
+ if (v1 !== (k === 0 ? e.a0 : e.a1)) return false
1328
+ k++
1329
+ }
1330
+ if (v2) {
1331
+ if (v2 !== (k === 0 ? e.a0 : k === 1 ? e.a1 : e.a2)) return false
1332
+ k++
1333
+ }
1334
+ return k === e.t
1335
+ }
1336
+
1337
+ // generic path for any arity: probes (loop form), clsx fallback for
1338
+ // non-string args, bucket lookup, insert, chain update
1339
+ // loop-form verify for any arity (identity compares; non-strings never match)
1340
+ const matchN = (e: ArgEntry, vals: ClassValue[]): boolean => {
1341
+ const ea = e.a
1342
+ let k = 0
1343
+ for (let i = 0; i < vals.length; i++) {
1344
+ const v = vals[i]
1345
+ if (!v) continue
1346
+ if (v !== ea[k]) return false
1347
+ k++
1348
+ }
1349
+ return k === e.t
1350
+ }
1351
+
1352
+ const resolveArgs = (vals: ClassValue[], probed: boolean): string => {
1353
+ const nArgs = vals.length
1354
+ const pred = lastHit === null ? null : lastHit.n
1355
+ if (!probed) {
1356
+ if (pred !== null && matchN(pred, vals)) {
1357
+ lastHit = pred
1358
+ return pred.r
1359
+ }
1360
+ if (lastHit !== null && lastHit !== pred && matchN(lastHit, vals))
1361
+ return lastHit.r
1362
+ }
1363
+ let first = ""
1364
+ let firstIdx = -1
1365
+ let truthy = 0
1366
+ let hasResolvedValue = false
1367
+ for (let i = 0; i < nArgs; i++) {
1368
+ let v = vals[i]
1369
+ if (!v) continue
1370
+ if (typeof v !== "string") {
1371
+ // objects and arrays resolve in place and ride the string path: a
1372
+ // one-key object resolves to that key string itself, whose identity
1373
+ // is stable across renders, so the arg cache still hits
1374
+ v = vals[i] = resolveValue(v as ClassValue, true)
1375
+ if (!v) continue
1376
+ hasResolvedValue = true
1377
+ }
1378
+ if (firstIdx < 0) {
1379
+ first = v
1380
+ firstIdx = i
1381
+ }
1382
+ truthy++
1383
+ }
1384
+ if (truthy === 0) return ""
1385
+ if (truthy === 1) return mergeString(first) // cheap path; chain untouched
1386
+ if (hasResolvedValue) {
1387
+ // the probes above saw the raw objects; retry them over the resolved
1388
+ // strings before paying for the bucket walk
1389
+ if (pred !== null && matchN(pred, vals)) {
1390
+ lastHit = pred
1391
+ return pred.r
1392
+ }
1393
+ if (lastHit !== null && lastHit !== pred && matchN(lastHit, vals))
1394
+ return lastHit.r
1395
+ }
1396
+ let bucket = argCache.get(first)
1397
+ if (bucket === undefined) {
1398
+ bucket = prevArgCache.get(first)
1399
+ if (bucket !== undefined) argCache.set(first, bucket) // promote
1400
+ }
1401
+ let hit: ArgEntry | null = null
1402
+ if (bucket !== undefined) {
1403
+ if (bucket.skip > 0) {
1404
+ // churning: the tuples here never repeat by identity, so the walk
1405
+ // and the insert are wasted. Join and go through the churn front;
1406
+ // the misses that got us here were all admitted strings, so the
1407
+ // doorkeeper pass is skipped (mergeString keeps its own on a miss).
1408
+ bucket.skip--
1409
+ lastHit = null // no chain to learn here; skip the probes next call
1410
+ let joined = first
1411
+ for (let i = firstIdx + 1; i < nArgs; i++) {
1412
+ const v = vals[i]
1413
+ if (v) joined += " " + (v as string)
1414
+ }
1415
+ return churnLookup(joined, "", false)
1416
+ }
1417
+ const entries = bucket.e
1418
+ outer: for (let b = 0; b < entries.length; b++) {
1419
+ const e = entries[b]!
1420
+ if (e.t !== truthy) continue
1421
+ const ea = e.a
1422
+ let k = 1
1423
+ for (let i = firstIdx + 1; i < nArgs; i++) {
1424
+ const v = vals[i]
1425
+ if (v && v !== ea[k++]) continue outer
1426
+ }
1427
+ hit = e
1428
+ break
1429
+ }
1430
+ if (hit !== null) bucket.miss = 0
1431
+ }
1432
+ if (hit === null) {
1433
+ let joined = first
1434
+ const a: string[] = [first]
1435
+ for (let i = firstIdx + 1; i < nArgs; i++) {
1436
+ const v = vals[i]
1437
+ if (!v) continue
1438
+ joined += " " + (v as string)
1439
+ a.push(v as string)
1440
+ }
1441
+ // a first sighting is merged straight through: no dictionary lookup
1442
+ // on a fresh key, no cache entry anywhere, chain left untouched. A
1443
+ // repeat pays the lookup once and caches like before.
1444
+ if (!seenBefore(joined)) return mergeUncached(joined)
1445
+ hit = {
1446
+ r: mergeString(joined),
1447
+ t: a.length,
1448
+ a0: a[0]!,
1449
+ a1: a[1]!,
1450
+ a2: a[2] ?? "",
1451
+ a,
1452
+ n: null,
1453
+ }
1454
+ if (bucket === undefined) {
1455
+ argCache.set(first, (bucket = { e: [], miss: 0, skip: 0, at: 0 }))
1456
+ } else if (++bucket.miss > BUCKET_MISS_CAP) {
1457
+ // the walk keeps coming up empty: skip it for a while. The stale
1458
+ // tuples go too, so the re-probe walks a short bucket, and one more
1459
+ // miss there re-arms the skip; a hit resets the count. The entry
1460
+ // still goes in so a site that turns stable is found on re-probe.
1461
+ bucket.miss = BUCKET_MISS_CAP
1462
+ bucket.skip = BUCKET_SKIP
1463
+ bucket.e.length = 0
1464
+ bucket.at = 0
1465
+ }
1466
+ // a component's base string is the first arg at every usage site, so
1467
+ // one key can carry dozens of tuples (54 in the largest corpus repo,
1468
+ // more once per-site className props count); a tight cap evicts them
1469
+ // faster than the sequence chain can learn them, at ~40x per call
1470
+ const entries = bucket.e
1471
+ if (entries.length < BUCKET_CAP) entries.push(hit)
1472
+ else {
1473
+ entries[bucket.at] = hit
1474
+ bucket.at = (bucket.at + 1) & (BUCKET_CAP - 1)
1475
+ }
1476
+ // two-generation rotation: a full generation ages out wholesale
1477
+ // instead of clearing everything; hot buckets get promoted on use,
1478
+ // so replayed sequences survive rotation and the chain stays warm
1479
+ if (++argCount > 1000) {
1480
+ argCount = 0
1481
+ prevArgCache = argCache
1482
+ argCache = new Map()
1483
+ }
1484
+ }
1485
+ if (lastHit !== null && lastHit !== hit) lastHit.n = hit
1486
+ lastHit = hit
1487
+ return hit.r
1488
+ }
1489
+
1490
+ // cn([a, b]) is cn(a, b) under clsx's flattening, so a lone array takes
1491
+ // the arg path and its stable element identities hit the cache
1492
+ const mergeSingleValue = (value: ClassValue): string =>
1493
+ Array.isArray(value)
1494
+ ? resolveArgs(value.slice(), false)
1495
+ : mergeString(resolveValue(value, true))
1496
+
1497
+ // named params make the hot path three register reads instead of three
1498
+ // `arguments` element loads; modules are strict, so params never alias
1499
+ // `arguments` (still used for arity and the 4+ overflow copy). Arity 2
1500
+ // rides the same branch as 3: an absent v2 is undefined, and a falsy pad
1501
+ // behaves identically to '' through the probes and the resolve path.
1502
+ return function (v0?: ClassValue, v1?: ClassValue, v2?: ClassValue): string {
1503
+ const nArgs = arguments.length
1504
+ if ((nArgs | 1) === 3) {
1505
+ // arity 2 or 3
1506
+ const lh = lastHit
1507
+ if (lh !== null) {
1508
+ // sequence prediction: does this call repeat what followed
1509
+ // last time?
1510
+ const pred = lh.n
1511
+ if (pred !== null && match3(pred, v0, v1, v2)) {
1512
+ lastHit = pred
1513
+ return pred.r
1514
+ }
1515
+ // self-repeat: the same call site firing again immediately.
1516
+ // Probed rather than stored as a self-link so an entry's
1517
+ // learned successor is never clobbered — a doubled site
1518
+ // (A, A, B) predicts all three calls: A→B via .n, the
1519
+ // repeat via this probe.
1520
+ if (lh !== pred && match3(lh, v0, v1, v2)) return lh.r
1521
+ }
1522
+ // arity-2 churn: cn(base, fresh) where the base's bucket has given
1523
+ // up on identity. Hash the fresh arg alone and identity-check the
1524
+ // base: no array, no join, no walk (the shape of an interpolated
1525
+ // arbitrary value or a className prop built per render). Buckets
1526
+ // are keyed by truthy strings only, so a non-string v0 finds none.
1527
+ if (nArgs === 2 && typeof v1 === "string" && v1 !== "") {
1528
+ const bucket = argCache.get(v0 as string)
1529
+ if (bucket !== undefined && bucket.skip > 0) {
1530
+ bucket.skip--
1531
+ lastHit = null
1532
+ return churnLookup(v1, v0 as string, true)
1533
+ }
1534
+ }
1535
+ return resolveArgs([v0, v1, v2], true)
1536
+ }
1537
+ if (nArgs === 1)
1538
+ return typeof v0 === "string" ? mergeString(v0) : mergeSingleValue(v0)
1539
+ // 4+ arity: probe predictions in place over `arguments` (indexed
1540
+ // reads only, so it never materializes) — a predicted render-loop
1541
+ // call allocates nothing. Only a genuine miss copies into an array
1542
+ // for the resolve path.
1543
+ const lh = lastHit
1544
+ if (lh !== null) {
1545
+ const pred = lh.n
1546
+ if (pred !== null) {
1547
+ const pa = pred.a
1548
+ let k = 0
1549
+ let ok = true
1550
+ for (let i = 0; i < nArgs; i++) {
1551
+ const v = arguments[i]
1552
+ if (!v) continue
1553
+ if (v !== pa[k]) {
1554
+ ok = false
1555
+ break
1556
+ }
1557
+ k++
1558
+ }
1559
+ if (ok && k === pred.t) {
1560
+ lastHit = pred
1561
+ return pred.r
1562
+ }
1563
+ }
1564
+ if (lh !== pred) {
1565
+ const la = lh.a
1566
+ let k = 0
1567
+ let ok = true
1568
+ for (let i = 0; i < nArgs; i++) {
1569
+ const v = arguments[i]
1570
+ if (!v) continue
1571
+ if (v !== la[k]) {
1572
+ ok = false
1573
+ break
1574
+ }
1575
+ k++
1576
+ }
1577
+ if (ok && k === lh.t) return lh.r
1578
+ }
1579
+ }
1580
+ const vals: ClassValue[] = []
1581
+ for (let i = 0; i < nArgs; i++) vals.push(arguments[i])
1582
+ return resolveArgs(vals, true)
1583
+ } as CnFunction
1584
+ }
1585
+
1586
+ /**
1587
+ * Create a `cn` function bound to compiled tables — the entry point for
1588
+ * project-compiled (`cn build`) tables:
1589
+ *
1590
+ * ```ts
1591
+ * import tables from "./cn-tables.js"
1592
+ * import { createCn } from "cn/engine"
1593
+ * export const cn = createCn(tables)
1594
+ * ```
1595
+ */
1596
+ export const createCn = (
1597
+ tables: Tables,
1598
+ validatorImpls?: ValidatorImpls,
1599
+ options?: EngineOptions
1600
+ ): CnFunction => {
1601
+ const engine = createEngine(tables, validatorImpls, options)
1602
+ return wrapClsx(engine.mergeString, engine)
1603
+ }