sugar-high 2.4.0 → 2.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -9,32 +9,6 @@ without requiring a DOM.
9
9
 
10
10
  ![Sugar High preview](https://repository-images.githubusercontent.com/453236442/9aa2144a-3a4c-4a93-a87f-92ca6a37ded6)
11
11
 
12
- ## Benchmarks
13
-
14
- Sugar High, PrismJS, and highlight.js highlighting the same generated TypeScript files:
15
-
16
- <!-- benchmark:start -->
17
- Measured 2026-09-04 with Node v24.18.0, darwin arm64, Apple M4 Pro.
18
-
19
- | TypeScript | Sugar High 2.2.2 | PrismJS 1.30.0 | highlight.js 11.12.0 |
20
- | --- | ---: | ---: | ---: |
21
- | Minified (KiB) | 9.90 | 14.63 | 29.54 |
22
- | Gzip (KiB) | 4.35 | 5.47 | 11.12 |
23
- | 11 KiB | 1.78 | 1.18 | 2.11 |
24
- | 100 KiB | 18.02 | 15.25 | 23.05 |
25
- | 500 KiB | 90.98 | 96.41 | 118.39 |
26
-
27
- Median milliseconds per file; lower is better. 5 timed samples after warmup.
28
- Sizes are TypeScript-only browser bundles, minified with Bun; gzip uses level 9. Theme CSS is excluded.
29
- Loading and initialization are excluded. Each library highlights the same generated TypeScript
30
- into HTML using an explicit language. Grammars and HTML output differ; this is not a measure
31
- of highlighting quality or browser rendering speed. Results vary by machine and workload.
32
- <!-- benchmark:end -->
33
-
34
- Run `pnpm --filter sugar-high benchmark:large --write` to refresh this table and the
35
- [website comparison](https://sugar-high.vercel.app/#benchmarks) from the same measurement.
36
- See [benchmark methodology and options](../../docs/BENCHMARK.md).
37
-
38
12
  ## Install
39
13
 
40
14
  ```sh
@@ -112,11 +86,12 @@ behavior.
112
86
  ## Built-in languages
113
87
 
114
88
  `javascript`, `typescript`, `css`, `python`, `c`, `go`, `java`, `rust`, `json`, `diff`, `shell`,
115
- `cpp`, `csharp`, `sql`, `html`, `yaml`, `markdown`, `plaintext`, `ruby`, `kotlin`, `swift`, `php`,
89
+ `cpp`, `csharp`, `sql`, `html`, `vue`, `svelte`, `yaml`, `markdown`, `plaintext`, `ruby`, `kotlin`, `swift`, `php`,
116
90
  `toml`, `powershell`, `dockerfile`, `graphql`, `hcl`, `zig`, and `lua`.
117
91
 
118
92
  Related dialects share one implementation: JavaScript includes JSX, TypeScript includes TSX, JSON
119
- includes JSONC comments, Shell includes sh/Bash/Zsh, and HCL includes Terraform.
93
+ includes JSONC comments, Shell includes sh/Bash/Zsh, HTML currently provides the base highlighting
94
+ for Vue and Svelte, and HCL includes Terraform.
120
95
 
121
96
  ## Composable core
122
97
 
@@ -266,6 +241,32 @@ See [`docs/API.md`](https://github.com/huozhi/sugar-high/blob/main/docs/API.md)
266
241
  options, and lower-level functions. Upgrading from v1? Read the
267
242
  [v2 migration guide](https://github.com/huozhi/sugar-high/blob/main/docs/MIGRATION.md).
268
243
 
244
+ ## Benchmarks
245
+
246
+ Sugar High, PrismJS, and highlight.js highlighting the same generated TypeScript files:
247
+
248
+ <!-- benchmark:start -->
249
+ Measured 2026-09-04 with Node v24.18.0, darwin arm64, Apple M4 Pro.
250
+
251
+ | TypeScript | Sugar High 2.2.2 | PrismJS 1.30.0 | highlight.js 11.12.0 |
252
+ | --- | ---: | ---: | ---: |
253
+ | Minified (KiB) | 9.90 | 14.63 | 29.54 |
254
+ | Gzip (KiB) | 4.35 | 5.47 | 11.12 |
255
+ | 11 KiB | 1.78 | 1.18 | 2.11 |
256
+ | 100 KiB | 18.02 | 15.25 | 23.05 |
257
+ | 500 KiB | 90.98 | 96.41 | 118.39 |
258
+
259
+ Median milliseconds per file; lower is better. 5 timed samples after warmup.
260
+ Sizes are TypeScript-only browser bundles, minified with Bun; gzip uses level 9. Theme CSS is excluded.
261
+ Loading and initialization are excluded. Each library highlights the same generated TypeScript
262
+ into HTML using an explicit language. Grammars and HTML output differ; this is not a measure
263
+ of highlighting quality or browser rendering speed. Results vary by machine and workload.
264
+ <!-- benchmark:end -->
265
+
266
+ Run `pnpm --filter sugar-high benchmark:large --write` to refresh this table and the
267
+ [website comparison](https://sugar-high.vercel.app/#benchmarks) from the same measurement.
268
+ See [benchmark methodology and options](../../docs/BENCHMARK.md).
269
+
269
270
  ## License
270
271
 
271
272
  MIT
package/lib/core.js CHANGED
@@ -83,7 +83,9 @@ function tokenize(code, options) {
83
83
  const quote = curr
84
84
  const start = i++
85
85
  while (i < code.length) {
86
- if (code[i] === quote && code[i - 1] !== '\\') {
86
+ if (code[i] === '\\') {
87
+ i++
88
+ } else if (code[i] === quote) {
87
89
  i++
88
90
  break
89
91
  }
package/lib/index.d.ts CHANGED
@@ -23,6 +23,8 @@ export type LanguageName =
23
23
  | 'csharp'
24
24
  | 'sql'
25
25
  | 'html'
26
+ | 'vue'
27
+ | 'svelte'
26
28
  | 'yaml'
27
29
  | 'markdown'
28
30
  | 'plaintext'
@@ -0,0 +1,98 @@
1
+ // @ts-check
2
+ import { tokenize as tokenizeCss } from './css.js'
3
+ import { tokenize as tokenizeJavaScript } from './javascript.js'
4
+
5
+ const htmlOptions = {
6
+ keywords: new Set(),
7
+ jsx: true,
8
+ regex: false,
9
+ templateStrings: false,
10
+ onCommentStart: (_currentChar, _nextChar, index, code) => code.startsWith('<!--', index) ? 2 : 0,
11
+ onCommentEnd: (_prevChar, _currChar, index, code) => code.slice(index - 2, index + 1) === '-->' ? 2 : 0,
12
+ }
13
+
14
+ const tokenizeHtml = (code) => tokenizeJavaScript(code, htmlOptions)
15
+
16
+ const embeddedOpeningTag = /^<\s*(script|style)\b/i
17
+ const embeddedClosingTag = (tag) => new RegExp(`</\\s*${tag}\\s*>`, 'ig')
18
+
19
+ /** Find the end of a markup tag without treating `>` inside quotes as its end. */
20
+ function findTagEnd(code, start) {
21
+ let quote = ''
22
+ for (let index = start; index < code.length; index++) {
23
+ const character = code[index]
24
+ if (quote) {
25
+ if (character === quote) quote = ''
26
+ } else if (character === '"' || character === "'") {
27
+ quote = character
28
+ } else if (character === '>') {
29
+ return index + 1
30
+ }
31
+ }
32
+ return -1
33
+ }
34
+
35
+ /** Find real script/style blocks while ignoring comments and quoted attributes. */
36
+ function findEmbeddedBlocks(code) {
37
+ const blocks = []
38
+ let index = 0
39
+ while (index < code.length) {
40
+ if (code.startsWith('<!--', index)) {
41
+ const commentEnd = code.indexOf('-->', index + 4)
42
+ index = commentEnd === -1 ? code.length : commentEnd + 3
43
+ continue
44
+ }
45
+ if (code[index] !== '<') {
46
+ index++
47
+ continue
48
+ }
49
+
50
+ const openEnd = findTagEnd(code, index)
51
+ if (openEnd === -1) break
52
+ const opening = code.slice(index, openEnd)
53
+ const match = opening.match(embeddedOpeningTag)
54
+ if (!match) {
55
+ index = openEnd
56
+ continue
57
+ }
58
+
59
+ const tag = match[1].toLowerCase()
60
+ const closing = embeddedClosingTag(tag)
61
+ closing.lastIndex = openEnd
62
+ const closeMatch = closing.exec(code)
63
+ if (!closeMatch) {
64
+ index = openEnd
65
+ continue
66
+ }
67
+ blocks.push({ start: index, openEnd, closeStart: closeMatch.index, end: closing.lastIndex, tag })
68
+ index = closing.lastIndex
69
+ }
70
+ return blocks
71
+ }
72
+
73
+ /**
74
+ * Tokenize HTML-like files while delegating script and style bodies to their
75
+ * existing language presets. Framework-specific template syntax remains HTML-like.
76
+ * @param {string} code
77
+ * @returns {Array<[number, string]>}
78
+ */
79
+ export function tokenizeEmbeddedHtml(code) {
80
+ /** @type {Array<[number, string]>} */
81
+ const tokens = []
82
+ let cursor = 0
83
+
84
+ for (const block of findEmbeddedBlocks(code)) {
85
+ const { start, openEnd, closeStart, end, tag } = block
86
+
87
+ tokens.push(...tokenizeHtml(code.slice(cursor, start)))
88
+ tokens.push(...tokenizeHtml(code.slice(start, openEnd)))
89
+ tokens.push(...(tag === 'style'
90
+ ? tokenizeCss(code.slice(openEnd, closeStart))
91
+ : tokenizeJavaScript(code.slice(openEnd, closeStart), { jsx: false })))
92
+ tokens.push(...tokenizeHtml(code.slice(closeStart, end)))
93
+ cursor = end
94
+ }
95
+
96
+ tokens.push(...tokenizeHtml(code.slice(cursor)))
97
+ return tokens
98
+ }
package/lib/lang/html.js CHANGED
@@ -1,11 +1,11 @@
1
1
  // @ts-check
2
- import { tokenize as tokenizeJavaScript } from '../presets/javascript-runtime.js'
2
+ import { tokenizeEmbeddedHtml } from './embedded-html.js'
3
3
 
4
4
  export const keywords = new Set([])
5
5
  export const jsx = true
6
6
  export const regex = false
7
7
  export const templateStrings = false
8
- export const tokenize = tokenizeJavaScript
8
+ export const tokenize = tokenizeEmbeddedHtml
9
9
 
10
10
  export const onCommentStart = (_currentChar, _nextChar, index, code) =>
11
11
  code.startsWith('<!--', index) ? 2 : 0
@@ -0,0 +1,2 @@
1
+ export * from './html.js'
2
+ export { tokenizeEmbeddedHtml as tokenize } from './embedded-html.js'
@@ -0,0 +1,2 @@
1
+ export * from './html.js'
2
+ export { tokenizeEmbeddedHtml as tokenize } from './embedded-html.js'
@@ -0,0 +1,2 @@
1
+ export * from './html.js'
2
+ export { tokenizeEmbeddedHtml as tokenize } from './embedded-html.js'
@@ -0,0 +1,2 @@
1
+ export * from './html.js'
2
+ export { tokenizeEmbeddedHtml as tokenize } from './embedded-html.js'
package/lib/lang.js CHANGED
@@ -3,7 +3,7 @@
3
3
  import {
4
4
  c, cpp, csharp, css, diff, dockerfile, go, graphql, hcl, html, java, javascript, json,
5
5
  kotlin, lua, markdown, nonJavaScript, php, plaintext, powershell, python, ruby, rust, shell, sql,
6
- swift, toml, typescript, yaml, zig,
6
+ svelte, swift, toml, typescript, vue, yaml, zig,
7
7
  } from './presets/configs.js'
8
8
 
9
9
  /**
@@ -33,6 +33,8 @@ const languages = [
33
33
  { id: 'csharp', extension: 'cs', aliases: ['c#', 'cs', 'dotnet'], config: nonJavaScript(csharp) },
34
34
  { id: 'sql', extension: 'sql', aliases: [], config: nonJavaScript(sql) },
35
35
  { id: 'html', extension: 'html', aliases: ['htm', 'xml'], config: html },
36
+ { id: 'vue', extension: 'vue', aliases: [], config: vue },
37
+ { id: 'svelte', extension: 'svelte', aliases: [], config: svelte },
36
38
  { id: 'yaml', extension: 'yaml', aliases: ['yml'], config: nonJavaScript(yaml) },
37
39
  { id: 'markdown', extension: 'md', aliases: ['md', 'mdx'], config: nonJavaScript(markdown) },
38
40
  { id: 'plaintext', extension: 'txt', aliases: ['text', 'plain'], config: nonJavaScript(plaintext) },
@@ -27,6 +27,8 @@ import * as sql from '../lang/sql.js'
27
27
  import * as swift from '../lang/swift.js'
28
28
  import * as toml from '../lang/toml.js'
29
29
  import * as typescript from '../lang/typescript.js'
30
+ import * as svelte from '../lang/svelte.js'
31
+ import * as vue from '../lang/vue.js'
30
32
  import * as yaml from '../lang/yaml.js'
31
33
  import * as zig from '../lang/zig.js'
32
34
 
@@ -61,6 +63,8 @@ function createConfigs() {
61
63
  csharp: nonJavaScript(csharp),
62
64
  sql: nonJavaScript(sql),
63
65
  html,
66
+ vue,
67
+ svelte,
64
68
  yaml: nonJavaScript(yaml),
65
69
  markdown: nonJavaScript(markdown),
66
70
  plaintext: nonJavaScript(plaintext),
@@ -88,5 +92,5 @@ function configFor(name) {
88
92
  export {
89
93
  c, configFor, configs, cpp, csharp, css, diff, dockerfile, go, graphql, hcl, html, java,
90
94
  javascript, json, kotlin, lua, markdown, nonJavaScript, php, plaintext, powershell, python, ruby,
91
- rust, shell, sql, swift, toml, typescript, yaml, zig,
95
+ rust, shell, sql, svelte, swift, toml, typescript, vue, yaml, zig,
92
96
  }
@@ -168,18 +168,11 @@ function isSign(ch) {
168
168
  }
169
169
 
170
170
  function isWord(chr) {
171
- return /^[\w_]+$/.test(chr) || hasUnicode(chr)
171
+ return /^[\w$\u0080-\uffff]+$/.test(chr || '')
172
172
  }
173
173
 
174
174
  function isCls(str) {
175
- const chr0 = str[0]
176
- return isWord(chr0) &&
177
- chr0 === chr0.toUpperCase() ||
178
- str === 'null'
179
- }
180
-
181
- function hasUnicode(s) {
182
- return /[^\u0000-\u007f]/.test(s);
175
+ return /^[0-9A-Z\p{Lu}]/u.test(str) || str === 'null'
183
176
  }
184
177
 
185
178
  function isAlpha(chr) {
@@ -187,11 +180,11 @@ function isAlpha(chr) {
187
180
  }
188
181
 
189
182
  function isIdentifierChar(chr) {
190
- return isAlpha(chr) || hasUnicode(chr)
183
+ return /^[$_A-Za-z\u0080-\uffff]$/.test(chr || '')
191
184
  }
192
185
 
193
186
  function isIdentifier(str) {
194
- return isIdentifierChar(str[0]) && (str.length === 1 || isWord(str.slice(1)))
187
+ return isIdentifierChar(str[0]) && isWord(str)
195
188
  }
196
189
 
197
190
  function isStrTemplateChr(chr) {
@@ -282,17 +275,17 @@ function tokenize(code, options) {
282
275
  let __jsxEnter = false
283
276
  /** @type {0 | 1 | 2} 0 = none; 1 = inside `<open`; 2 = inside `</close` */
284
277
  let __jsxTag = 0
285
- let __jsxExpr = false
278
+ let __jsxExprDepth = 0
286
279
  let __jsxTagExpr = 0
287
280
 
288
281
  /** Nested `<open>…</open>` depth (content between tags, including nested elements). */
289
282
  let __jsxStack = 0
290
283
 
291
- const __jsxChild = () => __jsxEnter && !__jsxExpr && !__jsxTag
284
+ const __jsxChild = () => __jsxEnter && !__jsxExprDepth && !__jsxTag
292
285
  // < __content__ >
293
- const inJsxTag = () => __jsxTag && !__jsxChild()
286
+ const inJsxTag = () => __jsxTag
294
287
  // {'__content__'}
295
- const inJsxLiterals = () => !__jsxTag && __jsxChild() && !__jsxExpr && __jsxStack > 0
288
+ const inJsxLiterals = () => __jsxChild() && __jsxStack
296
289
 
297
290
  /** @type {string | null} */
298
291
  let __strQuote = null
@@ -419,14 +412,12 @@ function tokenize(code, options) {
419
412
  if (isSingleQuotes(curr) && !inJsxLiterals() && !inStrTemplateLiterals()) {
420
413
  append()
421
414
  let isStringClose = false
422
- if (prev !== `\\`) {
423
- if (__strQuote && curr === __strQuote) {
424
- __strQuote = null
425
- isStringClose = true
426
- } else if (!__strQuote) {
427
- __strQuote = curr
428
- __strTokenStart = tokens.length
429
- }
415
+ if (__strQuote && curr === __strQuote) {
416
+ __strQuote = null
417
+ isStringClose = true
418
+ } else if (!__strQuote) {
419
+ __strQuote = curr
420
+ __strTokenStart = tokens.length
430
421
  }
431
422
 
432
423
  append(T_STRING, curr)
@@ -438,25 +429,14 @@ function tokenize(code, options) {
438
429
  continue
439
430
  }
440
431
 
441
- if (!inStrTemplateLiterals()) {
442
- if (prev !== '\\n' && isTemplateQuote(curr)) {
443
- append()
444
- append(T_STRING, curr)
445
- __strTemplateQuoteStack++
446
- continue
447
- }
432
+ if (isTemplateQuote(curr)) {
433
+ append()
434
+ __strTemplateQuoteStack += inStrTemplateLiterals() ? -1 : 1
435
+ append(T_STRING, curr)
436
+ continue
448
437
  }
449
438
 
450
439
  if (inStrTemplateLiterals()) {
451
- if (prev !== '\\n' && isTemplateQuote(curr)) {
452
- if (__strTemplateQuoteStack > 0) {
453
- append()
454
- __strTemplateQuoteStack--
455
- append(T_STRING, curr)
456
- continue
457
- }
458
- }
459
-
460
440
  if (c_n === '${') {
461
441
  __strTemplateExprStack++
462
442
  append(T_STRING)
@@ -477,7 +457,7 @@ function tokenize(code, options) {
477
457
  if (curr === '{') {
478
458
  append()
479
459
  append(T_SIGN, curr)
480
- __jsxExpr = true
460
+ __jsxExprDepth = 1
481
461
  continue
482
462
  }
483
463
  }
@@ -634,6 +614,8 @@ function tokenize(code, options) {
634
614
  // string quotation
635
615
  if (isQuotationChar || isStringTemplateLiterals || isSingleQuotes(__strQuote)) {
636
616
  current += curr
617
+ // Consume the escaped character before the next delimiter check.
618
+ if (curr === '\\') current += code[++i] || ''
637
619
  } else if (isRegexChar) {
638
620
  append()
639
621
  const [lastType, lastToken] = last
@@ -656,25 +638,25 @@ function tokenize(code, options) {
656
638
  __regexQuoteStart = true
657
639
  const start = i++
658
640
 
659
- // end of line of end of file
660
- const isEof = () => i >= code.length
661
- const isEol = () => isEof() || code[i] === '\n'
662
-
663
641
  let foundClose = false
664
642
 
665
643
  // `/` is literal inside regex character classes, e.g. `[/]`.
666
644
  let inCharClass = false
667
645
 
668
646
  // traverse to find closing regex slash
669
- for (; !isEol(); i++) {
647
+ for (; i < code.length && code[i] !== '\n'; i++) {
670
648
  const ch = code[i]
671
- const escaped = code[i - 1] === '\\'
672
- if (!escaped && ch === '[') inCharClass = true
673
- if (!escaped && ch === ']') inCharClass = false
674
- if (ch === '/' && !inCharClass && !escaped) {
649
+ if (ch === '\\') {
650
+ if (code[i + 1] === '\n') break
651
+ i++
652
+ continue
653
+ }
654
+ if (ch === '[') inCharClass = true
655
+ if (ch === ']') inCharClass = false
656
+ if (ch === '/' && !inCharClass) {
675
657
  foundClose = true
676
658
  // end of regex, append regex flags
677
- while (start !== i && /^[a-z]$/.test(code[i + 1]) && !isEol()) {
659
+ while (/^[a-z]$/.test(code[i + 1])) {
678
660
  i++
679
661
  }
680
662
  break
@@ -730,11 +712,11 @@ function tokenize(code, options) {
730
712
  append()
731
713
  }
732
714
  } else {
733
- if (__jsxExpr && curr === '}') {
715
+ if (__jsxExprDepth && curr === '}') {
734
716
  append()
735
717
  current = curr
736
718
  append()
737
- __jsxExpr = false
719
+ __jsxExprDepth--
738
720
  } else if (
739
721
  // it's jsx literals and is not a jsx bracket
740
722
  (isJsxLiterals && !JSXBrackets.has(curr)) ||
@@ -761,6 +743,7 @@ function tokenize(code, options) {
761
743
  }
762
744
  else if (JSXBrackets.has(curr)) append()
763
745
  }
746
+ if (__jsxExprDepth && curr === '{') __jsxExprDepth++
764
747
  }
765
748
  }
766
749
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sugar-high",
3
- "version": "2.4.0",
3
+ "version": "2.5.0",
4
4
  "repository": {
5
5
  "type": "git",
6
6
  "url": "git+https://github.com/huozhi/sugar-high.git",