@scalar/code-highlight 0.4.5 → 0.4.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,19 @@
1
1
  # @scalar/code-highlight
2
2
 
3
+ ## 0.4.6
4
+
5
+ ### Patch Changes
6
+
7
+ - [#10074](https://github.com/scalar/scalar/pull/10074): perf(code-highlight): reuse the markdown processor and skip the pipeline for plain paragraphs
8
+
9
+ `htmlFromMarkdown` rebuilt its twelve-plugin unified chain on every call, and each of those rebuilds ran `createLowlight(standardLanguages)`, re-registering all 57 syntax grammars before rendering. Rendering a single API description paid for a whole highlighter. Profiling an API reference schema put 34-44% of the time in markdown, roughly 70% of it in construction rather than parsing.
10
+
11
+ Two changes. One lazily created lowlight instance is now shared through `rehypeHighlight`'s `lowlight` option, and the processor is frozen and cached per option set (sorted `removeTags`, sorted `allowTags`, transform type). Calls that pass a `transform` callback still build per call, because the callback receives the AST. Separately, a description that is a single plain line with no markdown, HTML or collapsible whitespace now returns the paragraph the pipeline would have produced, without running it.
12
+
13
+ Output is unchanged. Every unique description in the Stripe OpenAPI document (5,167 of them), plus targeted probes across seven option sets and 40,000 random strings drawn from an alphabet of the excluded metacharacters, render byte-identically before and after.
14
+
15
+ Two caches now live for the process lifetime: the grammar registry and the processor map. Both are immutable once built, and the registry is only mutated through an `aliases` option this package never passes.
16
+
3
17
  ## 0.4.5
4
18
 
5
19
  ### Patch Changes
@@ -1,4 +1,4 @@
1
- import { i as lowlightLanguageMappings, t as rehypeHighlight } from "./rehype-highlight.js";
1
+ import { a as lowlightLanguageMappings, t as rehypeHighlight } from "./rehype-highlight.js";
2
2
  import { n as rehypeStringify, s as rehypeParse, t as unified } from "./lib.js";
3
3
  import { t as visit } from "./lib2.js";
4
4
  //#region src/code/line-numbers.ts
@@ -1,4 +1,4 @@
1
- import { c as __toESM, o as __commonJSMin, s as __exportAll } from "./rehype-highlight.js";
1
+ import { c as __exportAll, l as __toESM, s as __commonJSMin } from "./rehype-highlight.js";
2
2
  //#region ../../node_modules/.pnpm/property-information@7.1.0/node_modules/property-information/lib/util/schema.js
3
3
  /**
4
4
  * @import {Schema as SchemaType, Space} from 'property-information'
@@ -1,4 +1,4 @@
1
- import { n as convertElement, r as isElement, s as __exportAll, t as rehypeHighlight } from "./rehype-highlight.js";
1
+ import { c as __exportAll, i as isElement, n as createLowlight, r as convertElement, t as rehypeHighlight } from "./rehype-highlight.js";
2
2
  import { _ as stringify, a as zwitch, b as find, c as stringifyPosition, d as getTagID, f as TokenType, g as stringify$1, h as parse$1, i as ccount, l as Parser, m as webNamespaces, n as rehypeStringify, o as htmlVoidElements, p as fromParse5, r as whitespace, s as rehypeParse, t as unified, u as TokenizerMode, v as html$2, y as svg } from "./lib.js";
3
3
  import { i as convert, n as SKIP, r as visitParents, t as visit } from "./lib2.js";
4
4
  import { t as standardLanguages } from "./languages.js";
@@ -20829,35 +20829,100 @@ var transformInlineMarkdownInRawHtml = () => (tree) => {
20829
20829
  });
20830
20830
  };
20831
20831
  /**
20832
+ * One lowlight instance shared by every markdown pipeline.
20833
+ *
20834
+ * Registering the standard grammars is the most expensive part of building a
20835
+ * pipeline, and the registry is read-only once built (nothing here passes
20836
+ * `aliases`, the only option that mutates a given instance), so it is created
20837
+ * on first use and reused from then on.
20838
+ */
20839
+ var sharedLowlight;
20840
+ var getLowlight = () => {
20841
+ sharedLowlight ??= createLowlight(standardLanguages);
20842
+ return sharedLowlight;
20843
+ };
20844
+ /**
20845
+ * Build the markdown to HTML pipeline.
20846
+ */
20847
+ var createProcessor = (tagNames, transform, transformType) => unified().use(remarkParse).use(remarkGfm).use(transformNodes, {
20848
+ transform,
20849
+ type: transformType
20850
+ }).use(remarkRehype, { allowDangerousHtml: true }).use(rehypeAlert).use(transformInlineMarkdownInRawHtml).use(rehypeRaw).use(rehypeSanitize, {
20851
+ ...defaultSchema,
20852
+ clobberPrefix: "",
20853
+ tagNames,
20854
+ attributes: {
20855
+ ...defaultSchema.attributes,
20856
+ abbr: ["title"],
20857
+ "*": [...defaultSchema.attributes?.["*"] ?? [], "className"]
20858
+ },
20859
+ strip: [
20860
+ "script",
20861
+ "style",
20862
+ "object",
20863
+ "embed",
20864
+ "form"
20865
+ ]
20866
+ }).use(rehypeHighlight, {
20867
+ lowlight: getLowlight(),
20868
+ detect: true,
20869
+ className: "custom-scroll"
20870
+ }).use(rehypeExternalLinks, { target: "_blank" }).use(rehypeFormat).use(rehypeStringify);
20871
+ /**
20872
+ * Frozen pipelines keyed by the options that shape them.
20873
+ *
20874
+ * Every plugin in the chain is stateless once attached, so a pipeline can be
20875
+ * frozen and reused for every call that shares the same options. This matters
20876
+ * because API references render one description per schema row, and building
20877
+ * the pipeline used to cost far more than running it. Calls with a `transform`
20878
+ * callback are not cached, because that closure belongs to the caller.
20879
+ */
20880
+ var processorCache = /* @__PURE__ */ new Map();
20881
+ /**
20882
+ * Matches strings that CommonMark, GFM and this pipeline all render as a single plain paragraph.
20883
+ *
20884
+ * The point is to skip the whole markdown pipeline for the many short, plain
20885
+ * descriptions an API reference renders (property summaries such as "Integer
20886
+ * numbers."), where parsing, sanitising, highlighting and formatting cost far
20887
+ * more than the one paragraph they produce.
20888
+ *
20889
+ * The whitelist is deliberately conservative: a single line, no leading or
20890
+ * trailing whitespace, no run of two spaces, no character that markdown or the
20891
+ * serializer treats specially, and no block marker at the start. Every
20892
+ * CommonMark block construct needs a leading space, `#`, `>`, a bullet, an
20893
+ * ordered marker, `<`, a fence, a thematic break or a second line; every inline
20894
+ * construct needs `\`, a backtick, `*`, `_`, `[`, `]`, `<`, `&`, `~`, a hard
20895
+ * break or a GFM autolink literal (`www.`, `:/`, `mailto:`, `xmpp:`, `@`).
20896
+ * `rehype-stringify` escapes only `<` and `&` in text, and `rehype-format`
20897
+ * only collapses whitespace runs and trims block edges, both of which are
20898
+ * excluded here, so the fast path returns exactly what the pipeline returns.
20899
+ *
20900
+ * The `i` flag is load bearing: GFM matches the `www.`, `mailto:` and `xmpp:`
20901
+ * autolink prefixes case-insensitively, so the lookahead has to as well.
20902
+ * Unicode whitespace is written as escapes because a literal U+2028 or U+2029
20903
+ * inside a regular expression literal is a syntax error.
20904
+ */
20905
+ var PLAIN_PARAGRAPH = /^(?![-+=]|\d{1,9}[.)](?:\s|$))(?!.*(?: {2}|:\/|www\.|mailto:|xmpp:))[^\s\\`*_\[\]<>&#~|@\x00-\x1f\x7f\u00a0\u1680\u2000-\u200b\u2028\u2029\u202f\u205f\u3000\ufeff](?:[^\\`*_\[\]<>&#~|@\t\n\v\f\r\x00-\x1f\x7f\u00a0\u1680\u2000-\u200b\u2028\u2029\u202f\u205f\u3000\ufeff]*[^\s\\`*_\[\]<>&#~|@\x00-\x1f\x7f\u00a0\u1680\u2000-\u200b\u2028\u2029\u202f\u205f\u3000\ufeff])?$/i;
20906
+ /**
20832
20907
  * Take a Markdown string and generate HTML from it
20833
20908
  */
20834
20909
  function htmlFromMarkdown(markdown, options) {
20835
20910
  const removeTags = options?.removeTags ?? [];
20836
- const tagNames = [...defaultSchema.tagNames ?? [], ...options?.allowTags ?? []].filter((t) => !removeTags.includes(t));
20837
- return unified().use(remarkParse).use(remarkGfm).use(transformNodes, {
20838
- transform: options?.transform,
20839
- type: options?.transformType
20840
- }).use(remarkRehype, { allowDangerousHtml: true }).use(rehypeAlert).use(transformInlineMarkdownInRawHtml).use(rehypeRaw).use(rehypeSanitize, {
20841
- ...defaultSchema,
20842
- clobberPrefix: "",
20843
- tagNames,
20844
- attributes: {
20845
- ...defaultSchema.attributes,
20846
- abbr: ["title"],
20847
- "*": [...defaultSchema.attributes?.["*"] ?? [], "className"]
20848
- },
20849
- strip: [
20850
- "script",
20851
- "style",
20852
- "object",
20853
- "embed",
20854
- "form"
20855
- ]
20856
- }).use(rehypeHighlight, {
20857
- languages: standardLanguages,
20858
- detect: true,
20859
- className: "custom-scroll"
20860
- }).use(rehypeExternalLinks, { target: "_blank" }).use(rehypeFormat).use(rehypeStringify).processSync(markdown).toString();
20911
+ if (!options?.transform && !removeTags.includes("p") && PLAIN_PARAGRAPH.test(markdown)) return `\n<p>${markdown}</p>\n`;
20912
+ const allowTags = options?.allowTags ?? [];
20913
+ const tagNames = [...defaultSchema.tagNames ?? [], ...allowTags].filter((t) => !removeTags.includes(t));
20914
+ if (options?.transform) return createProcessor(tagNames, options.transform, options.transformType).processSync(markdown).toString();
20915
+ const key = [
20916
+ [...removeTags].sort().join(","),
20917
+ [...allowTags].sort().join(","),
20918
+ options?.transformType ?? ""
20919
+ ].join("|");
20920
+ let processor = processorCache.get(key);
20921
+ if (!processor) {
20922
+ processor = createProcessor(tagNames, void 0, options?.transformType).freeze();
20923
+ processorCache.set(key, processor);
20924
+ }
20925
+ return processor.processSync(markdown).toString();
20861
20926
  }
20862
20927
  /**
20863
20928
  * Create a Markdown AST from a string.