@flamingo-stack/openframe-frontend-core 0.0.488 → 0.0.490-1506.3944.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (311) hide show
  1. package/dist/chat-protocol/decode.d.ts +58 -0
  2. package/dist/chat-protocol/decode.d.ts.map +1 -0
  3. package/dist/chat-protocol/encode.d.ts +44 -0
  4. package/dist/chat-protocol/encode.d.ts.map +1 -0
  5. package/dist/chat-protocol/env-flag.d.ts +35 -0
  6. package/dist/chat-protocol/env-flag.d.ts.map +1 -0
  7. package/dist/chat-protocol/events.d.ts +196 -0
  8. package/dist/chat-protocol/events.d.ts.map +1 -0
  9. package/dist/chat-protocol/frames.d.ts +213 -0
  10. package/dist/chat-protocol/frames.d.ts.map +1 -0
  11. package/dist/chat-protocol/index.cjs +494 -0
  12. package/dist/chat-protocol/index.cjs.map +1 -0
  13. package/dist/chat-protocol/index.d.ts +18 -0
  14. package/dist/chat-protocol/index.d.ts.map +1 -0
  15. package/dist/chat-protocol/index.js +478 -0
  16. package/dist/chat-protocol/index.js.map +1 -0
  17. package/dist/chat-protocol/ip-normalize.d.ts +44 -0
  18. package/dist/chat-protocol/ip-normalize.d.ts.map +1 -0
  19. package/dist/chat-protocol/nats-decoder.d.ts +26 -0
  20. package/dist/chat-protocol/nats-decoder.d.ts.map +1 -0
  21. package/dist/{chunk-AJNDDXKC.js → chunk-4NWARV5I.js} +3 -3
  22. package/dist/{chunk-GQX6LFFA.cjs → chunk-5CJR2GZ4.cjs} +42 -42
  23. package/dist/{chunk-GQX6LFFA.cjs.map → chunk-5CJR2GZ4.cjs.map} +1 -1
  24. package/dist/{chunk-EUYIFALO.cjs → chunk-5EYRXSY5.cjs} +14 -14
  25. package/dist/{chunk-EUYIFALO.cjs.map → chunk-5EYRXSY5.cjs.map} +1 -1
  26. package/dist/{chunk-5RXKOGH6.cjs → chunk-5LMGFI4J.cjs} +7011 -4777
  27. package/dist/chunk-5LMGFI4J.cjs.map +1 -0
  28. package/dist/{chunk-KFQXO7QB.js → chunk-6GT557KN.js} +1 -1
  29. package/dist/chunk-6GT557KN.js.map +1 -0
  30. package/dist/{chunk-77DFPD35.js → chunk-BE6DK7QY.js} +12267 -10033
  31. package/dist/chunk-BE6DK7QY.js.map +1 -0
  32. package/dist/{chunk-A7IPON2V.cjs → chunk-BIKOJ4FH.cjs} +5 -5
  33. package/dist/{chunk-A7IPON2V.cjs.map → chunk-BIKOJ4FH.cjs.map} +1 -1
  34. package/dist/{chunk-HQYJINFV.js → chunk-CORIBKHG.js} +4 -4
  35. package/dist/{chunk-YEEHB4T6.cjs → chunk-CSYSYJCF.cjs} +40 -40
  36. package/dist/{chunk-YEEHB4T6.cjs.map → chunk-CSYSYJCF.cjs.map} +1 -1
  37. package/dist/{chunk-SIKCXKGB.cjs → chunk-DSDZ5T4P.cjs} +7 -7
  38. package/dist/{chunk-SIKCXKGB.cjs.map → chunk-DSDZ5T4P.cjs.map} +1 -1
  39. package/dist/{chunk-PNXNWGBY.js → chunk-GTWO5NAY.js} +2 -2
  40. package/dist/{chunk-SCCTJ2UI.js → chunk-HSACATCD.js} +4 -4
  41. package/dist/{chunk-PECGZWQR.cjs → chunk-KP5KRRD6.cjs} +3 -3
  42. package/dist/{chunk-PECGZWQR.cjs.map → chunk-KP5KRRD6.cjs.map} +1 -1
  43. package/dist/{chunk-2B6773QM.cjs → chunk-LILQJ3UZ.cjs} +1 -1
  44. package/dist/chunk-LILQJ3UZ.cjs.map +1 -0
  45. package/dist/{chunk-RZSA6MZL.cjs → chunk-NWX7MMF5.cjs} +15 -15
  46. package/dist/{chunk-RZSA6MZL.cjs.map → chunk-NWX7MMF5.cjs.map} +1 -1
  47. package/dist/{chunk-V3CNNMY3.cjs → chunk-P4WTZU6L.cjs} +18 -18
  48. package/dist/{chunk-V3CNNMY3.cjs.map → chunk-P4WTZU6L.cjs.map} +1 -1
  49. package/dist/{chunk-KLYHZRAD.js → chunk-PB3BTZDU.js} +2 -2
  50. package/dist/{chunk-FZOBQG6M.cjs → chunk-Q6E7VJ4W.cjs} +37 -36
  51. package/dist/chunk-Q6E7VJ4W.cjs.map +1 -0
  52. package/dist/{chunk-EYEIVYYF.cjs → chunk-QSCHCZL3.cjs} +7 -7
  53. package/dist/{chunk-EYEIVYYF.cjs.map → chunk-QSCHCZL3.cjs.map} +1 -1
  54. package/dist/{chunk-YJSKQ2ZV.js → chunk-RQLBHKXN.js} +3 -3
  55. package/dist/{chunk-CWQACI4Z.js → chunk-RW2LPL72.js} +2 -2
  56. package/dist/{chunk-DX6AOHNU.cjs → chunk-S2NHJDKQ.cjs} +77 -77
  57. package/dist/{chunk-DX6AOHNU.cjs.map → chunk-S2NHJDKQ.cjs.map} +1 -1
  58. package/dist/{chunk-AHDJMEEN.js → chunk-T2EGSZFN.js} +9 -8
  59. package/dist/chunk-T2EGSZFN.js.map +1 -0
  60. package/dist/{chunk-S52LJNMU.js → chunk-TKN5667N.js} +13 -13
  61. package/dist/chunk-TKN5667N.js.map +1 -0
  62. package/dist/{chunk-3BFX6HJM.js → chunk-UUYRJPGJ.js} +6 -6
  63. package/dist/{chunk-F6BYUO5U.js → chunk-VHG6AX5V.js} +3 -3
  64. package/dist/{chunk-FOQV6RCO.cjs → chunk-VQ523ORK.cjs} +29 -29
  65. package/dist/{chunk-FOQV6RCO.cjs.map → chunk-VQ523ORK.cjs.map} +1 -1
  66. package/dist/{chunk-PXF5J24T.js → chunk-XQ2B74SI.js} +2 -2
  67. package/dist/{chunk-H7JX2BG7.cjs → chunk-YHOIAA2E.cjs} +76 -76
  68. package/dist/chunk-YHOIAA2E.cjs.map +1 -0
  69. package/dist/{chunk-LLWBPPZ4.js → chunk-YZ2OIYQ3.js} +6 -6
  70. package/dist/components/case-studies/index.cjs +10 -10
  71. package/dist/components/case-studies/index.js +4 -4
  72. package/dist/components/chat/chat-message-enhanced.d.ts.map +1 -1
  73. package/dist/components/chat/chat-message-list.d.ts.map +1 -1
  74. package/dist/components/chat/hooks/index.d.ts +0 -1
  75. package/dist/components/chat/hooks/index.d.ts.map +1 -1
  76. package/dist/components/chat/hooks/use-chat-history-hydration.d.ts +47 -0
  77. package/dist/components/chat/hooks/use-chat-history-hydration.d.ts.map +1 -0
  78. package/dist/components/chat/hooks/use-chat.d.ts +1 -0
  79. package/dist/components/chat/hooks/use-chat.d.ts.map +1 -1
  80. package/dist/components/chat/hooks/use-nats-chat-adapter.d.ts +15 -48
  81. package/dist/components/chat/hooks/use-nats-chat-adapter.d.ts.map +1 -1
  82. package/dist/components/chat/hooks/use-sse-chat-adapter.d.ts +10 -53
  83. package/dist/components/chat/hooks/use-sse-chat-adapter.d.ts.map +1 -1
  84. package/dist/components/chat/index.cjs +12 -4
  85. package/dist/components/chat/index.cjs.map +1 -1
  86. package/dist/components/chat/index.d.ts +1 -0
  87. package/dist/components/chat/index.d.ts.map +1 -1
  88. package/dist/components/chat/index.js +23 -15
  89. package/dist/components/chat/stream/chat-dialog-store.d.ts +184 -0
  90. package/dist/components/chat/stream/chat-dialog-store.d.ts.map +1 -0
  91. package/dist/components/chat/stream/chat-stream-reducer.d.ts +367 -0
  92. package/dist/components/chat/stream/chat-stream-reducer.d.ts.map +1 -0
  93. package/dist/components/chat/stream/delta-batcher.d.ts +51 -0
  94. package/dist/components/chat/stream/delta-batcher.d.ts.map +1 -0
  95. package/dist/components/chat/stream/index.d.ts +17 -0
  96. package/dist/components/chat/stream/index.d.ts.map +1 -0
  97. package/dist/components/chat/stream/message-mutations.d.ts +77 -0
  98. package/dist/components/chat/stream/message-mutations.d.ts.map +1 -0
  99. package/dist/components/chat/stream/use-chat-stream-reducer.d.ts +22 -0
  100. package/dist/components/chat/stream/use-chat-stream-reducer.d.ts.map +1 -0
  101. package/dist/components/chat/types/api.types.d.ts +1 -81
  102. package/dist/components/chat/types/api.types.d.ts.map +1 -1
  103. package/dist/components/chat/types/processing.types.d.ts +2 -90
  104. package/dist/components/chat/types/processing.types.d.ts.map +1 -1
  105. package/dist/components/chat/types/unified-chat-state.types.d.ts +5 -0
  106. package/dist/components/chat/types/unified-chat-state.types.d.ts.map +1 -1
  107. package/dist/components/chat/utils/chat-conversation-storage.d.ts +34 -0
  108. package/dist/components/chat/utils/chat-conversation-storage.d.ts.map +1 -0
  109. package/dist/components/chat/utils/extract-incomplete-message-state.d.ts +34 -6
  110. package/dist/components/chat/utils/extract-incomplete-message-state.d.ts.map +1 -1
  111. package/dist/components/chat/utils/index.d.ts +2 -3
  112. package/dist/components/chat/utils/index.d.ts.map +1 -1
  113. package/dist/components/chat/utils/process-historical-messages.d.ts +59 -11
  114. package/dist/components/chat/utils/process-historical-messages.d.ts.map +1 -1
  115. package/dist/components/contact/index.cjs +5 -5
  116. package/dist/components/contact/index.js +4 -4
  117. package/dist/components/docs/index.cjs +7 -7
  118. package/dist/components/docs/index.js +6 -6
  119. package/dist/components/embeds/index.cjs +5 -5
  120. package/dist/components/embeds/index.js +4 -4
  121. package/dist/components/faq/index.cjs +6 -6
  122. package/dist/components/faq/index.js +5 -5
  123. package/dist/components/features/index.cjs +4 -4
  124. package/dist/components/features/index.js +3 -3
  125. package/dist/components/help-center-pages/index.cjs +23 -23
  126. package/dist/components/help-center-pages/index.js +14 -14
  127. package/dist/components/index.cjs +182 -140
  128. package/dist/components/index.cjs.map +1 -1
  129. package/dist/components/index.js +66 -24
  130. package/dist/components/index.js.map +1 -1
  131. package/dist/components/navigation/index.cjs +4 -4
  132. package/dist/components/navigation/index.js +3 -3
  133. package/dist/components/onboarding-guides/index.cjs +8 -8
  134. package/dist/components/onboarding-guides/index.js +7 -7
  135. package/dist/components/onboarding-guides/onboarding-guide-detail-view.d.ts.map +1 -1
  136. package/dist/components/related-content/index.cjs +6 -6
  137. package/dist/components/related-content/index.js +5 -5
  138. package/dist/components/shared/product-release/release-detail-page.d.ts.map +1 -1
  139. package/dist/components/tickets/index.cjs +7 -7
  140. package/dist/components/tickets/index.js +6 -6
  141. package/dist/components/ui/index.cjs +46 -4
  142. package/dist/components/ui/index.cjs.map +1 -1
  143. package/dist/components/ui/index.d.ts +1 -2
  144. package/dist/components/ui/index.d.ts.map +1 -1
  145. package/dist/components/ui/index.js +57 -15
  146. package/dist/components/ui/markdown/base-components.d.ts +51 -0
  147. package/dist/components/ui/markdown/base-components.d.ts.map +1 -0
  148. package/dist/components/ui/markdown/engine.d.ts +99 -0
  149. package/dist/components/ui/markdown/engine.d.ts.map +1 -0
  150. package/dist/components/ui/markdown/heading-ids.d.ts +65 -0
  151. package/dist/components/ui/markdown/heading-ids.d.ts.map +1 -0
  152. package/dist/components/ui/markdown/index.d.ts +18 -0
  153. package/dist/components/ui/markdown/index.d.ts.map +1 -0
  154. package/dist/components/ui/markdown/mermaid-diagram.d.ts +79 -0
  155. package/dist/components/ui/markdown/mermaid-diagram.d.ts.map +1 -0
  156. package/dist/components/ui/markdown/rich/embed-overrides.d.ts +14 -0
  157. package/dist/components/ui/markdown/rich/embed-overrides.d.ts.map +1 -0
  158. package/dist/components/ui/markdown/rich/rich-markdown-renderer.d.ts +40 -0
  159. package/dist/components/ui/markdown/rich/rich-markdown-renderer.d.ts.map +1 -0
  160. package/dist/components/ui/markdown/rich/shortcodes.d.ts +8 -0
  161. package/dist/components/ui/markdown/rich/shortcodes.d.ts.map +1 -0
  162. package/dist/components/ui/markdown/sanitize.d.ts +153 -0
  163. package/dist/components/ui/markdown/sanitize.d.ts.map +1 -0
  164. package/dist/components/ui/markdown/simple-markdown-renderer.d.ts +23 -0
  165. package/dist/components/ui/markdown/simple-markdown-renderer.d.ts.map +1 -0
  166. package/dist/components/ui/markdown/streaming.d.ts +78 -0
  167. package/dist/components/ui/markdown/streaming.d.ts.map +1 -0
  168. package/dist/components/ui/markdown/text-size.d.ts +34 -0
  169. package/dist/components/ui/markdown/text-size.d.ts.map +1 -0
  170. package/dist/contexts/chat-runtime-context.d.ts +8 -0
  171. package/dist/contexts/chat-runtime-context.d.ts.map +1 -1
  172. package/dist/contexts/index.cjs +2 -2
  173. package/dist/contexts/index.js +1 -1
  174. package/dist/hooks/index.cjs +3 -3
  175. package/dist/hooks/index.js +2 -2
  176. package/dist/index.cjs +62 -4
  177. package/dist/index.cjs.map +1 -1
  178. package/dist/index.js +73 -15
  179. package/dist/utils/index.cjs +169 -40
  180. package/dist/utils/index.cjs.map +1 -1
  181. package/dist/utils/index.d.ts +1 -0
  182. package/dist/utils/index.d.ts.map +1 -1
  183. package/dist/utils/index.js +162 -41
  184. package/dist/utils/index.js.map +1 -1
  185. package/dist/utils/markdown-fences.d.ts +42 -0
  186. package/dist/utils/markdown-fences.d.ts.map +1 -0
  187. package/dist/utils/markdown-heading-id.d.ts +85 -0
  188. package/dist/utils/markdown-heading-id.d.ts.map +1 -0
  189. package/dist/utils/markdown-section-extractor.d.ts.map +1 -1
  190. package/package.json +7 -1
  191. package/src/chat-protocol/__tests__/__snapshots__/nats-decoder-golden.test.ts.snap +307 -0
  192. package/src/chat-protocol/__tests__/chat-protocol.test.ts +516 -0
  193. package/src/chat-protocol/__tests__/env-flag.test.ts +52 -0
  194. package/src/chat-protocol/__tests__/ip-normalize.test.ts +136 -0
  195. package/src/chat-protocol/__tests__/nats-decoder-golden.test.ts +228 -0
  196. package/src/chat-protocol/decode.ts +278 -0
  197. package/src/chat-protocol/encode.ts +71 -0
  198. package/src/chat-protocol/env-flag.ts +39 -0
  199. package/src/chat-protocol/events.ts +232 -0
  200. package/src/chat-protocol/frames.ts +245 -0
  201. package/src/chat-protocol/index.ts +21 -0
  202. package/src/chat-protocol/ip-normalize.ts +146 -0
  203. package/src/chat-protocol/nats-decoder.ts +252 -0
  204. package/src/components/chat/__tests__/chat-message-list.test.tsx +293 -5
  205. package/src/components/chat/__tests__/chat-message-streaming-memo.test.tsx +207 -0
  206. package/src/components/chat/__tests__/chat-pending-turn.test.tsx +209 -0
  207. package/src/components/chat/chat-message-enhanced.tsx +111 -17
  208. package/src/components/chat/chat-message-list.tsx +358 -109
  209. package/src/components/chat/embeddable-chat.tsx +6 -0
  210. package/src/components/chat/hooks/.index.md +30 -33
  211. package/src/components/chat/hooks/.use-nats-chat-adapter.md +36 -55
  212. package/src/components/chat/hooks/__tests__/__snapshots__/conversation-id-persistence-golden.test.ts.snap +37 -0
  213. package/src/components/chat/hooks/__tests__/__snapshots__/sse-stream-golden.test.ts.snap +0 -0
  214. package/src/components/chat/hooks/__tests__/conversation-id-persistence-golden.test.ts +438 -0
  215. package/src/components/chat/hooks/__tests__/sse-stream-golden.test.ts +318 -0
  216. package/src/components/chat/hooks/index.ts +0 -1
  217. package/src/components/chat/hooks/use-chat-history-hydration.ts +142 -0
  218. package/src/components/chat/hooks/use-chat.ts +20 -0
  219. package/src/components/chat/hooks/use-nats-chat-adapter.ts +243 -823
  220. package/src/components/chat/hooks/use-sse-chat-adapter.ts +417 -774
  221. package/src/components/chat/index.ts +5 -0
  222. package/src/components/chat/stream/__tests__/__snapshots__/chat-stream-reducer-golden.test.ts.snap +940 -0
  223. package/src/components/chat/stream/__tests__/chat-dialog-store.test.ts +756 -0
  224. package/src/components/chat/stream/__tests__/chat-stream-reducer-golden.test.ts +460 -0
  225. package/src/components/chat/stream/__tests__/chat-stream-reducer.test.ts +894 -0
  226. package/src/components/chat/stream/__tests__/delta-batcher.test.ts +255 -0
  227. package/src/components/chat/stream/__tests__/use-chat-stream-reducer.test.ts +98 -0
  228. package/src/components/chat/stream/chat-dialog-store.ts +536 -0
  229. package/src/components/chat/stream/chat-stream-reducer.ts +1796 -0
  230. package/src/components/chat/stream/delta-batcher.ts +159 -0
  231. package/src/components/chat/stream/index.ts +55 -0
  232. package/src/components/chat/stream/message-mutations.ts +338 -0
  233. package/src/components/chat/stream/use-chat-stream-reducer.ts +126 -0
  234. package/src/components/chat/thinking-display.tsx +1 -1
  235. package/src/components/chat/types/.api.types.md +39 -49
  236. package/src/components/chat/types/.processing.types.md +29 -58
  237. package/src/components/chat/types/api.types.ts +6 -73
  238. package/src/components/chat/types/processing.types.ts +11 -52
  239. package/src/components/chat/types/unified-chat-state.types.ts +6 -0
  240. package/src/components/chat/utils/.index.md +31 -40
  241. package/src/components/chat/utils/__tests__/__snapshots__/process-historical-messages-golden.test.ts.snap +420 -0
  242. package/src/components/chat/utils/__tests__/__snapshots__/segment-accumulator-golden.test.ts.snap +605 -0
  243. package/src/components/chat/utils/__tests__/process-historical-messages-golden.test.ts +317 -0
  244. package/src/components/chat/utils/__tests__/segment-accumulator-golden.test.ts +270 -0
  245. package/src/components/chat/utils/chat-conversation-storage.ts +83 -0
  246. package/src/components/chat/utils/extract-incomplete-message-state.ts +63 -5
  247. package/src/components/chat/utils/index.ts +5 -8
  248. package/src/components/chat/utils/process-historical-messages.ts +352 -384
  249. package/src/components/onboarding-guides/onboarding-guide-detail-view.tsx +22 -4
  250. package/src/components/shared/legal-document/legal-document-page.tsx +1 -1
  251. package/src/components/shared/product-release/release-detail-page.tsx +19 -4
  252. package/src/components/ui/__tests__/__snapshots__/markdown-parity.test.tsx.snap +2420 -0
  253. package/src/components/ui/__tests__/markdown-parity.test.tsx +2528 -0
  254. package/src/components/ui/index.ts +1 -2
  255. package/src/components/ui/markdown/__tests__/mermaid-security.test.ts +115 -0
  256. package/src/components/ui/markdown/__tests__/mermaid-stale-render.test.tsx +95 -0
  257. package/src/components/ui/markdown/__tests__/sanitize-invariant.test.ts +309 -0
  258. package/src/components/ui/markdown/__tests__/streaming.test.tsx +388 -0
  259. package/src/components/ui/markdown/base-components.tsx +360 -0
  260. package/src/components/ui/markdown/engine.tsx +315 -0
  261. package/src/components/ui/markdown/heading-ids.ts +239 -0
  262. package/src/components/ui/markdown/index.ts +49 -0
  263. package/src/components/ui/markdown/mermaid-diagram.tsx +291 -0
  264. package/src/components/ui/markdown/rich/embed-overrides.tsx +199 -0
  265. package/src/components/ui/markdown/rich/rich-markdown-renderer.tsx +184 -0
  266. package/src/components/ui/markdown/rich/shortcodes.ts +170 -0
  267. package/src/components/ui/markdown/sanitize.ts +3499 -0
  268. package/src/components/ui/markdown/simple-markdown-renderer.tsx +28 -0
  269. package/src/components/ui/markdown/streaming.ts +362 -0
  270. package/src/components/ui/markdown/text-size.ts +106 -0
  271. package/src/components/ui/release-changelog-section.tsx +1 -1
  272. package/src/components/ui/ticket-info-section.tsx +1 -1
  273. package/src/contexts/chat-runtime-context.tsx +8 -0
  274. package/src/utils/index.ts +1 -0
  275. package/src/utils/markdown-fences.ts +127 -0
  276. package/src/utils/markdown-heading-id.ts +348 -0
  277. package/src/utils/markdown-section-extractor.ts +41 -56
  278. package/dist/chunk-2B6773QM.cjs.map +0 -1
  279. package/dist/chunk-5RXKOGH6.cjs.map +0 -1
  280. package/dist/chunk-77DFPD35.js.map +0 -1
  281. package/dist/chunk-AHDJMEEN.js.map +0 -1
  282. package/dist/chunk-FZOBQG6M.cjs.map +0 -1
  283. package/dist/chunk-H7JX2BG7.cjs.map +0 -1
  284. package/dist/chunk-KFQXO7QB.js.map +0 -1
  285. package/dist/chunk-S52LJNMU.js.map +0 -1
  286. package/dist/components/chat/hooks/use-realtime-chunk-processor.d.ts +0 -6
  287. package/dist/components/chat/hooks/use-realtime-chunk-processor.d.ts.map +0 -1
  288. package/dist/components/chat/utils/chunk-parser.d.ts +0 -25
  289. package/dist/components/chat/utils/chunk-parser.d.ts.map +0 -1
  290. package/dist/components/ui/rich-markdown-renderer.d.ts +0 -34
  291. package/dist/components/ui/rich-markdown-renderer.d.ts.map +0 -1
  292. package/dist/components/ui/simple-markdown-renderer.d.ts +0 -73
  293. package/dist/components/ui/simple-markdown-renderer.d.ts.map +0 -1
  294. package/src/components/chat/hooks/.use-realtime-chunk-processor.md +0 -77
  295. package/src/components/chat/hooks/use-realtime-chunk-processor.ts +0 -479
  296. package/src/components/chat/utils/.chunk-parser.md +0 -62
  297. package/src/components/chat/utils/chunk-parser.ts +0 -256
  298. package/src/components/ui/.simple-markdown-renderer.md +0 -52
  299. package/src/components/ui/rich-markdown-renderer.tsx +0 -1223
  300. package/src/components/ui/simple-markdown-renderer.tsx +0 -964
  301. /package/dist/{chunk-AJNDDXKC.js.map → chunk-4NWARV5I.js.map} +0 -0
  302. /package/dist/{chunk-HQYJINFV.js.map → chunk-CORIBKHG.js.map} +0 -0
  303. /package/dist/{chunk-PNXNWGBY.js.map → chunk-GTWO5NAY.js.map} +0 -0
  304. /package/dist/{chunk-SCCTJ2UI.js.map → chunk-HSACATCD.js.map} +0 -0
  305. /package/dist/{chunk-KLYHZRAD.js.map → chunk-PB3BTZDU.js.map} +0 -0
  306. /package/dist/{chunk-YJSKQ2ZV.js.map → chunk-RQLBHKXN.js.map} +0 -0
  307. /package/dist/{chunk-CWQACI4Z.js.map → chunk-RW2LPL72.js.map} +0 -0
  308. /package/dist/{chunk-3BFX6HJM.js.map → chunk-UUYRJPGJ.js.map} +0 -0
  309. /package/dist/{chunk-F6BYUO5U.js.map → chunk-VHG6AX5V.js.map} +0 -0
  310. /package/dist/{chunk-PXF5J24T.js.map → chunk-XQ2B74SI.js.map} +0 -0
  311. /package/dist/{chunk-LLWBPPZ4.js.map → chunk-YZ2OIYQ3.js.map} +0 -0
@@ -0,0 +1,3499 @@
1
+ /**
2
+ * Sanitization SSOT for the unified markdown engine.
3
+ *
4
+ * Layered defense (order matters, see engine.tsx):
5
+ * 1. `escapeUnknownHtmlTags` — TEXT pre-pass. Escapes `<tag>`s outside the
6
+ * effective allowlist so LLM-emitted pseudo-tags (`<their>`, `<ticket>`)
7
+ * never reach React as unknown elements (React 19 crash guard).
8
+ * NOT a security boundary.
9
+ * 2. `rehype-raw` parses remaining raw HTML into HAST.
10
+ * 3. `rehypeSanitize` with `buildSanitizeSchema(...)` — the audited
11
+ * allow-list boundary (hast-util-sanitize) with a schema extended to
12
+ * exactly what our surfaces need.
13
+ * 4. `rehypeStripUnsafe` — custom strip pass kept as defense-in-depth
14
+ * (srcset candidate scanning, iframe[srcdoc], belt-and-suspenders if
15
+ * the schema is ever loosened).
16
+ *
17
+ * COUPLED-ALLOWLIST INVARIANT (tested in __tests__/sanitize-invariant.test.ts):
18
+ * the two effective tag lists are EQUAL (case-insensitively), both computed
19
+ * AFTER merging `extraAllowedHtmlTags`. Both directions matter:
20
+ * - pre-pass ⊆ sanitizer: the pre-pass must never admit a raw tag the
21
+ * sanitizer then silently drops.
22
+ * - sanitizer ⊆ pre-pass: the pre-pass must never ESCAPE a tag the
23
+ * sanitizer would happily keep. This direction was broken before
24
+ * 2026-07: `strike` (and every other `defaultSchema`-only tag) survived
25
+ * the sanitizer but was escaped to `&lt;strike&gt;` source text by the
26
+ * pre-pass, so legacy authored markup regressed to visible tag soup.
27
+ * Both lists are now derived from the SINGLE `effectiveTagList()` below —
28
+ * never fork them.
29
+ *
30
+ * ONE documented exception, and it is CONTENT-dependent rather than
31
+ * list-level (so the invariant test still holds as an equality of tag SETS):
32
+ * an UNCLOSED RAWTEXT/RCDATA opener (`<textarea>`, `<iframe>`, `<title>`, …)
33
+ * is escaped by the pre-pass even though the sanitizer allowlists it —
34
+ * because parse5's tokenizer would otherwise swallow the remainder of the
35
+ * document into it before the sanitizer ever runs. See RAWTEXT_TAGS below.
36
+ */
37
+ import { defaultSchema } from 'rehype-sanitize'
38
+ import { visit } from 'unist-util-visit'
39
+ import { defaultUrlTransform } from 'react-markdown'
40
+ import { createFenceTracker, isBlankLine } from '../../../utils/markdown-fences'
41
+
42
+ // ---------------------------------------------------------------------------
43
+ // Shared tag allowlist (pre-pass baseline)
44
+ // ---------------------------------------------------------------------------
45
+ /**
46
+ * Tags the TEXT pre-pass forwards as raw HTML. Anything outside this set
47
+ * (plus per-composition `extraAllowedHtmlTags`) gets its angle brackets
48
+ * escaped and renders as plain text.
49
+ *
50
+ * `video` is deliberately NOT in the baseline: chat strips <video>
51
+ * server-side and playback goes through the <Video> SSOT. The rich
52
+ * composition opts back in via `extraAllowedHtmlTags={['video', 'source']}`
53
+ * so authored content (blog publisher video injection) keeps working.
54
+ */
55
+ export const SAFE_HTML_TAGS = new Set([
56
+ // Block + inline text
57
+ 'a', 'abbr', 'address', 'article', 'aside', 'b', 'bdi', 'bdo', 'blockquote',
58
+ 'br', 'caption', 'cite', 'code', 'col', 'colgroup', 'data', 'dd', 'del',
59
+ 'details', 'dfn', 'div', 'dl', 'dt', 'em', 'figcaption', 'figure', 'footer',
60
+ 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'i', 'ins',
61
+ 'kbd', 'li', 'main', 'mark', 'nav', 'ol', 'p', 'pre', 'q', 'rp', 'rt',
62
+ 'ruby', 's', 'samp', 'section', 'small', 'span', 'strong', 'sub', 'summary',
63
+ 'sup', 'table', 'tbody', 'td', 'tfoot', 'th', 'thead', 'time', 'tr', 'u',
64
+ 'ul', 'var', 'wbr',
65
+ // Deprecated presentational tags that REAL authored content still carries.
66
+ // They rendered before the unification (neither old renderer had a
67
+ // pre-pass), so escaping them to visible `&lt;center&gt;` source text was a
68
+ // regression. `font` gets its legacy attributes below so the sanitizer
69
+ // doesn't reduce it to a bare no-op tag. (`marquee` stays out — it is
70
+ // animated chrome, not text markup, and no audit hit found it.)
71
+ 'center', 'font', 'big',
72
+ // Media ('video' intentionally excluded — see the header comment)
73
+ 'img', 'picture', 'source', 'audio', 'iframe', 'track',
74
+ // Forms (rehype-raw allows them; mostly harmless for chat output)
75
+ 'button', 'input', 'label', 'select', 'option', 'optgroup', 'textarea', 'form', 'fieldset', 'legend',
76
+ ])
77
+
78
+ /**
79
+ * Inline SVG element set, in the CANONICAL case parse5 produces for SVG
80
+ * foreign content (`linearGradient`, `clipPath`, … are camelCase in the
81
+ * HTML parser's SVG adjustment table, so the sanitize schema must match
82
+ * that spelling; the text pre-pass lowercases before lookup).
83
+ *
84
+ * Inline `<svg>` renders in real published posts (hand-authored diagrams
85
+ * and inline icon markup — NOT `<use href>` sprite references, which are
86
+ * deliberately dropped; see SVG_ATTRIBUTES). The Rich renderer had NO pre-pass
87
+ * and NO sanitizer, so it always rendered; without this set the unified
88
+ * engine would escape it to visible source text.
89
+ *
90
+ * SEVERAL OF THESE NAMES ARE ALSO HTML ELEMENTS (`title`, `desc`, `text`,
91
+ * `g`, `line`, `use`, `symbol`, `marker`, `mask`, `pattern`) — admitting
92
+ * them UNCONSTRAINED let a post or a chat message emit a bare `<title>`,
93
+ * which React 19 hoists into `<head>` (browser-tab + SEO title hijack) and
94
+ * whose RAWTEXT content model swallows the rest of the document when
95
+ * unclosed. They are therefore pinned to an `svg` ancestor in
96
+ * `SVG_ONLY_ANCESTORS` below; outside `<svg>` the sanitizer drops them.
97
+ */
98
+ export const SVG_TAGS = new Set([
99
+ 'svg', 'path', 'circle', 'ellipse', 'g', 'rect', 'line', 'polyline',
100
+ 'polygon', 'text', 'tspan', 'defs', 'use', 'symbol', 'title', 'desc',
101
+ 'marker', 'mask', 'pattern', 'linearGradient', 'radialGradient', 'stop',
102
+ 'clipPath',
103
+ ])
104
+
105
+ /**
106
+ * Required-ancestor constraints for the SVG-only tags (hast-util-sanitize
107
+ * `ancestors`: a listed tag survives ONLY inside one of its ancestors).
108
+ *
109
+ * The TEXT pre-pass may still forward these — it is a flat regex over source
110
+ * text, cannot see nesting, and is explicitly NOT a security boundary. The
111
+ * coupled-allowlist invariant still holds because `ancestors` RESTRICTS a
112
+ * tag the schema already lists; it never adds one the pre-pass would escape.
113
+ */
114
+ const SVG_ONLY_ANCESTORS: Record<string, string[]> = {
115
+ title: ['svg'],
116
+ desc: ['svg'],
117
+ text: ['svg'],
118
+ tspan: ['svg', 'text'],
119
+ use: ['svg'],
120
+ symbol: ['svg'],
121
+ marker: ['svg'],
122
+ mask: ['svg'],
123
+ pattern: ['svg'],
124
+ g: ['svg'],
125
+ line: ['svg'],
126
+ path: ['svg'],
127
+ circle: ['svg'],
128
+ ellipse: ['svg'],
129
+ rect: ['svg'],
130
+ polyline: ['svg'],
131
+ polygon: ['svg'],
132
+ defs: ['svg'],
133
+ stop: ['svg'],
134
+ linearGradient: ['svg'],
135
+ radialGradient: ['svg'],
136
+ clipPath: ['svg'],
137
+ }
138
+
139
+ /**
140
+ * THE effective tag list for a composition, canonical case — the single
141
+ * source both the pre-pass set and the sanitize schema derive from
142
+ * (coupled-allowlist invariant, both directions).
143
+ *
144
+ * `defaultSchema.tagNames` is unioned in so the pre-pass can never escape a
145
+ * tag hast-util-sanitize would keep (`strike`, `tt`, …).
146
+ */
147
+ function effectiveTagList(extraAllowedHtmlTags?: string[]): string[] {
148
+ return [
149
+ ...(defaultSchema.tagNames ?? []),
150
+ ...SAFE_HTML_TAGS,
151
+ ...SVG_TAGS,
152
+ ...(extraAllowedHtmlTags ?? []),
153
+ ]
154
+ }
155
+
156
+ /** Effective pre-pass tag set for a composition (lowercased for lookup). */
157
+ export function buildEffectiveTagSet(extraAllowedHtmlTags?: string[]): Set<string> {
158
+ return new Set(effectiveTagList(extraAllowedHtmlTags).map((t) => t.toLowerCase()))
159
+ }
160
+
161
+ // ---------------------------------------------------------------------------
162
+ // rehype-sanitize schema (allow-list boundary)
163
+ // ---------------------------------------------------------------------------
164
+ /**
165
+ * Per-tag attribute allowances layered on top of hast-util-sanitize's
166
+ * defaultSchema. Property names are hast camelCase. Attribute survival
167
+ * matters as much as tag survival — an attribute-stripped `<video>` is a
168
+ * sourceless player (see plan: "video-survives-sanitize fixture").
169
+ */
170
+ const EXTRA_ATTRIBUTES: Record<string, Array<string | [string, ...unknown[]]>> = {
171
+ // `style` is allowed on the tags the 2026-07 content-store audit found it
172
+ // on in REAL published posts (div.takeaway, table styling, reddit
173
+ // blockquotes). This matches pre-unification behavior on BOTH surfaces —
174
+ // neither old renderer stripped style — so it is parity, not loosening;
175
+ // the URL-scheme guards in rehypeStripUnsafe still apply to attributes.
176
+ '*': ['className', 'id', 'data*', 'dir', 'title', 'lang'],
177
+ a: ['target', 'rel', 'href'],
178
+ div: ['style'],
179
+ span: ['style'],
180
+ p: ['style'],
181
+ blockquote: ['style', 'cite'],
182
+ td: ['colSpan', 'rowSpan', 'align', 'style'],
183
+ th: ['colSpan', 'rowSpan', 'align', 'scope', 'style'],
184
+ img: ['src', 'srcSet', 'sizes', 'alt', 'width', 'height', 'loading', 'decoding'],
185
+ iframe: ['src', 'width', 'height', 'allow', 'allowFullScreen', 'frameBorder', 'loading', 'referrerPolicy', 'style'],
186
+ video: ['src', 'poster', 'controls', 'width', 'height', 'loop', 'muted', 'autoPlay', 'playsInline', 'preload'],
187
+ source: ['src', 'type', 'media', 'srcSet', 'sizes'],
188
+ audio: ['src', 'controls', 'loop', 'muted', 'preload'],
189
+ track: ['src', 'kind', 'srcLang', 'label', 'default'],
190
+ time: ['dateTime'],
191
+ details: ['open'],
192
+ // Form elements (allow the benign presentational subset).
193
+ //
194
+ // `input` carries EXACTLY the GFM task-list contract and nothing else.
195
+ // Dropping the attribute widening alone was not enough: defaultSchema
196
+ // pins `required.input = { type:'checkbox', disabled:true }`, and
197
+ // `required` force-ADDS those properties regardless of what the author
198
+ // wrote — so `<input type="text" placeholder="email">` still came out as
199
+ // a disabled checkbox. `buildSanitizeSchema` therefore clears
200
+ // `required.input` (remark-gfm emits `type="checkbox" disabled` on task
201
+ // items itself, so the coercion was redundant) and the contract is
202
+ // expressed here instead: type is pinned to the literal `checkbox`, so a
203
+ // text input degrades to a bare `<input>` rather than a fake checkbox.
204
+ input: [['type', 'checkbox'], 'checked', 'disabled'],
205
+ button: ['type', 'disabled', 'name', 'value'],
206
+ // Legacy presentational tag — without its own attributes the sanitizer
207
+ // would keep `<font>` but strip everything that makes it do anything.
208
+ font: ['color', 'size', 'face'],
209
+ select: ['disabled', 'multiple', 'name'],
210
+ option: ['value', 'selected', 'disabled'],
211
+ optgroup: ['label', 'disabled'],
212
+ textarea: ['rows', 'cols', 'placeholder', 'disabled', 'readOnly', 'name'],
213
+ label: ['htmlFor'],
214
+ col: ['span'],
215
+ colgroup: ['span'],
216
+ }
217
+
218
+ /**
219
+ * SVG presentation/geometry attributes, keyed the way hast keys them:
220
+ * property-information normalizes `font-size` → `fontSize`,
221
+ * `stroke-dasharray` → `strokeDasharray`, … BEFORE the sanitizer sees the
222
+ * tree, so ONLY the camelCase spellings are load-bearing. The dashed
223
+ * spellings previously listed alongside them were dead weight (they never
224
+ * matched anything) and are gone; do not re-add them.
225
+ *
226
+ * `style` is allowed here for parity with div/span/p (same 2026-07 audit
227
+ * rationale — authored SVG carries inline `style` and both pre-unification
228
+ * renderers kept it; the URL guards in rehypeStripUnsafe still apply).
229
+ */
230
+ const SVG_ATTRIBUTES = [
231
+ 'viewBox', 'xmlns', 'd', 'fill', 'stroke', 'cx', 'cy', 'r', 'rx', 'ry',
232
+ 'x', 'y', 'x1', 'y1', 'x2', 'y2', 'points', 'transform', 'opacity',
233
+ 'offset', 'width', 'height', 'style',
234
+ // NOTE the exact casing: property-information's SVG map uses
235
+ // `strokeDashArray` / `strokeDashOffset` / `strokeMiterLimit` (capital
236
+ // A/O/L), NOT the react-DOM spellings. A near-miss here fails SILENTLY —
237
+ // the attribute is simply stripped. Verify against
238
+ // node_modules/property-information/lib/svg.js before adding one.
239
+ 'strokeWidth', 'strokeDashArray', 'strokeDashOffset', 'strokeMiterLimit',
240
+ 'strokeOpacity', 'strokeLinecap', 'strokeLinejoin',
241
+ 'fillRule', 'fillOpacity',
242
+ 'stopColor', 'stopOpacity',
243
+ 'fontSize', 'fontFamily', 'fontWeight', 'fontStyle', 'fontStretch',
244
+ 'textAnchor', 'dominantBaseline', 'alignmentBaseline', 'letterSpacing',
245
+ 'dx', 'dy', 'markerEnd', 'markerMid', 'markerStart',
246
+ 'gradientUnits', 'gradientTransform', 'patternUnits', 'maskUnits',
247
+ 'preserveAspectRatio',
248
+ 'clipPath', 'clipRule',
249
+ // `href` / `xlink:href` stay DELIBERATELY DISALLOWED on SVG elements:
250
+ // `<use href>` pulls in an external document fragment and the hast key
251
+ // (`xlinkHref`) is outside rehypeStripUnsafe's URL_ATTRS check, so it
252
+ // would be an unguarded URL sink. Consequence, stated plainly: PASTED
253
+ // ICON SPRITES THAT RELY ON `<use href="#id">` RENDER EMPTY. Hand-drawn
254
+ // inline SVG (the audited real-content case) is unaffected.
255
+ ]
256
+
257
+ export interface BuildSanitizeSchemaOptions {
258
+ extraAllowedHtmlTags?: string[]
259
+ }
260
+
261
+ /**
262
+ * The engine's sanitize schema: defaultSchema ∪ SAFE_HTML_TAGS ∪ extras.
263
+ * - `extraAllowedHtmlTags` is unioned into tagNames here AND into the
264
+ * pre-pass set (buildEffectiveTagSet) — the coupled-allowlist invariant.
265
+ * - `clobberPrefix: ''` + empty `clobber`: authored raw-HTML anchors
266
+ * (`<h2 id="…">`) keep their ids so `[jump](#anchor)` deep-links work.
267
+ * (The renderer's own heading ids are injected at the React layer,
268
+ * post-rehype, and were never affected.)
269
+ * - `card`/`mention` protocols registered for href so chat markers survive
270
+ * (the urlTransform below is the second gate).
271
+ */
272
+ export function buildSanitizeSchema(options: BuildSanitizeSchemaOptions = {}) {
273
+ // Canonical spelling AND lowercase for every tag: parse5 emits SVG
274
+ // foreign-content tags camelCased (`linearGradient`), HTML tags
275
+ // lowercased — admitting both keeps the schema list a superset of the
276
+ // (lowercased) pre-pass set, so the two are equal case-insensitively.
277
+ const tagNames = new Set<string>()
278
+ for (const tag of effectiveTagList(options.extraAllowedHtmlTags)) {
279
+ tagNames.add(tag)
280
+ tagNames.add(tag.toLowerCase())
281
+ }
282
+
283
+ const attributes: Record<string, Array<string | [string, ...unknown[]]>> = {
284
+ ...(defaultSchema.attributes as Record<string, Array<string | [string, ...unknown[]]>>),
285
+ }
286
+ for (const [tag, attrs] of Object.entries(EXTRA_ATTRIBUTES)) {
287
+ attributes[tag] = [...(attributes[tag] ?? []), ...attrs]
288
+ }
289
+ for (const tag of SVG_TAGS) {
290
+ attributes[tag] = [...(attributes[tag] ?? []), ...SVG_ATTRIBUTES]
291
+ const lower = tag.toLowerCase()
292
+ if (lower !== tag) attributes[lower] = [...(attributes[lower] ?? []), ...SVG_ATTRIBUTES]
293
+ }
294
+
295
+ // SVG-only tags are pinned to an `svg` ancestor (canonical AND lowercase
296
+ // spelling, matching the tagNames treatment above) so a bare `<title>` /
297
+ // `<text>` / `<g>` in prose is DROPPED instead of hijacking the page.
298
+ const ancestors: Record<string, string[]> = {
299
+ ...(defaultSchema.ancestors as Record<string, string[]> | undefined),
300
+ }
301
+ for (const [tag, required] of Object.entries(SVG_ONLY_ANCESTORS)) {
302
+ ancestors[tag] = required
303
+ ancestors[tag.toLowerCase()] = required
304
+ }
305
+
306
+ // `required.input` is CLEARED — see the `input` note in EXTRA_ATTRIBUTES.
307
+ // defaultSchema force-adds `type="checkbox" disabled` to every `<input>`,
308
+ // rewriting authored text inputs into fake disabled checkboxes; remark-gfm
309
+ // already emits both properties on real task-list items, so nothing is
310
+ // lost. The attribute allowlist pins `type` to the literal `checkbox`.
311
+ const required: Record<string, Record<string, unknown>> = {
312
+ ...(defaultSchema.required as Record<string, Record<string, unknown>> | undefined),
313
+ }
314
+ delete required.input
315
+
316
+ return {
317
+ ...defaultSchema,
318
+ tagNames: [...tagNames],
319
+ attributes,
320
+ ancestors,
321
+ required,
322
+ clobberPrefix: '',
323
+ clobber: [],
324
+ protocols: {
325
+ ...defaultSchema.protocols,
326
+ href: [...(defaultSchema.protocols?.href ?? []), 'card', 'mention'],
327
+ },
328
+ }
329
+ }
330
+
331
+ // ---------------------------------------------------------------------------
332
+ // rehypeStripUnsafe — defense-in-depth strip pass (kept verbatim from the
333
+ // pre-unification SimpleMarkdownRenderer)
334
+ // ---------------------------------------------------------------------------
335
+ const EVENT_HANDLER_ATTR_RE = /^on[a-z]+$/i
336
+ const JAVASCRIPT_URL_RE = /^[\s\x00-\x1f]*javascript:/i
337
+ const DATA_URL_RE = /^[\s\x00-\x1f]*data:/i
338
+ const URL_ATTRS = new Set([
339
+ 'href',
340
+ 'src',
341
+ 'srcset',
342
+ 'formaction',
343
+ 'xlink:href',
344
+ 'poster',
345
+ 'data',
346
+ 'action',
347
+ 'background',
348
+ ])
349
+
350
+ /**
351
+ * Returns true if any candidate in an `srcset` attribute has a dangerous
352
+ * URL scheme. srcset is a comma-separated candidate list — a single-URL
353
+ * check would miss a malicious second candidate
354
+ * (`"https://safe.png 1x, javascript:alert(1) 2x"`). Over-splitting on
355
+ * commas inside URL paths over-strips, which is the correct error bias.
356
+ */
357
+ function srcsetHasUnsafeCandidate(srcset: string): boolean {
358
+ for (const candidate of srcset.split(',')) {
359
+ const url = candidate.trim().split(/\s+/)[0] ?? ''
360
+ if (JAVASCRIPT_URL_RE.test(url) || DATA_URL_RE.test(url)) return true
361
+ }
362
+ return false
363
+ }
364
+
365
+ const STRIP_ELEMENTS = new Set([
366
+ 'script',
367
+ 'style',
368
+ 'noscript',
369
+ 'noembed',
370
+ 'object',
371
+ 'embed',
372
+ 'applet',
373
+ 'base',
374
+ 'meta',
375
+ ])
376
+
377
+ export function rehypeStripUnsafe() {
378
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
379
+ return (tree: any) => {
380
+ visit(tree, 'element', (node: any, index: number | undefined, parent: any) => {
381
+ const tag = String(node.tagName ?? '').toLowerCase()
382
+ if (STRIP_ELEMENTS.has(tag)) {
383
+ if (parent && typeof index === 'number') {
384
+ parent.children.splice(index, 1)
385
+ // Return the numeric index so the walker resumes at the slot the
386
+ // removed node vacated.
387
+ return index
388
+ }
389
+ // Root-level strip element — neutralize in place.
390
+ node.children = []
391
+ node.tagName = 'span'
392
+ node.properties = {}
393
+ return
394
+ }
395
+ if (!node.properties || typeof node.properties !== 'object') return
396
+ for (const key of Object.keys(node.properties)) {
397
+ if (EVENT_HANDLER_ATTR_RE.test(key)) {
398
+ delete node.properties[key]
399
+ continue
400
+ }
401
+ if (URL_ATTRS.has(key.toLowerCase())) {
402
+ const raw = node.properties[key]
403
+ const v = Array.isArray(raw) ? raw[0] : raw
404
+ if (typeof v === 'string') {
405
+ const unsafe =
406
+ key.toLowerCase() === 'srcset'
407
+ ? srcsetHasUnsafeCandidate(v)
408
+ : JAVASCRIPT_URL_RE.test(v) || DATA_URL_RE.test(v)
409
+ if (unsafe) {
410
+ delete node.properties[key]
411
+ continue
412
+ }
413
+ }
414
+ }
415
+ if (tag === 'iframe' && key.toLowerCase() === 'srcdoc') {
416
+ delete node.properties[key]
417
+ }
418
+ }
419
+ })
420
+ }
421
+ }
422
+
423
+ // ---------------------------------------------------------------------------
424
+ // escapeUnknownHtmlTags — TEXT pre-pass (React 19 crash guard)
425
+ // ---------------------------------------------------------------------------
426
+ // ReDoS-safe shape (CodeQL polynomial-regex hardening): every quantifier is
427
+ // hard-bounded (tag name ≤63 chars, attrs ≤4096) so matching is
428
+ // constant-time per tag. Anything longer falls through as plain text —
429
+ // the safe-degrade behavior for HTML-in-markdown.
430
+ const TAG_LIKE_REGEX = /<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})((?:\s[^>]{0,4096}?)?)(\/?)>/g
431
+
432
+ /**
433
+ * Tags whose HTML content model is RAWTEXT / RCDATA / PLAINTEXT: once parse5
434
+ * sees the start tag, EVERYTHING up to the matching end tag (or, if there is
435
+ * none, to end of input) is consumed as that element's text — headings,
436
+ * paragraphs, list items and all.
437
+ *
438
+ * The tokenizer runs BEFORE the sanitizer, so an allowlist entry (or an
439
+ * `ancestors` pin, as `title` got in round 2) cannot undo the damage: by the
440
+ * time the schema is consulted, the rest of the message is already a single
441
+ * text node hanging off the wrong element. Observed with the unclosed forms:
442
+ * `<textarea>` → the remainder of the message becomes the editable value
443
+ * of a live textarea
444
+ * `<iframe>` → the remainder is swallowed into an `about:blank` frame
445
+ * `<title>` → the remainder de-structures (headings stop being headings)
446
+ * Any chat message or post that merely MENTIONS one of these in prose — an
447
+ * LLM explaining HTML forms will — mangles everything after it.
448
+ *
449
+ * The TEXT pre-pass is the layer built for exactly this: it runs before
450
+ * parse5 and is purely textual. An opening tag from this set is escaped
451
+ * unless its matching `</tag>` appears LATER in the source, in which case the
452
+ * RAWTEXT span is bounded and the element renders normally (see the
453
+ * `closed-*` fixtures). This check is deliberately independent of the
454
+ * allowlist — it constrains tags the sanitizer WOULD keep.
455
+ */
456
+ const RAWTEXT_TAGS = new Set([
457
+ 'title', 'textarea', 'iframe', 'xmp', 'noembed', 'noframes', 'plaintext',
458
+ ])
459
+
460
+ /**
461
+ * Fenced code blocks and inline code spans — the regions whose `<tags>` are
462
+ * literal content and must survive the escaping pass verbatim.
463
+ *
464
+ * Drives the escaping CARVE in `escapeUnknownHtmlTags`. It is deliberately
465
+ * NARROWER than what the MASK now understands: the mask is the security
466
+ * boundary (too-narrow ⇒ a live `<textarea>` swallows the document) while the
467
+ * carve is cosmetic (too-narrow ⇒ a code sample renders as escaped text), so
468
+ * they are allowed to differ — but only in that direction, and that is now
469
+ * ENFORCED rather than assumed: `escapeUnknownHtmlTags` protects only the
470
+ * INTERSECTION of carve and mask, so a span this regex over-detects is escaped
471
+ * instead of sheltered. See the CARVE DECISION note on `buildCloserHaystack`.
472
+ *
473
+ * Deliberately NARROWER than `createFenceTracker`'s CommonMark notion: this is
474
+ * a flat regex over source TEXT with no line-state, so it only recognizes a
475
+ * fence that is CLOSED by a same-marker run. That is the correct bias here —
476
+ * an unclosed fence leaves its body UNPROTECTED, so a `<textarea>` inside it
477
+ * gets escaped (visible as escaped text) rather than left live. Using the real
478
+ * tracker would mean re-deriving character offsets from line state for a pass
479
+ * that is explicitly not a security boundary; the narrow form fails safe.
480
+ * `~{3,}` and the CommonMark 0..3-space indent ARE handled (they were not
481
+ * before: a `~~~` block containing `<textarea>` rendered as escaped text).
482
+ *
483
+ * The MASK does NOT use this regex's fence alternative at all any more — see
484
+ * `buildCloserHaystack`, which derives its code regions from the real
485
+ * `createFenceTracker` (plus indented / blockquoted / commented code). Only the
486
+ * INLINE-CODE region is derived separately, by `findInlineCodeRanges` below —
487
+ * which is no longer a regex and no longer shares this one's length cap.
488
+ */
489
+ const PROTECTED_SPAN_RE =
490
+ /^ {0,3}(`{3,}|~{3,})[\s\S]*?^ {0,3}\1[^\n]*$|(`+)[^\n]{0,4096}?\2/gm
491
+
492
+ /**
493
+ * The INLINE-CODE half of `PROTECTED_SPAN_RE`, on its own — the mask's only
494
+ * non-line-state code region. Everything block-level (fences, indented code,
495
+ * blockquoted code, HTML comments) is derived from line state instead, because
496
+ * a flat regex cannot express CommonMark's closer rules: `PROTECTED_SPAN_RE`
497
+ * ends a fenced span at the FIRST same-marker run even when that run carries an
498
+ * info string (```` ```html ````), which CommonMark forbids on a closer — so the
499
+ * span ended early and the real code content was left unmasked.
500
+ *
501
+ * NO LENGTH CAP, AND NO REGEX (round 16 — SECURITY). This used to be
502
+ * `` /(`+)[^\n]{0,4096}?\1/g ``, sharing `PROTECTED_SPAN_RE`'s 4096-char
503
+ * ReDoS bound. An inline span LONGER than the cap matched NEITHER regex, so the
504
+ * mask simply skipped it — leaving a `</textarea>` written inside that span
505
+ * VISIBLE in the closer haystack, `hasLaterCloser` true, the prose opener LIVE,
506
+ * and parse5 swallowing the rest of the message as the textarea's value.
507
+ * A clean cliff, padding length the only variable: span content ≤4094 chars ⇒
508
+ * blanked, opener escaped, 0 live textareas; ≥4099 ⇒ closer visible, 1 live
509
+ * textarea. That is a fail-OPEN in the security boundary, and it contradicts
510
+ * this module's own contract that every mask approximation "rounds towards
511
+ * blanking".
512
+ *
513
+ * THE BOUND COULD NOT SIMPLY BE DROPPED. `[^\n]` confines backtracking to one
514
+ * LINE, but a single line is not a small input — a chat message can be one.
515
+ * MEASURED (round 16, this repo's vitest env, one line of nothing but
516
+ * backticks — the pathological shape; figures from plain node are within 3%):
517
+ *
518
+ * input capped regex uncapped regex this linear scan
519
+ * 50K chars 295 ms (5.9 µs/c) 615 ms (12.3 µs/c) 0.69 ms (14 ns/c)
520
+ * 200K chars 1220 ms (6.1 µs/c) 9772 ms (48.9 µs/c) 0.42 ms (2 ns/c)
521
+ * 800K chars 5072 ms (6.3 µs/c) 158263 ms (198 µs/c) — (node)
522
+ *
523
+ * The capped regex is flat per char (linear, huge constant); the UNCAPPED one
524
+ * is plainly QUADRATIC — 31x the capped cost at 800KB and still climbing. Every
525
+ * other shape probed (lone tick + text, `` `` `` + text, one tick per 32 chars,
526
+ * one tick per line) is ≈2-4 ns/char in BOTH regex spellings, so the blowup is
527
+ * specific to long backtick runs — which an attacker controls. The cap was load
528
+ * bearing; the REGEX is what had to go. A realistic 260KB backtick-dense
529
+ * message (`Use `foo` and `bar` here.` × 10000) scans in 4.2 ms.
530
+ *
531
+ * `findInlineCodeRanges` is a LINEAR index scan that reproduces the old
532
+ * regex's match semantics exactly (verified by differential fuzz against the
533
+ * uncapped regex) with no backtracking and no cap: per line it collects the
534
+ * backtick RUNS, then for each opener run of length `n` picks the largest
535
+ * closer length `k ≤ n` that occurs later on the line — either inside the same
536
+ * run (needs `n ≥ 2k`, mirroring the regex giving back backticks from a greedy
537
+ * `` (`+) ``) or at the earliest following run of length ≥ k — and takes the
538
+ * EARLIEST such position (the lazy quantifier). The forward walk is amortized
539
+ * O(1) per run because the scan cursor jumps past every run it skipped.
540
+ *
541
+ * `PROTECTED_SPAN_RE` (the CARVE) KEEPS its cap, deliberately: the two
542
+ * consumers round in OPPOSITE directions, see the note on
543
+ * `escapeLeftoverTagStarts`. In the carve, "not known to be code" means ESCAPE,
544
+ * so an over-cap span there costs a code sample rendered as escaped text.
545
+ */
546
+ const BACKTICK_CODE = 0x60
547
+
548
+ /**
549
+ * PARAGRAPH SEGMENTS, NOT LINES (round 18 — SECURITY). A CommonMark code span
550
+ * CROSSES LINE BREAKS: `` `foo\n</textarea>` `` is one `inlineCode` node, so
551
+ * that `</textarea>` is a code sample and not a closer — yet this scan (and
552
+ * `PROTECTED_SPAN_RE`, whose body class is `[^\n]`) was strictly PER LINE, so
553
+ * the mask never saw the span, the closer stayed visible in the haystack,
554
+ * `hasLaterCloser` returned true, and a prose `<textarea>` above it stayed LIVE
555
+ * (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL; the renderer
556
+ * emitted `<code>foo </textarea></code>` — proving the closer is a code sample
557
+ * — beside a live textarea swallowing the prose). This is the shape that is not
558
+ * a CONTAINER at all, so no container sweep could ever have reached it.
559
+ *
560
+ * A code span CANNOT cross a paragraph break, so the scan unit is a maximal run
561
+ * of non-blank lines. Blank lines still terminate a segment, which keeps the
562
+ * fail-CLOSED direction (an unterminated opener consumes at most its own
563
+ * paragraph, never the rest of the document) and keeps the bound linear — the
564
+ * `suffMax` / cursor structure is unchanged, `\n` is simply an ordinary
565
+ * character inside a segment.
566
+ *
567
+ * `PROTECTED_SPAN_RE` (the CARVE) is deliberately left per-line: it rounds the
568
+ * other way, so at worst a multi-line code sample renders as escaped text.
569
+ */
570
+ function findInlineCodeRanges(source: string): Array<[number, number]> {
571
+ const ranges: Array<[number, number]> = []
572
+ const len = source.length
573
+ let pos = 0
574
+ while (pos <= len) {
575
+ // Grow one PARAGRAPH SEGMENT: the maximal run of non-blank lines starting
576
+ // at or after `pos`. `segStart`/`segEnd` bound it; blank lines never enter.
577
+ let segStart = -1
578
+ let segEnd = -1
579
+ while (pos <= len) {
580
+ let end = source.indexOf('\n', pos)
581
+ if (end === -1) end = len
582
+ const blank = isBlankLine(source.slice(pos, end))
583
+ if (blank && segStart !== -1) break
584
+ if (!blank) {
585
+ if (segStart === -1) segStart = pos
586
+ segEnd = end
587
+ }
588
+ pos = end === len ? len + 1 : end + 1
589
+ }
590
+ if (segStart === -1) break
591
+ const lineStart = segStart
592
+ const lineEnd = segEnd
593
+ const runStart: number[] = []
594
+ const runLen: number[] = []
595
+ for (let i = lineStart; i < lineEnd; i++) {
596
+ if (source.charCodeAt(i) !== BACKTICK_CODE) continue
597
+ let j = i + 1
598
+ while (j < lineEnd && source.charCodeAt(j) === BACKTICK_CODE) j++
599
+ runStart.push(i)
600
+ runLen.push(j - i)
601
+ i = j - 1
602
+ }
603
+ const n = runStart.length
604
+ if (n > 0) {
605
+ // suffMax[t] = longest run at or after t; 0 past the end.
606
+ const suffMax = new Array<number>(n + 1).fill(0)
607
+ for (let t = n - 1; t >= 0; t--) suffMax[t] = Math.max(runLen[t], suffMax[t + 1])
608
+ let cursor = lineStart
609
+ let idx = 0
610
+ while (idx < n) {
611
+ const runEnd = runStart[idx] + runLen[idx]
612
+ // The scan resumes at the END of the previous match, which can land
613
+ // MID-RUN — exactly as the global regex's `lastIndex` did. The
614
+ // REMAINDER of the run is then an opener in its own right (`` `a`` ``
615
+ // matches twice), so clamp rather than skip.
616
+ if (runEnd <= cursor) {
617
+ idx++
618
+ continue
619
+ }
620
+ const p = Math.max(runStart[idx], cursor)
621
+ const openLen = runEnd - p
622
+ // Largest closer length reachable via a LATER run, and via THIS one.
623
+ const kLater = Math.min(openLen, suffMax[idx + 1])
624
+ const kSame = openLen >> 1
625
+ const k = Math.max(kLater, kSame)
626
+ // No match is possible only when this is the last run and it is a
627
+ // single backtick — every longer run closes on itself, so advancing by
628
+ // one character (what the regex does) cannot find one either.
629
+ if (k < 1) {
630
+ cursor = runEnd
631
+ idx++
632
+ continue
633
+ }
634
+ let q = kSame >= k ? p + k : -1
635
+ if (q === -1)
636
+ for (let t = idx + 1; t < n; t++)
637
+ if (runLen[t] >= k) {
638
+ q = runStart[t]
639
+ break
640
+ }
641
+ ranges.push([p, q + k])
642
+ cursor = q + k
643
+ }
644
+ }
645
+ }
646
+ return ranges
647
+ }
648
+
649
+ /** Exported for the differential fuzz against the retired regex. */
650
+ export const __findInlineCodeRangesForTest = findInlineCodeRanges
651
+
652
+ /**
653
+ * ASCII-ONLY case fold. `String.prototype.toLowerCase()` is NOT
654
+ * length-preserving: U+0130 (Turkish dotted capital `İ`) expands to `i` +
655
+ * U+0307 (1 code unit → 2). It is the only BMP character that does so, and it
656
+ * is ordinary Turkish prose (`İstanbul`, `İzmir`) — so a message with enough
657
+ * of them ahead of a `<textarea>` shifted the whole haystack later than the
658
+ * `segmentOffset + index + match.length` the caller computes from the ORIGINAL
659
+ * text, `hasLaterCloser` began scanning in a window strictly BEFORE the
660
+ * opener, matched an already-consumed `</textarea>`, and left the opener LIVE
661
+ * — reopening the RAWTEXT swallow the mask exists to close.
662
+ *
663
+ * Tag names are ASCII by definition (`TAG_LIKE_REGEX` only matches
664
+ * `[a-zA-Z][a-zA-Z0-9-]*`), so folding ASCII alone loses nothing.
665
+ * `buildCloserHaystack(src).length === src.length` is asserted over the whole
666
+ * fixture corpus in the parity test — that invariant is the actual guard.
667
+ */
668
+ function foldAsciiCase(text: string): string {
669
+ return text.replace(/[A-Z]/g, (c) => String.fromCharCode(c.charCodeAt(0) + 32))
670
+ }
671
+
672
+ /**
673
+ * ---------------------------------------------------------------------------
674
+ * MASK-ONLY code-region blanking (never the carve)
675
+ * ---------------------------------------------------------------------------
676
+ * All of the blanking passes below share one contract:
677
+ *
678
+ * - they SCAN `source` (the folded but otherwise unmasked copy) and APPLY the
679
+ * resulting ranges to `masked`. That split is LOAD-BEARING: the inline-code
680
+ * pass chews a pair of backticks off an unclosed ```` ``` ```` opener (the
681
+ * opener run gives back backticks until a single one matches the next one as
682
+ * its closer), so a fence scan over the masked copy sees no fence at all. Both
683
+ * strings have identical indices, so offsets transfer verbatim.
684
+ * `blankComments` is the ONE deliberate exception (it is fed the masked
685
+ * copy, and runs last) — see its docblock for why the reasoning inverts.
686
+ * - they are LENGTH-PRESERVING (every non-newline char in a range becomes a
687
+ * space), because `escapeOutsideFences` indexes the mask with offsets it
688
+ * computed from the ORIGINAL text.
689
+ * - they fail CLOSED. Blanking too much can only make `hasLaterCloser` return
690
+ * false, i.e. ESCAPE a RAWTEXT opener that could have stayed live; blanking
691
+ * too little leaves a prose `<textarea>` live and lets parse5 swallow the
692
+ * rest of the message. Every approximation here therefore rounds towards
693
+ * blanking.
694
+ *
695
+ * ---------------------------------------------------------------------------
696
+ * THE SAFETY CLAIM IS LINE COVERAGE, NOT PASS COMPOSITION (round 18)
697
+ * ---------------------------------------------------------------------------
698
+ * Round 17 claimed this class was "closed by proof" and offered a PASS CALL
699
+ * MATRIX — which pass invokes which — as the proof. That was the wrong
700
+ * property, and round 18 found the eighth instance anyway. The matrix shows the
701
+ * passes COMPOSE SYMMETRICALLY; it says nothing about whether every line of the
702
+ * document is actually EXAMINED. Round 18's defect lived inside a pass that the
703
+ * matrix lists as present and symmetric (`blankListItemCode` calls and is called
704
+ * by `blankQuotedCode`): the pass simply never put the LIST-MARKER LINE into any
705
+ * run, so that one line was examined by nobody. A symmetric call graph over an
706
+ * incomplete line set is still incomplete. Do not restate the matrix as the
707
+ * safety argument.
708
+ *
709
+ * ---------------------------------------------------------------------------
710
+ * THE TABLE IS THE AUDITABLE ARTIFACT — AND TWO OF ITS ENTRIES WERE FALSE
711
+ * ---------------------------------------------------------------------------
712
+ * Round 19 audited the table below rather than the code, and found two entries
713
+ * literally untrue. Both were LOAD-BEARING: the `blankListItemCode` entry
714
+ * justified a `top >= 4` gate that hid three live instances (narrow `- ` / `1. `
715
+ * marker lines), and the `blankLinkDefinitions` entry's "absorbs … ONE list
716
+ * marker" justified never re-cutting nested markers. A table entry that is not
717
+ * LITERALLY TRUE is worse than no table: it converts an unexamined line into a
718
+ * documented decision. When you change a pass, restate what it NOW claims and
719
+ * re-derive every other entry from the code — do not copy the previous wording
720
+ * forward.
721
+ *
722
+ * THE INVARIANT THAT ACTUALLY MATTERS:
723
+ *
724
+ * For every line L of the document and every block-level construct that can
725
+ * OPEN on L, some pass must examine L at L's own CONTENT COLUMN — the column
726
+ * at which CommonMark itself would begin parsing L, after every enclosing
727
+ * container prefix (blockquote markers, list-item content columns) has been
728
+ * consumed. "Examined" means the line is a member of that pass's scanned run,
729
+ * cut at that column; being merely SKIPPED OVER while state is updated does
730
+ * not count. Constructs that are not line-anchored at all (inline code spans,
731
+ * HTML comments, link reference definitions, inline link/image payloads) are
732
+ * covered instead by a CONTAINER-AGNOSTIC pass that runs once over the whole
733
+ * document.
734
+ *
735
+ * AND: every region that CommonMark turns into an ATTRIBUTE OR AN IDENTIFIER
736
+ * rather than document text is a shelter of the same kind, whether or not it
737
+ * is line-anchored. A reference definition's destination/title and an INLINE
738
+ * link's destination/title are the same thing to remark — both become
739
+ * href/title and never appear as HTML — so both need a pass. Round 19 found
740
+ * the inline half entirely uncovered; it then implemented that generalization
741
+ * for only the PARENTHESISED half of the constructs the generalization names,
742
+ * and round 20 found the BRACKETED half — an image's alt, a reference label, a
743
+ * footnote label — uncovered in exactly the same way.
744
+ *
745
+ * So the checklist for a newly supported construct is: which of its text does
746
+ * remark consume into an attribute or an identifier — INCLUDING bracket text
747
+ * under `!` and reference/footnote labels — rather than emit as document HTML?
748
+ * Every such region needs a pass. The complement matters just as much: text
749
+ * remark DOES emit (an inline link's `[…]`, a bare shortcut reference's
750
+ * `[…]`) must stay VISIBLE, because a closer written there is real.
751
+ *
752
+ * AND: a length cap or a parse failure inside any of these passes must BLANK,
753
+ * never SKIP. `-1`-on-cap is the fail-OPEN shape `ba4a526b` closed for
754
+ * over-cap inline code spans and round 20 found reintroduced in
755
+ * `parseInlineLinkPayload` and the link-definition regexes. A cap is a
756
+ * BLANKING BOUNDARY: blank up to it. Only a genuinely unparseable SHAPE may
757
+ * decline, and only because remark will not read it as a link either.
758
+ *
759
+ * THAT RULE WAS STATED AND NOT STRUCTURALLY ENFORCED, so every new parser
760
+ * re-litigated it and sometimes lost: `ba4a526b` (over-cap code spans), round
761
+ * 20 (payload + definition CAPS), round 21 (the definition SHAPES the same
762
+ * round left behind). It is now enforced by SHAPE rather than by discipline —
763
+ * the container-agnostic passes have exactly TWO stages, and the stage
764
+ * decides the fail direction:
765
+ *
766
+ * RECOGNITION — "is this the construct at all?" MAY decline, and must,
767
+ * because every recognition decline is a spelling CommonMark ALSO refuses:
768
+ * remark emits the text as HTML, so a closer written in it is REAL and
769
+ * blanking it would over-escape a genuine element.
770
+ *
771
+ * CONSUMPTION — "the construct was recognized" may NEVER decline. Every
772
+ * give-up routes through ONE channel per pass, whose DEFAULT is blanking to
773
+ * the construct's CommonMark bound (the next blank line):
774
+ * `blankInlineLinkPayloads` → `paragraphEnd`, `blankLinkDefinitions` →
775
+ * `blankLinesToParagraphBound`. A pass cannot "forget" to blank, because
776
+ * the give-up path IS the blanking path; there is no `return -1` reachable
777
+ * after commitment.
778
+ *
779
+ * EXIT-PATH TABLE — every exit of the FOUR passes that can decline,
780
+ * classified. Keep it accurate when you touch them; an unclassified exit is
781
+ * the next instance.
782
+ *
783
+ * THE RULE THAT PRODUCED THIS ROUND: A NEW PASS MUST LAND IN BOTH TABLES —
784
+ * this one and "WHICH LINES EACH PASS CLAIMS" below — IN THE SAME COMMIT.
785
+ * Round 22 added `blankUnreferencedFootnotes` to the pipeline with an entry in
786
+ * NEITHER, and it is the pass that shipped a live fail-open (round 23: PASS 1
787
+ * counted PHANTOM references, so a definition holding a `</textarea>` stayed
788
+ * in the haystack and the opener above it stayed live, in ten spellings).
789
+ * These two tables have caught six literally-false or missing claims across
790
+ * six rounds; they are the instrument, and the round that skipped them is the
791
+ * round that regressed. Filling them in is not documentation, it is the audit.
792
+ *
793
+ * blankLinkDefinitions
794
+ * R no `[` on the line / `LINK_DEF_OPEN_RE` fails → not a definition line.
795
+ * R `\[^` (GFM footnote), REFERENCED → BLOCK-parsed body, may
796
+ * hold real HTML (r19).
797
+ * Only when the label is
798
+ * REFERENCED: r22 found
799
+ * the decline fail-OPEN
800
+ * for the unreferenced
801
+ * case, which
802
+ * `blankUnreferencedFootnotes`
803
+ * now blanks whole.
804
+ * R `findLabelClose`: `]` not followed by `:` → shortcut reference or
805
+ * plain text; remark
806
+ * EMITS it (the same
807
+ * exclusion
808
+ * `blankBracketLabels`
809
+ * documents).
810
+ * R `findLabelClose`: unescaped `[` in the label → CommonMark rejects the
811
+ * label → paragraph text.
812
+ * R `findLabelClose`: paragraph bound, no `]:` → an ordinary
813
+ * `[`-leading prose
814
+ * paragraph.
815
+ * R `parseDestOnLine` -1 (angle dest unclosed) → no line ending allowed
816
+ * in `<…>`, and a bare
817
+ * dest may not start with
818
+ * `<` → not a definition.
819
+ * R `parseDefTail`/`parseTitleTail` `decline` → trailing content, or a
820
+ * title neither
821
+ * space-separated nor
822
+ * delimiter-opened →
823
+ * remark reads a
824
+ * PARAGRAPH. On the
825
+ * OPENER line this
826
+ * unwinds the WHOLE
827
+ * construct (nothing is
828
+ * blanked). On a
829
+ * CONTINUATION line it
830
+ * splits in TWO, and the
831
+ * old single sentence
832
+ * was true of only one
833
+ * (r22):
834
+ * · after `needTitle`
835
+ * the definition WAS
836
+ * already complete
837
+ * (a title is
838
+ * optional), so
839
+ * stopping is exact;
840
+ * · after `needDest`
841
+ * it was NOT — with
842
+ * no parseable
843
+ * destination remark
844
+ * reads the whole run
845
+ * as a PARAGRAPH —
846
+ * and lines
847
+ * `i..close.line`
848
+ * are ALREADY blanked
849
+ * by the committed
850
+ * loop. So this exit
851
+ * OVER-blanks the
852
+ * label lines; safe
853
+ * because
854
+ * over-blanking only
855
+ * hides closers.
856
+ * C `openTitle` (title opens, never closes) → BLANK to the paragraph
857
+ * bound.
858
+ * C end of `lines` / blank line while continuing → everything up to the
859
+ * bound is already
860
+ * blanked.
861
+ * (no length cap exists in this pass at all)
862
+ *
863
+ * blankInlineLinkPayloads / parseInlineLinkPayload
864
+ * C `q >= limit`, input REMAINS past the cap → returns `limit`, and
865
+ * the caller widens to
866
+ * `paragraphEnd` (r20).
867
+ * R `q >= limit` because the INPUT IS EXHAUSTED → returns -1. Nothing
868
+ * closes the payload and
869
+ * nothing will, so remark
870
+ * reads text too. Split
871
+ * out in r22: it used to
872
+ * share the cap exit, so
873
+ * the ordinary STREAMING
874
+ * tail `see [a](/x`
875
+ * blanked its paragraph
876
+ * and flickered.
877
+ * R angle dest not closed before `\n`/end → CommonMark forbids a
878
+ * line ending in `<…>`.
879
+ * R bare dest with unbalanced `(` → not a link → text.
880
+ * R no `)` where the payload must end → not a link → text.
881
+ * - `s[q] !== close` after the title loop → UNREACHABLE: the loop
882
+ * exits only on the
883
+ * closer or on `q >=
884
+ * limit`, and the latter
885
+ * returns `overflow`
886
+ * first.
887
+ *
888
+ * blankUnreferencedFootnotes (round 23 — the entry round 22 never wrote)
889
+ * R `masked.indexOf('[^') === -1` (whole-pass skip) → the document contains
890
+ * no footnote SPELLING at
891
+ * all, so there is nothing
892
+ * to blank. Exact.
893
+ * R `FOOTNOTE_DEF_OPEN_RE` fails / no `:` after the
894
+ * label / `footnoteLabelEnd` -1 on the OPENER → not a definition line;
895
+ * remark reads a paragraph
896
+ * and any closer on it is
897
+ * REAL.
898
+ * R label IS referenced (in the REF-MASK) → round 19's case:
899
+ * remark keeps the
900
+ * definition, its body is
901
+ * BLOCK-parsed and may
902
+ * hold real HTML.
903
+ * C `footnoteLabelEnd === -1` mid-line → `break` → abandons the REST OF
904
+ * THE LINE's references.
905
+ * Fail-CLOSED (fewer
906
+ * references ⇒ more
907
+ * definitions blanked),
908
+ * but note the shape it
909
+ * gives up on: a line
910
+ * `[^x[ … [^f]` silently
911
+ * stops counting at the
912
+ * voided label, so a REAL
913
+ * `[^f]` after it can be
914
+ * missed and its
915
+ * definition over-blanked
916
+ * into escaped source
917
+ * (cosmetic).
918
+ * C body walk `break` on a SECOND definition line → the body ended; the
919
+ * neighbour is blanked (or
920
+ * not) on its OWN merits.
921
+ * C body walk `break` on a de-indented line after a
922
+ * blank one → GFM's own body bound.
923
+ * (both body `break`s only SHORTEN the blanked range, i.e. leave MORE
924
+ * haystack visible — the same direction as declining the definition
925
+ * entirely, which is round 19's shipped behaviour, never a new hole)
926
+ * (no length cap exists in this pass at all)
927
+ *
928
+ * blankBracketLabels
929
+ * R no `[` in the document → no bracket construct.
930
+ * R `]` with an empty stack → closes nothing.
931
+ * - no length cap and no parse that can fail: the walk is total over the
932
+ * document and crosses newlines, so it has NO give-up path to classify.
933
+ *
934
+ * WHICH LINES EACH PASS CLAIMS, AND AT WHAT COLUMN:
935
+ *
936
+ * findInlineCodeRanges — EVERY line, at column 0 of its PARAGRAPH SEGMENT
937
+ * (a maximal run of non-blank lines). Container-agnostic: backtick runs are
938
+ * matched with no column or prefix anchoring, so a `> ` / indent prefix is
939
+ * ordinary text between ticks. Spans CROSS line breaks (round 18) and stop
940
+ * at a paragraph break, which is exactly CommonMark's bound.
941
+ * blankFencedRegions — every line of the run it is GIVEN, at that run's
942
+ * column (the caller cut it). Absolute-column-limited by `FENCE_RE`'s 0..3
943
+ * indent cap, which is WHY the container passes must re-cut and re-run it.
944
+ * blankIndentedCode — every line of the run it is given, at that run's
945
+ * column, with a list-content-column stack for the +4 threshold.
946
+ * blankLinkDefinitions — EVERY line, in TWO dimensions that must both be
947
+ * stated, because round 21 found the entry true of the first and silently
948
+ * false of the second.
949
+ * COLUMN: at column 0 AND at the column its own prefix reaches.
950
+ * `LINK_DEF_CONTAINER_PREFIX` absorbs a blockquote run and AT MOST ONE
951
+ * list marker, so the top-level call covers a definition at nesting depth
952
+ * 0 or 1 directly. DEEPER nesting (`- - [a]: …`) is NOT covered by the
953
+ * top-level call — round 19's corrected entry — and is reached only
954
+ * because both container passes re-run this pass on their stripped runs,
955
+ * and `blankListItemCode` re-cuts nested markers by recursing into
956
+ * ITSELF. That re-cut is load-bearing, not redundancy.
957
+ * SHAPE: what the label, destination and title may CONTAIN — the
958
+ * dimension the old wording never mentioned, so five ESCAPED-delimiter
959
+ * spellings and five MULTI-LINE spellings were "covered" by an entry that
960
+ * had not examined them. The pass is now a CHARACTER PARSER, not a line
961
+ * regex: `\` + one character is consumed as a unit EVERYWHERE (so a
962
+ * title may hold `\"` / `\'` / `\)` and a label `\]`), the LABEL may
963
+ * span lines up to the paragraph bound, and an unterminated TITLE is
964
+ * blanked to that same bound. There is no length cap of any kind. What it
965
+ * does NOT claim, and why, is on the exits themselves (see below).
966
+ * blankUnreferencedFootnotes — EVERY line, container-agnostic and at ANY
967
+ * depth: `FOOTNOTE_DEF_OPEN_RE`'s own prefix absorbs a blockquote run plus
968
+ * ANY NUMBER of list markers, so unlike `blankLinkDefinitions` this pass
969
+ * needs no container re-cut — and could not use one, because its reference
970
+ * set is document-GLOBAL and a stripped run cannot see it. Exactly ONE
971
+ * top-level call. No length cap of any kind.
972
+ * IT READS TWO SOURCES, and that split is the security-load-bearing part
973
+ * (round 23):
974
+ * DEFINITIONS from the current MASK — a definition an earlier pass hid is
975
+ * not blanked, which leaves it in the haystack (round 19's direction).
976
+ * REFERENCES from a SEPARATE, MORE-BLANKED copy (`footnoteReferenceMask`),
977
+ * because every region remark consumes into an ATTRIBUTE or drops — image
978
+ * alt, full-reference label, inline link title / angle destination, HTML
979
+ * comment, raw HTML block, inline tag attribute, autolink — yields a
980
+ * PHANTOM reference, and a phantom keeps a dropped definition (and its
981
+ * `</textarea>`) in the haystack: fail-OPEN, reproduced live in ten
982
+ * spellings. Counting FEWER references only blanks MORE, so that copy may
983
+ * over-blank freely.
984
+ * CLAIMED BODY: the label line, its lazy paragraph continuations, and
985
+ * further blocks indented >= 4 columns past the blockquote run.
986
+ * blankInlineLinkPayloads — EVERY inline link/image payload in the document,
987
+ * container-agnostic: the scan is anchored on the `](` bigram with no column
988
+ * or prefix anchoring, so a container prefix is ordinary text ahead of it.
989
+ * Claims ONLY the `(…)` payload — never the `[…]` text of an INLINE LINK,
990
+ * which is inline-parsed and reaches the document as HTML. Its cap
991
+ * (`INLINE_LINK_PAYLOAD_MAX`) BLANKS THROUGH rather than declining; only an
992
+ * unparseable SHAPE declines (round 20).
993
+ * blankBracketLabels — EVERY `[…]` group in the document whose text remark
994
+ * consumes into an attribute or an identifier: an image's alt (`[` preceded
995
+ * by `!`), the second group of a `][` adjacency (a full reference's label),
996
+ * the first group of a `][]` adjacency (a collapsed reference's identifier),
997
+ * and a footnote label (`[^…]`, reference AND definition). Container-
998
+ * agnostic: one left-to-right bracket walk, no column or prefix anchoring.
999
+ * Claims NEITHER an inline link's `[…]` NOR a bare shortcut reference's —
1000
+ * remark emits both as HTML, so a closer there is real (round 20). The
1001
+ * `][` / `][]` adjacency is compared PER NESTING DEPTH (round 23 — a single
1002
+ * `prev` let a nested group clobber the sibling it had to be compared with,
1003
+ * so `[txt][[^f]]`'s label was never claimed). Its ONE option,
1004
+ * `{ footnoteLabels: false }`, is for `footnoteReferenceMask` only.
1005
+ * blankComments — EVERY line, container-agnostic: `HTML_COMMENT_RE` is
1006
+ * `[\s\S]`-based and anchored nowhere, so a comment matches straight through
1007
+ * any prefix. Runs LAST, over the masked copy (see its docblock).
1008
+ * blankQuotedCode — supplies runs cut at the BLOCKQUOTE content column,
1009
+ * for every maximal run of quote-prefixed lines, INCLUDING the line that
1010
+ * opens the quote (the prefix regex matches it like any other).
1011
+ * blankListItemCode — supplies runs cut at the LIST-ITEM content column
1012
+ * for EVERY line inside a list item at ANY content column >= 1 (round 19 —
1013
+ * the gate used to be `>= 4` on the claim that "below column 4 the top-level
1014
+ * passes already cover the line at the right column", which is true of a
1015
+ * CONTINUATION line and FALSE of the MARKER line: at content column 2 or 3
1016
+ * the marker line is examined only at column 0, where the leading `- ` /
1017
+ * `1. ` is not whitespace and `FENCE_RE` cannot match). Includes THE MARKER
1018
+ * LINE ITSELF (round 18) and re-cuts NESTED markers by recursing into itself
1019
+ * (round 19), since `LIST_MARKER_RE` matches only the FIRST marker on a
1020
+ * line.
1021
+ *
1022
+ * The two container passes call each other AND `blankListItemCode` calls itself,
1023
+ * and all of them call the fence + indented + link-definition passes, so a line
1024
+ * nested in any order and any DEPTH of containers is eventually cut to its own
1025
+ * content column. That composition is a MEANS to the invariant above, not a
1026
+ * substitute for it. When adding a pass or a container, the question to answer
1027
+ * is "which lines does it claim, at which column, and is any line now claimed by
1028
+ * nobody" — not "does the call graph look symmetric".
1029
+ */
1030
+
1031
+ /**
1032
+ * Length-preserving blank of MANY ranges in one pass. Ranges must be
1033
+ * non-overlapping and ascending.
1034
+ *
1035
+ * THE ONLY BLANKING PRIMITIVE (round 18 — performance). There used to be a
1036
+ * single-range `blankRange` beside it, and `blankIndentedCode` /
1037
+ * `blankFencedRegions` / `blankComments` each folded the document through it
1038
+ * ONCE PER LINE OR REGION. Every call rebuilds the entire string, so masking an
1039
+ * all-indented-code document was QUADRATIC — measured on
1040
+ * `__buildCloserHaystackForTest`: 37 KB → 6 ms, 151 KB → 178 ms, 389 KB →
1041
+ * 1127 ms, 989 KB → 3753 ms (2.5x input ⇒ ~6x time), and the two container
1042
+ * passes re-run both over every nested run, multiplying the constant. A ~400 KB
1043
+ * KB article or release-notes page — all of which go through this renderer —
1044
+ * blocked the main thread for over a second. Every pass now COLLECTS ranges and
1045
+ * applies them here exactly once, which is what the inline pass already did.
1046
+ * After, same four sizes and same harness: 2 ms / 5 ms / 12 ms / 28 ms — dead
1047
+ * linear at ~28 ns/char, a 134x improvement at 989 KB.
1048
+ *
1049
+ * Do not reintroduce a per-range helper; a pass that blanks in a loop is the
1050
+ * regression.
1051
+ */
1052
+ function blankRanges(masked: string, ranges: Array<[number, number]>): string {
1053
+ if (ranges.length === 0) return masked
1054
+ const parts: string[] = []
1055
+ let cursor = 0
1056
+ for (const [from, to] of ranges) {
1057
+ parts.push(masked.slice(cursor, from), masked.slice(from, to).replace(/[^\n]/g, ' '))
1058
+ cursor = to
1059
+ }
1060
+ parts.push(masked.slice(cursor))
1061
+ return parts.join('')
1062
+ }
1063
+
1064
+ /** One scannable line: where it starts, and (for container-nested scans) where
1065
+ * its scanned content starts once the container prefix is stripped. */
1066
+ interface MaskLine {
1067
+ start: number
1068
+ contentStart: number
1069
+ content: string
1070
+ }
1071
+
1072
+ function toMaskLines(source: string): MaskLine[] {
1073
+ const out: MaskLine[] = []
1074
+ let offset = 0
1075
+ for (const line of source.split('\n')) {
1076
+ out.push({ start: offset, contentStart: offset, content: line })
1077
+ offset += line.length + 1
1078
+ }
1079
+ return out
1080
+ }
1081
+
1082
+ /**
1083
+ * Blank every FENCED region in a line run, using the real CommonMark fence
1084
+ * state machine (`createFenceTracker`) rather than a regex.
1085
+ *
1086
+ * This replaces the old `blankUnclosedFence` + `PROTECTED_SPAN_RE` fence
1087
+ * alternative and subsumes both:
1088
+ * - a CLOSED fence is blanked from its opener line through its closer line;
1089
+ * - an EOF-terminated fence is blanked from its opener line to the end of the
1090
+ * run (the case `blankUnclosedFence` covered);
1091
+ * - a would-be closer carrying an INFO STRING (```` ```html ````) no longer
1092
+ * ends the region, because the tracker applies CommonMark's rule that a
1093
+ * closer may not have one. `PROTECTED_SPAN_RE` did end the span there, so
1094
+ * ` ```js … ```html\n</textarea>\n``` ` left the `</textarea>` unmasked and
1095
+ * a prose opener above it stayed live.
1096
+ */
1097
+ function blankFencedRegions(masked: string, lines: MaskLine[]): string {
1098
+ const fences = createFenceTracker()
1099
+ const ranges: Array<[number, number]> = []
1100
+ let openStart: number | null = null
1101
+ let lastEnd = 0
1102
+ for (const line of lines) {
1103
+ const role = fences.push(line.content)
1104
+ lastEnd = line.contentStart + line.content.length
1105
+ if (role === 'open') openStart = line.start
1106
+ else if (role === 'close' && openStart !== null) {
1107
+ ranges.push([openStart, lastEnd])
1108
+ openStart = null
1109
+ }
1110
+ }
1111
+ if (openStart !== null) ranges.push([openStart, lastEnd])
1112
+ return blankRanges(masked, ranges)
1113
+ }
1114
+
1115
+ /**
1116
+ * Blank INDENTED code blocks. A `</textarea>` written as an indented code
1117
+ * sample is code, not a closer — but `FENCE_RE` deliberately caps fence indent
1118
+ * at 3 spaces, so the tracker never sees these lines.
1119
+ *
1120
+ * The threshold is LIST-AWARE, not a flat 4 columns. CommonMark measures
1121
+ * indented code from the enclosing list item's CONTENT column, so under
1122
+ * `1. ` (content column 4) a 4-space line is a paragraph continuation, not
1123
+ * code — and `"1. Here is a form:\n\n <textarea>\n </textarea>\n"` had
1124
+ * its closer blanked, `hasLaterCloser` returned false, and a perfectly real
1125
+ * element got escaped. A numbered list containing markup is a very ordinary
1126
+ * chat answer, so "fail closed" is not a good enough excuse here.
1127
+ *
1128
+ * The walk mirrors `blankQuotedCode`'s line-state approach: a stack of open
1129
+ * list content columns, `code` meaning `indent >= top + 4`. Blank lines keep
1130
+ * the state (a list item survives them); a line indented below the top of the
1131
+ * stack pops it. A line indented past the code threshold is treated as code
1132
+ * BEFORE it is considered as a list marker.
1133
+ *
1134
+ * SCAN-SOURCE INVERSION (same reasoning as `blankComments`, and NOT the shared
1135
+ * contract): the caller must pass lines re-derived from the CURRENT mask, not
1136
+ * from `folded`. Fence content is already blanked by `blankFencedRegions`, but
1137
+ * that only holds for WRITING the mask — a walk over `folded` still SEES those
1138
+ * lines, so a `- x` written inside a fence pushed a content column of 2 and a
1139
+ * later top-level column-4 indented-code line then failed `indent >= top + 4`,
1140
+ * went unblanked, and its code-sample `</textarea>` kept a prose opener LIVE.
1141
+ * Over the masked copy those lines are all spaces, hit the `isBlankLine`
1142
+ * continue, preserve list state and push no bogus column.
1143
+ *
1144
+ * CONTENT-COLUMN CLAMP: CommonMark clamps an item's content column to
1145
+ * `markerEnd + 1` when the first block starts MORE than 4 spaces after the
1146
+ * marker — the remainder is indented code INSIDE the item. Taking the literal
1147
+ * column instead meant `-` + six spaces raised the threshold to 11, so a
1148
+ * column-7 `</textarea>` code sample was not blanked.
1149
+ *
1150
+ * KNOWN OMISSION (deliberate, fail-CLOSED): there is NO paragraph state. Under
1151
+ * CommonMark indented code cannot interrupt a paragraph, so a LAZY
1152
+ * continuation line — `'Here is a form: <textarea>\nsome paragraph\n </textarea>\n'`
1153
+ * — is paragraph text, yet this walk blanks it as code and the (real, properly
1154
+ * closed) element is escaped to visible source. That is cosmetic, and the
1155
+ * option NOT taken here is the fail-OPEN direction: skipping the code test in
1156
+ * paragraph state means blanking LESS, i.e. more closers visible to
1157
+ * `hasLaterCloser` and more openers left live. The list-awareness above was
1158
+ * worth its risk because it is unconditional over an entire list item; this
1159
+ * one is not, so it is documented rather than implemented.
1160
+ */
1161
+ const LIST_MARKER_RE = /^([ \t]*)(?:[-*+]|\d{1,9}[.)])([ \t]+)(?=\S)/
1162
+
1163
+ /** Visual column of `upTo` chars of `line`, expanding tabs to 4-col stops. */
1164
+ function visualColumn(line: string, upTo: number): number {
1165
+ let col = 0
1166
+ for (let i = 0; i < upTo; i++) col = line[i] === '\t' ? col + 4 - (col % 4) : col + 1
1167
+ return col
1168
+ }
1169
+
1170
+ function leadingIndent(line: string): number {
1171
+ const ws = /^[ \t]*/.exec(line)![0]
1172
+ return visualColumn(line, ws.length)
1173
+ }
1174
+
1175
+ /** Character index at which `line` reaches visual column `col`, or -1 when the
1176
+ * column falls INSIDE a tab (no exact character boundary) or the line is too
1177
+ * short. The mask is length-preserving, so a container prefix can only ever be
1178
+ * cut at a character boundary; -1 makes the caller decline to strip, which
1179
+ * leaves the line looking indented and therefore blanks MORE (fail-CLOSED). */
1180
+ function charIndexAtColumn(line: string, col: number): number {
1181
+ let c = 0
1182
+ for (let i = 0; i < line.length; i++) {
1183
+ if (c === col) return i
1184
+ c = line[i] === '\t' ? c + 4 - (c % 4) : c + 1
1185
+ if (c > col) return -1
1186
+ }
1187
+ return c === col ? line.length : -1
1188
+ }
1189
+
1190
+ function blankIndentedCode(masked: string, lines: MaskLine[]): string {
1191
+ const listContentCols: number[] = []
1192
+ const ranges: Array<[number, number]> = []
1193
+ for (const line of lines) {
1194
+ // CommonMark's blank line (spaces/tabs, `\r`-tolerant), NOT `trim()` — see
1195
+ // `isBlankLine`. An NBSP-only line is CONTENT, and skipping it here as
1196
+ // "blank" is the same one-character reopening documented there.
1197
+ if (isBlankLine(line.content)) continue
1198
+ const indent = leadingIndent(line.content)
1199
+ const top = listContentCols.length ? listContentCols[listContentCols.length - 1] : 0
1200
+ if (indent >= top + 4) {
1201
+ ranges.push([line.contentStart, line.contentStart + line.content.length])
1202
+ continue
1203
+ }
1204
+ while (listContentCols.length && indent < listContentCols[listContentCols.length - 1])
1205
+ listContentCols.pop()
1206
+ const marker = LIST_MARKER_RE.exec(line.content)
1207
+ if (marker) {
1208
+ const markerEndCol = visualColumn(line.content, marker[0].length - marker[2].length)
1209
+ const contentCol = visualColumn(line.content, marker[0].length)
1210
+ listContentCols.push(contentCol - markerEndCol > 4 ? markerEndCol + 1 : contentCol)
1211
+ }
1212
+ }
1213
+ return blankRanges(masked, ranges)
1214
+ }
1215
+
1216
+ /** Re-derive scannable lines from the CURRENT mask, preserving each line's
1217
+ * original `start` / `contentStart` (every pass is length-preserving, so the
1218
+ * offsets transfer verbatim). See `blankIndentedCode`'s SCAN-SOURCE
1219
+ * INVERSION. */
1220
+ function remapToMask(masked: string, lines: MaskLine[]): MaskLine[] {
1221
+ return lines.map((line) => ({
1222
+ ...line,
1223
+ content: masked.slice(line.contentStart, line.contentStart + line.content.length),
1224
+ }))
1225
+ }
1226
+
1227
+ /**
1228
+ * Blank HTML COMMENTS. `<!-- </textarea> -->` is not a closer — parse5 consumes
1229
+ * it as comment data — yet it satisfied the raw substring search. The
1230
+ * unterminated form is blanked to EOF, matching what the tokenizer does with a
1231
+ * comment that never ends (and, again, failing closed).
1232
+ *
1233
+ * SCAN-SOURCE EXCEPTION: this is the ONE pass fed the already-masked copy
1234
+ * rather than the unmasked one. The shared contract exists because
1235
+ * the inline-code pass chews backticks off an unclosed fence opener and would
1236
+ * blind a fence scan — but for comments the reasoning INVERTS: a `<!--` inside a code
1237
+ * region is not a comment start, and treating it as one blanked the document to
1238
+ * EOF. Both `` Use `<!--` to start a comment. `` and a truncated `<!-- todo`
1239
+ * inside a ```html fence disabled EVERY later RAWTEXT closer in the message.
1240
+ * Running last over the masked copy is safe: all prior passes are
1241
+ * length-preserving so offsets still transfer verbatim, a `<!--` inside
1242
+ * fenced / inline / indented / quoted code is spaces by now and matches
1243
+ * nothing, and a genuine prose comment is untouched by any of them.
1244
+ */
1245
+ const HTML_COMMENT_RE = /<!--[\s\S]*?-->|<!--[\s\S]*$/g
1246
+
1247
+ function blankComments(masked: string, source: string): string {
1248
+ HTML_COMMENT_RE.lastIndex = 0
1249
+ const ranges: Array<[number, number]> = []
1250
+ let m: RegExpExecArray | null
1251
+ while ((m = HTML_COMMENT_RE.exec(source)) !== null) {
1252
+ ranges.push([m.index, m.index + m[0].length])
1253
+ if (m[0].length === 0) HTML_COMMENT_RE.lastIndex++
1254
+ }
1255
+ return blankRanges(masked, ranges)
1256
+ }
1257
+
1258
+ /**
1259
+ * Blank LINK REFERENCE DEFINITIONS (round 18 — SECURITY).
1260
+ *
1261
+ * `remark` consumes a definition ENTIRELY and emits no node for it, so a
1262
+ * `</textarea>` written in a definition's DESTINATION or TITLE is never a real
1263
+ * closer — but it survived into the closer haystack, `hasLaterCloser` returned
1264
+ * true and a prose `<textarea>` above stayed LIVE. Reproduced byte-identical for
1265
+ * both `[a]: /x "</textarea>"` and `[a]: </textarea>`.
1266
+ *
1267
+ * FAIL DIRECTION: blank the WHOLE line on a definition-SHAPED match, without
1268
+ * modelling "a definition may not interrupt a paragraph". Over-blanking here can
1269
+ * only hide closers, i.e. escape MORE openers — the fail-CLOSED direction — so
1270
+ * the loose shape is the correct bias.
1271
+ *
1272
+ * MULTI-LINE SPELLINGS (round 19). CommonMark lets the destination AND/OR the
1273
+ * title sit on lines FOLLOWING the label. The previous continuation state was a
1274
+ * single `expectTitle` boolean checked against a BARE QUOTED TITLE, so
1275
+ * `[a]:\n/x "</textarea>"` blanked the `[a]:` line and left the whole
1276
+ * `destination + title` line visible (reproduced live, `escapeUnknownHtmlTags`
1277
+ * byte-identical); so did `[a]:\n</textarea>`. remark consumes the entire
1278
+ * definition and emits no link node at all, so the closer is fake in every one
1279
+ * of these spellings. The state is now a three-valued
1280
+ * `'none' | 'needDest' | 'needTitle'`:
1281
+ *
1282
+ * - a definition line with NO destination → `needDest`
1283
+ * - a definition line with a destination but NO title → `needTitle`
1284
+ * - `needDest` accepts a `destination [title]` line, then falls to
1285
+ * `needTitle` (or `none` when that line carried the title)
1286
+ * - `needTitle` accepts a bare quoted/parenthesised title line
1287
+ *
1288
+ * `needDest`'s continuation shape is deliberately loose (any single
1289
+ * non-whitespace run), which can over-blank ONE line after a bare `[a]:` — the
1290
+ * fail-CLOSED direction, and `[a]:` alone is not a shape prose produces.
1291
+ *
1292
+ * FOOTNOTES ARE EXCLUDED (round 19). `\[[^\]\n]{0,999}\]:` also matched a GFM
1293
+ * footnote definition `[^a]: …`, whose content is BLOCK-parsed and therefore
1294
+ * may contain REAL html: `[^a]: <textarea>hi</textarea>` rendered as escaped
1295
+ * visible source while the byte-identical pair in prose rendered correctly.
1296
+ * Cosmetic (fail-closed) rather than a security defect, but wrong, so the label
1297
+ * now rejects a leading `^`.
1298
+ *
1299
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankComments` — the shape absorbs
1300
+ * an optional blockquote run and one list marker, so a definition written inside
1301
+ * a quote or ON a list-marker line is covered by the SINGLE top-level call and
1302
+ * the pass never has to be threaded through the container recursion at a
1303
+ * container-relative column. The marker-line sweep dimension added in this round
1304
+ * found exactly that shape (`- [a]: /x "</textarea>"`) live.
1305
+ */
1306
+ /**
1307
+ * The two `{0,16}` / `{1,16}` whitespace bounds here are the ONE cap in this
1308
+ * pass that may still decline a line, and it is unreachable as a fail-open: an
1309
+ * indent or a marker gap above 16 columns is also ≥ 4 columns past the
1310
+ * enclosing content column, so the line is INDENTED CODE and
1311
+ * `blankIndentedCode` has already blanked it. Verified live at gap 17
1312
+ * (`- ` + 16 spaces + `[a]: /x "</textarea>"` escapes correctly). Do not raise
1313
+ * them into an unbounded `*` on the assumption that "more is safer" — that
1314
+ * would let a 4-column-indented definition line escape the code path it
1315
+ * currently falls into.
1316
+ */
1317
+ const LINK_DEF_CONTAINER_PREFIX = ' {0,3}(?:>[ \\t]?)*[ \\t]{0,16}(?:(?:[-*+]|\\d{1,9}[.)])[ \\t]{1,16})?'
1318
+ /**
1319
+ * NO REGEX, AND NO LENGTH BOUND, ON LABEL / DESTINATION / TITLE
1320
+ * (round 20 removed the `{0,999}` caps; round 21 removed the regexes).
1321
+ *
1322
+ * Round 20 removed the counted caps because a regex that fails to match leaves
1323
+ * the line VISIBLE — the fail-OPEN direction this module's contract forbids.
1324
+ * It left the SHAPE of those regexes untouched, and the shape was
1325
+ * BACKSLASH-BLIND: `[^"\n]*` / `[^'\n]*` / `[^)\n]*` / `[^>\n]*` / `[^\]\n]*`
1326
+ * each stop at the FIRST delimiter, escaped or not, while CommonMark lets a
1327
+ * title hold `\"` / `\'` / `\)` and a label hold `\]`. The class stopped early,
1328
+ * the full-line anchor `[ \t]*\r?$` then failed, and the line stayed VISIBLE
1329
+ * while remark still consumed the definition and emitted NOTHING — five live
1330
+ * spellings, each `escapeUnknownHtmlTags(md) === md` with one live
1331
+ * `<textarea>`:
1332
+ *
1333
+ * [a]: /x "a\"</textarea>" [a]: /x 'it\'s </textarea>'
1334
+ * [a]: /x (a\)</textarea>) [a\]b]: /x "</textarea>"
1335
+ * [a\]b]: </textarea>
1336
+ *
1337
+ * …while the unescaped CONTROL `[a]: /x "</textarea>"` blanked correctly, which
1338
+ * is what makes the escape (not the shape) the cause.
1339
+ *
1340
+ * AND THE LABEL AND TITLE MAY SPAN LINES. The old `'none' | 'needDest' |
1341
+ * 'needTitle'` state modelled continuation only AFTER the `]:`, so a LABEL that
1342
+ * opens on one line and closes on a later one, and a TITLE that opens
1343
+ * unterminated, were examined by NOBODY — `blankBracketLabels` deliberately
1344
+ * excludes a bare `[…]`, so the label had no other pass either. Live in both
1345
+ * renderers, byte-identical no-ops:
1346
+ *
1347
+ * [foo\n</textarea>]: /x [</textarea>\nfoo]: /x
1348
+ * [foo\n</textarea>\nbar]: /x > [foo\n> </textarea>]: /x
1349
+ * [a]: /x "line1\n</textarea>"
1350
+ *
1351
+ * …plus the escalation: a live `<iframe src=… width=… height=…>` behind
1352
+ * `[foo\n</iframe>]: /x`.
1353
+ *
1354
+ * THE PASS IS THEREFORE A CHARACTER PARSER, NOT A LINE REGEX. It is
1355
+ * ESCAPE-AWARE by construction (`skipEscaped` consumes `\` + one character
1356
+ * everywhere), has no length cap at all, and is LINEAR: every scan helper below
1357
+ * advances its cursor monotonically over one line, and the outer line loop
1358
+ * telescopes (see `findLabelClose`'s `stoppedAt` contract). No nested
1359
+ * quantifier survives, so the backtracking risk the counted caps used to
1360
+ * pretend to bound is gone rather than re-bounded. MEASURED, not assumed —
1361
+ * `__buildCloserHaystackForTest`, median of 7, at 37/151/389/989 KB, before →
1362
+ * after: list-dense 3.1/7.9/20.6/47.8 → 2.8/7.5/19.1/54.8 ms; realistic
1363
+ * 1.9/5.8/15.4/43.9 → 1.6/5.8/16.3/49.9 ms; definition-dense 1.3/5.5/15.0/40.0
1364
+ * → 1.4/5.9/17.8/45.2 ms. Every series stays DEAD LINEAR (2.5x input ⇒ ~2.7x
1365
+ * time) and the ~13-15% constant is the price of a character parser over a
1366
+ * regex. The ESCAPED-definition corpus is the outlier at 25.1 → 51.2 ms,
1367
+ * because HEAD did NO WORK on it: the backslash-blind regex failed to match and
1368
+ * left the line visible, which is precisely the defect. Adversarial shapes
1369
+ * (`[` + 40 backslash pairs per line, an all-unclosed-label document) are the
1370
+ * FASTEST corpora measured — 9.5 ms and 14.3 ms at 989 KB — because
1371
+ * `findLabelClose`'s `stoppedAt` contract makes the outer loop telescope
1372
+ * instead of rescanning the paragraph once per line. Do not remove `stoppedAt`;
1373
+ * a naive per-line lookahead is quadratic on exactly those inputs.
1374
+ *
1375
+ * EXIT DISCIPLINE — the structural point of this round. The parse has exactly
1376
+ * two stages, and the stage decides the fail direction:
1377
+ *
1378
+ * RECOGNITION (is this a definition at all?) may DECLINE. Every decline here
1379
+ * is a shape CommonMark also refuses, so remark emits the text as HTML and a
1380
+ * closer written in it is REAL — the same argument that keeps an inline
1381
+ * link's `[…]` and a bare shortcut reference visible. Declining is the
1382
+ * CORRECT answer, not a gap; the reachability argument for each is on the
1383
+ * exit itself.
1384
+ *
1385
+ * CONSUMPTION (a `[…]:` was recognized) may NEVER decline. Every give-up
1386
+ * routes through `blankLinesToParagraphBound` — the ONE give-up channel —
1387
+ * which blanks to the next blank line, CommonMark's own bound for a
1388
+ * definition, exactly as `blankInlineLinkPayloads` widens a capped payload to
1389
+ * `paragraphEnd`. A `decline` returned by the tail parser is a RECOGNITION
1390
+ * verdict delivered late (the line is not definition-shaped after all), and
1391
+ * it therefore unwinds the WHOLE construct — nothing is blanked — rather than
1392
+ * leaving a half-blanked span behind.
1393
+ */
1394
+
1395
+ /** `\` consumes the next character. THE escape primitive for this pass — every
1396
+ * scan below advances through it, which is what makes them all backslash-aware
1397
+ * and all monotonic. */
1398
+ function skipEscaped(s: string, i: number): number {
1399
+ return s[i] === '\\' ? i + 2 : i + 1
1400
+ }
1401
+
1402
+ /** First UNESCAPED occurrence of `ch` in `s` at or after `at`, else -1. */
1403
+ function findUnescaped(s: string, at: number, ch: string): number {
1404
+ for (let i = at; i < s.length; i = skipEscaped(s, i)) if (s[i] === ch) return i
1405
+ return -1
1406
+ }
1407
+
1408
+ const isSpaceTab = (ch: string): boolean => ch === ' ' || ch === '\t'
1409
+
1410
+ function skipSpaces(s: string, i: number): number {
1411
+ while (i < s.length && isSpaceTab(s[i])) i++
1412
+ return i
1413
+ }
1414
+
1415
+ /** Lines are `\n`-split, so a CRLF document leaves a trailing `\r`. The pass
1416
+ * blanks WHOLE lines, so dropping it costs no offset accuracy. */
1417
+ const stripCr = (s: string): string => (s.endsWith('\r') ? s.slice(0, -1) : s)
1418
+
1419
+ const TITLE_CLOSE: Record<string, string> = { '"': '"', "'": "'", '(': ')' }
1420
+
1421
+ /** Index just past a title whose opening delimiter is at `i`, or -1 when the
1422
+ * title does not close on this line. CommonMark ALLOWS a title to span lines,
1423
+ * so -1 is a CONTINUATION signal, never a decline. */
1424
+ function parseTitleOnLine(s: string, i: number): number {
1425
+ const close = TITLE_CLOSE[s[i]]
1426
+ for (let q = i + 1; q < s.length; q = skipEscaped(s, q)) if (s[q] === close) return q + 1
1427
+ return -1
1428
+ }
1429
+
1430
+ /** Index just past a destination at `i`, or -1.
1431
+ *
1432
+ * RECOGNITION DECLINE (-1), reachability: only an angle destination that never
1433
+ * closes on its line. CommonMark forbids a line ending inside `<…>` and a bare
1434
+ * destination may not START with `<`, so such a line is not a definition to
1435
+ * remark either — it is emitted as paragraph text and any closer in it is
1436
+ * REAL. Blanking it would over-escape a genuine element. */
1437
+ function parseDestOnLine(s: string, i: number): number {
1438
+ if (s[i] === '<') {
1439
+ for (let q = i + 1; q < s.length; q = skipEscaped(s, q)) if (s[q] === '>') return q + 1
1440
+ return -1
1441
+ }
1442
+ let q = i
1443
+ while (q < s.length && !isSpaceTab(s[q])) q = skipEscaped(s, q)
1444
+ return q > i ? Math.min(q, s.length) : -1
1445
+ }
1446
+
1447
+ /** What the remainder of ONE line says about the definition being consumed. */
1448
+ type DefTail =
1449
+ | { k: 'done' } // destination (+ optional title) complete; line ends
1450
+ | { k: 'needDest' } // nothing on this line; the destination follows
1451
+ | { k: 'needTitle' } // destination taken; a title MAY follow on a later line
1452
+ | { k: 'openTitle'; close: string } // a title opened here and did not close
1453
+ | { k: 'decline' } // not definition-shaped after all (see below)
1454
+
1455
+ /**
1456
+ * RECOGNITION DECLINE, reachability, for every `decline` this returns:
1457
+ *
1458
+ * - trailing content after a COMPLETE destination (+ title): CommonMark reads
1459
+ * a definition only when nothing but whitespace follows, so `[a]: /x junk
1460
+ * </textarea>` is a PARAGRAPH to remark and its closer is REAL;
1461
+ * - a title that is not space-separated from the destination (`[a]: <x>"t"`) —
1462
+ * same, remark reads no title and the trailing text invalidates the line.
1463
+ * The ANGLE spelling is the reachable one (round 22): `parseDestOnLine`
1464
+ * consumes a BARE destination to the next space/tab, so in `[a]: /x"t"` the
1465
+ * quote is part of the destination and the line returns `needTitle`, never
1466
+ * this decline;
1467
+ * - a non-delimiter where a title must begin — same;
1468
+ * - an unclosed angle destination — see `parseDestOnLine`.
1469
+ *
1470
+ * In every case remark EMITS the text, so leaving it visible is required, not
1471
+ * merely permitted. This is the same boundary `blankBracketLabels` draws
1472
+ * around an inline link's `[…]`.
1473
+ */
1474
+ function parseDefTail(c: string, at: number): DefTail {
1475
+ let q = skipSpaces(c, at)
1476
+ if (q >= c.length) return { k: 'needDest' }
1477
+ const destEnd = parseDestOnLine(c, q)
1478
+ if (destEnd < 0) return { k: 'decline' }
1479
+ q = destEnd
1480
+ const gap = skipSpaces(c, q)
1481
+ if (gap >= c.length) return { k: 'needTitle' }
1482
+ if (gap === q) return { k: 'decline' }
1483
+ return parseTitleTail(c, gap)
1484
+ }
1485
+
1486
+ /** The title half of `parseDefTail`, also used for a BARE title continuation
1487
+ * line. Same decline reachability. */
1488
+ function parseTitleTail(c: string, q: number): DefTail {
1489
+ const close = TITLE_CLOSE[c[q]]
1490
+ if (!close) return { k: 'decline' }
1491
+ const end = parseTitleOnLine(c, q)
1492
+ if (end < 0) return { k: 'openTitle', close }
1493
+ return skipSpaces(c, end) >= c.length ? { k: 'done' } : { k: 'decline' }
1494
+ }
1495
+
1496
+ /**
1497
+ * A definition's CONTINUATION lines carry the blockquote run and indent but
1498
+ * never a list marker — a marker would open a new item, not continue the
1499
+ * definition. The `{0,16}` indent bound is the same unreachable-as-fail-open
1500
+ * cap argued for `LINK_DEF_CONTAINER_PREFIX`: past 16 columns the line is ≥ 4
1501
+ * columns beyond the enclosing content column, i.e. INDENTED CODE that
1502
+ * `blankIndentedCode` has already blanked.
1503
+ */
1504
+ const LINK_DEF_CONT_PREFIX_RE = / {0,3}(?:>[ \t]?)*[ \t]{0,16}/y
1505
+ /** `\[(?!\^)` — a GFM FOOTNOTE definition is NOT a link reference definition;
1506
+ * its content is block-parsed and may hold real HTML (round 19). */
1507
+ const LINK_DEF_OPEN_RE = new RegExp(`^${LINK_DEF_CONTAINER_PREFIX}\\[(?!\\^)`)
1508
+
1509
+ function contPrefixLen(c: string): number {
1510
+ LINK_DEF_CONT_PREFIX_RE.lastIndex = 0
1511
+ return LINK_DEF_CONT_PREFIX_RE.exec(c)![0].length
1512
+ }
1513
+
1514
+ /**
1515
+ * Walk forward for the `]:` that turns an opened label into a DEFINITION.
1516
+ * Crosses lines (a CommonMark label may), bounded by the next blank line.
1517
+ *
1518
+ * Returns `{ line, colon }` on success, or `{ stoppedAt }` — a RECOGNITION
1519
+ * decline whose reachability is:
1520
+ * - a `]` not followed by `:` → a bare shortcut reference or ordinary text,
1521
+ * which remark EMITS, so a closer inside it is real (the exclusion
1522
+ * `blankBracketLabels` already documents);
1523
+ * - an unescaped `[` inside the label → CommonMark rejects the label, so the
1524
+ * whole run is paragraph text;
1525
+ * - the paragraph bound with neither → an ordinary `[`-leading prose
1526
+ * paragraph, which must stay untouched.
1527
+ *
1528
+ * `stoppedAt` also makes the outer loop LINEAR. Nothing in `[i, stoppedAt)`
1529
+ * holds a `]` or `[`, so no line in that window can open a definition either;
1530
+ * the caller resumes at `max(stoppedAt, i + 1)` and the per-line work
1531
+ * telescopes instead of rescanning the paragraph once per line.
1532
+ */
1533
+ function findLabelClose(
1534
+ lines: MaskLine[],
1535
+ i: number,
1536
+ from: number,
1537
+ ): { line: number; colon: number } | { stoppedAt: number } {
1538
+ for (let j = i; j < lines.length; j++) {
1539
+ const c = stripCr(lines[j].content)
1540
+ if (j > i && isBlankLine(c)) return { stoppedAt: j }
1541
+ for (let q = j === i ? from : contPrefixLen(c); q < c.length; q = skipEscaped(c, q)) {
1542
+ if (c[q] === '[') return { stoppedAt: j }
1543
+ if (c[q] !== ']') continue
1544
+ return c[q + 1] === ':' ? { line: j, colon: q + 1 } : { stoppedAt: j }
1545
+ }
1546
+ }
1547
+ return { stoppedAt: lines.length }
1548
+ }
1549
+
1550
+ /**
1551
+ * Blank LINK REFERENCE DEFINITIONS (round 18 — SECURITY).
1552
+ *
1553
+ * `remark` consumes a definition ENTIRELY and emits no node for it, so a
1554
+ * `</textarea>` written in a definition's LABEL, DESTINATION or TITLE is never
1555
+ * a real closer — but it survived into the closer haystack, `hasLaterCloser`
1556
+ * returned true and a prose `<textarea>` above stayed LIVE. Reproduced
1557
+ * byte-identical for `[a]: /x "</textarea>"`, `[a]: </textarea>`, the
1558
+ * multi-line spellings (round 19), the backslash-escaped delimiters and the
1559
+ * multi-line label/title (round 21).
1560
+ *
1561
+ * FAIL DIRECTION: blank the WHOLE line on a definition-SHAPED match, without
1562
+ * modelling "a definition may not interrupt a paragraph". Over-blanking here can
1563
+ * only hide closers, i.e. escape MORE openers — the fail-CLOSED direction — so
1564
+ * the loose shape is the correct bias.
1565
+ *
1566
+ * FOOTNOTES ARE EXCLUDED (round 19). The label rejects a leading `^`: a GFM
1567
+ * footnote definition's content is BLOCK-parsed and may contain REAL html.
1568
+ *
1569
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankComments` — the shape absorbs
1570
+ * an optional blockquote run and one list marker, so a definition written inside
1571
+ * a quote or ON a list-marker line is covered by the SINGLE top-level call.
1572
+ */
1573
+ function blankLinkDefinitions(masked: string, lines: MaskLine[]): string {
1574
+ const ranges: Array<[number, number]> = []
1575
+ const blankLine = (line: MaskLine): void => {
1576
+ ranges.push([line.contentStart, line.contentStart + line.content.length])
1577
+ }
1578
+
1579
+ /**
1580
+ * THE ONE GIVE-UP CHANNEL. A construct already recognized as a definition can
1581
+ * only ever stop blanking at CommonMark's own bound for it — the next blank
1582
+ * line — never by declining. Returns the index of the last line blanked.
1583
+ * `stop` lets a caller end EARLY on a line it recognises (a closing title
1584
+ * delimiter); returning false everywhere degrades to "blank to the bound",
1585
+ * which is the default this channel exists to guarantee.
1586
+ */
1587
+ const blankLinesToParagraphBound = (from: number, stop: (c: string) => boolean): number => {
1588
+ let k = from
1589
+ for (; k < lines.length; k++) {
1590
+ const c = stripCr(lines[k].content)
1591
+ if (isBlankLine(c)) break
1592
+ blankLine(lines[k])
1593
+ if (stop(c)) {
1594
+ k++
1595
+ break
1596
+ }
1597
+ }
1598
+ return k - 1
1599
+ }
1600
+
1601
+ let i = 0
1602
+ while (i < lines.length) {
1603
+ const line = lines[i]
1604
+ if (line.content.indexOf('[') === -1) {
1605
+ i++
1606
+ continue
1607
+ }
1608
+ const open = LINK_DEF_OPEN_RE.exec(line.content)
1609
+ if (!open) {
1610
+ i++
1611
+ continue
1612
+ }
1613
+ const close = findLabelClose(lines, i, open[0].length)
1614
+ if ('stoppedAt' in close) {
1615
+ i = Math.max(close.stoppedAt, i + 1)
1616
+ continue
1617
+ }
1618
+ const head = stripCr(lines[close.line].content)
1619
+ let tail = parseDefTail(head, close.colon + 1)
1620
+ // A late RECOGNITION verdict unwinds the WHOLE construct: remark emits every
1621
+ // line of it as text, so nothing may be blanked.
1622
+ if (tail.k === 'decline') {
1623
+ i = Math.max(close.line, i + 1)
1624
+ continue
1625
+ }
1626
+ // COMMITTED. From here every exit blanks.
1627
+ for (let j = i; j <= close.line; j++) blankLine(lines[j])
1628
+ let j = close.line
1629
+ while (tail.k !== 'done') {
1630
+ if (tail.k === 'openTitle') {
1631
+ const closer = tail.close
1632
+ j = blankLinesToParagraphBound(j + 1, (c) => findUnescaped(c, 0, closer) !== -1)
1633
+ break
1634
+ }
1635
+ const k = j + 1
1636
+ if (k >= lines.length) break
1637
+ const c = stripCr(lines[k].content)
1638
+ if (isBlankLine(c)) break
1639
+ const at = contPrefixLen(c)
1640
+ const next: DefTail =
1641
+ tail.k === 'needDest' ? parseDefTail(c, at) : parseTitleTail(c, skipSpaces(c, at))
1642
+ // The definition is already COMPLETE without this line (a destination-only
1643
+ // definition needs no title; a `[a]:` with no parseable destination is not
1644
+ // a definition at all and remark emits the following line as text), so this
1645
+ // is a RECOGNITION boundary, not a give-up.
1646
+ if (next.k === 'decline') break
1647
+ blankLine(lines[k])
1648
+ j = k
1649
+ tail = next
1650
+ }
1651
+ i = j + 1
1652
+ }
1653
+ return blankRanges(masked, mergeRanges(ranges))
1654
+ }
1655
+
1656
+ /**
1657
+ * `[^` after the container prefix — a GFM FOOTNOTE definition opener, the shape
1658
+ * `LINK_DEF_OPEN_RE`'s `\[(?!\^)` deliberately refuses.
1659
+ *
1660
+ * The prefix absorbs a blockquote run and ANY NUMBER of list markers, where
1661
+ * `LINK_DEF_CONTAINER_PREFIX` stops at one. It has to: this pass is called ONCE
1662
+ * at top level (its reference set is document-global, so it cannot be re-run on
1663
+ * a container pass's stripped run the way `blankLinkDefinitions` is), and the
1664
+ * sweep found `- - [^zz]: … </textarea>` swallowing live at depth 2. Over-
1665
+ * detection is fail-CLOSED here — an unreferenced footnote is emitted by
1666
+ * nothing, so blanking more of one costs nothing at all.
1667
+ *
1668
+ * THE MARKER GROUP CARRIES NO LEADING WHITESPACE QUANTIFIER, deliberately. The
1669
+ * indent is matched ONCE before the group and afterwards only by each marker's
1670
+ * OWN trailing `[ \t]{1,16}`, so no two quantifiers ever compete for the same
1671
+ * whitespace run and a gap has exactly one viable split. The naive spelling
1672
+ * (`(?:[ \t]{0,16}marker[ \t]{1,16})*`) splits a 2-space gap two ways and
1673
+ * backtracks 2^depth on a NON-matching line — `- - - …x` is ordinary prose.
1674
+ */
1675
+ const FOOTNOTE_DEF_OPEN_RE = new RegExp(
1676
+ '^ {0,3}(?:>[ \\t]?)*[ \\t]{0,16}(?:(?:[-*+]|\\d{1,9}[.)])[ \\t]{1,16})*\\[\\^',
1677
+ )
1678
+ /** A blockquote run, WITHOUT swallowing the indent after it — the footnote-body
1679
+ * continuation test has to MEASURE that indent, which `contPrefixLen` eats. */
1680
+ const FOOTNOTE_QUOTE_PREFIX_RE = / {0,3}(?:>[ \t]?)*/y
1681
+
1682
+ /**
1683
+ * micromark's `normalizeIdentifier`, byte-for-byte: collapse every whitespace
1684
+ * run to one space, trim, then case-fold via `toLowerCase().toUpperCase()` (the
1685
+ * double fold is what makes ß/ẞ and the Turkish dotted I agree). A reference and
1686
+ * a definition are the SAME footnote exactly when these agree, so matching on
1687
+ * anything looser (raw slices) would call a resolved footnote unreferenced.
1688
+ */
1689
+ function normalizeFootnoteLabel(label: string): string {
1690
+ return label
1691
+ .replace(/[\t\n\r ]+/g, ' ')
1692
+ .replace(/^ | $/g, '')
1693
+ .toLowerCase()
1694
+ .toUpperCase()
1695
+ }
1696
+
1697
+ /** End index of the label opened by `[^` at `open`, i.e. the index of its
1698
+ * closing `]`, or -1. Escape-aware and single-line, like the construct. An
1699
+ * unescaped `[` inside voids the label exactly as it does for a link label. */
1700
+ function footnoteLabelEnd(c: string, open: number): number {
1701
+ for (let q = open + 2; q < c.length; q = skipEscaped(c, q)) {
1702
+ if (c[q] === '[') return -1
1703
+ if (c[q] === ']') return q
1704
+ }
1705
+ return -1
1706
+ }
1707
+
1708
+ /**
1709
+ * Blank UNREFERENCED GFM FOOTNOTE DEFINITIONS (round 22 — SECURITY).
1710
+ *
1711
+ * Round 19 excluded `[^label]:` from `blankLinkDefinitions` and wrote the
1712
+ * reason on the exit: a footnote's body is BLOCK-parsed and may hold REAL html,
1713
+ * so blanking it would hide a genuine closer and over-escape a genuine element.
1714
+ * That reason is true of a REFERENCED footnote and FALSE of an unreferenced one:
1715
+ * `remark-gfm` resolves definitions against references and DROPS a definition
1716
+ * nothing points at, emitting no node and no footnote section for it. Nothing in
1717
+ * its body reaches the document — but the whole line stayed live in the closer
1718
+ * haystack, `hasLaterCloser` returned true, and the prose opener above it was
1719
+ * left UNESCAPED. Reproduced end-to-end through the real
1720
+ * `escapeUnknownHtmlTags → remarkGfm → rehypeRaw → rehypeSanitize` chain:
1721
+ *
1722
+ * Secret prose.
1723
+ *
1724
+ * <iframe src="https://evil.example/x" width="600">
1725
+ *
1726
+ * visible text
1727
+ *
1728
+ * [^f]: note body </iframe>
1729
+ *
1730
+ * → `<p>Secret prose.</p><iframe src="https://evil.example/x" width="600">
1731
+ * visible text</iframe>` — a LIVE iframe keeping both attributes and swallowing
1732
+ * the prose below. Delete the footnote line and the same input escapes
1733
+ * correctly. The lesson the round generalises: a RECOGNITION decline's
1734
+ * justification must hold for EVERY sub-case of the construct, not the common
1735
+ * one.
1736
+ *
1737
+ * SO THE DECLINE IS NARROWED, NOT DROPPED. A definition whose label IS
1738
+ * referenced keeps round 19's treatment (untouched, body live). A definition
1739
+ * whose label is referenced NOWHERE is blanked with its body. That blanking is
1740
+ * EXACT for every reference the mask can see — remark emits NONE of those bytes
1741
+ * — and OVER-blanks only where an earlier pass has already hidden a REAL
1742
+ * reference, which is the fail-CLOSED direction. (Round 23 checked the claim in
1743
+ * the over-blank direction, where the older "EXACT, not merely fail-closed"
1744
+ * wording was false: `blankLinkDefinitions`' `openTitle` exit blanks to the
1745
+ * paragraph bound, but an unclosed title makes CommonMark REJECT the definition
1746
+ * and read the run as a PARAGRAPH, whose `[^f]` is a genuine reference. Executed:
1747
+ * `[a]: /x "unclosed` + a lazy line holding `[^f]` + `[^f]: body </textarea>`
1748
+ * escapes its opener even though remark renders the closer live. Cosmetic, and
1749
+ * on the safe side — but do not re-read the sentence as a proof that it cannot
1750
+ * happen.)
1751
+ *
1752
+ * THE PASS READS TWO SOURCES, and the split is the security-load-bearing part
1753
+ * (round 23 — this pass's own fail-open):
1754
+ *
1755
+ * DEFINITIONS come from the CURRENT MASK (`text`). A definition line hidden
1756
+ * by an earlier pass is simply not seen, which leaves it in the haystack —
1757
+ * round 19's behaviour, the direction this pass was already in.
1758
+ *
1759
+ * REFERENCES come from a SEPARATE, MORE-BLANKED scratch copy (`refText`,
1760
+ * built by `footnoteReferenceMask`). Round 22 counted them on the current
1761
+ * mask and justified it with "a `[^f]` written inside code has already been
1762
+ * blanked" plus "a phantom reference degrades to round 19's behaviour, never
1763
+ * worse". Both sentences are true of CODE and FALSE of every region remark
1764
+ * consumes into an ATTRIBUTE or drops entirely: at this point in the pipeline
1765
+ * `blankInlineLinkPayloads`, `blankBracketLabels` and `blankComments` have
1766
+ * not run yet, so a `[^f]` written in an image ALT, a full-reference LABEL,
1767
+ * an inline link TITLE or angle DESTINATION, an HTML COMMENT or a raw HTML
1768
+ * BLOCK counted as a live reference. remark resolves NONE of those — the
1769
+ * definition it points at is dropped and never becomes document text — so
1770
+ * the phantom kept the definition (and its `</textarea>`) in the closer
1771
+ * haystack, `hasLaterCloser` returned true and the opener above stayed LIVE.
1772
+ * That is a fail-OPEN, not a degradation: TEN spellings reproduced a live
1773
+ * `<textarea>` (or, with an attribute-bearing `<iframe>` opener, a live
1774
+ * iframe) end-to-end. Verified phantom-ness independently —
1775
+ * `visible text ![x [^f] y](/i.png)` + `[^f]: body` emits NO `data-footnotes`
1776
+ * section at all, while the same input with a real reference does.
1777
+ *
1778
+ * The scratch copy may over-blank freely: counting FEWER references only
1779
+ * blanks MORE definitions, which is the fail-CLOSED direction.
1780
+ *
1781
+ * CONTAINER-NESTED CODE IS NOT A HOLE — round 22's hedge ("a reference inside
1782
+ * such a region still counts, because the container passes run AFTER this one")
1783
+ * was over-pessimistic and is retracted. MEASURED, with a type-6 opener so the
1784
+ * shape is not confounded by the opener's own HTML block: a `[^f]` inside a
1785
+ * blockquoted fence, a list-item fence at content column 2 AND at 4, a
1786
+ * quoted-list fence, a top-level fence and indented code all render 0 live
1787
+ * elements — every one of those regions is ALREADY blanked by
1788
+ * `blankFencedRegions` / `blankIndentedCode` before this pass reads the mask.
1789
+ *
1790
+ * THE ONE NAMED RESIDUAL is TRANSITIVE: a reference that exists ONLY inside
1791
+ * ANOTHER, itself-unreferenced, definition's body (`[^a]: see [^f]` with nothing
1792
+ * referencing `a`). remark drops both definitions, so `[^f]`'s is a phantom too,
1793
+ * but this pass counts references ONCE and would need a FIXPOINT (blank, rebuild
1794
+ * the ref-mask, recount) to see it. Deliberately not built: the loop is
1795
+ * unbounded in the number of definitions, and the shape needs an attacker to
1796
+ * plant a second dead definition. Reproduced and left standing knowingly — if it
1797
+ * is ever closed, close it with a bounded iteration count, not an unbounded one.
1798
+ *
1799
+ * BODY BOUND: the label line, its LAZY paragraph continuation lines, and any
1800
+ * further blocks indented >= 4 columns past the blockquote run — GFM's own
1801
+ * "content of a footnote is what an indented continuation would give a list
1802
+ * item". The walk stops at a de-indented line after a blank one, and at another
1803
+ * footnote-definition line, so a following REFERENCED footnote is not
1804
+ * over-blanked into escaped source.
1805
+ *
1806
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, and MORE so than `blankLinkDefinitions`:
1807
+ * the reference set is document-GLOBAL, so unlike that pass this one cannot be
1808
+ * re-run on a container pass's stripped run to reach deeper nesting. Its opener
1809
+ * prefix therefore absorbs any number of list markers itself (see
1810
+ * `FOOTNOTE_DEF_OPEN_RE`), and the single top-level call covers every depth.
1811
+ */
1812
+ function blankUnreferencedFootnotes(masked: string, lines: MaskLine[], folded: string): string {
1813
+ // `[^` is rare and the whole pass is a no-op without one, so one native scan
1814
+ // buys the overwhelming majority of documents a total skip. This pass runs
1815
+ // over EVERY line of the document; without the guard and the `indexOf` walk
1816
+ // below it would be the most expensive one in the mask, for a construct
1817
+ // almost nothing contains (measured: no regression at 989 KB on a corpus with
1818
+ // no footnotes at all).
1819
+ if (masked.indexOf('[^') === -1) return masked
1820
+ // The MASK's text for a line, sliced on demand. `remapToMask` would allocate a
1821
+ // second object per line of the whole document for a pass that usually touches
1822
+ // one of them.
1823
+ const text = (line: MaskLine): string =>
1824
+ stripCr(masked.slice(line.contentStart, line.contentStart + line.content.length))
1825
+ // …and the REFERENCE-ONLY copy, blanked further (see `footnoteReferenceMask`).
1826
+ // Built ONLY past the `[^` guard, so a document without footnotes pays nothing.
1827
+ // Length-preserving like every other mask, so an index means the same byte in
1828
+ // both copies.
1829
+ const refMask = footnoteReferenceMask(masked, folded)
1830
+ const refText = (line: MaskLine): string =>
1831
+ stripCr(refMask.slice(line.contentStart, line.contentStart + line.content.length))
1832
+
1833
+ // PASS 1 — the DEFINITION on each line (from the mask) and every `[^label]`
1834
+ // that is a REFERENCE (from the ref-mask). A group is a DEFINITION only where
1835
+ // the line-anchored opener shape puts it AND a `:` follows; every other
1836
+ // `[^…]`, including a mid-line `see [^f]: here`, is a reference to remark.
1837
+ // Occurrences are reached with `indexOf`, never a per-character walk: the pass
1838
+ // runs over EVERY line of the document and a char walk would make it the most
1839
+ // expensive one in the mask for a construct almost no line contains.
1840
+ const referenced = new Set<string>()
1841
+ const defAt: Array<number | null> = []
1842
+ for (const line of lines) {
1843
+ const c = text(line)
1844
+ let isDef: number | null = null
1845
+ if (c.indexOf('[^') !== -1) {
1846
+ const open = FOOTNOTE_DEF_OPEN_RE.exec(c)
1847
+ if (open !== null) {
1848
+ // The opener is line-anchored past a whitespace/marker prefix, so its
1849
+ // `[` can never carry a backslash escape — the prefix would not match.
1850
+ const defOpen = open[0].length - 2
1851
+ const end = footnoteLabelEnd(c, defOpen)
1852
+ if (end !== -1 && c[end + 1] === ':') isDef = defOpen
1853
+ }
1854
+ }
1855
+ defAt.push(isDef)
1856
+ const rc = refText(line)
1857
+ for (let q = rc.indexOf('[^'); q !== -1; q = rc.indexOf('[^', q)) {
1858
+ // Escape-aware without the walk: an ODD run of backslashes before the `[`
1859
+ // escapes it, an EVEN one is escaped backslashes and leaves `[` live.
1860
+ let back = q
1861
+ while (back > 0 && rc[back - 1] === '\\') back--
1862
+ if ((q - back) % 2 === 1) {
1863
+ q += 2
1864
+ continue
1865
+ }
1866
+ const end = footnoteLabelEnd(rc, q)
1867
+ if (end === -1) break
1868
+ if (q !== isDef) referenced.add(normalizeFootnoteLabel(rc.slice(q + 2, end)))
1869
+ q = end + 1
1870
+ }
1871
+ }
1872
+
1873
+ // PASS 2 — blank each definition whose label nothing references, body included.
1874
+ // Slices the REAL mask, never the ref-mask.
1875
+ const ranges: Array<[number, number]> = []
1876
+ for (let i = 0; i < lines.length; i++) {
1877
+ const at = defAt[i]
1878
+ if (at === null) continue
1879
+ const head = text(lines[i])
1880
+ const end = footnoteLabelEnd(head, at)
1881
+ if (referenced.has(normalizeFootnoteLabel(head.slice(at + 2, end)))) continue
1882
+ let j = i
1883
+ let sawBlank = false
1884
+ for (let k = i + 1; k < lines.length; k++) {
1885
+ const c = text(lines[k])
1886
+ if (isBlankLine(c)) {
1887
+ sawBlank = true
1888
+ continue
1889
+ }
1890
+ // A second definition ends this one; over-blanking a REFERENCED
1891
+ // neighbour's body would show it as escaped source (cosmetic, but avoidable).
1892
+ if (defAt[k] !== null) break
1893
+ // After a blank line only an INDENTED block continues the footnote; before
1894
+ // one, any non-blank line is a lazy paragraph continuation.
1895
+ if (sawBlank && footnoteIndentCols(c) < 4) break
1896
+ j = k
1897
+ }
1898
+ for (let k = i; k <= j; k++) {
1899
+ const line = lines[k]
1900
+ ranges.push([line.contentStart, line.contentStart + line.content.length])
1901
+ }
1902
+ }
1903
+ return blankRanges(masked, mergeRanges(ranges))
1904
+ }
1905
+
1906
+ /**
1907
+ * The REFERENCE-COUNTING copy of the mask for `blankUnreferencedFootnotes`
1908
+ * (round 23 — SECURITY, that pass's own fail-open).
1909
+ *
1910
+ * A `[^f]` only makes a definition REFERENCED if remark resolves it as a
1911
+ * reference. Everything remark consumes into an ATTRIBUTE or drops outright is
1912
+ * a PHANTOM, and at this point in the pipeline none of those regions are masked
1913
+ * yet, so this copy applies the four passes that hide them:
1914
+ *
1915
+ * `blankInlineLinkPayloads` — an inline link/image DESTINATION or TITLE
1916
+ * (`[a](/x "[^f]")`, `[a](<[^f]>)`, `![a](/x '[^f]')`) becomes href/title.
1917
+ * `blankBracketLabels`, WITHOUT its footnote-label ranges — an image ALT
1918
+ * (`![[^f]](/i.png)`) and a full-reference LABEL (`[txt][[^f]]`) become an
1919
+ * attribute or an identifier. The `[^…]` ranges MUST be excluded: that pass
1920
+ * blanks a footnote label "reference AND definition alike", which here
1921
+ * would erase EVERY real reference and over-blank every referenced
1922
+ * definition into escaped source.
1923
+ * HTML BLOCK ranges — inside a `<div>` … block the line `[^f]` is raw HTML
1924
+ * content, not a reference. Reuses `computeHtmlBlockRanges`, the module's
1925
+ * own CommonMark block model, so this copy cannot disagree with the carve.
1926
+ * `blankComments` — `<!-- [^f] -->` is dropped entirely. LAST, as everywhere
1927
+ * else, because it scans the masked copy. (The pipeline's own comment pass
1928
+ * still runs last over the REAL mask; this is a separate string.)
1929
+ *
1930
+ * OVER-BLANKING HERE IS FREE: fewer references means more definitions look
1931
+ * unreferenced, which blanks MORE of the haystack — the fail-CLOSED direction.
1932
+ * That is why this copy may apply passes out of the pipeline's order and may
1933
+ * use a block model that only approximates remark's.
1934
+ */
1935
+ function footnoteReferenceMask(masked: string, folded: string): string {
1936
+ let ref = blankInlineLinkPayloads(masked, folded)
1937
+ ref = blankBracketLabels(ref, folded, { footnoteLabels: false })
1938
+ ref = blankRanges(
1939
+ ref,
1940
+ mergeRanges(computeHtmlBlockRanges(folded).map(({ start, end }) => [start, end])),
1941
+ )
1942
+ ref = blankComments(ref, ref)
1943
+ ref = ref.replace(AUTOLINK_LIKE_RE, (m) => ' '.repeat(m.length))
1944
+ return blankTagAttributes(ref)
1945
+ }
1946
+
1947
+ /** AUTOLINKS, both spellings, DELIBERATELY over-wide (this regex is only ever
1948
+ * applied to the reference-counting copy, where over-blanking is free): a
1949
+ * CommonMark `<scheme:…>` autolink and a GFM LITERAL autolink both become an
1950
+ * `href`, so `<https://e.example/[^f]>` and `https://e.example/x[^f]y` are
1951
+ * phantom references — each reproduced a live iframe. It stops at whitespace,
1952
+ * so an ordinary `see https://e.example [^f]` keeps its REAL reference. Email
1953
+ * autolinks are deliberately absent: a `[` voids the email shape, so `[^f]`
1954
+ * inside one IS a real reference (verified — remark emits the footnote). */
1955
+ const AUTOLINK_LIKE_RE = /<[a-z][a-z0-9+.-]{1,31}:[^\s<>]*>|(?:https?:\/\/|www\.)[^\s<]*/gi
1956
+
1957
+ /** Blank the ATTRIBUTE RUN of every tag-like span, length-preserving. Blanking
1958
+ * the WHOLE tag would blank real `</tag>` closers too, so only the run between
1959
+ * the tag name and the `>` is cleared. Shared by `buildCloserHaystack`'s final
1960
+ * step and by `footnoteReferenceMask`, where an INLINE tag's attribute is one
1961
+ * more region remark never resolves a `[^f]` in (`<span title="[^f]">`
1962
+ * reproduced a live iframe). */
1963
+ function blankTagAttributes(masked: string): string {
1964
+ return masked.replace(
1965
+ TAG_LIKE_REGEX,
1966
+ (_m, slash: string, tag: string, rest: string, selfClose: string) =>
1967
+ `<${slash}${tag}${' '.repeat(rest.length)}${selfClose}>`,
1968
+ )
1969
+ }
1970
+
1971
+ /** Leading indent of `c` in COLUMNS (tabs advance to the next multiple of 4)
1972
+ * measured PAST the blockquote run, which is the column GFM measures a
1973
+ * footnote's continuation blocks at. */
1974
+ function footnoteIndentCols(c: string): number {
1975
+ FOOTNOTE_QUOTE_PREFIX_RE.lastIndex = 0
1976
+ let q = FOOTNOTE_QUOTE_PREFIX_RE.exec(c)![0].length
1977
+ let col = 0
1978
+ for (; q < c.length && isSpaceTab(c[q]); q++) col = c[q] === '\t' ? col + 4 - (col % 4) : col + 1
1979
+ return col
1980
+ }
1981
+
1982
+ /**
1983
+ * Blank the PARENTHESISED PAYLOAD of an INLINE link or image (round 19 —
1984
+ * SECURITY, a whole shelter class the table did not name).
1985
+ *
1986
+ * remark consumes an inline link's DESTINATION and TITLE exactly as it consumes
1987
+ * a reference definition's: both become href/title ATTRIBUTES on the emitted
1988
+ * node and never reach the document as HTML. So a `</textarea>` written in
1989
+ * either one is not a closer — but `blankLinkDefinitions` only covers the
1990
+ * DEFINITION spelling, and no pass covered the inline one. All eight spellings
1991
+ * reproduced live (`escapeUnknownHtmlTags` byte-identical, one live
1992
+ * `<textarea>` swallowing the prose above it):
1993
+ *
1994
+ * [a](/x "</textarea>") [a](/x '</textarea>') [a](/x (</textarea>))
1995
+ * ![a](/x "</textarea>") [a](</textarea>) > [a](/x "</textarea>")
1996
+ * - [a](/x "</textarea>") See [a](/x "</textarea>") for more.
1997
+ *
1998
+ * …and the escalation: an `<iframe src="…" width="600">` opener plus a title
1999
+ * shelter yields a LIVE iframe retaining both attributes.
2000
+ *
2001
+ * ONLY THE PAYLOAD IS BLANKED HERE, never the `[…]` text of an INLINE LINK
2002
+ * (`[text](dest)`). That text is INLINE-PARSED and reaches the document as
2003
+ * HTML, so blanking it would over-escape a paired `<textarea>…</textarea>`
2004
+ * written inside a link label. The bracket text of every OTHER spelling — an
2005
+ * IMAGE's alt, a reference LABEL, a footnote LABEL — is consumed into an
2006
+ * attribute or an identifier instead, and is blanked by `blankBracketLabels`
2007
+ * below. Round 20 found that half uncovered.
2008
+ *
2009
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankLinkDefinitions` and
2010
+ * `blankComments`: the scan is anchored on the `](` bigram with no column or
2011
+ * prefix anchoring, so a blockquote run or list marker is ordinary text ahead of
2012
+ * it and the single top-level call covers every container nesting.
2013
+ *
2014
+ * FAIL DIRECTION: blanks ONLY on a payload that parses through to its closing
2015
+ * `)`. A shape that does not parse is not a link to remark either, so its
2016
+ * `</textarea>` IS a real closer and must stay visible — declining to blank is
2017
+ * the correct answer there, not a gap. Conversely the parse is deliberately
2018
+ * LOOSER than CommonMark (it accepts payloads remark would reject, e.g. after a
2019
+ * `](`-shaped bigram in ordinary prose), and every such over-detection only
2020
+ * hides closers, i.e. escapes MORE openers.
2021
+ *
2022
+ * THE CAP IS A BLANKING BOUNDARY, NOT A REJECTION (round 20 — SECURITY).
2023
+ * `INLINE_LINK_PAYLOAD_MAX` bounds how far one `](` may blank, so a stray
2024
+ * bigram cannot blank an unbounded tail. It was originally spent as `return -1`
2025
+ * on every over-limit exit, which the caller reads as "not a link, leave
2026
+ * visible" — so `[a](/x "<1100 chars></textarea>")` sheltered its closer in
2027
+ * full view of the mask (`escapeUnknownHtmlTags` byte-identical, one live
2028
+ * RAWTEXT element; the `<iframe src=… width=…>` spelling kept both attributes).
2029
+ * CommonMark places NO length bound on a destination or a title, so that is
2030
+ * ordinary output, and this is the SAME fail-open shape `ba4a526b` closed for
2031
+ * over-cap inline code spans, reintroduced in newer code.
2032
+ *
2033
+ * A CAP-driven exit therefore returns `limit` — "blank through the cap" — while
2034
+ * a SHAPE-driven exit still returns -1. The two are distinguished by testing
2035
+ * `q >= limit` BEFORE the shape test at every exit; over-blanking a bounded
2036
+ * window is the fail-CLOSED direction, declining on a genuinely unparseable
2037
+ * shape is the deliberate one.
2038
+ *
2039
+ * AND THE CAP-DRIVEN EXIT MUST BLANK PAST THE CAP, not to it. Blanking exactly
2040
+ * `[from, limit)` still leaves the shelter live whenever the sheltered closer
2041
+ * sits BEYOND the cap — which is the ordinary case, since the filler is what
2042
+ * pushed the payload over it (measured: `[a](/x "<1100 y's></textarea>")` was
2043
+ * STILL a byte-identical no-op with a to-the-cap blank). So the caller widens a
2044
+ * capped payload to the end of its PARAGRAPH — CommonMark's own bound, since
2045
+ * neither a destination nor a title may contain a blank line. The cap therefore
2046
+ * only decides WHEN to stop parsing, never how little to blank, and a stray
2047
+ * `](` still cannot blank an unbounded tail: it fails on SHAPE and blanks
2048
+ * nothing.
2049
+ *
2050
+ * "FAILS ON SHAPE" HAS TO INCLUDE RUNNING OUT OF INPUT (round 22). It did not:
2051
+ * `limit` is `min(s.length, from + MAX)`, so a payload that simply reached the
2052
+ * END OF THE DOCUMENT hit the same `q >= limit` tests as a capped one and
2053
+ * returned `limit`, which the caller widened to `paragraphEnd`. `see [a](/x` at
2054
+ * end of input therefore blanked its paragraph tail (`see [a]( `) even though
2055
+ * nothing there is a link. Safe direction, but that is the COMMON shape while
2056
+ * STREAMING — the last token of a partial message is often a half-written link
2057
+ * — so an earlier opener was escaped mid-stream and unescaped when the link
2058
+ * completed, a visible flicker. The two are now distinguished by `overflow`:
2059
+ * `limit` only when input remains PAST the cap, -1 when the input is exhausted.
2060
+ * The cap path still blanks THROUGH (round 20's fix is untouched).
2061
+ */
2062
+ const INLINE_LINK_PAYLOAD_MAX = 1024
2063
+
2064
+ const isInlineSpace = (ch: string): boolean =>
2065
+ ch === ' ' || ch === '\t' || ch === '\n' || ch === '\r' || ch === '\f' || ch === '\v'
2066
+
2067
+ /** Index of the payload's closing `)`, or the cap index when the payload runs
2068
+ * past `INLINE_LINK_PAYLOAD_MAX` (blank through the cap), or -1 when the shape
2069
+ * does not parse. `from` is the index just past the `](`. */
2070
+ function parseInlineLinkPayload(s: string, from: number): number {
2071
+ const limit = Math.min(s.length, from + INLINE_LINK_PAYLOAD_MAX)
2072
+ // What a `q >= limit` exit MEANS, which is not one thing (round 22):
2073
+ // - the CAP truncated a payload that still has input after it → `limit`,
2074
+ // "blank through the cap" (round 20; the caller widens to `paragraphEnd`);
2075
+ // - the INPUT RAN OUT → -1, a SHAPE decline. Nothing closed the payload and
2076
+ // nothing ever will in this document, so it is not a link to remark
2077
+ // either. This is the ordinary STREAMING tail (`see [a](/x` as the last
2078
+ // token), where returning `limit` widened the blank to the paragraph end
2079
+ // and escaped an earlier opener that unescaped again once the link
2080
+ // completed — a visible flicker.
2081
+ const overflow = limit < s.length ? limit : -1
2082
+ let q = from
2083
+ while (q < limit && isInlineSpace(s[q])) q++
2084
+ // DESTINATION — angle-bracketed, or a bare run with BALANCED parens.
2085
+ if (s[q] === '<') {
2086
+ q++
2087
+ while (q < limit && s[q] !== '>' && s[q] !== '\n') q += s[q] === '\\' ? 2 : 1
2088
+ if (q >= limit) return overflow
2089
+ if (s[q] !== '>') return -1
2090
+ q++
2091
+ } else {
2092
+ let depth = 0
2093
+ while (q < limit) {
2094
+ const ch = s[q]
2095
+ if (ch === '\\') {
2096
+ q += 2
2097
+ continue
2098
+ }
2099
+ if (isInlineSpace(ch)) break
2100
+ if (ch === '(') depth++
2101
+ else if (ch === ')') {
2102
+ if (depth === 0) break
2103
+ depth--
2104
+ }
2105
+ q++
2106
+ }
2107
+ if (q >= limit) return overflow
2108
+ if (depth !== 0) return -1
2109
+ }
2110
+ // TITLE — `"…"`, `'…'` or `(…)`, separated from the destination by space.
2111
+ const beforeGap = q
2112
+ while (q < limit && isInlineSpace(s[q])) q++
2113
+ const open = s[q]
2114
+ if (q > beforeGap && (open === '"' || open === "'" || open === '(')) {
2115
+ const close = open === '(' ? ')' : open
2116
+ let depth = 1
2117
+ q++
2118
+ while (q < limit) {
2119
+ const ch = s[q]
2120
+ if (ch === '\\') {
2121
+ q += 2
2122
+ continue
2123
+ }
2124
+ if (open === '(' && ch === '(') depth++
2125
+ else if (ch === close && --depth === 0) break
2126
+ q++
2127
+ }
2128
+ if (q >= limit) return overflow
2129
+ if (s[q] !== close) return -1
2130
+ q++
2131
+ while (q < limit && isInlineSpace(s[q])) q++
2132
+ }
2133
+ if (q >= limit) return overflow
2134
+ return s[q] === ')' ? q : -1
2135
+ }
2136
+
2137
+ /** Start index of the first BLANK line at or after `from`, i.e. the end of the
2138
+ * paragraph `from` sits in — the widest span an inline construct may cover. */
2139
+ function paragraphEnd(s: string, from: number): number {
2140
+ let lineStart = s.indexOf('\n', from)
2141
+ while (lineStart !== -1) {
2142
+ lineStart += 1
2143
+ const next = s.indexOf('\n', lineStart)
2144
+ const line = s.slice(lineStart, next === -1 ? s.length : next)
2145
+ if (isBlankLine(line)) return lineStart
2146
+ if (next === -1) break
2147
+ lineStart = next
2148
+ }
2149
+ return s.length
2150
+ }
2151
+
2152
+ function blankInlineLinkPayloads(masked: string, source: string): string {
2153
+ const ranges: Array<[number, number]> = []
2154
+ let i = source.indexOf('](')
2155
+ while (i !== -1) {
2156
+ const from = i + 2
2157
+ const cap = Math.min(source.length, from + INLINE_LINK_PAYLOAD_MAX)
2158
+ const close = parseInlineLinkPayload(source, from)
2159
+ if (close < 0) {
2160
+ i = source.indexOf('](', i + 1)
2161
+ continue
2162
+ }
2163
+ // A CAPPED payload (`close === cap`) has an unknown end, so blank to the end
2164
+ // of the paragraph — see the docblock. A parsed one blanks exactly.
2165
+ const end = close >= cap ? paragraphEnd(source, from) : close
2166
+ if (end > from) ranges.push([from, end])
2167
+ // Ascending and non-overlapping: resume past the range just blanked.
2168
+ i = source.indexOf('](', Math.max(end, from))
2169
+ }
2170
+ return blankRanges(masked, ranges)
2171
+ }
2172
+
2173
+ /** Sort + merge so overlapping/nested finds satisfy `blankRanges`' contract
2174
+ * (non-overlapping, ascending). Empty ranges are dropped. */
2175
+ function mergeRanges(ranges: Array<[number, number]>): Array<[number, number]> {
2176
+ ranges.sort((a, b) => a[0] - b[0])
2177
+ const out: Array<[number, number]> = []
2178
+ for (const [from, to] of ranges) {
2179
+ if (to <= from) continue
2180
+ const last = out[out.length - 1]
2181
+ if (last && from <= last[1]) {
2182
+ if (to > last[1]) last[1] = to
2183
+ } else out.push([from, to])
2184
+ }
2185
+ return out
2186
+ }
2187
+
2188
+ /**
2189
+ * Blank the BRACKET TEXT of every spelling remark consumes into an ATTRIBUTE or
2190
+ * an IDENTIFIER (round 20 — SECURITY, the other half of the shelter class
2191
+ * `blankInlineLinkPayloads` opened).
2192
+ *
2193
+ * Round 19 wrote the general rule — "every region CommonMark turns into an
2194
+ * ATTRIBUTE rather than document text is a shelter of the same kind" — and then
2195
+ * implemented only the `(…)` payload half of it, on a rationale ("never the
2196
+ * `[…]` link TEXT, which is inline-parsed and may hold real HTML") that is true
2197
+ * of an INLINE LINK and false of every other bracket spelling. All seven
2198
+ * reproduced live, in BOTH renderers, with `escapeUnknownHtmlTags` returning the
2199
+ * input BYTE-IDENTICAL and a live RAWTEXT element swallowing the prose:
2200
+ *
2201
+ * ![</textarea>](/x) → alt="</textarea>" (string attribute)
2202
+ * ![</textarea>][r] → alt="…" (reference image)
2203
+ * [a][</textarea>] → label → identifier, never rendered
2204
+ * [</textarea>][] → collapsed reference, identifier again
2205
+ * See[^</textarea>] → href="#user-content-fn-%3C/textarea%3E"
2206
+ * > ![</textarea>](/x) · - ![</textarea>](/x) (container-nested)
2207
+ *
2208
+ * …plus the escalation: `<iframe src="https://evil.example/x" width="600">` in
2209
+ * prose above `![</iframe>](/x)` yielded a LIVE iframe retaining `src`, `width`
2210
+ * and `height`.
2211
+ *
2212
+ * WHAT IS CLAIMED, AND WHAT IS DELIBERATELY NOT:
2213
+ *
2214
+ * - a `[…]` whose `[` is immediately preceded by `!` — an image's alt is a
2215
+ * STRING attribute in every image spelling (inline, reference, collapsed,
2216
+ * shortcut), so the bracket text never reaches the document as HTML;
2217
+ * - the SECOND `[…]` of a `][` adjacency — a FULL reference's label, which
2218
+ * remark resolves to a definition and never renders;
2219
+ * - the FIRST `[…]` of a `][]` adjacency — a COLLAPSED reference, whose
2220
+ * bracket text IS the identifier. (remark also inline-parses it for display,
2221
+ * so unlike an alt this one is not purely an attribute; blanking it is the
2222
+ * fail-CLOSED direction and the reviewer-confirmed shelter, not a claim that
2223
+ * the text is unrendered.)
2224
+ * - a footnote LABEL, `[^…]`, in BOTH the reference and the definition —
2225
+ * remark percent-encodes it into `href`/`id`. Only the LABEL: round 19 was
2226
+ * right that a footnote definition's BODY is BLOCK-parsed and may hold real
2227
+ * HTML, which is why `blankLinkDefinitions` refuses the whole line.
2228
+ * - NOT the `[…]` of an inline `[text](…)` link, and NOT a bare SHORTCUT
2229
+ * reference `[label]`: in both, remark emits the bracket text as inline
2230
+ * HTML, so a `</textarea>` there IS a real closer and must stay visible.
2231
+ * (Verified: with a live opener above it, that closer pairs.)
2232
+ *
2233
+ * The reference spellings are NOT reachable from the `](`-anchored scan in
2234
+ * `blankInlineLinkPayloads` — there is no `](` in `![x][r]` or `[a][r]` at all —
2235
+ * so this pass carries its own anchors.
2236
+ *
2237
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankInlineLinkPayloads`,
2238
+ * `blankLinkDefinitions` and `blankComments`: a single left-to-right bracket
2239
+ * walk with no column or prefix anchoring, so a blockquote run or list marker is
2240
+ * ordinary text ahead of it and ONE top-level call covers every nesting.
2241
+ *
2242
+ * FAIL DIRECTION: brackets that do not resolve to a link/image at all (ordinary
2243
+ * prose `see [1][2]`) are still blanked. Every such over-detection only hides
2244
+ * closers, i.e. escapes MORE openers. A backslash escape is consumed as a pair,
2245
+ * so `\[` does not open a group; `\!` still leaves the following `[` looking
2246
+ * image-like, which over-blanks in the same safe direction.
2247
+ */
2248
+ function blankBracketLabels(
2249
+ masked: string,
2250
+ source: string,
2251
+ { footnoteLabels = true }: { footnoteLabels?: boolean } = {},
2252
+ ): string {
2253
+ if (source.indexOf('[') === -1) return masked
2254
+ const ranges: Array<[number, number]> = []
2255
+ /** Open `[` positions, innermost last. */
2256
+ const open: number[] = []
2257
+ /**
2258
+ * The most recently CLOSED group AT EACH NESTING DEPTH, for the `][` / `][]`
2259
+ * adjacencies. Round 23: this used to be a SINGLE `prev`, which any NESTED
2260
+ * group clobbered — so in `[txt][[^f]]` the outer second group (a full
2261
+ * reference's LABEL) was compared against the INNER `[^f]` instead of against
2262
+ * `[txt]`, the adjacency failed and the label was never blanked. That spelling
2263
+ * was a live fail-open through `blankUnreferencedFootnotes`' phantom count.
2264
+ * Depth-keyed, siblings are compared with siblings.
2265
+ */
2266
+ const prevByDepth: Array<{ open: number; close: number } | null> = []
2267
+ for (let i = 0; i < source.length; i++) {
2268
+ const ch = source[i]
2269
+ if (ch === '\\') {
2270
+ i++
2271
+ continue
2272
+ }
2273
+ if (ch === '[') {
2274
+ open.push(i)
2275
+ // Whatever closed at this depth before belongs OUTSIDE the group just
2276
+ // opened, so it cannot be adjacent to anything inside it.
2277
+ prevByDepth[open.length] = null
2278
+ continue
2279
+ }
2280
+ if (ch !== ']') continue
2281
+ const from = open.pop()
2282
+ if (from === undefined) {
2283
+ prevByDepth[0] = null
2284
+ continue
2285
+ }
2286
+ const prev = prevByDepth[open.length] ?? null
2287
+ // IMAGE alt — `![…]`, every image spelling.
2288
+ if (from > 0 && source[from - 1] === '!') ranges.push([from + 1, i])
2289
+ // FOOTNOTE label — `[^…]`, reference and definition alike. Suppressed for
2290
+ // the reference-counting copy only (`footnoteReferenceMask`), where blanking
2291
+ // the labels would erase the very references being counted.
2292
+ if (footnoteLabels && source[from + 1] === '^') ranges.push([from + 2, i])
2293
+ // REFERENCE label — the second group of `[…][…]`, or, when that group is
2294
+ // EMPTY (`[…][]`), the first group, which is then the identifier.
2295
+ if (prev !== null && prev.close === from - 1) {
2296
+ if (i === from + 1) ranges.push([prev.open + 1, prev.close])
2297
+ else ranges.push([from + 1, i])
2298
+ }
2299
+ prevByDepth[open.length] = { open: from, close: i }
2300
+ }
2301
+ return blankRanges(masked, mergeRanges(ranges))
2302
+ }
2303
+
2304
+ /** `> ` / `>` container prefixes, including nested ones (`> > `). */
2305
+ const BLOCKQUOTE_PREFIX_RE = /^(?: {0,3}>[ \t]?)+/
2306
+
2307
+ /**
2308
+ * EXACT NO-OP GUARDS for the container cross-calls (round 19 — performance).
2309
+ *
2310
+ * `blankQuotedCode` and `blankListItemCode` each call the other and
2311
+ * `blankListItemCode` now calls itself, and each of those calls walks the run
2312
+ * and folds the WHOLE document through `blankRanges` per flush. Widening the
2313
+ * list gate to `top >= 1` made every ordinary `- ` item open a run, so a
2314
+ * list-dense document paid that constant on every line (measured 3.1x at
2315
+ * 989 KB before these guards).
2316
+ *
2317
+ * Both guards are EXACT, not heuristic: `blankQuotedCode` only ever opens a run
2318
+ * on a line `BLOCKQUOTE_PREFIX_RE` matches and `blankListItemCode` only ever
2319
+ * pushes a column for a line `LIST_MARKER_RE` matches, so a run containing no
2320
+ * such line produces no runs at all and returns `masked` byte-identical. Skipping
2321
+ * a provable identity cannot change coverage — do NOT weaken either predicate
2322
+ * into an approximation of "probably nothing here"; that is how the eight
2323
+ * fail-open instances above were born.
2324
+ */
2325
+ const hasListMarker = (line: MaskLine): boolean => LIST_MARKER_RE.test(line.content)
2326
+ const hasQuotePrefix = (line: MaskLine): boolean => BLOCKQUOTE_PREFIX_RE.test(line.content)
2327
+
2328
+ /**
2329
+ * WINDOWED RUNS (round 19 — performance, and the same lesson as `blankRanges`
2330
+ * one level up).
2331
+ *
2332
+ * `blankRanges` is O(document): it rebuilds the whole string. The container
2333
+ * passes used to hand it the WHOLE document once per nested pass PER RUN, so a
2334
+ * document that is one long sequence of list/quote runs paid O(runs × document)
2335
+ * — a second quadratic, sitting directly above the one round 18 removed.
2336
+ * Widening the list gate to `top >= 1` tripled the run count and made it
2337
+ * visible: a 989 KB all-fenced-in-list-items document went 320 ms → 1006 ms.
2338
+ *
2339
+ * Runs are DISJOINT and ASCENDING, and every range any nested pass produces
2340
+ * lies inside its own run's span (fence ranges start at `line.start`, every
2341
+ * other pass at `line.contentStart`). So a run can be masked in ISOLATION, on a
2342
+ * window sliced out of the caller's baseline with all offsets rebased, and the
2343
+ * windows spliced back in ONE fold at the end. Same output, one document
2344
+ * rebuild per pass instead of one per run.
2345
+ *
2346
+ * Do not reintroduce a per-run fold; a container pass that reassigns the whole
2347
+ * `masked` inside its `flush` is the regression.
2348
+ */
2349
+ function rebaseRun(run: MaskLine[], from: number): MaskLine[] {
2350
+ return run.map((line) => ({
2351
+ start: line.start - from,
2352
+ contentStart: line.contentStart - from,
2353
+ content: line.content,
2354
+ }))
2355
+ }
2356
+
2357
+ function spliceWindows(masked: string, edits: Array<[number, number, string]>): string {
2358
+ if (edits.length === 0) return masked
2359
+ const parts: string[] = []
2360
+ let cursor = 0
2361
+ for (const [from, to, text] of edits) {
2362
+ if (from > cursor) parts.push(masked.slice(cursor, from))
2363
+ parts.push(text)
2364
+ cursor = to
2365
+ }
2366
+ parts.push(masked.slice(cursor))
2367
+ return parts.join('')
2368
+ }
2369
+
2370
+ /** The window a run occupies: from the first line's START (fence ranges are
2371
+ * anchored there, before any container prefix) to the last line's END. */
2372
+ function runWindow(run: MaskLine[]): [number, number] {
2373
+ const last = run[run.length - 1]
2374
+ return [run[0].start, last.contentStart + last.content.length]
2375
+ }
2376
+
2377
+ /**
2378
+ * CONTAINER NESTING DEPTH GUARD — and it FAILS CLOSED (round 19).
2379
+ *
2380
+ * The round-17 termination note claimed the mutual recursion was "verified
2381
+ * empirically on `> - ` alternation nested 1/2/5/20/100/500/2000/8000 levels
2382
+ * deep … no throw, ≤4 ms, and the observed recursion depth CAPPED AT 4". THAT
2383
+ * CLAIM IS FALSE and was false when written: HEAD throws `RangeError: Maximum
2384
+ * call stack size exceeded` on that exact input from depth ~2000 up. The
2385
+ * recursion terminates (the measure argument is sound) but its DEPTH is bounded
2386
+ * only by input length, and V8's stack is not. A `RangeError` out of the
2387
+ * sanitizer is a rendering crash, i.e. a denial of service on a 24 KB message.
2388
+ *
2389
+ * Round 19's list self-recursion widened the trigger (a single line of `- `
2390
+ * markers overflows from depth ~4000, where HEAD survived because
2391
+ * `LIST_MARKER_RE` matches only the first marker), so the guard lands here.
2392
+ *
2393
+ * THE GUARD IS NOT A COVERAGE HOLE. At the limit the run is not skipped — it is
2394
+ * BLANKED WHOLE, which is the strictly more aggressive answer and exactly the
2395
+ * fail direction this module rounds towards everywhere else. A markdown document
2396
+ * nested 64 containers deep is a code sample rendered as escaped text, not a
2397
+ * shelter. Do NOT convert this into a `return` / `continue`: skipping is the
2398
+ * fail-OPEN direction and would be a new instance of the class.
2399
+ */
2400
+ const CONTAINER_NEST_LIMIT = 64
2401
+
2402
+ /**
2403
+ * Blank code regions inside BLOCKQUOTES.
2404
+ *
2405
+ * `FENCE_RE` matches at column 0..3, so a fence inside a quote (```` > ```html ````)
2406
+ * is invisible to the top-level tracker — and a blockquoted code sample is an
2407
+ * utterly ordinary chat answer ("here's the markup:" followed by a quoted
2408
+ * fence). The closer inside it satisfied `hasLaterCloser` and the prose opener
2409
+ * above stayed live.
2410
+ *
2411
+ * CHOSEN APPROACH: strip the quote prefix off each run of quoted lines and run
2412
+ * a NESTED tracker (plus the indented-code rule) over the stripped content,
2413
+ * blanking only the code regions found. The blunter alternative — blank every
2414
+ * `^ {0,3}>` line — is also sound (it only over-blanks) but it would escape a
2415
+ * legitimately PAIRED `<textarea>…</textarea>` written inside a blockquote,
2416
+ * turning quoted HTML into visible `&lt;…&gt;` source. The nested scan costs
2417
+ * one extra line walk and keeps that shape rendering.
2418
+ *
2419
+ * ---------------------------------------------------------------------------
2420
+ * MUTUAL RECURSION — TERMINATION (round 17)
2421
+ * ---------------------------------------------------------------------------
2422
+ * `blankQuotedCode` and `blankListItemCode` now call EACH OTHER (the missing
2423
+ * quote→list direction was the seventh instance of the fail-open class). The
2424
+ * recursion terminates on the measure `M(run) = Σ line.content.length`:
2425
+ *
2426
+ * - `blankQuotedCode` only puts a line in a run when `BLOCKQUOTE_PREFIX_RE`
2427
+ * matches, and that pattern is `(?: {0,3}>[ \t]?)+` — at least one `>`, so
2428
+ * the stripped content is at least 1 char SHORTER. Blank lines never match
2429
+ * (they carry no `>`), so EVERY line in a quoted run strictly shortens.
2430
+ * - `blankListItemCode` only puts a line in a run when the content column
2431
+ * `top >= 1` (round 19 — was `>= 4`), and `charIndexAtColumn(content, top)`
2432
+ * with `top >= 1` returns an index `>= 1` (it can only return 0 when the
2433
+ * requested column is 0), so that line strictly shortens too. ROUND-19
2434
+ * RE-VERIFICATION: the same bound covers the new SELF-recursion — the run it
2435
+ * hands itself is cut at the same `top >= 1`, so `M` strictly decreases
2436
+ * across that call exactly as across the `blankQuotedCode` one. ROUND-18
2437
+ * RE-VERIFICATION: this also
2438
+ * covers the MARKER LINE, whose cut lands at `marker[0].length` (or
2439
+ * `markerEnd + 1` under the clamp) — both `>= 2` for every marker spelling,
2440
+ * so the bound `cut >= 1` is unchanged and the measure still strictly
2441
+ * decreases. The reorder moved WHICH lines join a run, not the shortening
2442
+ * property that makes the recursion finite. It also carries blank separators into an
2443
+ * ALREADY-OPEN run as `content: ''` (length 0 ≤ original), and a run is only
2444
+ * ever opened by a non-blank, strictly-shortened line.
2445
+ *
2446
+ * So each nested call is handed a run whose measure is strictly smaller than
2447
+ * the caller's, `M` is a non-negative integer, and the chain is finite.
2448
+ *
2449
+ * ---------------------------------------------------------------------------
2450
+ * FINITE IS NOT THE SAME AS SHALLOW (round 19 — the third false claim)
2451
+ * ---------------------------------------------------------------------------
2452
+ * Round 17 concluded here: "It is bounded by input length, so no depth guard is
2453
+ * added — there is no non-shortening case to guard against, and a speculative
2454
+ * bound would be a second, untested policy. Verified empirically on `> - `
2455
+ * alternation nested 1/2/5/20/100/500/2000/8000 levels deep (240 KB source):
2456
+ * length invariant held, no throw, ≤4 ms, and the observed recursion depth
2457
+ * CAPPED AT 4 regardless of nesting."
2458
+ *
2459
+ * THE EMPIRICAL PART OF THAT IS FALSE, and was false when written. Re-run on the
2460
+ * described input, HEAD raises `RangeError: Maximum call stack size exceeded`
2461
+ * from depth ~2000 up — a 24 KB message crashes the renderer. The depth cap of 4
2462
+ * held only for the shapes round 17 happened to try; `BLOCKQUOTE_PREFIX_RE`
2463
+ * consumes a `> > >` nest in one match, but an ALTERNATING `> - > - …` line
2464
+ * gives each pass exactly one level to strip and the chain is as deep as the
2465
+ * line is long. Round 19's list self-recursion widened it further (a plain `- `
2466
+ * run overflows from ~4000, where HEAD survived only because `LIST_MARKER_RE`
2467
+ * matches the first marker alone).
2468
+ *
2469
+ * Termination was never the property at risk — STACK DEPTH was, and "bounded by
2470
+ * input length" is precisely the bound that does not help. `CONTAINER_NEST_LIMIT`
2471
+ * now caps it, blanking an over-deep run WHOLE rather than recursing, which is
2472
+ * fail-CLOSED and therefore not a coverage hole. Pinned by
2473
+ * `masks arbitrarily deep container nesting without throwing` at depths up to
2474
+ * 40000 (469 KB, 11 ms, closer masked at every depth).
2475
+ */
2476
+ function blankQuotedCode(masked: string, lines: MaskLine[], depth = 0): string {
2477
+ let run: MaskLine[] = []
2478
+ // One edit per run, spliced in a SINGLE fold at the end — see `spliceWindows`.
2479
+ const edits: Array<[number, number, string]> = []
2480
+ const flush = () => {
2481
+ if (run.length === 0) return
2482
+ const [from, to] = runWindow(run)
2483
+ const wl = rebaseRun(run, from)
2484
+ let win = masked.slice(from, to)
2485
+ // Depth limit: blank the run WHOLE rather than recurse — see
2486
+ // `CONTAINER_NEST_LIMIT`. Fail-closed, never a skip.
2487
+ if (depth >= CONTAINER_NEST_LIMIT) {
2488
+ edits.push([from, to, blankRanges(win, [[0, win.length]])])
2489
+ run = []
2490
+ return
2491
+ }
2492
+ // The nested FENCE scan needs the unmasked `run` content (the inline-code
2493
+ // pass would have blinded it), but the nested INDENTED scan needs the CURRENT
2494
+ // mask — see `blankIndentedCode`'s SCAN-SOURCE INVERSION.
2495
+ const afterFences = blankFencedRegions(win, wl)
2496
+ win = blankIndentedCode(afterFences, remapToMask(afterFences, wl))
2497
+ win = blankLinkDefinitions(win, wl)
2498
+ // …and the LIST-container pass, mirroring the call `blankListItemCode`
2499
+ // already makes in the other direction. Without it a fenced sample inside a
2500
+ // LIST ITEM inside a QUOTE was seen by NO pass: `FENCE_RE` caps fence indent
2501
+ // at 3 ABSOLUTE columns, so at a quote-relative content column >= 4
2502
+ // (`> 1. ` / `> - ` / `> -\t`) the fence is invisible to the nested
2503
+ // tracker, and `blankIndentedCode`'s list-aware threshold (`contentCol + 4`)
2504
+ // starts at 8 and never reaches it either. Reproduced live for textarea and
2505
+ // iframe, at both list spellings, the tab spelling and depth-2 quotes;
2506
+ // `escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL.
2507
+ if (wl.some(hasListMarker)) win = blankListItemCode(win, wl, depth + 1)
2508
+ edits.push([from, to, win])
2509
+ run = []
2510
+ }
2511
+ for (const line of lines) {
2512
+ const prefix = BLOCKQUOTE_PREFIX_RE.exec(line.content)
2513
+ if (!prefix) {
2514
+ flush()
2515
+ continue
2516
+ }
2517
+ run.push({
2518
+ start: line.start,
2519
+ contentStart: line.contentStart + prefix[0].length,
2520
+ content: line.content.slice(prefix[0].length),
2521
+ })
2522
+ }
2523
+ flush()
2524
+ return spliceWindows(masked, edits)
2525
+ }
2526
+
2527
+ /**
2528
+ * Blank code regions nested inside LIST ITEMS, the list-container analogue of
2529
+ * `blankQuotedCode` (round 14).
2530
+ *
2531
+ * `FENCE_RE` caps fence indent at 3 columns ABSOLUTE, but CommonMark measures a
2532
+ * fence's indent from the enclosing item's CONTENT COLUMN. Every list wrapper
2533
+ * the corpus swept had a content column of 2 or 3 (`- `, `1. `), so the cap
2534
+ * happened to cover them and the gap was invisible; at content column 4 or more
2535
+ * — `-` + three spaces, `1.` + three spaces, or the TAB spelling `-\t`, all
2536
+ * ordinary ways to write a list — a fenced code sample inside the item is seen
2537
+ * by NO pass. Its `</textarea>` then satisfied `hasLaterCloser`, and a prose
2538
+ * `<textarea>` above stayed LIVE and swallowed the rest of the message
2539
+ * (reproduced end-to-end at content columns 4 and 5 in BOTH the space and tab
2540
+ * spellings; `escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL). The
2541
+ * same hole covers a BLOCKQUOTED fence inside such an item, since
2542
+ * `blankQuotedCode`'s own `BLOCKQUOTE_PREFIX_RE` is likewise anchored at
2543
+ * columns 0..3.
2544
+ *
2545
+ * Same shape as `blankQuotedCode`: strip the container prefix off each run of
2546
+ * lines that share a content column, then run the nested fence + indented scan
2547
+ * over the stripped content. The content-column stack is the one
2548
+ * `blankIndentedCode` keeps, INCLUDING CommonMark's `markerEnd + 1` clamp and
2549
+ * tab expansion, so the two passes cannot disagree about where an item's
2550
+ * content begins.
2551
+ *
2552
+ * FAIL DIRECTION: monotonic. Every pass it calls only ever blanks MORE of the
2553
+ * haystack, and more blanking means fewer visible closers, means more openers
2554
+ * escaped. So an over-detected run (a list marker written inside a fence
2555
+ * pushing a bogus column — the SCAN-SOURCE INVERSION `blankIndentedCode`
2556
+ * documents) costs at most a code sample rendered as escaped text.
2557
+ */
2558
+ function blankListItemCode(masked: string, lines: MaskLine[], depth = 0): string {
2559
+ const cols: number[] = []
2560
+ let run: MaskLine[] = []
2561
+ let runCol = 0
2562
+ // One edit per run, spliced in a SINGLE fold at the end — see `spliceWindows`.
2563
+ const edits: Array<[number, number, string]> = []
2564
+ const flush = () => {
2565
+ if (run.length === 0) return
2566
+ const [from, to] = runWindow(run)
2567
+ const wl = rebaseRun(run, from)
2568
+ let win = masked.slice(from, to)
2569
+ // Depth limit: blank the run WHOLE rather than recurse — see
2570
+ // `CONTAINER_NEST_LIMIT`. Fail-closed, never a skip.
2571
+ if (depth >= CONTAINER_NEST_LIMIT) {
2572
+ edits.push([from, to, blankRanges(win, [[0, win.length]])])
2573
+ run = []
2574
+ return
2575
+ }
2576
+ // The nested FENCE scan needs unmasked content; the nested INDENTED scan
2577
+ // needs the CURRENT mask — exactly `blankQuotedCode`'s split. The nested
2578
+ // QUOTED scan is needed too: `BLOCKQUOTE_PREFIX_RE` is anchored at columns
2579
+ // 0..3, so a quoted fence inside a column-4 item was missed by BOTH
2580
+ // containers' passes (`bq-in-col4-item`, reproduced live).
2581
+ const afterFences = blankFencedRegions(win, wl)
2582
+ win = blankIndentedCode(afterFences, remapToMask(afterFences, wl))
2583
+ win = blankLinkDefinitions(win, wl)
2584
+ if (wl.some(hasQuotePrefix)) win = blankQuotedCode(win, wl, depth + 1)
2585
+ // …and ITSELF, the symmetric counterpart of the `blankQuotedCode →
2586
+ // blankListItemCode` call above (round 19 — tenth instance of the fail-open
2587
+ // class). `LIST_MARKER_RE` is anchored at `^` and matches only the FIRST
2588
+ // marker on a line, so an INNER item's content column was never pushed and
2589
+ // a block opened on a nested marker line (`- - ```html`, `- - [a]: /x
2590
+ // "</textarea>"`) was cut to the OUTER item's column only — still short of
2591
+ // its own. Reproduced live for both shapes at zero quote depth
2592
+ // (`escapeUnknownHtmlTags` byte-identical, one live `<textarea>`, the
2593
+ // document below swallowed). Re-cutting the stripped run re-runs
2594
+ // `LIST_MARKER_RE` against content that now BEGINS at the outer item's
2595
+ // column, so the inner marker is the first one and its column is pushed.
2596
+ //
2597
+ // TERMINATION (self-recursion): every line put in a run is cut at
2598
+ // `charIndexAtColumn(content, top)` with `top >= 1`, which returns an index
2599
+ // `>= 1` (index 0 is only reachable for column 0), so EVERY member of the
2600
+ // run is strictly shorter than the line it came from. The measure
2601
+ // `M(run) = Σ line.content.length` from `blankQuotedCode`'s proof therefore
2602
+ // strictly decreases across this call exactly as it does across the
2603
+ // `blankQuotedCode` one — blank separators enter an already-open run as
2604
+ // `content: ''` (length 0 ≤ original) and never open one. `M` is a
2605
+ // non-negative integer, so the chain is finite; a run with no marker at all
2606
+ // pushes no column, leaves `top === 0`, opens no run and the recursion stops
2607
+ // one level down — which is exactly what `hasListMarker` short-circuits.
2608
+ if (wl.some(hasListMarker)) win = blankListItemCode(win, wl, depth + 1)
2609
+ edits.push([from, to, win])
2610
+ run = []
2611
+ }
2612
+ for (const line of lines) {
2613
+ // A blank line does not close a list item, so it stays in the run — the
2614
+ // nested tracker needs it to see the paragraph break. CommonMark's blank
2615
+ // line, not `trim()` (see `isBlankLine`).
2616
+ if (isBlankLine(line.content)) {
2617
+ if (run.length > 0) run.push({ ...line, content: '' })
2618
+ continue
2619
+ }
2620
+ const indent = leadingIndent(line.content)
2621
+ while (cols.length > 0 && indent < cols[cols.length - 1]) cols.pop()
2622
+ // THE MARKER LINE IS ITSELF ITEM CONTENT (round 18 — eighth instance of the
2623
+ // fail-open class). The marker used to be pushed AFTER the run-membership
2624
+ // decision, so `top` was read from the enclosing state and the marker line
2625
+ // NEVER entered a run — the run began on the line BELOW it. A block opened
2626
+ // ON the marker line (`- ```html`, `1. ```html`, `-\t```html`,
2627
+ // `> - ```html`) was therefore seen by no pass at all: `FENCE_RE` caps
2628
+ // fence indent at 3 ABSOLUTE columns so the top-level tracker misses it, and
2629
+ // `blankIndentedCode`'s `contentCol + 4` threshold overshoots it. Worse, the
2630
+ // run then STARTED after the opener, so the item's CLOSING fence read as an
2631
+ // `open` to the nested tracker, which blanked to EOF while leaving the code
2632
+ // BODY — and its `</textarea>` — live in the haystack. Reproduced at ZERO
2633
+ // nesting depth (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL,
2634
+ // one live editable `<textarea>` swallowing the prose above it), for both
2635
+ // textarea and iframe and at every marker spelling.
2636
+ //
2637
+ // Pushing the marker first makes `top` the column this line's own content
2638
+ // starts at, so `charIndexAtColumn(content, top)` cuts exactly at the marker
2639
+ // (`marker[0].length`, or `markerEnd + 1` under the clamp) and hands the
2640
+ // nested tracker precisely the item content.
2641
+ //
2642
+ // TERMINATION IS UNCHANGED: `top >= 4` still implies `cut >= 1` (the cut
2643
+ // index can only be 0 when the requested column is 0), so every line put in
2644
+ // a run still strictly shortens and the measure `M(run)` in
2645
+ // `blankQuotedCode`'s termination proof still strictly decreases.
2646
+ const marker = LIST_MARKER_RE.exec(line.content)
2647
+ if (marker) {
2648
+ const markerEndCol = visualColumn(line.content, marker[0].length - marker[2].length)
2649
+ const contentCol = visualColumn(line.content, marker[0].length)
2650
+ cols.push(contentCol - markerEndCol > 4 ? markerEndCol + 1 : contentCol)
2651
+ }
2652
+ const top = cols.length > 0 ? cols[cols.length - 1] : 0
2653
+ if (top !== runCol) {
2654
+ flush()
2655
+ runCol = top
2656
+ }
2657
+ // EVERY list item is re-scanned at its own content column (round 19 — ninth
2658
+ // instance of the fail-open class). The gate used to be `top >= 4`, on the
2659
+ // claim that "below column 4 the top-level passes already cover the line at
2660
+ // the right column". That is true for a CONTINUATION line — its absolute
2661
+ // indent of 2 or 3 falls inside `FENCE_RE`'s 0..3 cap — and FALSE for the
2662
+ // MARKER LINE, which the top-level passes examine only at column 0, where
2663
+ // the leading `- ` / `1. ` is not whitespace so `FENCE_RE` cannot match and
2664
+ // `blankIndentedCode`'s `contentCol + 4` overshoots. At content column 2 or
2665
+ // 3 — `- ` and `1. `, the two MOST COMMON spellings — the gate then denied
2666
+ // the marker line any run at all and no pass examined it. Reproduced live
2667
+ // for `- ```html` / `1. ```html` / the `<iframe>` spelling, EOF-terminated
2668
+ // (`escapeUnknownHtmlTags` byte-identical, one live RAWTEXT element, the
2669
+ // document below swallowed); the closed-fence spelling is rescued only
2670
+ // INCIDENTALLY by `findInlineCodeRanges` matching the two backtick runs.
2671
+ //
2672
+ // Round 18 fixed WHERE the column is pushed (the marker line now joins the
2673
+ // run) but kept a gate whose justification was untrue one column-range
2674
+ // lower. The gate was an unforced optimization: the pass is documented
2675
+ // monotonic, so re-scanning narrow items can only blank MORE, and
2676
+ // termination is unaffected (`top >= 1` still implies `cut >= 1`).
2677
+ if (top >= 1) {
2678
+ const cut = charIndexAtColumn(line.content, top)
2679
+ if (cut < 0) {
2680
+ flush()
2681
+ runCol = 0
2682
+ } else {
2683
+ run.push({
2684
+ start: line.start,
2685
+ contentStart: line.contentStart + cut,
2686
+ content: line.content.slice(cut),
2687
+ })
2688
+ }
2689
+ }
2690
+ }
2691
+ flush()
2692
+ return spliceWindows(masked, edits)
2693
+ }
2694
+
2695
+ /**
2696
+ * Build the haystack `hasLaterCloser` searches: a LENGTH-PRESERVING lowercased
2697
+ * copy of the document with every region that cannot contain a REAL closing
2698
+ * tag blanked to spaces.
2699
+ *
2700
+ * Why this exists: the escaping pass carefully carves code out, but the
2701
+ * closer search used to run over the RAW document. So a `</textarea>` sitting
2702
+ * inside a code fence, an inline-code span, or another tag's attribute string
2703
+ * satisfied "is closed later", the prose opener was left LIVE, and parse5's
2704
+ * RAWTEXT span swallowed the rest of the message anyway — the whole fix was
2705
+ * one code sample away from being bypassed, which is exactly what an LLM
2706
+ * answer about HTML looks like.
2707
+ *
2708
+ * Masking (rather than deleting) keeps every index identical to the original
2709
+ * string, so the caller's offset arithmetic is unchanged. THE LENGTH
2710
+ * INVARIANT IS LOAD-BEARING — see `foldAsciiCase`.
2711
+ *
2712
+ * CARVE DECISION (deliberate, do not "unify"): these tracker-derived regions
2713
+ * are NOT fed to the escaping carve, even though that would stop an authored
2714
+ * EOF-terminated fence body from rendering as literal `&lt;their&gt;`.
2715
+ *
2716
+ * The genuine asymmetry is the EOF-TERMINATED fence, and only that one. The
2717
+ * tracker protects an unclosed opener all the way to end of input, so a single
2718
+ * stray ``` line — mid-stream, or inside an open raw-HTML block where a ```
2719
+ * line is content rather than a fence — would carve the ENTIRE remainder of the
2720
+ * document out of the escaping pass. `PROTECTED_SPAN_RE` protects nothing at
2721
+ * all there (it only recognizes a fence CLOSED by a same-marker run), so its
2722
+ * failure mode is bounded: a code sample renders as escaped text. In the carve
2723
+ * an over-detected region is a region that is NOT escaped — a fail-OPEN, i.e.
2724
+ * exactly the swallow this module exists to prevent — so the materially larger
2725
+ * fail-open surface decides it.
2726
+ *
2727
+ * SHARED over-detection (e.g. a ``` line inside an HTML block — `<div>`,
2728
+ * `<pre>`, `<details>` — where CommonMark says the line is HTML content, not a
2729
+ * fence) was previously dismissed here as "not an argument either way". THAT
2730
+ * WAS WRONG: it is precisely the residual fail-open. The intersection guard
2731
+ * below only reconciles DISAGREEMENT, so when BOTH engines open the same bogus
2732
+ * fence the guard is a no-op and a live `<textarea>` inside it is pushed
2733
+ * verbatim, swallowing the rest of the message (reproduced for all three tags).
2734
+ * What actually closes it is the CARVE BALANCE GUARD in
2735
+ * `escapeUnknownHtmlTags`: a protected span may contain no UNBALANCED RAWTEXT
2736
+ * opener. Neither engine needs to learn about HTML blocks for that to hold.
2737
+ *
2738
+ * What makes keeping two engines SAFE is therefore the pair of guards in
2739
+ * `escapeUnknownHtmlTags`: a carve span the mask did not blank is escaped
2740
+ * rather than pushed through verbatim, and a span carrying an unbalanced
2741
+ * RAWTEXT opener is escaped even when both engines agree. Over-detection can
2742
+ * then only cost cosmetics. Before the first guard the regex's info-string-tolerant
2743
+ * closer let it desync and open a span from a line CommonMark treats as
2744
+ * ordinary text, sheltering a live `<textarea>` from escaping entirely
2745
+ * (`mismatched-fence-carve-does-not-shelter-opener`). The remaining tradeoff is
2746
+ * pinned by `unclosed-fence-body-renders-escaped` rather than left as prose.
2747
+ */
2748
+ function buildCloserHaystack(text: string): string {
2749
+ const folded = foldAsciiCase(text)
2750
+ const lines = toMaskLines(folded)
2751
+ // 1. Inline code spans (the only non-line-state code region). Uncapped and
2752
+ // backtracking-free — see `findInlineCodeRanges`; an over-cap span used to
2753
+ // be skipped entirely and sheltered a live RAWTEXT opener.
2754
+ let masked = blankRanges(folded, findInlineCodeRanges(folded))
2755
+ // 2. Every BLOCK-level code form, derived from line state over `folded`:
2756
+ // fences (tracker-accurate, closed and EOF-terminated alike), indented
2757
+ // code, blockquoted code, and HTML comments. Each of these carried a
2758
+ // reproduced live-textarea swallow before it was masked.
2759
+ masked = blankFencedRegions(masked, lines)
2760
+ // The indented pass walks the CURRENT mask (not `folded`) so a list marker
2761
+ // written inside a fence cannot shift its content-column stack — see its
2762
+ // SCAN-SOURCE INVERSION note. `blankQuotedCode` still gets the unmasked
2763
+ // lines because its NESTED fence scan needs them, and applies the same
2764
+ // inversion internally.
2765
+ masked = blankIndentedCode(masked, remapToMask(masked, lines))
2766
+ // …and LINK REFERENCE DEFINITIONS, which remark consumes whole and emits
2767
+ // nothing for, so a `</textarea>` in a destination or title is not a closer.
2768
+ masked = blankLinkDefinitions(masked, lines)
2769
+ // …and the GFM FOOTNOTE definitions that pass deliberately refuses, but only
2770
+ // the UNREFERENCED ones: remark-gfm drops those whole, so their bodies are
2771
+ // not document text either. It reads the CURRENT mask for DEFINITIONS and a
2772
+ // separate, more-blanked copy for REFERENCES (`footnoteReferenceMask`), both
2773
+ // behind a `[^` guard.
2774
+ //
2775
+ // ITS SLOT IS CONSTRAINED ON BOTH SIDES, and neither bound is cosmetic:
2776
+ // · it may not run EARLIER than the code passes, whose output is the
2777
+ // definition source;
2778
+ // · it may not simply be MOVED after `blankInlineLinkPayloads` /
2779
+ // `blankBracketLabels` to pick up the phantom-reference fix, because
2780
+ // `blankBracketLabels` blanks footnote labels "reference AND definition
2781
+ // alike" — after it, EVERY reference is gone and every referenced
2782
+ // definition would be over-blanked into escaped source. Hence the
2783
+ // separate scratch copy instead of a reorder.
2784
+ masked = blankUnreferencedFootnotes(masked, lines, folded)
2785
+ // …and the INLINE link/image spelling of the same shelter, which remark
2786
+ // likewise turns into href/title attributes. Container-agnostic, so like
2787
+ // the definition pass it needs exactly one top-level call.
2788
+ masked = blankInlineLinkPayloads(masked, folded)
2789
+ // …and the BRACKET half of that same class — an image's alt, a reference
2790
+ // label, a footnote label — which remark consumes into an attribute or an
2791
+ // identifier. Container-agnostic, so likewise exactly one top-level call.
2792
+ masked = blankBracketLabels(masked, folded)
2793
+ masked = blankQuotedCode(masked, lines)
2794
+ // …and the LIST-container analogue, for items whose content column exceeds
2795
+ // the 3-column fence-indent cap. Monotonic, so its position among the
2796
+ // block passes is not load-bearing.
2797
+ masked = blankListItemCode(masked, lines)
2798
+ // Comments scan the MASKED copy, not `folded` — see `blankComments`. Must
2799
+ // stay LAST: it relies on every code region already being blanked.
2800
+ masked = blankComments(masked, masked)
2801
+ // 3. Attribute regions. Blanking the WHOLE tag would blank real `</tag>`
2802
+ // closers too (and break the closed-form fixtures), so only the
2803
+ // attribute run between the tag name and the `>` is cleared.
2804
+ masked = blankTagAttributes(masked)
2805
+ return masked
2806
+ }
2807
+
2808
+ /** Exported for the length-preservation invariant test only. */
2809
+ export const __buildCloserHaystackForTest = buildCloserHaystack
2810
+
2811
+ /**
2812
+ * True when a well-formed `</tag>` (optional trailing whitespace) occurs at
2813
+ * or after `from` in the MASKED lowercased source (see `buildCloserHaystack`).
2814
+ * Substring search rather than a per-tag `RegExp` — the tag comes from
2815
+ * `RAWTEXT_TAGS`, but building regexes from tag names in a hot path invites
2816
+ * an injection footgun on the next edit.
2817
+ */
2818
+ function hasLaterCloser(lowerSource: string, tag: string, from: number): boolean {
2819
+ const needle = `</${tag}`
2820
+ let cursor = from
2821
+ for (;;) {
2822
+ const at = lowerSource.indexOf(needle, cursor)
2823
+ if (at === -1) return false
2824
+ // Only `</tag>` or `</tag >` closes it; `</tagfoo>` is a different tag.
2825
+ if (/^\s*>/.test(lowerSource.slice(at + needle.length, at + needle.length + 64)))
2826
+ return true
2827
+ cursor = at + needle.length
2828
+ }
2829
+ }
2830
+
2831
+ /**
2832
+ * True when the mask considers `[from, to)` entirely code — every character
2833
+ * blanked to a space (newlines are never blanked, so they count as blank).
2834
+ * Both strings are the same length by construction (see `foldAsciiCase`).
2835
+ */
2836
+ function isMaskedBlank(lowerSource: string, from: number, to: number): boolean {
2837
+ for (let i = from; i < to; i++) {
2838
+ const c = lowerSource[i]
2839
+ if (c !== ' ' && c !== '\n') return false
2840
+ }
2841
+ return true
2842
+ }
2843
+
2844
+ /**
2845
+ * True when `span` contains a RAWTEXT opener with no matching closer INSIDE
2846
+ * the span — the self-containment test the carve applies before pushing a
2847
+ * protected span through verbatim. See the CARVE BALANCE GUARD in
2848
+ * `escapeUnknownHtmlTags`.
2849
+ *
2850
+ * A closer with no opener before it is harmless (it cannot start a RAWTEXT
2851
+ * span), so the counter floors at zero rather than going negative.
2852
+ *
2853
+ * SELF-CLOSING IS AN OPENER (round 11). HTML ignores the self-closing flag on
2854
+ * non-void, non-foreign elements, so parse5 tokenizes `<textarea/>` as a START
2855
+ * tag and enters RAWTEXT exactly like `<textarea>`. Keying on `selfClose === ''`
2856
+ * therefore made this guard — and the closer check in `escapeOutsideFences` —
2857
+ * blind to the self-closed spelling of EVERY shape they defend against; the
2858
+ * round-9 HTML-block fixtures passed only because they used the bare spelling.
2859
+ * See the matching note on `escapeOutsideFences` for the one cosmetic cost.
2860
+ */
2861
+ function hasUnbalancedRawtextOpener(span: string): boolean {
2862
+ if (span.indexOf('<') === -1) return false
2863
+ const open = new Map<string, number>()
2864
+ TAG_LIKE_REGEX.lastIndex = 0
2865
+ let m: RegExpExecArray | null
2866
+ while ((m = TAG_LIKE_REGEX.exec(span)) !== null) {
2867
+ const [, slash, tag] = m
2868
+ const lower = tag.toLowerCase()
2869
+ if (!RAWTEXT_TAGS.has(lower)) continue
2870
+ if (slash === '') {
2871
+ open.set(lower, (open.get(lower) ?? 0) + 1)
2872
+ } else {
2873
+ open.set(lower, Math.max(0, (open.get(lower) ?? 0) - 1))
2874
+ }
2875
+ }
2876
+ for (const count of open.values()) if (count > 0) return true
2877
+ return false
2878
+ }
2879
+
2880
+ /**
2881
+ * ---------------------------------------------------------------------------
2882
+ * CommonMark HTML BLOCK ranges — the property the CARVE BALANCE GUARD gates on
2883
+ * ---------------------------------------------------------------------------
2884
+ * The guard exists because a protected span sitting inside an HTML BLOCK is not
2885
+ * really code: CommonMark says an HTML block runs to its own terminator, so
2886
+ * every line inside it is HTML CONTENT. Round 9 discovered that through the
2887
+ * FENCE spelling (a ``` line inside `<div>` is content, but both fence engines
2888
+ * call it a fence and shelter what follows). Round 11 then scoped the guard to
2889
+ * fences — and reopened the identical hole through INLINE CODE, whose
2890
+ * "an inline span can shelter nothing, remark emits it as an `inlineCode` TEXT
2891
+ * node" justification is precisely the invariant that fails inside an HTML
2892
+ * block, where remark emits raw HTML and backticks are not code at all.
2893
+ *
2894
+ * Gating on the span's FLAVOR was therefore the wrong property in both
2895
+ * directions. This walk supplies the right one: HTML-block membership, which
2896
+ * covers both spellings, while `` Use the `<title>` element `` in ordinary
2897
+ * prose keeps rendering verbatim (round 11's regression stays fixed).
2898
+ *
2899
+ * FAIL DIRECTION: a detected range only makes the guard ESCAPE a span, and
2900
+ * escaping inside a GENUINE HTML block is invisible (the surrounding content is
2901
+ * raw HTML, where `&lt;` is decoded as `<`). Over-detection is therefore
2902
+ * cosmetic ONLY when we are wrong about the block — so the walk tracks
2903
+ * CommonMark closely rather than blanket-detecting.
2904
+ *
2905
+ * START CONDITIONS IMPLEMENTED: all seven (1 `<script|pre|style|textarea`,
2906
+ * 2 `<!--`, 3 `<?`, 4 `<!LETTER`, 5 `<![CDATA[`, 6 the known block-tag list,
2907
+ * 7 a complete open/closing tag ALONE on its line). Condition 7 is the one that
2908
+ * needs paragraph state — it alone cannot interrupt a paragraph — and it is NOT
2909
+ * omissible: `<span>` is outside both the type-1 and type-6 tag lists, so
2910
+ * dropping 7 would leave `` <span>\n`<textarea>`\n</span> `` sheltering a live
2911
+ * opener (verified end-to-end before this walk existed). Paragraph state is
2912
+ * approximated by "the previous line was ordinary text", which is exact for the
2913
+ * shapes 7 cares about; where it errs it errs toward NOT being in a paragraph,
2914
+ * i.e. toward detecting a block, i.e. toward escaping.
2915
+ *
2916
+ * This walk is deliberately SEPARATE from the mask's line walk. The mask is the
2917
+ * closer-search security boundary and currently over-blanks a ``` line inside an
2918
+ * HTML block (fail-CLOSED there); teaching it about HTML blocks would UNBLANK
2919
+ * that region and turn a code-sample `</textarea>` into a live closer — the
2920
+ * fail-OPEN direction. Same line-state concept, opposite fail directions, so
2921
+ * they stay two walks.
2922
+ */
2923
+ interface HtmlBlockRange {
2924
+ start: number
2925
+ end: number
2926
+ }
2927
+
2928
+ /** CommonMark start-condition 6 tag list (verbatim from the spec). */
2929
+ const HTML_BLOCK_TYPE_6_TAGS = new Set([
2930
+ 'address', 'article', 'aside', 'base', 'basefont', 'blockquote', 'body',
2931
+ 'caption', 'center', 'col', 'colgroup', 'dd', 'details', 'dialog', 'dir',
2932
+ 'div', 'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form',
2933
+ 'frame', 'frameset', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'head', 'header',
2934
+ 'hr', 'html', 'iframe', 'legend', 'li', 'link', 'main', 'menu', 'menuitem',
2935
+ 'nav', 'noframes', 'ol', 'optgroup', 'option', 'p', 'param', 'search',
2936
+ 'section', 'summary', 'table', 'tbody', 'td', 'tfoot', 'th', 'thead',
2937
+ 'title', 'tr', 'track', 'ul',
2938
+ ])
2939
+
2940
+ const HTML_BLOCK_START_1 = /^ {0,3}<(?:script|pre|style|textarea)(?:[ \t>]|\r?$)/i
2941
+ const HTML_BLOCK_END_1 = /<\/(?:script|pre|style|textarea)>/i
2942
+ const HTML_BLOCK_START_2 = /^ {0,3}<!--/
2943
+ const HTML_BLOCK_START_3 = /^ {0,3}<\?/
2944
+ const HTML_BLOCK_START_4 = /^ {0,3}<![a-zA-Z]/
2945
+ const HTML_BLOCK_START_5 = /^ {0,3}<!\[CDATA\[/
2946
+ const HTML_BLOCK_START_6 = /^ {0,3}<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})(?:[ \t]|\/?>|\r?$)/
2947
+ /** A COMPLETE open or closing tag, alone on its line. Attribute run bounded for
2948
+ * the same ReDoS reason as `TAG_LIKE_REGEX`. */
2949
+ const HTML_BLOCK_START_7 =
2950
+ /^ {0,3}(?:<[a-zA-Z][a-zA-Z0-9-]{0,63}(?:\s[^>]{0,4096}?)?\/?>|<\/[a-zA-Z][a-zA-Z0-9-]{0,63}[ \t]{0,64}>)[ \t]*\r?$/
2951
+ /** Lines that are NOT ordinary paragraph text (so condition 7 may start after
2952
+ * them). A lone tag line is deliberately ABSENT: under CommonMark it cannot
2953
+ * interrupt a paragraph, so it continues one. */
2954
+ const NON_PARAGRAPH_LINE_RE =
2955
+ /^ {0,3}(?:#{1,6}(?:[ \t]|\r?$)|>|[-*+](?:[ \t]|\r?$)|\d{1,9}[.)](?:[ \t]|\r?$)|`{3,}|~{3,}|=+[ \t]*\r?$|(?:[-*_][ \t]*){3,}\r?$)/
2956
+
2957
+ /**
2958
+ * CONTAINER NORMALIZATION (round 13). Every start/end condition above is
2959
+ * anchored `^ {0,3}<…` and used to be matched against the RAW line, so inside a
2960
+ * BLOCKQUOTE or a LIST ITEM none of them ever fired — `> <div>` / `- <div>`
2961
+ * looked like ordinary text. CommonMark opens the block INSIDE the container, so
2962
+ * every following line is HTML content; the walk missed the whole range, the
2963
+ * balance guard stayed blind, and `` > `<textarea>` `` was pushed through
2964
+ * VERBATIM (`escapeUnknownHtmlTags` returned the input byte-identical).
2965
+ *
2966
+ * Detection here only ever causes ESCAPING, so a CONSERVATIVE strip is enough
2967
+ * and no container-stack model is needed: stripping more than CommonMark would
2968
+ * can only over-detect, and over-detection inside a genuine HTML block is
2969
+ * invisible (see FAIL DIRECTION above), while under-detection is the swallow.
2970
+ *
2971
+ * WHAT THIS MODELS: any run of blockquote markers (`>` with up to 3 spaces of
2972
+ * indent and one optional space after), then at most one list marker
2973
+ * (`-`/`*`/`+`/`1.`/`1)` plus its following spaces), then — for CONTINUATION
2974
+ * lines — up to `listContentCol` columns of leading whitespace, where
2975
+ * `listContentCol` is the width of the most recent list marker seen at the
2976
+ * current level.
2977
+ *
2978
+ * WHAT IT DOES NOT MODEL, and why the residual is fail-CLOSED:
2979
+ * - It keeps NO container stack, so it cannot tell a lazy-continuation line
2980
+ * from a line that genuinely left the container, and it does not verify that
2981
+ * a stripped prefix matches the prefix the enclosing block actually opened
2982
+ * with. Both errors strip TOO MUCH, i.e. detect MORE blocks, i.e. escape.
2983
+ * - `listContentCol` takes the literal marker width and does NOT apply
2984
+ * CommonMark's clamp to `markerEnd + 1` when the first block starts more than
2985
+ * 4 spaces after the marker. Under `-` + six spaces the real content column
2986
+ * is 2 and the remainder is indented code INSIDE the item; we strip 7 and may
2987
+ * call an indented-code line a block start. Again: more detection.
2988
+ * - TERMINATION strips by prefix WIDTH, not by prefix IDENTITY (see the
2989
+ * `blank` computation): a line carrying a different container's marker
2990
+ * within the opening line's prefix width and nothing after it still reads as
2991
+ * blank. That shape is a lone container marker at or left of the opening
2992
+ * content column, which under CommonMark closes the enclosing container (and
2993
+ * with it the HTML block) anyway — so the two agree on every shape checked.
2994
+ * It is the ONE bullet here whose error direction is under-detection, and it
2995
+ * is why the width is taken from the OPENING line rather than from a greedy
2996
+ * re-strip of each line.
2997
+ * - Offsets are NOT rewritten: `start` / `lastEnd` stay in the ORIGINAL
2998
+ * coordinate space (the stripped prefix is discarded, never subtracted), so
2999
+ * the ranges remain valid for the caller's overlap test. Line-granular
3000
+ * coordinates are sufficient there — `spanInsideHtmlBlock` only asks whether
3001
+ * a span intersects a range.
3002
+ *
3003
+ * TABS ARE EXPANDED TO 4-COLUMN STOPS FIRST (round 14). Every measurement here
3004
+ * is a COLUMN count, and CommonMark measures columns, so the walk cannot be fed
3005
+ * raw characters. Round 13 admitted the gap as a bounded residual and argued it
3006
+ * was fail-CLOSED; that argument was WRONG and the residual was exploitable.
3007
+ * `-\t-\tfoo` opens a list item whose real content column is 8 (each tab
3008
+ * advances to the next multiple of 4), but the character count is 4, so a
3009
+ * continuation line indented 8 spaces was stripped by only 4 and still looked
3010
+ * indented by 4 — `^ {0,3}<…` missed, `kind` stayed `null`, no range was
3011
+ * recorded, the balance guard never fired, and a live `<iframe>` / a swallowing
3012
+ * `<textarea>` reached the DOM inside a protected span (`escapeUnknownHtmlTags`
3013
+ * returned the input BYTE-IDENTICAL). `expandTabs` closes it: after expansion
3014
+ * the line contains no tabs at all, so `^ {0,3}` and every `[ \t]` class below
3015
+ * see true columns. Its arithmetic is the same 4-column stop rule as the mask's
3016
+ * `visualColumn`, so the two walks agree on what a column is.
3017
+ *
3018
+ * The blockquote half reuses `BLOCKQUOTE_PREFIX_RE` — the mask's existing
3019
+ * blockquote-stripping SSOT — so the two walks agree on what a quote marker is.
3020
+ */
3021
+ const LIST_MARKER_PREFIX_RE = /^[ \t]{0,3}(?:[-*+]|\d{1,9}[.)])(?:[ \t]{1,64}|\r?$)/
3022
+
3023
+ /** Expand tabs to 4-column tab stops, so character offsets in the result ARE
3024
+ * columns. Same stop rule as `visualColumn` (the mask's SSOT for this). */
3025
+ function expandTabs(s: string): string {
3026
+ if (s.indexOf('\t') === -1) return s
3027
+ let out = ''
3028
+ for (const ch of s) out += ch === '\t' ? ' '.repeat(4 - (out.length % 4)) : ch
3029
+ return out
3030
+ }
3031
+
3032
+ interface NormalizedLine {
3033
+ /** The line with its container prefix removed (start/end conditions match this). */
3034
+ text: string
3035
+ /** Content column of a list marker this line OPENED, or -1 if it opened none. */
3036
+ openedListCol: number
3037
+ }
3038
+
3039
+ /** `rawLine` is expanded to column stops FIRST, so every length taken below is a
3040
+ * column count. Callers that compare against the input must compare against
3041
+ * `expandTabs(rawLine)`, not `rawLine` — see `computeHtmlBlockRanges`. */
3042
+ function stripContainerPrefix(rawLine: string, listContentCol: number): NormalizedLine {
3043
+ const line = expandTabs(rawLine)
3044
+ let rest = line
3045
+ let openedListCol = -1
3046
+ // The enclosing item's continuation indent, consumable ONCE.
3047
+ let indentBudget = listContentCol
3048
+ // Containers nest in either order (`- > <div>`, `> - <div>`), so alternate
3049
+ // until nothing more is consumed. Bounded so a pathological line of markers
3050
+ // cannot make this super-linear.
3051
+ for (let depth = 0; depth < 16; depth++) {
3052
+ const bq = BLOCKQUOTE_PREFIX_RE.exec(rest)
3053
+ if (bq !== null && bq[0].length > 0) {
3054
+ rest = rest.slice(bq[0].length)
3055
+ if (openedListCol >= 0) openedListCol = line.length - rest.length
3056
+ continue
3057
+ }
3058
+ const li = LIST_MARKER_PREFIX_RE.exec(rest)
3059
+ if (li !== null) {
3060
+ rest = rest.slice(li[0].length)
3061
+ openedListCol = line.length - rest.length
3062
+ continue
3063
+ }
3064
+ // Continuation line of the open list item: drop up to the content column of
3065
+ // leading whitespace (never more, and never non-whitespace) — then KEEP
3066
+ // PEELING. Round 13 consumed this indent after the loop and returned, so a
3067
+ // container opened INSIDE the item (`-\\t> <div>` continued by ` > <div>`)
3068
+ // kept its `>` and no start condition could match: the range was missed and
3069
+ // the balance guard went blind, exactly the round-13 symptom one level down.
3070
+ // Both markers are anchored `^ {0,3}`, so the indent MUST come off first for
3071
+ // either to be seen. Guarded on "this line opened no list marker", which is
3072
+ // what makes it a continuation line at all.
3073
+ if (openedListCol < 0 && indentBudget > 0) {
3074
+ let i = 0
3075
+ while (i < indentBudget && i < rest.length && rest[i] === ' ') i++
3076
+ indentBudget = 0
3077
+ if (i > 0) {
3078
+ rest = rest.slice(i)
3079
+ continue
3080
+ }
3081
+ }
3082
+ break
3083
+ }
3084
+ return { text: rest, openedListCol }
3085
+ }
3086
+
3087
+ /**
3088
+ * `line` MUST be the CONTAINER-NORMALIZED text (`norm.text`), not the raw line.
3089
+ * Kind 4's end condition is a bare `>`, which EVERY blockquote prefix contains —
3090
+ * fed the raw line, `> <!DOCTYPE html` self-terminated on its own start line, so
3091
+ * the range was never recorded and the balance guard went blind for the rest of
3092
+ * the block. Kinds 1/2/3/5 cannot have their closers inside a container prefix,
3093
+ * so for them the two are equivalent; passing `norm.text` uniformly removes the
3094
+ * asymmetry rather than documenting it as a fifth under-detection residual.
3095
+ */
3096
+ function htmlBlockEnds(kind: number, line: string): boolean {
3097
+ switch (kind) {
3098
+ case 1:
3099
+ return HTML_BLOCK_END_1.test(line)
3100
+ case 2:
3101
+ return line.indexOf('-->') !== -1
3102
+ case 3:
3103
+ return line.indexOf('?>') !== -1
3104
+ case 4:
3105
+ return line.indexOf('>') !== -1
3106
+ default:
3107
+ return line.indexOf(']]>') !== -1
3108
+ }
3109
+ }
3110
+
3111
+ function htmlBlockStartKind(line: string, inParagraph: boolean): number | null {
3112
+ if (line.indexOf('<') === -1) return null
3113
+ if (HTML_BLOCK_START_1.test(line)) return 1
3114
+ if (HTML_BLOCK_START_2.test(line)) return 2
3115
+ if (HTML_BLOCK_START_3.test(line)) return 3
3116
+ if (HTML_BLOCK_START_5.test(line)) return 5
3117
+ if (HTML_BLOCK_START_4.test(line)) return 4
3118
+ const six = HTML_BLOCK_START_6.exec(line)
3119
+ if (six && HTML_BLOCK_TYPE_6_TAGS.has(six[2].toLowerCase())) return 6
3120
+ // Condition 7 is the ONLY one that cannot interrupt a paragraph.
3121
+ if (!inParagraph && HTML_BLOCK_START_7.test(line)) return 7
3122
+ return null
3123
+ }
3124
+
3125
+ function computeHtmlBlockRanges(text: string): HtmlBlockRange[] {
3126
+ if (text.indexOf('<') === -1) return []
3127
+ const ranges: HtmlBlockRange[] = []
3128
+ const fences = createFenceTracker()
3129
+ let kind: number | null = null
3130
+ let start = 0
3131
+ let lastEnd = 0
3132
+ let inParagraph = false
3133
+ let offset = 0
3134
+ // Container state for the normalization above. `listContentCol` is the width
3135
+ // of the innermost list marker seen; `openPrefixLen` is how many prefix
3136
+ // COLUMNS the CURRENTLY open block consumed on its OPENING line, which decides
3137
+ // whose notion of "blank line" terminates a type-6/7 block (see below).
3138
+ let listContentCol = 0
3139
+ let openPrefixLen = 0
3140
+ for (const line of text.split('\n')) {
3141
+ const lineStart = offset
3142
+ const lineEnd = offset + line.length
3143
+ offset = lineEnd + 1
3144
+ // Columns, not characters (see `expandTabs`). Every comparison against "the
3145
+ // line as written" below must use THIS, or a tab-prefixed container reads as
3146
+ // a container that opened nothing.
3147
+ const expanded = expandTabs(line)
3148
+ const norm = stripContainerPrefix(expanded, listContentCol)
3149
+ if (norm.openedListCol >= 0) listContentCol = norm.openedListCol
3150
+ else if (!isBlankLine(norm.text) && norm.text === expanded) listContentCol = 0
3151
+ const normBlank = isBlankLine(norm.text)
3152
+ // A type-6/7 block ends at the first blank line. At top level the raw line
3153
+ // decides — a bare `-` or `>` line inside a top-level HTML block is CONTENT,
3154
+ // and treating it as blank would END the range early (the one
3155
+ // under-detecting direction).
3156
+ //
3157
+ // Inside a container the container's own filler (`>`, `> >`, the item's
3158
+ // indent) IS that blank line, so the block must end there — but ONLY the
3159
+ // filler of the container the block actually opened in. Round 13 used the
3160
+ // fully-stripped `norm.text` here, and `stripContainerPrefix` strips ANY
3161
+ // container markers, not the ones that were open. So a line holding a
3162
+ // DIFFERENT container's opener (` >` under a `- <div>`, `> -` under a
3163
+ // `> <div>`) normalized to empty, read as blank, and ended the range early —
3164
+ // re-opening the very shelter the range exists to expose (verified: 1 live
3165
+ // `<textarea>`, where the same input WITHOUT the filler line rendered 0).
3166
+ // Under CommonMark that line is HTML content and the block continues.
3167
+ //
3168
+ // So termination strips at most the OPENING line's prefix width: ` >`
3169
+ // minus 2 columns is `>`, non-blank, block continues; a genuine filler (`>`
3170
+ // under `> `, two spaces under `- `) still normalizes to empty and still
3171
+ // terminates. Measured in expanded columns, consistently with `expandTabs`.
3172
+ // `isBlankLine`, NOT `trim()`. This is THE place the distinction bit: an
3173
+ // NBSP / VT / FF / BOM filler line ended a tracked type-6/7 range while
3174
+ // remark kept the HTML block open, so the inline-code shelter below it
3175
+ // stopped being "inside a tracked block", the balance guard went blind, and
3176
+ // a live `<textarea>` / third-party `<iframe>` reached the DOM
3177
+ // (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL). One
3178
+ // invisible character reopened the whole shelter class.
3179
+ const blank =
3180
+ openPrefixLen > 0 ? isBlankLine(expanded.slice(openPrefixLen)) : isBlankLine(line)
3181
+ if (kind !== null) {
3182
+ // Types 6 and 7 end at (and EXCLUDE) the first blank line; 1..5 end on
3183
+ // the line that satisfies their closer, INCLUSIVE.
3184
+ if (kind >= 6) {
3185
+ if (blank) {
3186
+ ranges.push({ start, end: lastEnd })
3187
+ kind = null
3188
+ openPrefixLen = 0
3189
+ inParagraph = false
3190
+ continue
3191
+ }
3192
+ lastEnd = lineEnd
3193
+ continue
3194
+ }
3195
+ lastEnd = lineEnd
3196
+ if (htmlBlockEnds(kind, norm.text)) {
3197
+ ranges.push({ start, end: lineEnd })
3198
+ kind = null
3199
+ openPrefixLen = 0
3200
+ inParagraph = false
3201
+ }
3202
+ continue
3203
+ }
3204
+ // Outside a block: keep fence state, so a start condition written inside a
3205
+ // genuine fenced code sample cannot open one. Fed the RAW line ON PURPOSE —
3206
+ // the tracker is a shared CommonMark machine with two other consumers and
3207
+ // feeding it normalized lines would let a `>`-prefixed delimiter INSIDE a
3208
+ // top-level fence close it early. The cost is that a fence written inside a
3209
+ // blockquote is invisible here, so its content lines can open a bogus block:
3210
+ // more detection, i.e. the fail-CLOSED direction.
3211
+ if (fences.push(line) !== 'text') {
3212
+ inParagraph = false
3213
+ continue
3214
+ }
3215
+ const started = htmlBlockStartKind(norm.text, inParagraph)
3216
+ if (started !== null) {
3217
+ inParagraph = false
3218
+ // 1..5 may satisfy their end condition on the START line itself.
3219
+ if (started < 6 && htmlBlockEnds(started, norm.text)) {
3220
+ ranges.push({ start: lineStart, end: lineEnd })
3221
+ continue
3222
+ }
3223
+ kind = started
3224
+ openPrefixLen = expanded.length - norm.text.length
3225
+ start = lineStart
3226
+ lastEnd = lineEnd
3227
+ continue
3228
+ }
3229
+ // Paragraph state uses the NORMALIZED blank: a container's own filler line
3230
+ // (`>`, `> >`) separates paragraphs inside the container. Erring toward
3231
+ // "not in a paragraph" only ENABLES the type-7 start condition — more
3232
+ // detection, the fail-CLOSED direction.
3233
+ inParagraph =
3234
+ !normBlank && leadingIndent(norm.text) < 4 && !NON_PARAGRAPH_LINE_RE.test(norm.text)
3235
+ }
3236
+ // An unterminated block runs to end of input, exactly as the tokenizer treats it.
3237
+ if (kind !== null) ranges.push({ start, end: lastEnd })
3238
+ return ranges
3239
+ }
3240
+
3241
+ /**
3242
+ * Tag-like starts the MAIN pass could not consume — `TAG_LIKE_REGEX` hard-bounds
3243
+ * its attribute run at 4096 chars (ReDoS hardening), so a longer run makes the
3244
+ * whole tag fail to match and NEITHER the allowlist NOR the RAWTEXT closer check
3245
+ * ever runs: a live `<iframe src="data:text/html;base64,…4KB+…">` reached the DOM
3246
+ * verbatim. The cap must stay (removing it reintroduces the backtracking blowup),
3247
+ * so instead an over-long tag FAILS CLOSED here: only its `<` is escaped, which
3248
+ * degrades it to visible text rather than a live opener.
3249
+ *
3250
+ * Applied ONLY to the gaps BETWEEN main-pass matches, so a tag the main pass
3251
+ * already decided on can never be touched twice (no `&amp;lt;`).
3252
+ *
3253
+ * Shape is deliberately trivial — one bounded quantifier over disjoint character
3254
+ * classes and a single-char lookahead, so there is no alternation to backtrack
3255
+ * across and failure costs at most 64 steps per candidate `<`.
3256
+ *
3257
+ * MEASURED COST (round 12, this repo's vitest/jsdom env, `escapeUnknownHtmlTags`
3258
+ * over a document of nothing but max-length never-closed tag names — the
3259
+ * pathological shape for this pass): 61.2 / 122.1 / 257.2 / 492.2 ms at 325KB /
3260
+ * 650KB / 1.3MB / 2.6MB. Dead linear at ~190 ns/char, so there is no
3261
+ * algorithmic blowup — only a large constant on an input no real message has.
3262
+ * Realistic chat/post output (≤256KB) lands around 50ms. An earlier note in the
3263
+ * remediation record claimed 5.2ms for the 1.3MB case; that figure was wrong by
3264
+ * ~50x and is corrected here.
3265
+ *
3266
+ * NOT APPLIED INSIDE CODE (round 12). The candidate shape here is ANY
3267
+ * `<[a-zA-Z…]` followed by whitespace or `>`, not just the over-long tag it was
3268
+ * written for — so ordinary pseudo-code (` if a <b then` in an indented
3269
+ * block) was escaped to a visible `&lt;`, since entity references are NOT
3270
+ * decoded inside code. That is the very argument that scoped the balance guard
3271
+ * away from inline code, applied here. The MASK already knows which regions are
3272
+ * code and the offsets are exact, so each candidate is checked against it
3273
+ * individually (per-candidate, not per-gap: a gap routinely spans both prose and
3274
+ * code).
3275
+ *
3276
+ * THE 4096-CHAR CAP HAS TWO CONSUMERS THAT ROUND IN OPPOSITE DIRECTIONS — and
3277
+ * getting that asymmetry wrong is what hid a live fail-open for six review
3278
+ * rounds (round 16). "This span is not KNOWN to be code" means:
3279
+ *
3280
+ * consumer | not-known-to-be-code ⇒ | fail direction
3281
+ * -------------------------------|------------------------|---------------
3282
+ * this pass (`isMaskedBlank`) | ESCAPE the `<` | CLOSED (cosmetic:
3283
+ * | | a visible `&lt;`)
3284
+ * the CARVE (`PROTECTED_SPAN_RE`)| ESCAPE the span | CLOSED (cosmetic:
3285
+ * | | code renders as
3286
+ * | | escaped text)
3287
+ * the MASK's closer haystack | ADMIT a `</tag>` closer| **OPEN** (a live
3288
+ * (`hasLaterCloser`) | | RAWTEXT opener
3289
+ * | | swallows the rest
3290
+ * | | of the message)
3291
+ *
3292
+ * The previous version of this note analysed the over-cap span for THIS pass
3293
+ * only, concluded "the safe direction", and stopped — true here, false for the
3294
+ * haystack, where the identical cap silently un-blanked a `</textarea>` written
3295
+ * inside an over-long inline span and re-opened the RAWTEXT swallow this module
3296
+ * exists to close. A residual note must state the fail direction PER CONSUMER;
3297
+ * a single "safe direction" verdict for a value read by passes that round
3298
+ * opposite ways is not a finding, it is an averaging error.
3299
+ *
3300
+ * RESOLVED for the haystack: the mask no longer uses a capped regex at all
3301
+ * (`findInlineCodeRanges` — linear, uncapped), so an over-long inline span is
3302
+ * blanked like any other and the haystack's fail-OPEN row above no longer has
3303
+ * an over-cap case. Pinned by the `spanLength` axis of the swallow sweep
3304
+ * (cap−k and cap+k for every shelter spelling).
3305
+ * RESIDUAL, deliberately kept: the CARVE keeps its cap, and so does this pass's
3306
+ * view of an over-cap span in a document the mask ALSO declines to blank — both
3307
+ * of those round CLOSED per the table, i.e. they cost at worst a visible `&lt;`.
3308
+ */
3309
+ const LEFTOVER_TAG_START_RE = /<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})(?=[\s>])/g
3310
+
3311
+ function escapeLeftoverTagStarts(gap: string, lowerSource: string, gapOffset: number): string {
3312
+ if (gap.indexOf('<') === -1) return gap
3313
+ LEFTOVER_TAG_START_RE.lastIndex = 0
3314
+ return gap.replace(
3315
+ LEFTOVER_TAG_START_RE,
3316
+ (m: string, slash: string, tag: string, at: number) =>
3317
+ isMaskedBlank(lowerSource, gapOffset + at, gapOffset + at + m.length)
3318
+ ? m
3319
+ : `&lt;${slash}${tag}`,
3320
+ )
3321
+ }
3322
+
3323
+ export function escapeUnknownHtmlTags(
3324
+ text: string,
3325
+ allowedTags: Set<string> = SAFE_HTML_TAGS,
3326
+ ): string {
3327
+ if (!text || text.indexOf('<') === -1) return text
3328
+ // Masked, length-preserving, lowercased whole-document copy for the RAWTEXT
3329
+ // closer lookup — the closer may live in a later segment than the opener,
3330
+ // so the search must span the ENTIRE source, not the segment being escaped,
3331
+ // and must ignore closers that are only code samples / attribute text.
3332
+ const lowerSource = buildCloserHaystack(text)
3333
+ // HTML-block ranges for the CARVE BALANCE GUARD below. Computed LAZILY: only
3334
+ // a protected span that actually carries an unbalanced RAWTEXT opener needs
3335
+ // them, which no ordinary message has.
3336
+ let htmlBlocks: HtmlBlockRange[] | null = null
3337
+ const spanInsideHtmlBlock = (from: number, to: number): boolean => {
3338
+ htmlBlocks ??= computeHtmlBlockRanges(text)
3339
+ return htmlBlocks.some((r) => r.start < to && r.end > from)
3340
+ }
3341
+ // Carve out fenced code blocks AND inline-backtick spans so `<their>`
3342
+ // examples inside code are preserved verbatim.
3343
+ const parts: string[] = []
3344
+ let cursor = 0
3345
+ PROTECTED_SPAN_RE.lastIndex = 0
3346
+ let span: RegExpExecArray | null
3347
+ while ((span = PROTECTED_SPAN_RE.exec(text)) !== null) {
3348
+ if (span.index > cursor) {
3349
+ parts.push(
3350
+ escapeOutsideFences(text.slice(cursor, span.index), allowedTags, lowerSource, cursor),
3351
+ )
3352
+ }
3353
+ // INTERSECTION GUARD (soundness, not an instance patch). A protected span
3354
+ // is pushed through VERBATIM, so a live RAWTEXT opener inside one never
3355
+ // reaches `escapeOutsideFences` at all and the mask's correctness is
3356
+ // bypassed. Carve and mask run different engines, so the carve CAN protect
3357
+ // a region the mask correctly blanked — `PROTECTED_SPAN_RE`'s closer
3358
+ // alternative accepts an info string, ends its span early, desyncs, and can
3359
+ // open a new span from a line CommonMark treats as ordinary text. Protect
3360
+ // only what BOTH engines call code: if the mask left anything non-blank
3361
+ // over this exact range, escape the span instead.
3362
+ //
3363
+ // CARVE BALANCE GUARD (the residual fail-open the intersection alone does
3364
+ // NOT close). The intersection only reconciles DISAGREEMENT; when BOTH
3365
+ // engines over-detect the SAME region it is a no-op. CommonMark says an
3366
+ // HTML block (type 1 `<pre>`/`<details>`, type 6 `<div>`) runs to its
3367
+ // terminator, so a ``` line inside one is HTML CONTENT and not a fence —
3368
+ // and NEITHER `createFenceTracker` nor `PROTECTED_SPAN_RE` models HTML
3369
+ // blocks, so both open a bogus fence at the same line and shelter whatever
3370
+ // follows. So the range check is paired with a self-containment check: a
3371
+ // protected span is by definition a complete code region, therefore any
3372
+ // RAWTEXT opener inside it must be BALANCED within it. An unbalanced one
3373
+ // means the span is not really code — route it through the escaper. This
3374
+ // is engine-independent (it needs no HTML-block tracking).
3375
+ //
3376
+ // GATED ON HTML-BLOCK MEMBERSHIP, NOT ON THE SPAN'S FLAVOR (round 12). The
3377
+ // property that makes a protected span "not really code" is that it sits
3378
+ // inside an HTML BLOCK — where CommonMark says every line is HTML content.
3379
+ // Two earlier rounds gated on flavor instead and traded one hole for the
3380
+ // other:
3381
+ // - round 9 applied the guard to BOTH alternatives. That over-applied to
3382
+ // inline code, where entity references are NOT recognized, so an escaped
3383
+ // `&lt;title&gt;` was shown to the reader LITERALLY — and naming a tag in
3384
+ // inline code (`` `<title>` ``) is the single most common way a docs
3385
+ // answer mentions one.
3386
+ // - round 11 scoped it to FENCES, justified by "an inline span cannot
3387
+ // shelter a live opener: remark emits it as an `inlineCode` TEXT node, so
3388
+ // parse5 never tokenizes its content". That invariant is asserted in a
3389
+ // comment and holds only OUTSIDE an HTML block. Inside one, remark emits
3390
+ // raw HTML, backticks are not code, and `` `<textarea>` `` on its own
3391
+ // line inside `<div>` / `<pre>` / `<details>` / `<span>` sheltered a live
3392
+ // opener that swallowed the rest of the message.
3393
+ // Membership covers BOTH spellings with one property, and leaves ordinary
3394
+ // prose inline code untouched. `isFence` is kept as an independent
3395
+ // sufficient condition: a fenced span that the mask blanked but the HTML
3396
+ // walk does not consider part of a block (the two engines can still desync)
3397
+ // must stay under the round-9 guarantee.
3398
+ // Group 1 is the fence marker, group 2 the inline backtick run.
3399
+ //
3400
+ // PROPERTY GUARANTEED: no protected span that is either a FENCE or inside an
3401
+ // HTML BLOCK can carry an unbalanced RAWTEXT opener into the output
3402
+ // verbatim. That is strictly weaker than "the carve never over-detects" — an
3403
+ // over-detected span with no RAWTEXT opener in it is still pushed verbatim,
3404
+ // which stays cosmetic-only.
3405
+ const isFence = span[1] !== undefined
3406
+ const spanEnd = span.index + span[0].length
3407
+ parts.push(
3408
+ isMaskedBlank(lowerSource, span.index, spanEnd) &&
3409
+ !(
3410
+ hasUnbalancedRawtextOpener(span[0]) &&
3411
+ (isFence || spanInsideHtmlBlock(span.index, spanEnd))
3412
+ )
3413
+ ? span[0]
3414
+ : escapeOutsideFences(span[0], allowedTags, lowerSource, span.index),
3415
+ )
3416
+ cursor = span.index + span[0].length
3417
+ }
3418
+ if (cursor < text.length) {
3419
+ parts.push(escapeOutsideFences(text.slice(cursor), allowedTags, lowerSource, cursor))
3420
+ }
3421
+ return parts.join('')
3422
+ }
3423
+
3424
+ /**
3425
+ * Walks `segment` tag by tag rather than using `String.replace`, so the regions
3426
+ * the main regex did NOT consume are addressable: each gap is handed to
3427
+ * `escapeLeftoverTagStarts` (see it for the over-long-attribute fail-open it
3428
+ * closes), while every matched tag keeps its ORIGINAL index. Preserving that
3429
+ * index matters — `hasLaterCloser` indexes `lowerSource`, which is built from
3430
+ * the untouched text, so any offset drift reopens the round-5 desync class.
3431
+ */
3432
+ function escapeOutsideFences(
3433
+ segment: string,
3434
+ allowedTags: Set<string>,
3435
+ lowerSource: string,
3436
+ segmentOffset: number,
3437
+ ): string {
3438
+ const out: string[] = []
3439
+ let cursor = 0
3440
+ TAG_LIKE_REGEX.lastIndex = 0
3441
+ let m: RegExpExecArray | null
3442
+ while ((m = TAG_LIKE_REGEX.exec(segment)) !== null) {
3443
+ const [match, slash, tag, rest, selfClose] = m
3444
+ if (m.index > cursor)
3445
+ out.push(
3446
+ escapeLeftoverTagStarts(
3447
+ segment.slice(cursor, m.index),
3448
+ lowerSource,
3449
+ segmentOffset + cursor,
3450
+ ),
3451
+ )
3452
+ const lower = tag.toLowerCase()
3453
+ const escaped = `&lt;${slash}${tag}${rest}${selfClose}&gt;`
3454
+ if (!allowedTags.has(lower)) {
3455
+ out.push(escaped)
3456
+ } else if (slash === '' && RAWTEXT_TAGS.has(lower)) {
3457
+ // Allowlisted — but an UNCLOSED RAWTEXT opener would swallow the rest of
3458
+ // the document during tokenization, before any allowlist applies.
3459
+ //
3460
+ // The SELF-CLOSED spelling counts as an opener (round 11): HTML ignores
3461
+ // the self-closing flag on non-void, non-foreign elements, so parse5
3462
+ // tokenizes `<textarea/>` as a start tag and enters RAWTEXT identically.
3463
+ // Excluding it here left the entire defense — prose openers, HTML-block
3464
+ // shelters, all of it — bypassable by one extra slash. `RAWTEXT_TAGS` has
3465
+ // no void members, so nothing legitimate self-closes.
3466
+ //
3467
+ // COSMETIC COST (accepted, fail-closed): self-closing IS honored in
3468
+ // foreign content, so an EMPTY `<title/>` inside `<svg>` now escapes
3469
+ // rather than rendering. It carries no accessible name either way, and
3470
+ // the real a11y form `<title>Chart</title>` is unaffected. SECOND COST
3471
+ // added by the same round: a protected span the balance guard deems
3472
+ // not-really-code is routed through this function whole, so bare `<tag`
3473
+ // starts in its GAPS are escaped too — the mask check in
3474
+ // `escapeLeftoverTagStarts` keeps that off genuine code regions.
3475
+ const afterTag = segmentOffset + m.index + match.length
3476
+ out.push(hasLaterCloser(lowerSource, lower, afterTag) ? match : escaped)
3477
+ } else {
3478
+ out.push(match)
3479
+ }
3480
+ cursor = m.index + match.length
3481
+ }
3482
+ if (cursor < segment.length)
3483
+ out.push(escapeLeftoverTagStarts(segment.slice(cursor), lowerSource, segmentOffset + cursor))
3484
+ return out.join('')
3485
+ }
3486
+
3487
+ // ---------------------------------------------------------------------------
3488
+ // URL transform
3489
+ // ---------------------------------------------------------------------------
3490
+ /**
3491
+ * Extends react-markdown's default safe-protocol allowlist with the two
3492
+ * internal schemes the chat remark plugins emit (`card://`, `mention://`),
3493
+ * for `href` ONLY. All other URLs go through `defaultUrlTransform`.
3494
+ */
3495
+ export function cardAwareUrlTransform(url: string, key: string): string {
3496
+ if (key === 'href' && typeof url === 'string' && (url.startsWith('card://') || url.startsWith('mention://')))
3497
+ return url
3498
+ return defaultUrlTransform(url)
3499
+ }