@aiquants/markdown-explorer 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/CHANGELOG.md +92 -0
  2. package/LICENSE +21 -0
  3. package/README.md +588 -2
  4. package/dist/ExplorerSplit-BESGFcu6.cjs +1 -0
  5. package/dist/ExplorerSplit-DwEFmvQM.js +1362 -0
  6. package/dist/MultiLayout-DxPiARsv.js +896 -0
  7. package/dist/MultiLayout-r_qx1Mfz.cjs +1 -0
  8. package/dist/TreeLayout-9-20bbSM.cjs +1 -0
  9. package/dist/TreeLayout-CW7lgV_e.js +19 -0
  10. package/dist/client/MarkdownExplorer.d.cts +30 -0
  11. package/dist/client/MarkdownExplorer.d.ts +30 -0
  12. package/dist/client/storage.d.cts +11 -0
  13. package/dist/client/storage.d.ts +11 -0
  14. package/dist/core/config.d.cts +71 -0
  15. package/dist/core/config.d.ts +71 -0
  16. package/dist/core/contracts.d.cts +155 -0
  17. package/dist/core/contracts.d.ts +155 -0
  18. package/dist/core/geometry.d.cts +51 -0
  19. package/dist/core/geometry.d.ts +51 -0
  20. package/dist/core/labels.d.cts +175 -0
  21. package/dist/core/labels.d.ts +175 -0
  22. package/dist/core/location.d.cts +82 -0
  23. package/dist/core/location.d.ts +82 -0
  24. package/dist/core/paths.d.cts +17 -0
  25. package/dist/core/paths.d.ts +17 -0
  26. package/dist/core/revalidation.d.cts +2 -0
  27. package/dist/core/revalidation.d.ts +2 -0
  28. package/dist/core/theme.d.cts +6 -0
  29. package/dist/core/theme.d.ts +6 -0
  30. package/dist/core/treeSettings.d.cts +48 -0
  31. package/dist/core/treeSettings.d.ts +48 -0
  32. package/dist/core/viewSettings.d.cts +15 -0
  33. package/dist/core/viewSettings.d.ts +15 -0
  34. package/dist/dataRoute-Bkfwfp1R.js +194 -0
  35. package/dist/dataRoute-CKbdwkQ7.cjs +1 -0
  36. package/dist/index-BgkZ9ATV.js +2061 -0
  37. package/dist/index-y9tVvDW7.cjs +3 -0
  38. package/dist/index.cjs +1 -0
  39. package/dist/index.d.cts +12 -0
  40. package/dist/index.d.ts +12 -0
  41. package/dist/index.js +44 -0
  42. package/dist/location-Cb2Ak55B.js +382 -0
  43. package/dist/location-DWHQYntR.cjs +1 -0
  44. package/dist/sanitizeWorker.cjs +1 -0
  45. package/dist/sanitizeWorker.js +291 -0
  46. package/dist/server/anonymous.d.cts +10 -0
  47. package/dist/server/anonymous.d.ts +10 -0
  48. package/dist/server/createMarkdownExplorer.d.cts +12 -0
  49. package/dist/server/createMarkdownExplorer.d.ts +12 -0
  50. package/dist/server/errors.d.cts +26 -0
  51. package/dist/server/errors.d.ts +26 -0
  52. package/dist/server/fileSystemSource.d.cts +38 -0
  53. package/dist/server/fileSystemSource.d.ts +38 -0
  54. package/dist/server/ports.d.cts +107 -0
  55. package/dist/server/ports.d.ts +107 -0
  56. package/dist/server/sanitizeProtocol.d.cts +17 -0
  57. package/dist/server/sanitizeProtocol.d.ts +17 -0
  58. package/dist/server/sanitizer.d.cts +51 -0
  59. package/dist/server/sanitizer.d.ts +51 -0
  60. package/dist/server/workerPool.d.cts +42 -0
  61. package/dist/server/workerPool.d.ts +42 -0
  62. package/dist/server.cjs +1 -0
  63. package/dist/server.d.cts +7 -0
  64. package/dist/server.d.ts +7 -0
  65. package/dist/server.js +1253 -0
  66. package/dist/storage-Bkb2Axdi.cjs +1 -0
  67. package/dist/storage-DA4ItHzl.js +60 -0
  68. package/dist/storage.cjs +1 -0
  69. package/dist/storage.d.cts +1 -0
  70. package/dist/storage.d.ts +1 -0
  71. package/dist/storage.js +4 -0
  72. package/dist/styles/markdown-explorer.css +3 -0
  73. package/docs/behavior/dom-hooks.md +95 -0
  74. package/docs/behavior/layout.md +203 -0
  75. package/docs/behavior/localization.md +47 -0
  76. package/docs/behavior/security.md +134 -0
  77. package/docs/behavior/url-contract.md +129 -0
  78. package/package.json +141 -3
@@ -0,0 +1,47 @@
1
+ # Localization
2
+
3
+ This page is part of the host-facing contract of `@aiquants/markdown-explorer`; its summary is in the [README](../../README.md#localization).
4
+
5
+ ## Languages
6
+
7
+ - `locale` is the UI language, `"en"` (the default) or `"ja"` — never a BCP 47 region tag. `"ja-JP"`, `"EN"` and any other value throw a `RangeError` on the first render, so a mistyped locale is found in development instead of silently showing English.
8
+ - The catalogs are `MARKDOWN_EXPLORER_LABEL_CATALOGS.en` and `MARKDOWN_EXPLORER_LABEL_CATALOGS.ja`, frozen objects with the same keys; `MARKDOWN_EXPLORER_LABEL_KEYS` lists the keys in declaration order.
9
+ - The explorer passes its document-viewer strings to `@aiquants/markdown`, which names every control and status text of the viewer with them, so the viewer follows the same language:
10
+ its status texts (`viewerLoading`, `viewerErrorPrefix`, `viewerErrorTitle`, `viewerMissingImage`), its toolbar's buttons (`viewerSelectAlign`, `viewerSelectStyle`, `viewerOpenToc`), the alignment and style menus (`viewerAlignTitle`, `viewerAlignLeft`, `viewerAlignCenter`, `viewerAlignRight`, `viewerPinAlign`, `viewerUnpinAlign`, `viewerStyleTitle`, `viewerPinStyle`, `viewerUnpinStyle`), the table of contents (`viewerTocTitle`, which also names its navigation, `viewerPinToc`, `viewerUnpinToc`)
11
+ and every Mermaid diagram's controls (`viewerMermaidInteractionMode` for the mouse-interaction menu and its button, `viewerMermaidModeNone`, `viewerMermaidModeZoom`, `viewerMermaidModePan`, `viewerMermaidModePanShift`, `viewerMermaidModeBoth`, `viewerMermaidModeBothShift` for its modes, `viewerMermaidZoomIn`, `viewerMermaidZoomOut`, `viewerMermaidResetView` and `viewerMermaidRetry`).
12
+ A label the viewer receives carries no `lang` of its own, so it is read in the page's language. Where the viewer's own built-in label is in a catalog's language, the catalog repeats it word for word (the English catalog the viewer's English labels, such as `Display Style` and `Zoom In`; the Japanese catalog its Japanese ones, such as `目次を開く`).
13
+ - The explorer passes `treeEntryKindDirectory` and `treeEntryKindFile` (the kind words in the rows' accessible names) and `treeScrollUp`, `treeScrollDown`, `treeScrollToTop` and `treeScrollToBottom` (the tree's scroll bar) to `@aiquants/directory-tree`.
14
+ - The explorer passes every string the multi view's drag-and-drop layout shows to `@aiquants/drag-drop-panels`: the panel headers' buttons (`panelSwitchToCustomDrag`, `panelSwitchToNormalDrag`, `panelMovePreviousColumn`, `panelMoveNextColumn`, `panelMoveUp`, `panelMoveDown`, `panelHide`, `panelMaximize`, `panelRestore`, `panelClose`) and the custom drag mode's badge (`panelCustomDragBadge`), the move announcements (`panelMoved`, `panelMoveUnavailable`), a touch or pen drag's lift and cancel announcements (`panelDragStarted`, `panelDragCancelled`), the bar of hidden panels (`hiddenPanelsTitle`), a hidden panel's stand-in while a panel is dragged (`hiddenPanelBadge`, `hiddenPanelNotice`), the drop placeholders (`panelDropHere`), an empty column's text (`multiColumnEmpty`) and the name of each column's region (`multiColumnRegion`).
15
+ The stage never lets the reader resize its columns, so the layout renders no column resize handle and the catalog has no key for that handle's name. A label the layout receives carries no `lang` of its own, so it is read in the page's language, like the explorer's other strings.
16
+ - `treeRegion` names the tree itself, apart from `explorerRegion`, which names the explorer's landmark (and the sheet on narrow frames).
17
+ - `refresh` is the visible text of the refresh buttons (the tree's and the single view's), which is also their accessible name; `refreshTree` and `refreshDocument` are their tooltips. A successful reload is announced with `treeRefreshed` or `documentRefreshed(name)`, and a failed one is shown as an alert with `treeRefreshFailed(reason)` or `documentRefreshFailed(reason)`, where `reason` is the failure's title (the matching `failure…Title` label).
18
+ - `switchToTree` and `switchToMulti` name their destination with the README's view names ("Open in the tree view" / "ツリービューで開く", "Open in the multi-panel view" / "マルチビューで開く"); no label names the single view, which no control inside the explorer opens.
19
+
20
+ ## Overrides
21
+
22
+ `labels` overlays single keys onto the catalog of `locale`:
23
+
24
+ ```tsx
25
+ const HANDBOOK_LABELS = {
26
+ explorerRegion: "Handbook",
27
+ selectionTitle: (count: number) => `${String(count)} picked`,
28
+ } satisfies MarkdownExplorerLabelOverrides
29
+
30
+ <MarkdownExplorer data={data} config={explorerConfig} storageNamespace="docs" locale="en" labels={HANDBOOK_LABELS} />
31
+ ```
32
+
33
+ - Strings that embed a value are functions — `foldersLoadFailedNamed(names)` (the announcement of the folders whose children failed to load within one frame, named in one sentence with the language's list format;
34
+ `folderLoadFailed` is the text shown in a row), `announcementSequence(messages)` (the live region's text for two or more messages announced within one frame, as separate sentences), `selectionTitle(count)`, `openSelectionInMulti(count)`, `removeFromSelection(name)`, `expandIconSizeValue(pixels)`, `settingAnnouncement(setting, value)`, `documentShown(name)`, `foldersLoading(count)` (the tree's progress line: the source's folder listings queued or in flight), `expandAllProgress(opened, loading)` (the progress line during an Expand all press), `expandAllDone(opened, failed, complete)` (the end of a press that no limit stopped; `complete` is whether every folder is open at the end, false when the reader closed one during the press, so the sentence then claims nothing about the rest; nothing opened, nothing failed and complete means every folder was already open), `expandAllFolderLimit(opened, limit, expandAll)` (a press stopped at `expandAll.maxFolders`; `expandAll` is the Expand all button's own label, so a host that renames the button keeps the invitation to press it again consistent), `expandAllDepthLimit(opened, depth)` (a press that left folders deeper than `expandAll.maxDepth` closed), `loopFolderName(name)` (the name of a loop row, a folder met again among its own ancestors), `documentFailed(name, title)` (a document that could not be shown:
35
+ the path's last segment and the failure's title — the text of the matching `failure…Title` label, not the reason's code), `imageZoomLevel(percent)`, `treeRefreshFailed(reason)`, `documentRefreshed(name)`, `documentRefreshFailed(reason)`, `panelLimit(max)`, `panelOpened(name)`, `panelClosed(name)`, `panelAlreadyOpen(name)`, `panelHidden(name)`, `panelShown(name)`, `panelMaximized(name)`, `panelRestored(name)` — so each language can place the value where its grammar needs it.
36
+ - The labels `@aiquants/drag-drop-panels` fills itself are templates instead, strings whose `{name}` placeholders mark where each value goes: `panelMoved` (`{title}` the panel's name, `{column}` its visible column, `{position}` its place among that column's shown panels), `panelMoveUnavailable` (`{title}`, and `{move}` the name of the move, one of the `panelMove…` labels), `panelDragStarted` and `panelDragCancelled` (`{title}`) and `multiColumnRegion` (`{column}`); columns and positions count from 1. `MARKDOWN_EXPLORER_LABEL_TEMPLATES` lists each template's placeholders, and braces in every other label are literal text.
37
+ - `labels` must be a plain object. An unknown key throws a `RangeError` that lists the valid keys, a value of the wrong type (a string where a function is expected, or the reverse) throws a `TypeError`, and a blank string throws a `RangeError`, because a blank name is neither shown nor announced. A template that uses a placeholder its key does not offer (`{name}` for `{title}`, say) throws a `RangeError` naming the placeholders it accepts, because a placeholder that is never filled would be read out as it is. A key set to `undefined` keeps the catalog's string.
38
+ - The overrides are compared by value, key by key: writing `labels={{ … }}` inline does not re-render the explorer on every render of the host, but a function created inline is a new value each time — hoist it, or memoize the object.
39
+
40
+ ## Formatting
41
+
42
+ Counts and percentages are formatted by the label functions themselves, so a host that needs grouping or another numeral system overrides the function. The tree keeps the order its source returns unless the configuration's `defaultTreeSettings` sets a sort order or the reader picks one in the tree settings; sorted entries are compared with `Intl.Collator("en", { numeric: true, sensitivity: "variant" })` — numeric-aware and independent of the UI language, so a source sorts the same for every reader.
43
+
44
+ ## Strings outside the catalog
45
+
46
+ - Names of sources (`label`), notices and document styles come from the host and are shown as given.
47
+ - The empty-list and horizontal-scroll strings of `@aiquants/directory-tree` are never shown by the explorer, so the catalog has no keys for them.
@@ -0,0 +1,134 @@
1
+ # Security
2
+
3
+ This page is part of the host-facing contract of `@aiquants/markdown-explorer`; its summary is in the [README](../../README.md#security-model).
4
+
5
+ The explorer renders documents that the reader did not write, from storage the reader may only partly be allowed to see. Each rule below holds without the host's help; the host's part is listed at the end.
6
+
7
+ ## Paths
8
+
9
+ - A document path from a URL must be acceptable before any source sees it: non-empty, at most 4096 UTF-16 code units, well-formed (no lone surrogate, which no URL can carry), without control characters (U+0000–U+001F, U+007F) and without backslashes (some filesystems read them as separators). Anything else is answered as `invalid`.
10
+ - Deny rules (`deny`, default `DEFAULT_DENIED_PATH_SEGMENTS`) hide matching entries from every listing — by entry name and by path, so a denied folder is pruned with everything under it — and refuse every document or children request for a matching path with a `forbidden` failure before any source is called.
11
+ - A rule is matched against each path segment (or entry name) at every percent-decoding level up to three decodes, against each level whole and against its parts split again on `/` or `\` (a storage may read either character as part of one name, as a Linux file name, a Drive name or a GCS object name does, or as a separator), and after NFKC normalization and full case folding (upper-casing then lower-casing, which folds `ſ` to `s` and `ı` to `i`). Decoding is lenient: a malformed escape stays as written while the valid ones around it are decoded. `%2Eenv`, `%252Eenv`, `.%65nv`, `docs%5C.env`, `SECRETS` and `ſecrets` are all denied like `.env` and `secrets`, whatever the storage backend decodes or folds before opening the path.
12
+ - Pattern syntax: a pattern matches one whole segment; `*` matches any run of characters inside it. `deny: false` turns the rules off; a custom array replaces the defaults (spread `DEFAULT_DENIED_PATH_SEGMENTS` to extend them).
13
+ - The defaults cover environment files (`.env*` — every name starting with `.env`, for example `.env~`, `.env-old`, `.envs/`, and also `.envelope.md` — and `*.env` with the names its backups and copies take: `*.env.*`, `*.env~*`, `*.env-*`, `*.env_*` and `*.env#*`, for example `prod.env.bak`, `prod.env.production`, `prod.env.1`, `prod.env~`, `prod.env-old`, `prod.env_backup` and `#prod.env#`; rules have no exceptions, so a document named like `deploy.env.md` is denied too, while names in which `.env` is followed by a letter or digit, such as `prod.environment.md`, or has no dot before it, such as `environment.md` and `my-env.md`, are not),
14
+ version-control metadata (`.git`, `.git-credentials`, `.svn`, `.hg`), `node_modules`, secrets files (`secrets`, `.secrets`, and `secrets.` with `json`, `yml`, `yaml`, `toml`, `ini` or `txt`), the credential folders and files of command-line tools (`.ssh`, `.aws`, `.gnupg`, `.docker`, `.kube`, `.terraform`, `.npmrc`, `.yarnrc`, `.yarnrc.yml`, `.netrc`, `_netrc`, `.pypirc`, `.pgpass`, `.my.cnf`, `.htpasswd`, `.vault-token`),
15
+ keys and keystores (`id_rsa*`, `id_dsa*`, `id_ecdsa*`, `id_ed25519*`, `*.pem`, `*.key`, `*.p8`, `*.p12`, `*.pfx`, `*.ppk`, `*.jks`, `*.keystore`, `*.kdbx`), credential files (`*credentials*.json`, `service-account*.json`), Terraform state and variables (`*.tfstate`, `*.tfstate.*`, `*.tfvars`, `*.tfvars.json`) and shell histories (`.*_history`, `fish_history`).
16
+ - `createFileSystemSource` additionally checks lexical containment before any filesystem call and again after resolving the real path, refuses symbolic links, and never follows a path outside its root. It opens a file without following a link (where the platform has `O_NOFOLLOW`) and reads it only when the opened file is the one the checks saw (the same device and inode; a document read refuses another file as `forbidden`).
17
+ Its `asset` port also requires the size and modification time the checks saw, answers `not-found` when the file is another one or has changed since, and streams a whole file no further than the size the checks saw.
18
+ The walk leaves out every name it cannot carry — not in NFC, with a colon, ending in a dot or a space, holding a control character or a backslash, too long, a symbolic link or a special file — and every subdirectory it cannot read (a permission error, or one that vanished or became a file during the walk), and reports each once per path through `reportSkipped` (a `console.warn` by default); only an unreadable root fails the tree. Only files with a known text extension or name are previewed as text; any other file is `unsupported`.
19
+
20
+ ## Source boundaries
21
+
22
+ - Every source declares a boundary: `contains(path)`, pure and synchronous (whether the path's text can lie inside the source's roots; it never reads storage), and an optional `confirm(context, path)`, the storage's answer `{ inside: false }` or `{ inside: true, readPath }` for a path `contains` accepted. `readPath` is what the host reads or lists (the object the storage placed the path at), never the path as written.
23
+ - A source's tree and children list only paths inside its own boundary: an entry outside it is left out with its subtree and logged.
24
+ - A children request names its source and is checked against that source alone, in this order: the deny rules, `contains` (outside: `forbidden`), `confirm` (outside: `forbidden`; an `ExplorerError` it throws keeps its reason; any other exception is logged and answered `failed`). The host's `children` is called only after both passed, with the confirmed `readPath`.
25
+ - A document carries no source (the union rule): a document request — any view, panel, document link or page seed — passes the deny rules, then opens only when the boundary of at least one source admits its path. A path no source contains is `forbidden` without a host or storage call; a containing source without `confirm` admits it at once with `readPath` equal to the path; otherwise the containing sources confirm it in configuration order and the first inside admits it with its `readPath`.
26
+ When none does, the first confirmation's failure is answered, else `forbidden`. The host's `document` receives the admitting source's `readPath`.
27
+ - A `contains` that throws or answers anything but a boolean is logged (operation `boundary`) and counts as outside (a promise it answers is not awaited, and its rejection is logged the same way); a malformed `confirm` answer is logged and counts as outside; a `readPath` that is not an acceptable document path or that the deny rules refuse is logged and answered `failed`. A broken boundary therefore never opens a document.
28
+ - The union is exactly what the deployment exposes: a source whose boundary admits every document a reader can read in a store opens every such document through every view, link and asset URL, so documents of that store are confined only when every source over it has a confined boundary.
29
+ - The bytes of a file take exactly the document admission: an `asset` request passes the deny rules (a refusal is logged once with the operation `asset`), then the same union of boundaries and confirmations, and the host's `asset` receives the admitting source's `readPath`. A boundary miss answers 403 without a log, since an embedded image outside every source is ordinary.
30
+
31
+ ## HTML
32
+
33
+ Markdown HTML returned by a source is sanitized on the server with DOMPurify, in worker threads, before it reaches the page loader's stream or the data route, so the browser never receives markup the allow-list rejects:
34
+
35
+ - removed elements: `script`, `style`, `noscript`, `template`, `form`, `button`, `textarea`, `select`, `option`, `object`, `embed`, `base`, `meta`, `link`, `frame`, `frameset`; `input` survives only as a disabled checkbox (task lists);
36
+ - removed attributes: every event handler (`on*`), `srcdoc`, `formaction`; `style` keeps only `text-align`;
37
+ - URLs keep only the `http:`, `https:`, `mailto:` and `tel:` schemes and relative references, and `data:` URLs are removed from every element;
38
+ - `iframe` survives only with an `https:` source whose host is listed exactly in `config.iframeHosts` (a longer host such as `www.youtube.com.example.net`, or a protocol-relative source, is removed);
39
+ - `allow`, `allowfullscreen`, `frameborder` and `referrerpolicy` survive only on iframes: `referrerpolicy` only as `no-referrer`, `origin`, `same-origin`, `strict-origin` or `strict-origin-when-cross-origin`, and `allow` reduced to `autoplay`, `clipboard-write`, `encrypted-media`, `fullscreen`, `picture-in-picture` and `web-share`;
40
+ - `target` survives only on `a` and `area`, and a link with any `target` other than `_self` gets `rel="noopener noreferrer"`;
41
+ - `id` attributes that would clobber `document` or form properties (`cookie`, `forms`, `location`, …) are removed — except on headings, which the table of contents needs and which cannot clobber `document`;
42
+ - the markup converters write for GitHub alerts and Zenn's message and details containers survives as written, through DOMPurify's own allow-lists: `div`, `p`, `aside`, `span`, `details` and `summary` with their `class` (and `aria-hidden` on a message's symbol), and an alert's inline SVG icon, `svg` with its `class`, `viewBox`, `version`, `width`, `height` and `aria-hidden` and `path` with its `d` (the README's Documents gives the markup, and its Load the styles the `@aiquants/markdown` stylesheets that draw it). Inside it, as anywhere else, event handlers, `script`, `foreignObject`, `use` and `animate` are removed and URLs follow the rules above;
43
+ - `data-*` attributes survive only for the Markdown renderer's contract — `data-link-kind`, `data-doc-path`, `data-doc-exists`, `data-footnote-ref` and `data-footnote-backref` on links, `data-asset-src` and `data-asset-exists` on images, `data-footnotes` on sections — so the renderer's plugin triggers (`data-plugin-type` and its companions) never reach the browser; SVG and MathML pass through DOMPurify's own allow-lists.
44
+ Of a link the renderer hands to the explorer's link component (every link with an `href` that is not external), the anchor the explorer renders carries the link's `title`, `lang`, `dir`, `role`, `aria-*`, `data-footnote-*` and an `id` in the `user-content-` namespace, which the renderer passes on, beside the explorer's own `data-link-kind`, `data-doc-path` and `data-doc-exists`; the renderer passes no event handler, `style`, `href`-like or other attribute (see the README's Documents).
45
+
46
+ ### The viewer's allow list
47
+
48
+ The document viewer (`@aiquants/markdown`) applies its own allow list to the sanitized HTML on every render, the server render and the browser alike, so the page holds the intersection of both rules. Where the viewer is stricter, its rule is the one readers see:
49
+
50
+ - `class` keeps only the viewer's own tokens — the converter's markup for GitHub alerts and their Octicon titles, Zenn's messages and details, footnotes (`footnotes`, `footnote-ref`, `footnote-backref`) and formulas (`math`, `inline`, `display`) — and a code block's `language-*`; every other token is dropped, so a document cannot restyle or position an element with the explorer's or the host's classes. A task list's `contains-task-list` and `task-list-item` do not survive either, so task-list items keep their list markers;
51
+ - `id` survives only on headings, on a converter's footnotes (`fnref:N`, or `fnrefK:N` for a later reference to the same note, on a reference's `sup`; `fn:N` on a note's `li`) and in the `user-content-` namespace (remark-gfm's footnote ids); every other element, a link included, loses its id, so a document cannot take a name the page or a script looks up;
52
+ - the author's `target` and `rel` never survive (the viewer opens an external link in a new tab itself, with `rel="noreferrer"`), and neither do `srcset`, `tabindex` or `name` outside links.
53
+
54
+ The sanitizer itself keeps `class` and `id` as DOMPurify does and leaves their reduction to the viewer. The viewer's lists are the single definition of the converter's class tokens and fragment targets, and `@aiquants/markdown` exports neither the lists nor the functions that apply them, so the sanitizer could adopt the rule only as a copy: a second definition that drifts from the converter and the viewer, keeping or dropping a token the other does not.
55
+ Every document the explorer shows is rendered by the viewer, on the server and in the browser, so no page holds a class or id outside the intersection without that copy.
56
+ The sanitized `htmlContent` of the page data and of the data route's answers still carries the classes and ids DOMPurify kept; a host that renders it outside the explorer reduces them itself (see [What the host must do](#what-the-host-must-do)).
57
+
58
+ A markdown document that lists more than 10,000 headings (`EXPLORER_MAX_HEADINGS`) is refused as `too-large` before it is sanitized: every listed heading travels with the document — copied by the server, sent in the data response, kept in the browser's document cache and rendered as an entry of the viewer's table of contents — and a parser can list headings its HTML never renders (`#` lines inside a math block), so none of the HTML's limits bounds the list. A host's own cache of parse results can rely on the same bound.
59
+ Sanitizing never runs on the request thread. HTML larger than `sanitizer.maxHtmlBytes` (4 MiB by default) is refused as `too-large` at once; the rest goes to a pool of worker threads (`sanitizer.workers`, 2 by default), shared by every explorer of the process with the same pool settings, with a per-document deadline (`sanitizer.deadlineMs`, 30 s by default): a document that misses it is `too-large`, and its worker is replaced.
60
+ In the worker the nesting limit is enforced while the HTML is parsed, by the parser and in the mode DOMPurify's jsdom uses (parse5, the input as a whole document, scripting disabled), before DOMPurify runs and within the document's deadline.
61
+ The parse stops, and the document is `too-large` without being sanitized, as soon as an element is placed deeper than 512 levels (the depth at which Chromium's HTML parser stops nesting elements; the children of `body` are at depth 1 and a `template`'s contents one level below it), or as soon as the parser has created more elements than the HTML has characters plus `html`, `head` and `body` (only formatting elements the parser reopens over and over can do that; an ordinary document stays far below).
62
+ Before DOMPurify runs, the guard also counts every element, attribute and comment the parser creates and refuses the document (`too-large`) once the count passes (`sanitizer.workerHeapMb` − 64 MiB − `sanitizer.maxHtmlBytes` × 32 bytes) ÷ 4 KiB, which is 81,920 nodes by default, or once it has seen more than 4,096 comments wherever they sit. jsdom inserts each comment the parser attaches to the document itself (before `<html>` or after `</html>`) in time that grows with those already there (4,000 took 0.37 s and 8,000 took 1.4 s in the worker, while 8,000 comments inside the body took 0.14 s), so the cap keeps that worst case near 0.4 s. The constants are measured: the worker needs about 48 MiB of its own, a parsed node costs at most about 3.4 KB at the sanitizer's peak and a byte of HTML at most about 31 bytes, so a document the guard accepts fits in the worker's heap. The budget does not depend on the worker's heap limit and holds even where a process-wide `--max-old-space-size` overrides it.
63
+ Before parsing, a linear scan follows every `<` or `</` followed by a letter through the tokenizer's tag states, quotes included, and refuses the document once a tag could hold more than 1,024 attributes, duplicates included; it also reads text the tokenizer would treat as text, so it may refuse more, never less. While parsing, `html` and `body` may receive at most 1,024 attributes through repeated tags. parse5, jsdom and DOMPurify look each attribute up in its element's list, so one element's attributes cost time that grows with the square of their number: without the cap, 40,000 attributes in one tag take 36 s and 79 repeated `body` tags of 1,000 attributes 90 s in the worker, while 78 tags of 1,024 take 1.8 s. The guard also adds up the name and value characters of every attribute the parser creates, including the copies it gives a formatting element it reopens or the adoption agency recreates, and refuses the document once they exceed `sanitizer.maxHtmlBytes` (a document whose elements each appear once never does): Without the budget, 1 MiB of copied `title` makes the worker write 210 million characters; the budget refuses it in about 0.1 s. Sanitized HTML longer than 4 × `sanitizer.maxHtmlBytes` UTF-16 code units (16,777,216 by default; ordinary HTML measured at most 2.05 characters per byte) is a cached `too-large`: the worker never posts it and the service refuses it should it arrive, so neither the cache (where the longest outcome weighs 32 MiB of the default 64 MiB) nor a caller ever holds more.
64
+ Every sanitizer worker starts with Node's `resourceLimits`: `workerHeapMb` MiB of old generation (512 by default) and 32 MiB of young generation. A worker whose heap runs out is ended with `ERR_WORKER_OUT_OF_MEMORY`, and its document is answered `too-large` and cached. The young generation is kept small on purpose: with Node's default for a worker (192 MiB), a worker at its limit could abort the whole process instead of ending alone, because one scavenge can promote more than the 16 MiB of leeway Node gives it.
65
+ An element counts at the depth where the parser first places it, so when the parser later moves misnested markup to a shallower place (the adoption agency for misnested formatting tags, a `frameset` replacing `body`), a document whose finished tree is at most 512 levels deep can still be refused; a deeper one is never accepted. DOMPurify's own walk also stops at the first element deeper than 512 levels.
66
+ Each request names its reader's scope. The pool admits at most `sanitizer.jobsPerScope` documents per reader running or waiting and `sanitizer.maxQueuedJobs` waiting in all; a document beyond either is `failed`, logged without the principal and not remembered, so a later request tries again.
67
+ A page sends the document of every panel to the sanitizer at once for one reader, so the admission defaults follow the multi view: `jobsPerScope` defaults to `multiView.maxPanels` + max(1, `workers` − 1) + 3 — the panels, the previous page's documents still running after a reload (a running job is not stopped when its requests leave, and one reader runs at most max(1, `workers` − 1) at once), the two intent prefetches and one spare; 16 with the default 12 panels and two workers. `maxQueuedJobs` defaults to four times `jobsPerScope`, also when it is set. Neither may be set below `multiView.maxPanels` (a `RangeError` naming both at startup). All anonymous readers share one scope and therefore one admission budget.
68
+ A reader runs at most `workers` − 1 documents at once (one when there is a single worker), and a free worker goes to the reader running the fewest, so one reader cannot starve the others; all anonymous readers share one scope.
69
+ Concurrent requests of one reader for the same input share one job; jobs of different readers are separate, and the first to finish answers the others; a waiting job is dropped when every request waiting on it has left. The first outcome remembered for an input stays: it answers every other job still pending for that input and drops those still waiting, and a duplicate already running finishes on its worker without answering anyone or replacing that outcome, so a document near the deadline or the heap limit cannot flip between `sanitized` and `too-large`.
70
+ Every outcome that describes the HTML (a refusal for nesting, the node budget, the heap or the deadline included) is remembered by the SHA-256 of the input in a cache bounded by count (1024 entries) and by bytes (`sanitizer.cacheBytes`, 64 MiB by default), so a document read by many readers is sanitized — or refused — once. A worker is sent a document only after it has reported that it is ready. A worker that fails before then (it cannot be started, is not ready within the deadline, or fails or exits while starting) never received the document: the document is answered `failed`, the cause is logged, and the outcome is not cached. A failure while a worker holds a document it was sent, such as an uncaught error or an unexpected exit, is that document's crash and stays a cached `failed`, whatever order Node delivers the worker's error and its messages in. Running out of heap and missing the deadline stay a cached `too-large`. Creating the explorer fails when the worker file is missing next to the server module; where DOMPurify cannot run, it refuses instead of returning its input unchanged, and the document fails.
71
+
72
+ Fixed or absolutely positioned markup in a document cannot cover the host or the explorer: the document region is its containing block and clips its painting (in a multi-view panel, the panel's frame clips it). Such markup can still cover the rest of its own document.
73
+
74
+ The rest of a source's output is validated before it is sent: image and media URLs must be `http:`, `https:` or relative, media types `video`, `audio` or `pdf`, notice links `http:`, `https:`, `mailto:` or relative, heading depths 1–6, and failure codes 1–64 lowercase letters, digits or hyphens. A violation is logged and the reader sees a generic failure (an empty notice list for notices).
75
+ In a tree or children listing, an entry with an empty name, or with a path the browser cannot address (empty, too long, a lone surrogate, a control character or a backslash — for example macOS's `Icon\r`), is left out together with its subtree and reported through the `logger` as `[markdown-explorer] <operation> left out an entry`, with the detail `{ operation, sourceId, path, error }`, whose `error` names the reason (an entry outside its source's boundary is reported the same way); any other malformed listing becomes `failed`. The browser applies the same checks to the data route's answers.
76
+
77
+ ## Responses
78
+
79
+ - Every JSON data response carries `Content-Type: application/json; charset=utf-8`, `Cache-Control: private, no-store`, `X-Content-Type-Options: nosniff`, `Content-Security-Policy: default-src 'none'; frame-ancestors 'none'; sandbox` and `Cross-Origin-Resource-Policy: same-origin`, and has a JSON body.
80
+ - The page loader's responses carry `Cache-Control: private, no-store` through the exported `headers`, to both the HTML document and the single-fetch data.
81
+ - The data route answers a React Router single-fetch request (`<dataPath>/<operation>.data`: its own URL path ends in `.data`, which React Router strips from the route's `url` argument and `params` but not from `request.url`) with a `not-found` failure (404) before authenticating or calling a source. React Router would run the route for it, read the whole response into memory and keep none of the headers above but `Set-Cookie`.
82
+ - Status codes: 200 for a tree, children, document or refresh answer (a refresh answers the reloaded tree, as a tree request does, failures included), 400 for a missing or malformed parameter, 401 when the reader is signed out, 403 when the reader may not use the explorer (or a refresh, or a document request marked as one's, lacks its proof), 404 for an unknown operation or source and for a children request to a source without `children()`, 405 (with `Allow`) for a wrong method.
83
+ When `authenticate`, `principalKey` or the route itself throws, the answer is the failure the exception maps to, with that failure's status: an `ExplorerError` keeps its reason (`too-large` is 413; `not-configured`, `unsupported` and `failed` are 500; the other reasons take the statuses above), and any other exception is logged and answered `failed` with 500.
84
+ A document or folder that cannot be shown — missing, denied, too large — is answered with status 200 and a failure in the JSON, so one inaccessible document does not break the multi-panel view.
85
+ - The tree refresh (`POST {dataPath}/refresh`) requires the custom request header `X-Markdown-Explorer: refresh` and refuses a request whose `Sec-Fetch-Site` is present and not `same-origin`, so a cross-site form cannot trigger it. It reloads the tree from the source (the source receives `refresh: true`) and answers a tree whose walk began after the refresh arrived; refreshes arriving while a walk runs share one trailing walk, so repeated refreshes never run more than one walk of a tree at a time.
86
+ Once the reload settled and unless the request went away, `onTreeRefreshed` receives the tree's cache key, which a multi-process host relays to its other processes' `forgetTree`; the response does not wait for a promise it returns, and a throw or that promise's rejection is logged (operation `refresh`).
87
+ - A document GET carrying `X-Markdown-Explorer: refresh` marks a request a refresh triggered: the host's `document` (and the admitting source's `confirm`) receive `refresh: true` and skip their caches. Such a request must carry the header's exact value and be same-origin (`Sec-Fetch-Site` absent or `same-origin`), as the refresh itself, or it is answered 403 before any host call, so a cross-site page cannot make the host re-parse documents.
88
+ - The `asset` operation answers bytes instead of JSON (its failures stay JSON failures with the statuses above). GET and HEAD only; a request whose own URL path ends in `.data` is 404 before authentication; signed out 401 and forbidden 403 with no redirect; a missing, repeated or unacceptable `path` 400; a denied or unadmitted path 403.
89
+ After admission it answers 304 for a matching `If-None-Match`, 200 for the whole file (without `Accept-Ranges` and `Content-Length` when the host reports no size; empty with `Content-Length: 0` for an empty file), 206 for one satisfiable byte range, and 416 (`Content-Range: bytes */<size>`, `Cache-Control: private, no-store`) for a range starting at or after the end; HEAD carries the GET's status and headers without a body.
90
+ `If-None-Match` matches the ETag (`"<version>"`) as RFC 9110 evaluates it: `*`, or a comma-separated list holding the ETag by the weak comparison, so `W/"<version>"` matches too, while a malformed element matches nothing; an `If-Range` counts only when it is the ETag itself (the strong comparison).
91
+ `If-None-Match` and `Range` are read only after admission, so a path outside every boundary is 403 even with a matching validator.
92
+ Every 200, 206 and 304 carries the declared type as a browser parses it (else `application/octet-stream`), `X-Content-Type-Options: nosniff`, `Content-Disposition` (`inline` or `attachment`, see [What the explorer serves](#what-the-explorer-serves)), `Content-Security-Policy: default-src 'none'; media-src 'self'; style-src 'unsafe-inline'; sandbox` except for PDF, `Cross-Origin-Resource-Policy: same-origin`, `Cache-Control: private, max-age=60, stale-while-revalidate=600, no-transform` and `Vary: Cookie`.
93
+ Inline audio and video get `sandbox allow-same-origin` instead of `sandbox`: the browser's media document of a file opened on its own fetches the file again, and under a plain `sandbox` its opaque origin would send that request cross-site, without the reader's `SameSite` cookie, to be answered 401; without `allow-scripts` no script runs.
94
+ The authentication's headers (a refreshed session cookie) go onto every asset response; the reader's scope header never does.
95
+ A body with a `Content-Length` carries exactly that many bytes: the bytes past it are dropped and the host's stream is cancelled, and a stream that ends short fails the response, so a file that changes while it is served never gets a body that disagrees with its length and ETag.
96
+ - Responses never contain tokens, principals, server paths or raw error messages. A failure carries only its reason and an optional host-defined code (`[a-z0-9-]{1,64}`); everything else goes to the server's `logger`. `new ExplorerError` throws a `RangeError` for a malformed code, and a failure document returned with one is logged and sent as `failed`.
97
+
98
+ ## Caches and identity
99
+
100
+ - The server's tree cache is keyed by the reader's scope (derived from the principal's stable key) and the source. `cacheScope: "shared"` lets readers share one entry when the tree does not depend on who reads it; `createFileSystemSource` uses `"principal"` unless told otherwise, which a source wrapped to filter its tree per reader must keep. Entries expire after `treeCache.ttlMs`, and the cache holds at most `treeCache.maxEntries` trees and `treeCache.maxBytes` of their JSON.
101
+ It never keeps an empty or failed tree and never answers a tree past its time to live, counted from the moment its walk began: a page load or tree request after it waits for one new walk, which every concurrent request shares. At most one walk of a tree runs at a time per key; a walk a refresh or `forgetTree` superseded keeps running for the requests already waiting on it (they receive its tree, which is not kept) and is aborted once none waits, and the next walk starts once it settles (at once for a source that honours its signal).
102
+ Each process keeps its own cache: `forgetTree(key)` drops a key another process refreshed, without starting a walk.
103
+ - Each data response carries an opaque, pseudonymous scope (`X-Markdown-Explorer-Scope`): an HMAC of the principal's key under `scopeSecret`, which nobody without the secret can link to a principal. When the browser sees a different scope — another reader signed in on the same browser — it drops every cached tree and document, clears the explorer's storage and reloads the page data (once per run of the loader, without cancelling a loader that is running). A 401 erases nothing and reloads the page data, whose loader answers a signed-out reader with `signedOutPage`.
104
+ - The page data carries the same scope, and the explorer records it in its storage namespace. On every page load, and whenever the page data names another reader, the recorded scope is compared during render — before any part of the explorer reads browser storage — and the namespace is erased unless it belongs to the page's reader (keys without a recorded scope are erased too).
105
+ A new scope also clears and reseeds both browser stores in a layout effect, before the browser paints, and rebuilds the explorer's components, so no document of the previous reader is drawn and no expansion state, selection or panel arrangement survives in memory either. Each reader also has a live-region announcer of their own: the rebuilt explorer's regions (the root's and the sheet's) start empty, and neither what the previous reader was last told (a folder or document name) nor a message their work posted just before the switch reaches it.
106
+ - The server keeps no failed tree and asks the source again for every document request (the sanitizer's cache only remembers what it made of a given HTML input). In the browser a failure is kept only until a pane has shown it: opening it again, or "Try again", asks the server again.
107
+
108
+ ## What the host must do
109
+
110
+ - Authenticate in `authenticate` and keep `principalKey` stable for one reader.
111
+ - Set `scopeSecret` from configuration, kept out of the source code and the browser: at least 32 characters, the same on every server instance.
112
+ - Keep the package external in your server bundle, so the sanitizer's worker file stays next to the server entry.
113
+ - Implement the ports: `document` and `asset` (and `children` for lazy folders) read what the explorer admitted, with the `readPath` it hands them, and report the same `version` for the same file.
114
+ The explorer serves every file's bytes itself (see [What the explorer serves](#what-the-explorer-serves)), so the host writes no asset route; the `asset` port only finds the file and opens its bytes, through a handle it checked (for example by comparing the opened file's device, inode, size and modification time with those of the checked path, refusing a file that has changed since it was found) and aborting with the request's signal.
115
+ - Refuse React Router single-fetch requests in any code your data route module runs before it hands the request to `dataLoader` or `dataAction` (which refuse them themselves), and in every resource route of your own: answer 404 for a request whose own URL path (`new URL(request.url).pathname`) ends in `.data`, before authenticating or touching storage.
116
+ React Router strips that suffix from `params` and from the loader's `url` argument, runs the loader for such a request, reads the whole `Response` body into memory and drops every header but `Set-Cookie`.
117
+ - When a source addresses documents by ID rather than by name (a cloud drive, for example), the deny rules can only filter its listings by entry name; its `document` function must refuse denied names itself, since the request carries only the ID.
118
+ - The explorer checks the deny rules on the document's own path, not on the files your parser reads while it renders it. If your Markdown parser expands include directives (for example `<!-- @import "…" -->`), apply the same rules (`isDeniedPath`) to every file it reads, recursively, or turn the directive off for the explorer's documents; otherwise a document can pull a denied file (an environment file, a key) into the page.
119
+ `@aiquants/markdown`'s `parseMarkdown` leaves the directive unexpanded unless `resolveImports` is set, and applies the `deniedPathSegments` it is given to every import, image and link target: give it the explorer's rules (`DEFAULT_DENIED_PATH_SEGMENTS`, or your `deny`).
120
+ - Render a document's `htmlContent` only through `<MarkdownExplorer>` or `@aiquants/markdown`'s renderers. Anywhere else (your own `dangerouslySetInnerHTML`, for example) it carries the classes and ids the sanitizer kept, so reduce them as the viewer does (see [The viewer's allow list](#the-viewers-allow-list)); otherwise a document can lay an element over your page with your own utility classes and take ids your scripts look up.
121
+ - Treat every GET route on the explorer's origin as reachable from a document. A rendered document makes the reader's browser send credentialed same-origin GET requests through its images (`![](/path)`, `<img src>`), its media (`<video poster>`, `<audio src>`) and the frames of listed hosts without the reader's interaction, and through its links once the reader follows one; the sanitizer and the viewer keep relative and root-relative URLs, since a document's own images are legitimate, so they cannot tell these requests apart.
122
+ An endpoint that changes state on a GET — signing the reader out, deleting, starting a download of the reader's files — can therefore be triggered by any document's author, without the reader's interaction.
123
+ Make such endpoints non-GET, or refuse (before changing anything, and without `Set-Cookie`) a request that carries `Sec-Purpose` — a prefetch or prerender, which a host's speculation rules can start for a document's same-origin links with `Sec-Fetch-Dest: document` before, or without, a click — or whose `Sec-Fetch-Dest` is present and is neither `document` (a typed address or a followed link) nor `empty` with `Sec-Fetch-Site: same-origin` (`fetch` from your own origin, including React Router's single-fetch data; another origin of the same site gets the `SameSite=Lax` cookies too).
124
+ - List only the iframe hosts you trust in `iframeHosts`, and list the host's own origin only when documents must frame it: every page of a listed host can be framed by a document, and a frame of your own origin is same-site, so it carries the reader's cookies (`SameSite=Lax` included) and the page's side effects happen in an invisible frame.
125
+ - Serve a content security policy for every page of the app — from the root route, the document handler or the edge — rather than on the explorer's route responses alone: a client-side navigation keeps the document it started in, and with it that document's policy, so a policy sent with the explorer's pages is missing when the reader arrives by a client-side navigation and stays on every page visited after them.
126
+ The explorer needs `frame-src`, `img-src` and `media-src` to allow `'self'`, the origin of its own asset URLs (PDF documents open in a frame from that URL), plus `frame-src https://<host>` for each `iframeHosts` entry, and works under `object-src 'none'` and `base-uri 'none'`.
127
+ - Call `clearExplorerStorage(storageNamespace)` when a reader signs out of a shared browser. It throws a `RangeError` for a malformed namespace (empty, containing `:` or spaces, longer than 64 characters), so a typo cannot clear another namespace's keys or nothing at all.
128
+
129
+ ## What the explorer serves
130
+
131
+ - Any regular file the document admission admits, whatever its kind: the bytes of image and media documents, the whole file behind a text preview, every `file` document (a file the explorer cannot show) and every image embedded in Markdown.
132
+ - Inline only for raster images, audio, video and PDF; everything else, SVG and an empty or unknown type included, as an attachment. Both dispositions name the file (`filename*=UTF-8''<name>`, the RFC 8187 encoding of the name with `'`, `(`, `)` and `*` percent-encoded too), so a saved file keeps its name although every asset URL's path ends in `asset`.
133
+ - The type is the one the host declares, parsed as a browser parses it, never guessed from the name; `nosniff` stops the browser from guessing either, and the sandboxing policy keeps an SVG or HTML file opened on its own from running script.
134
+ - For `createFileSystemSource`, files of unknown extensions are downloadable, and so are the whole bytes of every Markdown and text file the source admits: raw Markdown keeps what the rendered view drops (front matter, HTML comments), and a text file is served beyond its preview. Its deny rules — `DEFAULT_DENIED_PATH_SEGMENTS` (environment files, keys and certificates, credential files, Terraform state, shell histories) plus the host's `deny` patterns — are the only filter, so add a pattern for anything else the folder must withhold.
@@ -0,0 +1,129 @@
1
+ # URL contract
2
+
3
+ This page is part of the host-facing contract of `@aiquants/markdown-explorer`; its summary is in the [README](../../README.md#views-and-urls).
4
+
5
+ ## Shape
6
+
7
+ | View | Path | Query |
8
+ | --- | --- | --- |
9
+ | Tree | `{basePath}/{views.tree}` or `{basePath}/{views.tree}/{document}` | `source`, `style`, `align` |
10
+ | Multi-panel | `{basePath}/{views.multi}` | `source`, `doc` (repeated, in panel order), `maximized` |
11
+ | Single | `{basePath}/{views.single}/{document}` | `source`, `style`, `align`, `showToc`, `showStyle`, `showAlign`, `padding`, `width`, `maxWidth` |
12
+
13
+ - `basePath` and the view slugs come from `defineExplorerConfig` (`tree`, `multi` and `single` by default). The data route lives at `dataPath` and is not a page.
14
+ - `source` names the source shown in the tree: one of the server's sources, whose ids come from `createMarkdownExplorer`'s `sources` in their order (the browser reads them from the page data's `sources`). The first is the default and is written by **omission**; it is never written explicitly. A source id is lowercase letters, digits and `-`, starting with a letter or digit (`isExplorerSourceId`).
15
+ - The source never limits which documents can be open: a path names one document; it may appear in several sources' trees, and it opens when the boundary of at least one source admits it, whichever source the URL names. Switching the source therefore keeps every open document, and a document no source's boundary admits is answered `forbidden` in every view.
16
+ - `explorerHref(config, target)` is the only function that writes explorer URLs, and `resolveExplorerLocation(config, splat, search)` is the only one that reads them. The server's loader and the browser call the same two functions, each with a configuration joined with the server's source ids by `withExplorerSources(config, sourceIds)` (an `ExplorerLocationConfig`; no ids, an invalid id or a repeated one throws a `RangeError`).
17
+ `explorerHref` has two forms: any target with such a configuration, and a target without `source` (`ExplorerSourcelessHrefTarget`, the default source) with a plain `ExplorerConfig`, for links built where the server's sources are not known (a host menu). A target with a `source` and a configuration without ids throws a `TypeError` naming `withExplorerSources`, and so does `resolveExplorerLocation` given such a configuration.
18
+ - `explorerHref` writes only canonical URLs. It throws a `RangeError` for a target the resolver would redirect or partly ignore: an unknown source, an unacceptable document path, multi-panel documents that repeat or exceed `multiView.maxPanels`, a maximized document that is not open, a `style` naming no configured document style, an `align` other than `left`, `center` or `right`, and an embedding value outside the rules below (a non-boolean switch, a padding outside 0–256 or not writable in decimal digits, a width that is not one of the accepted CSS lengths).
19
+ Resolving a link it wrote therefore always yields the very location it was given.
20
+
21
+ ## Document segment
22
+
23
+ A document path occupies one URL segment of the tree and single views.
24
+
25
+ - Before percent-encoding, `%` is escaped to `%25`, and then the path is encoded with `encodeURIComponent`. React Router decodes a splat once and also turns `%2F` into `/`; because a literal `%` can never reach that step, the decoded splat is always the path with only the `%25` escape left to undo, and no name can be decoded twice.
26
+ - A segment that would end in `.data` has its last dot written as `%2Edata`, because React Router reads a request path ending in `.data` as a single-fetch data request.
27
+ - The remainder of the splat after the view slug is the document path, so `/docs/tree/guide%2Fintro.md` and `/docs/tree/guide/intro.md` name the same document (`guide/intro.md`). Only the first is what `explorerHref` writes; both are accepted and neither is redirected.
28
+ - In the multi-panel view documents are query values (`doc`, `maximized`) and need no segment encoding.
29
+
30
+ ## Canonical form
31
+
32
+ Every URL resolves either to a location or to a redirect to its canonical form. Decisions are made on the parsed result — never by comparing URL text — and unknown query parameters are ignored, so resolving a canonical URL always yields a location and canonicalization is idempotent.
33
+
34
+ | Input | Outcome |
35
+ | --- | --- |
36
+ | The base path, an unknown view slug, an empty splat | Redirect to the tree view without a document, keeping a valid `source` |
37
+ | A tree or single URL whose document segment is empty (a trailing slash) or not an acceptable path | Redirect to the tree view without a document |
38
+ | A single URL without a document | Redirect to the tree view without a document |
39
+ | `source` unknown, repeated, or equal to the default | Redirect to the same location with `source` dropped |
40
+ | Multi-panel `doc` values that are not acceptable paths, repeated, or beyond `multiView.maxPanels` | Redirect with only the first occurrence of each acceptable path, in order, up to the limit |
41
+ | `maximized` repeated, or not one of the open documents | Redirect without `maximized` |
42
+ | A path segment after the multi-panel slug | Redirect to the multi-panel view without it |
43
+ | `style` naming no configured document style, `align` other than `left`, `center` or `right` | Ignored (the reader's own setting applies) |
44
+ | `padding` outside 0–256 or not written in decimal digits (signs, exponents, hexadecimal, surrounding spaces and `Infinity` are not numbers here), `width` / `maxWidth` not a CSS length in `px`, `rem`, `em`, `ch`, `vw` or `%` | Ignored (the default applies) |
45
+ | `showToc`, `showStyle`, `showAlign` other than `false` | Ignored (the control is shown) |
46
+
47
+ An acceptable document path is non-empty, at most 4096 UTF-16 code units long, well-formed (no lone surrogate, which no URL can carry), and contains neither a control character (U+0000–U+001F, U+007F) nor a backslash. Whether the path exists, and whether it is denied, is decided by the server for the document request; the URL itself stays as written.
48
+
49
+ The page loader answers a non-canonical URL with a redirect, and the browser applies the same canonicalization with a replacing navigation when a URL is reached without the loader (a navigation that does not revalidate). A redirect result carries the location its URL names, and the browser draws that location while the replacing navigation catches up, so the host's frame stays mounted.
50
+
51
+ ## Document links
52
+
53
+ A Markdown link that should open another document inside the explorer is resolved by the host's parser, which marks it on the `a` element:
54
+
55
+ - `data-link-kind="doc"` and `data-doc-path`: the target's document path exactly as the explorer addresses documents — with the source's `pathPrefix`, resolved against the linking document's folder (`./` and `../` applied), percent-decoded once, without query or fragment, and an acceptable document path.
56
+ - The fragment of the `href` (`#section`) is kept: the explorer appends it to the link it writes. A link to a section of the document being shown is, in a page view, a router navigation like any other document link (one history entry, the hash in the URL), after which the pane scrolls to the section; in a multi-view panel a plain click only scrolls the panel and the URL keeps no hash.
57
+ Either way the section reaches the start of the pane's own scroller while every scroller around it moves only as far as needed, and the focus moves to it when it was in the document (see [Layout](layout.md#the-document-region)); a host's React Router `<ScrollRestoration>` may scroll a page view's section into view itself first (see the README's hosting notes). The rest of the `href` is not used for document links.
58
+ - `data-doc-exists="false"` marks a target the parser knows is missing; the link still opens it, and the document shows its `not-found` failure.
59
+ - A link that resolves outside every source (above the prefix, for example) keeps `data-link-kind="doc"` without `data-doc-path`; the explorer renders it inert, titled by `inertLink`, as it does a link whose `data-doc-path` is not an acceptable document path.
60
+ - Fragment-only links (`#section`) are ordinary links: leave them unmarked. In a page view a fragment-only link is the browser's own fragment navigation; in a multi-view panel its `href` is the tree view of the panel's document with the fragment (what a new tab or a copied address opens), and a plain click scrolls the panel instead. A fragment that names no element of the document but HTML's top of the document (`#top` in any ASCII case, or the empty `#`) takes the pane back to its start (see [Layout](layout.md#the-document-region)).
61
+ - Links to other sites need no mark: the viewer renders a link without `data-link-kind` itself, in a new tab with `rel="noreferrer"` and never through the explorer's link component, when its `href` names a scheme in any case (`https:`, `HTTPS:`, `mailto:`) or another host (`//host/…`); `data-link-kind="external"` marks one explicitly. The sanitizer removes an `href` whose scheme it does not allow (`javascript:`, for example).
62
+ - A link keeps the attributes a converter gives it for assistive technology and footnotes: `title`, `lang`, `dir`, `role`, `aria-*`, `data-footnote-*` and an `id` in the `user-content-` namespace reach the anchor the explorer renders (the viewer hands them to the explorer's link component, and the sanitizer keeps `data-footnote-ref` and `data-footnote-backref` among the `data-*` attributes); the viewer drops any other `id` of a link.
63
+ A footnote reference's `id` may sit on its link (remark-gfm, `user-content-fnref-…`) or on the `sup` around it (goldmark, `fnref:N`, and `fnrefK:N` for a later reference to the same note), and a note's on its `li` (`user-content-fn-…` or `fn:N`): the viewer keeps exactly these, so each is the target of its link's `href`, and the reference keeps its `aria-describedby` and the back-reference its `aria-label`. An inert link's `title` is the explorer's reason instead of the document's.
64
+
65
+ ## Document images
66
+
67
+ The parser marks an image whose file lies in a source with that file's document path; the explorer writes the URL. The viewer renders an image's `src` exactly as the sanitized HTML has it, so the rules are:
68
+
69
+ - A relative `src` is resolved against the document's own path (with the source's `pathPrefix`, `./` and `../` applied, percent-decoded once). The parser writes the resolved document path to `data-asset-src` and writes **no** `src`; the explorer's sanitizer then sets `src` to the explorer's own asset URL for that path (`explorerAssetHref`, without `v`; see [Asset URLs](#asset-urls)) and overwrites any `src` an author wrote beside the mark.
70
+ - A target the parser knows is missing is marked `data-asset-exists="false"` beside its `data-asset-src`; so is a `src` that resolves above every source, with the `src` as written in `data-asset-src`. The sanitizer writes no `src` for such an image, nor for one whose `data-asset-src` is not an acceptable document path (a lone surrogate, a control character, a backslash, more than `MAX_DOCUMENT_PATH_LENGTH` characters), and the viewer shows it as a labelled placeholder with its `alt` text and that target instead of a broken image.
71
+ - The parser strips author-written `data-asset-src` and `data-asset-exists` from every image before it marks its own, so only the parser decides which file an image loads.
72
+ - Absolute URLs, root-relative paths and protocol-relative URLs are left as they are, without a mark.
73
+ - A relative `src` left as written resolves against the page URL, not the document, and loads the explorer page instead of the image. In the tree and single views `images/diagram.svg` in `/docs/tree/guide%2FREADME.md` requests `/docs/tree/images/diagram.svg`, which the page route answers with an explorer page (HTML). In the multi-panel view it resolves against the base path (`/docs/images/diagram.svg`) and is redirected to the tree view. Either way the image is broken and every image costs a page request.
74
+
75
+ The sanitizer keeps only these two `data-*` attributes on images. The image branch of the demo's parser (`demo/app/markdown.server.ts` in the source repository) is a complete example.
76
+
77
+ ## Asset URLs
78
+
79
+ The explorer serves every file it shows or offers for download through the `asset` operation of its data route. `explorerAssetHref(dataPath, path, version)` is the only writer of these URLs; no package entry exports it, so a host never builds one.
80
+
81
+ | Part | Rule |
82
+ | --- | --- |
83
+ | Path | `<dataPath>/asset`. GET and HEAD only; any other method answers 405 with `Allow: GET, HEAD`. |
84
+ | `path` | Required exactly once: the document path exactly as the explorer shows it (workspace-relative, `gcs://bucket/object`, `gdrive://…`), an acceptable document path; otherwise 400. `filePath` is not a parameter (`?filePath=` alone answers 400). |
85
+ | `v` | Written only on the URL a document carries (image, media, text and file documents), from the version the host reported: a cache-buster the server never reads. The response's ETag is always the file's current version. |
86
+ | Encoding | `encodeURIComponent` once, decoded once by `URLSearchParams`: exact for `%`, `#`, `?`, `+`, spaces, NFC and NFD names and a trailing `.data`. The request's own path always ends in `asset`, never in `.data`, so no name needs another escape. |
87
+
88
+ Two writers call it: the copy of the host's `document` answer, for the `url` of image, media, text and file documents (with `v`), and the sanitizer, for images marked with `data-asset-src` (without `v`). With `dataPath` `/docs/_data`:
89
+
90
+ - `![](./images/diagram.png)` in `guide/README.md` → `<img data-asset-src="guide/images/diagram.png" data-asset-exists="true">` from the parser → `src="/docs/_data/asset?path=guide%2Fimages%2Fdiagram.png"` after the sanitizer.
91
+ - An image document opened in the tree view: `url: "/docs/_data/asset?path=guide%2Fimages%2Fdiagram.png&v=<version>"`.
92
+ - An image embedded in a `gcs://` document: `/docs/_data/asset?path=gcs%3A%2F%2Fbucket%2Freports%2Ffig%201.png`.
93
+ - A linked `notes/report.xlsx` opens as a `file` document whose `url` is `/docs/_data/asset?path=notes%2Freport.xlsx&v=<version>`, answered as an attachment.
94
+
95
+ The browser accepts the `url` of an image, media, text or file document only when it is root-relative on the page's own origin (`/x`, never `//host` or `/\host`) and spelled in visible ASCII, which is exactly what the explorer writes. What the operation admits and answers is in the README's [Serving files](../../README.md#serving-files) and in [Security](security.md#what-the-explorer-serves).
96
+
97
+ ## Table of contents
98
+
99
+ Each entry of the document viewer's table of contents links to its heading's own address: the shown document's address as the explorer writes it, with the router's basename, followed by `#` and the heading's id percent-encoded (`encodeURIComponent`). In the tree and single views that address is the page's URL (the same view, with its tree source and display state); in a multi-view panel it is the tree view of the panel's document, with the URL's tree source, as for the panel's fragment-only links. An entry whose heading has no id links to the document itself.
100
+
101
+ - A plain click on an entry, or Enter on it, makes the explorer's own jump to the heading (see [History](#history) and [Layout](layout.md#the-document-region)).
102
+ - A modified or middle click, a new tab or a copied address opens the entry's href, so it reaches the heading in a new page; a panel's entry opens the tree view and never duplicates the multi view.
103
+
104
+ ## History
105
+
106
+ | Action | History entry |
107
+ | --- | --- |
108
+ | Selecting a document in the tree | Replaces the current entry (no reload of the tree) |
109
+ | Following a document link inside the tree or single view | Pushes |
110
+ | Activating an entry of the document viewer's table of contents in the tree or single view | Pushes the heading's hash (nothing when the URL's hash already names that heading); in a multi-view panel the URL does not change |
111
+ | Switching the tree's source | Pushes |
112
+ | Switching between the views (the view switch, "open selection in the multi-panel view") | Pushes |
113
+ | Any multi-panel action (open, close, maximize, restore, navigate inside a panel, rearrange (drag or move)) | Replaces; hiding and showing a panel do not change the URL (browser storage remembers them) |
114
+
115
+ A multi-panel action that follows a source switch before the router reports it keeps the new source.
116
+
117
+ The page loader only seeds the first paint; after that the browser reads through the data route. Export `shouldRevalidateExplorer` as the page route's `shouldRevalidate`: it runs the loader again only for a revalidation of the very same URL (`useRevalidator().revalidate()`, or a navigation to the current URL) and never after a form or fetcher submission elsewhere on the page, so moving between views, documents and sources never reloads the tree. Entering the explorer from another route always runs the loader.
118
+ A revalidation keeps the shown tree and documents on screen and replaces each only when its new value differs (a different tree also drops its loaded lazy folders); a failure, or a value the loader gave up on, leaves the shown one alone.
119
+
120
+ ## Display settings
121
+
122
+ The reader's display settings — the explorer width per view, the document style and the alignment — live in a cookie named by `settingsCookie`, so the server renders the first paint with them:
123
+
124
+ - the value is percent-encoded JSON with the keys `treeExplorer`, `multiExplorer`, `style` and `align`, written in that order and only when set;
125
+ - widths are percentages of the split, greater than 0 and at most 1/φ² (38.197 %), rounded to 0.01 %;
126
+ - the cookie is written with `Path=/`, `Max-Age=31536000`, `SameSite=Lax`, and `Secure` on HTTPS;
127
+ - a malformed cookie, or a value outside these rules, is treated as absent (key by key).
128
+
129
+ The `style` and `align` query parameters override the cookie for one URL without changing it. Without either, a document takes `@aiquants/markdown`'s own default alignment, so the explorer and the viewer always agree.
package/package.json CHANGED
@@ -1,6 +1,144 @@
1
1
  {
2
2
  "name": "@aiquants/markdown-explorer",
3
- "version": "0.0.0-stage",
4
- "stub": true,
5
- "description": "Temporary package placeholder for staged publishing"
3
+ "version": "0.1.0",
4
+ "description": "Markdown document explorer for React Router 8 (framework mode): a source tree with a single-document view and a multi-panel reading view, storage-agnostic server ports, golden-ratio layout and an en / ja UI.",
5
+ "sideEffects": [
6
+ "**/*.css"
7
+ ],
8
+ "type": "module",
9
+ "main": "./dist/index.cjs",
10
+ "module": "./dist/index.js",
11
+ "types": "./dist/index.d.ts",
12
+ "exports": {
13
+ ".": {
14
+ "import": {
15
+ "types": "./dist/index.d.ts",
16
+ "default": "./dist/index.js"
17
+ },
18
+ "require": {
19
+ "types": "./dist/index.d.cts",
20
+ "default": "./dist/index.cjs"
21
+ },
22
+ "default": "./dist/index.js"
23
+ },
24
+ "./server": {
25
+ "import": {
26
+ "types": "./dist/server.d.ts",
27
+ "default": "./dist/server.js"
28
+ },
29
+ "require": {
30
+ "types": "./dist/server.d.cts",
31
+ "default": "./dist/server.cjs"
32
+ },
33
+ "default": "./dist/server.js"
34
+ },
35
+ "./storage": {
36
+ "import": {
37
+ "types": "./dist/storage.d.ts",
38
+ "default": "./dist/storage.js"
39
+ },
40
+ "require": {
41
+ "types": "./dist/storage.d.cts",
42
+ "default": "./dist/storage.cjs"
43
+ },
44
+ "default": "./dist/storage.js"
45
+ },
46
+ "./styles/markdown-explorer.css": "./dist/styles/markdown-explorer.css",
47
+ "./package.json": "./package.json"
48
+ },
49
+ "files": [
50
+ "dist",
51
+ "docs/behavior",
52
+ "README.md",
53
+ "CHANGELOG.md",
54
+ "LICENSE"
55
+ ],
56
+ "dependencies": {
57
+ "isomorphic-dompurify": "^2.36.0",
58
+ "parse5": "^8.0.0"
59
+ },
60
+ "peerDependencies": {
61
+ "@aiquants/directory-tree": "^4.1.0",
62
+ "@aiquants/drag-drop-panels": "^0.9.1",
63
+ "@aiquants/markdown": "^5.0.0",
64
+ "@aiquants/resize-panels": "^2.1.0",
65
+ "@aiquants/virtualscroll": "^3.13.0",
66
+ "react": "^19.2.7",
67
+ "react-dom": "^19.2.7",
68
+ "react-router": "^8.0.0"
69
+ },
70
+ "devDependencies": {
71
+ "@aiquants/directory-tree": "4.1.0",
72
+ "@aiquants/drag-drop-panels": "0.9.1",
73
+ "@aiquants/markdown": "5.0.0",
74
+ "@aiquants/resize-panels": "2.1.0",
75
+ "@aiquants/virtualscroll": "3.13.0",
76
+ "@heroicons/react": "^2.2.0",
77
+ "@playwright/test": "^1.59.1",
78
+ "@tailwindcss/cli": "^4.1.18",
79
+ "@testing-library/jest-dom": "^6.9.1",
80
+ "@testing-library/react": "^16.3.2",
81
+ "@testing-library/user-event": "^14.6.1",
82
+ "@types/node": "^24.10.3",
83
+ "@types/react": "^19.2.7",
84
+ "@types/react-dom": "^19.2.3",
85
+ "@vitejs/plugin-react": "^5.1.2",
86
+ "@vitest/coverage-v8": "^4.1.10",
87
+ "jsdom": "^27.4.0",
88
+ "react": "^19.2.7",
89
+ "react-dom": "^19.2.7",
90
+ "react-router": "^8.3.0",
91
+ "rimraf": "^6.1.2",
92
+ "tailwindcss": "^4.1.18",
93
+ "typescript": "^5.9.3",
94
+ "vite": "^7.3.1",
95
+ "vite-plugin-dts": "^4.5.4",
96
+ "vitest": "^4.1.10"
97
+ },
98
+ "keywords": [
99
+ "markdown",
100
+ "explorer",
101
+ "react-router",
102
+ "react",
103
+ "multi-panel",
104
+ "directory-tree",
105
+ "document-viewer",
106
+ "typescript"
107
+ ],
108
+ "author": {
109
+ "name": "fehde-k",
110
+ "url": "https://x.com/fehdek"
111
+ },
112
+ "license": "MIT",
113
+ "engines": {
114
+ "node": ">=22.22.0"
115
+ },
116
+ "publishConfig": {
117
+ "access": "public"
118
+ },
119
+ "scripts": {
120
+ "build": "vite build && pnpm run build:css && node ../../.config/scripts/strip-dts-comments.mjs dist && node scripts/verify-dist.mjs",
121
+ "build:css": "tailwindcss -i src/styles/markdown-explorer.css -o dist/styles/markdown-explorer.css --minify",
122
+ "dev": "vite build --watch",
123
+ "watch": "vite build --watch",
124
+ "demo:dev": "cd demo && pnpm run dev",
125
+ "demo:build": "cd demo && pnpm run build",
126
+ "demo:start": "cd demo && pnpm run start",
127
+ "typecheck": "tsc --noEmit && tsc --noEmit -p demo/tsconfig.json",
128
+ "test": "vitest run",
129
+ "test:run": "vitest run",
130
+ "test:watch": "vitest --watch",
131
+ "test:coverage": "vitest run --coverage",
132
+ "test:e2e": "playwright test",
133
+ "test:e2e:ui": "playwright test --ui",
134
+ "lint": "biome check src tests",
135
+ "lint:fix": "biome check --write src tests",
136
+ "format": "biome format --write src tests",
137
+ "format:check": "biome format src tests",
138
+ "license-check": "pnpm dlx license-checker --production --onlyAllow \"MIT;Apache-2.0;BSD-2-Clause;BSD-3-Clause;ISC;Unlicense\"",
139
+ "clean": "rimraf dist",
140
+ "publish:patch": "pnpm run typecheck && pnpm run --if-present test && pnpm run build && node ../../.config/scripts/check-publish-leaks.mjs && pnpm version patch --no-git-tag-version --no-git-checks && pnpm publish --no-git-checks",
141
+ "publish:minor": "pnpm run typecheck && pnpm run --if-present test && pnpm run build && node ../../.config/scripts/check-publish-leaks.mjs && pnpm version minor --no-git-tag-version --no-git-checks && pnpm publish --no-git-checks",
142
+ "publish:major": "pnpm run typecheck && pnpm run --if-present test && pnpm run build && node ../../.config/scripts/check-publish-leaks.mjs && pnpm version major --no-git-tag-version --no-git-checks && pnpm publish --no-git-checks"
143
+ }
6
144
  }