decant-core 1.10.0 → 1.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +27 -6
- package/package.json +1 -1
- package/utils/html-to-markdown.js +56 -21
package/README.md
CHANGED
|
@@ -41,7 +41,7 @@ AI platforms don't expose stable public APIs for reading conversation history. N
|
|
|
41
41
|
|
|
42
42
|
Maintaining that per-platform logic in every exporter is wasteful and fragile. `decant-core` centralizes it:
|
|
43
43
|
|
|
44
|
-
- ✅ **
|
|
44
|
+
- ✅ **18 AI chat platform parsers** with normalized output — you get structured messages, models, metadata and Markdown, not DOM soup.
|
|
45
45
|
- ✅ **Web article extraction** — Mozilla Readability, Defuddle, and Article-Extractor run in parallel and arbitrate by content-quality scoring.
|
|
46
46
|
- ✅ **Detection utilities** — tell an "AI chat page" apart from a "regular web page" before you decide which parser to run.
|
|
47
47
|
- ✅ **Math & Markdown handling** — LaTeX normalization plus GFM tables/code fencing that survive round-trips into Obsidian, Logseq and Notion.
|
|
@@ -117,13 +117,34 @@ import { normalizeLatexMath } from "decant-core";
|
|
|
117
117
|
|
|
118
118
|
## Supported Platforms
|
|
119
119
|
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
120
|
+
18 AI chat platform parsers plus generic web article extraction:
|
|
121
|
+
|
|
122
|
+
| Platform | Parser | Extraction strategy |
|
|
123
|
+
| :---------------------------------- | :------------------------ | :----------------------------------------- |
|
|
124
|
+
| **ChatGPT** | `ChatGPTParser` | DOM + internal API |
|
|
125
|
+
| **Claude** | `ClaudeParser` | DOM + internal API + React fiber |
|
|
126
|
+
| **Google Gemini** | `GeminiParser` | DOM + batchexecute RPC |
|
|
127
|
+
| **Microsoft Copilot** | `CopilotParser` | DOM (multi-domain) |
|
|
128
|
+
| **Perplexity** | `PerplexityParser` | Internal API + DOM fallback |
|
|
129
|
+
| **DeepSeek** | `DeepSeekParser` | DOM + internal API (`fragments[]`) |
|
|
130
|
+
| **Qwen** | `QwenParser` | DOM |
|
|
131
|
+
| **Meta AI** | `MetaParser` | Internal API (GraphQL) + DOM fallback |
|
|
132
|
+
| **Mistral / Le Chat** | `MistralParser` | DOM |
|
|
133
|
+
| **Proton Lumo** | `LumoParser` | DOM (API is E2E-encrypted, not readable) |
|
|
134
|
+
| **Z.ai** | `ZAiParser` | Internal API (chat + batch) + DOM fallback |
|
|
135
|
+
| **Grok** | `GrokParser` | Internal API (response-node + load) + DOM |
|
|
136
|
+
| **Google AI Studio** | `GoogleAIStudioParser` | DOM |
|
|
137
|
+
| **NotebookLM** | `NotebookLMParser` | DOM |
|
|
138
|
+
| **Google Search AI (AI Overviews)** | `GoogleSearchAIParser` | DOM |
|
|
139
|
+
| **Gemini Cloud Assist** | `GeminiCloudAssistParser` | DOM |
|
|
140
|
+
| **Joyland** | `JoylandParser` | DOM |
|
|
141
|
+
| **Chub** | `ChubParser` | DOM |
|
|
142
|
+
| **Generic Web Article** | `ArticleParser` | Readability + Defuddle + Article-Extractor |
|
|
123
143
|
|
|
124
144
|
All parsers extend the base [`ChatParser`](ai/base.js) interface — a consistent `isAvailable(url)` +
|
|
125
|
-
normalized `parse()` contract. For the extraction-strategy breakdown and maintenance model, see
|
|
126
|
-
[SUPPORTED_PLATFORMS.md](SUPPORTED_PLATFORMS.md)
|
|
145
|
+
normalized `parse()` contract. For the full extraction-strategy breakdown and maintenance model, see
|
|
146
|
+
[SUPPORTED_PLATFORMS.md](SUPPORTED_PLATFORMS.md), also published as the
|
|
147
|
+
[platform matrix](https://covai-labs.github.io/decant-core/platforms/) on the developer docs site.
|
|
127
148
|
|
|
128
149
|
---
|
|
129
150
|
|
package/package.json
CHANGED
|
@@ -192,28 +192,63 @@ export function convertToMarkdown(htmlContent, options = {}) {
|
|
|
192
192
|
registerMath(el, latex, isBlock, clone.ownerDocument);
|
|
193
193
|
});
|
|
194
194
|
|
|
195
|
-
// 3. Process Google Search SGE LaTeX images with [data-xpm-latex]
|
|
196
|
-
clone
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
195
|
+
// 3. Process Google Search SGE LaTeX images with [data-xpm-latex] or fallback math roots
|
|
196
|
+
clone
|
|
197
|
+
.querySelectorAll(
|
|
198
|
+
"[data-xpm-latex], [data-xpm-copy-root][data-xpm-copy-text], [data-xpm-copy-root] img[alt]",
|
|
199
|
+
)
|
|
200
|
+
.forEach((el) => {
|
|
201
|
+
if (!clone.contains(el)) return;
|
|
202
|
+
const copyRoot = el.hasAttribute("data-xpm-copy-root")
|
|
203
|
+
? el
|
|
204
|
+
: el.closest("[data-xpm-copy-root]");
|
|
205
|
+
if (!copyRoot) return;
|
|
206
|
+
const blockContainer = el.closest(".cPGBZb");
|
|
207
|
+
const inlineWrapper = el.closest(".mTEjhd") || el.closest(".dteT0b");
|
|
208
|
+
const container = blockContainer || inlineWrapper || copyRoot;
|
|
209
|
+
if (!container.parentNode) return;
|
|
210
|
+
|
|
211
|
+
const img = el.tagName === "IMG" ? el : copyRoot.querySelector("img");
|
|
212
|
+
const latex =
|
|
213
|
+
copyRoot.getAttribute("data-xpm-latex") ||
|
|
214
|
+
el.getAttribute("data-xpm-latex") ||
|
|
215
|
+
copyRoot.getAttribute("data-xpm-copy-text") ||
|
|
216
|
+
(img && img.getAttribute("data-xpm-latex")) ||
|
|
217
|
+
el.getAttribute("alt") ||
|
|
218
|
+
(img && img.getAttribute("alt")) ||
|
|
219
|
+
"";
|
|
220
|
+
if (!latex) return;
|
|
221
|
+
|
|
222
|
+
// Determine if block or inline based on DOM structure
|
|
223
|
+
let isBlock = false;
|
|
224
|
+
if (blockContainer) {
|
|
225
|
+
// Full block container (.cPGBZb)
|
|
226
|
+
isBlock = true;
|
|
227
|
+
} else if (inlineWrapper) {
|
|
228
|
+
// Inline math wrapper (.mTEjhd, .dteT0b)
|
|
229
|
+
isBlock = false;
|
|
230
|
+
} else {
|
|
231
|
+
const style = copyRoot.getAttribute("style") || "";
|
|
232
|
+
if (/display:\s*inline/i.test(style)) {
|
|
233
|
+
isBlock = false;
|
|
234
|
+
} else {
|
|
235
|
+
// Check enclosing block element (p, li, td, th, div) for surrounding text
|
|
236
|
+
const enclosingBlock = container.closest("p, li, td, th, div");
|
|
237
|
+
if (enclosingBlock) {
|
|
238
|
+
const cloneBlock = enclosingBlock.cloneNode(true);
|
|
239
|
+
const targetInClone =
|
|
240
|
+
cloneBlock
|
|
241
|
+
.querySelector("[data-xpm-latex]")
|
|
242
|
+
?.closest("[data-xpm-copy-root]") ||
|
|
243
|
+
cloneBlock.querySelector("[data-xpm-latex]");
|
|
244
|
+
if (targetInClone) targetInClone.remove();
|
|
245
|
+
isBlock = cloneBlock.textContent.trim() === "";
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
}
|
|
214
249
|
|
|
215
|
-
|
|
216
|
-
|
|
250
|
+
registerMath(container, latex, isBlock, clone.ownerDocument);
|
|
251
|
+
});
|
|
217
252
|
|
|
218
253
|
// 4. Process block display KaTeX (.katex-display)
|
|
219
254
|
clone.querySelectorAll(".katex-display").forEach((el) => {
|