decant-core 1.10.0 → 1.10.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -41,7 +41,7 @@ AI platforms don't expose stable public APIs for reading conversation history. N
41
41
 
42
42
  Maintaining that per-platform logic in every exporter is wasteful and fragile. `decant-core` centralizes it:
43
43
 
44
- - ✅ **10+ AI chat platform parsers** with normalized output — you get structured messages, models, metadata and Markdown, not DOM soup.
44
+ - ✅ **18 AI chat platform parsers** with normalized output — you get structured messages, models, metadata and Markdown, not DOM soup.
45
45
  - ✅ **Web article extraction** — Mozilla Readability, Defuddle, and Article-Extractor run in parallel and arbitrate by content-quality scoring.
46
46
  - ✅ **Detection utilities** — tell an "AI chat page" apart from a "regular web page" before you decide which parser to run.
47
47
  - ✅ **Math & Markdown handling** — LaTeX normalization plus GFM tables/code fencing that survive round-trips into Obsidian, Logseq and Notion.
@@ -117,13 +117,34 @@ import { normalizeLatexMath } from "decant-core";
117
117
 
118
118
  ## Supported Platforms
119
119
 
120
- 10+ AI chat platform parsers plus generic web article extraction:
121
-
122
- **ChatGPT · Claude · Google Gemini · Microsoft Copilot · Perplexity · DeepSeek · Qwen · Meta AI · Mistral (Le Chat) · Proton Lumo · Z.ai · Google AI Studio · NotebookLM · Google Search AI · Gemini Cloud Assist · Joyland · Chub**
120
+ 18 AI chat platform parsers plus generic web article extraction:
121
+
122
+ | Platform | Parser | Extraction strategy |
123
+ | :---------------------------------- | :------------------------ | :----------------------------------------- |
124
+ | **ChatGPT** | `ChatGPTParser` | DOM + internal API |
125
+ | **Claude** | `ClaudeParser` | DOM + internal API + React fiber |
126
+ | **Google Gemini** | `GeminiParser` | DOM + batchexecute RPC |
127
+ | **Microsoft Copilot** | `CopilotParser` | DOM (multi-domain) |
128
+ | **Perplexity** | `PerplexityParser` | Internal API + DOM fallback |
129
+ | **DeepSeek** | `DeepSeekParser` | DOM + internal API (`fragments[]`) |
130
+ | **Qwen** | `QwenParser` | DOM |
131
+ | **Meta AI** | `MetaParser` | Internal API (GraphQL) + DOM fallback |
132
+ | **Mistral / Le Chat** | `MistralParser` | DOM |
133
+ | **Proton Lumo** | `LumoParser` | DOM (API is E2E-encrypted, not readable) |
134
+ | **Z.ai** | `ZAiParser` | Internal API (chat + batch) + DOM fallback |
135
+ | **Grok** | `GrokParser` | Internal API (response-node + load) + DOM |
136
+ | **Google AI Studio** | `GoogleAIStudioParser` | DOM |
137
+ | **NotebookLM** | `NotebookLMParser` | DOM |
138
+ | **Google Search AI (AI Overviews)** | `GoogleSearchAIParser` | DOM |
139
+ | **Gemini Cloud Assist** | `GeminiCloudAssistParser` | DOM |
140
+ | **Joyland** | `JoylandParser` | DOM |
141
+ | **Chub** | `ChubParser` | DOM |
142
+ | **Generic Web Article** | `ArticleParser` | Readability + Defuddle + Article-Extractor |
123
143
 
124
144
  All parsers extend the base [`ChatParser`](ai/base.js) interface — a consistent `isAvailable(url)` +
125
- normalized `parse()` contract. For the extraction-strategy breakdown and maintenance model, see
126
- [SUPPORTED_PLATFORMS.md](SUPPORTED_PLATFORMS.md).
145
+ normalized `parse()` contract. For the full extraction-strategy breakdown and maintenance model, see
146
+ [SUPPORTED_PLATFORMS.md](SUPPORTED_PLATFORMS.md), also published as the
147
+ [platform matrix](https://covai-labs.github.io/decant-core/platforms/) on the developer docs site.
127
148
 
128
149
  ---
129
150
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "decant-core",
3
- "version": "1.10.0",
3
+ "version": "1.10.1",
4
4
  "description": "A shared web extraction layer for AI conversations and regular web pages.",
5
5
  "type": "module",
6
6
  "engines": {
@@ -192,28 +192,63 @@ export function convertToMarkdown(htmlContent, options = {}) {
192
192
  registerMath(el, latex, isBlock, clone.ownerDocument);
193
193
  });
194
194
 
195
- // 3. Process Google Search SGE LaTeX images with [data-xpm-latex]
196
- clone.querySelectorAll("[data-xpm-latex]").forEach((el) => {
197
- if (!clone.contains(el)) return;
198
- const copyRoot = el.closest("[data-xpm-copy-root]");
199
- if (!copyRoot) return;
200
- const container = el.closest(".cPGBZb") || copyRoot;
201
- if (!container.parentNode) return;
202
-
203
- const latex = el.getAttribute("data-xpm-latex");
204
-
205
- // Determine if it is block math
206
- let isBlock = false;
207
- const parent = container.parentNode;
208
- if (parent) {
209
- const parentText = parent.textContent
210
- .replace(container.textContent, "")
211
- .trim();
212
- isBlock = parentText === "";
213
- }
195
+ // 3. Process Google Search SGE LaTeX images with [data-xpm-latex] or fallback math roots
196
+ clone
197
+ .querySelectorAll(
198
+ "[data-xpm-latex], [data-xpm-copy-root][data-xpm-copy-text], [data-xpm-copy-root] img[alt]",
199
+ )
200
+ .forEach((el) => {
201
+ if (!clone.contains(el)) return;
202
+ const copyRoot = el.hasAttribute("data-xpm-copy-root")
203
+ ? el
204
+ : el.closest("[data-xpm-copy-root]");
205
+ if (!copyRoot) return;
206
+ const blockContainer = el.closest(".cPGBZb");
207
+ const inlineWrapper = el.closest(".mTEjhd") || el.closest(".dteT0b");
208
+ const container = blockContainer || inlineWrapper || copyRoot;
209
+ if (!container.parentNode) return;
210
+
211
+ const img = el.tagName === "IMG" ? el : copyRoot.querySelector("img");
212
+ const latex =
213
+ copyRoot.getAttribute("data-xpm-latex") ||
214
+ el.getAttribute("data-xpm-latex") ||
215
+ copyRoot.getAttribute("data-xpm-copy-text") ||
216
+ (img && img.getAttribute("data-xpm-latex")) ||
217
+ el.getAttribute("alt") ||
218
+ (img && img.getAttribute("alt")) ||
219
+ "";
220
+ if (!latex) return;
221
+
222
+ // Determine if block or inline based on DOM structure
223
+ let isBlock = false;
224
+ if (blockContainer) {
225
+ // Full block container (.cPGBZb)
226
+ isBlock = true;
227
+ } else if (inlineWrapper) {
228
+ // Inline math wrapper (.mTEjhd, .dteT0b)
229
+ isBlock = false;
230
+ } else {
231
+ const style = copyRoot.getAttribute("style") || "";
232
+ if (/display:\s*inline/i.test(style)) {
233
+ isBlock = false;
234
+ } else {
235
+ // Check enclosing block element (p, li, td, th, div) for surrounding text
236
+ const enclosingBlock = container.closest("p, li, td, th, div");
237
+ if (enclosingBlock) {
238
+ const cloneBlock = enclosingBlock.cloneNode(true);
239
+ const targetInClone =
240
+ cloneBlock
241
+ .querySelector("[data-xpm-latex]")
242
+ ?.closest("[data-xpm-copy-root]") ||
243
+ cloneBlock.querySelector("[data-xpm-latex]");
244
+ if (targetInClone) targetInClone.remove();
245
+ isBlock = cloneBlock.textContent.trim() === "";
246
+ }
247
+ }
248
+ }
214
249
 
215
- registerMath(container, latex, isBlock, clone.ownerDocument);
216
- });
250
+ registerMath(container, latex, isBlock, clone.ownerDocument);
251
+ });
217
252
 
218
253
  // 4. Process block display KaTeX (.katex-display)
219
254
  clone.querySelectorAll(".katex-display").forEach((el) => {