decant-core 1.5.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +109 -81
- package/ai/gemini.js +194 -23
- package/package.json +10 -3
package/README.md
CHANGED
|
@@ -1,116 +1,144 @@
|
|
|
1
1
|
# decant-core
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
**A shared extraction layer for the modern web — including AI conversations.**
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
[](https://www.npmjs.com/package/decant-core)
|
|
6
|
+
[](LICENSE)
|
|
7
|
+
[](https://github.com/Covai-Labs/decant-core/stargazers)
|
|
6
8
|
|
|
7
|
-
|
|
9
|
+
Every AI chat exporter ends up solving the same problem: extracting conversations from ChatGPT, Claude, Gemini, Perplexity, DeepSeek and other constantly changing AI interfaces.
|
|
10
|
+
|
|
11
|
+
And every time one of those platforms changes its UI, seriously re-renders a message, or ships a new feature, **somebody's parser breaks**.
|
|
8
12
|
|
|
9
|
-
|
|
10
|
-
| ------------------- | ------------------------- |
|
|
11
|
-
| ChatGPT | `ChatGPTParser` |
|
|
12
|
-
| Claude | `ClaudeParser` |
|
|
13
|
-
| Copilot | `CopilotParser` |
|
|
14
|
-
| DeepSeek | `DeepSeekParser` |
|
|
15
|
-
| Gemini | `GeminiParser` |
|
|
16
|
-
| Gemini Cloud Assist | `GeminiCloudAssistParser` |
|
|
17
|
-
| Google AI Studio | `GoogleAIStudioParser` |
|
|
18
|
-
| Google Search AI | `GoogleSearchAIParser` |
|
|
19
|
-
| Lumo | `LumoParser` |
|
|
20
|
-
| Meta AI | `MetaParser` |
|
|
21
|
-
| Mistral | `MistralParser` |
|
|
22
|
-
| NotebookLM | `NotebookLMParser` |
|
|
23
|
-
| Perplexity | `PerplexityParser` |
|
|
24
|
-
| Qwen | `QwenParser` |
|
|
25
|
-
| Z AI | `ZAiParser` |
|
|
26
|
-
|
|
27
|
-
## Installation
|
|
13
|
+
`decant-core` provides reusable parsers, platform detection, and web article extraction so developers don't have to build and maintain the same fragile parsing layer over and over again.
|
|
28
14
|
|
|
29
15
|
```bash
|
|
30
16
|
npm install decant-core
|
|
31
17
|
```
|
|
32
18
|
|
|
33
|
-
|
|
19
|
+
Use it as the parsing layer underneath your own:
|
|
34
20
|
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
21
|
+
- chat exporters
|
|
22
|
+
- browser extensions
|
|
23
|
+
- web clippers
|
|
24
|
+
- research & data-extraction tools
|
|
25
|
+
- content archivers
|
|
26
|
+
- knowledge-management and PKM applications
|
|
27
|
+
|
|
28
|
+
**Fix platform parsing once, and let the ecosystem benefit from the fix.**
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Why decant-core?
|
|
33
|
+
|
|
34
|
+
AI platforms don't expose stable public APIs for reading conversation history. No matter what you build, to extract a ChatGPT thread you need to walk the DOM, read internal RPC payloads, or traverse React component trees — and redo it when the frontend changes.
|
|
35
|
+
|
|
36
|
+
Maintaining that per-platform logic in every exporter is wasteful and fragile. `decant-core` centralizes it:
|
|
37
|
+
|
|
38
|
+
- ✅ **17 AI chat platform parsers** with normalized output — you get structured messages, models, metadata and Markdown, not DOM soup.
|
|
39
|
+
- ✅ **Web article extraction** — Mozilla Readability, Defuddle, and Article-Extractor run in parallel and arbitrate by content-quality scoring.
|
|
40
|
+
- ✅ **Detection utilities** — tell an "AI chat page" apart from a "regular web page" before you decide which parser to run.
|
|
41
|
+
- ✅ **Math & Markdown handling** — LaTeX normalization plus GFM tables/code fencing that survive round-trips into Obsidian, Logseq and Notion.
|
|
42
|
+
|
|
43
|
+
The payoff is maintenance: **when a platform changes, the fix happens once, in one place**, instead of being independently reimplemented across dozens of projects.
|
|
38
44
|
|
|
39
|
-
|
|
45
|
+
---
|
|
46
|
+
|
|
47
|
+
## Quickstart
|
|
48
|
+
|
|
49
|
+
### 1. Extract an AI conversation
|
|
40
50
|
|
|
41
51
|
```js
|
|
42
|
-
import { detectPlatform,
|
|
52
|
+
import { detectPlatform, isAiChatUrl } from "decant-core";
|
|
43
53
|
|
|
44
|
-
// Check if a URL is an AI chat platform
|
|
45
54
|
if (isAiChatUrl(window.location.href)) {
|
|
46
|
-
const
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
}
|
|
55
|
+
const match = detectPlatform(window.location.href);
|
|
56
|
+
|
|
57
|
+
if (match?.parser && match.parser.isAvailable(window.location.href)) {
|
|
58
|
+
const result = await match.parser.parse();
|
|
59
|
+
// result.title
|
|
60
|
+
// result.messages -> [{ role: 'User' | 'Assistant', content, ... }]
|
|
61
|
+
// result.model
|
|
62
|
+
// result.metadata -> platform-specific extras
|
|
55
63
|
}
|
|
56
64
|
}
|
|
57
65
|
```
|
|
58
66
|
|
|
59
|
-
###
|
|
67
|
+
### 2. Extract a regular web article
|
|
60
68
|
|
|
61
69
|
```js
|
|
62
|
-
|
|
63
|
-
|
|
70
|
+
import { extractArticle } from "decant-core";
|
|
71
|
+
|
|
72
|
+
const article = await extractArticle(document /* or an HTML string */, {
|
|
73
|
+
url: window.location.href,
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
// article.title, article.author, article.published
|
|
77
|
+
// article.markdown -> clean, ready-to-use Markdown
|
|
78
|
+
// article.content -> the body without the title prefix
|
|
79
|
+
// article.engine -> 'readability' | 'defuddle' | 'raw'
|
|
80
|
+
```
|
|
64
81
|
|
|
65
|
-
|
|
82
|
+
### 3. Detection
|
|
83
|
+
|
|
84
|
+
```js
|
|
66
85
|
import { detectPlatform, isAiChatUrl, AI_CHAT_DOMAINS } from "decant-core";
|
|
67
86
|
|
|
87
|
+
isAiChatUrl("https://chatgpt.com/c/abc-123"); // -> true
|
|
88
|
+
const detected = detectPlatform(url); // -> { type: 'ai-chat', platform: 'ChatGPT', parser }
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
### Subpath imports
|
|
92
|
+
|
|
93
|
+
```js
|
|
94
|
+
// Individual parsers (tree-shake the rest)
|
|
95
|
+
import { ChatGPTParser } from "decant-core/ai/chatgpt";
|
|
96
|
+
import { ClaudeParser } from "decant-core/ai/claude";
|
|
97
|
+
import { GeminiParser } from "decant-core/ai/gemini";
|
|
98
|
+
|
|
99
|
+
// Web & article extraction
|
|
100
|
+
import { extractArticle, ArticleParser, scoreContent } from "decant-core";
|
|
101
|
+
|
|
102
|
+
// Detection
|
|
103
|
+
import { detectPlatform, isAiChatUrl, parsers } from "decant-core";
|
|
104
|
+
|
|
68
105
|
// Utilities
|
|
69
106
|
import { convertToMarkdown, cleanMarkdown } from "decant-core";
|
|
107
|
+
import { normalizeLatexMath } from "decant-core";
|
|
70
108
|
```
|
|
71
109
|
|
|
72
|
-
|
|
110
|
+
---
|
|
73
111
|
|
|
74
|
-
|
|
75
|
-
parser-core/
|
|
76
|
-
ai/ # Parser classes
|
|
77
|
-
base.js # Base parser class
|
|
78
|
-
chatgpt.js # ChatGPT parser + linearize
|
|
79
|
-
chatgpt_helper.js # Injected helper for ChatGPT scroll collection
|
|
80
|
-
chatgpt_scroll_collector.js # Scroll/dedup logic for ChatGPT
|
|
81
|
-
claude.js # Claude parser (DOM + API)
|
|
82
|
-
claude_react_reader.js # Injected helper for Claude React tree
|
|
83
|
-
copilot.js # Copilot parser (multi-domain)
|
|
84
|
-
deepseek.js # DeepSeek parser (DOM + API)
|
|
85
|
-
gemini.js # Gemini parser
|
|
86
|
-
gemini_cloud_assist.js # Gemini Cloud Assist parser
|
|
87
|
-
google_ai_studio.js # Google AI Studio parser
|
|
88
|
-
google_search_ai.js # Google Search AI / SGE parser
|
|
89
|
-
index.js # Barrel export
|
|
90
|
-
lumo.js # Lumo parser
|
|
91
|
-
meta.js # Meta AI parser
|
|
92
|
-
mistral.js # Mistral parser
|
|
93
|
-
notebooklm.js # NotebookLM parser
|
|
94
|
-
perplexity.js # Perplexity parser
|
|
95
|
-
qwen.js # Qwen parser
|
|
96
|
-
z_ai.js # Z AI parser
|
|
97
|
-
detection/ # Platform detection
|
|
98
|
-
detect-platform.js # detectPlatform(), isAiChatUrl(), parsers[]
|
|
99
|
-
domains.js # AI_CHAT_DOMAINS, URL_PATTERNS
|
|
100
|
-
lib/ # Vendored libraries
|
|
101
|
-
turndown.js # Turndown HTML→Markdown converter
|
|
102
|
-
utils/ # Utilities
|
|
103
|
-
html-to-markdown.js # AI-specific Turndown rules
|
|
104
|
-
```
|
|
112
|
+
## Supported Platforms
|
|
105
113
|
|
|
106
|
-
|
|
114
|
+
17 AI chat platform parsers plus generic web article extraction:
|
|
107
115
|
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
116
|
+
**ChatGPT · Claude · Google Gemini · Microsoft Copilot · Perplexity · DeepSeek · Qwen · Meta AI · Mistral (Le Chat) · Proton Lumo · Z.ai · Google AI Studio · NotebookLM · Google Search AI · Gemini Cloud Assist · Joyland · Chub**
|
|
117
|
+
|
|
118
|
+
All parsers extend the base [`ChatParser`](ai/base.js) interface — a consistent `isAvailable(url)` +
|
|
119
|
+
normalized `parse()` contract. For the extraction-strategy breakdown and maintenance model, see
|
|
120
|
+
[SUPPORTED_PLATFORMS.md](SUPPORTED_PLATFORMS.md).
|
|
121
|
+
|
|
122
|
+
---
|
|
113
123
|
|
|
114
124
|
## License
|
|
115
125
|
|
|
116
|
-
|
|
126
|
+
`decant-core` is licensed under the **GNU Affero General Public License v3.0 (AGPL-3.0-only)**.
|
|
127
|
+
|
|
128
|
+
That choice is deliberate. AI platforms change constantly, and parser fixes belong in a shared commons so the whole ecosystem benefits — not siloed in a proprietary fork. If you use `decant-core`, network-based deployments that serve modified versions must also offer the corresponding source. Please review [`LICENSE`](LICENSE) before incorporating it into your project.
|
|
129
|
+
|
|
130
|
+
---
|
|
131
|
+
|
|
132
|
+
## Used by
|
|
133
|
+
|
|
134
|
+
- [AI Chat Exporter](https://github.com/Covai-Labs/ai-chat-exporter) — export, archive and transfer AI conversations between platforms.
|
|
135
|
+
- [Decant](https://github.com/Covai-Labs/decant) — the distraction-free web clipper and research batcher.
|
|
136
|
+
|
|
137
|
+
These products are demonstrations of the library, not its purpose. Yours can be next — see [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
138
|
+
|
|
139
|
+
---
|
|
140
|
+
|
|
141
|
+
## Development
|
|
142
|
+
|
|
143
|
+
Building, testing, and extending the library is covered in [DEVELOPMENT.md](DEVELOPMENT.md);
|
|
144
|
+
platform contributions follow the parser pattern and CLA in [CONTRIBUTING.md](CONTRIBUTING.md).
|
package/ai/gemini.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { ChatParser } from "./base.js";
|
|
2
2
|
import { convertToMarkdown } from "../utils/html-to-markdown.js";
|
|
3
|
+
import { normalizeLatexMath } from "../utils/latex-math.js";
|
|
3
4
|
|
|
4
5
|
const GEMINI_RPC_ID = "hNvQHb";
|
|
5
6
|
const DEFAULT_BARD_PATH = "/_/BardChatUi";
|
|
@@ -94,23 +95,7 @@ export class GeminiParser extends ChatParser {
|
|
|
94
95
|
|
|
95
96
|
const mode = options.parserMode || "auto";
|
|
96
97
|
|
|
97
|
-
// 1.
|
|
98
|
-
if (typeof document !== "undefined" && document.querySelector) {
|
|
99
|
-
const hasDomMessages = document.querySelector(
|
|
100
|
-
".conversation-container, user-query, model-response, deep-research-immersive-panel",
|
|
101
|
-
);
|
|
102
|
-
if (hasDomMessages && mode !== "api") {
|
|
103
|
-
const domResult = this.parseFromDom(currentUrl, options);
|
|
104
|
-
if (domResult && domResult.messages && domResult.messages.length > 0) {
|
|
105
|
-
console.log(
|
|
106
|
-
`[Gemini Parser] Successfully parsed ${domResult.messages.length} messages from DOM`,
|
|
107
|
-
);
|
|
108
|
-
return domResult;
|
|
109
|
-
}
|
|
110
|
-
}
|
|
111
|
-
}
|
|
112
|
-
|
|
113
|
-
// 2. Attempt API / RPC extraction if DOM parsing didn't find messages or mode is API
|
|
98
|
+
// 1. Attempt API / RPC extraction first when not explicitly in 'dom' mode
|
|
114
99
|
if (mode !== "dom" && typeof fetch === "function") {
|
|
115
100
|
try {
|
|
116
101
|
const convoId = this.getConversationId(currentUrl);
|
|
@@ -149,7 +134,7 @@ export class GeminiParser extends ChatParser {
|
|
|
149
134
|
}
|
|
150
135
|
}
|
|
151
136
|
|
|
152
|
-
// Fall back to robust DOM parsing
|
|
137
|
+
// 2. Fall back to robust DOM parsing
|
|
153
138
|
return this.parseFromDom(currentUrl, options);
|
|
154
139
|
}
|
|
155
140
|
|
|
@@ -175,14 +160,17 @@ export class GeminiParser extends ChatParser {
|
|
|
175
160
|
`&_reqid=${encodeURIComponent(reqId)}` +
|
|
176
161
|
`&rt=c`;
|
|
177
162
|
|
|
163
|
+
const formattedConvoId = convoId.startsWith("c_")
|
|
164
|
+
? convoId
|
|
165
|
+
: `c_${convoId}`;
|
|
178
166
|
const allItems = [];
|
|
179
167
|
let cursor = null;
|
|
180
168
|
let pageCount = 0;
|
|
181
169
|
|
|
182
|
-
while (pageCount <
|
|
170
|
+
while (pageCount < 250) {
|
|
183
171
|
pageCount++;
|
|
184
172
|
const payloadArg = JSON.stringify([
|
|
185
|
-
|
|
173
|
+
formattedConvoId,
|
|
186
174
|
100,
|
|
187
175
|
cursor,
|
|
188
176
|
1,
|
|
@@ -227,11 +215,11 @@ export class GeminiParser extends ChatParser {
|
|
|
227
215
|
const continueCursor = payload[1] || null;
|
|
228
216
|
|
|
229
217
|
if (items.length > 0) {
|
|
230
|
-
// Items are in reverse chronological order from API
|
|
218
|
+
// Items are in reverse chronological order from API; reverse to maintain oldest-first order
|
|
231
219
|
allItems.unshift(...items.slice().reverse());
|
|
232
220
|
}
|
|
233
221
|
|
|
234
|
-
if (!continueCursor
|
|
222
|
+
if (!continueCursor) {
|
|
235
223
|
break;
|
|
236
224
|
}
|
|
237
225
|
cursor = continueCursor;
|
|
@@ -241,7 +229,158 @@ export class GeminiParser extends ChatParser {
|
|
|
241
229
|
return null;
|
|
242
230
|
}
|
|
243
231
|
|
|
232
|
+
return this.formatApiResult(allItems, currentUrl, options);
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
enrichMessagesWithDomAttachments(messages) {
|
|
236
|
+
try {
|
|
237
|
+
if (typeof document === "undefined" || !document.querySelectorAll) {
|
|
238
|
+
return messages;
|
|
239
|
+
}
|
|
240
|
+
const parentContainers = document.querySelectorAll(
|
|
241
|
+
".conversation-container",
|
|
242
|
+
);
|
|
243
|
+
const domTurns = [];
|
|
244
|
+
|
|
245
|
+
if (parentContainers.length > 0) {
|
|
246
|
+
parentContainers.forEach((container, idx) => {
|
|
247
|
+
const rawId = (container.id || "").replace(/^r_/, "");
|
|
248
|
+
const fileElements = container.querySelectorAll(
|
|
249
|
+
"user-query-file-preview, .file-preview, .attachment-preview, [data-testid='file-preview']",
|
|
250
|
+
);
|
|
251
|
+
const fileNames = [];
|
|
252
|
+
fileElements.forEach((fe) => {
|
|
253
|
+
const name = (fe.textContent || "").trim();
|
|
254
|
+
if (name && !fileNames.includes(name)) fileNames.push(name);
|
|
255
|
+
});
|
|
256
|
+
|
|
257
|
+
const userQuery = container.querySelector("user-query") || container;
|
|
258
|
+
const clone = userQuery.cloneNode(true);
|
|
259
|
+
clone
|
|
260
|
+
.querySelectorAll(
|
|
261
|
+
"user-query-file-preview, .file-preview, .attachment-preview, button, .edit-button, model-response",
|
|
262
|
+
)
|
|
263
|
+
.forEach((el) => el.remove());
|
|
264
|
+
const queryText = (
|
|
265
|
+
clone.innerText !== undefined
|
|
266
|
+
? clone.innerText
|
|
267
|
+
: clone.textContent || ""
|
|
268
|
+
)
|
|
269
|
+
.replace(/^You said\s*/i, "")
|
|
270
|
+
.trim();
|
|
271
|
+
|
|
272
|
+
if (fileNames.length > 0) {
|
|
273
|
+
domTurns.push({
|
|
274
|
+
turnId: rawId,
|
|
275
|
+
queryText,
|
|
276
|
+
domIndex: idx,
|
|
277
|
+
totalDom: parentContainers.length,
|
|
278
|
+
fileNames,
|
|
279
|
+
});
|
|
280
|
+
}
|
|
281
|
+
});
|
|
282
|
+
} else {
|
|
283
|
+
const userQueries = document.querySelectorAll("user-query");
|
|
284
|
+
userQueries.forEach((uq, idx) => {
|
|
285
|
+
const fileElements = uq.querySelectorAll(
|
|
286
|
+
"user-query-file-preview, .file-preview, .attachment-preview, [data-testid='file-preview']",
|
|
287
|
+
);
|
|
288
|
+
const fileNames = [];
|
|
289
|
+
fileElements.forEach((fe) => {
|
|
290
|
+
const name = (fe.textContent || "").trim();
|
|
291
|
+
if (name && !fileNames.includes(name)) fileNames.push(name);
|
|
292
|
+
});
|
|
293
|
+
|
|
294
|
+
const clone = uq.cloneNode(true);
|
|
295
|
+
clone
|
|
296
|
+
.querySelectorAll(
|
|
297
|
+
"user-query-file-preview, .file-preview, .attachment-preview, button, .edit-button",
|
|
298
|
+
)
|
|
299
|
+
.forEach((el) => el.remove());
|
|
300
|
+
const queryText = (
|
|
301
|
+
clone.innerText !== undefined
|
|
302
|
+
? clone.innerText
|
|
303
|
+
: clone.textContent || ""
|
|
304
|
+
)
|
|
305
|
+
.replace(/^You said\s*/i, "")
|
|
306
|
+
.trim();
|
|
307
|
+
|
|
308
|
+
if (fileNames.length > 0) {
|
|
309
|
+
domTurns.push({
|
|
310
|
+
domIndex: idx,
|
|
311
|
+
queryText,
|
|
312
|
+
totalDom: userQueries.length,
|
|
313
|
+
fileNames,
|
|
314
|
+
});
|
|
315
|
+
}
|
|
316
|
+
});
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
if (domTurns.length === 0) return messages;
|
|
320
|
+
|
|
321
|
+
const userMessages = messages.filter((m) => m.role === "User");
|
|
322
|
+
|
|
323
|
+
for (const domTurn of domTurns) {
|
|
324
|
+
let matchedMsg = null;
|
|
325
|
+
|
|
326
|
+
// 1. Match by exact turnId if present
|
|
327
|
+
if (domTurn.turnId) {
|
|
328
|
+
matchedMsg = userMessages.find((m) => m.turnId === domTurn.turnId);
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
// 2. Fallback: match by queryText if present and non-empty
|
|
332
|
+
if (!matchedMsg && domTurn.queryText) {
|
|
333
|
+
matchedMsg = userMessages.find(
|
|
334
|
+
(m) =>
|
|
335
|
+
m.content &&
|
|
336
|
+
(m.content.includes(domTurn.queryText) ||
|
|
337
|
+
domTurn.queryText.includes(m.content)),
|
|
338
|
+
);
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
// 3. Fallback: match by trailing turn ordinal (mounted DOM elements align with latest turns)
|
|
342
|
+
if (!matchedMsg && typeof domTurn.domIndex === "number") {
|
|
343
|
+
const offset = userMessages.length - domTurn.totalDom;
|
|
344
|
+
const targetIndex =
|
|
345
|
+
offset >= 0 ? offset + domTurn.domIndex : domTurn.domIndex;
|
|
346
|
+
matchedMsg = userMessages[targetIndex];
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
if (matchedMsg) {
|
|
350
|
+
const attachBlock =
|
|
351
|
+
`\n\n**Attachments:**\n` +
|
|
352
|
+
domTurn.fileNames.map((fn) => `- ${fn}`).join("\n");
|
|
353
|
+
if (!matchedMsg.content.includes("**Attachments:**")) {
|
|
354
|
+
matchedMsg.content = matchedMsg.content
|
|
355
|
+
? `${matchedMsg.content}${attachBlock}`
|
|
356
|
+
: `**Attachments:**\n` +
|
|
357
|
+
domTurn.fileNames.map((fn) => `- ${fn}`).join("\n");
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
} catch (e) {
|
|
362
|
+
console.warn(
|
|
363
|
+
"[Gemini Parser] Failed to enrich messages with DOM attachments:",
|
|
364
|
+
e,
|
|
365
|
+
);
|
|
366
|
+
} finally {
|
|
367
|
+
// Clean internal turnId tracking before returning messages
|
|
368
|
+
for (const m of messages) {
|
|
369
|
+
delete m.turnId;
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
return messages;
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
formatApiResult(allItems, currentUrl, options = {}) {
|
|
377
|
+
if (!Array.isArray(allItems) || allItems.length === 0) {
|
|
378
|
+
return null;
|
|
379
|
+
}
|
|
380
|
+
|
|
244
381
|
const messages = this.convertApiItemsToMessages(allItems, options);
|
|
382
|
+
this.enrichMessagesWithDomAttachments(messages);
|
|
383
|
+
|
|
245
384
|
let title = this.extractTitleFromPage();
|
|
246
385
|
if (!title || title === "Gemini Conversation") {
|
|
247
386
|
const firstUserMsg = messages.find((m) => m.role === "User");
|
|
@@ -296,17 +435,28 @@ export class GeminiParser extends ChatParser {
|
|
|
296
435
|
return null;
|
|
297
436
|
}
|
|
298
437
|
|
|
438
|
+
extractTurnId(item) {
|
|
439
|
+
try {
|
|
440
|
+
const raw = item[0]?.[1] || item[1]?.[1] || "";
|
|
441
|
+
return String(raw).replace(/^r_/, "");
|
|
442
|
+
} catch {
|
|
443
|
+
return "";
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
|
|
299
447
|
convertApiItemsToMessages(items, options = {}) {
|
|
300
448
|
const messages = [];
|
|
301
449
|
|
|
302
450
|
for (const item of items) {
|
|
303
451
|
if (!Array.isArray(item)) continue;
|
|
304
452
|
|
|
453
|
+
const turnId = this.extractTurnId(item);
|
|
305
454
|
const userText = this.findUserTextInApiItem(item);
|
|
306
455
|
if (userText) {
|
|
307
456
|
messages.push({
|
|
308
457
|
role: "User",
|
|
309
458
|
content: userText.trim(),
|
|
459
|
+
turnId,
|
|
310
460
|
});
|
|
311
461
|
}
|
|
312
462
|
|
|
@@ -314,7 +464,8 @@ export class GeminiParser extends ChatParser {
|
|
|
314
464
|
if (modelText) {
|
|
315
465
|
messages.push({
|
|
316
466
|
role: "Model",
|
|
317
|
-
content: modelText.trim(),
|
|
467
|
+
content: normalizeLatexMath(modelText.trim()),
|
|
468
|
+
turnId,
|
|
318
469
|
});
|
|
319
470
|
}
|
|
320
471
|
}
|
|
@@ -324,6 +475,7 @@ export class GeminiParser extends ChatParser {
|
|
|
324
475
|
|
|
325
476
|
findUserTextInApiItem(item) {
|
|
326
477
|
try {
|
|
478
|
+
if (typeof item[2]?.[0]?.[0] === "string") return item[2][0][0];
|
|
327
479
|
if (typeof item[2]?.[0] === "string") return item[2][0];
|
|
328
480
|
if (typeof item[1]?.[0] === "string" && !Array.isArray(item[1][0]))
|
|
329
481
|
return item[1][0];
|
|
@@ -336,6 +488,25 @@ export class GeminiParser extends ChatParser {
|
|
|
336
488
|
|
|
337
489
|
findModelTextInApiItem(item) {
|
|
338
490
|
try {
|
|
491
|
+
// 1. Candidate responses in item[3]
|
|
492
|
+
if (Array.isArray(item[3])) {
|
|
493
|
+
const candidates = Array.isArray(item[3][0]) ? item[3][0] : item[3];
|
|
494
|
+
for (const cand of candidates) {
|
|
495
|
+
if (!Array.isArray(cand)) continue;
|
|
496
|
+
// Shape: ["rc_...", ["markdown text", ...], ...]
|
|
497
|
+
if (Array.isArray(cand[1]) && typeof cand[1][0] === "string") {
|
|
498
|
+
return cand[1][0];
|
|
499
|
+
}
|
|
500
|
+
if (typeof cand[1] === "string") {
|
|
501
|
+
return cand[1];
|
|
502
|
+
}
|
|
503
|
+
if (typeof cand[0] === "string" && cand[0].length > 50) {
|
|
504
|
+
return cand[0];
|
|
505
|
+
}
|
|
506
|
+
}
|
|
507
|
+
}
|
|
508
|
+
|
|
509
|
+
// 2. Fallback candidate in item[1]
|
|
339
510
|
if (Array.isArray(item[1])) {
|
|
340
511
|
const candidate = item[1][0];
|
|
341
512
|
if (typeof candidate === "string") return candidate;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "decant-core",
|
|
3
|
-
"version": "1.
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "1.7.0",
|
|
4
|
+
"description": "A shared web extraction layer for AI conversations and regular web pages.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"engines": {
|
|
7
7
|
"node": ">=26.0.0"
|
|
@@ -47,7 +47,14 @@
|
|
|
47
47
|
"gemini",
|
|
48
48
|
"deepseek",
|
|
49
49
|
"copilot",
|
|
50
|
-
"
|
|
50
|
+
"perplexity",
|
|
51
|
+
"web-clipper",
|
|
52
|
+
"extraction",
|
|
53
|
+
"markdown",
|
|
54
|
+
"readability",
|
|
55
|
+
"turndown",
|
|
56
|
+
"browser-extension",
|
|
57
|
+
"scraping"
|
|
51
58
|
],
|
|
52
59
|
"author": "deadrat-in",
|
|
53
60
|
"license": "AGPL-3.0-only",
|