decant-core 1.9.2 → 1.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +37 -6
- package/ai/deepseek.js +50 -5
- package/ai/grok.js +216 -0
- package/ai/index.js +1 -0
- package/ai/meta.js +355 -12
- package/ai/mistral.js +21 -20
- package/ai/qwen.js +23 -4
- package/ai/z_ai.js +164 -8
- package/detection/detect-platform.js +2 -0
- package/detection/domains.js +1 -0
- package/package.json +1 -1
- package/utils/html-to-markdown.js +56 -21
package/README.md
CHANGED
|
@@ -12,6 +12,12 @@ And every time one of those platforms changes its UI, seriously re-renders a mes
|
|
|
12
12
|
|
|
13
13
|
`decant-core` provides reusable parsers, platform detection, and web article extraction so developers don't have to build and maintain the same fragile parsing layer over and over again.
|
|
14
14
|
|
|
15
|
+
### Origin & Purpose
|
|
16
|
+
|
|
17
|
+
Existing chat exporters often suffer from two major flaws: they break whenever platform DOMs update, and many route user conversations through third-party servers.
|
|
18
|
+
|
|
19
|
+
`decant-core` was created to solve both at the foundation. Originally built to power local-first extensions like [AI Chat Exporter](https://ai-chat-exporter.covai.org/) and [Decant](https://decant.covai.org/), it decouples fragile platform parsing from presentation. By sharing this engine under AGPL-3.0, any browser extension, web clipper, archiver, or research tool can rely on a maintained, local-first extraction layer instead of reverse-engineering AI platforms in isolation.
|
|
20
|
+
|
|
15
21
|
```bash
|
|
16
22
|
npm install decant-core
|
|
17
23
|
```
|
|
@@ -35,7 +41,7 @@ AI platforms don't expose stable public APIs for reading conversation history. N
|
|
|
35
41
|
|
|
36
42
|
Maintaining that per-platform logic in every exporter is wasteful and fragile. `decant-core` centralizes it:
|
|
37
43
|
|
|
38
|
-
- ✅ **
|
|
44
|
+
- ✅ **18 AI chat platform parsers** with normalized output — you get structured messages, models, metadata and Markdown, not DOM soup.
|
|
39
45
|
- ✅ **Web article extraction** — Mozilla Readability, Defuddle, and Article-Extractor run in parallel and arbitrate by content-quality scoring.
|
|
40
46
|
- ✅ **Detection utilities** — tell an "AI chat page" apart from a "regular web page" before you decide which parser to run.
|
|
41
47
|
- ✅ **Math & Markdown handling** — LaTeX normalization plus GFM tables/code fencing that survive round-trips into Obsidian, Logseq and Notion.
|
|
@@ -111,13 +117,34 @@ import { normalizeLatexMath } from "decant-core";
|
|
|
111
117
|
|
|
112
118
|
## Supported Platforms
|
|
113
119
|
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
120
|
+
18 AI chat platform parsers plus generic web article extraction:
|
|
121
|
+
|
|
122
|
+
| Platform | Parser | Extraction strategy |
|
|
123
|
+
| :---------------------------------- | :------------------------ | :----------------------------------------- |
|
|
124
|
+
| **ChatGPT** | `ChatGPTParser` | DOM + internal API |
|
|
125
|
+
| **Claude** | `ClaudeParser` | DOM + internal API + React fiber |
|
|
126
|
+
| **Google Gemini** | `GeminiParser` | DOM + batchexecute RPC |
|
|
127
|
+
| **Microsoft Copilot** | `CopilotParser` | DOM (multi-domain) |
|
|
128
|
+
| **Perplexity** | `PerplexityParser` | Internal API + DOM fallback |
|
|
129
|
+
| **DeepSeek** | `DeepSeekParser` | DOM + internal API (`fragments[]`) |
|
|
130
|
+
| **Qwen** | `QwenParser` | DOM |
|
|
131
|
+
| **Meta AI** | `MetaParser` | Internal API (GraphQL) + DOM fallback |
|
|
132
|
+
| **Mistral / Le Chat** | `MistralParser` | DOM |
|
|
133
|
+
| **Proton Lumo** | `LumoParser` | DOM (API is E2E-encrypted, not readable) |
|
|
134
|
+
| **Z.ai** | `ZAiParser` | Internal API (chat + batch) + DOM fallback |
|
|
135
|
+
| **Grok** | `GrokParser` | Internal API (response-node + load) + DOM |
|
|
136
|
+
| **Google AI Studio** | `GoogleAIStudioParser` | DOM |
|
|
137
|
+
| **NotebookLM** | `NotebookLMParser` | DOM |
|
|
138
|
+
| **Google Search AI (AI Overviews)** | `GoogleSearchAIParser` | DOM |
|
|
139
|
+
| **Gemini Cloud Assist** | `GeminiCloudAssistParser` | DOM |
|
|
140
|
+
| **Joyland** | `JoylandParser` | DOM |
|
|
141
|
+
| **Chub** | `ChubParser` | DOM |
|
|
142
|
+
| **Generic Web Article** | `ArticleParser` | Readability + Defuddle + Article-Extractor |
|
|
117
143
|
|
|
118
144
|
All parsers extend the base [`ChatParser`](ai/base.js) interface — a consistent `isAvailable(url)` +
|
|
119
|
-
normalized `parse()` contract. For the extraction-strategy breakdown and maintenance model, see
|
|
120
|
-
[SUPPORTED_PLATFORMS.md](SUPPORTED_PLATFORMS.md)
|
|
145
|
+
normalized `parse()` contract. For the full extraction-strategy breakdown and maintenance model, see
|
|
146
|
+
[SUPPORTED_PLATFORMS.md](SUPPORTED_PLATFORMS.md), also published as the
|
|
147
|
+
[platform matrix](https://covai-labs.github.io/decant-core/platforms/) on the developer docs site.
|
|
121
148
|
|
|
122
149
|
---
|
|
123
150
|
|
|
@@ -127,6 +154,10 @@ normalized `parse()` contract. For the extraction-strategy breakdown and mainten
|
|
|
127
154
|
|
|
128
155
|
That choice is deliberate. AI platforms change constantly, and parser fixes belong in a shared commons so the whole ecosystem benefits — not siloed in a proprietary fork. If you use `decant-core`, network-based deployments that serve modified versions must also offer the corresponding source. Please review [`LICENSE`](LICENSE) before incorporating it into your project.
|
|
129
156
|
|
|
157
|
+
### Third-Party Test Fixtures Notice
|
|
158
|
+
|
|
159
|
+
Sample test fixtures located in [`tests/fixtures/`](tests/fixtures/) consist of third-party DOM snapshots and API response excerpts retained solely for automated regression testing and platform interoperability under fair use principles. They are excluded from the project's AGPL-3.0 license. See [`tests/fixtures/README.md`](tests/fixtures/README.md) for details.
|
|
160
|
+
|
|
130
161
|
---
|
|
131
162
|
|
|
132
163
|
## Used by
|
package/ai/deepseek.js
CHANGED
|
@@ -17,6 +17,27 @@ function getUserToken() {
|
|
|
17
17
|
}
|
|
18
18
|
}
|
|
19
19
|
|
|
20
|
+
export function extractDeepSeekMessageContent(msgNode) {
|
|
21
|
+
if (!msgNode) return "";
|
|
22
|
+
// Current API shape: content lives in fragments[] ({type, content}).
|
|
23
|
+
// REQUEST = user prompt, RESPONSE = final assistant answer.
|
|
24
|
+
// THINK / TOOL_* fragments are intermediate reasoning/tool output.
|
|
25
|
+
if (Array.isArray(msgNode.fragments) && msgNode.fragments.length > 0) {
|
|
26
|
+
const isUser = msgNode.role === "USER" || msgNode.role === "user";
|
|
27
|
+
const wanted = isUser ? "REQUEST" : "RESPONSE";
|
|
28
|
+
const picked = msgNode.fragments.filter((f) => f && f.type === wanted);
|
|
29
|
+
const fallback = picked.length > 0 ? picked : msgNode.fragments;
|
|
30
|
+
const text = fallback
|
|
31
|
+
.map((f) => (typeof f.content === "string" ? f.content : ""))
|
|
32
|
+
.join("\n\n")
|
|
33
|
+
.trim();
|
|
34
|
+
if (text) return text;
|
|
35
|
+
}
|
|
36
|
+
// Legacy shape: top-level content/text fields.
|
|
37
|
+
const legacy = msgNode.content || msgNode.text || "";
|
|
38
|
+
return typeof legacy === "string" ? legacy.trim() : "";
|
|
39
|
+
}
|
|
40
|
+
|
|
20
41
|
function getConversationId() {
|
|
21
42
|
try {
|
|
22
43
|
if (typeof window === "undefined" || !window.location) return null;
|
|
@@ -76,7 +97,7 @@ async function fetchDeepSeekConversation(sessionId, token) {
|
|
|
76
97
|
.map((msgNode) => {
|
|
77
98
|
const isUser = msgNode.role === "USER" || msgNode.role === "user";
|
|
78
99
|
const role = isUser ? "User" : "DeepSeek";
|
|
79
|
-
const content = msgNode
|
|
100
|
+
const content = extractDeepSeekMessageContent(msgNode);
|
|
80
101
|
return { role, content: content.trim() };
|
|
81
102
|
})
|
|
82
103
|
.filter((msg) => msg.content.length > 0);
|
|
@@ -130,12 +151,36 @@ export class DeepSeekParser extends ChatParser {
|
|
|
130
151
|
const userSelector = ".fbb737a4";
|
|
131
152
|
const assistantSelector = ".ds-markdown";
|
|
132
153
|
|
|
133
|
-
// We'll traverse the DOM to find these in order
|
|
134
|
-
|
|
135
|
-
|
|
154
|
+
// We'll traverse the DOM to find these in order. Nested matches
|
|
155
|
+
// (e.g. a .ds-markdown code fragment inside an assistant turn) are
|
|
156
|
+
// skipped so each turn is exported exactly once.
|
|
157
|
+
const allElements = Array.from(
|
|
158
|
+
document.querySelectorAll(`${userSelector}, ${assistantSelector}`),
|
|
136
159
|
);
|
|
160
|
+
const outerElements = allElements.filter((el) => {
|
|
161
|
+
// Skip collapsible thinking-chain blocks: they live inside
|
|
162
|
+
// ds-think-content containers and would otherwise duplicate turns.
|
|
163
|
+
let ancestor = el.parentElement;
|
|
164
|
+
while (ancestor) {
|
|
165
|
+
const cls =
|
|
166
|
+
typeof ancestor.className === "string" ? ancestor.className : "";
|
|
167
|
+
if (/think/i.test(cls)) return false;
|
|
168
|
+
ancestor = ancestor.parentElement;
|
|
169
|
+
}
|
|
170
|
+
let parent = el.parentElement;
|
|
171
|
+
while (parent) {
|
|
172
|
+
if (
|
|
173
|
+
parent.matches &&
|
|
174
|
+
(parent.matches(userSelector) || parent.matches(assistantSelector))
|
|
175
|
+
) {
|
|
176
|
+
return false;
|
|
177
|
+
}
|
|
178
|
+
parent = parent.parentElement;
|
|
179
|
+
}
|
|
180
|
+
return true;
|
|
181
|
+
});
|
|
137
182
|
|
|
138
|
-
|
|
183
|
+
outerElements.forEach((el) => {
|
|
139
184
|
let role = "Unknown";
|
|
140
185
|
if (el.matches(userSelector)) {
|
|
141
186
|
role = "User";
|
package/ai/grok.js
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
import { ChatParser } from "./base.js";
|
|
2
|
+
import { convertToMarkdown } from "../utils/html-to-markdown.js";
|
|
3
|
+
|
|
4
|
+
export const GROK_BATCH_SIZE = 20;
|
|
5
|
+
|
|
6
|
+
export function getGrokConversationId(url) {
|
|
7
|
+
if (!url || typeof url !== "string") return null;
|
|
8
|
+
return (
|
|
9
|
+
url.match(/grok\.com\/(?:chat|c|conversation)\/([a-f0-9-]+)/i)?.[1] ?? null
|
|
10
|
+
);
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Order response-node entries into a single parent-linked chain.
|
|
15
|
+
* Response-nodes form a linear branch via parentResponseId; walk from
|
|
16
|
+
* the leaf (the id that is never a parent) back to the root and reverse.
|
|
17
|
+
*/
|
|
18
|
+
export function orderGrokNodes(responseNodes) {
|
|
19
|
+
if (!Array.isArray(responseNodes) || responseNodes.length === 0) return [];
|
|
20
|
+
const byId = new Map(responseNodes.map((n) => [n?.responseId, n]));
|
|
21
|
+
const parentIds = new Set(
|
|
22
|
+
responseNodes.map((n) => n?.parentResponseId).filter(Boolean),
|
|
23
|
+
);
|
|
24
|
+
let leaf = responseNodes.find((n) => n && !parentIds.has(n.responseId));
|
|
25
|
+
if (!leaf) leaf = responseNodes[responseNodes.length - 1];
|
|
26
|
+
const chain = [];
|
|
27
|
+
const seen = new Set();
|
|
28
|
+
let current = leaf;
|
|
29
|
+
while (current && current.responseId && !seen.has(current.responseId)) {
|
|
30
|
+
seen.add(current.responseId);
|
|
31
|
+
chain.unshift(current);
|
|
32
|
+
current = byId.get(current.parentResponseId) ?? null;
|
|
33
|
+
}
|
|
34
|
+
return chain;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function formatGrokResponses(responses, orderedIds, fallbackTitle) {
|
|
38
|
+
const byId = new Map((responses || []).map((r) => [r?.responseId, r]));
|
|
39
|
+
const ids =
|
|
40
|
+
Array.isArray(orderedIds) && orderedIds.length > 0
|
|
41
|
+
? orderedIds
|
|
42
|
+
: (responses || []).map((r) => r?.responseId).filter(Boolean);
|
|
43
|
+
const messages = [];
|
|
44
|
+
let title = fallbackTitle || "Grok Conversation";
|
|
45
|
+
for (const id of ids) {
|
|
46
|
+
const r = byId.get(id);
|
|
47
|
+
if (!r) continue;
|
|
48
|
+
const content = (r.message || "").trim();
|
|
49
|
+
if (!content) continue;
|
|
50
|
+
const role = r.sender === "human" ? "User" : "Grok";
|
|
51
|
+
const msg = { role, content };
|
|
52
|
+
if (r.createTime) msg.timestamp = r.createTime;
|
|
53
|
+
messages.push(msg);
|
|
54
|
+
}
|
|
55
|
+
return { messages, title };
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export class GrokParser extends ChatParser {
|
|
59
|
+
name = "Grok";
|
|
60
|
+
isAvailable(url) {
|
|
61
|
+
return url.includes("grok.com");
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
async fetchConversationMeta(conversationId) {
|
|
65
|
+
const url =
|
|
66
|
+
`https://grok.com/rest/app-chat/conversations_v2/${conversationId}` +
|
|
67
|
+
`?includeWorkspaces=true&includeTaskResult=true`;
|
|
68
|
+
const response = await fetch(url, {
|
|
69
|
+
method: "GET",
|
|
70
|
+
credentials: "include",
|
|
71
|
+
headers: { Accept: "application/json" },
|
|
72
|
+
});
|
|
73
|
+
if (!response.ok) {
|
|
74
|
+
throw new Error(`Grok API request failed: ${response.status}`);
|
|
75
|
+
}
|
|
76
|
+
return response.json();
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
async fetchResponseNodes(conversationId) {
|
|
80
|
+
const url = `https://grok.com/rest/app-chat/conversations/${conversationId}/response-node`;
|
|
81
|
+
const response = await fetch(url, {
|
|
82
|
+
method: "GET",
|
|
83
|
+
credentials: "include",
|
|
84
|
+
headers: { Accept: "application/json" },
|
|
85
|
+
});
|
|
86
|
+
if (!response.ok) {
|
|
87
|
+
throw new Error(`Grok response-node request failed: ${response.status}`);
|
|
88
|
+
}
|
|
89
|
+
return response.json();
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
async fetchResponses(conversationId, responseIds) {
|
|
93
|
+
const url = `https://grok.com/rest/app-chat/conversations/${conversationId}/load-responses`;
|
|
94
|
+
const collected = [];
|
|
95
|
+
for (let i = 0; i < responseIds.length; i += GROK_BATCH_SIZE) {
|
|
96
|
+
const chunk = responseIds.slice(i, i + GROK_BATCH_SIZE);
|
|
97
|
+
const response = await fetch(url, {
|
|
98
|
+
method: "POST",
|
|
99
|
+
credentials: "include",
|
|
100
|
+
headers: {
|
|
101
|
+
Accept: "application/json",
|
|
102
|
+
"Content-Type": "application/json",
|
|
103
|
+
},
|
|
104
|
+
body: JSON.stringify({ responseIds: chunk }),
|
|
105
|
+
});
|
|
106
|
+
if (!response.ok) {
|
|
107
|
+
throw new Error(`Grok load-responses failed: ${response.status}`);
|
|
108
|
+
}
|
|
109
|
+
const json = await response.json();
|
|
110
|
+
if (Array.isArray(json?.responses)) collected.push(...json.responses);
|
|
111
|
+
}
|
|
112
|
+
return collected;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
async parse(options = {}) {
|
|
116
|
+
const parserMode = options.parserMode || "prefer_api";
|
|
117
|
+
const currentUrl =
|
|
118
|
+
typeof window !== "undefined" && window.location
|
|
119
|
+
? window.location.href || ""
|
|
120
|
+
: "";
|
|
121
|
+
const metadata = {
|
|
122
|
+
Source: "Grok",
|
|
123
|
+
Date: new Date().toLocaleString(),
|
|
124
|
+
Link: currentUrl,
|
|
125
|
+
};
|
|
126
|
+
const domTitle =
|
|
127
|
+
(document.title || "").replace(/\s*-\s*Grok\s*$/i, "").trim() ||
|
|
128
|
+
"Grok Conversation";
|
|
129
|
+
|
|
130
|
+
// Primary: internal API (response-node ordering + load-responses bodies)
|
|
131
|
+
if (parserMode !== "prefer_dom") {
|
|
132
|
+
try {
|
|
133
|
+
const conversationId = getGrokConversationId(currentUrl);
|
|
134
|
+
if (conversationId) {
|
|
135
|
+
const [meta, nodes] = await Promise.all([
|
|
136
|
+
this.fetchConversationMeta(conversationId).catch(() => null),
|
|
137
|
+
this.fetchResponseNodes(conversationId),
|
|
138
|
+
]);
|
|
139
|
+
const ordered = orderGrokNodes(nodes?.responseNodes);
|
|
140
|
+
const orderedIds = ordered.map((n) => n.responseId);
|
|
141
|
+
if (orderedIds.length > 0) {
|
|
142
|
+
const responses = await this.fetchResponses(
|
|
143
|
+
conversationId,
|
|
144
|
+
orderedIds,
|
|
145
|
+
);
|
|
146
|
+
const title = meta?.conversation?.title?.trim() || domTitle;
|
|
147
|
+
const { messages } = formatGrokResponses(
|
|
148
|
+
responses,
|
|
149
|
+
orderedIds,
|
|
150
|
+
title,
|
|
151
|
+
);
|
|
152
|
+
if (messages.length > 0) {
|
|
153
|
+
return {
|
|
154
|
+
title,
|
|
155
|
+
messages,
|
|
156
|
+
url: currentUrl,
|
|
157
|
+
metadata: { ...metadata, Method: "API" },
|
|
158
|
+
};
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
} catch (e) {
|
|
163
|
+
console.warn(
|
|
164
|
+
"[AI Exporter] Grok API fetch failed, falling back to DOM:",
|
|
165
|
+
e,
|
|
166
|
+
);
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
// Secondary: DOM fallback
|
|
171
|
+
const userSelector = '[data-testid="user-message"]';
|
|
172
|
+
const assistantSelector = '[data-testid="assistant-message"]';
|
|
173
|
+
const candidates = Array.from(
|
|
174
|
+
document.querySelectorAll(`${userSelector}, ${assistantSelector}`),
|
|
175
|
+
);
|
|
176
|
+
// Keep only outermost matches to avoid nested bubbles duplicating turns.
|
|
177
|
+
const elements = candidates.filter((el) => {
|
|
178
|
+
let parent = el.parentElement;
|
|
179
|
+
while (parent) {
|
|
180
|
+
if (
|
|
181
|
+
parent.matches &&
|
|
182
|
+
(parent.matches(userSelector) || parent.matches(assistantSelector))
|
|
183
|
+
) {
|
|
184
|
+
return false;
|
|
185
|
+
}
|
|
186
|
+
parent = parent.parentElement;
|
|
187
|
+
}
|
|
188
|
+
return true;
|
|
189
|
+
});
|
|
190
|
+
|
|
191
|
+
const messages = [];
|
|
192
|
+
for (const el of elements) {
|
|
193
|
+
const isUser = el.matches(userSelector);
|
|
194
|
+
const role = isUser ? "User" : "Grok";
|
|
195
|
+
const contentEl =
|
|
196
|
+
el.querySelector(".response-content-markdown") ||
|
|
197
|
+
el.querySelector(".message-bubble") ||
|
|
198
|
+
el;
|
|
199
|
+
const clone = contentEl.cloneNode(true);
|
|
200
|
+
clone
|
|
201
|
+
.querySelectorAll(
|
|
202
|
+
".thinking-container, button, [data-testid='canvas-trigger']",
|
|
203
|
+
)
|
|
204
|
+
.forEach((n) => n.remove());
|
|
205
|
+
const text = convertToMarkdown(clone).trim();
|
|
206
|
+
if (text) messages.push({ role, content: text });
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
return {
|
|
210
|
+
title: domTitle,
|
|
211
|
+
messages,
|
|
212
|
+
url: currentUrl,
|
|
213
|
+
metadata: { ...metadata, Method: "DOM" },
|
|
214
|
+
};
|
|
215
|
+
}
|
|
216
|
+
}
|
package/ai/index.js
CHANGED
|
@@ -27,6 +27,7 @@ export { GoogleSearchAIParser } from "./google_search_ai.js";
|
|
|
27
27
|
export { GeminiCloudAssistParser } from "./gemini_cloud_assist.js";
|
|
28
28
|
export { JoylandParser } from "./joyland.js";
|
|
29
29
|
export { ChubParser } from "./chub.js";
|
|
30
|
+
export { GrokParser } from "./grok.js";
|
|
30
31
|
|
|
31
32
|
// Utilities
|
|
32
33
|
export { convertToMarkdown, cleanMarkdown } from "../utils/html-to-markdown.js";
|
package/ai/meta.js
CHANGED
|
@@ -1,13 +1,361 @@
|
|
|
1
1
|
import { ChatParser } from "./base.js";
|
|
2
2
|
import { convertToMarkdown } from "../utils/html-to-markdown.js";
|
|
3
3
|
|
|
4
|
+
// Observed Relay doc_ids for the conversation message list (Sept 2026).
|
|
5
|
+
// These rotate when Meta redeploys the web client. The parser tries the
|
|
6
|
+
// pinned IDs first, then resolves fresh ones from the page's own JS chunks,
|
|
7
|
+
// and finally falls back to DOM extraction.
|
|
8
|
+
export const META_MESSAGES_DOC_ID = "d6d27be74bdbe5bec25ad7685dc65ac4";
|
|
9
|
+
export const META_MESSAGES_PAGE_DOC_ID = "bc523f4325577cba4ccbac4ffdec7ff4";
|
|
10
|
+
export const META_PAGE_SIZE = 10;
|
|
11
|
+
export const META_MAX_SCRIPT_CHUNKS = 25;
|
|
12
|
+
export const META_MAX_CHUNK_CHARS = 5_000_000;
|
|
13
|
+
export const META_MAX_DOCID_TRIALS = 8;
|
|
14
|
+
|
|
15
|
+
// Markers that identify the conversation-messages Relay artifact inside a
|
|
16
|
+
// JS chunk. Candidates are only trusted from chunks containing one of these.
|
|
17
|
+
const META_QUERY_MARKERS = [
|
|
18
|
+
"latestBranchPath",
|
|
19
|
+
"GenAIMarkdownTextUXPrimitive",
|
|
20
|
+
"GenAIBotThinkingStatusPrimitive",
|
|
21
|
+
];
|
|
22
|
+
|
|
23
|
+
let cachedMetaDocIds = null;
|
|
24
|
+
|
|
25
|
+
/** Test hook: clear the in-memory resolved doc_id cache. */
|
|
26
|
+
export function __resetMetaDocIdCache() {
|
|
27
|
+
cachedMetaDocIds = null;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Extract persisted-query doc_ids from a JS chunk — but only when the chunk
|
|
32
|
+
* contains the conversation-messages query markers, so unrelated Relay
|
|
33
|
+
* artifacts (analytics, surveys) are never mistaken for message queries.
|
|
34
|
+
*/
|
|
35
|
+
export function extractMetaDocIdCandidates(jsText) {
|
|
36
|
+
if (typeof jsText !== "string") return [];
|
|
37
|
+
if (!META_QUERY_MARKERS.some((marker) => jsText.includes(marker))) return [];
|
|
38
|
+
const ids = new Set();
|
|
39
|
+
const patterns = [
|
|
40
|
+
/["']id["']\s*:\s*["']([a-f0-9]{32})["']/g,
|
|
41
|
+
/doc_id["']?\s*[:=]\s*["']([a-f0-9]{32})["']/g,
|
|
42
|
+
];
|
|
43
|
+
for (const re of patterns) {
|
|
44
|
+
let match;
|
|
45
|
+
while ((match = re.exec(jsText)) !== null) ids.add(match[1]);
|
|
46
|
+
}
|
|
47
|
+
return [...ids];
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** Absolute URLs of the page's own script chunks (bounded). */
|
|
51
|
+
export function getMetaScriptUrls(limit = META_MAX_SCRIPT_CHUNKS) {
|
|
52
|
+
try {
|
|
53
|
+
const urls = [];
|
|
54
|
+
const base =
|
|
55
|
+
typeof window !== "undefined" && window.location?.href
|
|
56
|
+
? window.location.href
|
|
57
|
+
: "https://www.meta.ai/";
|
|
58
|
+
for (const script of document.querySelectorAll("script[src]")) {
|
|
59
|
+
const src = script.getAttribute("src");
|
|
60
|
+
if (!src || src.startsWith("data:") || src.startsWith("blob:")) continue;
|
|
61
|
+
try {
|
|
62
|
+
const absolute = new URL(src, base).href;
|
|
63
|
+
if (!urls.includes(absolute)) urls.push(absolute);
|
|
64
|
+
} catch {
|
|
65
|
+
// Ignore unresolvable src values
|
|
66
|
+
}
|
|
67
|
+
if (urls.length >= limit) break;
|
|
68
|
+
}
|
|
69
|
+
return urls;
|
|
70
|
+
} catch {
|
|
71
|
+
return [];
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export function getMetaConversationId(url) {
|
|
76
|
+
if (!url || typeof url !== "string") return null;
|
|
77
|
+
return (
|
|
78
|
+
url.match(/meta\.ai\/(?:c|chat|prompt)\/([a-f0-9-]+)/i)?.[1] ??
|
|
79
|
+
url.match(/[?&]conversationId=([a-f0-9-]+)/i)?.[1] ??
|
|
80
|
+
null
|
|
81
|
+
);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
export function extractMetaUserText(node) {
|
|
85
|
+
if (!node) return "";
|
|
86
|
+
for (const key of ["userContent", "content", "originalUserPrompt"]) {
|
|
87
|
+
const val = node[key];
|
|
88
|
+
if (typeof val === "string" && val.trim()) return val.trim();
|
|
89
|
+
}
|
|
90
|
+
return "";
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function extractMetaAssistantText(node) {
|
|
94
|
+
if (!node) return "";
|
|
95
|
+
const renderer = node.contentRenderer || {};
|
|
96
|
+
const message = renderer.message || {};
|
|
97
|
+
if (typeof message.content === "string" && message.content.trim()) {
|
|
98
|
+
return message.content.trim();
|
|
99
|
+
}
|
|
100
|
+
// Fallback: walk unified_response sections for markdown primitives.
|
|
101
|
+
const sections = renderer.unified_response?.sections;
|
|
102
|
+
if (Array.isArray(sections)) {
|
|
103
|
+
const parts = [];
|
|
104
|
+
for (const section of sections) {
|
|
105
|
+
const primitive = section?.view_model?.primitive;
|
|
106
|
+
if (
|
|
107
|
+
primitive &&
|
|
108
|
+
typeof primitive.text === "string" &&
|
|
109
|
+
primitive.text.trim()
|
|
110
|
+
) {
|
|
111
|
+
parts.push(primitive.text.trim());
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
if (parts.length > 0) return parts.join("\n\n");
|
|
115
|
+
}
|
|
116
|
+
if (typeof node.content === "string" && node.content.trim()) {
|
|
117
|
+
return node.content.trim();
|
|
118
|
+
}
|
|
119
|
+
return "";
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
export function formatMetaEdges(edges) {
|
|
123
|
+
const seen = new Set();
|
|
124
|
+
const nodes = [];
|
|
125
|
+
for (const edge of edges || []) {
|
|
126
|
+
const cursor = edge?.cursor;
|
|
127
|
+
if (cursor) {
|
|
128
|
+
if (seen.has(cursor)) continue;
|
|
129
|
+
seen.add(cursor);
|
|
130
|
+
}
|
|
131
|
+
if (edge?.node) nodes.push(edge.node);
|
|
132
|
+
}
|
|
133
|
+
// Relay returns oldest-first across pages; order by branchPath as a guard.
|
|
134
|
+
nodes.sort((a, b) => {
|
|
135
|
+
const pa = parseInt(a?.branchPath ?? "0", 10);
|
|
136
|
+
const pb = parseInt(b?.branchPath ?? "0", 10);
|
|
137
|
+
if (Number.isNaN(pa) || Number.isNaN(pb)) return 0;
|
|
138
|
+
return pa - pb;
|
|
139
|
+
});
|
|
140
|
+
const messages = [];
|
|
141
|
+
for (const node of nodes) {
|
|
142
|
+
if (node.__typename === "UserMessage") {
|
|
143
|
+
const content = extractMetaUserText(node);
|
|
144
|
+
if (content) messages.push({ role: "User", content });
|
|
145
|
+
} else if (node.__typename === "AssistantMessage") {
|
|
146
|
+
const content = extractMetaAssistantText(node);
|
|
147
|
+
if (content) messages.push({ role: "Meta AI", content });
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
return messages;
|
|
151
|
+
}
|
|
152
|
+
|
|
4
153
|
export class MetaParser extends ChatParser {
|
|
5
154
|
name = "Meta AI";
|
|
6
155
|
isAvailable(url) {
|
|
7
156
|
return url.includes("meta.ai");
|
|
8
157
|
}
|
|
9
158
|
|
|
10
|
-
async
|
|
159
|
+
async fetchGraphQL(docId, variables) {
|
|
160
|
+
const response = await fetch("https://www.meta.ai/api/graphql", {
|
|
161
|
+
method: "POST",
|
|
162
|
+
credentials: "include",
|
|
163
|
+
headers: {
|
|
164
|
+
Accept: "application/json",
|
|
165
|
+
"Content-Type": "application/json",
|
|
166
|
+
},
|
|
167
|
+
body: JSON.stringify({ doc_id: docId, variables }),
|
|
168
|
+
});
|
|
169
|
+
if (!response.ok) {
|
|
170
|
+
throw new Error(`Meta GraphQL request failed: ${response.status}`);
|
|
171
|
+
}
|
|
172
|
+
return response.json();
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
async fetchConversation(conversationId, docIds = {}) {
|
|
176
|
+
const initialId = docIds.initial || META_MESSAGES_DOC_ID;
|
|
177
|
+
const pageId = docIds.page || META_MESSAGES_PAGE_DOC_ID;
|
|
178
|
+
const allEdges = [];
|
|
179
|
+
// First page (latest messages).
|
|
180
|
+
const first = await this.fetchGraphQL(initialId, {
|
|
181
|
+
conversationId,
|
|
182
|
+
});
|
|
183
|
+
const conv = first?.data?.conversation;
|
|
184
|
+
if (!conv || !conv.messages) {
|
|
185
|
+
throw new Error("Meta GraphQL response missing conversation.messages");
|
|
186
|
+
}
|
|
187
|
+
const title = conv.displayTitle || conv.title || "";
|
|
188
|
+
allEdges.push(...(conv.messages.edges || []));
|
|
189
|
+
// Walk backwards through history while older pages exist.
|
|
190
|
+
let pageInfo = conv.messages.pageInfo;
|
|
191
|
+
let guard = 0;
|
|
192
|
+
while (pageInfo?.hasPreviousPage && pageInfo?.startCursor && guard < 50) {
|
|
193
|
+
guard += 1;
|
|
194
|
+
const page = await this.fetchGraphQL(pageId, {
|
|
195
|
+
conversationId,
|
|
196
|
+
before: pageInfo.startCursor,
|
|
197
|
+
last: META_PAGE_SIZE,
|
|
198
|
+
});
|
|
199
|
+
const pageConv = page?.data?.conversation;
|
|
200
|
+
if (!pageConv?.messages) break;
|
|
201
|
+
allEdges.push(...(pageConv.messages.edges || []));
|
|
202
|
+
pageInfo = pageConv.messages.pageInfo;
|
|
203
|
+
}
|
|
204
|
+
return { title, edges: allEdges };
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/** Verify a candidate doc_id returns a message list (self-validating). */
|
|
208
|
+
async trialMetaDocId(docId, variables) {
|
|
209
|
+
try {
|
|
210
|
+
const json = await this.fetchGraphQL(docId, variables);
|
|
211
|
+
const messages = json?.data?.conversation?.messages;
|
|
212
|
+
if (messages && Array.isArray(messages.edges)) return json;
|
|
213
|
+
} catch {
|
|
214
|
+
// Invalid/retired doc_id — caller tries the next candidate.
|
|
215
|
+
}
|
|
216
|
+
return null;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* Resolve fresh doc_ids from the page's own JS chunks when the pinned IDs
|
|
221
|
+
* stop working. Every candidate is trial-verified against the live API, so
|
|
222
|
+
* wrong guesses cost one request and never corrupt the export. Results are
|
|
223
|
+
* cached in-memory for the session.
|
|
224
|
+
*/
|
|
225
|
+
async resolveMetaDocIds(conversationId) {
|
|
226
|
+
if (cachedMetaDocIds) return cachedMetaDocIds;
|
|
227
|
+
const tried = new Set();
|
|
228
|
+
const candidates = [];
|
|
229
|
+
for (const chunkUrl of getMetaScriptUrls()) {
|
|
230
|
+
let jsText;
|
|
231
|
+
try {
|
|
232
|
+
const response = await fetch(chunkUrl, {
|
|
233
|
+
method: "GET",
|
|
234
|
+
credentials: "include",
|
|
235
|
+
});
|
|
236
|
+
if (!response.ok) continue;
|
|
237
|
+
jsText = await response.text();
|
|
238
|
+
} catch {
|
|
239
|
+
continue;
|
|
240
|
+
}
|
|
241
|
+
if (!jsText || jsText.length > META_MAX_CHUNK_CHARS) continue;
|
|
242
|
+
for (const id of extractMetaDocIdCandidates(jsText)) {
|
|
243
|
+
if (!tried.has(id)) {
|
|
244
|
+
tried.add(id);
|
|
245
|
+
candidates.push(id);
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
if (tried.size >= META_MAX_DOCID_TRIALS) break;
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
let initial = null;
|
|
252
|
+
let initialJson = null;
|
|
253
|
+
for (const id of candidates.slice(0, META_MAX_DOCID_TRIALS)) {
|
|
254
|
+
initialJson = await this.trialMetaDocId(id, { conversationId });
|
|
255
|
+
if (initialJson) {
|
|
256
|
+
initial = id;
|
|
257
|
+
break;
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
if (!initial) {
|
|
261
|
+
throw new Error("No working Meta doc_id found in page chunks");
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
// The pagination query may differ from the initial-page query: prefer the
|
|
265
|
+
// initial winner, then trial the remaining candidates with page vars.
|
|
266
|
+
let page = initial;
|
|
267
|
+
const pageInfo = initialJson.data.conversation.messages.pageInfo;
|
|
268
|
+
if (pageInfo?.hasPreviousPage && pageInfo?.startCursor) {
|
|
269
|
+
const pageVars = {
|
|
270
|
+
conversationId,
|
|
271
|
+
before: pageInfo.startCursor,
|
|
272
|
+
last: META_PAGE_SIZE,
|
|
273
|
+
};
|
|
274
|
+
if (!(await this.trialMetaDocId(initial, pageVars))) {
|
|
275
|
+
for (const id of candidates.slice(0, META_MAX_DOCID_TRIALS)) {
|
|
276
|
+
if (id === initial) continue;
|
|
277
|
+
if (await this.trialMetaDocId(id, pageVars)) {
|
|
278
|
+
page = id;
|
|
279
|
+
break;
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
cachedMetaDocIds = { initial, page };
|
|
285
|
+
return cachedMetaDocIds;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/** Pick the richest markdown candidate (thinking stubs are near-empty). */
|
|
289
|
+
pickAssistantContent(el) {
|
|
290
|
+
const candidates = Array.from(
|
|
291
|
+
el.querySelectorAll(".markdown-content, .ur-markdown, .prose"),
|
|
292
|
+
);
|
|
293
|
+
const pool = candidates.length > 0 ? candidates : [el];
|
|
294
|
+
let best = null;
|
|
295
|
+
let bestLen = -1;
|
|
296
|
+
for (const cand of pool) {
|
|
297
|
+
const len = (cand.textContent || "").trim().length;
|
|
298
|
+
if (len > bestLen) {
|
|
299
|
+
bestLen = len;
|
|
300
|
+
best = cand;
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
return best || el;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
async parse(options = {}) {
|
|
307
|
+
const parserMode = options.parserMode || "prefer_api";
|
|
308
|
+
const currentUrl =
|
|
309
|
+
typeof window !== "undefined" && window.location
|
|
310
|
+
? window.location.href || ""
|
|
311
|
+
: "";
|
|
312
|
+
const metadata = {
|
|
313
|
+
Source: "Meta AI",
|
|
314
|
+
Date: new Date().toLocaleString(),
|
|
315
|
+
Link: currentUrl,
|
|
316
|
+
};
|
|
317
|
+
|
|
318
|
+
// Primary: internal GraphQL API (same-session cookies required).
|
|
319
|
+
// Pinned doc_ids first; if Meta rotated them, resolve fresh ones from
|
|
320
|
+
// the page's own JS chunks before giving up to DOM extraction.
|
|
321
|
+
if (parserMode !== "prefer_dom") {
|
|
322
|
+
const conversationId = getMetaConversationId(currentUrl);
|
|
323
|
+
if (conversationId) {
|
|
324
|
+
const runApi = async (docIds) => {
|
|
325
|
+
const { title: apiTitle, edges } = await this.fetchConversation(
|
|
326
|
+
conversationId,
|
|
327
|
+
docIds,
|
|
328
|
+
);
|
|
329
|
+
const messages = formatMetaEdges(edges);
|
|
330
|
+
if (messages.length === 0) return null;
|
|
331
|
+
return {
|
|
332
|
+
title: apiTitle || "Meta AI Session",
|
|
333
|
+
messages,
|
|
334
|
+
url: currentUrl,
|
|
335
|
+
metadata: { ...metadata, Method: "API" },
|
|
336
|
+
};
|
|
337
|
+
};
|
|
338
|
+
try {
|
|
339
|
+
// Attempt 1: pinned doc_ids (fast path, no extra requests).
|
|
340
|
+
const pinned = await runApi();
|
|
341
|
+
if (pinned) return pinned;
|
|
342
|
+
} catch (e) {
|
|
343
|
+
console.warn("[AI Exporter] Meta pinned doc_ids failed:", e);
|
|
344
|
+
try {
|
|
345
|
+
// Attempt 2: resolve fresh doc_ids from the page's JS chunks.
|
|
346
|
+
const resolved = await runApi(
|
|
347
|
+
await this.resolveMetaDocIds(conversationId),
|
|
348
|
+
);
|
|
349
|
+
if (resolved) return resolved;
|
|
350
|
+
} catch (e2) {
|
|
351
|
+
console.warn("[AI Exporter] Meta doc_id resolution failed:", e2);
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
console.warn("[AI Exporter] Meta API exhausted, falling back to DOM");
|
|
355
|
+
}
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
// Secondary: DOM extraction fallback
|
|
11
359
|
// Try to get the conversation title from the input field or the header button
|
|
12
360
|
const titleInput = document.querySelector(
|
|
13
361
|
'input[placeholder="Conversation title"]',
|
|
@@ -70,11 +418,10 @@ export class MetaParser extends ChatParser {
|
|
|
70
418
|
}
|
|
71
419
|
} else if (isAssistant) {
|
|
72
420
|
role = "Meta AI";
|
|
73
|
-
// Assistant
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
el.querySelector(".prose");
|
|
421
|
+
// Assistant turns nest several .markdown-content wrappers (a
|
|
422
|
+
// near-empty thinking stub plus the real answer). Convert the
|
|
423
|
+
// richest candidate so thinking stubs never shadow the answer.
|
|
424
|
+
const contentEl = this.pickAssistantContent(el);
|
|
78
425
|
if (contentEl) {
|
|
79
426
|
const clone = contentEl.cloneNode(true);
|
|
80
427
|
|
|
@@ -101,17 +448,13 @@ export class MetaParser extends ChatParser {
|
|
|
101
448
|
}
|
|
102
449
|
});
|
|
103
450
|
|
|
104
|
-
const
|
|
105
|
-
typeof window !== "undefined" && window.location
|
|
106
|
-
? window.location.href || ""
|
|
107
|
-
: "";
|
|
108
|
-
const metadata = {
|
|
451
|
+
const domMetadata = {
|
|
109
452
|
Source: "Meta AI",
|
|
110
453
|
Date: new Date().toLocaleString(),
|
|
111
454
|
Link: currentUrl,
|
|
112
455
|
Method: "DOM",
|
|
113
456
|
};
|
|
114
457
|
|
|
115
|
-
return { title, messages, url: currentUrl, metadata };
|
|
458
|
+
return { title, messages, url: currentUrl, metadata: domMetadata };
|
|
116
459
|
}
|
|
117
460
|
}
|
package/ai/mistral.js
CHANGED
|
@@ -8,13 +8,17 @@ export class MistralParser extends ChatParser {
|
|
|
8
8
|
}
|
|
9
9
|
|
|
10
10
|
async parse() {
|
|
11
|
-
// Extract Title
|
|
12
|
-
|
|
13
|
-
let title =
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
11
|
+
// Extract Title: document.title is the most reliable source; the
|
|
12
|
+
// sidebar truncate span may match unrelated UI ("Upgrade to Pro").
|
|
13
|
+
let title = (document.title || "")
|
|
14
|
+
.replace(/\s*-\s*Mistral\s*$/i, "")
|
|
15
|
+
.trim();
|
|
16
|
+
if (!title) {
|
|
17
|
+
const titleElement = document.querySelector(
|
|
18
|
+
"span.truncate.text-sm, [data-testid='conversation-title']",
|
|
19
|
+
);
|
|
20
|
+
title =
|
|
21
|
+
(titleElement?.textContent || "").trim() || "Mistral Conversation";
|
|
18
22
|
}
|
|
19
23
|
|
|
20
24
|
const messages = [];
|
|
@@ -28,21 +32,18 @@ export class MistralParser extends ChatParser {
|
|
|
28
32
|
if (role === "user") {
|
|
29
33
|
const contentEl =
|
|
30
34
|
el.querySelector(".select-text") ||
|
|
31
|
-
el.querySelector(".whitespace-pre-wrap")
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
});
|
|
35
|
+
el.querySelector(".whitespace-pre-wrap") ||
|
|
36
|
+
el;
|
|
37
|
+
const text = convertToMarkdown(contentEl).trim();
|
|
38
|
+
if (text) {
|
|
39
|
+
messages.push({ role: "User", content: text });
|
|
37
40
|
}
|
|
38
41
|
} else if (role === "assistant") {
|
|
39
|
-
const answerEl =
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
content: markdown,
|
|
45
|
-
});
|
|
42
|
+
const answerEl =
|
|
43
|
+
el.querySelector('[data-message-part-type="answer"]') || el;
|
|
44
|
+
const markdown = convertToMarkdown(answerEl).trim();
|
|
45
|
+
if (markdown) {
|
|
46
|
+
messages.push({ role: "Mistral", content: markdown });
|
|
46
47
|
}
|
|
47
48
|
}
|
|
48
49
|
}
|
package/ai/qwen.js
CHANGED
|
@@ -24,25 +24,44 @@ export class QwenParser extends ChatParser {
|
|
|
24
24
|
if (element) {
|
|
25
25
|
const text = element.textContent || element.value || element.innerText;
|
|
26
26
|
if (text && text.trim() && text !== document.title) {
|
|
27
|
-
title = text.trim();
|
|
27
|
+
title = text.trim().replace(/\s+/g, " ");
|
|
28
28
|
break;
|
|
29
29
|
}
|
|
30
30
|
}
|
|
31
31
|
}
|
|
32
|
+
if (title === "Qwen Chat" && document.title) {
|
|
33
|
+
const docTitle = document.title.trim();
|
|
34
|
+
if (docTitle && docTitle.toLowerCase() !== "qwen studio") {
|
|
35
|
+
title = docTitle;
|
|
36
|
+
}
|
|
37
|
+
}
|
|
32
38
|
|
|
33
39
|
const messages = [];
|
|
34
40
|
|
|
35
41
|
// chat.qwen.ai uses specific class names
|
|
36
|
-
|
|
42
|
+
let chatMessages = document.querySelectorAll(".qwen-chat-message");
|
|
43
|
+
// Fallback for markup drift: user/assistant content blocks in order.
|
|
44
|
+
if (chatMessages.length === 0) {
|
|
45
|
+
chatMessages = document.querySelectorAll(
|
|
46
|
+
".user-message-content, .qwen-markdown",
|
|
47
|
+
);
|
|
48
|
+
}
|
|
37
49
|
|
|
38
50
|
chatMessages.forEach((message) => {
|
|
39
|
-
|
|
51
|
+
// Fallback nodes are the content blocks themselves.
|
|
52
|
+
const isFallbackContent =
|
|
53
|
+
message.matches?.(".user-message-content, .qwen-markdown") ?? false;
|
|
54
|
+
const isUser = isFallbackContent
|
|
55
|
+
? message.matches(".user-message-content")
|
|
56
|
+
: message.classList.contains("qwen-chat-message-user");
|
|
40
57
|
const role = isUser ? "User" : "Qwen";
|
|
41
58
|
|
|
42
59
|
let content = "";
|
|
43
60
|
let attachments = [];
|
|
44
61
|
|
|
45
|
-
if (
|
|
62
|
+
if (isFallbackContent) {
|
|
63
|
+
content = convertToMarkdown(message);
|
|
64
|
+
} else if (isUser) {
|
|
46
65
|
// Extract attachments first
|
|
47
66
|
const fileItems = message.querySelectorAll(
|
|
48
67
|
".index-module__file-message-document___OjWnc",
|
package/ai/z_ai.js
CHANGED
|
@@ -1,15 +1,138 @@
|
|
|
1
1
|
import { ChatParser } from "./base.js";
|
|
2
2
|
import { convertToMarkdown } from "../utils/html-to-markdown.js";
|
|
3
3
|
|
|
4
|
+
export const ZAI_BATCH_SIZE = 20;
|
|
5
|
+
|
|
6
|
+
export function getZaiChatId(url) {
|
|
7
|
+
if (!url || typeof url !== "string") return null;
|
|
8
|
+
return url.match(/chat\.z\.ai\/c\/([a-f0-9-]+)/i)?.[1] ?? null;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
/** Walk the skeleton history (id -> {parentId}) from currentId to root. */
|
|
12
|
+
export function orderZaiHistory(messagesById, currentId) {
|
|
13
|
+
if (!messagesById || typeof messagesById !== "object") return [];
|
|
14
|
+
const ordered = [];
|
|
15
|
+
const seen = new Set();
|
|
16
|
+
let current = currentId ?? null;
|
|
17
|
+
// Fallback: if no currentId, start from the node nobody parents.
|
|
18
|
+
if (!current || !messagesById[current]) {
|
|
19
|
+
const childIds = new Set(
|
|
20
|
+
Object.values(messagesById).flatMap((m) => m?.childrenIds || []),
|
|
21
|
+
);
|
|
22
|
+
const leaf = Object.keys(messagesById).find((id) => !childIds.has(id));
|
|
23
|
+
current = leaf ?? null;
|
|
24
|
+
}
|
|
25
|
+
while (current && messagesById[current] && !seen.has(current)) {
|
|
26
|
+
seen.add(current);
|
|
27
|
+
ordered.unshift(current);
|
|
28
|
+
current = messagesById[current].parentId ?? null;
|
|
29
|
+
}
|
|
30
|
+
return ordered;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function formatZaiMessage(entry) {
|
|
34
|
+
if (!entry) return null;
|
|
35
|
+
const role = entry.role === "user" ? "User" : "Z.ai";
|
|
36
|
+
if (entry.role === "user") {
|
|
37
|
+
const content = (entry.content || "").trim();
|
|
38
|
+
return content ? { role, content } : null;
|
|
39
|
+
}
|
|
40
|
+
const blocks = Array.isArray(entry.content_blocks)
|
|
41
|
+
? entry.content_blocks
|
|
42
|
+
: [];
|
|
43
|
+
const texts = blocks
|
|
44
|
+
.filter((b) => b && b.type === "text" && typeof b.content === "string")
|
|
45
|
+
.map((b) => b.content.trim())
|
|
46
|
+
.filter(Boolean);
|
|
47
|
+
// Fallback to legacy top-level content string when blocks are absent.
|
|
48
|
+
if (
|
|
49
|
+
texts.length === 0 &&
|
|
50
|
+
typeof entry.content === "string" &&
|
|
51
|
+
entry.content.trim()
|
|
52
|
+
) {
|
|
53
|
+
texts.push(entry.content.trim());
|
|
54
|
+
}
|
|
55
|
+
if (texts.length === 0) return null;
|
|
56
|
+
let content = texts.join("\n\n");
|
|
57
|
+
const reasoning = blocks
|
|
58
|
+
.filter((b) => b && b.type === "reasoning" && typeof b.content === "string")
|
|
59
|
+
.map((b) => b.content.trim())
|
|
60
|
+
.filter(Boolean)
|
|
61
|
+
.join("\n\n");
|
|
62
|
+
if (reasoning) {
|
|
63
|
+
const quoted = reasoning
|
|
64
|
+
.split("\n")
|
|
65
|
+
.map((line) => `> ${line}`)
|
|
66
|
+
.join("\n");
|
|
67
|
+
content += `\n\n> 🧠 Thinking\n${quoted}`;
|
|
68
|
+
}
|
|
69
|
+
const msg = { role, content };
|
|
70
|
+
if (entry.timestamp) {
|
|
71
|
+
try {
|
|
72
|
+
msg.timestamp = new Date(entry.timestamp * 1000).toISOString();
|
|
73
|
+
} catch {
|
|
74
|
+
// Ignore timestamp formatting errors
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
return msg;
|
|
78
|
+
}
|
|
79
|
+
|
|
4
80
|
export class ZAiParser extends ChatParser {
|
|
5
81
|
name = "Z.ai";
|
|
6
82
|
isAvailable(url) {
|
|
7
83
|
return url.includes("chat.z.ai");
|
|
8
84
|
}
|
|
9
85
|
|
|
10
|
-
async
|
|
11
|
-
|
|
86
|
+
async fetchChatSkeleton(chatId) {
|
|
87
|
+
const url = `https://chat.z.ai/api/v1/chats/${chatId}`;
|
|
88
|
+
const response = await fetch(url, {
|
|
89
|
+
method: "GET",
|
|
90
|
+
credentials: "include",
|
|
91
|
+
headers: { Accept: "application/json" },
|
|
92
|
+
});
|
|
93
|
+
if (!response.ok) {
|
|
94
|
+
throw new Error(`Z.ai chat request failed: ${response.status}`);
|
|
95
|
+
}
|
|
96
|
+
return response.json();
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
async fetchMessageBatch(chatId, ids) {
|
|
100
|
+
const url = `https://chat.z.ai/api/v1/chats/${chatId}/messages/batch`;
|
|
101
|
+
const collected = {};
|
|
102
|
+
for (let i = 0; i < ids.length; i += ZAI_BATCH_SIZE) {
|
|
103
|
+
const chunk = ids.slice(i, i + ZAI_BATCH_SIZE);
|
|
104
|
+
const response = await fetch(url, {
|
|
105
|
+
method: "POST",
|
|
106
|
+
credentials: "include",
|
|
107
|
+
headers: {
|
|
108
|
+
Accept: "application/json",
|
|
109
|
+
"Content-Type": "application/json",
|
|
110
|
+
},
|
|
111
|
+
body: JSON.stringify({ ids: chunk }),
|
|
112
|
+
});
|
|
113
|
+
if (!response.ok) {
|
|
114
|
+
throw new Error(`Z.ai batch request failed: ${response.status}`);
|
|
115
|
+
}
|
|
116
|
+
const json = await response.json();
|
|
117
|
+
Object.assign(collected, json?.data || {});
|
|
118
|
+
}
|
|
119
|
+
return collected;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
async parse(options = {}) {
|
|
123
|
+
const parserMode = options.parserMode || "prefer_api";
|
|
124
|
+
const currentUrl =
|
|
125
|
+
typeof window !== "undefined" && window.location
|
|
126
|
+
? window.location.href || ""
|
|
127
|
+
: "";
|
|
128
|
+
const metadata = {
|
|
129
|
+
Source: "Z.ai",
|
|
130
|
+
Date: new Date().toLocaleString(),
|
|
131
|
+
Link: currentUrl,
|
|
132
|
+
};
|
|
133
|
+
|
|
12
134
|
const titleEl = document.querySelector("title");
|
|
135
|
+
let title = "";
|
|
13
136
|
if (titleEl) {
|
|
14
137
|
title = titleEl.textContent.trim().replace(/\s+/g, " ");
|
|
15
138
|
}
|
|
@@ -17,6 +140,43 @@ export class ZAiParser extends ChatParser {
|
|
|
17
140
|
title = document.title.trim().replace(/\s+/g, " ");
|
|
18
141
|
}
|
|
19
142
|
title = title || "Z.ai Chat";
|
|
143
|
+
|
|
144
|
+
// Primary: internal API (chat skeleton for ordering + batched bodies)
|
|
145
|
+
if (parserMode !== "prefer_dom") {
|
|
146
|
+
try {
|
|
147
|
+
const chatId = getZaiChatId(currentUrl);
|
|
148
|
+
if (chatId) {
|
|
149
|
+
const skeleton = await this.fetchChatSkeleton(chatId);
|
|
150
|
+
const history = skeleton?.chat?.history || {};
|
|
151
|
+
const orderedIds = orderZaiHistory(
|
|
152
|
+
history.messages || {},
|
|
153
|
+
history.currentId,
|
|
154
|
+
);
|
|
155
|
+
if (orderedIds.length > 0) {
|
|
156
|
+
const bodies = await this.fetchMessageBatch(chatId, orderedIds);
|
|
157
|
+
const apiTitle = (skeleton?.title || "").trim() || title;
|
|
158
|
+
const apiMessages = orderedIds
|
|
159
|
+
.map((id) => formatZaiMessage(bodies[id]))
|
|
160
|
+
.filter(Boolean);
|
|
161
|
+
if (apiMessages.length > 0) {
|
|
162
|
+
return {
|
|
163
|
+
title: apiTitle,
|
|
164
|
+
messages: apiMessages,
|
|
165
|
+
url: currentUrl,
|
|
166
|
+
metadata: { ...metadata, Method: "API" },
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
} catch (e) {
|
|
172
|
+
console.warn(
|
|
173
|
+
"[AI Exporter] Z.ai API fetch failed, falling back to DOM:",
|
|
174
|
+
e,
|
|
175
|
+
);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// Secondary: DOM fallback
|
|
20
180
|
const messages = [];
|
|
21
181
|
|
|
22
182
|
// Selectors for z.ai messages
|
|
@@ -92,17 +252,13 @@ export class ZAiParser extends ChatParser {
|
|
|
92
252
|
}
|
|
93
253
|
});
|
|
94
254
|
|
|
95
|
-
const
|
|
96
|
-
typeof window !== "undefined" && window.location
|
|
97
|
-
? window.location.href || ""
|
|
98
|
-
: "";
|
|
99
|
-
const metadata = {
|
|
255
|
+
const domMetadata = {
|
|
100
256
|
Source: "Z.ai",
|
|
101
257
|
Date: new Date().toLocaleString(),
|
|
102
258
|
Link: currentUrl,
|
|
103
259
|
Method: "DOM",
|
|
104
260
|
};
|
|
105
261
|
|
|
106
|
-
return { title, messages, url: currentUrl, metadata };
|
|
262
|
+
return { title, messages, url: currentUrl, metadata: domMetadata };
|
|
107
263
|
}
|
|
108
264
|
}
|
|
@@ -16,6 +16,7 @@ import { GoogleSearchAIParser } from "../ai/google_search_ai.js";
|
|
|
16
16
|
import { GeminiCloudAssistParser } from "../ai/gemini_cloud_assist.js";
|
|
17
17
|
import { JoylandParser } from "../ai/joyland.js";
|
|
18
18
|
import { ChubParser } from "../ai/chub.js";
|
|
19
|
+
import { GrokParser } from "../ai/grok.js";
|
|
19
20
|
|
|
20
21
|
/**
|
|
21
22
|
* Ordered list of parsers. First match wins.
|
|
@@ -39,6 +40,7 @@ export const parsers = [
|
|
|
39
40
|
new GeminiCloudAssistParser(),
|
|
40
41
|
new JoylandParser(),
|
|
41
42
|
new ChubParser(),
|
|
43
|
+
new GrokParser(),
|
|
42
44
|
];
|
|
43
45
|
|
|
44
46
|
/**
|
package/detection/domains.js
CHANGED
package/package.json
CHANGED
|
@@ -192,28 +192,63 @@ export function convertToMarkdown(htmlContent, options = {}) {
|
|
|
192
192
|
registerMath(el, latex, isBlock, clone.ownerDocument);
|
|
193
193
|
});
|
|
194
194
|
|
|
195
|
-
// 3. Process Google Search SGE LaTeX images with [data-xpm-latex]
|
|
196
|
-
clone
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
195
|
+
// 3. Process Google Search SGE LaTeX images with [data-xpm-latex] or fallback math roots
|
|
196
|
+
clone
|
|
197
|
+
.querySelectorAll(
|
|
198
|
+
"[data-xpm-latex], [data-xpm-copy-root][data-xpm-copy-text], [data-xpm-copy-root] img[alt]",
|
|
199
|
+
)
|
|
200
|
+
.forEach((el) => {
|
|
201
|
+
if (!clone.contains(el)) return;
|
|
202
|
+
const copyRoot = el.hasAttribute("data-xpm-copy-root")
|
|
203
|
+
? el
|
|
204
|
+
: el.closest("[data-xpm-copy-root]");
|
|
205
|
+
if (!copyRoot) return;
|
|
206
|
+
const blockContainer = el.closest(".cPGBZb");
|
|
207
|
+
const inlineWrapper = el.closest(".mTEjhd") || el.closest(".dteT0b");
|
|
208
|
+
const container = blockContainer || inlineWrapper || copyRoot;
|
|
209
|
+
if (!container.parentNode) return;
|
|
210
|
+
|
|
211
|
+
const img = el.tagName === "IMG" ? el : copyRoot.querySelector("img");
|
|
212
|
+
const latex =
|
|
213
|
+
copyRoot.getAttribute("data-xpm-latex") ||
|
|
214
|
+
el.getAttribute("data-xpm-latex") ||
|
|
215
|
+
copyRoot.getAttribute("data-xpm-copy-text") ||
|
|
216
|
+
(img && img.getAttribute("data-xpm-latex")) ||
|
|
217
|
+
el.getAttribute("alt") ||
|
|
218
|
+
(img && img.getAttribute("alt")) ||
|
|
219
|
+
"";
|
|
220
|
+
if (!latex) return;
|
|
221
|
+
|
|
222
|
+
// Determine if block or inline based on DOM structure
|
|
223
|
+
let isBlock = false;
|
|
224
|
+
if (blockContainer) {
|
|
225
|
+
// Full block container (.cPGBZb)
|
|
226
|
+
isBlock = true;
|
|
227
|
+
} else if (inlineWrapper) {
|
|
228
|
+
// Inline math wrapper (.mTEjhd, .dteT0b)
|
|
229
|
+
isBlock = false;
|
|
230
|
+
} else {
|
|
231
|
+
const style = copyRoot.getAttribute("style") || "";
|
|
232
|
+
if (/display:\s*inline/i.test(style)) {
|
|
233
|
+
isBlock = false;
|
|
234
|
+
} else {
|
|
235
|
+
// Check enclosing block element (p, li, td, th, div) for surrounding text
|
|
236
|
+
const enclosingBlock = container.closest("p, li, td, th, div");
|
|
237
|
+
if (enclosingBlock) {
|
|
238
|
+
const cloneBlock = enclosingBlock.cloneNode(true);
|
|
239
|
+
const targetInClone =
|
|
240
|
+
cloneBlock
|
|
241
|
+
.querySelector("[data-xpm-latex]")
|
|
242
|
+
?.closest("[data-xpm-copy-root]") ||
|
|
243
|
+
cloneBlock.querySelector("[data-xpm-latex]");
|
|
244
|
+
if (targetInClone) targetInClone.remove();
|
|
245
|
+
isBlock = cloneBlock.textContent.trim() === "";
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
}
|
|
214
249
|
|
|
215
|
-
|
|
216
|
-
|
|
250
|
+
registerMath(container, latex, isBlock, clone.ownerDocument);
|
|
251
|
+
});
|
|
217
252
|
|
|
218
253
|
// 4. Process block display KaTeX (.katex-display)
|
|
219
254
|
clone.querySelectorAll(".katex-display").forEach((el) => {
|