decant-core 1.12.0 → 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +16 -0
- package/README.md +8 -2
- package/ai/chatgpt.js +250 -7
- package/ai/chatgpt_helper.js +21 -7
- package/ai/claude.js +249 -4
- package/ai/deepseek.js +7 -0
- package/ai/gemini.js +311 -12
- package/ai/meta.js +14 -2
- package/package.json +1 -1
- package/utils/timestamps.js +65 -0
package/LICENSE
CHANGED
|
@@ -1,3 +1,19 @@
|
|
|
1
|
+
ADDITIONAL PERMISSION UNDER GNU AGPL VERSION 3 SECTION 7:
|
|
2
|
+
FOSS LINKING EXCEPTION
|
|
3
|
+
|
|
4
|
+
Permission is granted to link, import, or bundle decant-core into projects
|
|
5
|
+
distributed under any OSI-approved open source license (including MPL-2.0, MIT,
|
|
6
|
+
Apache-2.0, and BSD) and distribute the resulting work under that project's
|
|
7
|
+
license, without requiring the enclosing project to be licensed under AGPLv3.
|
|
8
|
+
Any modifications directly made to decant-core source files remain subject to AGPLv3.
|
|
9
|
+
|
|
10
|
+
COMMERCIAL USE
|
|
11
|
+
If you wish to use decant-core in closed-source, proprietary, or commercial
|
|
12
|
+
software that cannot comply with the AGPLv3, a commercial license is available.
|
|
13
|
+
Please contact office@covai.org for licensing terms.
|
|
14
|
+
|
|
15
|
+
==============================================================================
|
|
16
|
+
|
|
1
17
|
GNU AFFERO GENERAL PUBLIC LICENSE
|
|
2
18
|
Version 3, 19 November 2007
|
|
3
19
|
|
package/README.md
CHANGED
|
@@ -151,9 +151,15 @@ normalized `parse()` contract. For the full extraction-strategy breakdown and ma
|
|
|
151
151
|
|
|
152
152
|
## License
|
|
153
153
|
|
|
154
|
-
`decant-core` is licensed under the **GNU Affero General Public License v3.0 (AGPL-3.0-only)
|
|
154
|
+
`decant-core` is licensed under the **GNU Affero General Public License v3.0 (AGPL-3.0-only)** with a **FOSS Linking Exception**, alongside a **Commercial License** option.
|
|
155
155
|
|
|
156
|
-
|
|
156
|
+
### FOSS Linking Exception (Open Source)
|
|
157
|
+
|
|
158
|
+
Permission is granted to link, import, or bundle `decant-core` into projects distributed under any OSI-approved open source license (including **MPL-2.0**, **MIT**, **Apache-2.0**, and **BSD**) and distribute the resulting work under that project's license, without requiring the enclosing project to be licensed under AGPLv3. Any modifications directly made to `decant-core` source files remain subject to AGPLv3.
|
|
159
|
+
|
|
160
|
+
### Commercial License
|
|
161
|
+
|
|
162
|
+
If you wish to use `decant-core` in closed-source, proprietary, or commercial software that cannot comply with the AGPLv3, a commercial license is available. Please contact `office@covai.org` for licensing terms.
|
|
157
163
|
|
|
158
164
|
### Third-Party Test Fixtures Notice
|
|
159
165
|
|
package/ai/chatgpt.js
CHANGED
|
@@ -196,6 +196,221 @@ export function extractSharedConversationFromDom(
|
|
|
196
196
|
return null;
|
|
197
197
|
}
|
|
198
198
|
|
|
199
|
+
function cleanApiPartText(partText) {
|
|
200
|
+
return partText
|
|
201
|
+
.replace(/\u{E0000}[\u{E0000}-\u{E007F}]*/gu, "")
|
|
202
|
+
.replace(/citeturn\d+\w*/g, "")
|
|
203
|
+
.trim();
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
// Linearize the newer backend-api conversation shape, which returns a
|
|
207
|
+
// `messages` array instead of a `mapping` tree. Output segments match the
|
|
208
|
+
// mapping-based linearize() format ({ type: "text" | "thought" | "image" })
|
|
209
|
+
// so both flow through the same formatApiResult().
|
|
210
|
+
export function linearizeMessagesArray(apiMessages, includeImages) {
|
|
211
|
+
const messages = [];
|
|
212
|
+
if (!Array.isArray(apiMessages)) return messages;
|
|
213
|
+
|
|
214
|
+
const pushOrMerge = (entry, isThoughtMsg) => {
|
|
215
|
+
if (
|
|
216
|
+
entry.role === "ChatGPT" &&
|
|
217
|
+
messages.length > 0 &&
|
|
218
|
+
messages[messages.length - 1].role === "ChatGPT"
|
|
219
|
+
) {
|
|
220
|
+
const prevMsg = messages[messages.length - 1];
|
|
221
|
+
if (isThoughtMsg) {
|
|
222
|
+
prevMsg.segments.unshift(...entry.segments);
|
|
223
|
+
} else {
|
|
224
|
+
prevMsg.segments.push(...entry.segments);
|
|
225
|
+
}
|
|
226
|
+
Object.assign(prevMsg.citeMap, entry.citeMap);
|
|
227
|
+
Object.assign(prevMsg.imageGroupMap, entry.imageGroupMap);
|
|
228
|
+
if (entry.timestamp && !prevMsg.timestamp) {
|
|
229
|
+
prevMsg.timestamp = entry.timestamp;
|
|
230
|
+
}
|
|
231
|
+
} else {
|
|
232
|
+
messages.push(entry);
|
|
233
|
+
}
|
|
234
|
+
};
|
|
235
|
+
|
|
236
|
+
for (const msg of apiMessages) {
|
|
237
|
+
if (!msg) continue;
|
|
238
|
+
const role = msg?.author?.role;
|
|
239
|
+
if (role !== "user" && role !== "assistant" && role !== "tool") continue;
|
|
240
|
+
if (msg.metadata?.is_visually_hidden_from_conversation === true) continue;
|
|
241
|
+
|
|
242
|
+
const content = msg.content || {};
|
|
243
|
+
const contentType = content.content_type;
|
|
244
|
+
const segments = [];
|
|
245
|
+
// Mirror the mapping path's thought detection (content types plus
|
|
246
|
+
// author/recipient markers).
|
|
247
|
+
const isThoughtMsg =
|
|
248
|
+
msg?.author?.name === "thought" ||
|
|
249
|
+
msg?.recipient === "thought" ||
|
|
250
|
+
contentType === "thought" ||
|
|
251
|
+
contentType === "thoughts" ||
|
|
252
|
+
contentType === "reasoning_recap" ||
|
|
253
|
+
msg?.metadata?.reasoning_status === "is_reasoning";
|
|
254
|
+
|
|
255
|
+
// Reasoning summaries: thoughts = [{ summary, content }]
|
|
256
|
+
if (Array.isArray(content.thoughts) && content.thoughts.length > 0) {
|
|
257
|
+
const thoughtParts = content.thoughts
|
|
258
|
+
.map((t) =>
|
|
259
|
+
t.summary ? `**${t.summary}**\n${t.content || ""}` : t.content || "",
|
|
260
|
+
)
|
|
261
|
+
.map((t) => t.trim())
|
|
262
|
+
.filter(Boolean);
|
|
263
|
+
if (thoughtParts.length > 0) {
|
|
264
|
+
segments.push({ type: "thought", content: thoughtParts.join("\n\n") });
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
// Reasoning recap (e.g. "Worked for 11s")
|
|
269
|
+
if (
|
|
270
|
+
contentType === "reasoning_recap" &&
|
|
271
|
+
typeof content.content === "string" &&
|
|
272
|
+
content.content.trim()
|
|
273
|
+
) {
|
|
274
|
+
segments.push({ type: "thought", content: content.content.trim() });
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
// Tool invocations (content_type "code", e.g. Deep Research args JSON)
|
|
278
|
+
// and tool-role messages are not user-visible prose — skip them.
|
|
279
|
+
if (contentType !== "code" && role !== "tool") {
|
|
280
|
+
const parts = Array.isArray(content.parts) ? content.parts : [];
|
|
281
|
+
for (const part of parts) {
|
|
282
|
+
let partText = "";
|
|
283
|
+
let isThoughtPart = isThoughtMsg;
|
|
284
|
+
if (typeof part === "string") {
|
|
285
|
+
partText = part;
|
|
286
|
+
} else if (part && typeof part === "object") {
|
|
287
|
+
if (part.content_type === "text" && typeof part.text === "string") {
|
|
288
|
+
partText = part.text;
|
|
289
|
+
} else if (
|
|
290
|
+
part.content_type === "thought" &&
|
|
291
|
+
typeof part.text === "string"
|
|
292
|
+
) {
|
|
293
|
+
partText = part.text;
|
|
294
|
+
isThoughtPart = true;
|
|
295
|
+
} else if (
|
|
296
|
+
part.content_type === "audio_transcription" &&
|
|
297
|
+
typeof part.text === "string"
|
|
298
|
+
) {
|
|
299
|
+
partText = part.text;
|
|
300
|
+
} else if (
|
|
301
|
+
includeImages &&
|
|
302
|
+
part?.content_type === "image_asset_pointer" &&
|
|
303
|
+
part?.asset_pointer
|
|
304
|
+
) {
|
|
305
|
+
segments.push({
|
|
306
|
+
type: "image",
|
|
307
|
+
fileId: part.asset_pointer.split("://")[1],
|
|
308
|
+
});
|
|
309
|
+
continue;
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
const text = partText ? cleanApiPartText(partText) : "";
|
|
313
|
+
if (text) {
|
|
314
|
+
segments.push({
|
|
315
|
+
type: isThoughtPart ? "thought" : "text",
|
|
316
|
+
content: text,
|
|
317
|
+
});
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// Standalone content.text without parts (plain text / execution
|
|
322
|
+
// output), mirroring the mapping path.
|
|
323
|
+
if (
|
|
324
|
+
parts.length === 0 &&
|
|
325
|
+
typeof content.text === "string" &&
|
|
326
|
+
content.text.trim()
|
|
327
|
+
) {
|
|
328
|
+
segments.push({ type: "text", content: content.text.trim() });
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
// Deep Research reports (widget_state), attachments, and Canvas
|
|
333
|
+
// documents from message metadata, mirroring the mapping path.
|
|
334
|
+
if (role !== "tool") {
|
|
335
|
+
const widgetRaw =
|
|
336
|
+
msg.metadata?.chatgpt_sdk?.widget_state ||
|
|
337
|
+
msg.metadata?.tool_response_metadata?.venus_widget_state;
|
|
338
|
+
if (widgetRaw) {
|
|
339
|
+
try {
|
|
340
|
+
const widget =
|
|
341
|
+
typeof widgetRaw === "string" ? JSON.parse(widgetRaw) : widgetRaw;
|
|
342
|
+
const reportText =
|
|
343
|
+
widget.report_message?.content?.parts?.[0] || widget.markdown;
|
|
344
|
+
const steering = widget.steering_acknowledgement;
|
|
345
|
+
let researchContent = "";
|
|
346
|
+
if (steering) researchContent += `${steering}\n\n`;
|
|
347
|
+
if (reportText) researchContent += reportText;
|
|
348
|
+
if (researchContent.trim()) {
|
|
349
|
+
segments.push({ type: "text", content: researchContent.trim() });
|
|
350
|
+
}
|
|
351
|
+
} catch {
|
|
352
|
+
// Ignore widget state JSON parse errors
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
if (
|
|
357
|
+
Array.isArray(msg.metadata?.attachments) &&
|
|
358
|
+
msg.metadata.attachments.length > 0
|
|
359
|
+
) {
|
|
360
|
+
const fileNames = msg.metadata.attachments
|
|
361
|
+
.map((att) => att.name)
|
|
362
|
+
.filter(Boolean);
|
|
363
|
+
if (fileNames.length > 0) {
|
|
364
|
+
segments.push({
|
|
365
|
+
type: "text",
|
|
366
|
+
content: `[Attached: ${fileNames.join(", ")}]`,
|
|
367
|
+
});
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
if (msg.metadata?.canvas?.title) {
|
|
372
|
+
segments.push({
|
|
373
|
+
type: "text",
|
|
374
|
+
content: `[Canvas: ${msg.metadata.canvas.title}]`,
|
|
375
|
+
});
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
if (segments.length === 0) continue;
|
|
380
|
+
|
|
381
|
+
const displayRole = role === "user" ? "User" : "ChatGPT";
|
|
382
|
+
const timestamp = msg?.create_time
|
|
383
|
+
? new Date(msg.create_time * 1000).toLocaleString()
|
|
384
|
+
: null;
|
|
385
|
+
// Citation / image-group references, mirroring the mapping path.
|
|
386
|
+
const citeMap = {};
|
|
387
|
+
const imageGroupMap = {};
|
|
388
|
+
for (const ref of msg?.metadata?.content_references ?? []) {
|
|
389
|
+
if (ref.matched_text) {
|
|
390
|
+
if (ref.items?.length) citeMap[ref.matched_text] = ref.items;
|
|
391
|
+
if (
|
|
392
|
+
ref.type === "image_group" ||
|
|
393
|
+
ref.matched_text.includes("image_group")
|
|
394
|
+
) {
|
|
395
|
+
imageGroupMap[ref.matched_text] = ref;
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
pushOrMerge(
|
|
400
|
+
{
|
|
401
|
+
role: displayRole,
|
|
402
|
+
segments,
|
|
403
|
+
citeMap,
|
|
404
|
+
imageGroupMap,
|
|
405
|
+
timestamp,
|
|
406
|
+
},
|
|
407
|
+
isThoughtMsg,
|
|
408
|
+
);
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
return messages;
|
|
412
|
+
}
|
|
413
|
+
|
|
199
414
|
export function linearize(mapping, includeImages, currentNodeId) {
|
|
200
415
|
let path = [];
|
|
201
416
|
const leafId = resolveActiveLeafNode(mapping, currentNodeId);
|
|
@@ -268,8 +483,20 @@ export function linearize(mapping, includeImages, currentNodeId) {
|
|
|
268
483
|
) {
|
|
269
484
|
const segments = [];
|
|
270
485
|
const parts = msg?.content?.parts ?? [];
|
|
486
|
+
// Tool-invocation payloads (content_type "code") are not user-visible
|
|
487
|
+
// prose, whether carried as standalone text or inside parts.
|
|
488
|
+
const isToolInvocation = msg?.content?.content_type === "code";
|
|
271
489
|
|
|
272
490
|
for (const part of parts) {
|
|
491
|
+
if (isToolInvocation) {
|
|
492
|
+
if (!(
|
|
493
|
+
includeImages &&
|
|
494
|
+
part?.content_type === "image_asset_pointer" &&
|
|
495
|
+
part?.asset_pointer
|
|
496
|
+
)) {
|
|
497
|
+
continue;
|
|
498
|
+
}
|
|
499
|
+
}
|
|
273
500
|
let partText = "";
|
|
274
501
|
let isThoughtPart = isThoughtMsg;
|
|
275
502
|
|
|
@@ -316,11 +543,15 @@ export function linearize(mapping, includeImages, currentNodeId) {
|
|
|
316
543
|
}
|
|
317
544
|
}
|
|
318
545
|
|
|
319
|
-
// Handle standalone content.text (e.g. execution_output or plain text)
|
|
546
|
+
// Handle standalone content.text (e.g. execution_output or plain text).
|
|
547
|
+
// Tool-invocation payloads (content_type "code", e.g. Deep Research
|
|
548
|
+
// "/Deep Research App/start" args JSON) are not user-visible prose —
|
|
549
|
+
// the site renders a status card instead — so never dump them as text.
|
|
320
550
|
if (
|
|
321
551
|
typeof msg.content?.text === "string" &&
|
|
322
552
|
msg.content.text.trim() &&
|
|
323
|
-
parts.length === 0
|
|
553
|
+
parts.length === 0 &&
|
|
554
|
+
msg.content?.content_type !== "code"
|
|
324
555
|
) {
|
|
325
556
|
segments.push({ type: "text", content: msg.content.text.trim() });
|
|
326
557
|
}
|
|
@@ -887,6 +1118,7 @@ export class ChatGPTParser extends ChatParser {
|
|
|
887
1118
|
Link: currentUrl,
|
|
888
1119
|
Model:
|
|
889
1120
|
convoData?.model_slug ||
|
|
1121
|
+
convoData?.default_model_slug ||
|
|
890
1122
|
(typeof document !== "undefined" && document.querySelector
|
|
891
1123
|
? document.querySelector('[data-testid="model-selector-dropdown"]')
|
|
892
1124
|
?.innerText
|
|
@@ -961,11 +1193,22 @@ export class ChatGPTParser extends ChatParser {
|
|
|
961
1193
|
};
|
|
962
1194
|
}
|
|
963
1195
|
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
1196
|
+
// The backend returns either the legacy `mapping` tree or the newer
|
|
1197
|
+
// `messages` array shape — support both.
|
|
1198
|
+
let apiMessages = [];
|
|
1199
|
+
if (result.data.mapping) {
|
|
1200
|
+
apiMessages = linearize(
|
|
1201
|
+
result.data.mapping,
|
|
1202
|
+
includeImages,
|
|
1203
|
+
result.data.current_node,
|
|
1204
|
+
);
|
|
1205
|
+
}
|
|
1206
|
+
if (apiMessages.length === 0 && Array.isArray(result.data.messages)) {
|
|
1207
|
+
apiMessages = linearizeMessagesArray(
|
|
1208
|
+
result.data.messages,
|
|
1209
|
+
includeImages,
|
|
1210
|
+
);
|
|
1211
|
+
}
|
|
969
1212
|
if (apiMessages.length > 0) {
|
|
970
1213
|
return this.formatApiResult(
|
|
971
1214
|
result.data,
|
package/ai/chatgpt_helper.js
CHANGED
|
@@ -84,16 +84,30 @@ if (!window.__chatgptHelperInjected) {
|
|
|
84
84
|
let images = {};
|
|
85
85
|
if (includeImages) {
|
|
86
86
|
const fileIds = new Set();
|
|
87
|
-
|
|
87
|
+
const collectImagePointer = (part) => {
|
|
88
|
+
if (
|
|
89
|
+
part &&
|
|
90
|
+
part.content_type === "image_asset_pointer" &&
|
|
91
|
+
part.asset_pointer
|
|
92
|
+
) {
|
|
93
|
+
fileIds.add(part.asset_pointer.split("://")[1]);
|
|
94
|
+
}
|
|
95
|
+
};
|
|
96
|
+
// Legacy `mapping` tree shape…
|
|
97
|
+
for (const node of Object.values(data.mapping || {})) {
|
|
88
98
|
const msg = node.message;
|
|
89
99
|
if (msg && msg.content && Array.isArray(msg.content.parts)) {
|
|
90
100
|
for (const part of msg.content.parts) {
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
101
|
+
collectImagePointer(part);
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
// …and the newer `messages` array shape.
|
|
106
|
+
if (Array.isArray(data.messages)) {
|
|
107
|
+
for (const msg of data.messages) {
|
|
108
|
+
if (msg && msg.content && Array.isArray(msg.content.parts)) {
|
|
109
|
+
for (const part of msg.content.parts) {
|
|
110
|
+
collectImagePointer(part);
|
|
97
111
|
}
|
|
98
112
|
}
|
|
99
113
|
}
|
package/ai/claude.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { ChatParser } from "./base.js";
|
|
2
2
|
import { convertToMarkdown } from "../utils/html-to-markdown.js";
|
|
3
|
+
import { pickTimestamp } from "../utils/timestamps.js";
|
|
3
4
|
|
|
4
5
|
async function getOrganizationId() {
|
|
5
6
|
try {
|
|
@@ -103,6 +104,124 @@ const MIME_TO_LANG = {
|
|
|
103
104
|
"application/vnd.ant.code": "text",
|
|
104
105
|
};
|
|
105
106
|
|
|
107
|
+
// Binary archives cannot be inlined as text in any export format; they are
|
|
108
|
+
// listed by name so they at least appear in markdown/html/json exports.
|
|
109
|
+
const BINARY_ARCHIVE_MIMES = new Set([
|
|
110
|
+
"application/x-tar",
|
|
111
|
+
"application/gzip",
|
|
112
|
+
"application/zip",
|
|
113
|
+
"application/x-7z-compressed",
|
|
114
|
+
"application/x-rar-compressed",
|
|
115
|
+
]);
|
|
116
|
+
|
|
117
|
+
function formatFileSize(bytes) {
|
|
118
|
+
if (typeof bytes !== "number" || !Number.isFinite(bytes) || bytes < 0) {
|
|
119
|
+
return "";
|
|
120
|
+
}
|
|
121
|
+
if (bytes < 1024) return `${bytes} B`;
|
|
122
|
+
return `${(bytes / 1024).toFixed(1)} KB`;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
function basenameOfPath(filePath) {
|
|
126
|
+
if (typeof filePath !== "string" || !filePath) return "file";
|
|
127
|
+
const base = filePath.split("/").pop();
|
|
128
|
+
return base || "file";
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function isBinaryArchive(mimeType, filePath) {
|
|
132
|
+
if (mimeType && BINARY_ARCHIVE_MIMES.has(mimeType)) return true;
|
|
133
|
+
return /\.(tar\.gz|tgz|tar|zip|gz|7z|rar)$/i.test(filePath || "");
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// Collect files surfaced via the `present_files` tool (File Creation
|
|
137
|
+
// integration). The tool_use block carries `input.filepaths`; the matching
|
|
138
|
+
// tool_result block (joined via tool_use_id) carries display names + mime
|
|
139
|
+
// types as `local_resource` entries. Either side may be missing, so resolve
|
|
140
|
+
// metadata when available and fall back to bare paths otherwise.
|
|
141
|
+
function collectPresentedFiles(branch) {
|
|
142
|
+
const pathsByToolUseId = new Map();
|
|
143
|
+
const resourcesByToolUseId = new Map();
|
|
144
|
+
for (const msg of branch) {
|
|
145
|
+
if (!Array.isArray(msg?.content)) continue;
|
|
146
|
+
for (const block of msg.content) {
|
|
147
|
+
if (block?.type === "tool_use" && block?.name === "present_files") {
|
|
148
|
+
const filepaths = block?.input?.filepaths;
|
|
149
|
+
if (Array.isArray(filepaths) && filepaths.length > 0 && block.id) {
|
|
150
|
+
pathsByToolUseId.set(block.id, filepaths);
|
|
151
|
+
}
|
|
152
|
+
} else if (
|
|
153
|
+
block?.type === "tool_result" &&
|
|
154
|
+
Array.isArray(block.content)
|
|
155
|
+
) {
|
|
156
|
+
const resources = block.content.filter(
|
|
157
|
+
(item) => item && item.type === "local_resource" && item.file_path,
|
|
158
|
+
);
|
|
159
|
+
if (resources.length > 0) {
|
|
160
|
+
const toolUseId = block.tool_use_id || block.id;
|
|
161
|
+
if (toolUseId) resourcesByToolUseId.set(toolUseId, resources);
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
const filesByToolUseId = new Map();
|
|
168
|
+
const seenPaths = new Set();
|
|
169
|
+
const toEntry = (resource, fallbackPath) => {
|
|
170
|
+
const filePath = resource?.file_path || fallbackPath || "";
|
|
171
|
+
const name = resource?.name || basenameOfPath(filePath);
|
|
172
|
+
return {
|
|
173
|
+
key: resource?.uuid || filePath || `${name}`,
|
|
174
|
+
name,
|
|
175
|
+
path: filePath,
|
|
176
|
+
mime: resource?.mime_type || "",
|
|
177
|
+
};
|
|
178
|
+
};
|
|
179
|
+
|
|
180
|
+
for (const [toolUseId, resources] of resourcesByToolUseId.entries()) {
|
|
181
|
+
const entries = [];
|
|
182
|
+
for (const resource of resources) {
|
|
183
|
+
// Skip the inline "say they are below" helper text item (type: text).
|
|
184
|
+
if (!resource.file_path) continue;
|
|
185
|
+
if (seenPaths.has(resource.file_path)) continue;
|
|
186
|
+
seenPaths.add(resource.file_path);
|
|
187
|
+
entries.push(toEntry(resource));
|
|
188
|
+
}
|
|
189
|
+
if (entries.length > 0) filesByToolUseId.set(toolUseId, entries);
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
for (const [toolUseId, filepaths] of pathsByToolUseId.entries()) {
|
|
193
|
+
const existing = filesByToolUseId.get(toolUseId) || [];
|
|
194
|
+
const entries = [...existing];
|
|
195
|
+
for (const filePath of filepaths) {
|
|
196
|
+
if (typeof filePath !== "string" || !filePath) continue;
|
|
197
|
+
if (seenPaths.has(filePath)) continue;
|
|
198
|
+
seenPaths.add(filePath);
|
|
199
|
+
entries.push(toEntry(null, filePath));
|
|
200
|
+
}
|
|
201
|
+
if (entries.length > 0) filesByToolUseId.set(toolUseId, entries);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
return filesByToolUseId;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function formatGeneratedFilesSection(files) {
|
|
208
|
+
if (!Array.isArray(files) || files.length === 0) return "";
|
|
209
|
+
const lines = files.map((file) => {
|
|
210
|
+
const displayName = file.name || basenameOfPath(file.path);
|
|
211
|
+
const meta = [];
|
|
212
|
+
if (file.mime) meta.push(file.mime);
|
|
213
|
+
if (isBinaryArchive(file.mime, file.path || displayName)) {
|
|
214
|
+
meta.push("binary archive — download from Claude UI");
|
|
215
|
+
}
|
|
216
|
+
const suffix = meta.length > 0 ? ` _(${meta.join(", ")})_` : "";
|
|
217
|
+
const pathSuffix =
|
|
218
|
+
file.path && file.path !== displayName ? ` — \`${file.path}\`` : "";
|
|
219
|
+
return `- \`${displayName}\`${suffix}${pathSuffix}`;
|
|
220
|
+
});
|
|
221
|
+
const label = files.length === 1 ? "Generated file:" : "Generated files:";
|
|
222
|
+
return `\n\n**${label}**\n${lines.join("\n")}\n\n`;
|
|
223
|
+
}
|
|
224
|
+
|
|
106
225
|
function extractArtifactsFromText(text) {
|
|
107
226
|
const artifactRegex = /<antArtifact[^>]*>([\s\S]*?)<\/antArtifact>/g;
|
|
108
227
|
const artifacts = [];
|
|
@@ -267,6 +386,44 @@ function extractArtifacts(message, foldedArtifacts = new Map()) {
|
|
|
267
386
|
return artifacts;
|
|
268
387
|
}
|
|
269
388
|
|
|
389
|
+
// File cards rendered for generated files (markdown docs, tarballs, …).
|
|
390
|
+
// They carry no text content for convertToMarkdown, so extract their display
|
|
391
|
+
// names (aria-label="View <name>") and drop the nodes to avoid button noise.
|
|
392
|
+
function extractFileCards(root) {
|
|
393
|
+
if (!root || typeof root.querySelectorAll !== "function") return [];
|
|
394
|
+
const names = [];
|
|
395
|
+
const cards = root.querySelectorAll('[data-testid="file-card-open"]');
|
|
396
|
+
for (const card of cards) {
|
|
397
|
+
const label = card.getAttribute && card.getAttribute("aria-label");
|
|
398
|
+
const match = typeof label === "string" && label.match(/^View\s+(.+)$/i);
|
|
399
|
+
const name = match ? match[1].trim() : "";
|
|
400
|
+
// No name-based dedup: two cards may legitimately share a display name
|
|
401
|
+
// (same basename in different directories). Each card element is visited
|
|
402
|
+
// exactly once, so nothing is double-counted here.
|
|
403
|
+
if (name) {
|
|
404
|
+
names.push(name);
|
|
405
|
+
}
|
|
406
|
+
// Remove the whole card element so download buttons / type badges
|
|
407
|
+
// don't leak into the markdown conversion — but only when the parent
|
|
408
|
+
// is a dedicated card wrapper. If the button shares its parent with
|
|
409
|
+
// other prose, remove just the button to avoid deleting message text.
|
|
410
|
+
const parent =
|
|
411
|
+
card.parentElement && card.parentElement !== root
|
|
412
|
+
? card.parentElement
|
|
413
|
+
: null;
|
|
414
|
+
const isDedicatedCard =
|
|
415
|
+
!!parent &&
|
|
416
|
+
parent.querySelectorAll('[data-testid="file-card-open"]').length === 1;
|
|
417
|
+
const target = isDedicatedCard ? parent : card;
|
|
418
|
+
if (target && typeof target.remove === "function") {
|
|
419
|
+
target.remove();
|
|
420
|
+
} else if (card.parentNode) {
|
|
421
|
+
card.parentNode.removeChild(card);
|
|
422
|
+
}
|
|
423
|
+
}
|
|
424
|
+
return names;
|
|
425
|
+
}
|
|
426
|
+
|
|
270
427
|
function unrollInteractiveElements(root, doc) {
|
|
271
428
|
if (!root || !doc) return;
|
|
272
429
|
|
|
@@ -371,6 +528,8 @@ export class ClaudeParser extends ChatParser {
|
|
|
371
528
|
|
|
372
529
|
const branch = getCurrentBranch(data);
|
|
373
530
|
const foldedArtifacts = collectArtifacts(branch);
|
|
531
|
+
const presentedFilesByToolUseId = collectPresentedFiles(branch);
|
|
532
|
+
const emittedFileKeys = new Set();
|
|
374
533
|
|
|
375
534
|
const toolResultMap = new Map();
|
|
376
535
|
for (const msg of branch) {
|
|
@@ -473,6 +632,37 @@ export class ClaudeParser extends ChatParser {
|
|
|
473
632
|
: `> - ${qText}\n`;
|
|
474
633
|
}
|
|
475
634
|
contentStr += `${qStr}\n`;
|
|
635
|
+
} else if (
|
|
636
|
+
block.name === "present_files" &&
|
|
637
|
+
presentedFilesByToolUseId.has(block.id)
|
|
638
|
+
) {
|
|
639
|
+
// Files surfaced via the File Creation integration
|
|
640
|
+
// (markdown docs, tarball, …). Without this they are
|
|
641
|
+
// silently dropped from every export format.
|
|
642
|
+
const files = presentedFilesByToolUseId.get(block.id);
|
|
643
|
+
const fresh = files.filter(
|
|
644
|
+
(file) => !emittedFileKeys.has(file.key),
|
|
645
|
+
);
|
|
646
|
+
fresh.forEach((file) => emittedFileKeys.add(file.key));
|
|
647
|
+
if (fresh.length > 0) {
|
|
648
|
+
contentStr += formatGeneratedFilesSection(fresh);
|
|
649
|
+
}
|
|
650
|
+
}
|
|
651
|
+
} else if (
|
|
652
|
+
block.type === "tool_result" &&
|
|
653
|
+
Array.isArray(block.content)
|
|
654
|
+
) {
|
|
655
|
+
// Orphan file presentation: the matching tool_use block may
|
|
656
|
+
// sit outside the current branch, so emit unseen
|
|
657
|
+
// local_resource entries directly from the result.
|
|
658
|
+
const toolUseId = block.tool_use_id || block.id;
|
|
659
|
+
const files = presentedFilesByToolUseId.get(toolUseId) || [];
|
|
660
|
+
const fresh = files.filter(
|
|
661
|
+
(file) => !emittedFileKeys.has(file.key),
|
|
662
|
+
);
|
|
663
|
+
fresh.forEach((file) => emittedFileKeys.add(file.key));
|
|
664
|
+
if (fresh.length > 0) {
|
|
665
|
+
contentStr += formatGeneratedFilesSection(fresh);
|
|
476
666
|
}
|
|
477
667
|
}
|
|
478
668
|
}
|
|
@@ -491,8 +681,9 @@ export class ClaudeParser extends ChatParser {
|
|
|
491
681
|
if (attachment.file_name) {
|
|
492
682
|
let header = `### Attachment: ${attachment.file_name}`;
|
|
493
683
|
const meta = [];
|
|
494
|
-
|
|
495
|
-
|
|
684
|
+
const sizeLabel = formatFileSize(attachment.file_size);
|
|
685
|
+
if (sizeLabel) {
|
|
686
|
+
meta.push(sizeLabel);
|
|
496
687
|
}
|
|
497
688
|
if (attachment.file_type) {
|
|
498
689
|
meta.push(attachment.file_type);
|
|
@@ -505,7 +696,9 @@ export class ClaudeParser extends ChatParser {
|
|
|
505
696
|
contentStr += `\`\`\`\`\n${attachment.extracted_content}\n\`\`\`\`\n\n`;
|
|
506
697
|
}
|
|
507
698
|
} else if (attachment.extracted_content) {
|
|
508
|
-
|
|
699
|
+
const sizeLabel = formatFileSize(attachment.file_size);
|
|
700
|
+
const sizeSuffix = sizeLabel ? ` _(${sizeLabel})_` : "";
|
|
701
|
+
contentStr += `\n\n### Pasted content${sizeSuffix}\n\`\`\`\`\n${attachment.extracted_content}\n\`\`\`\`\n\n`;
|
|
509
702
|
}
|
|
510
703
|
}
|
|
511
704
|
}
|
|
@@ -544,6 +737,13 @@ export class ClaudeParser extends ChatParser {
|
|
|
544
737
|
if (thinkingStr) {
|
|
545
738
|
msgObj.thinking = thinkingStr;
|
|
546
739
|
}
|
|
740
|
+
const timestamp = pickTimestamp(message, [
|
|
741
|
+
"created_at",
|
|
742
|
+
"updated_at",
|
|
743
|
+
]);
|
|
744
|
+
if (timestamp) {
|
|
745
|
+
msgObj.timestamp = timestamp;
|
|
746
|
+
}
|
|
547
747
|
messages.push(msgObj);
|
|
548
748
|
}
|
|
549
749
|
|
|
@@ -568,6 +768,14 @@ export class ClaudeParser extends ChatParser {
|
|
|
568
768
|
messages.push({
|
|
569
769
|
role: "Claude Artifact",
|
|
570
770
|
content: artContent.trim(),
|
|
771
|
+
...(pickTimestamp(message, ["created_at", "updated_at"])
|
|
772
|
+
? {
|
|
773
|
+
timestamp: pickTimestamp(message, [
|
|
774
|
+
"created_at",
|
|
775
|
+
"updated_at",
|
|
776
|
+
]),
|
|
777
|
+
}
|
|
778
|
+
: {}),
|
|
571
779
|
});
|
|
572
780
|
}
|
|
573
781
|
}
|
|
@@ -678,8 +886,16 @@ export class ClaudeParser extends ChatParser {
|
|
|
678
886
|
) {
|
|
679
887
|
role = "Claude";
|
|
680
888
|
const clone = el.cloneNode(true);
|
|
889
|
+
// Extract file cards first: unrollInteractiveElements strips all
|
|
890
|
+
// buttons, which would destroy the card markers.
|
|
891
|
+
const fileCardNames = extractFileCards(clone);
|
|
681
892
|
unrollInteractiveElements(clone, el.ownerDocument || document);
|
|
682
893
|
content = convertToMarkdown(clone);
|
|
894
|
+
if (fileCardNames.length > 0) {
|
|
895
|
+
content += formatGeneratedFilesSection(
|
|
896
|
+
fileCardNames.map((name) => ({ name, path: "", mime: "" })),
|
|
897
|
+
);
|
|
898
|
+
}
|
|
683
899
|
} else if (el.matches(".artifact-block-cell")) {
|
|
684
900
|
role = "Claude Artifact";
|
|
685
901
|
|
|
@@ -712,7 +928,36 @@ export class ClaudeParser extends ChatParser {
|
|
|
712
928
|
}
|
|
713
929
|
|
|
714
930
|
if (content) {
|
|
715
|
-
|
|
931
|
+
const msgObj = { role, content };
|
|
932
|
+
// Best-effort DOM timestamp. <time datetime> usually lives in the
|
|
933
|
+
// sibling MessageActions toolbar inside the same transcript row —
|
|
934
|
+
// not inside the message element itself. Scope to the closest row
|
|
935
|
+
// so we never borrow the previous/next turn's timestamp. Sparse in
|
|
936
|
+
// practice (hover-only on some turns) — absent stays absent.
|
|
937
|
+
try {
|
|
938
|
+
let timeEl =
|
|
939
|
+
typeof el.querySelector === "function"
|
|
940
|
+
? el.querySelector("time[datetime]")
|
|
941
|
+
: null;
|
|
942
|
+
if (!timeEl && typeof el.closest === "function") {
|
|
943
|
+
// Narrow containers only: transcript-row holds one turn, so its
|
|
944
|
+
// time belongs to this message. Never fall back to broad
|
|
945
|
+
// containers like article (many turns) — wrong date is worse
|
|
946
|
+
// than no date.
|
|
947
|
+
const row = el.closest(
|
|
948
|
+
'[data-testid="transcript-row"], .group\\/message-row',
|
|
949
|
+
);
|
|
950
|
+
timeEl = row?.querySelector?.("time[datetime]") || null;
|
|
951
|
+
}
|
|
952
|
+
const datetime = timeEl?.getAttribute?.("datetime");
|
|
953
|
+
const timestamp = pickTimestamp({ datetime }, ["datetime"]);
|
|
954
|
+
if (timestamp) {
|
|
955
|
+
msgObj.timestamp = timestamp;
|
|
956
|
+
}
|
|
957
|
+
} catch {
|
|
958
|
+
// Ignore DOM timestamp lookup errors
|
|
959
|
+
}
|
|
960
|
+
messages.push(msgObj);
|
|
716
961
|
}
|
|
717
962
|
}
|
|
718
963
|
|
package/ai/deepseek.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { ChatParser } from "./base.js";
|
|
2
2
|
import { convertToMarkdown } from "../utils/html-to-markdown.js";
|
|
3
|
+
import { normalizeTimestamp } from "../utils/timestamps.js";
|
|
3
4
|
|
|
4
5
|
function getUserToken() {
|
|
5
6
|
try {
|
|
@@ -120,6 +121,12 @@ async function fetchDeepSeekConversation(sessionId, token) {
|
|
|
120
121
|
if (thinking) {
|
|
121
122
|
msg.thinking = thinking;
|
|
122
123
|
}
|
|
124
|
+
// Turn granularity: inserted_at is seconds-epoch, ~identical within a
|
|
125
|
+
// USER/ASSISTANT pair, so both sides share the turn's timestamp.
|
|
126
|
+
const timestamp = normalizeTimestamp(msgNode?.inserted_at);
|
|
127
|
+
if (timestamp) {
|
|
128
|
+
msg.timestamp = timestamp;
|
|
129
|
+
}
|
|
123
130
|
return msg;
|
|
124
131
|
})
|
|
125
132
|
.filter((msg) => msg.content.length > 0);
|
package/ai/gemini.js
CHANGED
|
@@ -461,10 +461,15 @@ export class GeminiParser extends ChatParser {
|
|
|
461
461
|
}
|
|
462
462
|
|
|
463
463
|
const modelText = this.findModelTextInApiItem(item, options);
|
|
464
|
-
|
|
464
|
+
const researchExtras = this.extractDeepResearchExtras(item);
|
|
465
|
+
const combined = [modelText, researchExtras]
|
|
466
|
+
.map((t) => this.stripChipPlaceholders(t))
|
|
467
|
+
.filter((t) => t && t.trim())
|
|
468
|
+
.join("\n\n");
|
|
469
|
+
if (combined.trim()) {
|
|
465
470
|
messages.push({
|
|
466
471
|
role: "Model",
|
|
467
|
-
content: normalizeLatexMath(
|
|
472
|
+
content: normalizeLatexMath(combined.trim()),
|
|
468
473
|
turnId,
|
|
469
474
|
});
|
|
470
475
|
}
|
|
@@ -473,6 +478,213 @@ export class GeminiParser extends ChatParser {
|
|
|
473
478
|
return messages;
|
|
474
479
|
}
|
|
475
480
|
|
|
481
|
+
// Deep-research turns render a short summary plus placeholder chip links
|
|
482
|
+
// (e.g. http://googleusercontent.com/immersive_entry_chip/0) whose real
|
|
483
|
+
// content — research plan, full report, citation map — lives in adjacent
|
|
484
|
+
// candidate slots. Only the known chip placeholders are stripped; every
|
|
485
|
+
// other URL (including googleusercontent subdomains hosting real images
|
|
486
|
+
// and links inside markdown) is left intact.
|
|
487
|
+
stripChipPlaceholders(text) {
|
|
488
|
+
if (typeof text !== "string" || !text) return text;
|
|
489
|
+
const chipPattern =
|
|
490
|
+
/<?https?:\/\/googleusercontent\.com\/(?:immersive_entry_chip|deep_research_confirmation_content)(?:\/\d*)?>?/;
|
|
491
|
+
const chipPatternGlobal = new RegExp(chipPattern.source, "g");
|
|
492
|
+
return text
|
|
493
|
+
.split("\n")
|
|
494
|
+
.filter((line) => {
|
|
495
|
+
if (!chipPattern.test(line)) return true;
|
|
496
|
+
return line.replace(chipPatternGlobal, "").trim() !== "";
|
|
497
|
+
})
|
|
498
|
+
.map((line) => line.replace(chipPatternGlobal, ""))
|
|
499
|
+
.join("\n")
|
|
500
|
+
.replace(/\n{3,}/g, "\n\n");
|
|
501
|
+
}
|
|
502
|
+
|
|
503
|
+
getApiCandidates(item) {
|
|
504
|
+
try {
|
|
505
|
+
if (!Array.isArray(item[3])) return [];
|
|
506
|
+
const candidates = Array.isArray(item[3][0]) ? item[3][0] : item[3];
|
|
507
|
+
return candidates.filter((cand) => Array.isArray(cand));
|
|
508
|
+
} catch {
|
|
509
|
+
return [];
|
|
510
|
+
}
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
buildDeepResearchCiteMap(citeGroups) {
|
|
514
|
+
const map = new Map();
|
|
515
|
+
try {
|
|
516
|
+
const groups = Array.isArray(citeGroups) ? citeGroups : [citeGroups];
|
|
517
|
+
for (const group of groups) {
|
|
518
|
+
if (!group || typeof group !== "object" || Array.isArray(group))
|
|
519
|
+
continue;
|
|
520
|
+
for (const entries of Object.values(group)) {
|
|
521
|
+
if (!Array.isArray(entries)) continue;
|
|
522
|
+
for (const entry of entries) {
|
|
523
|
+
if (!Array.isArray(entry) || !Array.isArray(entry[1])) continue;
|
|
524
|
+
for (const source of entry[1]) {
|
|
525
|
+
// Source shape: [null, null, null,
|
|
526
|
+
// [detail, number, ...]] where detail = [favicon, url, title].
|
|
527
|
+
if (!Array.isArray(source) || !Array.isArray(source[3])) continue;
|
|
528
|
+
const detail = source[3][0];
|
|
529
|
+
const url = Array.isArray(detail) ? detail[1] : null;
|
|
530
|
+
const number = source[3][1];
|
|
531
|
+
if (
|
|
532
|
+
typeof url === "string" &&
|
|
533
|
+
url.startsWith("http") &&
|
|
534
|
+
typeof number === "number" &&
|
|
535
|
+
!map.has(number)
|
|
536
|
+
) {
|
|
537
|
+
map.set(number, {
|
|
538
|
+
url,
|
|
539
|
+
title:
|
|
540
|
+
typeof detail[2] === "string" && detail[2]
|
|
541
|
+
? detail[2]
|
|
542
|
+
: url,
|
|
543
|
+
});
|
|
544
|
+
}
|
|
545
|
+
}
|
|
546
|
+
}
|
|
547
|
+
}
|
|
548
|
+
}
|
|
549
|
+
} catch {
|
|
550
|
+
// Ignore malformed citation maps
|
|
551
|
+
}
|
|
552
|
+
return map;
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
resolveDeepResearchCites(markdown, citeMap) {
|
|
556
|
+
if (typeof markdown !== "string" || !(citeMap instanceof Map)) {
|
|
557
|
+
return markdown;
|
|
558
|
+
}
|
|
559
|
+
return markdown.replace(/ ?\[cite: ([\d,\s]+)\]/g, (match, nums) => {
|
|
560
|
+
const numbers = [
|
|
561
|
+
...new Set(
|
|
562
|
+
nums
|
|
563
|
+
.split(",")
|
|
564
|
+
.map((n) => parseInt(n.trim(), 10))
|
|
565
|
+
.filter((n) => Number.isFinite(n)),
|
|
566
|
+
),
|
|
567
|
+
];
|
|
568
|
+
if (numbers.length === 0) return "";
|
|
569
|
+
const links = numbers.map((n) => {
|
|
570
|
+
const cite = citeMap.get(n);
|
|
571
|
+
return cite ? `[[${n}]](${cite.url})` : `[${n}]`;
|
|
572
|
+
});
|
|
573
|
+
return ` ${links.join(" ")}`;
|
|
574
|
+
});
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
extractImmersiveDocFromCandidate(cand) {
|
|
578
|
+
try {
|
|
579
|
+
if (!Array.isArray(cand[30]) || cand[30].length === 0) return "";
|
|
580
|
+
const doc = cand[30][0];
|
|
581
|
+
if (!Array.isArray(doc)) return "";
|
|
582
|
+
// Guard: immersive research documents carry this task marker.
|
|
583
|
+
if (doc[3] !== "agency-placeholder-task-id") return "";
|
|
584
|
+
if (typeof doc[4] !== "string" || doc[4].trim().length < 100) return "";
|
|
585
|
+
const title = typeof doc[2] === "string" && doc[2] ? doc[2] : "Report";
|
|
586
|
+
const citeMap = this.buildDeepResearchCiteMap(doc[5]);
|
|
587
|
+
const markdown = this.resolveDeepResearchCites(doc[4].trim(), citeMap);
|
|
588
|
+
return `## ${title}\n\n${markdown}`;
|
|
589
|
+
} catch {
|
|
590
|
+
return "";
|
|
591
|
+
}
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
extractResearchPlanFromCandidate(cand) {
|
|
595
|
+
try {
|
|
596
|
+
if (!Array.isArray(cand[12])) return "";
|
|
597
|
+
for (const annotation of cand[12]) {
|
|
598
|
+
if (
|
|
599
|
+
!annotation ||
|
|
600
|
+
typeof annotation !== "object" ||
|
|
601
|
+
Array.isArray(annotation) ||
|
|
602
|
+
!Array.isArray(annotation["56"])
|
|
603
|
+
) {
|
|
604
|
+
continue;
|
|
605
|
+
}
|
|
606
|
+
const [planTitle, steps] = annotation["56"];
|
|
607
|
+
if (!Array.isArray(steps) || steps.length === 0) continue;
|
|
608
|
+
const lines = steps.map((step, idx) => {
|
|
609
|
+
if (!Array.isArray(step)) return null;
|
|
610
|
+
const stepTitle = step[1] || `Step ${idx + 1}`;
|
|
611
|
+
const desc =
|
|
612
|
+
typeof step[2] === "string" && step[2].trim()
|
|
613
|
+
? `: ${step[2].trim()}`
|
|
614
|
+
: "";
|
|
615
|
+
return `${idx + 1}. **${stepTitle}**${desc}`;
|
|
616
|
+
});
|
|
617
|
+
const valid = lines.filter(Boolean);
|
|
618
|
+
if (valid.length === 0) continue;
|
|
619
|
+
const heading =
|
|
620
|
+
typeof planTitle === "string" && planTitle
|
|
621
|
+
? `### ${planTitle} — research plan`
|
|
622
|
+
: "### Research plan";
|
|
623
|
+
return `${heading}\n${valid.join("\n")}`;
|
|
624
|
+
}
|
|
625
|
+
} catch {
|
|
626
|
+
// Ignore malformed plan annotations
|
|
627
|
+
}
|
|
628
|
+
return "";
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
extractActivitySources(item, limit = 40) {
|
|
632
|
+
const seen = new Map();
|
|
633
|
+
try {
|
|
634
|
+
const trail = item[3]?.[4];
|
|
635
|
+
if (!Array.isArray(trail)) return [];
|
|
636
|
+
for (const entry of trail) {
|
|
637
|
+
const detail = entry?.[4]?.[2];
|
|
638
|
+
const url = Array.isArray(detail) ? detail[1] : null;
|
|
639
|
+
if (typeof url !== "string" || !url.startsWith("http")) continue;
|
|
640
|
+
if (seen.has(url)) continue;
|
|
641
|
+
const title =
|
|
642
|
+
typeof detail[2] === "string" && detail[2] ? detail[2] : url;
|
|
643
|
+
seen.set(url, title);
|
|
644
|
+
}
|
|
645
|
+
} catch {
|
|
646
|
+
// Ignore malformed activity trails
|
|
647
|
+
}
|
|
648
|
+
const all = [...seen.entries()];
|
|
649
|
+
const shown = all.slice(0, limit);
|
|
650
|
+
const lines = shown.map(([url, title]) => `- [${title}](${url})`);
|
|
651
|
+
if (all.length > shown.length) {
|
|
652
|
+
lines.push(`- …and ${all.length - shown.length} more`);
|
|
653
|
+
}
|
|
654
|
+
return lines;
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
extractDeepResearchExtras(item) {
|
|
658
|
+
const parts = [];
|
|
659
|
+
try {
|
|
660
|
+
// Extras must come from the same candidate that supplied the visible
|
|
661
|
+
// text (findModelTextInApiItem uses the first candidate with text),
|
|
662
|
+
// never mixed in from alternate drafts.
|
|
663
|
+
const candidates = this.getApiCandidates(item);
|
|
664
|
+
let anchorIdx = candidates.findIndex(
|
|
665
|
+
(cand) =>
|
|
666
|
+
(Array.isArray(cand[1]) && typeof cand[1][0] === "string") ||
|
|
667
|
+
typeof cand[1] === "string" ||
|
|
668
|
+
(typeof cand[0] === "string" && cand[0].length > 50),
|
|
669
|
+
);
|
|
670
|
+
if (anchorIdx === -1) anchorIdx = 0;
|
|
671
|
+
const anchor = candidates[anchorIdx];
|
|
672
|
+
if (anchor) {
|
|
673
|
+
const plan = this.extractResearchPlanFromCandidate(anchor);
|
|
674
|
+
if (plan) parts.push(plan);
|
|
675
|
+
const doc = this.extractImmersiveDocFromCandidate(anchor);
|
|
676
|
+
if (doc) parts.push(doc);
|
|
677
|
+
}
|
|
678
|
+
const sourceLines = this.extractActivitySources(item);
|
|
679
|
+
if (sourceLines.length > 0) {
|
|
680
|
+
parts.push(`**Sources consulted:**\n${sourceLines.join("\n")}`);
|
|
681
|
+
}
|
|
682
|
+
} catch {
|
|
683
|
+
// Never let research extras break the base message
|
|
684
|
+
}
|
|
685
|
+
return parts.filter(Boolean).join("\n\n");
|
|
686
|
+
}
|
|
687
|
+
|
|
476
688
|
findUserTextInApiItem(item) {
|
|
477
689
|
try {
|
|
478
690
|
if (typeof item[2]?.[0]?.[0] === "string") return item[2][0][0];
|
|
@@ -693,20 +905,26 @@ export class GeminiParser extends ChatParser {
|
|
|
693
905
|
if (markdownDiv) {
|
|
694
906
|
const clone = markdownDiv.cloneNode(true);
|
|
695
907
|
|
|
696
|
-
// Remove UI buttons, thought overlays,
|
|
908
|
+
// Remove UI buttons, thought overlays, follow-up suggestion
|
|
909
|
+
// widgets, and interactive toolbars.
|
|
910
|
+
// Note: .hide-from-message-actions is NOT removed — it wraps
|
|
911
|
+
// deep-research plan widgets whose text must be kept (buttons
|
|
912
|
+
// inside are still stripped above).
|
|
697
913
|
clone
|
|
698
914
|
.querySelectorAll(
|
|
699
|
-
"button, .thoughts-container, .thoughts-wrapper, model-thoughts, .table-footer,
|
|
915
|
+
"button, follow-up, .follow-up-container, .thoughts-container, .thoughts-wrapper, model-thoughts, .table-footer, message-actions, election-info-disclaimer, finance-info-disclaimer, .sources-list",
|
|
700
916
|
)
|
|
701
917
|
.forEach((el) => el.remove());
|
|
702
918
|
|
|
703
|
-
// Unwrap response-element wrappers
|
|
704
|
-
clone
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
919
|
+
// Unwrap response-element wrappers and message-action guards
|
|
920
|
+
clone
|
|
921
|
+
.querySelectorAll("response-element, .hide-from-message-actions")
|
|
922
|
+
.forEach((el) => {
|
|
923
|
+
while (el.firstChild) {
|
|
924
|
+
el.parentNode.insertBefore(el.firstChild, el);
|
|
925
|
+
}
|
|
926
|
+
el.remove();
|
|
927
|
+
});
|
|
710
928
|
|
|
711
929
|
const text = convertToMarkdown(clone);
|
|
712
930
|
const trimmed = text.trim();
|
|
@@ -722,7 +940,27 @@ export class GeminiParser extends ChatParser {
|
|
|
722
940
|
});
|
|
723
941
|
}
|
|
724
942
|
|
|
725
|
-
// Strategy 2: Deep Research immersive panel
|
|
943
|
+
// Strategy 2: Deep Research immersive panel (full report document).
|
|
944
|
+
// Runs even when chat shells were found above — the panel holds the
|
|
945
|
+
// report body, which never appears in the chat transcript.
|
|
946
|
+
const immersiveSections = this.extractImmersivePanelMessages(document);
|
|
947
|
+
immersiveSections.forEach((section) => {
|
|
948
|
+
if (!section.content || seenTexts.has(section.content)) return;
|
|
949
|
+
// The panel body passes through a different conversion path than chat
|
|
950
|
+
// messages, so exact-match dedup never fires. Skip the section when a
|
|
951
|
+
// Model message already carries the report (e.g. panel content also
|
|
952
|
+
// rendered inside a chat model-response).
|
|
953
|
+
const body = section.content.replace(/^## .*\n\n/, "");
|
|
954
|
+
const probe = body.slice(0, 300);
|
|
955
|
+
const alreadyExported =
|
|
956
|
+
probe.length > 0 &&
|
|
957
|
+
messages.some((m) => m.role === "Model" && m.content.includes(probe));
|
|
958
|
+
if (!alreadyExported) {
|
|
959
|
+
seenTexts.add(section.content);
|
|
960
|
+
messages.push(section);
|
|
961
|
+
}
|
|
962
|
+
});
|
|
963
|
+
|
|
726
964
|
if (messages.length === 0) {
|
|
727
965
|
const deepResearchPanel = document.querySelector(
|
|
728
966
|
"deep-research-immersive-panel",
|
|
@@ -838,6 +1076,67 @@ export class GeminiParser extends ChatParser {
|
|
|
838
1076
|
return sections;
|
|
839
1077
|
}
|
|
840
1078
|
|
|
1079
|
+
// Extracts the open Deep Research immersive panel (the full report
|
|
1080
|
+
// document). Returns [] when no panel is rendered in the DOM.
|
|
1081
|
+
extractImmersivePanelMessages(doc) {
|
|
1082
|
+
const sections = [];
|
|
1083
|
+
try {
|
|
1084
|
+
if (!doc || typeof doc.querySelector !== "function") return sections;
|
|
1085
|
+
const panel =
|
|
1086
|
+
doc.querySelector("immersive-panel deep-research-immersive-panel") ||
|
|
1087
|
+
doc.querySelector("deep-research-immersive-panel");
|
|
1088
|
+
if (!panel) return sections;
|
|
1089
|
+
|
|
1090
|
+
const titleEl =
|
|
1091
|
+
panel.querySelector("toolbar .title-text") ||
|
|
1092
|
+
panel.querySelector(".title-text");
|
|
1093
|
+
const title = (titleEl?.textContent || "").trim();
|
|
1094
|
+
|
|
1095
|
+
const bodyRoot =
|
|
1096
|
+
panel.querySelector('[data-test-id="message-content"] .markdown') ||
|
|
1097
|
+
panel.querySelector("#extended-response-markdown-content") ||
|
|
1098
|
+
panel.querySelector("message-content .markdown") ||
|
|
1099
|
+
panel.querySelector("message-content");
|
|
1100
|
+
if (!bodyRoot) return sections;
|
|
1101
|
+
|
|
1102
|
+
const clone = bodyRoot.cloneNode(true);
|
|
1103
|
+
// Inline citation footnotes carry only a source index — render it as
|
|
1104
|
+
// text so references survive markdown conversion.
|
|
1105
|
+
clone.querySelectorAll("sup[data-turn-source-index]").forEach((sup) => {
|
|
1106
|
+
const idx = sup.getAttribute("data-turn-source-index");
|
|
1107
|
+
if (idx && sup.parentNode) {
|
|
1108
|
+
sup.parentNode.replaceChild(doc.createTextNode(`[${idx}]`), sup);
|
|
1109
|
+
}
|
|
1110
|
+
});
|
|
1111
|
+
clone
|
|
1112
|
+
.querySelectorAll(
|
|
1113
|
+
"button, toolbar, toc-menu, mat-menu, message-actions, follow-up, .follow-up-container, .hide-from-message-actions button",
|
|
1114
|
+
)
|
|
1115
|
+
.forEach((el) => el.remove());
|
|
1116
|
+
clone.querySelectorAll("response-element").forEach((el) => {
|
|
1117
|
+
while (el.firstChild) {
|
|
1118
|
+
el.parentNode.insertBefore(el.firstChild, el);
|
|
1119
|
+
}
|
|
1120
|
+
el.remove();
|
|
1121
|
+
});
|
|
1122
|
+
|
|
1123
|
+
const body = convertToMarkdown(clone)
|
|
1124
|
+
.trim()
|
|
1125
|
+
// Turndown escapes the [N] citation markers inserted above;
|
|
1126
|
+
// restore them (they render identically either way).
|
|
1127
|
+
.replace(/\\\[(\d+)\\\]/g, "[$1]");
|
|
1128
|
+
if (body && body.length > 100) {
|
|
1129
|
+
sections.push({
|
|
1130
|
+
role: "Model",
|
|
1131
|
+
content: title ? `## ${title}\n\n${body}` : body,
|
|
1132
|
+
});
|
|
1133
|
+
}
|
|
1134
|
+
} catch (error) {
|
|
1135
|
+
console.error("[Gemini Parser] Error extracting immersive panel:", error);
|
|
1136
|
+
}
|
|
1137
|
+
return sections;
|
|
1138
|
+
}
|
|
1139
|
+
|
|
841
1140
|
extractDeepResearchPanelContent(panelElement) {
|
|
842
1141
|
const sections = [];
|
|
843
1142
|
try {
|
package/ai/meta.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { ChatParser } from "./base.js";
|
|
2
2
|
import { convertToMarkdown } from "../utils/html-to-markdown.js";
|
|
3
|
+
import { pickTimestamp } from "../utils/timestamps.js";
|
|
3
4
|
|
|
4
5
|
// Observed Relay doc_ids for the conversation message list (Sept 2026).
|
|
5
6
|
// These rotate when Meta redeploys the web client. The parser tries the
|
|
@@ -139,12 +140,23 @@ export function formatMetaEdges(edges) {
|
|
|
139
140
|
});
|
|
140
141
|
const messages = [];
|
|
141
142
|
for (const node of nodes) {
|
|
143
|
+
// Turn granularity: user + assistant in the same turn share createdAt,
|
|
144
|
+
// so both messages carry the same timestamp (matches Perplexity pattern).
|
|
145
|
+
const timestamp = pickTimestamp(node, ["createdAt", "userCreatedAt"]);
|
|
142
146
|
if (node.__typename === "UserMessage") {
|
|
143
147
|
const content = extractMetaUserText(node);
|
|
144
|
-
if (content)
|
|
148
|
+
if (content) {
|
|
149
|
+
const msg = { role: "User", content };
|
|
150
|
+
if (timestamp) msg.timestamp = timestamp;
|
|
151
|
+
messages.push(msg);
|
|
152
|
+
}
|
|
145
153
|
} else if (node.__typename === "AssistantMessage") {
|
|
146
154
|
const content = extractMetaAssistantText(node);
|
|
147
|
-
if (content)
|
|
155
|
+
if (content) {
|
|
156
|
+
const msg = { role: "Meta AI", content };
|
|
157
|
+
if (timestamp) msg.timestamp = timestamp;
|
|
158
|
+
messages.push(msg);
|
|
159
|
+
}
|
|
148
160
|
}
|
|
149
161
|
}
|
|
150
162
|
return messages;
|
package/package.json
CHANGED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared per-message timestamp normalization (decant-core).
|
|
3
|
+
*
|
|
4
|
+
* Parsers attach `timestamp` as ISO strings, locale strings, or numeric
|
|
5
|
+
* epochs (seconds or milliseconds). Display layers (ace) format whatever
|
|
6
|
+
* string they receive, so this helper keeps ISO/locale strings as-is and
|
|
7
|
+
* converts numeric epochs to ISO.
|
|
8
|
+
*
|
|
9
|
+
* Turn-granularity sources (Perplexity entries, Meta edges, DeepSeek pairs)
|
|
10
|
+
* reuse the same timestamp for both prompt and response — callers pass the
|
|
11
|
+
* same value to both messages.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
/** Epoch values with abs < 1e11 are treated as seconds, else milliseconds. */
|
|
15
|
+
export function normalizeEpochToMs(value) {
|
|
16
|
+
if (Math.abs(value) < 1e11) return value * 1000;
|
|
17
|
+
return value;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Normalizes a raw timestamp to a display-ready string.
|
|
22
|
+
* @param {unknown} value ISO string, locale string, epoch number, or numeric string.
|
|
23
|
+
* @returns {string|null} ISO string for epochs, trimmed input for date strings, else null.
|
|
24
|
+
*/
|
|
25
|
+
export function normalizeTimestamp(value) {
|
|
26
|
+
if (typeof value === "number" && Number.isFinite(value)) {
|
|
27
|
+
try {
|
|
28
|
+
return new Date(normalizeEpochToMs(value)).toISOString();
|
|
29
|
+
} catch {
|
|
30
|
+
return null;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
if (typeof value === "string") {
|
|
34
|
+
const trimmed = value.trim();
|
|
35
|
+
if (!trimmed) return null;
|
|
36
|
+
if (/^[+-]?\d+(\.\d+)?$/.test(trimmed)) {
|
|
37
|
+
const numeric = Number(trimmed);
|
|
38
|
+
if (Number.isFinite(numeric)) {
|
|
39
|
+
try {
|
|
40
|
+
return new Date(normalizeEpochToMs(numeric)).toISOString();
|
|
41
|
+
} catch {
|
|
42
|
+
return null;
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
return null;
|
|
46
|
+
}
|
|
47
|
+
return trimmed;
|
|
48
|
+
}
|
|
49
|
+
return null;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Picks the first available timestamp from candidate fields.
|
|
54
|
+
* @param {object} source Parser payload object.
|
|
55
|
+
* @param {string[]} keys Field names to try in order.
|
|
56
|
+
* @returns {string|null}
|
|
57
|
+
*/
|
|
58
|
+
export function pickTimestamp(source, keys) {
|
|
59
|
+
if (!source || !Array.isArray(keys)) return null;
|
|
60
|
+
for (const key of keys) {
|
|
61
|
+
const normalized = normalizeTimestamp(source[key]);
|
|
62
|
+
if (normalized) return normalized;
|
|
63
|
+
}
|
|
64
|
+
return null;
|
|
65
|
+
}
|