decant-core 1.12.0 → 1.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE CHANGED
@@ -1,3 +1,19 @@
1
+ ADDITIONAL PERMISSION UNDER GNU AGPL VERSION 3 SECTION 7:
2
+ FOSS LINKING EXCEPTION
3
+
4
+ Permission is granted to link, import, or bundle decant-core into projects
5
+ distributed under any OSI-approved open source license (including MPL-2.0, MIT,
6
+ Apache-2.0, and BSD) and distribute the resulting work under that project's
7
+ license, without requiring the enclosing project to be licensed under AGPLv3.
8
+ Any modifications directly made to decant-core source files remain subject to AGPLv3.
9
+
10
+ COMMERCIAL USE
11
+ If you wish to use decant-core in closed-source, proprietary, or commercial
12
+ software that cannot comply with the AGPLv3, a commercial license is available.
13
+ Please contact office@covai.org for licensing terms.
14
+
15
+ ==============================================================================
16
+
1
17
  GNU AFFERO GENERAL PUBLIC LICENSE
2
18
  Version 3, 19 November 2007
3
19
 
package/README.md CHANGED
@@ -151,9 +151,15 @@ normalized `parse()` contract. For the full extraction-strategy breakdown and ma
151
151
 
152
152
  ## License
153
153
 
154
- `decant-core` is licensed under the **GNU Affero General Public License v3.0 (AGPL-3.0-only)**.
154
+ `decant-core` is licensed under the **GNU Affero General Public License v3.0 (AGPL-3.0-only)** with a **FOSS Linking Exception**, alongside a **Commercial License** option.
155
155
 
156
- That choice is deliberate. AI platforms change constantly, and parser fixes belong in a shared commons so the whole ecosystem benefits — not siloed in a proprietary fork. If you use `decant-core`, network-based deployments that serve modified versions must also offer the corresponding source. Please review [`LICENSE`](LICENSE) before incorporating it into your project.
156
+ ### FOSS Linking Exception (Open Source)
157
+
158
+ Permission is granted to link, import, or bundle `decant-core` into projects distributed under any OSI-approved open source license (including **MPL-2.0**, **MIT**, **Apache-2.0**, and **BSD**) and distribute the resulting work under that project's license, without requiring the enclosing project to be licensed under AGPLv3. Any modifications directly made to `decant-core` source files remain subject to AGPLv3.
159
+
160
+ ### Commercial License
161
+
162
+ If you wish to use `decant-core` in closed-source, proprietary, or commercial software that cannot comply with the AGPLv3, a commercial license is available. Please contact `office@covai.org` for licensing terms.
157
163
 
158
164
  ### Third-Party Test Fixtures Notice
159
165
 
package/ai/chatgpt.js CHANGED
@@ -196,6 +196,221 @@ export function extractSharedConversationFromDom(
196
196
  return null;
197
197
  }
198
198
 
199
+ function cleanApiPartText(partText) {
200
+ return partText
201
+ .replace(/\u{E0000}[\u{E0000}-\u{E007F}]*/gu, "")
202
+ .replace(/citeturn\d+\w*/g, "")
203
+ .trim();
204
+ }
205
+
206
+ // Linearize the newer backend-api conversation shape, which returns a
207
+ // `messages` array instead of a `mapping` tree. Output segments match the
208
+ // mapping-based linearize() format ({ type: "text" | "thought" | "image" })
209
+ // so both flow through the same formatApiResult().
210
+ export function linearizeMessagesArray(apiMessages, includeImages) {
211
+ const messages = [];
212
+ if (!Array.isArray(apiMessages)) return messages;
213
+
214
+ const pushOrMerge = (entry, isThoughtMsg) => {
215
+ if (
216
+ entry.role === "ChatGPT" &&
217
+ messages.length > 0 &&
218
+ messages[messages.length - 1].role === "ChatGPT"
219
+ ) {
220
+ const prevMsg = messages[messages.length - 1];
221
+ if (isThoughtMsg) {
222
+ prevMsg.segments.unshift(...entry.segments);
223
+ } else {
224
+ prevMsg.segments.push(...entry.segments);
225
+ }
226
+ Object.assign(prevMsg.citeMap, entry.citeMap);
227
+ Object.assign(prevMsg.imageGroupMap, entry.imageGroupMap);
228
+ if (entry.timestamp && !prevMsg.timestamp) {
229
+ prevMsg.timestamp = entry.timestamp;
230
+ }
231
+ } else {
232
+ messages.push(entry);
233
+ }
234
+ };
235
+
236
+ for (const msg of apiMessages) {
237
+ if (!msg) continue;
238
+ const role = msg?.author?.role;
239
+ if (role !== "user" && role !== "assistant" && role !== "tool") continue;
240
+ if (msg.metadata?.is_visually_hidden_from_conversation === true) continue;
241
+
242
+ const content = msg.content || {};
243
+ const contentType = content.content_type;
244
+ const segments = [];
245
+ // Mirror the mapping path's thought detection (content types plus
246
+ // author/recipient markers).
247
+ const isThoughtMsg =
248
+ msg?.author?.name === "thought" ||
249
+ msg?.recipient === "thought" ||
250
+ contentType === "thought" ||
251
+ contentType === "thoughts" ||
252
+ contentType === "reasoning_recap" ||
253
+ msg?.metadata?.reasoning_status === "is_reasoning";
254
+
255
+ // Reasoning summaries: thoughts = [{ summary, content }]
256
+ if (Array.isArray(content.thoughts) && content.thoughts.length > 0) {
257
+ const thoughtParts = content.thoughts
258
+ .map((t) =>
259
+ t.summary ? `**${t.summary}**\n${t.content || ""}` : t.content || "",
260
+ )
261
+ .map((t) => t.trim())
262
+ .filter(Boolean);
263
+ if (thoughtParts.length > 0) {
264
+ segments.push({ type: "thought", content: thoughtParts.join("\n\n") });
265
+ }
266
+ }
267
+
268
+ // Reasoning recap (e.g. "Worked for 11s")
269
+ if (
270
+ contentType === "reasoning_recap" &&
271
+ typeof content.content === "string" &&
272
+ content.content.trim()
273
+ ) {
274
+ segments.push({ type: "thought", content: content.content.trim() });
275
+ }
276
+
277
+ // Tool invocations (content_type "code", e.g. Deep Research args JSON)
278
+ // and tool-role messages are not user-visible prose — skip them.
279
+ if (contentType !== "code" && role !== "tool") {
280
+ const parts = Array.isArray(content.parts) ? content.parts : [];
281
+ for (const part of parts) {
282
+ let partText = "";
283
+ let isThoughtPart = isThoughtMsg;
284
+ if (typeof part === "string") {
285
+ partText = part;
286
+ } else if (part && typeof part === "object") {
287
+ if (part.content_type === "text" && typeof part.text === "string") {
288
+ partText = part.text;
289
+ } else if (
290
+ part.content_type === "thought" &&
291
+ typeof part.text === "string"
292
+ ) {
293
+ partText = part.text;
294
+ isThoughtPart = true;
295
+ } else if (
296
+ part.content_type === "audio_transcription" &&
297
+ typeof part.text === "string"
298
+ ) {
299
+ partText = part.text;
300
+ } else if (
301
+ includeImages &&
302
+ part?.content_type === "image_asset_pointer" &&
303
+ part?.asset_pointer
304
+ ) {
305
+ segments.push({
306
+ type: "image",
307
+ fileId: part.asset_pointer.split("://")[1],
308
+ });
309
+ continue;
310
+ }
311
+ }
312
+ const text = partText ? cleanApiPartText(partText) : "";
313
+ if (text) {
314
+ segments.push({
315
+ type: isThoughtPart ? "thought" : "text",
316
+ content: text,
317
+ });
318
+ }
319
+ }
320
+
321
+ // Standalone content.text without parts (plain text / execution
322
+ // output), mirroring the mapping path.
323
+ if (
324
+ parts.length === 0 &&
325
+ typeof content.text === "string" &&
326
+ content.text.trim()
327
+ ) {
328
+ segments.push({ type: "text", content: content.text.trim() });
329
+ }
330
+ }
331
+
332
+ // Deep Research reports (widget_state), attachments, and Canvas
333
+ // documents from message metadata, mirroring the mapping path.
334
+ if (role !== "tool") {
335
+ const widgetRaw =
336
+ msg.metadata?.chatgpt_sdk?.widget_state ||
337
+ msg.metadata?.tool_response_metadata?.venus_widget_state;
338
+ if (widgetRaw) {
339
+ try {
340
+ const widget =
341
+ typeof widgetRaw === "string" ? JSON.parse(widgetRaw) : widgetRaw;
342
+ const reportText =
343
+ widget.report_message?.content?.parts?.[0] || widget.markdown;
344
+ const steering = widget.steering_acknowledgement;
345
+ let researchContent = "";
346
+ if (steering) researchContent += `${steering}\n\n`;
347
+ if (reportText) researchContent += reportText;
348
+ if (researchContent.trim()) {
349
+ segments.push({ type: "text", content: researchContent.trim() });
350
+ }
351
+ } catch {
352
+ // Ignore widget state JSON parse errors
353
+ }
354
+ }
355
+
356
+ if (
357
+ Array.isArray(msg.metadata?.attachments) &&
358
+ msg.metadata.attachments.length > 0
359
+ ) {
360
+ const fileNames = msg.metadata.attachments
361
+ .map((att) => att.name)
362
+ .filter(Boolean);
363
+ if (fileNames.length > 0) {
364
+ segments.push({
365
+ type: "text",
366
+ content: `[Attached: ${fileNames.join(", ")}]`,
367
+ });
368
+ }
369
+ }
370
+
371
+ if (msg.metadata?.canvas?.title) {
372
+ segments.push({
373
+ type: "text",
374
+ content: `[Canvas: ${msg.metadata.canvas.title}]`,
375
+ });
376
+ }
377
+ }
378
+
379
+ if (segments.length === 0) continue;
380
+
381
+ const displayRole = role === "user" ? "User" : "ChatGPT";
382
+ const timestamp = msg?.create_time
383
+ ? new Date(msg.create_time * 1000).toLocaleString()
384
+ : null;
385
+ // Citation / image-group references, mirroring the mapping path.
386
+ const citeMap = {};
387
+ const imageGroupMap = {};
388
+ for (const ref of msg?.metadata?.content_references ?? []) {
389
+ if (ref.matched_text) {
390
+ if (ref.items?.length) citeMap[ref.matched_text] = ref.items;
391
+ if (
392
+ ref.type === "image_group" ||
393
+ ref.matched_text.includes("image_group")
394
+ ) {
395
+ imageGroupMap[ref.matched_text] = ref;
396
+ }
397
+ }
398
+ }
399
+ pushOrMerge(
400
+ {
401
+ role: displayRole,
402
+ segments,
403
+ citeMap,
404
+ imageGroupMap,
405
+ timestamp,
406
+ },
407
+ isThoughtMsg,
408
+ );
409
+ }
410
+
411
+ return messages;
412
+ }
413
+
199
414
  export function linearize(mapping, includeImages, currentNodeId) {
200
415
  let path = [];
201
416
  const leafId = resolveActiveLeafNode(mapping, currentNodeId);
@@ -268,8 +483,20 @@ export function linearize(mapping, includeImages, currentNodeId) {
268
483
  ) {
269
484
  const segments = [];
270
485
  const parts = msg?.content?.parts ?? [];
486
+ // Tool-invocation payloads (content_type "code") are not user-visible
487
+ // prose, whether carried as standalone text or inside parts.
488
+ const isToolInvocation = msg?.content?.content_type === "code";
271
489
 
272
490
  for (const part of parts) {
491
+ if (isToolInvocation) {
492
+ if (!(
493
+ includeImages &&
494
+ part?.content_type === "image_asset_pointer" &&
495
+ part?.asset_pointer
496
+ )) {
497
+ continue;
498
+ }
499
+ }
273
500
  let partText = "";
274
501
  let isThoughtPart = isThoughtMsg;
275
502
 
@@ -316,11 +543,15 @@ export function linearize(mapping, includeImages, currentNodeId) {
316
543
  }
317
544
  }
318
545
 
319
- // Handle standalone content.text (e.g. execution_output or plain text)
546
+ // Handle standalone content.text (e.g. execution_output or plain text).
547
+ // Tool-invocation payloads (content_type "code", e.g. Deep Research
548
+ // "/Deep Research App/start" args JSON) are not user-visible prose —
549
+ // the site renders a status card instead — so never dump them as text.
320
550
  if (
321
551
  typeof msg.content?.text === "string" &&
322
552
  msg.content.text.trim() &&
323
- parts.length === 0
553
+ parts.length === 0 &&
554
+ msg.content?.content_type !== "code"
324
555
  ) {
325
556
  segments.push({ type: "text", content: msg.content.text.trim() });
326
557
  }
@@ -887,6 +1118,7 @@ export class ChatGPTParser extends ChatParser {
887
1118
  Link: currentUrl,
888
1119
  Model:
889
1120
  convoData?.model_slug ||
1121
+ convoData?.default_model_slug ||
890
1122
  (typeof document !== "undefined" && document.querySelector
891
1123
  ? document.querySelector('[data-testid="model-selector-dropdown"]')
892
1124
  ?.innerText
@@ -961,11 +1193,22 @@ export class ChatGPTParser extends ChatParser {
961
1193
  };
962
1194
  }
963
1195
 
964
- const apiMessages = linearize(
965
- result.data.mapping,
966
- includeImages,
967
- result.data.current_node,
968
- );
1196
+ // The backend returns either the legacy `mapping` tree or the newer
1197
+ // `messages` array shape — support both.
1198
+ let apiMessages = [];
1199
+ if (result.data.mapping) {
1200
+ apiMessages = linearize(
1201
+ result.data.mapping,
1202
+ includeImages,
1203
+ result.data.current_node,
1204
+ );
1205
+ }
1206
+ if (apiMessages.length === 0 && Array.isArray(result.data.messages)) {
1207
+ apiMessages = linearizeMessagesArray(
1208
+ result.data.messages,
1209
+ includeImages,
1210
+ );
1211
+ }
969
1212
  if (apiMessages.length > 0) {
970
1213
  return this.formatApiResult(
971
1214
  result.data,
@@ -84,16 +84,30 @@ if (!window.__chatgptHelperInjected) {
84
84
  let images = {};
85
85
  if (includeImages) {
86
86
  const fileIds = new Set();
87
- for (const node of Object.values(data.mapping)) {
87
+ const collectImagePointer = (part) => {
88
+ if (
89
+ part &&
90
+ part.content_type === "image_asset_pointer" &&
91
+ part.asset_pointer
92
+ ) {
93
+ fileIds.add(part.asset_pointer.split("://")[1]);
94
+ }
95
+ };
96
+ // Legacy `mapping` tree shape…
97
+ for (const node of Object.values(data.mapping || {})) {
88
98
  const msg = node.message;
89
99
  if (msg && msg.content && Array.isArray(msg.content.parts)) {
90
100
  for (const part of msg.content.parts) {
91
- if (
92
- part &&
93
- part.content_type === "image_asset_pointer" &&
94
- part.asset_pointer
95
- ) {
96
- fileIds.add(part.asset_pointer.split("://")[1]);
101
+ collectImagePointer(part);
102
+ }
103
+ }
104
+ }
105
+ // …and the newer `messages` array shape.
106
+ if (Array.isArray(data.messages)) {
107
+ for (const msg of data.messages) {
108
+ if (msg && msg.content && Array.isArray(msg.content.parts)) {
109
+ for (const part of msg.content.parts) {
110
+ collectImagePointer(part);
97
111
  }
98
112
  }
99
113
  }
package/ai/claude.js CHANGED
@@ -103,6 +103,124 @@ const MIME_TO_LANG = {
103
103
  "application/vnd.ant.code": "text",
104
104
  };
105
105
 
106
+ // Binary archives cannot be inlined as text in any export format; they are
107
+ // listed by name so they at least appear in markdown/html/json exports.
108
+ const BINARY_ARCHIVE_MIMES = new Set([
109
+ "application/x-tar",
110
+ "application/gzip",
111
+ "application/zip",
112
+ "application/x-7z-compressed",
113
+ "application/x-rar-compressed",
114
+ ]);
115
+
116
+ function formatFileSize(bytes) {
117
+ if (typeof bytes !== "number" || !Number.isFinite(bytes) || bytes < 0) {
118
+ return "";
119
+ }
120
+ if (bytes < 1024) return `${bytes} B`;
121
+ return `${(bytes / 1024).toFixed(1)} KB`;
122
+ }
123
+
124
+ function basenameOfPath(filePath) {
125
+ if (typeof filePath !== "string" || !filePath) return "file";
126
+ const base = filePath.split("/").pop();
127
+ return base || "file";
128
+ }
129
+
130
+ function isBinaryArchive(mimeType, filePath) {
131
+ if (mimeType && BINARY_ARCHIVE_MIMES.has(mimeType)) return true;
132
+ return /\.(tar\.gz|tgz|tar|zip|gz|7z|rar)$/i.test(filePath || "");
133
+ }
134
+
135
+ // Collect files surfaced via the `present_files` tool (File Creation
136
+ // integration). The tool_use block carries `input.filepaths`; the matching
137
+ // tool_result block (joined via tool_use_id) carries display names + mime
138
+ // types as `local_resource` entries. Either side may be missing, so resolve
139
+ // metadata when available and fall back to bare paths otherwise.
140
+ function collectPresentedFiles(branch) {
141
+ const pathsByToolUseId = new Map();
142
+ const resourcesByToolUseId = new Map();
143
+ for (const msg of branch) {
144
+ if (!Array.isArray(msg?.content)) continue;
145
+ for (const block of msg.content) {
146
+ if (block?.type === "tool_use" && block?.name === "present_files") {
147
+ const filepaths = block?.input?.filepaths;
148
+ if (Array.isArray(filepaths) && filepaths.length > 0 && block.id) {
149
+ pathsByToolUseId.set(block.id, filepaths);
150
+ }
151
+ } else if (
152
+ block?.type === "tool_result" &&
153
+ Array.isArray(block.content)
154
+ ) {
155
+ const resources = block.content.filter(
156
+ (item) => item && item.type === "local_resource" && item.file_path,
157
+ );
158
+ if (resources.length > 0) {
159
+ const toolUseId = block.tool_use_id || block.id;
160
+ if (toolUseId) resourcesByToolUseId.set(toolUseId, resources);
161
+ }
162
+ }
163
+ }
164
+ }
165
+
166
+ const filesByToolUseId = new Map();
167
+ const seenPaths = new Set();
168
+ const toEntry = (resource, fallbackPath) => {
169
+ const filePath = resource?.file_path || fallbackPath || "";
170
+ const name = resource?.name || basenameOfPath(filePath);
171
+ return {
172
+ key: resource?.uuid || filePath || `${name}`,
173
+ name,
174
+ path: filePath,
175
+ mime: resource?.mime_type || "",
176
+ };
177
+ };
178
+
179
+ for (const [toolUseId, resources] of resourcesByToolUseId.entries()) {
180
+ const entries = [];
181
+ for (const resource of resources) {
182
+ // Skip the inline "say they are below" helper text item (type: text).
183
+ if (!resource.file_path) continue;
184
+ if (seenPaths.has(resource.file_path)) continue;
185
+ seenPaths.add(resource.file_path);
186
+ entries.push(toEntry(resource));
187
+ }
188
+ if (entries.length > 0) filesByToolUseId.set(toolUseId, entries);
189
+ }
190
+
191
+ for (const [toolUseId, filepaths] of pathsByToolUseId.entries()) {
192
+ const existing = filesByToolUseId.get(toolUseId) || [];
193
+ const entries = [...existing];
194
+ for (const filePath of filepaths) {
195
+ if (typeof filePath !== "string" || !filePath) continue;
196
+ if (seenPaths.has(filePath)) continue;
197
+ seenPaths.add(filePath);
198
+ entries.push(toEntry(null, filePath));
199
+ }
200
+ if (entries.length > 0) filesByToolUseId.set(toolUseId, entries);
201
+ }
202
+
203
+ return filesByToolUseId;
204
+ }
205
+
206
+ function formatGeneratedFilesSection(files) {
207
+ if (!Array.isArray(files) || files.length === 0) return "";
208
+ const lines = files.map((file) => {
209
+ const displayName = file.name || basenameOfPath(file.path);
210
+ const meta = [];
211
+ if (file.mime) meta.push(file.mime);
212
+ if (isBinaryArchive(file.mime, file.path || displayName)) {
213
+ meta.push("binary archive — download from Claude UI");
214
+ }
215
+ const suffix = meta.length > 0 ? ` _(${meta.join(", ")})_` : "";
216
+ const pathSuffix =
217
+ file.path && file.path !== displayName ? ` — \`${file.path}\`` : "";
218
+ return `- \`${displayName}\`${suffix}${pathSuffix}`;
219
+ });
220
+ const label = files.length === 1 ? "Generated file:" : "Generated files:";
221
+ return `\n\n**${label}**\n${lines.join("\n")}\n\n`;
222
+ }
223
+
106
224
  function extractArtifactsFromText(text) {
107
225
  const artifactRegex = /<antArtifact[^>]*>([\s\S]*?)<\/antArtifact>/g;
108
226
  const artifacts = [];
@@ -267,6 +385,44 @@ function extractArtifacts(message, foldedArtifacts = new Map()) {
267
385
  return artifacts;
268
386
  }
269
387
 
388
+ // File cards rendered for generated files (markdown docs, tarballs, …).
389
+ // They carry no text content for convertToMarkdown, so extract their display
390
+ // names (aria-label="View <name>") and drop the nodes to avoid button noise.
391
+ function extractFileCards(root) {
392
+ if (!root || typeof root.querySelectorAll !== "function") return [];
393
+ const names = [];
394
+ const cards = root.querySelectorAll('[data-testid="file-card-open"]');
395
+ for (const card of cards) {
396
+ const label = card.getAttribute && card.getAttribute("aria-label");
397
+ const match = typeof label === "string" && label.match(/^View\s+(.+)$/i);
398
+ const name = match ? match[1].trim() : "";
399
+ // No name-based dedup: two cards may legitimately share a display name
400
+ // (same basename in different directories). Each card element is visited
401
+ // exactly once, so nothing is double-counted here.
402
+ if (name) {
403
+ names.push(name);
404
+ }
405
+ // Remove the whole card element so download buttons / type badges
406
+ // don't leak into the markdown conversion — but only when the parent
407
+ // is a dedicated card wrapper. If the button shares its parent with
408
+ // other prose, remove just the button to avoid deleting message text.
409
+ const parent =
410
+ card.parentElement && card.parentElement !== root
411
+ ? card.parentElement
412
+ : null;
413
+ const isDedicatedCard =
414
+ !!parent &&
415
+ parent.querySelectorAll('[data-testid="file-card-open"]').length === 1;
416
+ const target = isDedicatedCard ? parent : card;
417
+ if (target && typeof target.remove === "function") {
418
+ target.remove();
419
+ } else if (card.parentNode) {
420
+ card.parentNode.removeChild(card);
421
+ }
422
+ }
423
+ return names;
424
+ }
425
+
270
426
  function unrollInteractiveElements(root, doc) {
271
427
  if (!root || !doc) return;
272
428
 
@@ -371,6 +527,8 @@ export class ClaudeParser extends ChatParser {
371
527
 
372
528
  const branch = getCurrentBranch(data);
373
529
  const foldedArtifacts = collectArtifacts(branch);
530
+ const presentedFilesByToolUseId = collectPresentedFiles(branch);
531
+ const emittedFileKeys = new Set();
374
532
 
375
533
  const toolResultMap = new Map();
376
534
  for (const msg of branch) {
@@ -473,6 +631,37 @@ export class ClaudeParser extends ChatParser {
473
631
  : `> - ${qText}\n`;
474
632
  }
475
633
  contentStr += `${qStr}\n`;
634
+ } else if (
635
+ block.name === "present_files" &&
636
+ presentedFilesByToolUseId.has(block.id)
637
+ ) {
638
+ // Files surfaced via the File Creation integration
639
+ // (markdown docs, tarball, …). Without this they are
640
+ // silently dropped from every export format.
641
+ const files = presentedFilesByToolUseId.get(block.id);
642
+ const fresh = files.filter(
643
+ (file) => !emittedFileKeys.has(file.key),
644
+ );
645
+ fresh.forEach((file) => emittedFileKeys.add(file.key));
646
+ if (fresh.length > 0) {
647
+ contentStr += formatGeneratedFilesSection(fresh);
648
+ }
649
+ }
650
+ } else if (
651
+ block.type === "tool_result" &&
652
+ Array.isArray(block.content)
653
+ ) {
654
+ // Orphan file presentation: the matching tool_use block may
655
+ // sit outside the current branch, so emit unseen
656
+ // local_resource entries directly from the result.
657
+ const toolUseId = block.tool_use_id || block.id;
658
+ const files = presentedFilesByToolUseId.get(toolUseId) || [];
659
+ const fresh = files.filter(
660
+ (file) => !emittedFileKeys.has(file.key),
661
+ );
662
+ fresh.forEach((file) => emittedFileKeys.add(file.key));
663
+ if (fresh.length > 0) {
664
+ contentStr += formatGeneratedFilesSection(fresh);
476
665
  }
477
666
  }
478
667
  }
@@ -491,8 +680,9 @@ export class ClaudeParser extends ChatParser {
491
680
  if (attachment.file_name) {
492
681
  let header = `### Attachment: ${attachment.file_name}`;
493
682
  const meta = [];
494
- if (attachment.file_size) {
495
- meta.push(`${(attachment.file_size / 1024).toFixed(1)} KB`);
683
+ const sizeLabel = formatFileSize(attachment.file_size);
684
+ if (sizeLabel) {
685
+ meta.push(sizeLabel);
496
686
  }
497
687
  if (attachment.file_type) {
498
688
  meta.push(attachment.file_type);
@@ -505,7 +695,9 @@ export class ClaudeParser extends ChatParser {
505
695
  contentStr += `\`\`\`\`\n${attachment.extracted_content}\n\`\`\`\`\n\n`;
506
696
  }
507
697
  } else if (attachment.extracted_content) {
508
- contentStr += `\n\n### Pasted\n\`\`\`\`\n${attachment.extracted_content}\n\`\`\`\`\n\n`;
698
+ const sizeLabel = formatFileSize(attachment.file_size);
699
+ const sizeSuffix = sizeLabel ? ` _(${sizeLabel})_` : "";
700
+ contentStr += `\n\n### Pasted content${sizeSuffix}\n\`\`\`\`\n${attachment.extracted_content}\n\`\`\`\`\n\n`;
509
701
  }
510
702
  }
511
703
  }
@@ -678,8 +870,16 @@ export class ClaudeParser extends ChatParser {
678
870
  ) {
679
871
  role = "Claude";
680
872
  const clone = el.cloneNode(true);
873
+ // Extract file cards first: unrollInteractiveElements strips all
874
+ // buttons, which would destroy the card markers.
875
+ const fileCardNames = extractFileCards(clone);
681
876
  unrollInteractiveElements(clone, el.ownerDocument || document);
682
877
  content = convertToMarkdown(clone);
878
+ if (fileCardNames.length > 0) {
879
+ content += formatGeneratedFilesSection(
880
+ fileCardNames.map((name) => ({ name, path: "", mime: "" })),
881
+ );
882
+ }
683
883
  } else if (el.matches(".artifact-block-cell")) {
684
884
  role = "Claude Artifact";
685
885
 
package/ai/gemini.js CHANGED
@@ -461,10 +461,15 @@ export class GeminiParser extends ChatParser {
461
461
  }
462
462
 
463
463
  const modelText = this.findModelTextInApiItem(item, options);
464
- if (modelText) {
464
+ const researchExtras = this.extractDeepResearchExtras(item);
465
+ const combined = [modelText, researchExtras]
466
+ .map((t) => this.stripChipPlaceholders(t))
467
+ .filter((t) => t && t.trim())
468
+ .join("\n\n");
469
+ if (combined.trim()) {
465
470
  messages.push({
466
471
  role: "Model",
467
- content: normalizeLatexMath(modelText.trim()),
472
+ content: normalizeLatexMath(combined.trim()),
468
473
  turnId,
469
474
  });
470
475
  }
@@ -473,6 +478,213 @@ export class GeminiParser extends ChatParser {
473
478
  return messages;
474
479
  }
475
480
 
481
+ // Deep-research turns render a short summary plus placeholder chip links
482
+ // (e.g. http://googleusercontent.com/immersive_entry_chip/0) whose real
483
+ // content — research plan, full report, citation map — lives in adjacent
484
+ // candidate slots. Only the known chip placeholders are stripped; every
485
+ // other URL (including googleusercontent subdomains hosting real images
486
+ // and links inside markdown) is left intact.
487
+ stripChipPlaceholders(text) {
488
+ if (typeof text !== "string" || !text) return text;
489
+ const chipPattern =
490
+ /<?https?:\/\/googleusercontent\.com\/(?:immersive_entry_chip|deep_research_confirmation_content)(?:\/\d*)?>?/;
491
+ const chipPatternGlobal = new RegExp(chipPattern.source, "g");
492
+ return text
493
+ .split("\n")
494
+ .filter((line) => {
495
+ if (!chipPattern.test(line)) return true;
496
+ return line.replace(chipPatternGlobal, "").trim() !== "";
497
+ })
498
+ .map((line) => line.replace(chipPatternGlobal, ""))
499
+ .join("\n")
500
+ .replace(/\n{3,}/g, "\n\n");
501
+ }
502
+
503
+ getApiCandidates(item) {
504
+ try {
505
+ if (!Array.isArray(item[3])) return [];
506
+ const candidates = Array.isArray(item[3][0]) ? item[3][0] : item[3];
507
+ return candidates.filter((cand) => Array.isArray(cand));
508
+ } catch {
509
+ return [];
510
+ }
511
+ }
512
+
513
+ buildDeepResearchCiteMap(citeGroups) {
514
+ const map = new Map();
515
+ try {
516
+ const groups = Array.isArray(citeGroups) ? citeGroups : [citeGroups];
517
+ for (const group of groups) {
518
+ if (!group || typeof group !== "object" || Array.isArray(group))
519
+ continue;
520
+ for (const entries of Object.values(group)) {
521
+ if (!Array.isArray(entries)) continue;
522
+ for (const entry of entries) {
523
+ if (!Array.isArray(entry) || !Array.isArray(entry[1])) continue;
524
+ for (const source of entry[1]) {
525
+ // Source shape: [null, null, null,
526
+ // [detail, number, ...]] where detail = [favicon, url, title].
527
+ if (!Array.isArray(source) || !Array.isArray(source[3])) continue;
528
+ const detail = source[3][0];
529
+ const url = Array.isArray(detail) ? detail[1] : null;
530
+ const number = source[3][1];
531
+ if (
532
+ typeof url === "string" &&
533
+ url.startsWith("http") &&
534
+ typeof number === "number" &&
535
+ !map.has(number)
536
+ ) {
537
+ map.set(number, {
538
+ url,
539
+ title:
540
+ typeof detail[2] === "string" && detail[2]
541
+ ? detail[2]
542
+ : url,
543
+ });
544
+ }
545
+ }
546
+ }
547
+ }
548
+ }
549
+ } catch {
550
+ // Ignore malformed citation maps
551
+ }
552
+ return map;
553
+ }
554
+
555
+ resolveDeepResearchCites(markdown, citeMap) {
556
+ if (typeof markdown !== "string" || !(citeMap instanceof Map)) {
557
+ return markdown;
558
+ }
559
+ return markdown.replace(/ ?\[cite: ([\d,\s]+)\]/g, (match, nums) => {
560
+ const numbers = [
561
+ ...new Set(
562
+ nums
563
+ .split(",")
564
+ .map((n) => parseInt(n.trim(), 10))
565
+ .filter((n) => Number.isFinite(n)),
566
+ ),
567
+ ];
568
+ if (numbers.length === 0) return "";
569
+ const links = numbers.map((n) => {
570
+ const cite = citeMap.get(n);
571
+ return cite ? `[[${n}]](${cite.url})` : `[${n}]`;
572
+ });
573
+ return ` ${links.join(" ")}`;
574
+ });
575
+ }
576
+
577
+ extractImmersiveDocFromCandidate(cand) {
578
+ try {
579
+ if (!Array.isArray(cand[30]) || cand[30].length === 0) return "";
580
+ const doc = cand[30][0];
581
+ if (!Array.isArray(doc)) return "";
582
+ // Guard: immersive research documents carry this task marker.
583
+ if (doc[3] !== "agency-placeholder-task-id") return "";
584
+ if (typeof doc[4] !== "string" || doc[4].trim().length < 100) return "";
585
+ const title = typeof doc[2] === "string" && doc[2] ? doc[2] : "Report";
586
+ const citeMap = this.buildDeepResearchCiteMap(doc[5]);
587
+ const markdown = this.resolveDeepResearchCites(doc[4].trim(), citeMap);
588
+ return `## ${title}\n\n${markdown}`;
589
+ } catch {
590
+ return "";
591
+ }
592
+ }
593
+
594
+ extractResearchPlanFromCandidate(cand) {
595
+ try {
596
+ if (!Array.isArray(cand[12])) return "";
597
+ for (const annotation of cand[12]) {
598
+ if (
599
+ !annotation ||
600
+ typeof annotation !== "object" ||
601
+ Array.isArray(annotation) ||
602
+ !Array.isArray(annotation["56"])
603
+ ) {
604
+ continue;
605
+ }
606
+ const [planTitle, steps] = annotation["56"];
607
+ if (!Array.isArray(steps) || steps.length === 0) continue;
608
+ const lines = steps.map((step, idx) => {
609
+ if (!Array.isArray(step)) return null;
610
+ const stepTitle = step[1] || `Step ${idx + 1}`;
611
+ const desc =
612
+ typeof step[2] === "string" && step[2].trim()
613
+ ? `: ${step[2].trim()}`
614
+ : "";
615
+ return `${idx + 1}. **${stepTitle}**${desc}`;
616
+ });
617
+ const valid = lines.filter(Boolean);
618
+ if (valid.length === 0) continue;
619
+ const heading =
620
+ typeof planTitle === "string" && planTitle
621
+ ? `### ${planTitle} — research plan`
622
+ : "### Research plan";
623
+ return `${heading}\n${valid.join("\n")}`;
624
+ }
625
+ } catch {
626
+ // Ignore malformed plan annotations
627
+ }
628
+ return "";
629
+ }
630
+
631
+ extractActivitySources(item, limit = 40) {
632
+ const seen = new Map();
633
+ try {
634
+ const trail = item[3]?.[4];
635
+ if (!Array.isArray(trail)) return [];
636
+ for (const entry of trail) {
637
+ const detail = entry?.[4]?.[2];
638
+ const url = Array.isArray(detail) ? detail[1] : null;
639
+ if (typeof url !== "string" || !url.startsWith("http")) continue;
640
+ if (seen.has(url)) continue;
641
+ const title =
642
+ typeof detail[2] === "string" && detail[2] ? detail[2] : url;
643
+ seen.set(url, title);
644
+ }
645
+ } catch {
646
+ // Ignore malformed activity trails
647
+ }
648
+ const all = [...seen.entries()];
649
+ const shown = all.slice(0, limit);
650
+ const lines = shown.map(([url, title]) => `- [${title}](${url})`);
651
+ if (all.length > shown.length) {
652
+ lines.push(`- …and ${all.length - shown.length} more`);
653
+ }
654
+ return lines;
655
+ }
656
+
657
+ extractDeepResearchExtras(item) {
658
+ const parts = [];
659
+ try {
660
+ // Extras must come from the same candidate that supplied the visible
661
+ // text (findModelTextInApiItem uses the first candidate with text),
662
+ // never mixed in from alternate drafts.
663
+ const candidates = this.getApiCandidates(item);
664
+ let anchorIdx = candidates.findIndex(
665
+ (cand) =>
666
+ (Array.isArray(cand[1]) && typeof cand[1][0] === "string") ||
667
+ typeof cand[1] === "string" ||
668
+ (typeof cand[0] === "string" && cand[0].length > 50),
669
+ );
670
+ if (anchorIdx === -1) anchorIdx = 0;
671
+ const anchor = candidates[anchorIdx];
672
+ if (anchor) {
673
+ const plan = this.extractResearchPlanFromCandidate(anchor);
674
+ if (plan) parts.push(plan);
675
+ const doc = this.extractImmersiveDocFromCandidate(anchor);
676
+ if (doc) parts.push(doc);
677
+ }
678
+ const sourceLines = this.extractActivitySources(item);
679
+ if (sourceLines.length > 0) {
680
+ parts.push(`**Sources consulted:**\n${sourceLines.join("\n")}`);
681
+ }
682
+ } catch {
683
+ // Never let research extras break the base message
684
+ }
685
+ return parts.filter(Boolean).join("\n\n");
686
+ }
687
+
476
688
  findUserTextInApiItem(item) {
477
689
  try {
478
690
  if (typeof item[2]?.[0]?.[0] === "string") return item[2][0][0];
@@ -693,20 +905,26 @@ export class GeminiParser extends ChatParser {
693
905
  if (markdownDiv) {
694
906
  const clone = markdownDiv.cloneNode(true);
695
907
 
696
- // Remove UI buttons, thought overlays, and interactive toolbars
908
+ // Remove UI buttons, thought overlays, follow-up suggestion
909
+ // widgets, and interactive toolbars.
910
+ // Note: .hide-from-message-actions is NOT removed — it wraps
911
+ // deep-research plan widgets whose text must be kept (buttons
912
+ // inside are still stripped above).
697
913
  clone
698
914
  .querySelectorAll(
699
- "button, .thoughts-container, .thoughts-wrapper, model-thoughts, .table-footer, .hide-from-message-actions, message-actions, election-info-disclaimer, finance-info-disclaimer, .sources-list",
915
+ "button, follow-up, .follow-up-container, .thoughts-container, .thoughts-wrapper, model-thoughts, .table-footer, message-actions, election-info-disclaimer, finance-info-disclaimer, .sources-list",
700
916
  )
701
917
  .forEach((el) => el.remove());
702
918
 
703
- // Unwrap response-element wrappers
704
- clone.querySelectorAll("response-element").forEach((el) => {
705
- while (el.firstChild) {
706
- el.parentNode.insertBefore(el.firstChild, el);
707
- }
708
- el.remove();
709
- });
919
+ // Unwrap response-element wrappers and message-action guards
920
+ clone
921
+ .querySelectorAll("response-element, .hide-from-message-actions")
922
+ .forEach((el) => {
923
+ while (el.firstChild) {
924
+ el.parentNode.insertBefore(el.firstChild, el);
925
+ }
926
+ el.remove();
927
+ });
710
928
 
711
929
  const text = convertToMarkdown(clone);
712
930
  const trimmed = text.trim();
@@ -722,7 +940,27 @@ export class GeminiParser extends ChatParser {
722
940
  });
723
941
  }
724
942
 
725
- // Strategy 2: Deep Research immersive panel structure fallback
943
+ // Strategy 2: Deep Research immersive panel (full report document).
944
+ // Runs even when chat shells were found above — the panel holds the
945
+ // report body, which never appears in the chat transcript.
946
+ const immersiveSections = this.extractImmersivePanelMessages(document);
947
+ immersiveSections.forEach((section) => {
948
+ if (!section.content || seenTexts.has(section.content)) return;
949
+ // The panel body passes through a different conversion path than chat
950
+ // messages, so exact-match dedup never fires. Skip the section when a
951
+ // Model message already carries the report (e.g. panel content also
952
+ // rendered inside a chat model-response).
953
+ const body = section.content.replace(/^## .*\n\n/, "");
954
+ const probe = body.slice(0, 300);
955
+ const alreadyExported =
956
+ probe.length > 0 &&
957
+ messages.some((m) => m.role === "Model" && m.content.includes(probe));
958
+ if (!alreadyExported) {
959
+ seenTexts.add(section.content);
960
+ messages.push(section);
961
+ }
962
+ });
963
+
726
964
  if (messages.length === 0) {
727
965
  const deepResearchPanel = document.querySelector(
728
966
  "deep-research-immersive-panel",
@@ -838,6 +1076,67 @@ export class GeminiParser extends ChatParser {
838
1076
  return sections;
839
1077
  }
840
1078
 
1079
+ // Extracts the open Deep Research immersive panel (the full report
1080
+ // document). Returns [] when no panel is rendered in the DOM.
1081
+ extractImmersivePanelMessages(doc) {
1082
+ const sections = [];
1083
+ try {
1084
+ if (!doc || typeof doc.querySelector !== "function") return sections;
1085
+ const panel =
1086
+ doc.querySelector("immersive-panel deep-research-immersive-panel") ||
1087
+ doc.querySelector("deep-research-immersive-panel");
1088
+ if (!panel) return sections;
1089
+
1090
+ const titleEl =
1091
+ panel.querySelector("toolbar .title-text") ||
1092
+ panel.querySelector(".title-text");
1093
+ const title = (titleEl?.textContent || "").trim();
1094
+
1095
+ const bodyRoot =
1096
+ panel.querySelector('[data-test-id="message-content"] .markdown') ||
1097
+ panel.querySelector("#extended-response-markdown-content") ||
1098
+ panel.querySelector("message-content .markdown") ||
1099
+ panel.querySelector("message-content");
1100
+ if (!bodyRoot) return sections;
1101
+
1102
+ const clone = bodyRoot.cloneNode(true);
1103
+ // Inline citation footnotes carry only a source index — render it as
1104
+ // text so references survive markdown conversion.
1105
+ clone.querySelectorAll("sup[data-turn-source-index]").forEach((sup) => {
1106
+ const idx = sup.getAttribute("data-turn-source-index");
1107
+ if (idx && sup.parentNode) {
1108
+ sup.parentNode.replaceChild(doc.createTextNode(`[${idx}]`), sup);
1109
+ }
1110
+ });
1111
+ clone
1112
+ .querySelectorAll(
1113
+ "button, toolbar, toc-menu, mat-menu, message-actions, follow-up, .follow-up-container, .hide-from-message-actions button",
1114
+ )
1115
+ .forEach((el) => el.remove());
1116
+ clone.querySelectorAll("response-element").forEach((el) => {
1117
+ while (el.firstChild) {
1118
+ el.parentNode.insertBefore(el.firstChild, el);
1119
+ }
1120
+ el.remove();
1121
+ });
1122
+
1123
+ const body = convertToMarkdown(clone)
1124
+ .trim()
1125
+ // Turndown escapes the [N] citation markers inserted above;
1126
+ // restore them (they render identically either way).
1127
+ .replace(/\\\[(\d+)\\\]/g, "[$1]");
1128
+ if (body && body.length > 100) {
1129
+ sections.push({
1130
+ role: "Model",
1131
+ content: title ? `## ${title}\n\n${body}` : body,
1132
+ });
1133
+ }
1134
+ } catch (error) {
1135
+ console.error("[Gemini Parser] Error extracting immersive panel:", error);
1136
+ }
1137
+ return sections;
1138
+ }
1139
+
841
1140
  extractDeepResearchPanelContent(panelElement) {
842
1141
  const sections = [];
843
1142
  try {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "decant-core",
3
- "version": "1.12.0",
3
+ "version": "1.13.0",
4
4
  "description": "A shared web extraction layer for AI conversations and regular web pages.",
5
5
  "type": "module",
6
6
  "engines": {