decant-core 1.11.0 → 1.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/ai/gemini.js CHANGED
@@ -461,10 +461,15 @@ export class GeminiParser extends ChatParser {
461
461
  }
462
462
 
463
463
  const modelText = this.findModelTextInApiItem(item, options);
464
- if (modelText) {
464
+ const researchExtras = this.extractDeepResearchExtras(item);
465
+ const combined = [modelText, researchExtras]
466
+ .map((t) => this.stripChipPlaceholders(t))
467
+ .filter((t) => t && t.trim())
468
+ .join("\n\n");
469
+ if (combined.trim()) {
465
470
  messages.push({
466
471
  role: "Model",
467
- content: normalizeLatexMath(modelText.trim()),
472
+ content: normalizeLatexMath(combined.trim()),
468
473
  turnId,
469
474
  });
470
475
  }
@@ -473,6 +478,213 @@ export class GeminiParser extends ChatParser {
473
478
  return messages;
474
479
  }
475
480
 
481
+ // Deep-research turns render a short summary plus placeholder chip links
482
+ // (e.g. http://googleusercontent.com/immersive_entry_chip/0) whose real
483
+ // content — research plan, full report, citation map — lives in adjacent
484
+ // candidate slots. Only the known chip placeholders are stripped; every
485
+ // other URL (including googleusercontent subdomains hosting real images
486
+ // and links inside markdown) is left intact.
487
+ stripChipPlaceholders(text) {
488
+ if (typeof text !== "string" || !text) return text;
489
+ const chipPattern =
490
+ /<?https?:\/\/googleusercontent\.com\/(?:immersive_entry_chip|deep_research_confirmation_content)(?:\/\d*)?>?/;
491
+ const chipPatternGlobal = new RegExp(chipPattern.source, "g");
492
+ return text
493
+ .split("\n")
494
+ .filter((line) => {
495
+ if (!chipPattern.test(line)) return true;
496
+ return line.replace(chipPatternGlobal, "").trim() !== "";
497
+ })
498
+ .map((line) => line.replace(chipPatternGlobal, ""))
499
+ .join("\n")
500
+ .replace(/\n{3,}/g, "\n\n");
501
+ }
502
+
503
+ getApiCandidates(item) {
504
+ try {
505
+ if (!Array.isArray(item[3])) return [];
506
+ const candidates = Array.isArray(item[3][0]) ? item[3][0] : item[3];
507
+ return candidates.filter((cand) => Array.isArray(cand));
508
+ } catch {
509
+ return [];
510
+ }
511
+ }
512
+
513
+ buildDeepResearchCiteMap(citeGroups) {
514
+ const map = new Map();
515
+ try {
516
+ const groups = Array.isArray(citeGroups) ? citeGroups : [citeGroups];
517
+ for (const group of groups) {
518
+ if (!group || typeof group !== "object" || Array.isArray(group))
519
+ continue;
520
+ for (const entries of Object.values(group)) {
521
+ if (!Array.isArray(entries)) continue;
522
+ for (const entry of entries) {
523
+ if (!Array.isArray(entry) || !Array.isArray(entry[1])) continue;
524
+ for (const source of entry[1]) {
525
+ // Source shape: [null, null, null,
526
+ // [detail, number, ...]] where detail = [favicon, url, title].
527
+ if (!Array.isArray(source) || !Array.isArray(source[3])) continue;
528
+ const detail = source[3][0];
529
+ const url = Array.isArray(detail) ? detail[1] : null;
530
+ const number = source[3][1];
531
+ if (
532
+ typeof url === "string" &&
533
+ url.startsWith("http") &&
534
+ typeof number === "number" &&
535
+ !map.has(number)
536
+ ) {
537
+ map.set(number, {
538
+ url,
539
+ title:
540
+ typeof detail[2] === "string" && detail[2]
541
+ ? detail[2]
542
+ : url,
543
+ });
544
+ }
545
+ }
546
+ }
547
+ }
548
+ }
549
+ } catch {
550
+ // Ignore malformed citation maps
551
+ }
552
+ return map;
553
+ }
554
+
555
+ resolveDeepResearchCites(markdown, citeMap) {
556
+ if (typeof markdown !== "string" || !(citeMap instanceof Map)) {
557
+ return markdown;
558
+ }
559
+ return markdown.replace(/ ?\[cite: ([\d,\s]+)\]/g, (match, nums) => {
560
+ const numbers = [
561
+ ...new Set(
562
+ nums
563
+ .split(",")
564
+ .map((n) => parseInt(n.trim(), 10))
565
+ .filter((n) => Number.isFinite(n)),
566
+ ),
567
+ ];
568
+ if (numbers.length === 0) return "";
569
+ const links = numbers.map((n) => {
570
+ const cite = citeMap.get(n);
571
+ return cite ? `[[${n}]](${cite.url})` : `[${n}]`;
572
+ });
573
+ return ` ${links.join(" ")}`;
574
+ });
575
+ }
576
+
577
+ extractImmersiveDocFromCandidate(cand) {
578
+ try {
579
+ if (!Array.isArray(cand[30]) || cand[30].length === 0) return "";
580
+ const doc = cand[30][0];
581
+ if (!Array.isArray(doc)) return "";
582
+ // Guard: immersive research documents carry this task marker.
583
+ if (doc[3] !== "agency-placeholder-task-id") return "";
584
+ if (typeof doc[4] !== "string" || doc[4].trim().length < 100) return "";
585
+ const title = typeof doc[2] === "string" && doc[2] ? doc[2] : "Report";
586
+ const citeMap = this.buildDeepResearchCiteMap(doc[5]);
587
+ const markdown = this.resolveDeepResearchCites(doc[4].trim(), citeMap);
588
+ return `## ${title}\n\n${markdown}`;
589
+ } catch {
590
+ return "";
591
+ }
592
+ }
593
+
594
+ extractResearchPlanFromCandidate(cand) {
595
+ try {
596
+ if (!Array.isArray(cand[12])) return "";
597
+ for (const annotation of cand[12]) {
598
+ if (
599
+ !annotation ||
600
+ typeof annotation !== "object" ||
601
+ Array.isArray(annotation) ||
602
+ !Array.isArray(annotation["56"])
603
+ ) {
604
+ continue;
605
+ }
606
+ const [planTitle, steps] = annotation["56"];
607
+ if (!Array.isArray(steps) || steps.length === 0) continue;
608
+ const lines = steps.map((step, idx) => {
609
+ if (!Array.isArray(step)) return null;
610
+ const stepTitle = step[1] || `Step ${idx + 1}`;
611
+ const desc =
612
+ typeof step[2] === "string" && step[2].trim()
613
+ ? `: ${step[2].trim()}`
614
+ : "";
615
+ return `${idx + 1}. **${stepTitle}**${desc}`;
616
+ });
617
+ const valid = lines.filter(Boolean);
618
+ if (valid.length === 0) continue;
619
+ const heading =
620
+ typeof planTitle === "string" && planTitle
621
+ ? `### ${planTitle} — research plan`
622
+ : "### Research plan";
623
+ return `${heading}\n${valid.join("\n")}`;
624
+ }
625
+ } catch {
626
+ // Ignore malformed plan annotations
627
+ }
628
+ return "";
629
+ }
630
+
631
+ extractActivitySources(item, limit = 40) {
632
+ const seen = new Map();
633
+ try {
634
+ const trail = item[3]?.[4];
635
+ if (!Array.isArray(trail)) return [];
636
+ for (const entry of trail) {
637
+ const detail = entry?.[4]?.[2];
638
+ const url = Array.isArray(detail) ? detail[1] : null;
639
+ if (typeof url !== "string" || !url.startsWith("http")) continue;
640
+ if (seen.has(url)) continue;
641
+ const title =
642
+ typeof detail[2] === "string" && detail[2] ? detail[2] : url;
643
+ seen.set(url, title);
644
+ }
645
+ } catch {
646
+ // Ignore malformed activity trails
647
+ }
648
+ const all = [...seen.entries()];
649
+ const shown = all.slice(0, limit);
650
+ const lines = shown.map(([url, title]) => `- [${title}](${url})`);
651
+ if (all.length > shown.length) {
652
+ lines.push(`- …and ${all.length - shown.length} more`);
653
+ }
654
+ return lines;
655
+ }
656
+
657
+ extractDeepResearchExtras(item) {
658
+ const parts = [];
659
+ try {
660
+ // Extras must come from the same candidate that supplied the visible
661
+ // text (findModelTextInApiItem uses the first candidate with text),
662
+ // never mixed in from alternate drafts.
663
+ const candidates = this.getApiCandidates(item);
664
+ let anchorIdx = candidates.findIndex(
665
+ (cand) =>
666
+ (Array.isArray(cand[1]) && typeof cand[1][0] === "string") ||
667
+ typeof cand[1] === "string" ||
668
+ (typeof cand[0] === "string" && cand[0].length > 50),
669
+ );
670
+ if (anchorIdx === -1) anchorIdx = 0;
671
+ const anchor = candidates[anchorIdx];
672
+ if (anchor) {
673
+ const plan = this.extractResearchPlanFromCandidate(anchor);
674
+ if (plan) parts.push(plan);
675
+ const doc = this.extractImmersiveDocFromCandidate(anchor);
676
+ if (doc) parts.push(doc);
677
+ }
678
+ const sourceLines = this.extractActivitySources(item);
679
+ if (sourceLines.length > 0) {
680
+ parts.push(`**Sources consulted:**\n${sourceLines.join("\n")}`);
681
+ }
682
+ } catch {
683
+ // Never let research extras break the base message
684
+ }
685
+ return parts.filter(Boolean).join("\n\n");
686
+ }
687
+
476
688
  findUserTextInApiItem(item) {
477
689
  try {
478
690
  if (typeof item[2]?.[0]?.[0] === "string") return item[2][0][0];
@@ -693,20 +905,26 @@ export class GeminiParser extends ChatParser {
693
905
  if (markdownDiv) {
694
906
  const clone = markdownDiv.cloneNode(true);
695
907
 
696
- // Remove UI buttons, thought overlays, and interactive toolbars
908
+ // Remove UI buttons, thought overlays, follow-up suggestion
909
+ // widgets, and interactive toolbars.
910
+ // Note: .hide-from-message-actions is NOT removed — it wraps
911
+ // deep-research plan widgets whose text must be kept (buttons
912
+ // inside are still stripped above).
697
913
  clone
698
914
  .querySelectorAll(
699
- "button, .thoughts-container, .thoughts-wrapper, model-thoughts, .table-footer, .hide-from-message-actions, message-actions, election-info-disclaimer, finance-info-disclaimer, .sources-list",
915
+ "button, follow-up, .follow-up-container, .thoughts-container, .thoughts-wrapper, model-thoughts, .table-footer, message-actions, election-info-disclaimer, finance-info-disclaimer, .sources-list",
700
916
  )
701
917
  .forEach((el) => el.remove());
702
918
 
703
- // Unwrap response-element wrappers
704
- clone.querySelectorAll("response-element").forEach((el) => {
705
- while (el.firstChild) {
706
- el.parentNode.insertBefore(el.firstChild, el);
707
- }
708
- el.remove();
709
- });
919
+ // Unwrap response-element wrappers and message-action guards
920
+ clone
921
+ .querySelectorAll("response-element, .hide-from-message-actions")
922
+ .forEach((el) => {
923
+ while (el.firstChild) {
924
+ el.parentNode.insertBefore(el.firstChild, el);
925
+ }
926
+ el.remove();
927
+ });
710
928
 
711
929
  const text = convertToMarkdown(clone);
712
930
  const trimmed = text.trim();
@@ -722,7 +940,27 @@ export class GeminiParser extends ChatParser {
722
940
  });
723
941
  }
724
942
 
725
- // Strategy 2: Deep Research immersive panel structure fallback
943
+ // Strategy 2: Deep Research immersive panel (full report document).
944
+ // Runs even when chat shells were found above — the panel holds the
945
+ // report body, which never appears in the chat transcript.
946
+ const immersiveSections = this.extractImmersivePanelMessages(document);
947
+ immersiveSections.forEach((section) => {
948
+ if (!section.content || seenTexts.has(section.content)) return;
949
+ // The panel body passes through a different conversion path than chat
950
+ // messages, so exact-match dedup never fires. Skip the section when a
951
+ // Model message already carries the report (e.g. panel content also
952
+ // rendered inside a chat model-response).
953
+ const body = section.content.replace(/^## .*\n\n/, "");
954
+ const probe = body.slice(0, 300);
955
+ const alreadyExported =
956
+ probe.length > 0 &&
957
+ messages.some((m) => m.role === "Model" && m.content.includes(probe));
958
+ if (!alreadyExported) {
959
+ seenTexts.add(section.content);
960
+ messages.push(section);
961
+ }
962
+ });
963
+
726
964
  if (messages.length === 0) {
727
965
  const deepResearchPanel = document.querySelector(
728
966
  "deep-research-immersive-panel",
@@ -838,6 +1076,67 @@ export class GeminiParser extends ChatParser {
838
1076
  return sections;
839
1077
  }
840
1078
 
1079
+ // Extracts the open Deep Research immersive panel (the full report
1080
+ // document). Returns [] when no panel is rendered in the DOM.
1081
+ extractImmersivePanelMessages(doc) {
1082
+ const sections = [];
1083
+ try {
1084
+ if (!doc || typeof doc.querySelector !== "function") return sections;
1085
+ const panel =
1086
+ doc.querySelector("immersive-panel deep-research-immersive-panel") ||
1087
+ doc.querySelector("deep-research-immersive-panel");
1088
+ if (!panel) return sections;
1089
+
1090
+ const titleEl =
1091
+ panel.querySelector("toolbar .title-text") ||
1092
+ panel.querySelector(".title-text");
1093
+ const title = (titleEl?.textContent || "").trim();
1094
+
1095
+ const bodyRoot =
1096
+ panel.querySelector('[data-test-id="message-content"] .markdown') ||
1097
+ panel.querySelector("#extended-response-markdown-content") ||
1098
+ panel.querySelector("message-content .markdown") ||
1099
+ panel.querySelector("message-content");
1100
+ if (!bodyRoot) return sections;
1101
+
1102
+ const clone = bodyRoot.cloneNode(true);
1103
+ // Inline citation footnotes carry only a source index — render it as
1104
+ // text so references survive markdown conversion.
1105
+ clone.querySelectorAll("sup[data-turn-source-index]").forEach((sup) => {
1106
+ const idx = sup.getAttribute("data-turn-source-index");
1107
+ if (idx && sup.parentNode) {
1108
+ sup.parentNode.replaceChild(doc.createTextNode(`[${idx}]`), sup);
1109
+ }
1110
+ });
1111
+ clone
1112
+ .querySelectorAll(
1113
+ "button, toolbar, toc-menu, mat-menu, message-actions, follow-up, .follow-up-container, .hide-from-message-actions button",
1114
+ )
1115
+ .forEach((el) => el.remove());
1116
+ clone.querySelectorAll("response-element").forEach((el) => {
1117
+ while (el.firstChild) {
1118
+ el.parentNode.insertBefore(el.firstChild, el);
1119
+ }
1120
+ el.remove();
1121
+ });
1122
+
1123
+ const body = convertToMarkdown(clone)
1124
+ .trim()
1125
+ // Turndown escapes the [N] citation markers inserted above;
1126
+ // restore them (they render identically either way).
1127
+ .replace(/\\\[(\d+)\\\]/g, "[$1]");
1128
+ if (body && body.length > 100) {
1129
+ sections.push({
1130
+ role: "Model",
1131
+ content: title ? `## ${title}\n\n${body}` : body,
1132
+ });
1133
+ }
1134
+ } catch (error) {
1135
+ console.error("[Gemini Parser] Error extracting immersive panel:", error);
1136
+ }
1137
+ return sections;
1138
+ }
1139
+
841
1140
  extractDeepResearchPanelContent(panelElement) {
842
1141
  const sections = [];
843
1142
  try {
package/ai/z_ai.js CHANGED
@@ -60,13 +60,12 @@ export function formatZaiMessage(entry) {
60
60
  .filter(Boolean)
61
61
  .join("\n\n");
62
62
  if (reasoning) {
63
- const quoted = reasoning
64
- .split("\n")
65
- .map((line) => `> ${line}`)
66
- .join("\n");
67
- content += `\n\n> 🧠 Thinking\n${quoted}`;
63
+ content = `<think>\n${reasoning}\n</think>\n\n${content}`;
68
64
  }
69
65
  const msg = { role, content };
66
+ if (reasoning) {
67
+ msg.thinking = reasoning;
68
+ }
70
69
  if (entry.timestamp) {
71
70
  try {
72
71
  msg.timestamp = new Date(entry.timestamp * 1000).toISOString();
@@ -0,0 +1,162 @@
1
+ [
2
+ {
3
+ "id": "chatgpt",
4
+ "platform": "ChatGPT",
5
+ "parser": "ChatGPTParser",
6
+ "module": "decant-core/ai/chatgpt",
7
+ "strategy": "DOM + internal API",
8
+ "notes": "Fallback and RPC-assisted extraction; scroll/dedup helper for long threads."
9
+ },
10
+ {
11
+ "id": "claude",
12
+ "platform": "Claude",
13
+ "parser": "ClaudeParser",
14
+ "module": "decant-core/ai/claude",
15
+ "strategy": "DOM + internal API + React fiber",
16
+ "notes": "Internal API first with DOM fallback; reads the React tree for artifacts and structured blocks."
17
+ },
18
+ {
19
+ "id": "gemini",
20
+ "platform": "Google Gemini",
21
+ "parser": "GeminiParser",
22
+ "module": "decant-core/ai/gemini",
23
+ "strategy": "DOM + batchexecute RPC",
24
+ "notes": "Uses batchexecute RPC pagination with resilient DOM fallback."
25
+ },
26
+ {
27
+ "id": "copilot",
28
+ "platform": "Microsoft Copilot",
29
+ "parser": "CopilotParser",
30
+ "module": "decant-core/ai/copilot",
31
+ "strategy": "DOM",
32
+ "notes": "Multi-domain (bing + copilot) DOM extraction."
33
+ },
34
+ {
35
+ "id": "perplexity",
36
+ "platform": "Perplexity",
37
+ "parser": "PerplexityParser",
38
+ "module": "decant-core/ai/perplexity",
39
+ "strategy": "Internal API + DOM",
40
+ "notes": "Internal API first, DOM fallback for source citations and answers."
41
+ },
42
+ {
43
+ "id": "deepseek",
44
+ "platform": "DeepSeek",
45
+ "parser": "DeepSeekParser",
46
+ "module": "decant-core/ai/deepseek",
47
+ "strategy": "DOM + internal API",
48
+ "notes": "API-assisted parsing (fragments[]) with DOM fallback."
49
+ },
50
+ {
51
+ "id": "qwen",
52
+ "platform": "Qwen",
53
+ "parser": "QwenParser",
54
+ "module": "decant-core/ai/qwen",
55
+ "strategy": "DOM",
56
+ "notes": "Structured DOM extraction incl. file attachments."
57
+ },
58
+ {
59
+ "id": "meta",
60
+ "platform": "Meta AI",
61
+ "parser": "MetaParser",
62
+ "module": "decant-core/ai/meta",
63
+ "strategy": "Internal API + DOM",
64
+ "notes": "Internal GraphQL API (pinned + auto-resolved doc_ids) with DOM fallback."
65
+ },
66
+ {
67
+ "id": "mistral",
68
+ "platform": "Mistral / Le Chat",
69
+ "parser": "MistralParser",
70
+ "module": "decant-core/ai/mistral",
71
+ "strategy": "DOM",
72
+ "notes": "DOM extraction."
73
+ },
74
+ {
75
+ "id": "lumo",
76
+ "platform": "Proton Lumo",
77
+ "parser": "LumoParser",
78
+ "module": "decant-core/ai/lumo",
79
+ "strategy": "DOM",
80
+ "notes": "DOM only — API responses are E2E-encrypted and not readable."
81
+ },
82
+ {
83
+ "id": "z-ai",
84
+ "platform": "Z.ai",
85
+ "parser": "ZAiParser",
86
+ "module": "decant-core/ai/z_ai",
87
+ "strategy": "Internal API + DOM",
88
+ "notes": "Internal API (chat skeleton + batched bodies) with DOM fallback."
89
+ },
90
+ {
91
+ "id": "grok",
92
+ "platform": "Grok",
93
+ "parser": "GrokParser",
94
+ "module": "decant-core/ai/grok",
95
+ "strategy": "Internal API + DOM",
96
+ "notes": "Internal API (response-node ordering + load-responses bodies) with DOM fallback."
97
+ },
98
+ {
99
+ "id": "google-ai-studio",
100
+ "platform": "Google AI Studio",
101
+ "parser": "GoogleAIStudioParser",
102
+ "module": "decant-core/ai/google_ai_studio",
103
+ "strategy": "DOM",
104
+ "notes": "DOM extraction for aistudio.google.com sessions."
105
+ },
106
+ {
107
+ "id": "notebooklm",
108
+ "platform": "NotebookLM",
109
+ "parser": "NotebookLMParser",
110
+ "module": "decant-core/ai/notebooklm",
111
+ "strategy": "DOM",
112
+ "notes": "DOM extraction incl. notes and citations."
113
+ },
114
+ {
115
+ "id": "google-search-ai",
116
+ "platform": "Google Search AI (AI Overviews)",
117
+ "parser": "GoogleSearchAIParser",
118
+ "module": "decant-core/ai/google_search_ai",
119
+ "strategy": "DOM",
120
+ "notes": "DOM extraction for SGE overviews."
121
+ },
122
+ {
123
+ "id": "gemini-cloud-assist",
124
+ "platform": "Gemini Cloud Assist",
125
+ "parser": "GeminiCloudAssistParser",
126
+ "module": "decant-core/ai/gemini_cloud_assist",
127
+ "strategy": "DOM",
128
+ "notes": "DOM extraction for console.cloud.google.com assistants."
129
+ },
130
+ {
131
+ "id": "joyland",
132
+ "platform": "Joyland",
133
+ "parser": "JoylandParser",
134
+ "module": "decant-core/ai/joyland",
135
+ "strategy": "DOM",
136
+ "notes": "DOM extraction for character chat platforms."
137
+ },
138
+ {
139
+ "id": "chub",
140
+ "platform": "Chub",
141
+ "parser": "ChubParser",
142
+ "module": "decant-core/ai/chub",
143
+ "strategy": "DOM",
144
+ "notes": "DOM extraction."
145
+ },
146
+ {
147
+ "id": "duck-ai",
148
+ "platform": "Duck.ai (DuckDuckGo AI)",
149
+ "parser": "DuckAIParser",
150
+ "module": "decant-core/ai/duck_ai",
151
+ "strategy": "DOM",
152
+ "notes": "DOM (client-side privacy, no server chat history API)."
153
+ },
154
+ {
155
+ "id": "article",
156
+ "platform": "Generic Web Article",
157
+ "parser": "ArticleParser",
158
+ "module": "decant-core/article",
159
+ "strategy": "Readability + Defuddle + Article-Extractor",
160
+ "notes": "Runs three extractors concurrently and arbitrates by content-quality scoring."
161
+ }
162
+ ]
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "decant-core",
3
- "version": "1.11.0",
3
+ "version": "1.13.0",
4
4
  "description": "A shared web extraction layer for AI conversations and regular web pages.",
5
5
  "type": "module",
6
6
  "engines": {
@@ -13,7 +13,8 @@
13
13
  "./web": "./web/index.js",
14
14
  "./web/*": "./web/*.js",
15
15
  "./article": "./web/article.js",
16
- "./detection/*": "./detection/*.js"
16
+ "./detection/*": "./detection/*.js",
17
+ "./platforms": "./data/platforms.json"
17
18
  },
18
19
  "files": [
19
20
  "ai",
@@ -21,6 +22,7 @@
21
22
  "detection",
22
23
  "lib",
23
24
  "utils",
25
+ "data",
24
26
  "LICENSE",
25
27
  "README.md"
26
28
  ],
@@ -36,7 +38,9 @@
36
38
  "test": "node --test tests/*.test.js",
37
39
  "lint": "eslint .",
38
40
  "format": "prettier --write .",
39
- "format:check": "prettier --check ."
41
+ "format:check": "prettier --check .",
42
+ "sync:platforms": "node scripts/sync-platforms.js",
43
+ "sync:check": "node scripts/sync-platforms.js --check"
40
44
  },
41
45
  "keywords": [
42
46
  "ai",