decant-core 1.12.0 → 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE CHANGED
@@ -1,3 +1,19 @@
1
+ ADDITIONAL PERMISSION UNDER GNU AGPL VERSION 3 SECTION 7:
2
+ FOSS LINKING EXCEPTION
3
+
4
+ Permission is granted to link, import, or bundle decant-core into projects
5
+ distributed under any OSI-approved open source license (including MPL-2.0, MIT,
6
+ Apache-2.0, and BSD) and distribute the resulting work under that project's
7
+ license, without requiring the enclosing project to be licensed under AGPLv3.
8
+ Any modifications directly made to decant-core source files remain subject to AGPLv3.
9
+
10
+ COMMERCIAL USE
11
+ If you wish to use decant-core in closed-source, proprietary, or commercial
12
+ software that cannot comply with the AGPLv3, a commercial license is available.
13
+ Please contact office@covai.org for licensing terms.
14
+
15
+ ==============================================================================
16
+
1
17
  GNU AFFERO GENERAL PUBLIC LICENSE
2
18
  Version 3, 19 November 2007
3
19
 
package/README.md CHANGED
@@ -151,9 +151,15 @@ normalized `parse()` contract. For the full extraction-strategy breakdown and ma
151
151
 
152
152
  ## License
153
153
 
154
- `decant-core` is licensed under the **GNU Affero General Public License v3.0 (AGPL-3.0-only)**.
154
+ `decant-core` is licensed under the **GNU Affero General Public License v3.0 (AGPL-3.0-only)** with a **FOSS Linking Exception**, alongside a **Commercial License** option.
155
155
 
156
- That choice is deliberate. AI platforms change constantly, and parser fixes belong in a shared commons so the whole ecosystem benefits — not siloed in a proprietary fork. If you use `decant-core`, network-based deployments that serve modified versions must also offer the corresponding source. Please review [`LICENSE`](LICENSE) before incorporating it into your project.
156
+ ### FOSS Linking Exception (Open Source)
157
+
158
+ Permission is granted to link, import, or bundle `decant-core` into projects distributed under any OSI-approved open source license (including **MPL-2.0**, **MIT**, **Apache-2.0**, and **BSD**) and distribute the resulting work under that project's license, without requiring the enclosing project to be licensed under AGPLv3. Any modifications directly made to `decant-core` source files remain subject to AGPLv3.
159
+
160
+ ### Commercial License
161
+
162
+ If you wish to use `decant-core` in closed-source, proprietary, or commercial software that cannot comply with the AGPLv3, a commercial license is available. Please contact `office@covai.org` for licensing terms.
157
163
 
158
164
  ### Third-Party Test Fixtures Notice
159
165
 
package/ai/chatgpt.js CHANGED
@@ -196,6 +196,221 @@ export function extractSharedConversationFromDom(
196
196
  return null;
197
197
  }
198
198
 
199
+ function cleanApiPartText(partText) {
200
+ return partText
201
+ .replace(/\u{E0000}[\u{E0000}-\u{E007F}]*/gu, "")
202
+ .replace(/citeturn\d+\w*/g, "")
203
+ .trim();
204
+ }
205
+
206
+ // Linearize the newer backend-api conversation shape, which returns a
207
+ // `messages` array instead of a `mapping` tree. Output segments match the
208
+ // mapping-based linearize() format ({ type: "text" | "thought" | "image" })
209
+ // so both flow through the same formatApiResult().
210
+ export function linearizeMessagesArray(apiMessages, includeImages) {
211
+ const messages = [];
212
+ if (!Array.isArray(apiMessages)) return messages;
213
+
214
+ const pushOrMerge = (entry, isThoughtMsg) => {
215
+ if (
216
+ entry.role === "ChatGPT" &&
217
+ messages.length > 0 &&
218
+ messages[messages.length - 1].role === "ChatGPT"
219
+ ) {
220
+ const prevMsg = messages[messages.length - 1];
221
+ if (isThoughtMsg) {
222
+ prevMsg.segments.unshift(...entry.segments);
223
+ } else {
224
+ prevMsg.segments.push(...entry.segments);
225
+ }
226
+ Object.assign(prevMsg.citeMap, entry.citeMap);
227
+ Object.assign(prevMsg.imageGroupMap, entry.imageGroupMap);
228
+ if (entry.timestamp && !prevMsg.timestamp) {
229
+ prevMsg.timestamp = entry.timestamp;
230
+ }
231
+ } else {
232
+ messages.push(entry);
233
+ }
234
+ };
235
+
236
+ for (const msg of apiMessages) {
237
+ if (!msg) continue;
238
+ const role = msg?.author?.role;
239
+ if (role !== "user" && role !== "assistant" && role !== "tool") continue;
240
+ if (msg.metadata?.is_visually_hidden_from_conversation === true) continue;
241
+
242
+ const content = msg.content || {};
243
+ const contentType = content.content_type;
244
+ const segments = [];
245
+ // Mirror the mapping path's thought detection (content types plus
246
+ // author/recipient markers).
247
+ const isThoughtMsg =
248
+ msg?.author?.name === "thought" ||
249
+ msg?.recipient === "thought" ||
250
+ contentType === "thought" ||
251
+ contentType === "thoughts" ||
252
+ contentType === "reasoning_recap" ||
253
+ msg?.metadata?.reasoning_status === "is_reasoning";
254
+
255
+ // Reasoning summaries: thoughts = [{ summary, content }]
256
+ if (Array.isArray(content.thoughts) && content.thoughts.length > 0) {
257
+ const thoughtParts = content.thoughts
258
+ .map((t) =>
259
+ t.summary ? `**${t.summary}**\n${t.content || ""}` : t.content || "",
260
+ )
261
+ .map((t) => t.trim())
262
+ .filter(Boolean);
263
+ if (thoughtParts.length > 0) {
264
+ segments.push({ type: "thought", content: thoughtParts.join("\n\n") });
265
+ }
266
+ }
267
+
268
+ // Reasoning recap (e.g. "Worked for 11s")
269
+ if (
270
+ contentType === "reasoning_recap" &&
271
+ typeof content.content === "string" &&
272
+ content.content.trim()
273
+ ) {
274
+ segments.push({ type: "thought", content: content.content.trim() });
275
+ }
276
+
277
+ // Tool invocations (content_type "code", e.g. Deep Research args JSON)
278
+ // and tool-role messages are not user-visible prose — skip them.
279
+ if (contentType !== "code" && role !== "tool") {
280
+ const parts = Array.isArray(content.parts) ? content.parts : [];
281
+ for (const part of parts) {
282
+ let partText = "";
283
+ let isThoughtPart = isThoughtMsg;
284
+ if (typeof part === "string") {
285
+ partText = part;
286
+ } else if (part && typeof part === "object") {
287
+ if (part.content_type === "text" && typeof part.text === "string") {
288
+ partText = part.text;
289
+ } else if (
290
+ part.content_type === "thought" &&
291
+ typeof part.text === "string"
292
+ ) {
293
+ partText = part.text;
294
+ isThoughtPart = true;
295
+ } else if (
296
+ part.content_type === "audio_transcription" &&
297
+ typeof part.text === "string"
298
+ ) {
299
+ partText = part.text;
300
+ } else if (
301
+ includeImages &&
302
+ part?.content_type === "image_asset_pointer" &&
303
+ part?.asset_pointer
304
+ ) {
305
+ segments.push({
306
+ type: "image",
307
+ fileId: part.asset_pointer.split("://")[1],
308
+ });
309
+ continue;
310
+ }
311
+ }
312
+ const text = partText ? cleanApiPartText(partText) : "";
313
+ if (text) {
314
+ segments.push({
315
+ type: isThoughtPart ? "thought" : "text",
316
+ content: text,
317
+ });
318
+ }
319
+ }
320
+
321
+ // Standalone content.text without parts (plain text / execution
322
+ // output), mirroring the mapping path.
323
+ if (
324
+ parts.length === 0 &&
325
+ typeof content.text === "string" &&
326
+ content.text.trim()
327
+ ) {
328
+ segments.push({ type: "text", content: content.text.trim() });
329
+ }
330
+ }
331
+
332
+ // Deep Research reports (widget_state), attachments, and Canvas
333
+ // documents from message metadata, mirroring the mapping path.
334
+ if (role !== "tool") {
335
+ const widgetRaw =
336
+ msg.metadata?.chatgpt_sdk?.widget_state ||
337
+ msg.metadata?.tool_response_metadata?.venus_widget_state;
338
+ if (widgetRaw) {
339
+ try {
340
+ const widget =
341
+ typeof widgetRaw === "string" ? JSON.parse(widgetRaw) : widgetRaw;
342
+ const reportText =
343
+ widget.report_message?.content?.parts?.[0] || widget.markdown;
344
+ const steering = widget.steering_acknowledgement;
345
+ let researchContent = "";
346
+ if (steering) researchContent += `${steering}\n\n`;
347
+ if (reportText) researchContent += reportText;
348
+ if (researchContent.trim()) {
349
+ segments.push({ type: "text", content: researchContent.trim() });
350
+ }
351
+ } catch {
352
+ // Ignore widget state JSON parse errors
353
+ }
354
+ }
355
+
356
+ if (
357
+ Array.isArray(msg.metadata?.attachments) &&
358
+ msg.metadata.attachments.length > 0
359
+ ) {
360
+ const fileNames = msg.metadata.attachments
361
+ .map((att) => att.name)
362
+ .filter(Boolean);
363
+ if (fileNames.length > 0) {
364
+ segments.push({
365
+ type: "text",
366
+ content: `[Attached: ${fileNames.join(", ")}]`,
367
+ });
368
+ }
369
+ }
370
+
371
+ if (msg.metadata?.canvas?.title) {
372
+ segments.push({
373
+ type: "text",
374
+ content: `[Canvas: ${msg.metadata.canvas.title}]`,
375
+ });
376
+ }
377
+ }
378
+
379
+ if (segments.length === 0) continue;
380
+
381
+ const displayRole = role === "user" ? "User" : "ChatGPT";
382
+ const timestamp = msg?.create_time
383
+ ? new Date(msg.create_time * 1000).toLocaleString()
384
+ : null;
385
+ // Citation / image-group references, mirroring the mapping path.
386
+ const citeMap = {};
387
+ const imageGroupMap = {};
388
+ for (const ref of msg?.metadata?.content_references ?? []) {
389
+ if (ref.matched_text) {
390
+ if (ref.items?.length) citeMap[ref.matched_text] = ref.items;
391
+ if (
392
+ ref.type === "image_group" ||
393
+ ref.matched_text.includes("image_group")
394
+ ) {
395
+ imageGroupMap[ref.matched_text] = ref;
396
+ }
397
+ }
398
+ }
399
+ pushOrMerge(
400
+ {
401
+ role: displayRole,
402
+ segments,
403
+ citeMap,
404
+ imageGroupMap,
405
+ timestamp,
406
+ },
407
+ isThoughtMsg,
408
+ );
409
+ }
410
+
411
+ return messages;
412
+ }
413
+
199
414
  export function linearize(mapping, includeImages, currentNodeId) {
200
415
  let path = [];
201
416
  const leafId = resolveActiveLeafNode(mapping, currentNodeId);
@@ -268,8 +483,20 @@ export function linearize(mapping, includeImages, currentNodeId) {
268
483
  ) {
269
484
  const segments = [];
270
485
  const parts = msg?.content?.parts ?? [];
486
+ // Tool-invocation payloads (content_type "code") are not user-visible
487
+ // prose, whether carried as standalone text or inside parts.
488
+ const isToolInvocation = msg?.content?.content_type === "code";
271
489
 
272
490
  for (const part of parts) {
491
+ if (isToolInvocation) {
492
+ if (!(
493
+ includeImages &&
494
+ part?.content_type === "image_asset_pointer" &&
495
+ part?.asset_pointer
496
+ )) {
497
+ continue;
498
+ }
499
+ }
273
500
  let partText = "";
274
501
  let isThoughtPart = isThoughtMsg;
275
502
 
@@ -316,11 +543,15 @@ export function linearize(mapping, includeImages, currentNodeId) {
316
543
  }
317
544
  }
318
545
 
319
- // Handle standalone content.text (e.g. execution_output or plain text)
546
+ // Handle standalone content.text (e.g. execution_output or plain text).
547
+ // Tool-invocation payloads (content_type "code", e.g. Deep Research
548
+ // "/Deep Research App/start" args JSON) are not user-visible prose —
549
+ // the site renders a status card instead — so never dump them as text.
320
550
  if (
321
551
  typeof msg.content?.text === "string" &&
322
552
  msg.content.text.trim() &&
323
- parts.length === 0
553
+ parts.length === 0 &&
554
+ msg.content?.content_type !== "code"
324
555
  ) {
325
556
  segments.push({ type: "text", content: msg.content.text.trim() });
326
557
  }
@@ -887,6 +1118,7 @@ export class ChatGPTParser extends ChatParser {
887
1118
  Link: currentUrl,
888
1119
  Model:
889
1120
  convoData?.model_slug ||
1121
+ convoData?.default_model_slug ||
890
1122
  (typeof document !== "undefined" && document.querySelector
891
1123
  ? document.querySelector('[data-testid="model-selector-dropdown"]')
892
1124
  ?.innerText
@@ -961,11 +1193,22 @@ export class ChatGPTParser extends ChatParser {
961
1193
  };
962
1194
  }
963
1195
 
964
- const apiMessages = linearize(
965
- result.data.mapping,
966
- includeImages,
967
- result.data.current_node,
968
- );
1196
+ // The backend returns either the legacy `mapping` tree or the newer
1197
+ // `messages` array shape — support both.
1198
+ let apiMessages = [];
1199
+ if (result.data.mapping) {
1200
+ apiMessages = linearize(
1201
+ result.data.mapping,
1202
+ includeImages,
1203
+ result.data.current_node,
1204
+ );
1205
+ }
1206
+ if (apiMessages.length === 0 && Array.isArray(result.data.messages)) {
1207
+ apiMessages = linearizeMessagesArray(
1208
+ result.data.messages,
1209
+ includeImages,
1210
+ );
1211
+ }
969
1212
  if (apiMessages.length > 0) {
970
1213
  return this.formatApiResult(
971
1214
  result.data,
@@ -84,16 +84,30 @@ if (!window.__chatgptHelperInjected) {
84
84
  let images = {};
85
85
  if (includeImages) {
86
86
  const fileIds = new Set();
87
- for (const node of Object.values(data.mapping)) {
87
+ const collectImagePointer = (part) => {
88
+ if (
89
+ part &&
90
+ part.content_type === "image_asset_pointer" &&
91
+ part.asset_pointer
92
+ ) {
93
+ fileIds.add(part.asset_pointer.split("://")[1]);
94
+ }
95
+ };
96
+ // Legacy `mapping` tree shape…
97
+ for (const node of Object.values(data.mapping || {})) {
88
98
  const msg = node.message;
89
99
  if (msg && msg.content && Array.isArray(msg.content.parts)) {
90
100
  for (const part of msg.content.parts) {
91
- if (
92
- part &&
93
- part.content_type === "image_asset_pointer" &&
94
- part.asset_pointer
95
- ) {
96
- fileIds.add(part.asset_pointer.split("://")[1]);
101
+ collectImagePointer(part);
102
+ }
103
+ }
104
+ }
105
+ // …and the newer `messages` array shape.
106
+ if (Array.isArray(data.messages)) {
107
+ for (const msg of data.messages) {
108
+ if (msg && msg.content && Array.isArray(msg.content.parts)) {
109
+ for (const part of msg.content.parts) {
110
+ collectImagePointer(part);
97
111
  }
98
112
  }
99
113
  }
package/ai/claude.js CHANGED
@@ -1,5 +1,6 @@
1
1
  import { ChatParser } from "./base.js";
2
2
  import { convertToMarkdown } from "../utils/html-to-markdown.js";
3
+ import { pickTimestamp } from "../utils/timestamps.js";
3
4
 
4
5
  async function getOrganizationId() {
5
6
  try {
@@ -103,6 +104,124 @@ const MIME_TO_LANG = {
103
104
  "application/vnd.ant.code": "text",
104
105
  };
105
106
 
107
+ // Binary archives cannot be inlined as text in any export format; they are
108
+ // listed by name so they at least appear in markdown/html/json exports.
109
+ const BINARY_ARCHIVE_MIMES = new Set([
110
+ "application/x-tar",
111
+ "application/gzip",
112
+ "application/zip",
113
+ "application/x-7z-compressed",
114
+ "application/x-rar-compressed",
115
+ ]);
116
+
117
+ function formatFileSize(bytes) {
118
+ if (typeof bytes !== "number" || !Number.isFinite(bytes) || bytes < 0) {
119
+ return "";
120
+ }
121
+ if (bytes < 1024) return `${bytes} B`;
122
+ return `${(bytes / 1024).toFixed(1)} KB`;
123
+ }
124
+
125
+ function basenameOfPath(filePath) {
126
+ if (typeof filePath !== "string" || !filePath) return "file";
127
+ const base = filePath.split("/").pop();
128
+ return base || "file";
129
+ }
130
+
131
+ function isBinaryArchive(mimeType, filePath) {
132
+ if (mimeType && BINARY_ARCHIVE_MIMES.has(mimeType)) return true;
133
+ return /\.(tar\.gz|tgz|tar|zip|gz|7z|rar)$/i.test(filePath || "");
134
+ }
135
+
136
+ // Collect files surfaced via the `present_files` tool (File Creation
137
+ // integration). The tool_use block carries `input.filepaths`; the matching
138
+ // tool_result block (joined via tool_use_id) carries display names + mime
139
+ // types as `local_resource` entries. Either side may be missing, so resolve
140
+ // metadata when available and fall back to bare paths otherwise.
141
+ function collectPresentedFiles(branch) {
142
+ const pathsByToolUseId = new Map();
143
+ const resourcesByToolUseId = new Map();
144
+ for (const msg of branch) {
145
+ if (!Array.isArray(msg?.content)) continue;
146
+ for (const block of msg.content) {
147
+ if (block?.type === "tool_use" && block?.name === "present_files") {
148
+ const filepaths = block?.input?.filepaths;
149
+ if (Array.isArray(filepaths) && filepaths.length > 0 && block.id) {
150
+ pathsByToolUseId.set(block.id, filepaths);
151
+ }
152
+ } else if (
153
+ block?.type === "tool_result" &&
154
+ Array.isArray(block.content)
155
+ ) {
156
+ const resources = block.content.filter(
157
+ (item) => item && item.type === "local_resource" && item.file_path,
158
+ );
159
+ if (resources.length > 0) {
160
+ const toolUseId = block.tool_use_id || block.id;
161
+ if (toolUseId) resourcesByToolUseId.set(toolUseId, resources);
162
+ }
163
+ }
164
+ }
165
+ }
166
+
167
+ const filesByToolUseId = new Map();
168
+ const seenPaths = new Set();
169
+ const toEntry = (resource, fallbackPath) => {
170
+ const filePath = resource?.file_path || fallbackPath || "";
171
+ const name = resource?.name || basenameOfPath(filePath);
172
+ return {
173
+ key: resource?.uuid || filePath || `${name}`,
174
+ name,
175
+ path: filePath,
176
+ mime: resource?.mime_type || "",
177
+ };
178
+ };
179
+
180
+ for (const [toolUseId, resources] of resourcesByToolUseId.entries()) {
181
+ const entries = [];
182
+ for (const resource of resources) {
183
+ // Skip the inline "say they are below" helper text item (type: text).
184
+ if (!resource.file_path) continue;
185
+ if (seenPaths.has(resource.file_path)) continue;
186
+ seenPaths.add(resource.file_path);
187
+ entries.push(toEntry(resource));
188
+ }
189
+ if (entries.length > 0) filesByToolUseId.set(toolUseId, entries);
190
+ }
191
+
192
+ for (const [toolUseId, filepaths] of pathsByToolUseId.entries()) {
193
+ const existing = filesByToolUseId.get(toolUseId) || [];
194
+ const entries = [...existing];
195
+ for (const filePath of filepaths) {
196
+ if (typeof filePath !== "string" || !filePath) continue;
197
+ if (seenPaths.has(filePath)) continue;
198
+ seenPaths.add(filePath);
199
+ entries.push(toEntry(null, filePath));
200
+ }
201
+ if (entries.length > 0) filesByToolUseId.set(toolUseId, entries);
202
+ }
203
+
204
+ return filesByToolUseId;
205
+ }
206
+
207
+ function formatGeneratedFilesSection(files) {
208
+ if (!Array.isArray(files) || files.length === 0) return "";
209
+ const lines = files.map((file) => {
210
+ const displayName = file.name || basenameOfPath(file.path);
211
+ const meta = [];
212
+ if (file.mime) meta.push(file.mime);
213
+ if (isBinaryArchive(file.mime, file.path || displayName)) {
214
+ meta.push("binary archive — download from Claude UI");
215
+ }
216
+ const suffix = meta.length > 0 ? ` _(${meta.join(", ")})_` : "";
217
+ const pathSuffix =
218
+ file.path && file.path !== displayName ? ` — \`${file.path}\`` : "";
219
+ return `- \`${displayName}\`${suffix}${pathSuffix}`;
220
+ });
221
+ const label = files.length === 1 ? "Generated file:" : "Generated files:";
222
+ return `\n\n**${label}**\n${lines.join("\n")}\n\n`;
223
+ }
224
+
106
225
  function extractArtifactsFromText(text) {
107
226
  const artifactRegex = /<antArtifact[^>]*>([\s\S]*?)<\/antArtifact>/g;
108
227
  const artifacts = [];
@@ -267,6 +386,44 @@ function extractArtifacts(message, foldedArtifacts = new Map()) {
267
386
  return artifacts;
268
387
  }
269
388
 
389
+ // File cards rendered for generated files (markdown docs, tarballs, …).
390
+ // They carry no text content for convertToMarkdown, so extract their display
391
+ // names (aria-label="View <name>") and drop the nodes to avoid button noise.
392
+ function extractFileCards(root) {
393
+ if (!root || typeof root.querySelectorAll !== "function") return [];
394
+ const names = [];
395
+ const cards = root.querySelectorAll('[data-testid="file-card-open"]');
396
+ for (const card of cards) {
397
+ const label = card.getAttribute && card.getAttribute("aria-label");
398
+ const match = typeof label === "string" && label.match(/^View\s+(.+)$/i);
399
+ const name = match ? match[1].trim() : "";
400
+ // No name-based dedup: two cards may legitimately share a display name
401
+ // (same basename in different directories). Each card element is visited
402
+ // exactly once, so nothing is double-counted here.
403
+ if (name) {
404
+ names.push(name);
405
+ }
406
+ // Remove the whole card element so download buttons / type badges
407
+ // don't leak into the markdown conversion — but only when the parent
408
+ // is a dedicated card wrapper. If the button shares its parent with
409
+ // other prose, remove just the button to avoid deleting message text.
410
+ const parent =
411
+ card.parentElement && card.parentElement !== root
412
+ ? card.parentElement
413
+ : null;
414
+ const isDedicatedCard =
415
+ !!parent &&
416
+ parent.querySelectorAll('[data-testid="file-card-open"]').length === 1;
417
+ const target = isDedicatedCard ? parent : card;
418
+ if (target && typeof target.remove === "function") {
419
+ target.remove();
420
+ } else if (card.parentNode) {
421
+ card.parentNode.removeChild(card);
422
+ }
423
+ }
424
+ return names;
425
+ }
426
+
270
427
  function unrollInteractiveElements(root, doc) {
271
428
  if (!root || !doc) return;
272
429
 
@@ -371,6 +528,8 @@ export class ClaudeParser extends ChatParser {
371
528
 
372
529
  const branch = getCurrentBranch(data);
373
530
  const foldedArtifacts = collectArtifacts(branch);
531
+ const presentedFilesByToolUseId = collectPresentedFiles(branch);
532
+ const emittedFileKeys = new Set();
374
533
 
375
534
  const toolResultMap = new Map();
376
535
  for (const msg of branch) {
@@ -473,6 +632,37 @@ export class ClaudeParser extends ChatParser {
473
632
  : `> - ${qText}\n`;
474
633
  }
475
634
  contentStr += `${qStr}\n`;
635
+ } else if (
636
+ block.name === "present_files" &&
637
+ presentedFilesByToolUseId.has(block.id)
638
+ ) {
639
+ // Files surfaced via the File Creation integration
640
+ // (markdown docs, tarball, …). Without this they are
641
+ // silently dropped from every export format.
642
+ const files = presentedFilesByToolUseId.get(block.id);
643
+ const fresh = files.filter(
644
+ (file) => !emittedFileKeys.has(file.key),
645
+ );
646
+ fresh.forEach((file) => emittedFileKeys.add(file.key));
647
+ if (fresh.length > 0) {
648
+ contentStr += formatGeneratedFilesSection(fresh);
649
+ }
650
+ }
651
+ } else if (
652
+ block.type === "tool_result" &&
653
+ Array.isArray(block.content)
654
+ ) {
655
+ // Orphan file presentation: the matching tool_use block may
656
+ // sit outside the current branch, so emit unseen
657
+ // local_resource entries directly from the result.
658
+ const toolUseId = block.tool_use_id || block.id;
659
+ const files = presentedFilesByToolUseId.get(toolUseId) || [];
660
+ const fresh = files.filter(
661
+ (file) => !emittedFileKeys.has(file.key),
662
+ );
663
+ fresh.forEach((file) => emittedFileKeys.add(file.key));
664
+ if (fresh.length > 0) {
665
+ contentStr += formatGeneratedFilesSection(fresh);
476
666
  }
477
667
  }
478
668
  }
@@ -491,8 +681,9 @@ export class ClaudeParser extends ChatParser {
491
681
  if (attachment.file_name) {
492
682
  let header = `### Attachment: ${attachment.file_name}`;
493
683
  const meta = [];
494
- if (attachment.file_size) {
495
- meta.push(`${(attachment.file_size / 1024).toFixed(1)} KB`);
684
+ const sizeLabel = formatFileSize(attachment.file_size);
685
+ if (sizeLabel) {
686
+ meta.push(sizeLabel);
496
687
  }
497
688
  if (attachment.file_type) {
498
689
  meta.push(attachment.file_type);
@@ -505,7 +696,9 @@ export class ClaudeParser extends ChatParser {
505
696
  contentStr += `\`\`\`\`\n${attachment.extracted_content}\n\`\`\`\`\n\n`;
506
697
  }
507
698
  } else if (attachment.extracted_content) {
508
- contentStr += `\n\n### Pasted\n\`\`\`\`\n${attachment.extracted_content}\n\`\`\`\`\n\n`;
699
+ const sizeLabel = formatFileSize(attachment.file_size);
700
+ const sizeSuffix = sizeLabel ? ` _(${sizeLabel})_` : "";
701
+ contentStr += `\n\n### Pasted content${sizeSuffix}\n\`\`\`\`\n${attachment.extracted_content}\n\`\`\`\`\n\n`;
509
702
  }
510
703
  }
511
704
  }
@@ -544,6 +737,13 @@ export class ClaudeParser extends ChatParser {
544
737
  if (thinkingStr) {
545
738
  msgObj.thinking = thinkingStr;
546
739
  }
740
+ const timestamp = pickTimestamp(message, [
741
+ "created_at",
742
+ "updated_at",
743
+ ]);
744
+ if (timestamp) {
745
+ msgObj.timestamp = timestamp;
746
+ }
547
747
  messages.push(msgObj);
548
748
  }
549
749
 
@@ -568,6 +768,14 @@ export class ClaudeParser extends ChatParser {
568
768
  messages.push({
569
769
  role: "Claude Artifact",
570
770
  content: artContent.trim(),
771
+ ...(pickTimestamp(message, ["created_at", "updated_at"])
772
+ ? {
773
+ timestamp: pickTimestamp(message, [
774
+ "created_at",
775
+ "updated_at",
776
+ ]),
777
+ }
778
+ : {}),
571
779
  });
572
780
  }
573
781
  }
@@ -678,8 +886,16 @@ export class ClaudeParser extends ChatParser {
678
886
  ) {
679
887
  role = "Claude";
680
888
  const clone = el.cloneNode(true);
889
+ // Extract file cards first: unrollInteractiveElements strips all
890
+ // buttons, which would destroy the card markers.
891
+ const fileCardNames = extractFileCards(clone);
681
892
  unrollInteractiveElements(clone, el.ownerDocument || document);
682
893
  content = convertToMarkdown(clone);
894
+ if (fileCardNames.length > 0) {
895
+ content += formatGeneratedFilesSection(
896
+ fileCardNames.map((name) => ({ name, path: "", mime: "" })),
897
+ );
898
+ }
683
899
  } else if (el.matches(".artifact-block-cell")) {
684
900
  role = "Claude Artifact";
685
901
 
@@ -712,7 +928,36 @@ export class ClaudeParser extends ChatParser {
712
928
  }
713
929
 
714
930
  if (content) {
715
- messages.push({ role, content });
931
+ const msgObj = { role, content };
932
+ // Best-effort DOM timestamp. <time datetime> usually lives in the
933
+ // sibling MessageActions toolbar inside the same transcript row —
934
+ // not inside the message element itself. Scope to the closest row
935
+ // so we never borrow the previous/next turn's timestamp. Sparse in
936
+ // practice (hover-only on some turns) — absent stays absent.
937
+ try {
938
+ let timeEl =
939
+ typeof el.querySelector === "function"
940
+ ? el.querySelector("time[datetime]")
941
+ : null;
942
+ if (!timeEl && typeof el.closest === "function") {
943
+ // Narrow containers only: transcript-row holds one turn, so its
944
+ // time belongs to this message. Never fall back to broad
945
+ // containers like article (many turns) — wrong date is worse
946
+ // than no date.
947
+ const row = el.closest(
948
+ '[data-testid="transcript-row"], .group\\/message-row',
949
+ );
950
+ timeEl = row?.querySelector?.("time[datetime]") || null;
951
+ }
952
+ const datetime = timeEl?.getAttribute?.("datetime");
953
+ const timestamp = pickTimestamp({ datetime }, ["datetime"]);
954
+ if (timestamp) {
955
+ msgObj.timestamp = timestamp;
956
+ }
957
+ } catch {
958
+ // Ignore DOM timestamp lookup errors
959
+ }
960
+ messages.push(msgObj);
716
961
  }
717
962
  }
718
963
 
package/ai/deepseek.js CHANGED
@@ -1,5 +1,6 @@
1
1
  import { ChatParser } from "./base.js";
2
2
  import { convertToMarkdown } from "../utils/html-to-markdown.js";
3
+ import { normalizeTimestamp } from "../utils/timestamps.js";
3
4
 
4
5
  function getUserToken() {
5
6
  try {
@@ -120,6 +121,12 @@ async function fetchDeepSeekConversation(sessionId, token) {
120
121
  if (thinking) {
121
122
  msg.thinking = thinking;
122
123
  }
124
+ // Turn granularity: inserted_at is seconds-epoch, ~identical within a
125
+ // USER/ASSISTANT pair, so both sides share the turn's timestamp.
126
+ const timestamp = normalizeTimestamp(msgNode?.inserted_at);
127
+ if (timestamp) {
128
+ msg.timestamp = timestamp;
129
+ }
123
130
  return msg;
124
131
  })
125
132
  .filter((msg) => msg.content.length > 0);
package/ai/gemini.js CHANGED
@@ -461,10 +461,15 @@ export class GeminiParser extends ChatParser {
461
461
  }
462
462
 
463
463
  const modelText = this.findModelTextInApiItem(item, options);
464
- if (modelText) {
464
+ const researchExtras = this.extractDeepResearchExtras(item);
465
+ const combined = [modelText, researchExtras]
466
+ .map((t) => this.stripChipPlaceholders(t))
467
+ .filter((t) => t && t.trim())
468
+ .join("\n\n");
469
+ if (combined.trim()) {
465
470
  messages.push({
466
471
  role: "Model",
467
- content: normalizeLatexMath(modelText.trim()),
472
+ content: normalizeLatexMath(combined.trim()),
468
473
  turnId,
469
474
  });
470
475
  }
@@ -473,6 +478,213 @@ export class GeminiParser extends ChatParser {
473
478
  return messages;
474
479
  }
475
480
 
481
+ // Deep-research turns render a short summary plus placeholder chip links
482
+ // (e.g. http://googleusercontent.com/immersive_entry_chip/0) whose real
483
+ // content — research plan, full report, citation map — lives in adjacent
484
+ // candidate slots. Only the known chip placeholders are stripped; every
485
+ // other URL (including googleusercontent subdomains hosting real images
486
+ // and links inside markdown) is left intact.
487
+ stripChipPlaceholders(text) {
488
+ if (typeof text !== "string" || !text) return text;
489
+ const chipPattern =
490
+ /<?https?:\/\/googleusercontent\.com\/(?:immersive_entry_chip|deep_research_confirmation_content)(?:\/\d*)?>?/;
491
+ const chipPatternGlobal = new RegExp(chipPattern.source, "g");
492
+ return text
493
+ .split("\n")
494
+ .filter((line) => {
495
+ if (!chipPattern.test(line)) return true;
496
+ return line.replace(chipPatternGlobal, "").trim() !== "";
497
+ })
498
+ .map((line) => line.replace(chipPatternGlobal, ""))
499
+ .join("\n")
500
+ .replace(/\n{3,}/g, "\n\n");
501
+ }
502
+
503
+ getApiCandidates(item) {
504
+ try {
505
+ if (!Array.isArray(item[3])) return [];
506
+ const candidates = Array.isArray(item[3][0]) ? item[3][0] : item[3];
507
+ return candidates.filter((cand) => Array.isArray(cand));
508
+ } catch {
509
+ return [];
510
+ }
511
+ }
512
+
513
+ buildDeepResearchCiteMap(citeGroups) {
514
+ const map = new Map();
515
+ try {
516
+ const groups = Array.isArray(citeGroups) ? citeGroups : [citeGroups];
517
+ for (const group of groups) {
518
+ if (!group || typeof group !== "object" || Array.isArray(group))
519
+ continue;
520
+ for (const entries of Object.values(group)) {
521
+ if (!Array.isArray(entries)) continue;
522
+ for (const entry of entries) {
523
+ if (!Array.isArray(entry) || !Array.isArray(entry[1])) continue;
524
+ for (const source of entry[1]) {
525
+ // Source shape: [null, null, null,
526
+ // [detail, number, ...]] where detail = [favicon, url, title].
527
+ if (!Array.isArray(source) || !Array.isArray(source[3])) continue;
528
+ const detail = source[3][0];
529
+ const url = Array.isArray(detail) ? detail[1] : null;
530
+ const number = source[3][1];
531
+ if (
532
+ typeof url === "string" &&
533
+ url.startsWith("http") &&
534
+ typeof number === "number" &&
535
+ !map.has(number)
536
+ ) {
537
+ map.set(number, {
538
+ url,
539
+ title:
540
+ typeof detail[2] === "string" && detail[2]
541
+ ? detail[2]
542
+ : url,
543
+ });
544
+ }
545
+ }
546
+ }
547
+ }
548
+ }
549
+ } catch {
550
+ // Ignore malformed citation maps
551
+ }
552
+ return map;
553
+ }
554
+
555
+ resolveDeepResearchCites(markdown, citeMap) {
556
+ if (typeof markdown !== "string" || !(citeMap instanceof Map)) {
557
+ return markdown;
558
+ }
559
+ return markdown.replace(/ ?\[cite: ([\d,\s]+)\]/g, (match, nums) => {
560
+ const numbers = [
561
+ ...new Set(
562
+ nums
563
+ .split(",")
564
+ .map((n) => parseInt(n.trim(), 10))
565
+ .filter((n) => Number.isFinite(n)),
566
+ ),
567
+ ];
568
+ if (numbers.length === 0) return "";
569
+ const links = numbers.map((n) => {
570
+ const cite = citeMap.get(n);
571
+ return cite ? `[[${n}]](${cite.url})` : `[${n}]`;
572
+ });
573
+ return ` ${links.join(" ")}`;
574
+ });
575
+ }
576
+
577
+ extractImmersiveDocFromCandidate(cand) {
578
+ try {
579
+ if (!Array.isArray(cand[30]) || cand[30].length === 0) return "";
580
+ const doc = cand[30][0];
581
+ if (!Array.isArray(doc)) return "";
582
+ // Guard: immersive research documents carry this task marker.
583
+ if (doc[3] !== "agency-placeholder-task-id") return "";
584
+ if (typeof doc[4] !== "string" || doc[4].trim().length < 100) return "";
585
+ const title = typeof doc[2] === "string" && doc[2] ? doc[2] : "Report";
586
+ const citeMap = this.buildDeepResearchCiteMap(doc[5]);
587
+ const markdown = this.resolveDeepResearchCites(doc[4].trim(), citeMap);
588
+ return `## ${title}\n\n${markdown}`;
589
+ } catch {
590
+ return "";
591
+ }
592
+ }
593
+
594
+ extractResearchPlanFromCandidate(cand) {
595
+ try {
596
+ if (!Array.isArray(cand[12])) return "";
597
+ for (const annotation of cand[12]) {
598
+ if (
599
+ !annotation ||
600
+ typeof annotation !== "object" ||
601
+ Array.isArray(annotation) ||
602
+ !Array.isArray(annotation["56"])
603
+ ) {
604
+ continue;
605
+ }
606
+ const [planTitle, steps] = annotation["56"];
607
+ if (!Array.isArray(steps) || steps.length === 0) continue;
608
+ const lines = steps.map((step, idx) => {
609
+ if (!Array.isArray(step)) return null;
610
+ const stepTitle = step[1] || `Step ${idx + 1}`;
611
+ const desc =
612
+ typeof step[2] === "string" && step[2].trim()
613
+ ? `: ${step[2].trim()}`
614
+ : "";
615
+ return `${idx + 1}. **${stepTitle}**${desc}`;
616
+ });
617
+ const valid = lines.filter(Boolean);
618
+ if (valid.length === 0) continue;
619
+ const heading =
620
+ typeof planTitle === "string" && planTitle
621
+ ? `### ${planTitle} — research plan`
622
+ : "### Research plan";
623
+ return `${heading}\n${valid.join("\n")}`;
624
+ }
625
+ } catch {
626
+ // Ignore malformed plan annotations
627
+ }
628
+ return "";
629
+ }
630
+
631
+ extractActivitySources(item, limit = 40) {
632
+ const seen = new Map();
633
+ try {
634
+ const trail = item[3]?.[4];
635
+ if (!Array.isArray(trail)) return [];
636
+ for (const entry of trail) {
637
+ const detail = entry?.[4]?.[2];
638
+ const url = Array.isArray(detail) ? detail[1] : null;
639
+ if (typeof url !== "string" || !url.startsWith("http")) continue;
640
+ if (seen.has(url)) continue;
641
+ const title =
642
+ typeof detail[2] === "string" && detail[2] ? detail[2] : url;
643
+ seen.set(url, title);
644
+ }
645
+ } catch {
646
+ // Ignore malformed activity trails
647
+ }
648
+ const all = [...seen.entries()];
649
+ const shown = all.slice(0, limit);
650
+ const lines = shown.map(([url, title]) => `- [${title}](${url})`);
651
+ if (all.length > shown.length) {
652
+ lines.push(`- …and ${all.length - shown.length} more`);
653
+ }
654
+ return lines;
655
+ }
656
+
657
+ extractDeepResearchExtras(item) {
658
+ const parts = [];
659
+ try {
660
+ // Extras must come from the same candidate that supplied the visible
661
+ // text (findModelTextInApiItem uses the first candidate with text),
662
+ // never mixed in from alternate drafts.
663
+ const candidates = this.getApiCandidates(item);
664
+ let anchorIdx = candidates.findIndex(
665
+ (cand) =>
666
+ (Array.isArray(cand[1]) && typeof cand[1][0] === "string") ||
667
+ typeof cand[1] === "string" ||
668
+ (typeof cand[0] === "string" && cand[0].length > 50),
669
+ );
670
+ if (anchorIdx === -1) anchorIdx = 0;
671
+ const anchor = candidates[anchorIdx];
672
+ if (anchor) {
673
+ const plan = this.extractResearchPlanFromCandidate(anchor);
674
+ if (plan) parts.push(plan);
675
+ const doc = this.extractImmersiveDocFromCandidate(anchor);
676
+ if (doc) parts.push(doc);
677
+ }
678
+ const sourceLines = this.extractActivitySources(item);
679
+ if (sourceLines.length > 0) {
680
+ parts.push(`**Sources consulted:**\n${sourceLines.join("\n")}`);
681
+ }
682
+ } catch {
683
+ // Never let research extras break the base message
684
+ }
685
+ return parts.filter(Boolean).join("\n\n");
686
+ }
687
+
476
688
  findUserTextInApiItem(item) {
477
689
  try {
478
690
  if (typeof item[2]?.[0]?.[0] === "string") return item[2][0][0];
@@ -693,20 +905,26 @@ export class GeminiParser extends ChatParser {
693
905
  if (markdownDiv) {
694
906
  const clone = markdownDiv.cloneNode(true);
695
907
 
696
- // Remove UI buttons, thought overlays, and interactive toolbars
908
+ // Remove UI buttons, thought overlays, follow-up suggestion
909
+ // widgets, and interactive toolbars.
910
+ // Note: .hide-from-message-actions is NOT removed — it wraps
911
+ // deep-research plan widgets whose text must be kept (buttons
912
+ // inside are still stripped above).
697
913
  clone
698
914
  .querySelectorAll(
699
- "button, .thoughts-container, .thoughts-wrapper, model-thoughts, .table-footer, .hide-from-message-actions, message-actions, election-info-disclaimer, finance-info-disclaimer, .sources-list",
915
+ "button, follow-up, .follow-up-container, .thoughts-container, .thoughts-wrapper, model-thoughts, .table-footer, message-actions, election-info-disclaimer, finance-info-disclaimer, .sources-list",
700
916
  )
701
917
  .forEach((el) => el.remove());
702
918
 
703
- // Unwrap response-element wrappers
704
- clone.querySelectorAll("response-element").forEach((el) => {
705
- while (el.firstChild) {
706
- el.parentNode.insertBefore(el.firstChild, el);
707
- }
708
- el.remove();
709
- });
919
+ // Unwrap response-element wrappers and message-action guards
920
+ clone
921
+ .querySelectorAll("response-element, .hide-from-message-actions")
922
+ .forEach((el) => {
923
+ while (el.firstChild) {
924
+ el.parentNode.insertBefore(el.firstChild, el);
925
+ }
926
+ el.remove();
927
+ });
710
928
 
711
929
  const text = convertToMarkdown(clone);
712
930
  const trimmed = text.trim();
@@ -722,7 +940,27 @@ export class GeminiParser extends ChatParser {
722
940
  });
723
941
  }
724
942
 
725
- // Strategy 2: Deep Research immersive panel structure fallback
943
+ // Strategy 2: Deep Research immersive panel (full report document).
944
+ // Runs even when chat shells were found above — the panel holds the
945
+ // report body, which never appears in the chat transcript.
946
+ const immersiveSections = this.extractImmersivePanelMessages(document);
947
+ immersiveSections.forEach((section) => {
948
+ if (!section.content || seenTexts.has(section.content)) return;
949
+ // The panel body passes through a different conversion path than chat
950
+ // messages, so exact-match dedup never fires. Skip the section when a
951
+ // Model message already carries the report (e.g. panel content also
952
+ // rendered inside a chat model-response).
953
+ const body = section.content.replace(/^## .*\n\n/, "");
954
+ const probe = body.slice(0, 300);
955
+ const alreadyExported =
956
+ probe.length > 0 &&
957
+ messages.some((m) => m.role === "Model" && m.content.includes(probe));
958
+ if (!alreadyExported) {
959
+ seenTexts.add(section.content);
960
+ messages.push(section);
961
+ }
962
+ });
963
+
726
964
  if (messages.length === 0) {
727
965
  const deepResearchPanel = document.querySelector(
728
966
  "deep-research-immersive-panel",
@@ -838,6 +1076,67 @@ export class GeminiParser extends ChatParser {
838
1076
  return sections;
839
1077
  }
840
1078
 
1079
+ // Extracts the open Deep Research immersive panel (the full report
1080
+ // document). Returns [] when no panel is rendered in the DOM.
1081
+ extractImmersivePanelMessages(doc) {
1082
+ const sections = [];
1083
+ try {
1084
+ if (!doc || typeof doc.querySelector !== "function") return sections;
1085
+ const panel =
1086
+ doc.querySelector("immersive-panel deep-research-immersive-panel") ||
1087
+ doc.querySelector("deep-research-immersive-panel");
1088
+ if (!panel) return sections;
1089
+
1090
+ const titleEl =
1091
+ panel.querySelector("toolbar .title-text") ||
1092
+ panel.querySelector(".title-text");
1093
+ const title = (titleEl?.textContent || "").trim();
1094
+
1095
+ const bodyRoot =
1096
+ panel.querySelector('[data-test-id="message-content"] .markdown') ||
1097
+ panel.querySelector("#extended-response-markdown-content") ||
1098
+ panel.querySelector("message-content .markdown") ||
1099
+ panel.querySelector("message-content");
1100
+ if (!bodyRoot) return sections;
1101
+
1102
+ const clone = bodyRoot.cloneNode(true);
1103
+ // Inline citation footnotes carry only a source index — render it as
1104
+ // text so references survive markdown conversion.
1105
+ clone.querySelectorAll("sup[data-turn-source-index]").forEach((sup) => {
1106
+ const idx = sup.getAttribute("data-turn-source-index");
1107
+ if (idx && sup.parentNode) {
1108
+ sup.parentNode.replaceChild(doc.createTextNode(`[${idx}]`), sup);
1109
+ }
1110
+ });
1111
+ clone
1112
+ .querySelectorAll(
1113
+ "button, toolbar, toc-menu, mat-menu, message-actions, follow-up, .follow-up-container, .hide-from-message-actions button",
1114
+ )
1115
+ .forEach((el) => el.remove());
1116
+ clone.querySelectorAll("response-element").forEach((el) => {
1117
+ while (el.firstChild) {
1118
+ el.parentNode.insertBefore(el.firstChild, el);
1119
+ }
1120
+ el.remove();
1121
+ });
1122
+
1123
+ const body = convertToMarkdown(clone)
1124
+ .trim()
1125
+ // Turndown escapes the [N] citation markers inserted above;
1126
+ // restore them (they render identically either way).
1127
+ .replace(/\\\[(\d+)\\\]/g, "[$1]");
1128
+ if (body && body.length > 100) {
1129
+ sections.push({
1130
+ role: "Model",
1131
+ content: title ? `## ${title}\n\n${body}` : body,
1132
+ });
1133
+ }
1134
+ } catch (error) {
1135
+ console.error("[Gemini Parser] Error extracting immersive panel:", error);
1136
+ }
1137
+ return sections;
1138
+ }
1139
+
841
1140
  extractDeepResearchPanelContent(panelElement) {
842
1141
  const sections = [];
843
1142
  try {
package/ai/meta.js CHANGED
@@ -1,5 +1,6 @@
1
1
  import { ChatParser } from "./base.js";
2
2
  import { convertToMarkdown } from "../utils/html-to-markdown.js";
3
+ import { pickTimestamp } from "../utils/timestamps.js";
3
4
 
4
5
  // Observed Relay doc_ids for the conversation message list (Sept 2026).
5
6
  // These rotate when Meta redeploys the web client. The parser tries the
@@ -139,12 +140,23 @@ export function formatMetaEdges(edges) {
139
140
  });
140
141
  const messages = [];
141
142
  for (const node of nodes) {
143
+ // Turn granularity: user + assistant in the same turn share createdAt,
144
+ // so both messages carry the same timestamp (matches Perplexity pattern).
145
+ const timestamp = pickTimestamp(node, ["createdAt", "userCreatedAt"]);
142
146
  if (node.__typename === "UserMessage") {
143
147
  const content = extractMetaUserText(node);
144
- if (content) messages.push({ role: "User", content });
148
+ if (content) {
149
+ const msg = { role: "User", content };
150
+ if (timestamp) msg.timestamp = timestamp;
151
+ messages.push(msg);
152
+ }
145
153
  } else if (node.__typename === "AssistantMessage") {
146
154
  const content = extractMetaAssistantText(node);
147
- if (content) messages.push({ role: "Meta AI", content });
155
+ if (content) {
156
+ const msg = { role: "Meta AI", content };
157
+ if (timestamp) msg.timestamp = timestamp;
158
+ messages.push(msg);
159
+ }
148
160
  }
149
161
  }
150
162
  return messages;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "decant-core",
3
- "version": "1.12.0",
3
+ "version": "1.14.0",
4
4
  "description": "A shared web extraction layer for AI conversations and regular web pages.",
5
5
  "type": "module",
6
6
  "engines": {
@@ -0,0 +1,65 @@
1
+ /**
2
+ * Shared per-message timestamp normalization (decant-core).
3
+ *
4
+ * Parsers attach `timestamp` as ISO strings, locale strings, or numeric
5
+ * epochs (seconds or milliseconds). Display layers (ace) format whatever
6
+ * string they receive, so this helper keeps ISO/locale strings as-is and
7
+ * converts numeric epochs to ISO.
8
+ *
9
+ * Turn-granularity sources (Perplexity entries, Meta edges, DeepSeek pairs)
10
+ * reuse the same timestamp for both prompt and response — callers pass the
11
+ * same value to both messages.
12
+ */
13
+
14
+ /** Epoch values with abs < 1e11 are treated as seconds, else milliseconds. */
15
+ export function normalizeEpochToMs(value) {
16
+ if (Math.abs(value) < 1e11) return value * 1000;
17
+ return value;
18
+ }
19
+
20
+ /**
21
+ * Normalizes a raw timestamp to a display-ready string.
22
+ * @param {unknown} value ISO string, locale string, epoch number, or numeric string.
23
+ * @returns {string|null} ISO string for epochs, trimmed input for date strings, else null.
24
+ */
25
+ export function normalizeTimestamp(value) {
26
+ if (typeof value === "number" && Number.isFinite(value)) {
27
+ try {
28
+ return new Date(normalizeEpochToMs(value)).toISOString();
29
+ } catch {
30
+ return null;
31
+ }
32
+ }
33
+ if (typeof value === "string") {
34
+ const trimmed = value.trim();
35
+ if (!trimmed) return null;
36
+ if (/^[+-]?\d+(\.\d+)?$/.test(trimmed)) {
37
+ const numeric = Number(trimmed);
38
+ if (Number.isFinite(numeric)) {
39
+ try {
40
+ return new Date(normalizeEpochToMs(numeric)).toISOString();
41
+ } catch {
42
+ return null;
43
+ }
44
+ }
45
+ return null;
46
+ }
47
+ return trimmed;
48
+ }
49
+ return null;
50
+ }
51
+
52
+ /**
53
+ * Picks the first available timestamp from candidate fields.
54
+ * @param {object} source Parser payload object.
55
+ * @param {string[]} keys Field names to try in order.
56
+ * @returns {string|null}
57
+ */
58
+ export function pickTimestamp(source, keys) {
59
+ if (!source || !Array.isArray(keys)) return null;
60
+ for (const key of keys) {
61
+ const normalized = normalizeTimestamp(source[key]);
62
+ if (normalized) return normalized;
63
+ }
64
+ return null;
65
+ }