@memberjunction/core-actions 2.103.0 → 2.105.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/dist/config.d.ts +82 -0
  2. package/dist/config.d.ts.map +1 -0
  3. package/dist/config.js +83 -0
  4. package/dist/config.js.map +1 -0
  5. package/dist/custom/files/base-file-storage.action.d.ts +70 -0
  6. package/dist/custom/files/base-file-storage.action.d.ts.map +1 -0
  7. package/dist/custom/files/base-file-storage.action.js +140 -0
  8. package/dist/custom/files/base-file-storage.action.js.map +1 -0
  9. package/dist/custom/files/copy-object.action.d.ts +38 -0
  10. package/dist/custom/files/copy-object.action.d.ts.map +1 -0
  11. package/dist/custom/files/copy-object.action.js +84 -0
  12. package/dist/custom/files/copy-object.action.js.map +1 -0
  13. package/dist/custom/files/create-directory.action.d.ts +34 -0
  14. package/dist/custom/files/create-directory.action.d.ts.map +1 -0
  15. package/dist/custom/files/create-directory.action.js +75 -0
  16. package/dist/custom/files/create-directory.action.js.map +1 -0
  17. package/dist/custom/files/delete-directory.action.d.ts +38 -0
  18. package/dist/custom/files/delete-directory.action.d.ts.map +1 -0
  19. package/dist/custom/files/delete-directory.action.js +82 -0
  20. package/dist/custom/files/delete-directory.action.js.map +1 -0
  21. package/dist/custom/files/delete-object.action.d.ts +34 -0
  22. package/dist/custom/files/delete-object.action.d.ts.map +1 -0
  23. package/dist/custom/files/delete-object.action.js +75 -0
  24. package/dist/custom/files/delete-object.action.js.map +1 -0
  25. package/dist/custom/files/directory-exists.action.d.ts +34 -0
  26. package/dist/custom/files/directory-exists.action.d.ts.map +1 -0
  27. package/dist/custom/files/directory-exists.action.js +75 -0
  28. package/dist/custom/files/directory-exists.action.js.map +1 -0
  29. package/dist/custom/files/get-download-url.action.d.ts +34 -0
  30. package/dist/custom/files/get-download-url.action.d.ts.map +1 -0
  31. package/dist/custom/files/get-download-url.action.js +75 -0
  32. package/dist/custom/files/get-download-url.action.js.map +1 -0
  33. package/dist/custom/files/get-metadata.action.d.ts +42 -0
  34. package/dist/custom/files/get-metadata.action.d.ts.map +1 -0
  35. package/dist/custom/files/get-metadata.action.js +95 -0
  36. package/dist/custom/files/get-metadata.action.js.map +1 -0
  37. package/dist/custom/files/get-object.action.d.ts +35 -0
  38. package/dist/custom/files/get-object.action.d.ts.map +1 -0
  39. package/dist/custom/files/get-object.action.js +80 -0
  40. package/dist/custom/files/get-object.action.js.map +1 -0
  41. package/dist/custom/files/get-upload-url.action.d.ts +35 -0
  42. package/dist/custom/files/get-upload-url.action.d.ts.map +1 -0
  43. package/dist/custom/files/get-upload-url.action.js +80 -0
  44. package/dist/custom/files/get-upload-url.action.js.map +1 -0
  45. package/dist/custom/files/index.d.ts +23 -0
  46. package/dist/custom/files/index.d.ts.map +1 -0
  47. package/dist/custom/files/index.js +39 -0
  48. package/dist/custom/files/index.js.map +1 -0
  49. package/dist/custom/files/list-objects.action.d.ts +50 -0
  50. package/dist/custom/files/list-objects.action.d.ts.map +1 -0
  51. package/dist/custom/files/list-objects.action.js +95 -0
  52. package/dist/custom/files/list-objects.action.js.map +1 -0
  53. package/dist/custom/files/move-object.action.d.ts +38 -0
  54. package/dist/custom/files/move-object.action.d.ts.map +1 -0
  55. package/dist/custom/files/move-object.action.js +84 -0
  56. package/dist/custom/files/move-object.action.js.map +1 -0
  57. package/dist/custom/files/object-exists.action.d.ts +34 -0
  58. package/dist/custom/files/object-exists.action.d.ts.map +1 -0
  59. package/dist/custom/files/object-exists.action.js +75 -0
  60. package/dist/custom/files/object-exists.action.js.map +1 -0
  61. package/dist/custom/integration/gamma-generate-presentation.action.d.ts +108 -0
  62. package/dist/custom/integration/gamma-generate-presentation.action.d.ts.map +1 -0
  63. package/dist/custom/integration/gamma-generate-presentation.action.js +315 -0
  64. package/dist/custom/integration/gamma-generate-presentation.action.js.map +1 -0
  65. package/dist/custom/web/perplexity-search.action.d.ts +98 -0
  66. package/dist/custom/web/perplexity-search.action.d.ts.map +1 -0
  67. package/dist/custom/web/perplexity-search.action.js +269 -0
  68. package/dist/custom/web/perplexity-search.action.js.map +1 -0
  69. package/dist/custom/web/web-page-content.action.d.ts +74 -27
  70. package/dist/custom/web/web-page-content.action.d.ts.map +1 -1
  71. package/dist/custom/web/web-page-content.action.js +458 -226
  72. package/dist/custom/web/web-page-content.action.js.map +1 -1
  73. package/dist/index.d.ts +3 -0
  74. package/dist/index.d.ts.map +1 -1
  75. package/dist/index.js +32 -0
  76. package/dist/index.js.map +1 -1
  77. package/package.json +16 -10
@@ -5,58 +5,88 @@ var __decorate = (this && this.__decorate) || function (decorators, target, key,
5
5
  else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
6
6
  return c > 3 && r && Object.defineProperty(target, key, r), r;
7
7
  };
8
+ var __metadata = (this && this.__metadata) || function (k, v) {
9
+ if (typeof Reflect === "object" && typeof Reflect.metadata === "function") return Reflect.metadata(k, v);
10
+ };
11
+ var __importDefault = (this && this.__importDefault) || function (mod) {
12
+ return (mod && mod.__esModule) ? mod : { "default": mod };
13
+ };
8
14
  Object.defineProperty(exports, "__esModule", { value: true });
9
15
  exports.LoadWebPageContentAction = exports.WebPageContentAction = void 0;
10
16
  const actions_1 = require("@memberjunction/actions");
11
17
  const global_1 = require("@memberjunction/global");
18
+ const turndown_1 = __importDefault(require("turndown"));
19
+ const jsdom_1 = require("jsdom");
20
+ const pdfParse = require('pdf-parse');
21
+ const mammoth = require('mammoth');
22
+ const xml2js = require('xml2js');
23
+ const Papa = require('papaparse');
12
24
  /**
13
- * Action that retrieves and processes web page content with various output formats
14
- * Supports text extraction, HTML parsing, and basic document conversion
25
+ * Action that retrieves and processes web content in various formats
26
+ * Supports JSON, PDF, DOCX, XML, CSV, HTML and more - similar to Claude's web reading capabilities
15
27
  *
16
28
  * @example
17
29
  * ```typescript
18
- * // Get page content as text
30
+ * // Get JSON API response
19
31
  * await runAction({
20
32
  * ActionName: 'Web Page Content',
21
33
  * Params: [{
22
34
  * Name: 'URL',
23
- * Value: 'https://example.com'
35
+ * Value: 'https://api.example.com/data'
36
+ * }, {
37
+ * Name: 'ContentType',
38
+ * Value: 'json'
39
+ * }]
40
+ * });
41
+ *
42
+ * // Get PDF content as text
43
+ * await runAction({
44
+ * ActionName: 'Web Page Content',
45
+ * Params: [{
46
+ * Name: 'URL',
47
+ * Value: 'https://example.com/document.pdf'
24
48
  * }, {
25
49
  * Name: 'ContentType',
26
50
  * Value: 'text'
27
51
  * }]
28
52
  * });
29
53
  *
30
- * // Get main content only (removes navigation, ads, etc.)
54
+ * // Get DOCX content as markdown
31
55
  * await runAction({
32
56
  * ActionName: 'Web Page Content',
33
57
  * Params: [{
34
58
  * Name: 'URL',
35
- * Value: 'https://news.example.com/article'
59
+ * Value: 'https://example.com/document.docx'
36
60
  * }, {
37
61
  * Name: 'ContentType',
38
62
  * Value: 'markdown'
39
- * }, {
40
- * Name: 'ExtractMainContent',
41
- * Value: true
42
63
  * }]
43
64
  * });
44
65
  * ```
45
66
  */
46
67
  let WebPageContentAction = class WebPageContentAction extends actions_1.BaseAction {
47
- MAX_CONTENT_SIZE = 5 * 1024 * 1024; // 5MB limit
48
- MAX_IMAGE_SIZE = 2 * 1024 * 1024; // 2MB limit for images
68
+ MAX_CONTENT_SIZE = 10 * 1024 * 1024; // 10MB limit
69
+ MAX_IMAGE_SIZE = 5 * 1024 * 1024; // 5MB limit for images
70
+ turndown;
71
+ constructor() {
72
+ super();
73
+ this.turndown = new turndown_1.default({
74
+ headingStyle: 'atx',
75
+ codeBlockStyle: 'fenced',
76
+ emDelimiter: '*'
77
+ });
78
+ }
49
79
  /**
50
- * Executes the web page content retrieval and processing
80
+ * Executes web content retrieval and processing
51
81
  *
52
82
  * @param params - The action parameters containing:
53
- * - URL: Web page URL to fetch (required)
54
- * - ContentType: Output format - 'text', 'html', 'markdown', 'json' (default: 'text')
83
+ * - URL: Web resource URL to fetch (required)
84
+ * - ContentType: Output format - 'auto', 'text', 'html', 'markdown', 'json' (default: 'auto')
55
85
  * - ExtractMainContent: Extract only main content vs full page (default: false)
56
- * - IncludeMetadata: Include page metadata in response (default: true)
57
- * - MaxContentLength: Maximum content length to return (default: 50000 chars)
86
+ * - IncludeMetadata: Include resource metadata in response (default: true)
87
+ * - MaxContentLength: Maximum content length to return (default: 100000 chars)
58
88
  *
59
- * @returns Processed web page content in the specified format
89
+ * @returns Processed content in the specified format
60
90
  */
61
91
  async InternalRunAction(params) {
62
92
  try {
@@ -73,10 +103,10 @@ let WebPageContentAction = class WebPageContentAction extends actions_1.BaseActi
73
103
  };
74
104
  }
75
105
  const url = urlParam.Value.toString().trim();
76
- const contentType = contentTypeParam?.Value?.toString().toLowerCase() || 'text';
106
+ const requestedContentType = contentTypeParam?.Value?.toString().toLowerCase() || 'auto';
77
107
  const extractMain = extractMainParam?.Value === true || extractMainParam?.Value === 'true';
78
108
  const includeMetadata = includeMetadataParam?.Value !== false && includeMetadataParam?.Value !== 'false';
79
- const maxLength = parseInt(maxLengthParam?.Value?.toString() || '50000');
109
+ const maxLength = parseInt(maxLengthParam?.Value?.toString() || '100000');
80
110
  // Validate URL
81
111
  let parsedUrl;
82
112
  try {
@@ -93,18 +123,18 @@ let WebPageContentAction = class WebPageContentAction extends actions_1.BaseActi
93
123
  };
94
124
  }
95
125
  // Validate content type
96
- if (!['text', 'html', 'markdown', 'json'].includes(contentType)) {
126
+ if (!['auto', 'text', 'html', 'markdown', 'json'].includes(requestedContentType)) {
97
127
  return {
98
128
  Success: false,
99
- Message: "ContentType must be one of: text, html, markdown, json",
129
+ Message: "ContentType must be one of: auto, text, html, markdown, json",
100
130
  ResultCode: "INVALID_CONTENT_TYPE"
101
131
  };
102
132
  }
103
- // Fetch the web page
133
+ // Fetch the resource
104
134
  const response = await fetch(url, {
105
135
  headers: {
106
- 'User-Agent': 'Mozilla/5.0 (compatible; MemberJunction/1.0; +https://memberjunction.org)',
107
- 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
136
+ 'User-Agent': 'Mozilla/5.0 (compatible; MemberJunction/2.0; +https://memberjunction.org)',
137
+ 'Accept': '*/*',
108
138
  'Accept-Language': 'en-US,en;q=0.5',
109
139
  'Accept-Encoding': 'gzip, deflate'
110
140
  },
@@ -117,7 +147,7 @@ let WebPageContentAction = class WebPageContentAction extends actions_1.BaseActi
117
147
  ResultCode: "FETCH_FAILED"
118
148
  };
119
149
  }
120
- // Check content type and size
150
+ // Get response metadata
121
151
  const responseContentType = response.headers.get('content-type') || '';
122
152
  const contentLength = parseInt(response.headers.get('content-length') || '0');
123
153
  if (contentLength > this.MAX_CONTENT_SIZE) {
@@ -127,270 +157,471 @@ let WebPageContentAction = class WebPageContentAction extends actions_1.BaseActi
127
157
  ResultCode: "CONTENT_TOO_LARGE"
128
158
  };
129
159
  }
130
- // Handle different content types
131
- if (responseContentType.startsWith('image/')) {
132
- if (contentLength > this.MAX_IMAGE_SIZE) {
133
- return {
134
- Success: false,
135
- Message: `Image too large: ${contentLength} bytes (max: ${this.MAX_IMAGE_SIZE})`,
136
- ResultCode: "IMAGE_TOO_LARGE"
137
- };
138
- }
139
- // For images, return base64 encoded data
140
- const buffer = await response.arrayBuffer();
141
- const base64 = Buffer.from(buffer).toString('base64');
142
- const result = {
143
- url,
144
- contentType: responseContentType,
145
- contentLength,
146
- isImage: true,
147
- base64Data: base64,
148
- dataUrl: `data:${responseContentType};base64,${base64}`
149
- };
150
- return {
151
- Success: true,
152
- ResultCode: "SUCCESS",
153
- Message: JSON.stringify(result, null, 2)
154
- };
155
- }
156
- // Handle text-based content
157
- if (!responseContentType.includes('text/html') && !responseContentType.includes('application/xhtml') && !responseContentType.includes('text/')) {
158
- return {
159
- Success: false,
160
- Message: `Unsupported content type: ${responseContentType}`,
161
- ResultCode: "UNSUPPORTED_CONTENT_TYPE"
162
- };
163
- }
164
- const html = await response.text();
165
- if (html.length > this.MAX_CONTENT_SIZE) {
166
- return {
167
- Success: false,
168
- Message: `Content too large: ${html.length} characters (max: ${this.MAX_CONTENT_SIZE})`,
169
- ResultCode: "CONTENT_TOO_LARGE"
170
- };
171
- }
172
- // Process the content based on requested format
173
- const processedContent = await this.processContent(html, contentType, extractMain, includeMetadata, maxLength);
174
- processedContent.url = url;
175
- processedContent.fetchedAt = new Date().toISOString();
160
+ // Process based on detected content type
161
+ const detectedType = this.detectContentType(responseContentType, parsedUrl.pathname);
162
+ const processResult = await this.processResponse(response, detectedType, requestedContentType, extractMain, includeMetadata, maxLength);
163
+ processResult.url = url;
164
+ processResult.fetchedAt = new Date().toISOString();
165
+ processResult.responseContentType = responseContentType;
176
166
  return {
177
167
  Success: true,
178
168
  ResultCode: "SUCCESS",
179
- Message: JSON.stringify(processedContent, null, 2)
169
+ Message: JSON.stringify(processResult, null, 2)
180
170
  };
181
171
  }
182
172
  catch (error) {
183
173
  return {
184
174
  Success: false,
185
- Message: `Failed to retrieve web page content: ${error instanceof Error ? error.message : String(error)}`,
175
+ Message: `Failed to retrieve web content: ${error instanceof Error ? error.message : String(error)}`,
186
176
  ResultCode: "FAILED"
187
177
  };
188
178
  }
189
179
  }
190
180
  /**
191
- * Processes HTML content into the requested format
181
+ * Detects content type from response headers and URL
192
182
  */
193
- async processContent(html, contentType, extractMain, includeMetadata, maxLength) {
183
+ detectContentType(contentType, pathname) {
184
+ const lowerContentType = contentType.toLowerCase();
185
+ const lowerPath = pathname.toLowerCase();
186
+ // Check content-type header first
187
+ if (lowerContentType.includes('application/json'))
188
+ return 'json';
189
+ if (lowerContentType.includes('application/pdf'))
190
+ return 'pdf';
191
+ if (lowerContentType.includes('application/vnd.openxmlformats-officedocument.wordprocessingml'))
192
+ return 'docx';
193
+ if (lowerContentType.includes('application/msword'))
194
+ return 'doc';
195
+ if (lowerContentType.includes('text/csv') || lowerContentType.includes('application/csv'))
196
+ return 'csv';
197
+ if (lowerContentType.includes('text/xml') || lowerContentType.includes('application/xml'))
198
+ return 'xml';
199
+ if (lowerContentType.includes('text/html') || lowerContentType.includes('application/xhtml'))
200
+ return 'html';
201
+ if (lowerContentType.includes('text/plain'))
202
+ return 'text';
203
+ if (lowerContentType.includes('image/'))
204
+ return 'image';
205
+ // Check file extension if content-type is ambiguous
206
+ if (lowerPath.endsWith('.json'))
207
+ return 'json';
208
+ if (lowerPath.endsWith('.pdf'))
209
+ return 'pdf';
210
+ if (lowerPath.endsWith('.docx'))
211
+ return 'docx';
212
+ if (lowerPath.endsWith('.doc'))
213
+ return 'doc';
214
+ if (lowerPath.endsWith('.csv'))
215
+ return 'csv';
216
+ if (lowerPath.endsWith('.xml'))
217
+ return 'xml';
218
+ if (lowerPath.endsWith('.html') || lowerPath.endsWith('.htm'))
219
+ return 'html';
220
+ if (lowerPath.endsWith('.txt'))
221
+ return 'text';
222
+ // Default to text
223
+ return 'text';
224
+ }
225
+ /**
226
+ * Processes response based on detected content type
227
+ */
228
+ async processResponse(response, detectedType, requestedType, extractMain, includeMetadata, maxLength) {
229
+ const outputType = requestedType === 'auto' ? this.inferOutputType(detectedType) : requestedType;
230
+ switch (detectedType) {
231
+ case 'json':
232
+ return this.processJson(response, outputType, includeMetadata);
233
+ case 'pdf':
234
+ return this.processPdf(response, outputType, includeMetadata, maxLength);
235
+ case 'docx':
236
+ return this.processDocx(response, outputType, includeMetadata, maxLength);
237
+ case 'csv':
238
+ return this.processCsv(response, outputType, includeMetadata);
239
+ case 'xml':
240
+ return this.processXml(response, outputType, includeMetadata);
241
+ case 'html':
242
+ return this.processHtml(response, outputType, extractMain, includeMetadata, maxLength);
243
+ case 'image':
244
+ return this.processImage(response, includeMetadata);
245
+ case 'text':
246
+ default:
247
+ return this.processText(response, outputType, maxLength);
248
+ }
249
+ }
250
+ /**
251
+ * Infers best output type based on detected content type
252
+ */
253
+ inferOutputType(detectedType) {
254
+ switch (detectedType) {
255
+ case 'json':
256
+ case 'csv':
257
+ case 'xml':
258
+ return 'json';
259
+ case 'html':
260
+ return 'markdown';
261
+ case 'pdf':
262
+ case 'docx':
263
+ case 'text':
264
+ default:
265
+ return 'text';
266
+ }
267
+ }
268
+ /**
269
+ * Process JSON content
270
+ */
271
+ async processJson(response, outputType, includeMetadata) {
272
+ const text = await response.text();
273
+ const parsed = JSON.parse(text);
194
274
  const result = {
195
- contentType,
196
- extractMainContent: extractMain,
197
- includeMetadata
275
+ detectedType: 'json',
276
+ outputType
277
+ };
278
+ if (includeMetadata) {
279
+ result.metadata = {
280
+ size: text.length,
281
+ isArray: Array.isArray(parsed),
282
+ isObject: typeof parsed === 'object' && !Array.isArray(parsed),
283
+ keys: typeof parsed === 'object' && !Array.isArray(parsed) ? Object.keys(parsed) : undefined
284
+ };
285
+ }
286
+ if (outputType === 'json') {
287
+ result.content = parsed;
288
+ }
289
+ else if (outputType === 'text' || outputType === 'markdown') {
290
+ result.content = JSON.stringify(parsed, null, 2);
291
+ }
292
+ else {
293
+ result.content = text;
294
+ }
295
+ return result;
296
+ }
297
+ /**
298
+ * Process PDF content
299
+ */
300
+ async processPdf(response, outputType, includeMetadata, maxLength) {
301
+ const buffer = await response.arrayBuffer();
302
+ const pdfData = await pdfParse(Buffer.from(buffer));
303
+ const result = {
304
+ detectedType: 'pdf',
305
+ outputType
198
306
  };
199
- // Extract metadata if requested
200
307
  if (includeMetadata) {
201
- result.metadata = this.extractMetadata(html);
308
+ result.metadata = {
309
+ pages: pdfData.numpages,
310
+ info: pdfData.info,
311
+ version: pdfData.version
312
+ };
313
+ }
314
+ const text = pdfData.text || '';
315
+ if (outputType === 'json') {
316
+ result.content = {
317
+ text: this.truncateContent(text, maxLength),
318
+ pages: pdfData.numpages,
319
+ metadata: pdfData.info
320
+ };
321
+ }
322
+ else {
323
+ result.content = this.truncateContent(text, maxLength);
324
+ }
325
+ return result;
326
+ }
327
+ /**
328
+ * Process DOCX content
329
+ */
330
+ async processDocx(response, outputType, includeMetadata, maxLength) {
331
+ const buffer = await response.arrayBuffer();
332
+ const docxResult = await mammoth.convertToHtml({ buffer: Buffer.from(buffer) });
333
+ const htmlContent = docxResult.value;
334
+ const result = {
335
+ detectedType: 'docx',
336
+ outputType
337
+ };
338
+ if (includeMetadata) {
339
+ result.metadata = {
340
+ warnings: docxResult.messages,
341
+ htmlLength: htmlContent.length
342
+ };
343
+ }
344
+ let content;
345
+ if (outputType === 'markdown') {
346
+ content = this.turndown.turndown(htmlContent);
347
+ }
348
+ else if (outputType === 'html') {
349
+ content = htmlContent;
350
+ }
351
+ else {
352
+ content = this.htmlToText(htmlContent);
353
+ }
354
+ if (outputType === 'json') {
355
+ result.content = {
356
+ text: this.truncateContent(content, maxLength),
357
+ html: this.truncateContent(htmlContent, maxLength)
358
+ };
359
+ }
360
+ else {
361
+ result.content = this.truncateContent(content, maxLength);
362
+ }
363
+ return result;
364
+ }
365
+ /**
366
+ * Process CSV content
367
+ */
368
+ async processCsv(response, outputType, includeMetadata) {
369
+ const text = await response.text();
370
+ const parsed = Papa.parse(text, {
371
+ header: true,
372
+ skipEmptyLines: true,
373
+ dynamicTyping: true
374
+ });
375
+ const result = {
376
+ detectedType: 'csv',
377
+ outputType
378
+ };
379
+ if (includeMetadata) {
380
+ result.metadata = {
381
+ rows: parsed.data.length,
382
+ fields: parsed.meta.fields,
383
+ errors: parsed.errors
384
+ };
385
+ }
386
+ if (outputType === 'json') {
387
+ result.content = parsed.data;
388
+ }
389
+ else if (outputType === 'markdown') {
390
+ result.content = this.csvToMarkdownTable(parsed.data, parsed.meta.fields || []);
391
+ }
392
+ else {
393
+ result.content = text;
394
+ }
395
+ return result;
396
+ }
397
+ /**
398
+ * Process XML content
399
+ */
400
+ async processXml(response, outputType, includeMetadata) {
401
+ const text = await response.text();
402
+ const parser = new xml2js.Parser({
403
+ explicitArray: false,
404
+ mergeAttrs: true,
405
+ normalizeTags: true
406
+ });
407
+ const parsed = await parser.parseStringPromise(text);
408
+ const result = {
409
+ detectedType: 'xml',
410
+ outputType
411
+ };
412
+ if (includeMetadata) {
413
+ result.metadata = {
414
+ size: text.length,
415
+ rootElement: Object.keys(parsed)[0]
416
+ };
417
+ }
418
+ if (outputType === 'json') {
419
+ result.content = parsed;
420
+ }
421
+ else if (outputType === 'markdown' || outputType === 'text') {
422
+ result.content = JSON.stringify(parsed, null, 2);
423
+ }
424
+ else {
425
+ result.content = text;
426
+ }
427
+ return result;
428
+ }
429
+ /**
430
+ * Process HTML content
431
+ */
432
+ async processHtml(response, outputType, extractMain, includeMetadata, maxLength) {
433
+ const html = await response.text();
434
+ const result = {
435
+ detectedType: 'html',
436
+ outputType,
437
+ extractMainContent: extractMain
438
+ };
439
+ if (includeMetadata) {
440
+ result.metadata = this.extractHtmlMetadata(html);
202
441
  }
203
- // Extract main content or use full HTML
204
442
  let contentHtml = html;
205
443
  if (extractMain) {
206
444
  contentHtml = this.extractMainContent(html);
207
445
  }
208
- // Convert based on content type
209
- switch (contentType) {
210
- case 'html':
211
- result.content = this.truncateContent(contentHtml, maxLength);
212
- break;
213
- case 'text':
214
- result.content = this.truncateContent(this.htmlToText(contentHtml), maxLength);
215
- break;
216
- case 'markdown':
217
- result.content = this.truncateContent(this.htmlToMarkdown(contentHtml), maxLength);
218
- break;
219
- case 'json':
220
- result.content = {
221
- html: this.truncateContent(contentHtml, maxLength),
222
- text: this.truncateContent(this.htmlToText(contentHtml), maxLength),
223
- structure: this.extractStructure(contentHtml)
224
- };
225
- break;
446
+ let content;
447
+ if (outputType === 'markdown') {
448
+ content = this.turndown.turndown(contentHtml);
226
449
  }
227
- result.contentLength = typeof result.content === 'string' ? result.content.length : JSON.stringify(result.content).length;
450
+ else if (outputType === 'text') {
451
+ content = this.htmlToText(contentHtml);
452
+ }
453
+ else if (outputType === 'json') {
454
+ const dom = new jsdom_1.JSDOM(contentHtml);
455
+ result.content = {
456
+ html: this.truncateContent(contentHtml, maxLength),
457
+ text: this.truncateContent(this.htmlToText(contentHtml), maxLength),
458
+ markdown: this.truncateContent(this.turndown.turndown(contentHtml), maxLength),
459
+ structure: this.extractStructure(dom.window.document)
460
+ };
461
+ return result;
462
+ }
463
+ else {
464
+ content = contentHtml;
465
+ }
466
+ result.content = this.truncateContent(content, maxLength);
228
467
  return result;
229
468
  }
230
469
  /**
231
- * Extracts page metadata from HTML
470
+ * Process image content
232
471
  */
233
- extractMetadata(html) {
472
+ async processImage(response, includeMetadata) {
473
+ const buffer = await response.arrayBuffer();
474
+ const contentType = response.headers.get('content-type') || 'image/unknown';
475
+ if (buffer.byteLength > this.MAX_IMAGE_SIZE) {
476
+ throw new Error(`Image too large: ${buffer.byteLength} bytes (max: ${this.MAX_IMAGE_SIZE})`);
477
+ }
478
+ const base64 = Buffer.from(buffer).toString('base64');
479
+ const result = {
480
+ detectedType: 'image',
481
+ outputType: 'base64',
482
+ contentType,
483
+ size: buffer.byteLength,
484
+ base64Data: base64,
485
+ dataUrl: `data:${contentType};base64,${base64}`
486
+ };
487
+ if (includeMetadata) {
488
+ result.metadata = {
489
+ mimeType: contentType,
490
+ sizeBytes: buffer.byteLength,
491
+ sizeMB: (buffer.byteLength / (1024 * 1024)).toFixed(2)
492
+ };
493
+ }
494
+ return result;
495
+ }
496
+ /**
497
+ * Process plain text content
498
+ */
499
+ async processText(response, outputType, maxLength) {
500
+ const text = await response.text();
501
+ const result = {
502
+ detectedType: 'text',
503
+ outputType
504
+ };
505
+ if (outputType === 'json') {
506
+ result.content = {
507
+ text: this.truncateContent(text, maxLength),
508
+ lines: text.split('\n').length
509
+ };
510
+ }
511
+ else {
512
+ result.content = this.truncateContent(text, maxLength);
513
+ }
514
+ return result;
515
+ }
516
+ /**
517
+ * Extracts HTML metadata
518
+ */
519
+ extractHtmlMetadata(html) {
520
+ const dom = new jsdom_1.JSDOM(html);
521
+ const doc = dom.window.document;
234
522
  const metadata = {};
235
523
  // Title
236
- const titleMatch = html.match(/<title[^>]*>(.*?)<\/title>/is);
237
- if (titleMatch) {
238
- metadata.title = this.stripHtml(titleMatch[1]).trim();
524
+ const title = doc.querySelector('title');
525
+ if (title) {
526
+ metadata.title = title.textContent?.trim();
239
527
  }
240
528
  // Meta tags
241
- const metaRegex = /<meta[^>]+>/gi;
242
- const metas = html.match(metaRegex) || [];
243
- for (const meta of metas) {
244
- const nameMatch = meta.match(/name=['"]([^'"]+)['"]/i);
245
- const propertyMatch = meta.match(/property=['"]([^'"]+)['"]/i);
246
- const contentMatch = meta.match(/content=['"]([^'"]*)['"]/i);
247
- if (contentMatch) {
248
- const content = contentMatch[1];
249
- if (nameMatch) {
250
- metadata[nameMatch[1]] = content;
251
- }
252
- else if (propertyMatch) {
253
- metadata[propertyMatch[1]] = content;
254
- }
529
+ const metas = doc.querySelectorAll('meta');
530
+ metas.forEach(meta => {
531
+ const name = meta.getAttribute('name') || meta.getAttribute('property');
532
+ const content = meta.getAttribute('content');
533
+ if (name && content) {
534
+ metadata[name] = content;
255
535
  }
256
- }
536
+ });
257
537
  // Canonical URL
258
- const canonicalMatch = html.match(/<link[^>]+rel=['"]canonical['"][^>]+href=['"]([^'"]+)['"]/i);
259
- if (canonicalMatch) {
260
- metadata.canonical = canonicalMatch[1];
538
+ const canonical = doc.querySelector('link[rel="canonical"]');
539
+ if (canonical) {
540
+ metadata.canonical = canonical.getAttribute('href');
261
541
  }
262
542
  return metadata;
263
543
  }
264
544
  /**
265
- * Attempts to extract main content from HTML (removes navigation, ads, etc.)
545
+ * Attempts to extract main content from HTML
266
546
  */
267
547
  extractMainContent(html) {
548
+ const dom = new jsdom_1.JSDOM(html);
549
+ const doc = dom.window.document;
268
550
  // Remove scripts and styles
269
- let content = html.replace(/<script[^>]*>[\s\S]*?<\/script>/gi, '');
270
- content = content.replace(/<style[^>]*>[\s\S]*?<\/style>/gi, '');
551
+ doc.querySelectorAll('script, style, nav, header, footer, aside').forEach(el => el.remove());
271
552
  // Try to find main content containers
272
- const mainSelectors = [
273
- /<main[^>]*>([\s\S]*?)<\/main>/i,
274
- /<article[^>]*>([\s\S]*?)<\/article>/i,
275
- /<div[^>]*class=['"][^'"]*content[^'"]*['"][^>]*>([\s\S]*?)<\/div>/i,
276
- /<div[^>]*id=['"][^'"]*content[^'"]*['"][^>]*>([\s\S]*?)<\/div>/i,
277
- /<div[^>]*class=['"][^'"]*post[^'"]*['"][^>]*>([\s\S]*?)<\/div>/i
278
- ];
553
+ const mainSelectors = ['main', 'article', '[role="main"]', '.content', '#content', '.post', '.article'];
279
554
  for (const selector of mainSelectors) {
280
- const match = content.match(selector);
281
- if (match && match[1].trim().length > 100) {
282
- return match[1];
555
+ const element = doc.querySelector(selector);
556
+ if (element && element.textContent && element.textContent.trim().length > 100) {
557
+ return element.innerHTML;
283
558
  }
284
559
  }
285
- // If no main content found, remove common non-content elements
286
- content = content.replace(/<nav[^>]*>[\s\S]*?<\/nav>/gi, '');
287
- content = content.replace(/<header[^>]*>[\s\S]*?<\/header>/gi, '');
288
- content = content.replace(/<footer[^>]*>[\s\S]*?<\/footer>/gi, '');
289
- content = content.replace(/<aside[^>]*>[\s\S]*?<\/aside>/gi, '');
290
- content = content.replace(/<div[^>]*class=['"][^'"]*nav[^'"]*['"][^>]*>[\s\S]*?<\/div>/gi, '');
291
- return content;
560
+ // If no main content found, return body without nav/header/footer
561
+ const body = doc.querySelector('body');
562
+ return body ? body.innerHTML : html;
292
563
  }
293
564
  /**
294
565
  * Converts HTML to plain text
295
566
  */
296
567
  htmlToText(html) {
297
- return html
298
- .replace(/<script[^>]*>[\s\S]*?<\/script>/gi, '')
299
- .replace(/<style[^>]*>[\s\S]*?<\/style>/gi, '')
300
- .replace(/<br\s*\/?>/gi, '\n')
301
- .replace(/<\/p>/gi, '\n\n')
302
- .replace(/<\/h[1-6]>/gi, '\n\n')
303
- .replace(/<\/div>/gi, '\n')
304
- .replace(/<\/li>/gi, '\n')
305
- .replace(/<[^>]*>/g, '')
306
- .replace(/&amp;/g, '&')
307
- .replace(/&lt;/g, '<')
308
- .replace(/&gt;/g, '>')
309
- .replace(/&quot;/g, '"')
310
- .replace(/&#39;/g, "'")
311
- .replace(/&nbsp;/g, ' ')
312
- .replace(/\s+/g, ' ')
313
- .replace(/\n\s+/g, '\n')
314
- .replace(/\n{3,}/g, '\n\n')
315
- .trim();
316
- }
317
- /**
318
- * Converts HTML to basic Markdown
319
- */
320
- htmlToMarkdown(html) {
321
- return html
322
- .replace(/<script[^>]*>[\s\S]*?<\/script>/gi, '')
323
- .replace(/<style[^>]*>[\s\S]*?<\/style>/gi, '')
324
- .replace(/<h1[^>]*>(.*?)<\/h1>/gi, '# $1\n\n')
325
- .replace(/<h2[^>]*>(.*?)<\/h2>/gi, '## $1\n\n')
326
- .replace(/<h3[^>]*>(.*?)<\/h3>/gi, '### $1\n\n')
327
- .replace(/<h4[^>]*>(.*?)<\/h4>/gi, '#### $1\n\n')
328
- .replace(/<h5[^>]*>(.*?)<\/h5>/gi, '##### $1\n\n')
329
- .replace(/<h6[^>]*>(.*?)<\/h6>/gi, '###### $1\n\n')
330
- .replace(/<strong[^>]*>(.*?)<\/strong>/gi, '**$1**')
331
- .replace(/<b[^>]*>(.*?)<\/b>/gi, '**$1**')
332
- .replace(/<em[^>]*>(.*?)<\/em>/gi, '*$1*')
333
- .replace(/<i[^>]*>(.*?)<\/i>/gi, '*$1*')
334
- .replace(/<a[^>]+href=['"]([^'"]+)['"][^>]*>(.*?)<\/a>/gi, '[$2]($1)')
335
- .replace(/<img[^>]+src=['"]([^'"]+)['"][^>]*alt=['"]([^'"]*)['"]/gi, '![$2]($1)')
336
- .replace(/<img[^>]+src=['"]([^'"]+)['"][^>]*/gi, '![]($1)')
337
- .replace(/<br\s*\/?>/gi, '\n')
338
- .replace(/<\/p>/gi, '\n\n')
339
- .replace(/<\/div>/gi, '\n')
340
- .replace(/<li[^>]*>(.*?)<\/li>/gi, '- $1\n')
341
- .replace(/<[^>]*>/g, '')
342
- .replace(/&amp;/g, '&')
343
- .replace(/&lt;/g, '<')
344
- .replace(/&gt;/g, '>')
345
- .replace(/&quot;/g, '"')
346
- .replace(/&#39;/g, "'")
347
- .replace(/&nbsp;/g, ' ')
348
- .replace(/\s+/g, ' ')
349
- .replace(/\n\s+/g, '\n')
350
- .replace(/\n{3,}/g, '\n\n')
351
- .trim();
568
+ const dom = new jsdom_1.JSDOM(html);
569
+ const doc = dom.window.document;
570
+ // Remove script and style elements
571
+ doc.querySelectorAll('script, style').forEach(el => el.remove());
572
+ return doc.body.textContent || '';
352
573
  }
353
574
  /**
354
- * Extracts basic document structure
575
+ * Extracts document structure from HTML
355
576
  */
356
- extractStructure(html) {
577
+ extractStructure(doc) {
357
578
  const structure = {
358
579
  headings: [],
359
580
  links: [],
360
581
  images: []
361
582
  };
362
583
  // Extract headings
363
- const headingRegex = /<h([1-6])[^>]*>(.*?)<\/h[1-6]>/gi;
364
- let match;
365
- while ((match = headingRegex.exec(html)) !== null) {
366
- structure.headings.push({
367
- level: parseInt(match[1]),
368
- text: this.stripHtml(match[2]).trim()
369
- });
370
- }
584
+ const headings = doc.querySelectorAll('h1, h2, h3, h4, h5, h6');
585
+ structure.headings = Array.from(headings).map(h => ({
586
+ level: parseInt(h.tagName[1]),
587
+ text: h.textContent?.trim() || ''
588
+ }));
371
589
  // Extract links
372
- const linkRegex = /<a[^>]+href=['"]([^'"]+)['"][^>]*>(.*?)<\/a>/gi;
373
- while ((match = linkRegex.exec(html)) !== null) {
374
- structure.links.push({
375
- url: match[1],
376
- text: this.stripHtml(match[2]).trim()
377
- });
378
- }
590
+ const links = doc.querySelectorAll('a[href]');
591
+ structure.links = Array.from(links).map(a => ({
592
+ url: a.getAttribute('href'),
593
+ text: a.textContent?.trim() || ''
594
+ }));
379
595
  // Extract images
380
- const imgRegex = /<img[^>]+src=['"]([^'"]+)['"][^>]*(?:alt=['"]([^'"]*)['"]*)?/gi;
381
- while ((match = imgRegex.exec(html)) !== null) {
382
- structure.images.push({
383
- src: match[1],
384
- alt: match[2] || ''
385
- });
386
- }
596
+ const images = doc.querySelectorAll('img[src]');
597
+ structure.images = Array.from(images).map(img => ({
598
+ src: img.getAttribute('src'),
599
+ alt: img.getAttribute('alt') || ''
600
+ }));
387
601
  return structure;
388
602
  }
389
603
  /**
390
- * Strips HTML tags from text
604
+ * Converts CSV data to markdown table
391
605
  */
392
- stripHtml(html) {
393
- return html.replace(/<[^>]*>/g, '');
606
+ csvToMarkdownTable(data, fields) {
607
+ if (data.length === 0)
608
+ return '';
609
+ const headers = fields.length > 0 ? fields : Object.keys(data[0] || {});
610
+ if (headers.length === 0)
611
+ return '';
612
+ // Header row
613
+ let markdown = '| ' + headers.join(' | ') + ' |\n';
614
+ // Separator row
615
+ markdown += '| ' + headers.map(() => '---').join(' | ') + ' |\n';
616
+ // Data rows
617
+ data.forEach(row => {
618
+ const values = headers.map(header => {
619
+ const value = row[header];
620
+ return value != null ? String(value) : '';
621
+ });
622
+ markdown += '| ' + values.join(' | ') + ' |\n';
623
+ });
624
+ return markdown;
394
625
  }
395
626
  /**
396
627
  * Truncates content to specified length
@@ -404,7 +635,8 @@ let WebPageContentAction = class WebPageContentAction extends actions_1.BaseActi
404
635
  };
405
636
  exports.WebPageContentAction = WebPageContentAction;
406
637
  exports.WebPageContentAction = WebPageContentAction = __decorate([
407
- (0, global_1.RegisterClass)(actions_1.BaseAction, "__WebPageContent")
638
+ (0, global_1.RegisterClass)(actions_1.BaseAction, "__WebPageContent"),
639
+ __metadata("design:paramtypes", [])
408
640
  ], WebPageContentAction);
409
641
  /**
410
642
  * Loader function to ensure the WebPageContentAction class is included in the bundle.