@lobehub/chat 0.27.1 → 0.27.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,64 @@
2
2
 
3
3
  # Changelog
4
4
 
5
+ ### [Version 0.27.3](https://github.com/lobehub/lobe-chat/compare/v0.27.2...v0.27.3)
6
+
7
+ <sup>Released on **2023-07-29**</sup>
8
+
9
+ #### 🐛 Bug Fixes
10
+
11
+ - **misc**: 修正返回结果导致插件无法正常识别的问题.
12
+
13
+ #### 💄 Styles
14
+
15
+ - **misc**: 优化样式.
16
+
17
+ <br/>
18
+
19
+ <details>
20
+ <summary><kbd>Improvements and Fixes</kbd></summary>
21
+
22
+ #### What's fixed
23
+
24
+ - **misc**: 修正返回结果导致插件无法正常识别的问题 ([b183188](https://github.com/lobehub/lobe-chat/commit/b183188))
25
+
26
+ #### Styles
27
+
28
+ - **misc**: 优化样式 ([9ce5d1d](https://github.com/lobehub/lobe-chat/commit/9ce5d1d))
29
+
30
+ </details>
31
+
32
+ <div align="right">
33
+
34
+ [![](https://img.shields.io/badge/-BACK_TO_TOP-151515?style=flat-square)](#readme-top)
35
+
36
+ </div>
37
+
38
+ ### [Version 0.27.2](https://github.com/lobehub/lobe-chat/compare/v0.27.1...v0.27.2)
39
+
40
+ <sup>Released on **2023-07-29**</sup>
41
+
42
+ #### ♻ Code Refactoring
43
+
44
+ - **misc**: 重构并优化文档抓取插件能力.
45
+
46
+ <br/>
47
+
48
+ <details>
49
+ <summary><kbd>Improvements and Fixes</kbd></summary>
50
+
51
+ #### Code refactoring
52
+
53
+ - **misc**: 重构并优化文档抓取插件能力 ([ff56348](https://github.com/lobehub/lobe-chat/commit/ff56348))
54
+
55
+ </details>
56
+
57
+ <div align="right">
58
+
59
+ [![](https://img.shields.io/badge/-BACK_TO_TOP-151515?style=flat-square)](#readme-top)
60
+
61
+ </div>
62
+
5
63
  ### [Version 0.27.1](https://github.com/lobehub/lobe-chat/compare/v0.27.0...v0.27.1)
6
64
 
7
65
  <sup>Released on **2023-07-29**</sup>
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lobehub/chat",
3
- "version": "0.27.1",
3
+ "version": "0.27.3",
4
4
  "description": "Lobe Chat is an open-source chatbot client using LangChain, Typescript and Next.js",
5
5
  "keywords": [
6
6
  "chatbot",
@@ -49,7 +49,6 @@ const FunctionCall = memo<FunctionCallProps>(({ function_call, loading }) => {
49
49
  avatar
50
50
  )}
51
51
  {t(`plugins.${function_call?.name}` as any, { ns: 'plugin' })}
52
- {loading ? `(${t('loading.plugin')})` : null}
53
52
  <Icon icon={open ? LucideChevronUp : LucideChevronDown} />
54
53
  </Flexbox>
55
54
  {open && <Highlighter language={'json'}>{args}</Highlighter>}
@@ -1,50 +1,45 @@
1
1
  import { PluginRunner } from '@/plugins/type';
2
2
 
3
- import { DataResults, Result } from './type';
3
+ import { ParserResponse, Result } from './type';
4
4
 
5
- const BASE_URL =
6
- 'https://api.apify.com/v2/acts/apify~website-content-crawler/run-sync-get-dataset-items';
7
- const token = process.env.APIFY_API_KEY;
5
+ const BASE_URL = process.env.BROWSERLESS_URL ?? 'https://chrome.browserless.io';
6
+ const BROWSERLESS_TOKEN = process.env.BROWSERLESS_TOKEN;
7
+
8
+ // service from: https://github.com/lobehub/html-parser/tree/master
9
+ const HTML_PARSER_URL = process.env.HTML_PARSER_URL;
8
10
 
9
11
  const runner: PluginRunner<{ url: string }, Result> = async ({ url }) => {
10
- // Prepare Actor input
11
12
  const input = {
12
- aggressivePrune: false,
13
- clickElementsCssSelector: '[aria-expanded="false"]',
14
- debugMode: false,
15
- dynamicContentWaitSecs: 3,
16
- proxyConfiguration: {
17
- useApifyProxy: true,
18
- },
19
- removeCookieWarnings: true,
20
- removeElementsCssSelector:
21
- 'nav, footer, script, style, noscript, svg,\n[role="alert"],\n[role="banner"],\n[role="dialog"],\n[role="alertdialog"],\n[role="region"][aria-label*="skip" i],\n[aria-modal="true"]',
22
- saveFiles: false,
23
- saveHtml: false,
24
- saveMarkdown: true,
25
- saveScreenshots: false,
26
- startUrls: [{ url }],
13
+ gotoOptions: { waitUntil: 'networkidle2' },
14
+ url,
27
15
  };
28
16
 
29
17
  try {
30
- const data = await fetch(`${BASE_URL}?token=${token}`, {
18
+ const res = await fetch(`${BASE_URL}/content?token=${BROWSERLESS_TOKEN}`, {
31
19
  body: JSON.stringify(input),
32
20
  headers: {
33
21
  'Content-Type': 'application/json',
34
22
  },
35
23
  method: 'POST',
36
24
  });
37
- const result = (await data.json()) as DataResults;
38
-
39
- const item = result[0];
40
- return {
41
- content: item.markdown,
42
- title: item.metadata.title,
43
- url: item.url,
44
- };
25
+ const html = await res.text();
26
+
27
+ const parserBody = { html, url };
28
+
29
+ const parseRes = await fetch(`${HTML_PARSER_URL}`, {
30
+ body: JSON.stringify(parserBody),
31
+ headers: {
32
+ 'Content-Type': 'application/json',
33
+ },
34
+ method: 'POST',
35
+ });
36
+
37
+ const { title, textContent, siteName } = (await parseRes.json()) as ParserResponse;
38
+
39
+ return { content: textContent, title, url, website: siteName };
45
40
  } catch (error) {
46
41
  console.error(error);
47
- return { content: '抓取失败', errorMessage: (error as any).message };
42
+ return { content: '抓取失败', errorMessage: (error as any).message, url };
48
43
  }
49
44
  };
50
45
 
@@ -1,32 +1,35 @@
1
- export type DataResults = DataItem[];
2
-
3
1
  export type Result = {
4
- content?: string;
2
+ content: string;
5
3
  title?: string;
6
- url?: string;
7
- };
8
- export interface DataItem {
9
- crawl: Crawl;
10
- markdown: string;
11
- metadata: Metadata;
12
- screenshotUrl: any;
13
- text: string;
14
4
  url: string;
15
- }
5
+ website?: string;
6
+ };
16
7
 
17
- export interface Crawl {
18
- depth: number;
19
- httpStatusCode: number;
20
- loadedTime: string;
21
- loadedUrl: string;
22
- referrerUrl: string;
23
- }
8
+ export interface ParserResponse {
9
+ /** author metadata */
10
+ byline: string;
11
+
12
+ /** HTML string of processed article content */
13
+ content: string;
14
+
15
+ /** content direction */
16
+ dir: string;
17
+
18
+ /** article description, or short excerpt from the content */
19
+ excerpt: string;
20
+
21
+ /** content language */
22
+ lang: string;
23
+
24
+ /** length of an article, in characters */
25
+ length: number;
26
+
27
+ /** name of the site */
28
+ siteName: string;
29
+
30
+ /** text content of the article, with all the HTML tags removed */
31
+ textContent: string;
24
32
 
25
- export interface Metadata {
26
- author: any;
27
- canonicalUrl: string;
28
- description: string;
29
- keywords: string;
30
- languageCode: string;
33
+ /** article title */
31
34
  title: string;
32
35
  }