koishi-plugin-ban-qrcode 1.2.0 → 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/document.d.ts +2 -1
- package/lib/document.js +19 -6
- package/lib/index.d.ts +1 -0
- package/lib/index.js +2 -1
- package/package.json +1 -1
- package/readme.md +2 -1
package/lib/document.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
export declare const DEFAULT_MAX_OFFICE_BYTES: number;
|
|
1
2
|
export interface DocContent {
|
|
2
3
|
title: string;
|
|
3
4
|
text: string;
|
|
@@ -17,5 +18,5 @@ export declare function parseTencentDocUrl(url: string): {
|
|
|
17
18
|
export declare function collectTencentDocUrls(text: string): string[];
|
|
18
19
|
export declare function parseOpendocBody(body: string): DocContent | null;
|
|
19
20
|
export declare function fetchTencentDoc(http: DocHttp, url: string): Promise<DocContent | null>;
|
|
20
|
-
export declare function extractOfficeText(buffer: Buffer, name: string): DocContent | null;
|
|
21
|
+
export declare function extractOfficeText(buffer: Buffer, name: string, maxBytes?: number): DocContent | null;
|
|
21
22
|
export declare function extractReadableText(input: string | Buffer): string;
|
package/lib/document.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.DEFAULT_MAX_OFFICE_BYTES = void 0;
|
|
3
4
|
exports.parseTencentDocUrl = parseTencentDocUrl;
|
|
4
5
|
exports.collectTencentDocUrls = collectTencentDocUrls;
|
|
5
6
|
exports.parseOpendocBody = parseOpendocBody;
|
|
@@ -12,6 +13,7 @@ const UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML,
|
|
|
12
13
|
const DOC_RE = /https?:\/\/(?:docs\.qq\.com|doc\.weixin\.qq\.com)\/(doc|sheet|slide|pdf|form|mind)\/([A-Za-z0-9]+)/i;
|
|
13
14
|
const ZIP_LOCAL = Buffer.from([0x50, 0x4b, 0x03, 0x04]);
|
|
14
15
|
const IMAGE_RE = /https?:\/\/docimg\d+\.docs\.qq\.com\/image\/[A-Za-z0-9_-]+(?:\.(?:jpe?g|png|webp))?/gi;
|
|
16
|
+
exports.DEFAULT_MAX_OFFICE_BYTES = 5 * 1024 * 1024;
|
|
15
17
|
function parseTencentDocUrl(url) {
|
|
16
18
|
const match = DOC_RE.exec(url);
|
|
17
19
|
if (!match)
|
|
@@ -83,14 +85,16 @@ async function fetchTencentDoc(http, url) {
|
|
|
83
85
|
const opendoc = await http.text(opendocUrl, headers);
|
|
84
86
|
return parseOpendocBody(opendoc.body);
|
|
85
87
|
}
|
|
86
|
-
function extractOfficeText(buffer, name) {
|
|
88
|
+
function extractOfficeText(buffer, name, maxBytes = exports.DEFAULT_MAX_OFFICE_BYTES) {
|
|
89
|
+
if (buffer.length > maxBytes)
|
|
90
|
+
return null;
|
|
87
91
|
const lower = name.toLowerCase();
|
|
88
92
|
if (lower.endsWith('.txt')) {
|
|
89
93
|
const text = buffer.toString('utf8').trim();
|
|
90
94
|
return text ? { title: name, text, images: [] } : null;
|
|
91
95
|
}
|
|
92
96
|
if (lower.endsWith('.docx')) {
|
|
93
|
-
const xml = readZipEntry(buffer, 'word/document.xml');
|
|
97
|
+
const xml = readZipEntry(buffer, 'word/document.xml', maxBytes);
|
|
94
98
|
if (!xml)
|
|
95
99
|
return null;
|
|
96
100
|
const text = xml
|
|
@@ -165,7 +169,7 @@ function extractUtf16LeText(buffer) {
|
|
|
165
169
|
chars.push(run);
|
|
166
170
|
return chars.join('\n');
|
|
167
171
|
}
|
|
168
|
-
function readZipEntry(buffer, suffix) {
|
|
172
|
+
function readZipEntry(buffer, suffix, maxBytes) {
|
|
169
173
|
let offset = 0;
|
|
170
174
|
while (offset < buffer.length) {
|
|
171
175
|
const found = buffer.indexOf(ZIP_LOCAL, offset);
|
|
@@ -175,6 +179,7 @@ function readZipEntry(buffer, suffix) {
|
|
|
175
179
|
return null;
|
|
176
180
|
const method = buffer.readUInt16LE(found + 8);
|
|
177
181
|
const compSize = buffer.readUInt32LE(found + 18);
|
|
182
|
+
const uncompSize = buffer.readUInt32LE(found + 22);
|
|
178
183
|
const nameLen = buffer.readUInt16LE(found + 26);
|
|
179
184
|
const extraLen = buffer.readUInt16LE(found + 28);
|
|
180
185
|
const nameStart = found + 30;
|
|
@@ -183,11 +188,19 @@ function readZipEntry(buffer, suffix) {
|
|
|
183
188
|
return null;
|
|
184
189
|
const name = buffer.subarray(nameStart, nameStart + nameLen).toString('utf8');
|
|
185
190
|
if (name === suffix || name.endsWith(`/${suffix}`)) {
|
|
191
|
+
if (uncompSize > 0 && uncompSize <= 0xffff_fffe && uncompSize > maxBytes)
|
|
192
|
+
return null;
|
|
186
193
|
const data = buffer.subarray(dataStart, dataStart + compSize);
|
|
187
194
|
if (method === 0)
|
|
188
|
-
return data.toString('utf8');
|
|
189
|
-
if (method === 8)
|
|
190
|
-
|
|
195
|
+
return data.length > maxBytes ? null : data.toString('utf8');
|
|
196
|
+
if (method === 8) {
|
|
197
|
+
try {
|
|
198
|
+
return (0, node_zlib_1.inflateRawSync)(data, { maxOutputLength: maxBytes }).toString('utf8');
|
|
199
|
+
}
|
|
200
|
+
catch {
|
|
201
|
+
return null;
|
|
202
|
+
}
|
|
203
|
+
}
|
|
191
204
|
}
|
|
192
205
|
offset = dataStart + Math.max(compSize, 1);
|
|
193
206
|
}
|
package/lib/index.d.ts
CHANGED
package/lib/index.js
CHANGED
|
@@ -29,6 +29,7 @@ exports.Config = koishi_1.Schema.object({
|
|
|
29
29
|
scanGroupInvite: koishi_1.Schema.boolean().default(true).description('拦截邀请 / 推荐群聊 / 群名片分享卡。'),
|
|
30
30
|
scanDocs: koishi_1.Schema.boolean().default(true).description('检查腾讯文档和 Word / 文本附件里的广告。'),
|
|
31
31
|
adKeywords: koishi_1.Schema.array(koishi_1.Schema.string()).role('table').default([]).description('额外广告关键词。命中即撤回。'),
|
|
32
|
+
maxOfficeMb: koishi_1.Schema.number().min(1).default(5).description('解析 Word / 文本附件的大小上限(MB)。超出则跳过,防止解压占用过多内存。'),
|
|
32
33
|
debug: koishi_1.Schema.boolean().default(true).description('输出调试日志:跳过原因、消息结构、下载/扫码/文档结果。'),
|
|
33
34
|
});
|
|
34
35
|
function apply(ctx, config) {
|
|
@@ -146,7 +147,7 @@ function apply(ctx, config) {
|
|
|
146
147
|
}
|
|
147
148
|
for (const file of officeFiles) {
|
|
148
149
|
try {
|
|
149
|
-
const doc = (0, document_1.extractOfficeText)(await (0, qrcode_1.downloadFile)(ctx.http, { ...file, groupId: guildId }, resolveFile), file.name);
|
|
150
|
+
const doc = (0, document_1.extractOfficeText)(await (0, qrcode_1.downloadFile)(ctx.http, { ...file, groupId: guildId }, resolveFile), file.name, Math.floor(config.maxOfficeMb * 1024 * 1024));
|
|
150
151
|
const ad = doc ? (0, ad_1.detectAdContent)(doc.text, doc.title || file.name, config.adKeywords) : null;
|
|
151
152
|
if (config.debug)
|
|
152
153
|
logger.info('office %s ad=%s', file.name, Boolean(ad));
|
package/package.json
CHANGED
package/readme.md
CHANGED
|
@@ -94,6 +94,7 @@ npm run build
|
|
|
94
94
|
| `scanGroupInvite` | `boolean` | `true` | 拦截邀请 / 推荐群聊分享卡 |
|
|
95
95
|
| `scanDocs` | `boolean` | `true` | 检查腾讯文档和 Word / 文本附件 |
|
|
96
96
|
| `adKeywords` | `string[]` | `[]` | 额外广告关键词,命中即撤回 |
|
|
97
|
+
| `maxOfficeMb` | `number` | `5` | Word / 文本附件的大小上限(MB),超出则跳过 |
|
|
97
98
|
| `debug` | `boolean` | `true` | 输出调试日志:跳过原因、消息结构、下载 / 扫码 / 文档结果 |
|
|
98
99
|
|
|
99
100
|
内置词在 `src/ad-keywords.txt`,一行一条。`[strong]` 命中任意一条即撤回;`[commerce]` 要同时出现床品类用词且至少两条。控制台 `adKeywords` 按强匹配叠加。
|
|
@@ -278,7 +279,7 @@ Actions 会 `npm ci` → `npm test` → `npm run build` → `npm publish --acces
|
|
|
278
279
|
|
|
279
280
|
## Changelog
|
|
280
281
|
|
|
281
|
-
当前版本 **1.2.0
|
|
282
|
+
当前版本 **1.2.1**:Word / 文本附件默认限制 5MB(`maxOfficeMb`),防止解压占用过多内存。1.2.0 起内置广告词改到 `ad-keywords.txt` 维护。1.1.3 补拦 QQ 群名片。1.1.2 修复 OneBot `getImage` 未绑定导致的生产崩溃。1.1.1 补齐跳过原因日志。1.1.0 起拦截拉群分享卡和腾讯文档 / Word 广告。
|
|
282
283
|
|
|
283
284
|
## License
|
|
284
285
|
|