koishi-plugin-kkk 3.2.1 → 3.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/lib/index.js +33 -11
- package/lib/karin/module/utils/CardParser.js +115 -25
- package/package.json +1 -1
- package/scripts/smoke-cardparse.cjs +58 -0
- package/src/index.ts +33 -11
- package/src/karin/module/utils/CardParser.ts +105 -23
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,14 @@
|
|
|
1
1
|
# 更新日志
|
|
2
2
|
|
|
3
|
+
## 3.2.2
|
|
4
|
+
|
|
5
|
+
### 卡片解析:按 UP 主模糊匹配,候选也发得出来了
|
|
6
|
+
|
|
7
|
+
- **认错 UP 主的问题修了**:B站卡片的 OCR 文本里,「UP主」的上一行**不一定是昵称** —— 你那张卡的上一行是封面上的「半身像」,真名「雾小霜暗区突围」在**第一行**。以前只认「UP主前一行」,于是拿「半身像」去搜,标题又对不上,六个候选一个都不敢选(就是你看到的「没能唯一定位…候选 6 条」)。现在**首行、UP主前后行**都当候选,搜索时**任意一个对上**就算作者命中,命中的那个会写进日志和提示语。
|
|
8
|
+
- **候选列表在个人号上发不出去的问题修了**:候选以前是**一条 markdown 表格**,只有 QQ 官方机器人渲染,个人号(NapCat 等)发过去整条消息都出不去 —— 用户只看到「正在提取卡片信息…」然后就没了下文。现在**按平台分开发**:官方 QQ 还是表格 + 按钮(点一下直接解析);个人号改成纯文本,一条条列「标题 —— UP」+ 完整链接,点链接或复制发回来就能解析。
|
|
9
|
+
- 顺手:抖音搜索也带上作者加分(之前只看标题,噪声大时容易选错)。
|
|
10
|
+
|
|
11
|
+
实测(用你那条 OCR):候选名首选「雾小霜暗区突围」;拿它去 B站搜「根本就没有这些大金」,命中作者=雾小霜暗区突围、标题=根本就没有这些大金,**自动定位成功**。
|
|
3
12
|
## 3.2.1
|
|
4
13
|
|
|
5
14
|
### 出错时自动上报,进群报个编号就行
|
package/lib/index.js
CHANGED
|
@@ -704,18 +704,40 @@ function registerCommands(ctx, logger, autoParse) {
|
|
|
704
704
|
return;
|
|
705
705
|
}
|
|
706
706
|
if (resolved && resolved.candidates && resolved.candidates.length) {
|
|
707
|
-
const
|
|
708
|
-
const
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
707
|
+
const top = resolved.candidates.slice(0, 6);
|
|
708
|
+
const tip = '没能唯一确定这个作品(识别到:' + ((resolved.upName) || '未知') + '),下面是候选:';
|
|
709
|
+
const linkOf = (item) => item.platform === 'bilibili'
|
|
710
|
+
? 'https://www.bilibili.com/video/' + item.id
|
|
711
|
+
: 'https://www.douyin.com/video/' + item.id;
|
|
712
|
+
/**
|
|
713
|
+
* **按平台分成两种发法**。
|
|
714
|
+
*
|
|
715
|
+
* QQ 官方机器人认 markdown,候选直接排成表格、操作列是按钮,点一下就走解析(最省事)。
|
|
716
|
+
* 个人号(OneBot / NapCat)**不渲染 markdown**:实测这条候选消息发过去整条都发不出来 ——
|
|
717
|
+
* 用户只看到「正在提取卡片信息…」然后再无音讯,六个候选白搜。那边改成纯文本 + 完整链接,
|
|
718
|
+
* QQ 客户端会把裸链接变成可点的蓝色链接,复制粘贴也方便。
|
|
719
|
+
*/
|
|
720
|
+
const officialQq = /^qq/i.test(String(session.platform ?? ''));
|
|
721
|
+
if (officialQq) {
|
|
722
|
+
const { cmdInput } = await Promise.resolve().then(() => __importStar(require('./karin/module/utils/QqPanel')));
|
|
723
|
+
const table = ['| # | 标题 | UP / 作者 | 操作 |', '| :---: | :--- | :--- | :---: |'];
|
|
724
|
+
top.forEach((item, index) => {
|
|
725
|
+
const title = String(item.title || '(无标题)').replace(/[|\n]/g, ' ').slice(0, 26);
|
|
726
|
+
const author = String(item.author || '-').replace(/[|\n]/g, ' ').slice(0, 12);
|
|
727
|
+
table.push('| ' + (index + 1) + ' | ' + title + ' | ' + author + ' | ' + cmdInput('解析 ' + linkOf(item), '解析') + ' |');
|
|
728
|
+
});
|
|
729
|
+
await send([{ type: 'markdown', attrs: { content: tip + '点按钮直接解析' + String.fromCharCode(10) + table.join(String.fromCharCode(10)) } }]);
|
|
730
|
+
return;
|
|
731
|
+
}
|
|
732
|
+
const lines = [tip];
|
|
733
|
+
top.forEach((item, index) => {
|
|
734
|
+
const title = String(item.title || '(无标题)').replace(/\s+/g, ' ').slice(0, 40);
|
|
735
|
+
const author = String(item.author || '-').replace(/\s+/g, ' ').slice(0, 16);
|
|
736
|
+
lines.push((index + 1) + '. ' + title + ' —— ' + author);
|
|
737
|
+
lines.push(linkOf(item));
|
|
716
738
|
});
|
|
717
|
-
|
|
718
|
-
await send(
|
|
739
|
+
lines.push('把想看的那个链接发给我就能解析。');
|
|
740
|
+
await send(lines.join(String.fromCharCode(10)));
|
|
719
741
|
return;
|
|
720
742
|
}
|
|
721
743
|
await send('没能从这张卡片里认出作品,直接发链接给我吧');
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.resolveCardToUrl = exports.searchDouyinWorks = exports.searchBiliVideos = exports.extractUpName = exports.ocrImageText = exports.extractCardInfo = void 0;
|
|
3
|
+
exports.resolveCardToUrl = exports.searchDouyinWorks = exports.searchBiliVideos = exports.extractUpNames = exports.extractUpName = exports.ocrImageText = exports.extractCardInfo = void 0;
|
|
4
4
|
/**
|
|
5
5
|
* QQ 卡片消息解析(小程序卡片 / 分享卡片)。
|
|
6
6
|
*
|
|
@@ -239,19 +239,66 @@ exports.ocrImageText = ocrImageText;
|
|
|
239
239
|
* 从 OCR 文本里认 UP 主名。
|
|
240
240
|
* B站个人卡片的排版是「昵称 / UP主 / 粉丝数…」,所以拿「UP主」上一行最稳。
|
|
241
241
|
*/
|
|
242
|
-
const extractUpName = (text) =>
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
242
|
+
const extractUpName = (text) => (0, exports.extractUpNames)(text)[0] ?? '';
|
|
243
|
+
exports.extractUpName = extractUpName;
|
|
244
|
+
/** 明显不是昵称的标签词 */
|
|
245
|
+
const NICK_LABELS = new Set(['up', 'up主', 'upzhu', 'v', '粉丝', '关注', '获赞', '播放', '点赞', '弹幕', '投币', '收藏', '转发', '评论', '分享', '投稿', '作品', '简介', '主页', '更多', '展开', '未知', '半身像']);
|
|
246
|
+
/**
|
|
247
|
+
* 从 OCR 文本里认**所有可能**的 UP 主名(按可能性排序)。
|
|
248
|
+
*
|
|
249
|
+
* 为什么要多个:B站个人卡片实测长这样(OCR 把结构压扁了、顺序也不一定)——
|
|
250
|
+
*
|
|
251
|
+
* 雾小霜暗区突围 1,052 半身像 UP主 4993粉丝 1,087 *未知" 5.3万播放1806点赞 11弹幕
|
|
252
|
+
*
|
|
253
|
+
* 真名是**第一行的「雾小霜暗区突围」**,而「UP主」前一行是「半身像」(封面上的字)。
|
|
254
|
+
* 只取「UP主前一行」这条老规则就会拿「半身像」去搜,标题又对不上,于是六个候选一个都不敢选。
|
|
255
|
+
* 现在两条都当候选(首行 + UP主前后行),搜索端**任一命中**就算作者对上,
|
|
256
|
+
* 谁真的搜得到就用谁 —— 不再赌某一种排版。
|
|
257
|
+
*/
|
|
258
|
+
const extractUpNames = (text) => {
|
|
259
|
+
const raw = String(text ?? '');
|
|
260
|
+
// OCR 有时整段只有一行(空格分隔),这时按空白切
|
|
261
|
+
const byLine = raw.split(/\r?\n/).map((line) => line.trim()).filter(Boolean);
|
|
262
|
+
const tokens = byLine.length >= 2 ? byLine : raw.split(/[\s\u3000|/]+/).map((item) => item.trim()).filter(Boolean);
|
|
263
|
+
const isStat = (line) => /(粉丝|播放|点赞|弹幕|投币|收藏|关注|转发|评论|分享|投稿)/.test(line) ||
|
|
264
|
+
/^[\d,.]+(\.\d+)?[万亿]?$/.test(line) ||
|
|
265
|
+
/^[**·.]+$/.test(line);
|
|
266
|
+
const clean = (line) => line
|
|
267
|
+
.replace(/^[**·\s]+/, '')
|
|
268
|
+
.replace(/[**·\s]+$/, '')
|
|
269
|
+
.replace(/["“”'']/g, '')
|
|
270
|
+
.trim();
|
|
271
|
+
const candidates = [];
|
|
272
|
+
const push = (line) => {
|
|
273
|
+
if (!line)
|
|
274
|
+
return;
|
|
275
|
+
const value = clean(line);
|
|
276
|
+
if (value.length < 2 || value.length > 20)
|
|
277
|
+
return;
|
|
278
|
+
if (isStat(value))
|
|
279
|
+
return;
|
|
280
|
+
if (NICK_LABELS.has(value.toLowerCase()))
|
|
281
|
+
return;
|
|
282
|
+
if (/^[\d,.万]+$/.test(value))
|
|
283
|
+
return;
|
|
284
|
+
if (!candidates.includes(value))
|
|
285
|
+
candidates.push(value);
|
|
286
|
+
};
|
|
287
|
+
// ① UP主 前后各一行/一个词
|
|
288
|
+
const idx = tokens.findIndex((token) => /^(up主|up|upzhu)$/i.test(clean(token)));
|
|
248
289
|
if (idx > 0)
|
|
249
|
-
|
|
250
|
-
if (idx
|
|
251
|
-
|
|
252
|
-
|
|
290
|
+
push(tokens[idx - 1]);
|
|
291
|
+
if (idx >= 0)
|
|
292
|
+
push(tokens[idx + 1]);
|
|
293
|
+
// ② 第一行(B站卡片昵称就在最上面)
|
|
294
|
+
push(tokens[0]);
|
|
295
|
+
// ③ 任何一行像「粉丝数」这种统计的**前面**一行
|
|
296
|
+
const statIdx = tokens.findIndex((token) => /粉丝|关注/.test(token));
|
|
297
|
+
if (statIdx > 0)
|
|
298
|
+
push(tokens[statIdx - 1]);
|
|
299
|
+
return candidates;
|
|
253
300
|
};
|
|
254
|
-
exports.
|
|
301
|
+
exports.extractUpNames = extractUpNames;
|
|
255
302
|
/* ------------------------------------------------------------------ *
|
|
256
303
|
* B站搜索(自带 Wbi 签名)
|
|
257
304
|
* ------------------------------------------------------------------ */
|
|
@@ -331,7 +378,13 @@ const searchBiliVideos = async (keyword, title = '', author = '', limit = 8) =>
|
|
|
331
378
|
}
|
|
332
379
|
const raw = Array.isArray(json?.data?.result) ? json.data.result : [];
|
|
333
380
|
const titleKey = normalizeText(title || keyword);
|
|
334
|
-
|
|
381
|
+
/**
|
|
382
|
+
* 作者可以有**多个候选**(卡片摘要里的那个 + OCR 认出来的那几个),
|
|
383
|
+
* 结果只要命中任意一个就算作者对上了,用命中的那个算分。
|
|
384
|
+
*/
|
|
385
|
+
const authorKeys = (Array.isArray(author) ? author : [author])
|
|
386
|
+
.map((item) => normalizeText(item))
|
|
387
|
+
.filter((item, index, list) => item.length >= 2 && list.indexOf(item) === index);
|
|
335
388
|
return raw
|
|
336
389
|
.filter((item) => item?.bvid)
|
|
337
390
|
.map((item, index) => {
|
|
@@ -342,13 +395,22 @@ const searchBiliVideos = async (keyword, title = '', author = '', limit = 8) =>
|
|
|
342
395
|
let score = 0;
|
|
343
396
|
let titleMatch = false;
|
|
344
397
|
let authorMatch = false;
|
|
345
|
-
|
|
398
|
+
let matchedAuthor = '';
|
|
399
|
+
for (const authorKey of authorKeys) {
|
|
400
|
+
if (!a)
|
|
401
|
+
break;
|
|
346
402
|
if (a === authorKey) {
|
|
347
403
|
score += 120;
|
|
348
404
|
authorMatch = true;
|
|
405
|
+
matchedAuthor = authorKey;
|
|
406
|
+
break;
|
|
349
407
|
}
|
|
350
|
-
|
|
351
|
-
|
|
408
|
+
if (a.includes(authorKey) || authorKey.includes(a)) {
|
|
409
|
+
// 模糊命中:分少一点,并且只取最好的那次
|
|
410
|
+
if (!authorMatch) {
|
|
411
|
+
score += 70;
|
|
412
|
+
matchedAuthor = authorKey;
|
|
413
|
+
}
|
|
352
414
|
authorMatch = true;
|
|
353
415
|
}
|
|
354
416
|
}
|
|
@@ -368,7 +430,7 @@ const searchBiliVideos = async (keyword, title = '', author = '', limit = 8) =>
|
|
|
368
430
|
score += Math.round(sim * 40);
|
|
369
431
|
}
|
|
370
432
|
}
|
|
371
|
-
return { bvid: String(item.bvid), title: itemTitle, author: itemAuthor, score: score - index, titleMatch, authorMatch };
|
|
433
|
+
return { bvid: String(item.bvid), title: itemTitle, author: itemAuthor, score: score - index, titleMatch, authorMatch, matchedAuthor };
|
|
372
434
|
})
|
|
373
435
|
.sort((left, right) => right.score - left.score)
|
|
374
436
|
.slice(0, Math.max(1, Math.min(20, limit)));
|
|
@@ -382,7 +444,7 @@ exports.searchBiliVideos = searchBiliVideos;
|
|
|
382
444
|
/* ------------------------------------------------------------------ *
|
|
383
445
|
* 抖音搜索(直接用 amagi 的 search 端点)
|
|
384
446
|
* ------------------------------------------------------------------ */
|
|
385
|
-
const searchDouyinWorks = async (keyword, limit = 8) => {
|
|
447
|
+
const searchDouyinWorks = async (keyword, limit = 8, authors = []) => {
|
|
386
448
|
try {
|
|
387
449
|
// 这版接口库的抖音 fetcher 不一定有 search(实测 6.6.0 上没有),没有就干脆跳过
|
|
388
450
|
const fetcher = amagiClient_1.douyinFetcher;
|
|
@@ -393,16 +455,30 @@ const searchDouyinWorks = async (keyword, limit = 8) => {
|
|
|
393
455
|
const res = await fetcher.search({ query: String(keyword ?? '').trim(), type: 'video', number: limit });
|
|
394
456
|
const list = res?.data?.data?.aweme_list ?? res?.data?.aweme_list ?? res?.aweme_list ?? [];
|
|
395
457
|
const titleKey = normalizeText(keyword);
|
|
458
|
+
const authorKeys = authors.map((item) => normalizeText(item)).filter((item) => item.length >= 2);
|
|
396
459
|
return list
|
|
397
460
|
.filter((item) => item?.aweme_id)
|
|
398
461
|
.map((item, index) => {
|
|
399
462
|
const desc = String(item.desc ?? '');
|
|
400
463
|
const sim = titleSimilarity(normalizeText(desc), titleKey);
|
|
464
|
+
// 作者对上一样加分:抖音搜索噪声大,光靠标题常常分不出是哪一条
|
|
465
|
+
const authorKey = normalizeText(String(item.author?.nickname ?? ''));
|
|
466
|
+
let authorScore = 0;
|
|
467
|
+
for (const key of authorKeys) {
|
|
468
|
+
if (!authorKey)
|
|
469
|
+
break;
|
|
470
|
+
if (authorKey === key) {
|
|
471
|
+
authorScore = 60;
|
|
472
|
+
break;
|
|
473
|
+
}
|
|
474
|
+
if (authorKey.includes(key) || key.includes(authorKey))
|
|
475
|
+
authorScore = Math.max(authorScore, 35);
|
|
476
|
+
}
|
|
401
477
|
return {
|
|
402
478
|
aweme_id: String(item.aweme_id),
|
|
403
479
|
desc,
|
|
404
480
|
author: String(item.author?.nickname ?? ''),
|
|
405
|
-
score: Math.round(sim * 100) - index
|
|
481
|
+
score: Math.round(sim * 100) + authorScore - index
|
|
406
482
|
};
|
|
407
483
|
})
|
|
408
484
|
.sort((left, right) => right.score - left.score)
|
|
@@ -429,7 +505,17 @@ const resolveCardToUrl = async (content) => {
|
|
|
429
505
|
}
|
|
430
506
|
// ② OCR 封面拿文字线索
|
|
431
507
|
const ocrText = await (0, exports.ocrImageText)(card.cover);
|
|
432
|
-
|
|
508
|
+
/**
|
|
509
|
+
* UP 主名可以有好几个来源:卡片摘要里的 author、OCR 里「UP主」前后行、OCR 首行。
|
|
510
|
+
*
|
|
511
|
+
* 实测踩过的坑:卡片摘要给的是「半身像」(封面上的字),OCR 首行才是真昵称
|
|
512
|
+
* 「雾小霜暗区突围」—— 只认一个来源时,作者永远匹配不上,六个候选一个都不敢选。
|
|
513
|
+
* 这里全部当候选,谁匹配上算谁的。
|
|
514
|
+
*/
|
|
515
|
+
const upNames = [card.author, ...(0, exports.extractUpNames)(ocrText)]
|
|
516
|
+
.map((item) => String(item ?? '').trim())
|
|
517
|
+
.filter((item, index, list) => item.length >= 2 && list.indexOf(item) === index);
|
|
518
|
+
const upName = upNames[0] ?? '';
|
|
433
519
|
const keyword = card.title || upName || ocrText.replace(/\s+/g, ' ').slice(0, 40);
|
|
434
520
|
if (!keyword) {
|
|
435
521
|
node_karin_1.logger.mark('[卡片解析] OCR 没有给出可用关键词');
|
|
@@ -441,19 +527,21 @@ const resolveCardToUrl = async (content) => {
|
|
|
441
527
|
const looksBili = /bilibili|哔哩|B站|UP主/i.test(hint);
|
|
442
528
|
// ③ 先按最可能的平台搜,命中就返回
|
|
443
529
|
const tryBili = async () => {
|
|
444
|
-
const hits = await (0, exports.searchBiliVideos)(keyword, card.title,
|
|
530
|
+
const hits = await (0, exports.searchBiliVideos)(keyword, card.title, upNames, 8);
|
|
445
531
|
const candidates = hits.slice(0, 6).map((item) => ({
|
|
446
532
|
platform: 'bilibili', id: item.bvid, title: item.title, author: item.author, score: item.score
|
|
447
533
|
}));
|
|
448
534
|
const strict = hits.filter((item) => item.authorMatch || item.titleMatch);
|
|
449
535
|
// 只有「标题和作者都命中」才敢自动继续,否则交给用户挑
|
|
450
536
|
const best = (strict.length ? strict : hits)[0];
|
|
451
|
-
if (best && best.titleMatch && best.authorMatch)
|
|
452
|
-
|
|
537
|
+
if (best && best.titleMatch && best.authorMatch) {
|
|
538
|
+
node_karin_1.logger.mark('[卡片解析] 作者命中「' + (best.matchedAuthor || '') + '」:' + best.author + ',标题: ' + best.title);
|
|
539
|
+
return { url: 'https://www.bilibili.com/video/' + best.bvid, candidates, matchedAuthor: best.author };
|
|
540
|
+
}
|
|
453
541
|
return { candidates };
|
|
454
542
|
};
|
|
455
543
|
const tryDouyin = async () => {
|
|
456
|
-
const hits = await (0, exports.searchDouyinWorks)(keyword || card.title, 8);
|
|
544
|
+
const hits = await (0, exports.searchDouyinWorks)(keyword || card.title, 8, upNames);
|
|
457
545
|
const candidates = hits.slice(0, 6).map((item) => ({
|
|
458
546
|
platform: 'douyin', id: item.aweme_id, title: item.desc, author: item.author, score: item.score
|
|
459
547
|
}));
|
|
@@ -471,7 +559,9 @@ const resolveCardToUrl = async (content) => {
|
|
|
471
559
|
candidates.push(...hit.candidates);
|
|
472
560
|
if (hit.url) {
|
|
473
561
|
node_karin_1.logger.mark('[卡片解析] 定位成功: ' + hit.url);
|
|
474
|
-
|
|
562
|
+
// 提示语里报「真正匹配上的那个 UP 名」,而不是卡片摘要里那个不准的
|
|
563
|
+
const matched = hit.matchedAuthor || upName;
|
|
564
|
+
return { url: hit.url, platform: hit.url.includes('bilibili') ? 'bilibili' : 'douyin', card, candidates, ocrText, upName: matched };
|
|
475
565
|
}
|
|
476
566
|
}
|
|
477
567
|
}
|
package/package.json
CHANGED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 卡片解析(OCR 找 UP 主)冒烟测试。
|
|
3
|
+
*
|
|
4
|
+
* 覆盖线上踩过的那个坑:B站卡片的 OCR 文本里「UP主」**上一行不是昵称**,
|
|
5
|
+
* 真名在第一行 —— 老规则只看「UP主前一行」,于是拿封面上的「半身像」去搜,
|
|
6
|
+
* 标题又对不上,六个候选一个都不敢选(用户看到的「没能唯一定位…候选 6 条」)。
|
|
7
|
+
*
|
|
8
|
+
* 这里只测纯函数(不联网):候选名的顺序、噪声过滤、去重。
|
|
9
|
+
* 用法:node scripts/smoke-cardparse.cjs
|
|
10
|
+
*/
|
|
11
|
+
const path = require('node:path')
|
|
12
|
+
|
|
13
|
+
const pluginRoot = path.resolve(__dirname, '..')
|
|
14
|
+
const libRoot = path.join(pluginRoot, 'lib')
|
|
15
|
+
|
|
16
|
+
/** 兼容层要先绑定运行时,CardParser 读配置/日志都走它 */
|
|
17
|
+
const runtime = require(path.join(libRoot, 'compat', 'runtime.js'))
|
|
18
|
+
const noop = () => {}
|
|
19
|
+
runtime.bindRuntime({
|
|
20
|
+
ctx: { get: () => undefined, logger: () => ({ info: noop, warn: noop, error: noop, debug: noop, mark: noop }), bots: [], registry: new Map() },
|
|
21
|
+
config: { app: {} },
|
|
22
|
+
dataRoot: path.join(pluginRoot, 'data-smoke-cardparse'),
|
|
23
|
+
master: () => []
|
|
24
|
+
})
|
|
25
|
+
|
|
26
|
+
const { extractUpNames, extractUpName } = require(path.join(libRoot, 'karin/module/utils/CardParser.js'))
|
|
27
|
+
|
|
28
|
+
const results = []
|
|
29
|
+
const check = (name, ok, detail) => {
|
|
30
|
+
results.push({ name, ok })
|
|
31
|
+
console.log((ok ? ' ✅ ' : ' ❌ ') + name + (detail ? ' —— ' + detail : ''))
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
console.log('\n[1] 线上那条真实 OCR(真名在第一行、「UP主」前一行是封面文字)')
|
|
35
|
+
const live = '雾小霜暗区突围\n1,052\n半身像\nUP主\n4993粉丝\n1,087\n*未知"\n5.3万播放1806点赞\n11弹幕'
|
|
36
|
+
const liveNames = extractUpNames(live)
|
|
37
|
+
check('首选就是真昵称「雾小霜暗区突围」', liveNames[0] === '雾小霜暗区突围', liveNames.join(' / '))
|
|
38
|
+
check('封面文字「半身像」不会混进来', !liveNames.includes('半身像'), liveNames.join(' / '))
|
|
39
|
+
check('统计行(4993粉丝 / 5.3万播放…)不当昵称', !liveNames.some((item) => /粉丝|播放|点赞|弹幕|^[\d,.]+$/.test(item)), liveNames.join(' / '))
|
|
40
|
+
check('旧入口 extractUpName 与首选一致', extractUpName(live) === '雾小霜暗区突围', extractUpName(live))
|
|
41
|
+
|
|
42
|
+
console.log('\n[2] OCR 把整段压成一行(空格分隔)也要认出来')
|
|
43
|
+
const flat = '雾小霜暗区突围 1,052 半身像 UP主 4993粉丝 1,087 *未知" 5.3万播放1806点赞 11弹幕'
|
|
44
|
+
check('单行也能拿到真昵称', extractUpNames(flat)[0] === '雾小霜暗区突围', JSON.stringify(extractUpNames(flat)))
|
|
45
|
+
|
|
46
|
+
console.log('\n[3] 标准 B站排版(昵称在「UP主」上一行)保持原样')
|
|
47
|
+
const classic = '影视飓风\nUP主\n123.4万粉丝\n1.2亿播放'
|
|
48
|
+
check('昵称「影视飓风」', extractUpNames(classic)[0] === '影视飓风', JSON.stringify(extractUpNames(classic)))
|
|
49
|
+
|
|
50
|
+
console.log('\n[4] 边界:空文本 / 只有统计 / 重复行')
|
|
51
|
+
check('空文本返回空数组', extractUpNames('').length === 0)
|
|
52
|
+
check('只有统计行时没有候选', extractUpNames('1.2万粉丝\n3,456\n7.8万播放').length === 0, JSON.stringify(extractUpNames('1.2万粉丝\n3,456\n7.8万播放')))
|
|
53
|
+
check('大小写不同的 UP主 都能识别', extractUpNames('某某某\nup主\n100粉丝')[0] === '某某某', JSON.stringify(extractUpNames('某某某\nup主\n100粉丝')))
|
|
54
|
+
check('候选不重复', (() => { const list = extractUpNames('小明明\nUP主\n小明明\n99粉丝'); return list.length === new Set(list).size })(), JSON.stringify(extractUpNames('小明明\nUP主\n小明明\n99粉丝')))
|
|
55
|
+
|
|
56
|
+
const failed = results.filter((item) => !item.ok)
|
|
57
|
+
console.log('\n=== ' + (results.length - failed.length) + '/' + results.length + ' 通过 ===')
|
|
58
|
+
process.exitCode = failed.length ? 1 : 0
|
package/src/index.ts
CHANGED
|
@@ -705,18 +705,40 @@ function registerCommands (
|
|
|
705
705
|
if (await runTextCommand(session, '#解析 ' + resolved.url)) return
|
|
706
706
|
}
|
|
707
707
|
if (resolved && resolved.candidates && resolved.candidates.length) {
|
|
708
|
-
const
|
|
709
|
-
const
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
708
|
+
const top = resolved.candidates.slice(0, 6)
|
|
709
|
+
const tip = '没能唯一确定这个作品(识别到:' + ((resolved.upName) || '未知') + '),下面是候选:'
|
|
710
|
+
const linkOf = (item: any) => item.platform === 'bilibili'
|
|
711
|
+
? 'https://www.bilibili.com/video/' + item.id
|
|
712
|
+
: 'https://www.douyin.com/video/' + item.id
|
|
713
|
+
/**
|
|
714
|
+
* **按平台分成两种发法**。
|
|
715
|
+
*
|
|
716
|
+
* QQ 官方机器人认 markdown,候选直接排成表格、操作列是按钮,点一下就走解析(最省事)。
|
|
717
|
+
* 个人号(OneBot / NapCat)**不渲染 markdown**:实测这条候选消息发过去整条都发不出来 ——
|
|
718
|
+
* 用户只看到「正在提取卡片信息…」然后再无音讯,六个候选白搜。那边改成纯文本 + 完整链接,
|
|
719
|
+
* QQ 客户端会把裸链接变成可点的蓝色链接,复制粘贴也方便。
|
|
720
|
+
*/
|
|
721
|
+
const officialQq = /^qq/i.test(String((session as any).platform ?? ''))
|
|
722
|
+
if (officialQq) {
|
|
723
|
+
const { cmdInput } = await import('./karin/module/utils/QqPanel')
|
|
724
|
+
const table = ['| # | 标题 | UP / 作者 | 操作 |', '| :---: | :--- | :--- | :---: |']
|
|
725
|
+
top.forEach((item: any, index: number) => {
|
|
726
|
+
const title = String(item.title || '(无标题)').replace(/[|\n]/g, ' ').slice(0, 26)
|
|
727
|
+
const author = String(item.author || '-').replace(/[|\n]/g, ' ').slice(0, 12)
|
|
728
|
+
table.push('| ' + (index + 1) + ' | ' + title + ' | ' + author + ' | ' + cmdInput('解析 ' + linkOf(item), '解析') + ' |')
|
|
729
|
+
})
|
|
730
|
+
await send([{ type: 'markdown', attrs: { content: tip + '点按钮直接解析' + String.fromCharCode(10) + table.join(String.fromCharCode(10)) } }])
|
|
731
|
+
return
|
|
732
|
+
}
|
|
733
|
+
const lines = [tip]
|
|
734
|
+
top.forEach((item: any, index: number) => {
|
|
735
|
+
const title = String(item.title || '(无标题)').replace(/\s+/g, ' ').slice(0, 40)
|
|
736
|
+
const author = String(item.author || '-').replace(/\s+/g, ' ').slice(0, 16)
|
|
737
|
+
lines.push((index + 1) + '. ' + title + ' —— ' + author)
|
|
738
|
+
lines.push(linkOf(item))
|
|
717
739
|
})
|
|
718
|
-
|
|
719
|
-
await send(
|
|
740
|
+
lines.push('把想看的那个链接发给我就能解析。')
|
|
741
|
+
await send(lines.join(String.fromCharCode(10)))
|
|
720
742
|
return
|
|
721
743
|
}
|
|
722
744
|
await send('没能从这张卡片里认出作品,直接发链接给我吧')
|
|
@@ -236,15 +236,57 @@ export const ocrImageText = async (imageUrl: string): Promise<string> => {
|
|
|
236
236
|
* 从 OCR 文本里认 UP 主名。
|
|
237
237
|
* B站个人卡片的排版是「昵称 / UP主 / 粉丝数…」,所以拿「UP主」上一行最稳。
|
|
238
238
|
*/
|
|
239
|
-
export const extractUpName = (text: string): string =>
|
|
240
|
-
|
|
241
|
-
|
|
239
|
+
export const extractUpName = (text: string): string => extractUpNames(text)[0] ?? ''
|
|
240
|
+
|
|
241
|
+
/** 明显不是昵称的标签词 */
|
|
242
|
+
const NICK_LABELS = new Set(['up', 'up主', 'upzhu', 'v', '粉丝', '关注', '获赞', '播放', '点赞', '弹幕', '投币', '收藏', '转发', '评论', '分享', '投稿', '作品', '简介', '主页', '更多', '展开', '未知', '半身像'])
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* 从 OCR 文本里认**所有可能**的 UP 主名(按可能性排序)。
|
|
246
|
+
*
|
|
247
|
+
* 为什么要多个:B站个人卡片实测长这样(OCR 把结构压扁了、顺序也不一定)——
|
|
248
|
+
*
|
|
249
|
+
* 雾小霜暗区突围 1,052 半身像 UP主 4993粉丝 1,087 *未知" 5.3万播放1806点赞 11弹幕
|
|
250
|
+
*
|
|
251
|
+
* 真名是**第一行的「雾小霜暗区突围」**,而「UP主」前一行是「半身像」(封面上的字)。
|
|
252
|
+
* 只取「UP主前一行」这条老规则就会拿「半身像」去搜,标题又对不上,于是六个候选一个都不敢选。
|
|
253
|
+
* 现在两条都当候选(首行 + UP主前后行),搜索端**任一命中**就算作者对上,
|
|
254
|
+
* 谁真的搜得到就用谁 —— 不再赌某一种排版。
|
|
255
|
+
*/
|
|
256
|
+
export const extractUpNames = (text: string): string[] => {
|
|
257
|
+
const raw = String(text ?? '')
|
|
258
|
+
// OCR 有时整段只有一行(空格分隔),这时按空白切
|
|
259
|
+
const byLine = raw.split(/\r?\n/).map((line) => line.trim()).filter(Boolean)
|
|
260
|
+
const tokens = byLine.length >= 2 ? byLine : raw.split(/[\s\u3000|/]+/).map((item) => item.trim()).filter(Boolean)
|
|
242
261
|
const isStat = (line: string) =>
|
|
243
|
-
/(
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
262
|
+
/(粉丝|播放|点赞|弹幕|投币|收藏|关注|转发|评论|分享|投稿)/.test(line) ||
|
|
263
|
+
/^[\d,.]+(\.\d+)?[万亿]?$/.test(line) ||
|
|
264
|
+
/^[**·.]+$/.test(line)
|
|
265
|
+
const clean = (line: string): string => line
|
|
266
|
+
.replace(/^[**·\s]+/, '')
|
|
267
|
+
.replace(/[**·\s]+$/, '')
|
|
268
|
+
.replace(/["“”'']/g, '')
|
|
269
|
+
.trim()
|
|
270
|
+
const candidates: string[] = []
|
|
271
|
+
const push = (line?: string) => {
|
|
272
|
+
if (!line) return
|
|
273
|
+
const value = clean(line)
|
|
274
|
+
if (value.length < 2 || value.length > 20) return
|
|
275
|
+
if (isStat(value)) return
|
|
276
|
+
if (NICK_LABELS.has(value.toLowerCase())) return
|
|
277
|
+
if (/^[\d,.万]+$/.test(value)) return
|
|
278
|
+
if (!candidates.includes(value)) candidates.push(value)
|
|
279
|
+
}
|
|
280
|
+
// ① UP主 前后各一行/一个词
|
|
281
|
+
const idx = tokens.findIndex((token) => /^(up主|up|upzhu)$/i.test(clean(token)))
|
|
282
|
+
if (idx > 0) push(tokens[idx - 1])
|
|
283
|
+
if (idx >= 0) push(tokens[idx + 1])
|
|
284
|
+
// ② 第一行(B站卡片昵称就在最上面)
|
|
285
|
+
push(tokens[0])
|
|
286
|
+
// ③ 任何一行像「粉丝数」这种统计的**前面**一行
|
|
287
|
+
const statIdx = tokens.findIndex((token) => /粉丝|关注/.test(token))
|
|
288
|
+
if (statIdx > 0) push(tokens[statIdx - 1])
|
|
289
|
+
return candidates
|
|
248
290
|
}
|
|
249
291
|
|
|
250
292
|
/* ------------------------------------------------------------------ *
|
|
@@ -305,9 +347,9 @@ const getWbiKeys = async (): Promise<{ imgKey: string; subKey: string; buvid3: s
|
|
|
305
347
|
export const searchBiliVideos = async (
|
|
306
348
|
keyword: string,
|
|
307
349
|
title = '',
|
|
308
|
-
author = '',
|
|
350
|
+
author: string | string[] = '',
|
|
309
351
|
limit = 8
|
|
310
|
-
): Promise<Array<{ bvid: string; title: string; author: string; score: number; titleMatch: boolean; authorMatch: boolean }>> => {
|
|
352
|
+
): Promise<Array<{ bvid: string; title: string; author: string; score: number; titleMatch: boolean; authorMatch: boolean; matchedAuthor: string }>> => {
|
|
311
353
|
try {
|
|
312
354
|
const { imgKey, subKey, buvid3 } = await getWbiKeys()
|
|
313
355
|
const mixinKey = MIXIN_KEY_ENC_TAB.map((n) => (imgKey + subKey)[n]).join('').slice(0, 32)
|
|
@@ -329,7 +371,13 @@ export const searchBiliVideos = async (
|
|
|
329
371
|
}
|
|
330
372
|
const raw = Array.isArray(json?.data?.result) ? json.data.result : []
|
|
331
373
|
const titleKey = normalizeText(title || keyword)
|
|
332
|
-
|
|
374
|
+
/**
|
|
375
|
+
* 作者可以有**多个候选**(卡片摘要里的那个 + OCR 认出来的那几个),
|
|
376
|
+
* 结果只要命中任意一个就算作者对上了,用命中的那个算分。
|
|
377
|
+
*/
|
|
378
|
+
const authorKeys = (Array.isArray(author) ? author : [author])
|
|
379
|
+
.map((item) => normalizeText(item))
|
|
380
|
+
.filter((item, index, list) => item.length >= 2 && list.indexOf(item) === index)
|
|
333
381
|
return raw
|
|
334
382
|
.filter((item: any) => item?.bvid)
|
|
335
383
|
.map((item: any, index: number) => {
|
|
@@ -340,9 +388,15 @@ export const searchBiliVideos = async (
|
|
|
340
388
|
let score = 0
|
|
341
389
|
let titleMatch = false
|
|
342
390
|
let authorMatch = false
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
391
|
+
let matchedAuthor = ''
|
|
392
|
+
for (const authorKey of authorKeys) {
|
|
393
|
+
if (!a) break
|
|
394
|
+
if (a === authorKey) { score += 120; authorMatch = true; matchedAuthor = authorKey; break }
|
|
395
|
+
if (a.includes(authorKey) || authorKey.includes(a)) {
|
|
396
|
+
// 模糊命中:分少一点,并且只取最好的那次
|
|
397
|
+
if (!authorMatch) { score += 70; matchedAuthor = authorKey }
|
|
398
|
+
authorMatch = true
|
|
399
|
+
}
|
|
346
400
|
}
|
|
347
401
|
if (titleKey && t) {
|
|
348
402
|
if (t === titleKey) { score += 100; titleMatch = true }
|
|
@@ -353,7 +407,7 @@ export const searchBiliVideos = async (
|
|
|
353
407
|
score += Math.round(sim * 40)
|
|
354
408
|
}
|
|
355
409
|
}
|
|
356
|
-
return { bvid: String(item.bvid), title: itemTitle, author: itemAuthor, score: score - index, titleMatch, authorMatch }
|
|
410
|
+
return { bvid: String(item.bvid), title: itemTitle, author: itemAuthor, score: score - index, titleMatch, authorMatch, matchedAuthor }
|
|
357
411
|
})
|
|
358
412
|
.sort((left: any, right: any) => right.score - left.score)
|
|
359
413
|
.slice(0, Math.max(1, Math.min(20, limit)))
|
|
@@ -367,7 +421,11 @@ export const searchBiliVideos = async (
|
|
|
367
421
|
* 抖音搜索(直接用 amagi 的 search 端点)
|
|
368
422
|
* ------------------------------------------------------------------ */
|
|
369
423
|
|
|
370
|
-
export const searchDouyinWorks = async (
|
|
424
|
+
export const searchDouyinWorks = async (
|
|
425
|
+
keyword: string,
|
|
426
|
+
limit = 8,
|
|
427
|
+
authors: string[] = []
|
|
428
|
+
): Promise<Array<{ aweme_id: string; desc: string; author: string; score: number }>> => {
|
|
371
429
|
try {
|
|
372
430
|
// 这版接口库的抖音 fetcher 不一定有 search(实测 6.6.0 上没有),没有就干脆跳过
|
|
373
431
|
const fetcher: any = douyinFetcher as any
|
|
@@ -379,16 +437,25 @@ export const searchDouyinWorks = async (keyword: string, limit = 8): Promise<Arr
|
|
|
379
437
|
const list: any[] =
|
|
380
438
|
res?.data?.data?.aweme_list ?? res?.data?.aweme_list ?? res?.aweme_list ?? []
|
|
381
439
|
const titleKey = normalizeText(keyword)
|
|
440
|
+
const authorKeys = authors.map((item) => normalizeText(item)).filter((item) => item.length >= 2)
|
|
382
441
|
return list
|
|
383
442
|
.filter((item: any) => item?.aweme_id)
|
|
384
443
|
.map((item: any, index: number) => {
|
|
385
444
|
const desc = String(item.desc ?? '')
|
|
386
445
|
const sim = titleSimilarity(normalizeText(desc), titleKey)
|
|
446
|
+
// 作者对上一样加分:抖音搜索噪声大,光靠标题常常分不出是哪一条
|
|
447
|
+
const authorKey = normalizeText(String(item.author?.nickname ?? ''))
|
|
448
|
+
let authorScore = 0
|
|
449
|
+
for (const key of authorKeys) {
|
|
450
|
+
if (!authorKey) break
|
|
451
|
+
if (authorKey === key) { authorScore = 60; break }
|
|
452
|
+
if (authorKey.includes(key) || key.includes(authorKey)) authorScore = Math.max(authorScore, 35)
|
|
453
|
+
}
|
|
387
454
|
return {
|
|
388
455
|
aweme_id: String(item.aweme_id),
|
|
389
456
|
desc,
|
|
390
457
|
author: String(item.author?.nickname ?? ''),
|
|
391
|
-
score: Math.round(sim * 100) - index
|
|
458
|
+
score: Math.round(sim * 100) + authorScore - index
|
|
392
459
|
}
|
|
393
460
|
})
|
|
394
461
|
.sort((left, right) => right.score - left.score)
|
|
@@ -443,7 +510,17 @@ export const resolveCardToUrl = async (
|
|
|
443
510
|
|
|
444
511
|
// ② OCR 封面拿文字线索
|
|
445
512
|
const ocrText = await ocrImageText(card.cover)
|
|
446
|
-
|
|
513
|
+
/**
|
|
514
|
+
* UP 主名可以有好几个来源:卡片摘要里的 author、OCR 里「UP主」前后行、OCR 首行。
|
|
515
|
+
*
|
|
516
|
+
* 实测踩过的坑:卡片摘要给的是「半身像」(封面上的字),OCR 首行才是真昵称
|
|
517
|
+
* 「雾小霜暗区突围」—— 只认一个来源时,作者永远匹配不上,六个候选一个都不敢选。
|
|
518
|
+
* 这里全部当候选,谁匹配上算谁的。
|
|
519
|
+
*/
|
|
520
|
+
const upNames = [card.author, ...extractUpNames(ocrText)]
|
|
521
|
+
.map((item) => String(item ?? '').trim())
|
|
522
|
+
.filter((item, index, list) => item.length >= 2 && list.indexOf(item) === index)
|
|
523
|
+
const upName = upNames[0] ?? ''
|
|
447
524
|
const keyword = card.title || upName || ocrText.replace(/\s+/g, ' ').slice(0, 40)
|
|
448
525
|
if (!keyword) {
|
|
449
526
|
logger.mark('[卡片解析] OCR 没有给出可用关键词')
|
|
@@ -456,19 +533,22 @@ export const resolveCardToUrl = async (
|
|
|
456
533
|
const looksBili = /bilibili|哔哩|B站|UP主/i.test(hint)
|
|
457
534
|
|
|
458
535
|
// ③ 先按最可能的平台搜,命中就返回
|
|
459
|
-
const tryBili = async (): Promise<{ url?: string; candidates: CardCandidate[] }> => {
|
|
460
|
-
const hits = await searchBiliVideos(keyword, card.title,
|
|
536
|
+
const tryBili = async (): Promise<{ url?: string; candidates: CardCandidate[]; matchedAuthor?: string }> => {
|
|
537
|
+
const hits = await searchBiliVideos(keyword, card.title, upNames, 8)
|
|
461
538
|
const candidates = hits.slice(0, 6).map((item) => ({
|
|
462
539
|
platform: 'bilibili' as const, id: item.bvid, title: item.title, author: item.author, score: item.score
|
|
463
540
|
}))
|
|
464
541
|
const strict = hits.filter((item) => item.authorMatch || item.titleMatch)
|
|
465
542
|
// 只有「标题和作者都命中」才敢自动继续,否则交给用户挑
|
|
466
543
|
const best = (strict.length ? strict : hits)[0]
|
|
467
|
-
if (best && best.titleMatch && best.authorMatch)
|
|
544
|
+
if (best && best.titleMatch && best.authorMatch) {
|
|
545
|
+
logger.mark('[卡片解析] 作者命中「' + (best.matchedAuthor || '') + '」:' + best.author + ',标题: ' + best.title)
|
|
546
|
+
return { url: 'https://www.bilibili.com/video/' + best.bvid, candidates, matchedAuthor: best.author }
|
|
547
|
+
}
|
|
468
548
|
return { candidates }
|
|
469
549
|
}
|
|
470
550
|
const tryDouyin = async (): Promise<{ url?: string; candidates: CardCandidate[] }> => {
|
|
471
|
-
const hits = await searchDouyinWorks(keyword || card.title, 8)
|
|
551
|
+
const hits = await searchDouyinWorks(keyword || card.title, 8, upNames)
|
|
472
552
|
const candidates = hits.slice(0, 6).map((item) => ({
|
|
473
553
|
platform: 'douyin' as const, id: item.aweme_id, title: item.desc, author: item.author, score: item.score
|
|
474
554
|
}))
|
|
@@ -486,7 +566,9 @@ export const resolveCardToUrl = async (
|
|
|
486
566
|
candidates.push(...hit.candidates)
|
|
487
567
|
if (hit.url) {
|
|
488
568
|
logger.mark('[卡片解析] 定位成功: ' + hit.url)
|
|
489
|
-
|
|
569
|
+
// 提示语里报「真正匹配上的那个 UP 名」,而不是卡片摘要里那个不准的
|
|
570
|
+
const matched = (hit as any).matchedAuthor || upName
|
|
571
|
+
return { url: hit.url, platform: hit.url.includes('bilibili') ? 'bilibili' : 'douyin', card, candidates, ocrText, upName: matched }
|
|
490
572
|
}
|
|
491
573
|
}
|
|
492
574
|
}
|