pixivflow 3.0.3 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/TopicCommand.js +41 -9
- package/dist/config/types.d.ts +54 -0
- package/dist/config/types.js +3 -0
- package/dist/config/validation.js +44 -0
- package/dist/domain/media/NovelCoverPolicy.d.ts +28 -0
- package/dist/domain/media/NovelCoverPolicy.js +70 -0
- package/dist/download/DownloadManager.js +5 -1
- package/dist/download/NovelDownloader.d.ts +15 -4
- package/dist/download/NovelDownloader.js +43 -11
- package/dist/download/novelCover.d.ts +5 -6
- package/dist/download/novelCover.js +9 -12
- package/dist/package.json +1 -1
- package/dist/topic/TopicPipeline.d.ts +38 -7
- package/dist/topic/TopicPipeline.js +177 -30
- package/dist/topic/TopicResolver.js +4 -1
- package/dist/topic/TopicTagScorer.js +12 -1
- package/dist/topic/types.d.ts +48 -0
- package/dist/version.js +1 -1
- package/dist/webui/package.json +1 -1
- package/examples/gateway/README.md +4 -0
- package/examples/onebot-adapter/README.md +139 -0
- package/examples/onebot-adapter/server.mjs +612 -0
- package/package.json +2 -1
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.TopicPipeline = void 0;
|
|
4
|
+
exports.seedFallbackTag = seedFallbackTag;
|
|
5
|
+
exports.selectWalkedTags = selectWalkedTags;
|
|
6
|
+
exports.recallChannels = recallChannels;
|
|
4
7
|
const promises_1 = require("node:timers/promises");
|
|
5
8
|
const logger_1 = require("../logger");
|
|
6
9
|
const pixiv_utils_1 = require("../utils/pixiv-utils");
|
|
@@ -14,6 +17,105 @@ const STOP_TAGS = new Set([
|
|
|
14
17
|
'1000users入り', '5000users入り', '10000users入り', '500users入り', '100users入り',
|
|
15
18
|
'pixiv', 'commission', 'skeb', '依頼絵', '仕事絵',
|
|
16
19
|
]);
|
|
20
|
+
/** Normalization shared by the tag-relation filter and the pipeline itself. */
|
|
21
|
+
function normalizeKey(value) {
|
|
22
|
+
return value.trim().normalize('NFKC').toLocaleLowerCase();
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* The tag the operator asked for, even when `tagRelations` filters it out.
|
|
26
|
+
*
|
|
27
|
+
* `deny` may drop the seed, but a space filtered down to nothing must still be
|
|
28
|
+
* usable: the pipeline then falls back to the seed tag alone, exactly like the
|
|
29
|
+
* resolver's `seedOnly()` degradation. `weight` defaults to the tag's `score`
|
|
30
|
+
* for spaces persisted before provenance existed.
|
|
31
|
+
*/
|
|
32
|
+
function seedFallbackTag(seedTag) {
|
|
33
|
+
if (seedTag.source === 'seed' && seedTag.weight !== undefined)
|
|
34
|
+
return seedTag;
|
|
35
|
+
return { ...seedTag, source: 'seed', weight: seedTag.weight ?? seedTag.score, seed: true };
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Applies `topicDiscovery.tagRelations` to a resolved space (§tag-provenance).
|
|
39
|
+
*
|
|
40
|
+
* Deny beats allow, and deny is the only thing that can drop the seed tag. A
|
|
41
|
+
* tag whose provenance is unknown (a space persisted before `source` existed, or
|
|
42
|
+
* a hand-written space) is treated as `cooccurrence` so that `allowSources` can
|
|
43
|
+
* still narrow it while the default (all sources) keeps walking everything.
|
|
44
|
+
*/
|
|
45
|
+
function selectWalkedTags(tags, seedKey, relations = {}) {
|
|
46
|
+
const seedTag = tags.find((tag) => normalizeKey(tag.name) === seedKey);
|
|
47
|
+
const allowed = tags.filter((tag) => normalizeKey(tag.name) !== seedKey);
|
|
48
|
+
const deny = (relations.deny ?? []).map(normalizeKey);
|
|
49
|
+
const deniedKeys = new Set(deny);
|
|
50
|
+
const seedDenied = seedTag !== undefined && deniedKeys.has(seedKey);
|
|
51
|
+
if (seedDenied)
|
|
52
|
+
deniedKeys.clear(); // the seed fallback ignores deny
|
|
53
|
+
const allow = (relations.allow ?? []).map(normalizeKey);
|
|
54
|
+
const allowKeys = new Set(allow);
|
|
55
|
+
const open = allowKeys.size === 0;
|
|
56
|
+
const sources = relations.allowSources;
|
|
57
|
+
const sourceAllowed = (tag) => {
|
|
58
|
+
if (!sources || sources.length === 0)
|
|
59
|
+
return true;
|
|
60
|
+
const name = normalizeKey(tag.name);
|
|
61
|
+
if (allowKeys.has(name))
|
|
62
|
+
return true; // an explicit allow beats allowSources
|
|
63
|
+
const provenance = tag.source;
|
|
64
|
+
if (provenance === undefined)
|
|
65
|
+
return sources.includes('cooccurrence');
|
|
66
|
+
if (provenance === 'cooccurrence+autocomplete') {
|
|
67
|
+
return sources.includes('cooccurrence') || sources.includes('autocomplete');
|
|
68
|
+
}
|
|
69
|
+
return sources.includes(provenance);
|
|
70
|
+
};
|
|
71
|
+
const walked = allowed.filter((tag) => {
|
|
72
|
+
const name = normalizeKey(tag.name);
|
|
73
|
+
if (deniedKeys.has(name))
|
|
74
|
+
return false;
|
|
75
|
+
if (!open && !allowKeys.has(name))
|
|
76
|
+
return false;
|
|
77
|
+
return sourceAllowed(tag);
|
|
78
|
+
});
|
|
79
|
+
// No seed tag in the space at all: there is nothing to dominate, so fall back
|
|
80
|
+
// to walking the whole (filtered) space — the pre-provenance behaviour.
|
|
81
|
+
if (!seedTag)
|
|
82
|
+
return walked;
|
|
83
|
+
// A denied seed empties the space, exactly like the resolver's seedOnly():
|
|
84
|
+
// the operator still gets their own topic, and no related tag is walked.
|
|
85
|
+
if (seedDenied)
|
|
86
|
+
return [seedFallbackTag(seedTag)];
|
|
87
|
+
return [seedTag, ...walked];
|
|
88
|
+
}
|
|
89
|
+
/** Every spelling under which a set of resolved tags can be recognized. */
|
|
90
|
+
function buildTagAliases(tags, matchTranslatedNames) {
|
|
91
|
+
const aliases = new Map();
|
|
92
|
+
for (const tag of tags) {
|
|
93
|
+
const primary = normalizeKey(tag.name);
|
|
94
|
+
const entry = { primary, weight: tag.weight ?? tag.score };
|
|
95
|
+
if (!aliases.has(primary))
|
|
96
|
+
aliases.set(primary, entry);
|
|
97
|
+
if (!matchTranslatedNames)
|
|
98
|
+
continue;
|
|
99
|
+
// The work-side `translated_name` for this tag is the same resolved tag: the
|
|
100
|
+
// space still exposes one tag, it is simply recognized under two keys.
|
|
101
|
+
const translated = tag.translatedName ? normalizeKey(tag.translatedName) : '';
|
|
102
|
+
if (translated && !aliases.has(translated))
|
|
103
|
+
aliases.set(translated, entry);
|
|
104
|
+
}
|
|
105
|
+
return aliases;
|
|
106
|
+
}
|
|
107
|
+
/** Which of the walked tags a `relatedTags` mode will actually search today. */
|
|
108
|
+
function recallChannels(walked, seedKey, relatedMode) {
|
|
109
|
+
const seedTags = walked.filter((tag) => normalizeKey(tag.name) === seedKey);
|
|
110
|
+
// A space with no seed tag has nothing to protect: every remaining tag is the
|
|
111
|
+
// only channel there is, so the whole space is walked whatever the mode
|
|
112
|
+
// (pre-provenance fallback for a hand-written space).
|
|
113
|
+
if (seedTags.length === 0)
|
|
114
|
+
return [...walked];
|
|
115
|
+
if (relatedMode === 'never' || relatedMode === 'when_seed_insufficient')
|
|
116
|
+
return seedTags;
|
|
117
|
+
return [...seedTags, ...walked.filter((tag) => normalizeKey(tag.name) !== seedKey)];
|
|
118
|
+
}
|
|
17
119
|
/**
|
|
18
120
|
* Resolves a topic to a tag space, collects that day's works across the tags,
|
|
19
121
|
* filters by lightweight metadata relevance and ranks by local popularity, then
|
|
@@ -42,7 +144,6 @@ class TopicPipeline {
|
|
|
42
144
|
async selectWorks(target, contentType, day, limit, discovery, collect) {
|
|
43
145
|
const topic = (target.topic ?? '').trim();
|
|
44
146
|
const { space } = await this.resolver.resolve(topic, contentType, discovery);
|
|
45
|
-
const tagScores = new Map(space.tags.map((t) => [this.key(t.name), t.score]));
|
|
46
147
|
const maxPerTag = this.bound(collect.maxPerTag, COLLECT_DEFAULTS.maxPerTag, 5, 100);
|
|
47
148
|
const maxCandidates = this.bound(collect.maxCandidates, COLLECT_DEFAULTS.maxCandidates, 20, 500);
|
|
48
149
|
const minMetadataScore = collect.minMetadataScore ?? COLLECT_DEFAULTS.minMetadataScore;
|
|
@@ -55,11 +156,22 @@ class TopicPipeline {
|
|
|
55
156
|
let rawCount = 0;
|
|
56
157
|
let aiExcludedCount = 0;
|
|
57
158
|
let duplicateRemovedCount = 0;
|
|
58
|
-
const tagNames = space.tags.map((t) => t.name);
|
|
59
159
|
const seedKey = this.key(topic);
|
|
160
|
+
// §tag-provenance: apply tagRelations where the walked list is built, so the
|
|
161
|
+
// filter governs seedTags/relatedTags BEFORE any search is issued. The seed
|
|
162
|
+
// tag survives allow/allowSources; a deny entry that empties the space falls
|
|
163
|
+
// back to the seed tag alone, exactly like the resolver's seedOnly().
|
|
164
|
+
const matchTranslatedNames = discovery.matchTranslatedNames === true;
|
|
165
|
+
const walked = selectWalkedTags(space.tags, seedKey, discovery.tagRelations);
|
|
166
|
+
const tagNames = walked.map((t) => t.name);
|
|
60
167
|
const seedTags = tagNames.filter((name) => this.key(name) === seedKey);
|
|
61
168
|
const relatedTags = tagNames.filter((name) => this.key(name) !== seedKey);
|
|
169
|
+
// Ranking reads the documented semantic weight, falling back to `score` for
|
|
170
|
+
// spaces persisted before provenance existed. The two are always equal.
|
|
171
|
+
const tagScores = buildTagAliases(walked, matchTranslatedNames);
|
|
62
172
|
const searchedTags = [];
|
|
173
|
+
// A hard seed tier: 'on' always, and the seed-first modes keep their layer.
|
|
174
|
+
const hardSeedTier = discovery.seedTier === 'on' || relatedMode !== 'always';
|
|
63
175
|
const collectTag = async (tag) => {
|
|
64
176
|
// Cancellation is checked between tags, so a cancelled run stops issuing
|
|
65
177
|
// new searches even when the aborted request itself had already returned.
|
|
@@ -100,11 +212,11 @@ class TopicPipeline {
|
|
|
100
212
|
if (seedTags.length === 0 || relatedMode === 'always') {
|
|
101
213
|
// No seed tag in the space (hand-written space): keep walking everything
|
|
102
214
|
// rather than returning nothing.
|
|
103
|
-
await runTags(
|
|
215
|
+
await runTags(recallChannels(walked, seedKey, relatedMode).map((t) => t.name));
|
|
104
216
|
}
|
|
105
217
|
else {
|
|
106
218
|
await runTags(seedTags);
|
|
107
|
-
const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType).length;
|
|
219
|
+
const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames).length;
|
|
108
220
|
if (relatedMode === 'never') {
|
|
109
221
|
logger_1.logger.info('[TopicRecall] mode=never tag=' + topic + ' day=' + day + ' accepted=' + seedAccepted);
|
|
110
222
|
}
|
|
@@ -118,19 +230,19 @@ class TopicPipeline {
|
|
|
118
230
|
}
|
|
119
231
|
const dedupedCount = byId.size;
|
|
120
232
|
logger_1.logger.info('[TopicCollector] type=' + contentType + ' raw=' + rawCount + ' deduplicated=' + dedupedCount + ' aiExcluded=' + aiExcludedCount + ' searchedTags=' + searchedTags.length);
|
|
121
|
-
const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType);
|
|
122
|
-
const chosen = this.topByPopularity(accepted, limit,
|
|
233
|
+
const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames);
|
|
234
|
+
const chosen = this.topByPopularity(accepted, limit, hardSeedTier ? seedKey : undefined);
|
|
123
235
|
const selected = chosen.map((e) => e.candidate);
|
|
124
236
|
logger_1.logger.info('[MetadataTopicFilter] accepted=' + accepted.length);
|
|
125
237
|
if (selected[0]) {
|
|
126
|
-
logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + ' title=' + selected[0].title);
|
|
238
|
+
logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + (hardSeedTier ? ' seedTier=on' : '') + ' title=' + selected[0].title);
|
|
127
239
|
}
|
|
128
240
|
return {
|
|
129
241
|
works: chosen.map((e) => e.work),
|
|
130
242
|
selection: {
|
|
131
243
|
candidates: [...byId.values()].map((e) => e.candidate),
|
|
132
244
|
selected,
|
|
133
|
-
resolvedTagCount:
|
|
245
|
+
resolvedTagCount: walked.length,
|
|
134
246
|
searchedTags,
|
|
135
247
|
rawCount,
|
|
136
248
|
dedupedCount,
|
|
@@ -163,12 +275,16 @@ class TopicPipeline {
|
|
|
163
275
|
}
|
|
164
276
|
toCandidate(work, type) {
|
|
165
277
|
const popularity = (0, pixiv_utils_1.calculatePopularityScore)(work);
|
|
278
|
+
const translatedTags = (work.tags ?? [])
|
|
279
|
+
.map((t) => t.translated_name)
|
|
280
|
+
.filter((name) => Boolean(name));
|
|
166
281
|
return {
|
|
167
282
|
id: work.id,
|
|
168
283
|
type,
|
|
169
284
|
title: work.title ?? '',
|
|
170
285
|
caption: work.caption ?? '',
|
|
171
286
|
tags: (work.tags ?? []).map((t) => t.name).filter(Boolean),
|
|
287
|
+
...(translatedTags.length > 0 ? { translatedTags } : {}),
|
|
172
288
|
bookmarks: Number(work.total_bookmarks ?? work.bookmark_count ?? 0) || 0,
|
|
173
289
|
views: Number(work.total_view ?? work.view_count ?? 0) || 0,
|
|
174
290
|
popularity,
|
|
@@ -184,16 +300,23 @@ class TopicPipeline {
|
|
|
184
300
|
* the seed pass can be evaluated before deciding whether the related channel
|
|
185
301
|
* is needed at all (§topic-recall).
|
|
186
302
|
*/
|
|
187
|
-
acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType) {
|
|
303
|
+
acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames) {
|
|
188
304
|
const accepted = [];
|
|
305
|
+
const hitsSeed = (candidate) => {
|
|
306
|
+
if (candidate.tags.some((t) => this.key(t) === seedKey))
|
|
307
|
+
return true;
|
|
308
|
+
if (!matchTranslatedNames)
|
|
309
|
+
return false;
|
|
310
|
+
return (candidate.translatedTags ?? []).some((t) => this.key(t) === seedKey);
|
|
311
|
+
};
|
|
189
312
|
for (const entry of byId.values()) {
|
|
190
|
-
entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores);
|
|
313
|
+
entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores, matchTranslatedNames);
|
|
191
314
|
if (entry.candidate.metadataScore >= minMetadataScore)
|
|
192
315
|
accepted.push(entry);
|
|
193
316
|
}
|
|
194
317
|
if (accepted.length === 0 && byId.size > 0) {
|
|
195
318
|
const fallback = [...byId.values()]
|
|
196
|
-
.filter((e) => e.candidate
|
|
319
|
+
.filter((e) => hitsSeed(e.candidate))
|
|
197
320
|
.sort((a, b) => b.candidate.popularity - a.candidate.popularity);
|
|
198
321
|
accepted.push(...fallback.slice(0, Math.max(limit, 1)));
|
|
199
322
|
logger_1.logger.warn('[MetadataTopicFilter] type=' + contentType + ' none above threshold ' + minMetadataScore + '; kept ' + accepted.length + ' seed-tag fallback');
|
|
@@ -204,25 +327,49 @@ class TopicPipeline {
|
|
|
204
327
|
* Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
|
|
205
328
|
* title/caption add smaller boosts. The seed tag is strong evidence.
|
|
206
329
|
* No text model — case/symbol-insensitive substring matching only.
|
|
330
|
+
*
|
|
331
|
+
* `matchTranslatedNames` (default false) additionally counts a work's
|
|
332
|
+
* `translated_name` as a hit for the resolved tag it translates to, so a
|
|
333
|
+
* topic written in one language can still match works tagged in another. The
|
|
334
|
+
* translated name is matched against the resolved space — never added to
|
|
335
|
+
* `candidate.tags` — so the work is not reported as carrying a tag it lacks.
|
|
336
|
+
*
|
|
337
|
+
* A resolved tag counts at most ONCE per work, whether it matched through the
|
|
338
|
+
* tag name or through its translation (and a work that repeats a tag, as Pixiv
|
|
339
|
+
* payloads do for a tag plus its translation, does not double its weight).
|
|
207
340
|
*/
|
|
208
|
-
metadataScore(candidate, seedKey, tagScores) {
|
|
341
|
+
metadataScore(candidate, seedKey, tagScores, matchTranslatedNames) {
|
|
209
342
|
let seedHit = false;
|
|
210
343
|
let relatedSum = 0;
|
|
211
344
|
let relatedHits = 0;
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
345
|
+
let strongRelated = false;
|
|
346
|
+
// A resolved tag is matched through its own name, and — when the caller
|
|
347
|
+
// opted in — through the work-side translation of that tag. Either spelling
|
|
348
|
+
// maps back to the SAME resolved tag, so a work carrying both spellings
|
|
349
|
+
// counts once and never gets a double weight.
|
|
350
|
+
const countedTagKeys = new Set();
|
|
351
|
+
const consider = (candidateKey) => {
|
|
352
|
+
if (candidateKey === '' || STOP_TAGS.has(candidateKey))
|
|
353
|
+
return;
|
|
354
|
+
if (candidateKey === seedKey) {
|
|
217
355
|
seedHit = true;
|
|
218
|
-
|
|
219
|
-
}
|
|
220
|
-
const related = tagScores.get(k);
|
|
221
|
-
if (related !== undefined) {
|
|
222
|
-
relatedSum += Math.min(related, 0.6);
|
|
223
|
-
relatedHits += 1;
|
|
356
|
+
return;
|
|
224
357
|
}
|
|
358
|
+
const aliases = tagScores.get(candidateKey);
|
|
359
|
+
if (!aliases || countedTagKeys.has(aliases.primary))
|
|
360
|
+
return;
|
|
361
|
+
countedTagKeys.add(aliases.primary);
|
|
362
|
+
relatedSum += Math.min(aliases.weight, 0.6);
|
|
363
|
+
relatedHits += 1;
|
|
364
|
+
if (aliases.weight >= 0.6)
|
|
365
|
+
strongRelated = true;
|
|
366
|
+
};
|
|
367
|
+
if (matchTranslatedNames) {
|
|
368
|
+
for (const tag of candidate.translatedTags ?? [])
|
|
369
|
+
consider(this.key(tag));
|
|
225
370
|
}
|
|
371
|
+
for (const tag of candidate.tags)
|
|
372
|
+
consider(this.key(tag));
|
|
226
373
|
const hayTitle = this.normalize(candidate.title);
|
|
227
374
|
const hayCaption = this.normalize(candidate.caption);
|
|
228
375
|
const titleSeed = !!seedKey && hayTitle.includes(seedKey);
|
|
@@ -239,7 +386,6 @@ class TopicPipeline {
|
|
|
239
386
|
// deliberately does NOT clear the bar, so a hugely popular tangential work
|
|
240
387
|
// cannot crowd out core topic works.
|
|
241
388
|
let score = 0;
|
|
242
|
-
const strongRelated = [...tagScores.entries()].some(([k, w]) => w >= 0.6 && candidate.tags.some((t) => this.key(t) === k));
|
|
243
389
|
if (titleSeed)
|
|
244
390
|
score += 0.8;
|
|
245
391
|
if (captionSeed)
|
|
@@ -256,12 +402,13 @@ class TopicPipeline {
|
|
|
256
402
|
* alone. A work with a higher metadata score does NOT outrank a more popular
|
|
257
403
|
* accepted work.
|
|
258
404
|
*
|
|
259
|
-
* `seedKey` adds a single tier in front of that popularity order and is
|
|
260
|
-
* passed by the seed-first recall modes (§topic-recall)
|
|
261
|
-
*
|
|
262
|
-
*
|
|
263
|
-
*
|
|
264
|
-
* is
|
|
405
|
+
* `seedKey` adds a single tier in front of that popularity order and is
|
|
406
|
+
* passed by the seed-first recall modes (§topic-recall) and by
|
|
407
|
+
* `topicDiscovery.seedTier: 'on'` (§tag-provenance): a work that actually
|
|
408
|
+
* carries the topic tag outranks a related-only work, however popular the
|
|
409
|
+
* latter is, and popularity decides within each tier. With `seedTier: 'off'`
|
|
410
|
+
* and the default `relatedTags: 'always'` no `seedKey` is passed, so the
|
|
411
|
+
* documented popularity-only ranking is unchanged.
|
|
265
412
|
*/
|
|
266
413
|
popCompare(a, b) {
|
|
267
414
|
return b.popularity - a.popularity;
|
|
@@ -97,6 +97,9 @@ class TopicResolver {
|
|
|
97
97
|
name: seed,
|
|
98
98
|
translatedName: suggested.find((t) => this.scorer.key(t.name) === this.scorer.key(seed))?.translated_name,
|
|
99
99
|
score: 1,
|
|
100
|
+
weight: 1,
|
|
101
|
+
// The tag the operator asked for: the strongest provenance, never a hint.
|
|
102
|
+
source: 'seed',
|
|
100
103
|
occurrences: topicWorks.length,
|
|
101
104
|
coverage: 1,
|
|
102
105
|
specificity: 1,
|
|
@@ -168,7 +171,7 @@ class TopicResolver {
|
|
|
168
171
|
expiresAt: new Date(now + cacheDays * 24 * 60 * 60_000).toISOString(),
|
|
169
172
|
sampleSize: 0,
|
|
170
173
|
sampledWorks: 0,
|
|
171
|
-
tags: [{ name: seed, score: 1, occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
|
|
174
|
+
tags: [{ name: seed, score: 1, weight: 1, source: 'seed', occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
|
|
172
175
|
};
|
|
173
176
|
}
|
|
174
177
|
ageDays(space) {
|
|
@@ -99,10 +99,17 @@ class TopicTagScorer {
|
|
|
99
99
|
const suggestionWeight = stat.suggested ? 1.1 : 1.0;
|
|
100
100
|
const genericPenalty = GENERIC_TAG_PENALTY.has(k) ? 0.4 : 1.0;
|
|
101
101
|
const raw = recall * specificity * suggestionWeight * genericPenalty;
|
|
102
|
+
const score = Number(raw.toFixed(4));
|
|
102
103
|
resolved.push({
|
|
103
104
|
name: stat.name,
|
|
104
105
|
translatedName: stat.translatedName,
|
|
105
|
-
score
|
|
106
|
+
score,
|
|
107
|
+
// `weight` is the same semantic number; ranking and diagnostics read it
|
|
108
|
+
// while `score` stays for backward compatibility (§tag-provenance).
|
|
109
|
+
weight: score,
|
|
110
|
+
// Both channels may agree on a tag; the combined provenance records
|
|
111
|
+
// that, which is strictly more informative than either alone.
|
|
112
|
+
source: stat.suggested ? 'cooccurrence+autocomplete' : 'cooccurrence',
|
|
106
113
|
occurrences: stat.topicDocs,
|
|
107
114
|
coverage: Number(coverage.toFixed(4)),
|
|
108
115
|
specificity: Number(specificity.toFixed(4)),
|
|
@@ -127,6 +134,10 @@ class TopicTagScorer {
|
|
|
127
134
|
name: name,
|
|
128
135
|
translatedName: sug.translated_name?.trim() || undefined,
|
|
129
136
|
score: AUTOCOMPLETE_ONLY_SCORE,
|
|
137
|
+
weight: AUTOCOMPLETE_ONLY_SCORE,
|
|
138
|
+
// Pixiv autocomplete is the only evidence for this tag: it never
|
|
139
|
+
// co-occurred in the bounded sample.
|
|
140
|
+
source: 'autocomplete',
|
|
130
141
|
occurrences: 0,
|
|
131
142
|
coverage: 0,
|
|
132
143
|
specificity: 1.0, // Pixiv-endorsed related; treated as specific but unobserved
|
package/dist/topic/types.d.ts
CHANGED
|
@@ -4,12 +4,34 @@
|
|
|
4
4
|
* local models are used anywhere in this module.
|
|
5
5
|
*/
|
|
6
6
|
export type TopicContentType = 'illustration' | 'novel';
|
|
7
|
+
/**
|
|
8
|
+
* Where a resolved tag came from. Provenance is what lets a caller treat a weak
|
|
9
|
+
* expansion differently from the topic it was asked for (§tag-provenance):
|
|
10
|
+
*
|
|
11
|
+
* - `'seed'`: the tag the operator asked for. Always the strongest key.
|
|
12
|
+
* - `'cooccurrence'`: sampled together with the seed, but Pixiv autocomplete
|
|
13
|
+
* does not relate it to the seed. Co-occurrence evidence only.
|
|
14
|
+
* - `'autocomplete'`: Pixiv autocomplete relates it to the seed, but it never
|
|
15
|
+
* appeared in the sample. No co-occurrence evidence.
|
|
16
|
+
* - `'cooccurrence+autocomplete'`: both channels agree — the strongest related
|
|
17
|
+
* provenance available.
|
|
18
|
+
*/
|
|
19
|
+
export type TagSource = 'seed' | 'cooccurrence' | 'autocomplete' | 'cooccurrence+autocomplete';
|
|
7
20
|
/** A single related tag with a 0..1 relatedness score and provenance. */
|
|
8
21
|
export interface ResolvedTag {
|
|
9
22
|
name: string;
|
|
10
23
|
translatedName?: string;
|
|
11
24
|
/** Combined relatedness score (co-occurrence * specificity * suggestion). */
|
|
12
25
|
score: number;
|
|
26
|
+
/**
|
|
27
|
+
* Semantic weight used for ranking, filtering and diagnostics. Always the
|
|
28
|
+
* same number as `score`; kept as a separate, documented field so ranking can
|
|
29
|
+
* be explained (and, later, adjusted) without redefining `score`.
|
|
30
|
+
* Optional: spaces persisted before provenance existed have neither field.
|
|
31
|
+
*/
|
|
32
|
+
weight?: number;
|
|
33
|
+
/** Provenance of the tag. Optional for the same reason as `weight`. */
|
|
34
|
+
source?: TagSource;
|
|
13
35
|
/** How many sampled works (of the seed search) carried this tag. */
|
|
14
36
|
occurrences: number;
|
|
15
37
|
/** Coverage of the sampled seed works (occurrences / sample size). */
|
|
@@ -57,6 +79,26 @@ export interface TopicDiscoveryOptions {
|
|
|
57
79
|
* `'never'` searches the seed tag alone.
|
|
58
80
|
*/
|
|
59
81
|
relatedTags?: RelatedTagMode;
|
|
82
|
+
/**
|
|
83
|
+
* Which resolved tags may become recall channels (§tag-provenance). Deny wins
|
|
84
|
+
* over allow; the seed tag is never dropped by `allow`/`allowSources`.
|
|
85
|
+
*/
|
|
86
|
+
tagRelations?: TopicRelationsOptions;
|
|
87
|
+
/** Make the seed tag a hard ranking tier (default `'off'`). */
|
|
88
|
+
seedTier?: 'off' | 'on';
|
|
89
|
+
/** Count a work's translated tag names as tag hits (default false). */
|
|
90
|
+
matchTranslatedNames?: boolean;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Runtime shape of `TopicDiscoveryConfig.tagRelations`, declared here so the
|
|
94
|
+
* topic module does not depend on the config layer (the topic pipeline is also
|
|
95
|
+
* driven by hand-written targets in tests and by callers that never load a
|
|
96
|
+
* config file).
|
|
97
|
+
*/
|
|
98
|
+
export interface TopicRelationsOptions {
|
|
99
|
+
allowSources?: TagSource[];
|
|
100
|
+
allow?: string[];
|
|
101
|
+
deny?: string[];
|
|
60
102
|
}
|
|
61
103
|
export interface TopicCollectOptions {
|
|
62
104
|
maxPerTag?: number;
|
|
@@ -76,6 +118,12 @@ export interface TopicCandidate {
|
|
|
76
118
|
popularity: number;
|
|
77
119
|
/** Metadata topic-relevance score computed by the filter stage. */
|
|
78
120
|
metadataScore: number;
|
|
121
|
+
/**
|
|
122
|
+
* Translated tag names carried by the work, kept beside `tags` so
|
|
123
|
+
* `matchTranslatedNames` can compare them WITHOUT claiming the work itself
|
|
124
|
+
* carries a tag it does not (§tag-provenance). Default-off.
|
|
125
|
+
*/
|
|
126
|
+
translatedTags?: string[];
|
|
79
127
|
/** Pixiv AI classification copied from illustration search metadata. */
|
|
80
128
|
aiType?: number;
|
|
81
129
|
}
|
package/dist/version.js
CHANGED
|
@@ -2,5 +2,5 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.BUILD = void 0;
|
|
4
4
|
// GENERATED by scripts/write-version.js — do not edit manually.
|
|
5
|
-
exports.BUILD = { version: '3.0
|
|
5
|
+
exports.BUILD = { version: '3.2.0', commit: 'c195b909063c' };
|
|
6
6
|
//# sourceMappingURL=version.js.map
|
package/dist/webui/package.json
CHANGED
|
@@ -85,6 +85,10 @@ npx jest src/__tests__/delivery/gateway-reference-e2e.test.ts
|
|
|
85
85
|
重启后带同一个 `idempotencyKey` 收敛;一条坏路由的失败不会影响另一条。写自己的网关时,
|
|
86
86
|
让这个文件继续通过就是「你接对了」的最强证据。
|
|
87
87
|
|
|
88
|
+
接平台那一步请用 [`examples/onebot-adapter/`](../onebot-adapter/README.md):它是本节所述
|
|
89
|
+
「最小转换进程」的可运行实例(QQ / OneBot v11),同样零依赖、同样不实现 QQ 协议。本参考网关
|
|
90
|
+
与它是**互补**关系:一个证明「契约本身通不通」,另一个证明「平台映射写对了没有」。
|
|
91
|
+
|
|
88
92
|
## 它有意不做什么
|
|
89
93
|
|
|
90
94
|
- 不保存任何东西到磁盘(去重表在内存里,进程重启即丢)—— 生产网关必须持久化,
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
# Example OneBot v11 adapter(QQ)
|
|
2
|
+
|
|
3
|
+
一个**零依赖**的 OneBot v11 投递适配器,把 [Gateway Contract v1](../../docs/GATEWAY_CONTRACT.md)
|
|
4
|
+
的三个端点翻译成 OneBot v11 的 HTTP API 调用。它是 [docs/GATEWAY.md](../../docs/GATEWAY.md) §5.3
|
|
5
|
+
所描述的「最小转换进程」的**可运行实例**。
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
PixivFlow ──POST /deliver(契约 v1,带签名/幂等键)──▶ 本进程
|
|
9
|
+
│ POST /send_group_msg
|
|
10
|
+
▼
|
|
11
|
+
NapCat / Lagrange / LLOneBot ──▶ QQ
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
## 它不是什么
|
|
15
|
+
|
|
16
|
+
- **不实现 QQ 协议**,也不实现 OneBot 本身:QQ 会话属于你已经在跑的 OneBot 实现。
|
|
17
|
+
- **不做扫码登录**:二维码由 NapCat 自己的面板显示,凭据不会经过本进程,也不会经过 PixivFlow
|
|
18
|
+
(见 GATEWAY.md §5.1/§5.2)。
|
|
19
|
+
- **不做重试**:重试由 PixivFlow 的 outbox 负责;本进程只负责把一次请求翻译成一次 OneBot 调用,
|
|
20
|
+
并把「成功 / 待定 / 永久失败」如实翻译回契约词汇。
|
|
21
|
+
- **不按平台长分支**:这就是适配器独立成进程的原因 —— PixivFlow 侧的契约里没有 QQ。
|
|
22
|
+
|
|
23
|
+
只想先验证「契约本身通不通」、还不想碰 QQ,请先用
|
|
24
|
+
[`examples/gateway/`](../gateway/README.md)(它只打印消息,不接平台)。
|
|
25
|
+
|
|
26
|
+
## 跑起来
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
# 1. 先让 NapCat(或其它 OneBot v11 实现)的 HTTP API 在 3000 端口可用,并记下它的 token
|
|
30
|
+
# 2. 起适配器
|
|
31
|
+
ONEBOT_URL=http://127.0.0.1:3000 \
|
|
32
|
+
ONEBOT_TOKEN=<napcat-token> \
|
|
33
|
+
ONEBOT_TARGET=group:987654 \
|
|
34
|
+
ADAPTER_TOKEN=<给 PixivFlow 用的 token> \
|
|
35
|
+
node examples/onebot-adapter/server.mjs
|
|
36
|
+
|
|
37
|
+
# 自检:起一个假 OneBot,跑完三个端点与去重逻辑,打印结果并退出(0 = 通过)
|
|
38
|
+
node examples/onebot-adapter/server.mjs --selftest
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
启动时会拒绝「半配置」:`ONEBOT_URL` 缺失或 `ONEBOT_TARGET` 还是占位值 `group:0` 时,
|
|
42
|
+
进程会打印 `config.problem` 并以退出码 2 结束 —— 发错群比不启动更糟。
|
|
43
|
+
|
|
44
|
+
## 环境变量
|
|
45
|
+
|
|
46
|
+
| 变量 | 作用 |
|
|
47
|
+
| --- | --- |
|
|
48
|
+
| `PORT` | 监听端口(默认 `8791`) |
|
|
49
|
+
| `HOST` | 绑定地址(默认 `127.0.0.1`) |
|
|
50
|
+
| `ADAPTER_TOKEN` | **给 PixivFlow 用的** token,校验 `Authorization: Bearer <token>`;未设置只告警(端点无鉴权) |
|
|
51
|
+
| `ADAPTER_SECRET` | 设置后强制校验 `X-Webhook-Signature`(对**原始字节**做 HMAC,5 分钟时间窗) |
|
|
52
|
+
| `ONEBOT_URL` | OneBot HTTP API 基址,如 `http://127.0.0.1:3000` |
|
|
53
|
+
| `ONEBOT_TOKEN` | **给 OneBot 用的** token(与 `ADAPTER_TOKEN` 不要复用同一个值) |
|
|
54
|
+
| `ONEBOT_TARGET` | 投递目标:`group:987654`(默认)或 `private:987654` |
|
|
55
|
+
| `ONEBOT_TIMEOUT_MS` | 单次 OneBot 调用超时(默认 `15000`) |
|
|
56
|
+
| `ONEBOT_MIN_SEND_INTERVAL_MS` | 两次 OneBot 调用之间的最小间隔(默认 `500`),用于限速 |
|
|
57
|
+
| `ONEBOT_STATE_FILE` | 幂等账本(默认 `./.onebot-adapter-state.jsonl`,追加写 JSONL) |
|
|
58
|
+
|
|
59
|
+
## 端点
|
|
60
|
+
|
|
61
|
+
| 端点 | 行为 |
|
|
62
|
+
| --- | --- |
|
|
63
|
+
| `POST /deliver` | 验签 → 校验 Bearer → 解析 JSON → 校验 `schemaVersion` → 按 `idempotencyKey` 去重 → 发送消息段 → 上传附件 → 回契约 ACK 词 |
|
|
64
|
+
| `GET /pairing` | 调 `get_login_info`:成功 `{status:"connected", account, nickname}`;可达但未登录 `{status:"waiting", reason}`(HTTP 200);不可达 HTTP 503 `{status:"unreachable"}` |
|
|
65
|
+
| `GET /health` | 调 `get_status`:`data.online !== false && data.good === true` 时 `{status:"connected", contractVersion:1, gateway:"pixivflow-onebot-adapter", onebot:{…}}`,否则 HTTP 503。**供运维用**,PixivFlow 的投递链路从不调用它 |
|
|
66
|
+
|
|
67
|
+
未知路由回 `404 {status:"invalid"}`。
|
|
68
|
+
|
|
69
|
+
## 翻译规则(这是整个文件的重点)
|
|
70
|
+
|
|
71
|
+
**消息段**(契约 `message.parts` → OneBot `message` 数组,顺序保留):
|
|
72
|
+
|
|
73
|
+
| 契约 part | OneBot 段 |
|
|
74
|
+
| --- | --- |
|
|
75
|
+
| `{kind:"text"}` | `{type:"text", data:{text}}` —— 文案取自 `message.text`(契约里正文只出现在 `message.text`,`parts` 里只是一个位置标记),只消费一次 |
|
|
76
|
+
| `{kind:"image", media}` | `{type:"image", data:{file}}` |
|
|
77
|
+
| `{kind:"video", media}` | `{type:"video", data:{file}}` |
|
|
78
|
+
| `{kind:"album"}` | PixivFlow 会展开成 N 个各自带 `media` 的 part,因此这里就是 N 个 `image`/`video` 段 |
|
|
79
|
+
| `{kind:"file", media}` | **不是消息段**:走 `upload_group_file {group_id, file, name}`,再补发一条 `📎 附件:<name>` 提示消息 |
|
|
80
|
+
|
|
81
|
+
`media.file` 的取值:`base64://<...>`(`base64` 传输,永远可用)或 `file://<绝对路径>`
|
|
82
|
+
(`reference` 传输,要求 OneBot 能读到 PixivFlow 的磁盘 —— 这正是该传输的取舍,不做静默降级)。
|
|
83
|
+
|
|
84
|
+
**ACK 映射**(OneBot 的 HTTP 状态码几乎永远是 200,成败在 `retcode`;契约的规则是「先看词,再看码」):
|
|
85
|
+
|
|
86
|
+
| OneBot 回答 | 本适配器回给 PixivFlow | 结果 |
|
|
87
|
+
| --- | --- | --- |
|
|
88
|
+
| `retcode 0` | `200 {status:"accepted", id:<message_id>}` | `delivered`,`remote_id` 就是平台消息号 |
|
|
89
|
+
| `status:"async"` 或 `retcode 1` | `200 {status:"pending", reason}` | 保持待投递,**绝不报成功** |
|
|
90
|
+
| `retcode 100/102/103/104/105/1400/1404` | `200 {status:"failed", reason}` | 永久失败,进死信,不无限重试 |
|
|
91
|
+
| 未知 `retcode` | `502 {reason}`(**无状态词**) | 可重试,不猜 |
|
|
92
|
+
| 网络不可达 / 超时 / HTTP ≥ 500 | `502 {reason}`(无状态词) | 可重试 |
|
|
93
|
+
| HTTP `401/403/404` | 原样 4xx(无状态词) | 令牌/地址写错,修好即可重试 |
|
|
94
|
+
| HTTP `429` | `429`(无状态词) | 限速,可重试 |
|
|
95
|
+
| 非 JSON 响应体 | `502 {reason}`(无状态词) | 可重试 |
|
|
96
|
+
|
|
97
|
+
**「无状态词」是刻意的**:契约里状态词优先于 HTTP 码,一旦回了 `failed` 就会进死信;
|
|
98
|
+
所以凡是「请求本身有问题、但改配置后能成功」的情形,都只回一个裸 HTTP 码,让 PixivFlow 继续重试。
|
|
99
|
+
|
|
100
|
+
## PixivFlow 侧配置
|
|
101
|
+
|
|
102
|
+
```json
|
|
103
|
+
{
|
|
104
|
+
"delivery": {
|
|
105
|
+
"targets": {
|
|
106
|
+
"qq-main": {
|
|
107
|
+
"type": "webhook",
|
|
108
|
+
"url": "http://127.0.0.1:8791/deliver",
|
|
109
|
+
"token": "${QQ_ADAPTER_TOKEN}",
|
|
110
|
+
"pairingUrl": "http://127.0.0.1:8791/pairing",
|
|
111
|
+
"capabilities": {
|
|
112
|
+
"maxTextLength": 4000,
|
|
113
|
+
"maxAttachmentsPerMessage": 9,
|
|
114
|
+
"album": true,
|
|
115
|
+
"albumMin": 2,
|
|
116
|
+
"albumMax": 9,
|
|
117
|
+
"minSendIntervalMs": 500
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
跨机部署时把 `url`/`pairingUrl` 换成适配器所在主机的地址;若两者不在同一台机器上,
|
|
126
|
+
`reference` 传输的本地路径对端读不到,请改用 `base64`
|
|
127
|
+
(见 GATEWAY.md §6「`reference` 传输的文件可见性」)。
|
|
128
|
+
|
|
129
|
+
## 验证
|
|
130
|
+
|
|
131
|
+
| 层 | 命令/动作 | 证明的是什么 |
|
|
132
|
+
| --- | --- | --- |
|
|
133
|
+
| 适配器自身 | `node examples/onebot-adapter/server.mjs --selftest` | 段构造、ACK 映射、去重、`/pairing`、`/health`(假 OneBot,不接 QQ) |
|
|
134
|
+
| 契约到底 | `npx jest src/__tests__/delivery/onebot-adapter-e2e.test.ts` | **真**投递运行时 → outbox → 适配器进程 → OneBot HTTP:`remote_id`、`pending` 不落地、`failed` 进死信、重放 `duplicate_existing` |
|
|
135
|
+
| QQ 登录态 | NapCat 自己的面板 | 账号在线;**不证明** PixivFlow 能投递 |
|
|
136
|
+
| 真实投递 | 跑一次下载 + `pixivflow delivery status` | 账本上的 `delivered` 与群里的那条消息 |
|
|
137
|
+
|
|
138
|
+
`--selftest` 与上面的 jest 套件都不需要 QQ、不需要 NapCat:它们证明的是**翻译**正确,
|
|
139
|
+
不是「QQ 已经通了」。
|