pixivflow 3.0.2 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/TopicCommand.js +41 -9
- package/dist/config/types.d.ts +65 -1
- package/dist/config/types.js +3 -0
- package/dist/config/validation.js +47 -0
- package/dist/domain/media/NovelCoverPolicy.d.ts +28 -0
- package/dist/domain/media/NovelCoverPolicy.js +70 -0
- package/dist/download/DownloadManager.js +5 -1
- package/dist/download/NovelDownloader.d.ts +23 -1
- package/dist/download/NovelDownloader.js +69 -5
- package/dist/download/novelCover.d.ts +29 -5
- package/dist/download/novelCover.js +34 -5
- package/dist/package.json +1 -1
- package/dist/topic/TopicPipeline.d.ts +55 -1
- package/dist/topic/TopicPipeline.js +256 -41
- package/dist/topic/TopicResolver.js +4 -1
- package/dist/topic/TopicTagScorer.js +12 -1
- package/dist/topic/types.d.ts +62 -0
- package/dist/utils/imageDimensions.d.ts +19 -0
- package/dist/utils/imageDimensions.js +95 -0
- package/dist/version.js +1 -1
- package/dist/webui/package.json +1 -1
- package/package.json +1 -1
package/dist/package.json
CHANGED
|
@@ -1,10 +1,16 @@
|
|
|
1
1
|
import type { TargetConfig } from '../config';
|
|
2
2
|
import type { TopicResolver } from './TopicResolver';
|
|
3
|
-
import type { TopicCandidate, TopicClient, TopicCollectOptions, TopicContentType, TopicDiscoveryOptions, WorkLike } from './types';
|
|
3
|
+
import type { RelatedTagMode, ResolvedTag, TopicCandidate, TopicClient, TopicCollectOptions, TopicContentType, TopicDiscoveryOptions, TopicRelationsOptions, WorkLike } from './types';
|
|
4
4
|
export interface TopicSelection {
|
|
5
5
|
candidates: TopicCandidate[];
|
|
6
6
|
selected: TopicCandidate[];
|
|
7
7
|
resolvedTagCount: number;
|
|
8
|
+
/**
|
|
9
|
+
* Tags actually searched for the day. Under `relatedTags: 'always'` this is
|
|
10
|
+
* the whole resolved space; under the seed-first modes it is the seed tag and
|
|
11
|
+
* only the related tags that were really needed (§topic-recall).
|
|
12
|
+
*/
|
|
13
|
+
searchedTags?: string[];
|
|
8
14
|
rawCount: number;
|
|
9
15
|
dedupedCount: number;
|
|
10
16
|
acceptedCount: number;
|
|
@@ -12,6 +18,26 @@ export interface TopicSelection {
|
|
|
12
18
|
/** Works seen more than once across the topic tag space (recorded, dropped). */
|
|
13
19
|
duplicateRemovedCount: number;
|
|
14
20
|
}
|
|
21
|
+
/**
|
|
22
|
+
* The tag the operator asked for, even when `tagRelations` filters it out.
|
|
23
|
+
*
|
|
24
|
+
* `deny` may drop the seed, but a space filtered down to nothing must still be
|
|
25
|
+
* usable: the pipeline then falls back to the seed tag alone, exactly like the
|
|
26
|
+
* resolver's `seedOnly()` degradation. `weight` defaults to the tag's `score`
|
|
27
|
+
* for spaces persisted before provenance existed.
|
|
28
|
+
*/
|
|
29
|
+
export declare function seedFallbackTag(seedTag: ResolvedTag): ResolvedTag;
|
|
30
|
+
/**
|
|
31
|
+
* Applies `topicDiscovery.tagRelations` to a resolved space (§tag-provenance).
|
|
32
|
+
*
|
|
33
|
+
* Deny beats allow, and deny is the only thing that can drop the seed tag. A
|
|
34
|
+
* tag whose provenance is unknown (a space persisted before `source` existed, or
|
|
35
|
+
* a hand-written space) is treated as `cooccurrence` so that `allowSources` can
|
|
36
|
+
* still narrow it while the default (all sources) keeps walking everything.
|
|
37
|
+
*/
|
|
38
|
+
export declare function selectWalkedTags(tags: readonly ResolvedTag[], seedKey: string, relations?: TopicRelationsOptions): ResolvedTag[];
|
|
39
|
+
/** Which of the walked tags a `relatedTags` mode will actually search today. */
|
|
40
|
+
export declare function recallChannels(walked: readonly ResolvedTag[], seedKey: string, relatedMode: RelatedTagMode): ResolvedTag[];
|
|
15
41
|
/**
|
|
16
42
|
* Resolves a topic to a tag space, collects that day's works across the tags,
|
|
17
43
|
* filters by lightweight metadata relevance and ranks by local popularity, then
|
|
@@ -44,10 +70,29 @@ export declare class TopicPipeline {
|
|
|
44
70
|
}>;
|
|
45
71
|
private searchDay;
|
|
46
72
|
private toCandidate;
|
|
73
|
+
/**
|
|
74
|
+
* Applies the metadata gate to everything collected so far. When nothing at
|
|
75
|
+
* all clears the threshold but the seed tag is present, the seed-tag works are
|
|
76
|
+
* kept anyway: a sparse day must stay usable instead of reporting "no
|
|
77
|
+
* candidates" for a topic that visibly has works. Extracted from selection so
|
|
78
|
+
* the seed pass can be evaluated before deciding whether the related channel
|
|
79
|
+
* is needed at all (§topic-recall).
|
|
80
|
+
*/
|
|
81
|
+
private acceptedWorks;
|
|
47
82
|
/**
|
|
48
83
|
* Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
|
|
49
84
|
* title/caption add smaller boosts. The seed tag is strong evidence.
|
|
50
85
|
* No text model — case/symbol-insensitive substring matching only.
|
|
86
|
+
*
|
|
87
|
+
* `matchTranslatedNames` (default false) additionally counts a work's
|
|
88
|
+
* `translated_name` as a hit for the resolved tag it translates to, so a
|
|
89
|
+
* topic written in one language can still match works tagged in another. The
|
|
90
|
+
* translated name is matched against the resolved space — never added to
|
|
91
|
+
* `candidate.tags` — so the work is not reported as carrying a tag it lacks.
|
|
92
|
+
*
|
|
93
|
+
* A resolved tag counts at most ONCE per work, whether it matched through the
|
|
94
|
+
* tag name or through its translation (and a work that repeats a tag, as Pixiv
|
|
95
|
+
* payloads do for a tag plus its translation, does not double its weight).
|
|
51
96
|
*/
|
|
52
97
|
private metadataScore;
|
|
53
98
|
/**
|
|
@@ -56,8 +101,17 @@ export declare class TopicPipeline {
|
|
|
56
101
|
* as on-topic, and the choice between accepted works is decided by popularity
|
|
57
102
|
* alone. A work with a higher metadata score does NOT outrank a more popular
|
|
58
103
|
* accepted work.
|
|
104
|
+
*
|
|
105
|
+
* `seedKey` adds a single tier in front of that popularity order and is
|
|
106
|
+
* passed by the seed-first recall modes (§topic-recall) and by
|
|
107
|
+
* `topicDiscovery.seedTier: 'on'` (§tag-provenance): a work that actually
|
|
108
|
+
* carries the topic tag outranks a related-only work, however popular the
|
|
109
|
+
* latter is, and popularity decides within each tier. With `seedTier: 'off'`
|
|
110
|
+
* and the default `relatedTags: 'always'` no `seedKey` is passed, so the
|
|
111
|
+
* documented popularity-only ranking is unchanged.
|
|
59
112
|
*/
|
|
60
113
|
private popCompare;
|
|
114
|
+
private rankCompare;
|
|
61
115
|
private topByPopularity;
|
|
62
116
|
private onDay;
|
|
63
117
|
private normalize;
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.TopicPipeline = void 0;
|
|
4
|
+
exports.seedFallbackTag = seedFallbackTag;
|
|
5
|
+
exports.selectWalkedTags = selectWalkedTags;
|
|
6
|
+
exports.recallChannels = recallChannels;
|
|
4
7
|
const promises_1 = require("node:timers/promises");
|
|
5
8
|
const logger_1 = require("../logger");
|
|
6
9
|
const pixiv_utils_1 = require("../utils/pixiv-utils");
|
|
@@ -14,6 +17,105 @@ const STOP_TAGS = new Set([
|
|
|
14
17
|
'1000users入り', '5000users入り', '10000users入り', '500users入り', '100users入り',
|
|
15
18
|
'pixiv', 'commission', 'skeb', '依頼絵', '仕事絵',
|
|
16
19
|
]);
|
|
20
|
+
/** Normalization shared by the tag-relation filter and the pipeline itself. */
|
|
21
|
+
function normalizeKey(value) {
|
|
22
|
+
return value.trim().normalize('NFKC').toLocaleLowerCase();
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* The tag the operator asked for, even when `tagRelations` filters it out.
|
|
26
|
+
*
|
|
27
|
+
* `deny` may drop the seed, but a space filtered down to nothing must still be
|
|
28
|
+
* usable: the pipeline then falls back to the seed tag alone, exactly like the
|
|
29
|
+
* resolver's `seedOnly()` degradation. `weight` defaults to the tag's `score`
|
|
30
|
+
* for spaces persisted before provenance existed.
|
|
31
|
+
*/
|
|
32
|
+
function seedFallbackTag(seedTag) {
|
|
33
|
+
if (seedTag.source === 'seed' && seedTag.weight !== undefined)
|
|
34
|
+
return seedTag;
|
|
35
|
+
return { ...seedTag, source: 'seed', weight: seedTag.weight ?? seedTag.score, seed: true };
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Applies `topicDiscovery.tagRelations` to a resolved space (§tag-provenance).
|
|
39
|
+
*
|
|
40
|
+
* Deny beats allow, and deny is the only thing that can drop the seed tag. A
|
|
41
|
+
* tag whose provenance is unknown (a space persisted before `source` existed, or
|
|
42
|
+
* a hand-written space) is treated as `cooccurrence` so that `allowSources` can
|
|
43
|
+
* still narrow it while the default (all sources) keeps walking everything.
|
|
44
|
+
*/
|
|
45
|
+
function selectWalkedTags(tags, seedKey, relations = {}) {
|
|
46
|
+
const seedTag = tags.find((tag) => normalizeKey(tag.name) === seedKey);
|
|
47
|
+
const allowed = tags.filter((tag) => normalizeKey(tag.name) !== seedKey);
|
|
48
|
+
const deny = (relations.deny ?? []).map(normalizeKey);
|
|
49
|
+
const deniedKeys = new Set(deny);
|
|
50
|
+
const seedDenied = seedTag !== undefined && deniedKeys.has(seedKey);
|
|
51
|
+
if (seedDenied)
|
|
52
|
+
deniedKeys.clear(); // the seed fallback ignores deny
|
|
53
|
+
const allow = (relations.allow ?? []).map(normalizeKey);
|
|
54
|
+
const allowKeys = new Set(allow);
|
|
55
|
+
const open = allowKeys.size === 0;
|
|
56
|
+
const sources = relations.allowSources;
|
|
57
|
+
const sourceAllowed = (tag) => {
|
|
58
|
+
if (!sources || sources.length === 0)
|
|
59
|
+
return true;
|
|
60
|
+
const name = normalizeKey(tag.name);
|
|
61
|
+
if (allowKeys.has(name))
|
|
62
|
+
return true; // an explicit allow beats allowSources
|
|
63
|
+
const provenance = tag.source;
|
|
64
|
+
if (provenance === undefined)
|
|
65
|
+
return sources.includes('cooccurrence');
|
|
66
|
+
if (provenance === 'cooccurrence+autocomplete') {
|
|
67
|
+
return sources.includes('cooccurrence') || sources.includes('autocomplete');
|
|
68
|
+
}
|
|
69
|
+
return sources.includes(provenance);
|
|
70
|
+
};
|
|
71
|
+
const walked = allowed.filter((tag) => {
|
|
72
|
+
const name = normalizeKey(tag.name);
|
|
73
|
+
if (deniedKeys.has(name))
|
|
74
|
+
return false;
|
|
75
|
+
if (!open && !allowKeys.has(name))
|
|
76
|
+
return false;
|
|
77
|
+
return sourceAllowed(tag);
|
|
78
|
+
});
|
|
79
|
+
// No seed tag in the space at all: there is nothing to dominate, so fall back
|
|
80
|
+
// to walking the whole (filtered) space — the pre-provenance behaviour.
|
|
81
|
+
if (!seedTag)
|
|
82
|
+
return walked;
|
|
83
|
+
// A denied seed empties the space, exactly like the resolver's seedOnly():
|
|
84
|
+
// the operator still gets their own topic, and no related tag is walked.
|
|
85
|
+
if (seedDenied)
|
|
86
|
+
return [seedFallbackTag(seedTag)];
|
|
87
|
+
return [seedTag, ...walked];
|
|
88
|
+
}
|
|
89
|
+
/** Every spelling under which a set of resolved tags can be recognized. */
|
|
90
|
+
function buildTagAliases(tags, matchTranslatedNames) {
|
|
91
|
+
const aliases = new Map();
|
|
92
|
+
for (const tag of tags) {
|
|
93
|
+
const primary = normalizeKey(tag.name);
|
|
94
|
+
const entry = { primary, weight: tag.weight ?? tag.score };
|
|
95
|
+
if (!aliases.has(primary))
|
|
96
|
+
aliases.set(primary, entry);
|
|
97
|
+
if (!matchTranslatedNames)
|
|
98
|
+
continue;
|
|
99
|
+
// The work-side `translated_name` for this tag is the same resolved tag: the
|
|
100
|
+
// space still exposes one tag, it is simply recognized under two keys.
|
|
101
|
+
const translated = tag.translatedName ? normalizeKey(tag.translatedName) : '';
|
|
102
|
+
if (translated && !aliases.has(translated))
|
|
103
|
+
aliases.set(translated, entry);
|
|
104
|
+
}
|
|
105
|
+
return aliases;
|
|
106
|
+
}
|
|
107
|
+
/** Which of the walked tags a `relatedTags` mode will actually search today. */
|
|
108
|
+
function recallChannels(walked, seedKey, relatedMode) {
|
|
109
|
+
const seedTags = walked.filter((tag) => normalizeKey(tag.name) === seedKey);
|
|
110
|
+
// A space with no seed tag has nothing to protect: every remaining tag is the
|
|
111
|
+
// only channel there is, so the whole space is walked whatever the mode
|
|
112
|
+
// (pre-provenance fallback for a hand-written space).
|
|
113
|
+
if (seedTags.length === 0)
|
|
114
|
+
return [...walked];
|
|
115
|
+
if (relatedMode === 'never' || relatedMode === 'when_seed_insufficient')
|
|
116
|
+
return seedTags;
|
|
117
|
+
return [...seedTags, ...walked.filter((tag) => normalizeKey(tag.name) !== seedKey)];
|
|
118
|
+
}
|
|
17
119
|
/**
|
|
18
120
|
* Resolves a topic to a tag space, collects that day's works across the tags,
|
|
19
121
|
* filters by lightweight metadata relevance and ranks by local popularity, then
|
|
@@ -42,23 +144,39 @@ class TopicPipeline {
|
|
|
42
144
|
async selectWorks(target, contentType, day, limit, discovery, collect) {
|
|
43
145
|
const topic = (target.topic ?? '').trim();
|
|
44
146
|
const { space } = await this.resolver.resolve(topic, contentType, discovery);
|
|
45
|
-
const tagScores = new Map(space.tags.map((t) => [this.key(t.name), t.score]));
|
|
46
147
|
const maxPerTag = this.bound(collect.maxPerTag, COLLECT_DEFAULTS.maxPerTag, 5, 100);
|
|
47
148
|
const maxCandidates = this.bound(collect.maxCandidates, COLLECT_DEFAULTS.maxCandidates, 20, 500);
|
|
48
149
|
const minMetadataScore = collect.minMetadataScore ?? COLLECT_DEFAULTS.minMetadataScore;
|
|
49
150
|
const includeR18 = discovery.includeR18 === true;
|
|
151
|
+
// An unknown mode (hand-written config bypassing validation) falls back to
|
|
152
|
+
// the historical behaviour rather than silently narrowing recall.
|
|
153
|
+
const requestedMode = discovery.relatedTags;
|
|
154
|
+
const relatedMode = requestedMode === 'when_seed_insufficient' || requestedMode === 'never' ? requestedMode : 'always';
|
|
50
155
|
const byId = new Map();
|
|
51
156
|
let rawCount = 0;
|
|
52
157
|
let aiExcludedCount = 0;
|
|
53
158
|
let duplicateRemovedCount = 0;
|
|
54
|
-
const
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
159
|
+
const seedKey = this.key(topic);
|
|
160
|
+
// §tag-provenance: apply tagRelations where the walked list is built, so the
|
|
161
|
+
// filter governs seedTags/relatedTags BEFORE any search is issued. The seed
|
|
162
|
+
// tag survives allow/allowSources; a deny entry that empties the space falls
|
|
163
|
+
// back to the seed tag alone, exactly like the resolver's seedOnly().
|
|
164
|
+
const matchTranslatedNames = discovery.matchTranslatedNames === true;
|
|
165
|
+
const walked = selectWalkedTags(space.tags, seedKey, discovery.tagRelations);
|
|
166
|
+
const tagNames = walked.map((t) => t.name);
|
|
167
|
+
const seedTags = tagNames.filter((name) => this.key(name) === seedKey);
|
|
168
|
+
const relatedTags = tagNames.filter((name) => this.key(name) !== seedKey);
|
|
169
|
+
// Ranking reads the documented semantic weight, falling back to `score` for
|
|
170
|
+
// spaces persisted before provenance existed. The two are always equal.
|
|
171
|
+
const tagScores = buildTagAliases(walked, matchTranslatedNames);
|
|
172
|
+
const searchedTags = [];
|
|
173
|
+
// A hard seed tier: 'on' always, and the seed-first modes keep their layer.
|
|
174
|
+
const hardSeedTier = discovery.seedTier === 'on' || relatedMode !== 'always';
|
|
175
|
+
const collectTag = async (tag) => {
|
|
58
176
|
// Cancellation is checked between tags, so a cancelled run stops issuing
|
|
59
177
|
// new searches even when the aborted request itself had already returned.
|
|
60
178
|
(0, errors_1.throwIfAborted)(this.signal, 'topic collection cancelled');
|
|
61
|
-
|
|
179
|
+
searchedTags.push(tag);
|
|
62
180
|
const works = await this.searchDay(contentType, tag, day, maxPerTag, includeR18);
|
|
63
181
|
rawCount += works.length;
|
|
64
182
|
for (const work of works) {
|
|
@@ -75,37 +193,57 @@ class TopicPipeline {
|
|
|
75
193
|
break;
|
|
76
194
|
}
|
|
77
195
|
logger_1.logger.debug('[TopicCollector] type=' + contentType + ' tag=' + tag + ' day=' + day + ' fetched=' + works.length + ' pool=' + byId.size);
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
196
|
+
};
|
|
197
|
+
const runTags = async (names) => {
|
|
198
|
+
for (let i = 0; i < names.length; i++) {
|
|
199
|
+
if (byId.size >= maxCandidates)
|
|
200
|
+
break;
|
|
201
|
+
await collectTag(names[i]);
|
|
202
|
+
if (i < names.length - 1 && this.requestDelayMs > 0)
|
|
203
|
+
await (0, promises_1.setTimeout)(this.requestDelayMs);
|
|
204
|
+
}
|
|
205
|
+
};
|
|
206
|
+
// §topic-recall: a resolved tag space is a hierarchy, not a bag of
|
|
207
|
+
// interchangeable tags. 'always' keeps the historical behaviour — every
|
|
208
|
+
// resolved tag is a recall channel for the day. The seed-first modes search
|
|
209
|
+
// the topic tag the operator actually asked for and only walk the related
|
|
210
|
+
// channel when that cannot fill the target, so a second high-weight tag
|
|
211
|
+
// (丸吞) cannot take the only slot of a 西瓜肚 target.
|
|
212
|
+
if (seedTags.length === 0 || relatedMode === 'always') {
|
|
213
|
+
// No seed tag in the space (hand-written space): keep walking everything
|
|
214
|
+
// rather than returning nothing.
|
|
215
|
+
await runTags(recallChannels(walked, seedKey, relatedMode).map((t) => t.name));
|
|
89
216
|
}
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
217
|
+
else {
|
|
218
|
+
await runTags(seedTags);
|
|
219
|
+
const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames).length;
|
|
220
|
+
if (relatedMode === 'never') {
|
|
221
|
+
logger_1.logger.info('[TopicRecall] mode=never tag=' + topic + ' day=' + day + ' accepted=' + seedAccepted);
|
|
222
|
+
}
|
|
223
|
+
else if (seedAccepted < limit) {
|
|
224
|
+
logger_1.logger.info('[TopicRecall] mode=when_seed_insufficient tag=' + topic + ' seedAccepted=' + seedAccepted + '/' + limit + ' relatedTags=' + relatedTags.length + '; expanding');
|
|
225
|
+
await runTags(relatedTags);
|
|
226
|
+
}
|
|
227
|
+
else {
|
|
228
|
+
logger_1.logger.info('[TopicRecall] mode=when_seed_insufficient tag=' + topic + ' seedAccepted=' + seedAccepted + '/' + limit + '; related tags not searched');
|
|
229
|
+
}
|
|
96
230
|
}
|
|
97
|
-
const
|
|
231
|
+
const dedupedCount = byId.size;
|
|
232
|
+
logger_1.logger.info('[TopicCollector] type=' + contentType + ' raw=' + rawCount + ' deduplicated=' + dedupedCount + ' aiExcluded=' + aiExcludedCount + ' searchedTags=' + searchedTags.length);
|
|
233
|
+
const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames);
|
|
234
|
+
const chosen = this.topByPopularity(accepted, limit, hardSeedTier ? seedKey : undefined);
|
|
98
235
|
const selected = chosen.map((e) => e.candidate);
|
|
99
236
|
logger_1.logger.info('[MetadataTopicFilter] accepted=' + accepted.length);
|
|
100
237
|
if (selected[0]) {
|
|
101
|
-
logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + ' title=' + selected[0].title);
|
|
238
|
+
logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + (hardSeedTier ? ' seedTier=on' : '') + ' title=' + selected[0].title);
|
|
102
239
|
}
|
|
103
240
|
return {
|
|
104
241
|
works: chosen.map((e) => e.work),
|
|
105
242
|
selection: {
|
|
106
243
|
candidates: [...byId.values()].map((e) => e.candidate),
|
|
107
244
|
selected,
|
|
108
|
-
resolvedTagCount:
|
|
245
|
+
resolvedTagCount: walked.length,
|
|
246
|
+
searchedTags,
|
|
109
247
|
rawCount,
|
|
110
248
|
dedupedCount,
|
|
111
249
|
acceptedCount: accepted.length,
|
|
@@ -137,12 +275,16 @@ class TopicPipeline {
|
|
|
137
275
|
}
|
|
138
276
|
toCandidate(work, type) {
|
|
139
277
|
const popularity = (0, pixiv_utils_1.calculatePopularityScore)(work);
|
|
278
|
+
const translatedTags = (work.tags ?? [])
|
|
279
|
+
.map((t) => t.translated_name)
|
|
280
|
+
.filter((name) => Boolean(name));
|
|
140
281
|
return {
|
|
141
282
|
id: work.id,
|
|
142
283
|
type,
|
|
143
284
|
title: work.title ?? '',
|
|
144
285
|
caption: work.caption ?? '',
|
|
145
286
|
tags: (work.tags ?? []).map((t) => t.name).filter(Boolean),
|
|
287
|
+
...(translatedTags.length > 0 ? { translatedTags } : {}),
|
|
146
288
|
bookmarks: Number(work.total_bookmarks ?? work.bookmark_count ?? 0) || 0,
|
|
147
289
|
views: Number(work.total_view ?? work.view_count ?? 0) || 0,
|
|
148
290
|
popularity,
|
|
@@ -150,29 +292,84 @@ class TopicPipeline {
|
|
|
150
292
|
...(work.illust_ai_type !== undefined ? { aiType: work.illust_ai_type } : {}),
|
|
151
293
|
};
|
|
152
294
|
}
|
|
295
|
+
/**
|
|
296
|
+
* Applies the metadata gate to everything collected so far. When nothing at
|
|
297
|
+
* all clears the threshold but the seed tag is present, the seed-tag works are
|
|
298
|
+
* kept anyway: a sparse day must stay usable instead of reporting "no
|
|
299
|
+
* candidates" for a topic that visibly has works. Extracted from selection so
|
|
300
|
+
* the seed pass can be evaluated before deciding whether the related channel
|
|
301
|
+
* is needed at all (§topic-recall).
|
|
302
|
+
*/
|
|
303
|
+
acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames) {
|
|
304
|
+
const accepted = [];
|
|
305
|
+
const hitsSeed = (candidate) => {
|
|
306
|
+
if (candidate.tags.some((t) => this.key(t) === seedKey))
|
|
307
|
+
return true;
|
|
308
|
+
if (!matchTranslatedNames)
|
|
309
|
+
return false;
|
|
310
|
+
return (candidate.translatedTags ?? []).some((t) => this.key(t) === seedKey);
|
|
311
|
+
};
|
|
312
|
+
for (const entry of byId.values()) {
|
|
313
|
+
entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores, matchTranslatedNames);
|
|
314
|
+
if (entry.candidate.metadataScore >= minMetadataScore)
|
|
315
|
+
accepted.push(entry);
|
|
316
|
+
}
|
|
317
|
+
if (accepted.length === 0 && byId.size > 0) {
|
|
318
|
+
const fallback = [...byId.values()]
|
|
319
|
+
.filter((e) => hitsSeed(e.candidate))
|
|
320
|
+
.sort((a, b) => b.candidate.popularity - a.candidate.popularity);
|
|
321
|
+
accepted.push(...fallback.slice(0, Math.max(limit, 1)));
|
|
322
|
+
logger_1.logger.warn('[MetadataTopicFilter] type=' + contentType + ' none above threshold ' + minMetadataScore + '; kept ' + accepted.length + ' seed-tag fallback');
|
|
323
|
+
}
|
|
324
|
+
return accepted;
|
|
325
|
+
}
|
|
153
326
|
/**
|
|
154
327
|
* Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
|
|
155
328
|
* title/caption add smaller boosts. The seed tag is strong evidence.
|
|
156
329
|
* No text model — case/symbol-insensitive substring matching only.
|
|
330
|
+
*
|
|
331
|
+
* `matchTranslatedNames` (default false) additionally counts a work's
|
|
332
|
+
* `translated_name` as a hit for the resolved tag it translates to, so a
|
|
333
|
+
* topic written in one language can still match works tagged in another. The
|
|
334
|
+
* translated name is matched against the resolved space — never added to
|
|
335
|
+
* `candidate.tags` — so the work is not reported as carrying a tag it lacks.
|
|
336
|
+
*
|
|
337
|
+
* A resolved tag counts at most ONCE per work, whether it matched through the
|
|
338
|
+
* tag name or through its translation (and a work that repeats a tag, as Pixiv
|
|
339
|
+
* payloads do for a tag plus its translation, does not double its weight).
|
|
157
340
|
*/
|
|
158
|
-
metadataScore(candidate, seedKey, tagScores) {
|
|
341
|
+
metadataScore(candidate, seedKey, tagScores, matchTranslatedNames) {
|
|
159
342
|
let seedHit = false;
|
|
160
343
|
let relatedSum = 0;
|
|
161
344
|
let relatedHits = 0;
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
345
|
+
let strongRelated = false;
|
|
346
|
+
// A resolved tag is matched through its own name, and — when the caller
|
|
347
|
+
// opted in — through the work-side translation of that tag. Either spelling
|
|
348
|
+
// maps back to the SAME resolved tag, so a work carrying both spellings
|
|
349
|
+
// counts once and never gets a double weight.
|
|
350
|
+
const countedTagKeys = new Set();
|
|
351
|
+
const consider = (candidateKey) => {
|
|
352
|
+
if (candidateKey === '' || STOP_TAGS.has(candidateKey))
|
|
353
|
+
return;
|
|
354
|
+
if (candidateKey === seedKey) {
|
|
167
355
|
seedHit = true;
|
|
168
|
-
|
|
169
|
-
}
|
|
170
|
-
const related = tagScores.get(k);
|
|
171
|
-
if (related !== undefined) {
|
|
172
|
-
relatedSum += Math.min(related, 0.6);
|
|
173
|
-
relatedHits += 1;
|
|
356
|
+
return;
|
|
174
357
|
}
|
|
358
|
+
const aliases = tagScores.get(candidateKey);
|
|
359
|
+
if (!aliases || countedTagKeys.has(aliases.primary))
|
|
360
|
+
return;
|
|
361
|
+
countedTagKeys.add(aliases.primary);
|
|
362
|
+
relatedSum += Math.min(aliases.weight, 0.6);
|
|
363
|
+
relatedHits += 1;
|
|
364
|
+
if (aliases.weight >= 0.6)
|
|
365
|
+
strongRelated = true;
|
|
366
|
+
};
|
|
367
|
+
if (matchTranslatedNames) {
|
|
368
|
+
for (const tag of candidate.translatedTags ?? [])
|
|
369
|
+
consider(this.key(tag));
|
|
175
370
|
}
|
|
371
|
+
for (const tag of candidate.tags)
|
|
372
|
+
consider(this.key(tag));
|
|
176
373
|
const hayTitle = this.normalize(candidate.title);
|
|
177
374
|
const hayCaption = this.normalize(candidate.caption);
|
|
178
375
|
const titleSeed = !!seedKey && hayTitle.includes(seedKey);
|
|
@@ -189,7 +386,6 @@ class TopicPipeline {
|
|
|
189
386
|
// deliberately does NOT clear the bar, so a hugely popular tangential work
|
|
190
387
|
// cannot crowd out core topic works.
|
|
191
388
|
let score = 0;
|
|
192
|
-
const strongRelated = [...tagScores.entries()].some(([k, w]) => w >= 0.6 && candidate.tags.some((t) => this.key(t) === k));
|
|
193
389
|
if (titleSeed)
|
|
194
390
|
score += 0.8;
|
|
195
391
|
if (captionSeed)
|
|
@@ -205,13 +401,32 @@ class TopicPipeline {
|
|
|
205
401
|
* as on-topic, and the choice between accepted works is decided by popularity
|
|
206
402
|
* alone. A work with a higher metadata score does NOT outrank a more popular
|
|
207
403
|
* accepted work.
|
|
404
|
+
*
|
|
405
|
+
* `seedKey` adds a single tier in front of that popularity order and is
|
|
406
|
+
* passed by the seed-first recall modes (§topic-recall) and by
|
|
407
|
+
* `topicDiscovery.seedTier: 'on'` (§tag-provenance): a work that actually
|
|
408
|
+
* carries the topic tag outranks a related-only work, however popular the
|
|
409
|
+
* latter is, and popularity decides within each tier. With `seedTier: 'off'`
|
|
410
|
+
* and the default `relatedTags: 'always'` no `seedKey` is passed, so the
|
|
411
|
+
* documented popularity-only ranking is unchanged.
|
|
208
412
|
*/
|
|
209
413
|
popCompare(a, b) {
|
|
210
414
|
return b.popularity - a.popularity;
|
|
211
415
|
}
|
|
212
|
-
|
|
416
|
+
rankCompare(seedKey) {
|
|
417
|
+
if (!seedKey)
|
|
418
|
+
return (a, b) => this.popCompare(a, b);
|
|
419
|
+
const tier = (c) => (c.tags.some((t) => this.key(t) === seedKey) ? 0 : 1);
|
|
420
|
+
return (a, b) => {
|
|
421
|
+
const diff = tier(a) - tier(b);
|
|
422
|
+
return diff !== 0 ? diff : this.popCompare(a, b);
|
|
423
|
+
};
|
|
424
|
+
}
|
|
425
|
+
topByPopularity(items, limit, seedKey) {
|
|
213
426
|
if (items.length <= limit)
|
|
214
|
-
return items.sort((a, b) => this.
|
|
427
|
+
return items.sort((a, b) => this.rankCompare(seedKey)(a.candidate, b.candidate));
|
|
428
|
+
if (seedKey)
|
|
429
|
+
return items.sort((a, b) => this.rankCompare(seedKey)(a.candidate, b.candidate)).slice(0, limit);
|
|
215
430
|
// O(n) top-`limit` selection (limit is tiny, e.g. 1); avoids a full sort.
|
|
216
431
|
const top = [];
|
|
217
432
|
for (const item of items) {
|
|
@@ -97,6 +97,9 @@ class TopicResolver {
|
|
|
97
97
|
name: seed,
|
|
98
98
|
translatedName: suggested.find((t) => this.scorer.key(t.name) === this.scorer.key(seed))?.translated_name,
|
|
99
99
|
score: 1,
|
|
100
|
+
weight: 1,
|
|
101
|
+
// The tag the operator asked for: the strongest provenance, never a hint.
|
|
102
|
+
source: 'seed',
|
|
100
103
|
occurrences: topicWorks.length,
|
|
101
104
|
coverage: 1,
|
|
102
105
|
specificity: 1,
|
|
@@ -168,7 +171,7 @@ class TopicResolver {
|
|
|
168
171
|
expiresAt: new Date(now + cacheDays * 24 * 60 * 60_000).toISOString(),
|
|
169
172
|
sampleSize: 0,
|
|
170
173
|
sampledWorks: 0,
|
|
171
|
-
tags: [{ name: seed, score: 1, occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
|
|
174
|
+
tags: [{ name: seed, score: 1, weight: 1, source: 'seed', occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
|
|
172
175
|
};
|
|
173
176
|
}
|
|
174
177
|
ageDays(space) {
|
|
@@ -99,10 +99,17 @@ class TopicTagScorer {
|
|
|
99
99
|
const suggestionWeight = stat.suggested ? 1.1 : 1.0;
|
|
100
100
|
const genericPenalty = GENERIC_TAG_PENALTY.has(k) ? 0.4 : 1.0;
|
|
101
101
|
const raw = recall * specificity * suggestionWeight * genericPenalty;
|
|
102
|
+
const score = Number(raw.toFixed(4));
|
|
102
103
|
resolved.push({
|
|
103
104
|
name: stat.name,
|
|
104
105
|
translatedName: stat.translatedName,
|
|
105
|
-
score
|
|
106
|
+
score,
|
|
107
|
+
// `weight` is the same semantic number; ranking and diagnostics read it
|
|
108
|
+
// while `score` stays for backward compatibility (§tag-provenance).
|
|
109
|
+
weight: score,
|
|
110
|
+
// Both channels may agree on a tag; the combined provenance records
|
|
111
|
+
// that, which is strictly more informative than either alone.
|
|
112
|
+
source: stat.suggested ? 'cooccurrence+autocomplete' : 'cooccurrence',
|
|
106
113
|
occurrences: stat.topicDocs,
|
|
107
114
|
coverage: Number(coverage.toFixed(4)),
|
|
108
115
|
specificity: Number(specificity.toFixed(4)),
|
|
@@ -127,6 +134,10 @@ class TopicTagScorer {
|
|
|
127
134
|
name: name,
|
|
128
135
|
translatedName: sug.translated_name?.trim() || undefined,
|
|
129
136
|
score: AUTOCOMPLETE_ONLY_SCORE,
|
|
137
|
+
weight: AUTOCOMPLETE_ONLY_SCORE,
|
|
138
|
+
// Pixiv autocomplete is the only evidence for this tag: it never
|
|
139
|
+
// co-occurred in the bounded sample.
|
|
140
|
+
source: 'autocomplete',
|
|
130
141
|
occurrences: 0,
|
|
131
142
|
coverage: 0,
|
|
132
143
|
specificity: 1.0, // Pixiv-endorsed related; treated as specific but unobserved
|
package/dist/topic/types.d.ts
CHANGED
|
@@ -4,12 +4,34 @@
|
|
|
4
4
|
* local models are used anywhere in this module.
|
|
5
5
|
*/
|
|
6
6
|
export type TopicContentType = 'illustration' | 'novel';
|
|
7
|
+
/**
|
|
8
|
+
* Where a resolved tag came from. Provenance is what lets a caller treat a weak
|
|
9
|
+
* expansion differently from the topic it was asked for (§tag-provenance):
|
|
10
|
+
*
|
|
11
|
+
* - `'seed'`: the tag the operator asked for. Always the strongest key.
|
|
12
|
+
* - `'cooccurrence'`: sampled together with the seed, but Pixiv autocomplete
|
|
13
|
+
* does not relate it to the seed. Co-occurrence evidence only.
|
|
14
|
+
* - `'autocomplete'`: Pixiv autocomplete relates it to the seed, but it never
|
|
15
|
+
* appeared in the sample. No co-occurrence evidence.
|
|
16
|
+
* - `'cooccurrence+autocomplete'`: both channels agree — the strongest related
|
|
17
|
+
* provenance available.
|
|
18
|
+
*/
|
|
19
|
+
export type TagSource = 'seed' | 'cooccurrence' | 'autocomplete' | 'cooccurrence+autocomplete';
|
|
7
20
|
/** A single related tag with a 0..1 relatedness score and provenance. */
|
|
8
21
|
export interface ResolvedTag {
|
|
9
22
|
name: string;
|
|
10
23
|
translatedName?: string;
|
|
11
24
|
/** Combined relatedness score (co-occurrence * specificity * suggestion). */
|
|
12
25
|
score: number;
|
|
26
|
+
/**
|
|
27
|
+
* Semantic weight used for ranking, filtering and diagnostics. Always the
|
|
28
|
+
* same number as `score`; kept as a separate, documented field so ranking can
|
|
29
|
+
* be explained (and, later, adjusted) without redefining `score`.
|
|
30
|
+
* Optional: spaces persisted before provenance existed have neither field.
|
|
31
|
+
*/
|
|
32
|
+
weight?: number;
|
|
33
|
+
/** Provenance of the tag. Optional for the same reason as `weight`. */
|
|
34
|
+
source?: TagSource;
|
|
13
35
|
/** How many sampled works (of the seed search) carried this tag. */
|
|
14
36
|
occurrences: number;
|
|
15
37
|
/** Coverage of the sampled seed works (occurrences / sample size). */
|
|
@@ -35,6 +57,8 @@ export interface TopicSpace {
|
|
|
35
57
|
sampledWorks: number;
|
|
36
58
|
tags: ResolvedTag[];
|
|
37
59
|
}
|
|
60
|
+
/** When related tags may be used as their own recall channel (§topic-recall). */
|
|
61
|
+
export type RelatedTagMode = 'always' | 'when_seed_insufficient' | 'never';
|
|
38
62
|
export interface TopicDiscoveryOptions {
|
|
39
63
|
/** Include R-18 works in topic sampling and collection (default false). */
|
|
40
64
|
includeR18?: boolean;
|
|
@@ -43,6 +67,38 @@ export interface TopicDiscoveryOptions {
|
|
|
43
67
|
cacheDays?: number;
|
|
44
68
|
minScore?: number;
|
|
45
69
|
refresh?: boolean;
|
|
70
|
+
/**
|
|
71
|
+
* Related-tag recall mode (default `'always'`).
|
|
72
|
+
*
|
|
73
|
+
* A resolved tag space is a hierarchy, not a bag of interchangeable tags: the
|
|
74
|
+
* seed tag is the topic the operator asked for and every other tag is a hint.
|
|
75
|
+
* Under `'always'` each resolved tag is searched for the day's works, so a
|
|
76
|
+
* second high-weight tag (丸吞) can occupy the only slot of a 西瓜肚 target.
|
|
77
|
+
* `'when_seed_insufficient'` searches the seed tag first and only walks the
|
|
78
|
+
* related channel when the seed cannot fill the limit for that day;
|
|
79
|
+
* `'never'` searches the seed tag alone.
|
|
80
|
+
*/
|
|
81
|
+
relatedTags?: RelatedTagMode;
|
|
82
|
+
/**
|
|
83
|
+
* Which resolved tags may become recall channels (§tag-provenance). Deny wins
|
|
84
|
+
* over allow; the seed tag is never dropped by `allow`/`allowSources`.
|
|
85
|
+
*/
|
|
86
|
+
tagRelations?: TopicRelationsOptions;
|
|
87
|
+
/** Make the seed tag a hard ranking tier (default `'off'`). */
|
|
88
|
+
seedTier?: 'off' | 'on';
|
|
89
|
+
/** Count a work's translated tag names as tag hits (default false). */
|
|
90
|
+
matchTranslatedNames?: boolean;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Runtime shape of `TopicDiscoveryConfig.tagRelations`, declared here so the
|
|
94
|
+
* topic module does not depend on the config layer (the topic pipeline is also
|
|
95
|
+
* driven by hand-written targets in tests and by callers that never load a
|
|
96
|
+
* config file).
|
|
97
|
+
*/
|
|
98
|
+
export interface TopicRelationsOptions {
|
|
99
|
+
allowSources?: TagSource[];
|
|
100
|
+
allow?: string[];
|
|
101
|
+
deny?: string[];
|
|
46
102
|
}
|
|
47
103
|
export interface TopicCollectOptions {
|
|
48
104
|
maxPerTag?: number;
|
|
@@ -62,6 +118,12 @@ export interface TopicCandidate {
|
|
|
62
118
|
popularity: number;
|
|
63
119
|
/** Metadata topic-relevance score computed by the filter stage. */
|
|
64
120
|
metadataScore: number;
|
|
121
|
+
/**
|
|
122
|
+
* Translated tag names carried by the work, kept beside `tags` so
|
|
123
|
+
* `matchTranslatedNames` can compare them WITHOUT claiming the work itself
|
|
124
|
+
* carries a tag it does not (§tag-provenance). Default-off.
|
|
125
|
+
*/
|
|
126
|
+
translatedTags?: string[];
|
|
65
127
|
/** Pixiv AI classification copied from illustration search metadata. */
|
|
66
128
|
aiType?: number;
|
|
67
129
|
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Intrinsic image dimensions read straight from the container header.
|
|
3
|
+
*
|
|
4
|
+
* PixivFlow ships no image decoder on purpose (§novel-cover): the only question
|
|
5
|
+
* it ever asks about a remote image is "what canvas is this?", and a handful of
|
|
6
|
+
* header bytes answer it for JPEG/PNG/GIF without decoding any pixels. Unknown
|
|
7
|
+
* containers, truncated input and non-image payloads return `undefined` so
|
|
8
|
+
* callers can fail open instead of guessing.
|
|
9
|
+
*/
|
|
10
|
+
export interface ImageDimensions {
|
|
11
|
+
width: number;
|
|
12
|
+
height: number;
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Reads the intrinsic canvas of a JPEG, PNG or GIF payload.
|
|
16
|
+
* Returns `undefined` when the format is unknown or the header is incomplete.
|
|
17
|
+
*/
|
|
18
|
+
export declare function readImageDimensions(input: ArrayBuffer | Uint8Array | null | undefined): ImageDimensions | undefined;
|
|
19
|
+
//# sourceMappingURL=imageDimensions.d.ts.map
|