pixivflow 3.0.3 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/TopicCommand.js +41 -9
- package/dist/config/types.d.ts +54 -0
- package/dist/config/types.js +3 -0
- package/dist/config/validation.js +44 -0
- package/dist/domain/media/NovelCoverPolicy.d.ts +28 -0
- package/dist/domain/media/NovelCoverPolicy.js +70 -0
- package/dist/download/DownloadManager.js +5 -1
- package/dist/download/NovelDownloader.d.ts +15 -4
- package/dist/download/NovelDownloader.js +43 -11
- package/dist/download/novelCover.d.ts +5 -6
- package/dist/download/novelCover.js +9 -12
- package/dist/package.json +1 -1
- package/dist/topic/TopicPipeline.d.ts +38 -7
- package/dist/topic/TopicPipeline.js +177 -30
- package/dist/topic/TopicResolver.js +4 -1
- package/dist/topic/TopicTagScorer.js +12 -1
- package/dist/topic/types.d.ts +48 -0
- package/dist/version.js +1 -1
- package/dist/webui/package.json +1 -1
- package/package.json +1 -1
|
@@ -7,6 +7,7 @@ const Database_1 = require("../storage/Database");
|
|
|
7
7
|
const PixivAuth_1 = require("../auth/PixivAuth");
|
|
8
8
|
const createPixivFlowClient_1 = require("../pixiv-client/createPixivFlowClient");
|
|
9
9
|
const createTopicPipeline_1 = require("../topic/createTopicPipeline");
|
|
10
|
+
const TopicPipeline_1 = require("../topic/TopicPipeline");
|
|
10
11
|
const TopicResolver_1 = require("../topic/TopicResolver");
|
|
11
12
|
const TopicCache_1 = require("../topic/TopicCache");
|
|
12
13
|
const node_path_1 = require("node:path");
|
|
@@ -50,27 +51,48 @@ class TopicCommand extends Command_1.BaseCommand {
|
|
|
50
51
|
database.close();
|
|
51
52
|
}
|
|
52
53
|
}
|
|
53
|
-
async resolve(context,
|
|
54
|
+
async resolve(context, args, topic) {
|
|
54
55
|
if (!topic) {
|
|
55
56
|
console.error('\nUsage: pixivflow topic resolve <topic> [--type illustration|novel|all] [--refresh]\n');
|
|
56
57
|
return this.failure('Missing topic for resolve');
|
|
57
58
|
}
|
|
58
|
-
const type = (String(
|
|
59
|
-
const refresh = Boolean(
|
|
59
|
+
const type = (String(args.options.type ?? 'all'));
|
|
60
|
+
const refresh = Boolean(args.options.refresh);
|
|
60
61
|
const types = type === 'all' ? ['illustration', 'novel'] : [type];
|
|
61
62
|
return this.withClient(context, async ({ client, database }) => {
|
|
62
63
|
const cache = new TopicCache_1.TopicCache((0, node_path_1.dirname)(database.getDatabasePath()) + '/topic-cache');
|
|
64
|
+
const discovery = context.config.targets?.find((target) => target.mode === 'topic')?.topicDiscovery ?? {};
|
|
65
|
+
// §tag-provenance: the audit surface for "the original tag dominates and no
|
|
66
|
+
// related tag outranks it". Each row exposes where the tag came from, its
|
|
67
|
+
// semantic weight and whether the current tagRelations/relatedTags settings
|
|
68
|
+
// would actually search it today (the `searched` column).
|
|
69
|
+
const rows = [];
|
|
70
|
+
const lines = [];
|
|
63
71
|
for (const contentType of types) {
|
|
64
72
|
const resolver = new TopicResolver_1.TopicResolver(client, cache, context.config.download?.requestDelay ?? 500);
|
|
65
73
|
const { space, fromCache, degraded } = await resolver.resolve(topic, contentType, { refresh });
|
|
66
|
-
|
|
67
|
-
|
|
74
|
+
const walked = (0, TopicPipeline_1.selectWalkedTags)(space.tags, key(topic), discovery.tagRelations);
|
|
75
|
+
const channels = new Set((0, TopicPipeline_1.recallChannels)(walked, key(topic), relatedMode(discovery.relatedTags)).map((tag) => key(tag.name)));
|
|
76
|
+
lines.push('');
|
|
77
|
+
lines.push(`Topic: ${topic} (${contentType}) ${fromCache ? (degraded ? '· stale cache' : '· cache') : '· fresh'}`);
|
|
78
|
+
lines.push('Name'.padEnd(24) + 'Trans'.padEnd(16) + 'Source'.padEnd(28) + 'Weight'.padStart(7) + 'Score'.padStart(7) + 'Seed'.padStart(6) + 'Searched'.padStart(10));
|
|
79
|
+
const tagRows = [];
|
|
68
80
|
for (const tag of space.tags) {
|
|
69
|
-
const
|
|
70
|
-
|
|
81
|
+
const source = tag.source ?? (tag.seed ? 'seed' : 'cooccurrence');
|
|
82
|
+
const weight = tag.weight ?? tag.score;
|
|
83
|
+
const searched = channels.has(key(tag.name));
|
|
84
|
+
tagRows.push({ name: tag.name, translatedName: tag.translatedName, source, weight, score: tag.score, seed: tag.seed, searched });
|
|
85
|
+
lines.push(tag.name.slice(0, 23).padEnd(24)
|
|
86
|
+
+ (tag.translatedName ?? '-').slice(0, 15).padEnd(16)
|
|
87
|
+
+ source.padEnd(28)
|
|
88
|
+
+ weight.toFixed(2).padStart(7)
|
|
89
|
+
+ tag.score.toFixed(2).padStart(7)
|
|
90
|
+
+ (tag.seed ? 'yes' : 'no').padStart(6)
|
|
91
|
+
+ (searched ? 'yes' : 'no').padStart(10));
|
|
71
92
|
}
|
|
93
|
+
rows.push({ contentType, fromCache, degraded, tags: tagRows });
|
|
72
94
|
}
|
|
73
|
-
return this.success('
|
|
95
|
+
return this.success(lines.join('\n'), { topic, types: rows });
|
|
74
96
|
});
|
|
75
97
|
}
|
|
76
98
|
async test(context, args, topic) {
|
|
@@ -92,6 +114,7 @@ class TopicCommand extends Command_1.BaseCommand {
|
|
|
92
114
|
const { works, selection } = await pipeline.selectWorks(target, contentType, day, limit, { refresh }, {});
|
|
93
115
|
console.log(`\n=== ${contentType} topic "${topic}" day ${day} ===`);
|
|
94
116
|
console.log(`resolvedTags=${selection.resolvedTagCount} raw=${selection.rawCount} deduped=${selection.dedupedCount} accepted=${selection.acceptedCount} selected=${works.length}`);
|
|
117
|
+
console.log(`searchedTags=${(selection.searchedTags ?? []).join(', ')}`);
|
|
95
118
|
selection.selected.forEach((c, i) => {
|
|
96
119
|
console.log(` #${i + 1} id=${c.id} pop=${c.popularity.toFixed(1)} meta=${c.metadataScore.toFixed(2)} ${c.title}`);
|
|
97
120
|
});
|
|
@@ -103,7 +126,8 @@ class TopicCommand extends Command_1.BaseCommand {
|
|
|
103
126
|
getUsage() {
|
|
104
127
|
return [
|
|
105
128
|
'topic resolve <topic> [--type all|illustration|novel] [--refresh]',
|
|
106
|
-
' Show the Pixiv-derived related tag space for a topic (cached)
|
|
129
|
+
' Show the Pixiv-derived related tag space for a topic (cached), with each',
|
|
130
|
+
' tag\'s provenance (source), semantic weight, score and whether it is searched.',
|
|
107
131
|
'',
|
|
108
132
|
'topic test <topic> [--type all|illustration|novel] [--date YESTERDAY|YYYY-MM-DD] [--limit N] [--refresh]',
|
|
109
133
|
' Dry-run a daily selection: resolved tags, candidate counts, Top N (no downloads).',
|
|
@@ -111,4 +135,12 @@ class TopicCommand extends Command_1.BaseCommand {
|
|
|
111
135
|
}
|
|
112
136
|
}
|
|
113
137
|
exports.TopicCommand = TopicCommand;
|
|
138
|
+
/** Same normalization as the pipeline/resolver keys (trim + NFKC + lowercase). */
|
|
139
|
+
function key(value) {
|
|
140
|
+
return value.normalize('NFKC').trim().toLocaleLowerCase();
|
|
141
|
+
}
|
|
142
|
+
/** Unknown modes fall back to the historical 'always', as in the pipeline. */
|
|
143
|
+
function relatedMode(value) {
|
|
144
|
+
return value === 'when_seed_insufficient' || value === 'never' ? value : 'always';
|
|
145
|
+
}
|
|
114
146
|
//# sourceMappingURL=TopicCommand.js.map
|
package/dist/config/types.d.ts
CHANGED
|
@@ -4,6 +4,25 @@
|
|
|
4
4
|
export type TargetType = 'illustration' | 'novel';
|
|
5
5
|
import type { DeliveryCapabilityOverrides } from '../delivery/capabilities';
|
|
6
6
|
export type DeliveryFieldValue = string | number | boolean | string[];
|
|
7
|
+
/** Provenance names accepted by `TopicDiscoveryConfig.tagRelations`. */
|
|
8
|
+
export declare const TAG_RELATION_SOURCES: readonly ["seed", "cooccurrence", "autocomplete"];
|
|
9
|
+
export type TagRelationSource = (typeof TAG_RELATION_SOURCES)[number];
|
|
10
|
+
/**
|
|
11
|
+
* Which resolved tags may become recall channels (§tag-provenance). Every list
|
|
12
|
+
* is optional and the defaults reproduce the pre-existing behaviour: walk the
|
|
13
|
+
* whole resolved space.
|
|
14
|
+
*
|
|
15
|
+
* `deny` always wins, then `allowSources`, then `allow`. The seed tag is never
|
|
16
|
+
* dropped by `allow`/`allowSources` — only `deny` can drop it.
|
|
17
|
+
*/
|
|
18
|
+
export interface TagRelationsConfig {
|
|
19
|
+
/** Provenance categories allowed to be walked (default: all three). */
|
|
20
|
+
allowSources?: TagRelationSource[];
|
|
21
|
+
/** If non-empty: ONLY these tag names are walked (the seed tag is kept). */
|
|
22
|
+
allow?: string[];
|
|
23
|
+
/** Always dropped, wins over `allow`/`allowSources` (default: none). */
|
|
24
|
+
deny?: string[];
|
|
25
|
+
}
|
|
7
26
|
/** Discovery tuning for mode='topic'. Every value has a safe default. */
|
|
8
27
|
export interface TopicDiscoveryConfig {
|
|
9
28
|
/** Max related tags used to build the search space (default 12). */
|
|
@@ -28,6 +47,26 @@ export interface TopicDiscoveryConfig {
|
|
|
28
47
|
* - `'never'`: search the topic tag alone.
|
|
29
48
|
*/
|
|
30
49
|
relatedTags?: 'always' | 'when_seed_insufficient' | 'never';
|
|
50
|
+
/**
|
|
51
|
+
* Make the seed tag a HARD ranking tier (default `'off'`).
|
|
52
|
+
*
|
|
53
|
+
* `'off'` keeps the documented popularity-only ranking, where relevance is
|
|
54
|
+
* only an acceptance gate. `'on'` guarantees the invariant "原始 Tag 权重最高,
|
|
55
|
+
* 相关 Tag 权重不得超过原始 Tag": a work carrying the seed tag always outranks
|
|
56
|
+
* a work that only matched expanded tags, however popular the latter is.
|
|
57
|
+
*/
|
|
58
|
+
seedTier?: 'off' | 'on';
|
|
59
|
+
/** Which resolved tags may be walked as recall channels (default: all). */
|
|
60
|
+
tagRelations?: TagRelationsConfig;
|
|
61
|
+
/**
|
|
62
|
+
* Also treat a work's `translated_name` as a tag hit (default false).
|
|
63
|
+
*
|
|
64
|
+
* Today only the work's `name` is compared with the resolved tag keys. When
|
|
65
|
+
* enabled, a work whose translated tag name matches a resolved tag counts as
|
|
66
|
+
* carrying that tag (and a translated match of the seed counts as a seed hit).
|
|
67
|
+
* Default `false` keeps today's matching exactly.
|
|
68
|
+
*/
|
|
69
|
+
matchTranslatedNames?: boolean;
|
|
31
70
|
}
|
|
32
71
|
/** Candidate collection tuning for mode='topic'. */
|
|
33
72
|
export interface CandidateCollectionConfig {
|
|
@@ -876,6 +915,21 @@ export interface StandaloneConfig {
|
|
|
876
915
|
* Default: 'eager'
|
|
877
916
|
*/
|
|
878
917
|
materializationPolicy?: 'eager' | 'on-demand';
|
|
918
|
+
/**
|
|
919
|
+
* Novel cover content policy (§media-asset-pipeline / §novel-cover).
|
|
920
|
+
*
|
|
921
|
+
* Pixiv serves author covers and its own generated design covers from the
|
|
922
|
+
* same URL shape, so the content type is classified from the fetched bytes:
|
|
923
|
+
* a generated design (exactly 640x900) is never delivered as Telegram media.
|
|
924
|
+
* This key governs the UNCERTAIN case only — a cover whose header cannot be
|
|
925
|
+
* classified. 'skip' (default) is safe mode: never ship an unclassifiable
|
|
926
|
+
* cover, and log `coverType=unknown` so a future Pixiv format change is
|
|
927
|
+
* visible instead of silently leaking designs again. 'keep' prefers
|
|
928
|
+
* availability over certainty. A failed probe always keeps the cover.
|
|
929
|
+
*/
|
|
930
|
+
novelCover?: {
|
|
931
|
+
unknown?: 'skip' | 'keep';
|
|
932
|
+
};
|
|
879
933
|
};
|
|
880
934
|
}
|
|
881
935
|
//# sourceMappingURL=types.d.ts.map
|
package/dist/config/types.js
CHANGED
|
@@ -3,4 +3,7 @@
|
|
|
3
3
|
* Configuration type definitions
|
|
4
4
|
*/
|
|
5
5
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
+
exports.TAG_RELATION_SOURCES = void 0;
|
|
7
|
+
/** Provenance names accepted by `TopicDiscoveryConfig.tagRelations`. */
|
|
8
|
+
exports.TAG_RELATION_SOURCES = ['seed', 'cooccurrence', 'autocomplete'];
|
|
6
9
|
//# sourceMappingURL=types.js.map
|
|
@@ -194,6 +194,45 @@ function validateConfig(config, location, databasePath) {
|
|
|
194
194
|
if (td.relatedTags !== undefined && !['always', 'when_seed_insufficient', 'never'].includes(td.relatedTags)) {
|
|
195
195
|
errors.push(`targets[${index}].topicDiscovery.relatedTags: Must be "always", "when_seed_insufficient" or "never" (got ${String(td.relatedTags)})`);
|
|
196
196
|
}
|
|
197
|
+
// §tag-provenance: all optional; the defaults reproduce the pre-change
|
|
198
|
+
// behaviour (whole space walked, popularity-only ranking, name-only match).
|
|
199
|
+
if (td.seedTier !== undefined && !['off', 'on'].includes(td.seedTier)) {
|
|
200
|
+
errors.push(`targets[${index}].topicDiscovery.seedTier: Must be "off" or "on" (got ${String(td.seedTier)})`);
|
|
201
|
+
}
|
|
202
|
+
if (td.matchTranslatedNames !== undefined && typeof td.matchTranslatedNames !== 'boolean') {
|
|
203
|
+
errors.push(`targets[${index}].topicDiscovery.matchTranslatedNames: Must be a boolean (got ${typeof td.matchTranslatedNames})`);
|
|
204
|
+
}
|
|
205
|
+
const relations = td.tagRelations;
|
|
206
|
+
if (relations !== undefined) {
|
|
207
|
+
if (typeof relations !== 'object' || relations === null || Array.isArray(relations)) {
|
|
208
|
+
errors.push(`targets[${index}].topicDiscovery.tagRelations: Must be an object (got ${Array.isArray(relations) ? 'array' : typeof relations})`);
|
|
209
|
+
}
|
|
210
|
+
else {
|
|
211
|
+
const knownSources = ['seed', 'cooccurrence', 'autocomplete'];
|
|
212
|
+
const listErrors = (field) => (field === 'allowSources'
|
|
213
|
+
? `targets[${index}].topicDiscovery.tagRelations.allowSources: Must be an array of tag sources: ${knownSources.map((s) => `"${s}"`).join(', ')} (got ${JSON.stringify(relations.allowSources)})`
|
|
214
|
+
: `targets[${index}].topicDiscovery.tagRelations.${field}: Must be an array of tag names (got ${JSON.stringify(relations[field])})`);
|
|
215
|
+
const checkList = (field) => {
|
|
216
|
+
const value = relations[field];
|
|
217
|
+
if (value === undefined)
|
|
218
|
+
return undefined;
|
|
219
|
+
if (!Array.isArray(value) || value.some((entry) => typeof entry !== 'string' || entry.trim() === '')) {
|
|
220
|
+
errors.push(listErrors(field));
|
|
221
|
+
return undefined;
|
|
222
|
+
}
|
|
223
|
+
return value;
|
|
224
|
+
};
|
|
225
|
+
const allowSources = checkList('allowSources');
|
|
226
|
+
if (allowSources) {
|
|
227
|
+
const unknown = allowSources.filter((source) => !knownSources.includes(source));
|
|
228
|
+
if (unknown.length > 0) {
|
|
229
|
+
errors.push(`targets[${index}].topicDiscovery.tagRelations.allowSources: Unknown tag source${unknown.length > 1 ? 's' : ''} ${unknown.map((source) => `"${source}"`).join(', ')}; known sources are ${knownSources.map((s) => `"${s}"`).join(', ')}`);
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
checkList('allow');
|
|
233
|
+
checkList('deny');
|
|
234
|
+
}
|
|
235
|
+
}
|
|
197
236
|
}
|
|
198
237
|
const cc = target.candidateCollection;
|
|
199
238
|
if (cc) {
|
|
@@ -660,6 +699,11 @@ function validateConfig(config, location, databasePath) {
|
|
|
660
699
|
config.download.candidateScanLimit > 100)) {
|
|
661
700
|
warnings.push('download.candidateScanLimit: Should be an integer between 1 and 100');
|
|
662
701
|
}
|
|
702
|
+
if (config.download.novelCover?.unknown !== undefined &&
|
|
703
|
+
config.download.novelCover.unknown !== 'skip' &&
|
|
704
|
+
config.download.novelCover.unknown !== 'keep') {
|
|
705
|
+
errors.push(`download.novelCover.unknown: Must be "skip" or "keep" (got ${String(config.download.novelCover.unknown)})`);
|
|
706
|
+
}
|
|
663
707
|
}
|
|
664
708
|
// The bounded candidate scan is what stops a page full of duplicates from
|
|
665
709
|
// burning a whole scheduled slot, so a non-integer value is reported here.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/** What a fetched novel cover actually is. */
|
|
2
|
+
export type NovelCoverType = 'custom' | 'pixiv_generated' | 'unknown';
|
|
3
|
+
/**
|
|
4
|
+
* The canvas Pixiv renders its built-in novel cover designs on. Every design
|
|
5
|
+
* observed in production (floral, seasonal sweets, treasure map, genre label,
|
|
6
|
+
* city night) is delivered at exactly this size, while author covers keep their
|
|
7
|
+
* own dimensions (512x512, 768x768, 800x1200, 822x1200, 826x1169, 1024x1024 …).
|
|
8
|
+
*/
|
|
9
|
+
export declare const PIXIV_GENERATED_COVER_WIDTH = 640;
|
|
10
|
+
export declare const PIXIV_GENERATED_COVER_HEIGHT = 900;
|
|
11
|
+
export interface NovelCoverPolicy {
|
|
12
|
+
/**
|
|
13
|
+
* What to do with a cover whose content type could not be classified.
|
|
14
|
+
* 'skip' (default): safe mode — never ship an unclassifiable cover.
|
|
15
|
+
* 'keep': prefer availability over certainty.
|
|
16
|
+
*/
|
|
17
|
+
unknownCover: 'skip' | 'keep';
|
|
18
|
+
}
|
|
19
|
+
/** Production-safe default: an unclassifiable cover is never shipped. */
|
|
20
|
+
export declare const DEFAULT_NOVEL_COVER_POLICY: NovelCoverPolicy;
|
|
21
|
+
/**
|
|
22
|
+
* Classify fetched cover bytes. Returns `unknown` when the image header cannot
|
|
23
|
+
* be read (no decoder in this project, so only the header is inspected).
|
|
24
|
+
*/
|
|
25
|
+
export declare function classifyNovelCover(cover: ArrayBuffer | Uint8Array | null | undefined): NovelCoverType;
|
|
26
|
+
/** The delivery decision for one classified cover under a policy. */
|
|
27
|
+
export declare function coverDeliveryDecision(policy: NovelCoverPolicy, type: NovelCoverType): 'deliver' | 'skip';
|
|
28
|
+
//# sourceMappingURL=NovelCoverPolicy.d.ts.map
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.DEFAULT_NOVEL_COVER_POLICY = exports.PIXIV_GENERATED_COVER_HEIGHT = exports.PIXIV_GENERATED_COVER_WIDTH = void 0;
|
|
4
|
+
exports.classifyNovelCover = classifyNovelCover;
|
|
5
|
+
exports.coverDeliveryDecision = coverDeliveryDecision;
|
|
6
|
+
/**
|
|
7
|
+
* Novel cover content policy (§media-asset-pipeline / §novel-cover).
|
|
8
|
+
*
|
|
9
|
+
* Pixiv exposes a novel cover as a single URL and gives NO field that says what
|
|
10
|
+
* the image actually IS. Two very different things arrive through that one
|
|
11
|
+
* field:
|
|
12
|
+
*
|
|
13
|
+
* custom the author's own uploaded artwork — real content, must ship
|
|
14
|
+
* pixiv_generated Pixiv's built-in design cover, rendered per novel on a fixed
|
|
15
|
+
* 640x900 canvas with the title typeset on it — zero content
|
|
16
|
+
* value, must never ship as Telegram media
|
|
17
|
+
*
|
|
18
|
+
* The URL cannot separate them (both live under `novel-cover-master/img/...`
|
|
19
|
+
* behind a unique hash, and the old `novel-cover-master-default` placeholder no
|
|
20
|
+
* longer occurs), so the classification is made from the fetched bytes' header.
|
|
21
|
+
* A third outcome exists on purpose:
|
|
22
|
+
*
|
|
23
|
+
* unknown the payload could not be classified (unrecognized header,
|
|
24
|
+
* format Pixiv does not normally serve)
|
|
25
|
+
*
|
|
26
|
+
* `unknown` is a POLICY decision, not an accident: sending it can leak a design
|
|
27
|
+
* cover into a public channel (the exact bug this pipeline fixes), while
|
|
28
|
+
* dropping it costs one illustration on the card. The production-safe default is
|
|
29
|
+
* therefore to skip it and log `coverType=unknown` loudly, so a future Pixiv
|
|
30
|
+
* cover-format change surfaces as a visible log line instead of silently
|
|
31
|
+
* shipping generated covers again.
|
|
32
|
+
*
|
|
33
|
+
* A FAILED probe (network / auth / rate limit) is deliberately NOT `unknown`:
|
|
34
|
+
* it is `probe_failed` and always keeps the cover, because a transient fetch
|
|
35
|
+
* error must never cost a real one.
|
|
36
|
+
*/
|
|
37
|
+
const imageDimensions_1 = require("../../utils/imageDimensions");
|
|
38
|
+
/**
|
|
39
|
+
* The canvas Pixiv renders its built-in novel cover designs on. Every design
|
|
40
|
+
* observed in production (floral, seasonal sweets, treasure map, genre label,
|
|
41
|
+
* city night) is delivered at exactly this size, while author covers keep their
|
|
42
|
+
* own dimensions (512x512, 768x768, 800x1200, 822x1200, 826x1169, 1024x1024 …).
|
|
43
|
+
*/
|
|
44
|
+
exports.PIXIV_GENERATED_COVER_WIDTH = 640;
|
|
45
|
+
exports.PIXIV_GENERATED_COVER_HEIGHT = 900;
|
|
46
|
+
/** Production-safe default: an unclassifiable cover is never shipped. */
|
|
47
|
+
exports.DEFAULT_NOVEL_COVER_POLICY = { unknownCover: 'skip' };
|
|
48
|
+
/**
|
|
49
|
+
* Classify fetched cover bytes. Returns `unknown` when the image header cannot
|
|
50
|
+
* be read (no decoder in this project, so only the header is inspected).
|
|
51
|
+
*/
|
|
52
|
+
function classifyNovelCover(cover) {
|
|
53
|
+
const dimensions = (0, imageDimensions_1.readImageDimensions)(cover);
|
|
54
|
+
if (!dimensions)
|
|
55
|
+
return 'unknown';
|
|
56
|
+
if (dimensions.width === exports.PIXIV_GENERATED_COVER_WIDTH &&
|
|
57
|
+
dimensions.height === exports.PIXIV_GENERATED_COVER_HEIGHT) {
|
|
58
|
+
return 'pixiv_generated';
|
|
59
|
+
}
|
|
60
|
+
return 'custom';
|
|
61
|
+
}
|
|
62
|
+
/** The delivery decision for one classified cover under a policy. */
|
|
63
|
+
function coverDeliveryDecision(policy, type) {
|
|
64
|
+
if (type === 'pixiv_generated')
|
|
65
|
+
return 'skip';
|
|
66
|
+
if (type === 'unknown')
|
|
67
|
+
return policy.unknownCover === 'keep' ? 'deliver' : 'skip';
|
|
68
|
+
return 'deliver';
|
|
69
|
+
}
|
|
70
|
+
//# sourceMappingURL=NovelCoverPolicy.js.map
|
|
@@ -6,6 +6,7 @@ const RankingService_1 = require("./RankingService");
|
|
|
6
6
|
const IllustrationDownloader_1 = require("./IllustrationDownloader");
|
|
7
7
|
const NovelDownloader_1 = require("./NovelDownloader");
|
|
8
8
|
const MaterializationPolicy_1 = require("../domain/media/MaterializationPolicy");
|
|
9
|
+
const NovelCoverPolicy_1 = require("../domain/media/NovelCoverPolicy");
|
|
9
10
|
const ProgressReporter_1 = require("./report/ProgressReporter");
|
|
10
11
|
const DownloadPlanner_1 = require("./plan/DownloadPlanner");
|
|
11
12
|
const DownloadExecutor_1 = require("./exec/DownloadExecutor");
|
|
@@ -129,7 +130,10 @@ class DownloadManager {
|
|
|
129
130
|
const materialization = config.download?.materializationPolicy
|
|
130
131
|
? { mode: config.download?.materializationPolicy }
|
|
131
132
|
: MaterializationPolicy_1.DEFAULT_MATERIALIZATION_POLICY;
|
|
132
|
-
|
|
133
|
+
const novelCoverPolicy = {
|
|
134
|
+
unknownCover: config.download?.novelCover?.unknown ?? NovelCoverPolicy_1.DEFAULT_NOVEL_COVER_POLICY.unknownCover,
|
|
135
|
+
};
|
|
136
|
+
this.novelDownloader = new NovelDownloader_1.NovelDownloader(client, database, fileService, database, undefined, materialization, novelCoverPolicy);
|
|
133
137
|
this.planner = new DownloadPlanner_1.DownloadPlanner(database, {
|
|
134
138
|
// `scope` is a single legacy target name, or the full fan-out array for a
|
|
135
139
|
// multi-platform target (then only works confirmed on EVERY platform count).
|
|
@@ -6,6 +6,7 @@ import { PixivNovel } from '@redtidev/pixiv-client';
|
|
|
6
6
|
import { DownloadedArtifact } from '../delivery/types';
|
|
7
7
|
import { type MaterializationPolicy } from '../domain/media/MaterializationPolicy';
|
|
8
8
|
import { type MediaMaterializer } from './materialization/MediaMaterializer';
|
|
9
|
+
import { type NovelCoverPolicy } from '../domain/media/NovelCoverPolicy';
|
|
9
10
|
import type { Database } from '../storage/Database';
|
|
10
11
|
export declare class NovelDownloader {
|
|
11
12
|
private readonly client;
|
|
@@ -14,7 +15,8 @@ export declare class NovelDownloader {
|
|
|
14
15
|
private readonly metadataDb?;
|
|
15
16
|
private readonly materializer;
|
|
16
17
|
private readonly materializationPolicy;
|
|
17
|
-
|
|
18
|
+
private readonly novelCoverPolicy;
|
|
19
|
+
constructor(client: IPixivClient, database: IDatabase, fileService: IFileService, metadataDb?: Database | undefined, materializer?: MediaMaterializer, materializationPolicy?: MaterializationPolicy, novelCoverPolicy?: NovelCoverPolicy);
|
|
18
20
|
download(novel: PixivNovel, tag: string, target: TargetConfig): Promise<DownloadedArtifact | undefined>;
|
|
19
21
|
/**
|
|
20
22
|
* Resolves the cover URL a novel should ship with (§novel-cover).
|
|
@@ -22,9 +24,18 @@ export declare class NovelDownloader {
|
|
|
22
24
|
* Pixiv's "author set no cover" case is invisible in the API: the design it
|
|
23
25
|
* renders (title typeset on a template) is served from the same CDN path with
|
|
24
26
|
* a unique hash as a real cover, so the URL cannot decide. The candidate cover
|
|
25
|
-
* is therefore fetched once and its header
|
|
26
|
-
* exactly 640x900.
|
|
27
|
-
*
|
|
27
|
+
* is therefore fetched once and classified from its header; Pixiv's design
|
|
28
|
+
* canvas is exactly 640x900.
|
|
29
|
+
*
|
|
30
|
+
* Policy (§media-asset-pipeline):
|
|
31
|
+
* - custom → deliver the cover
|
|
32
|
+
* - pixiv_generated → never deliver (a design cover is not content and must
|
|
33
|
+
* not become Telegram media)
|
|
34
|
+
* - unknown → policy decision, default 'skip' (safe mode), so a future
|
|
35
|
+
* Pixiv cover-format change surfaces as a loud
|
|
36
|
+
* `coverType=unknown` log instead of leaking silently
|
|
37
|
+
* A FAILED probe (network / auth / rate limit) is a separate case and always
|
|
38
|
+
* keeps the cover: a transient fetch error must never cost a real one.
|
|
28
39
|
*/
|
|
29
40
|
private resolveCoverUrl;
|
|
30
41
|
}
|
|
@@ -43,6 +43,7 @@ const Artifact_1 = require("../domain/media/Artifact");
|
|
|
43
43
|
const MediaMaterializer_1 = require("./materialization/MediaMaterializer");
|
|
44
44
|
const novelMarkers_1 = require("./novelMarkers");
|
|
45
45
|
const novelCover_1 = require("./novelCover");
|
|
46
|
+
const NovelCoverPolicy_1 = require("../domain/media/NovelCoverPolicy");
|
|
46
47
|
const zip_1 = require("../utils/zip");
|
|
47
48
|
const LANGUAGE_CACHE_TTL_MS = 7 * 24 * 60 * 60 * 1000;
|
|
48
49
|
class NovelDownloader {
|
|
@@ -52,13 +53,15 @@ class NovelDownloader {
|
|
|
52
53
|
metadataDb;
|
|
53
54
|
materializer;
|
|
54
55
|
materializationPolicy;
|
|
55
|
-
|
|
56
|
+
novelCoverPolicy;
|
|
57
|
+
constructor(client, database, fileService, metadataDb, materializer, materializationPolicy = MaterializationPolicy_1.DEFAULT_MATERIALIZATION_POLICY, novelCoverPolicy = NovelCoverPolicy_1.DEFAULT_NOVEL_COVER_POLICY) {
|
|
56
58
|
this.client = client;
|
|
57
59
|
this.database = database;
|
|
58
60
|
this.fileService = fileService;
|
|
59
61
|
this.metadataDb = metadataDb;
|
|
60
62
|
this.materializer = materializer ?? new MediaMaterializer_1.PixivMediaMaterializer(client, fileService);
|
|
61
63
|
this.materializationPolicy = materializationPolicy;
|
|
64
|
+
this.novelCoverPolicy = novelCoverPolicy;
|
|
62
65
|
}
|
|
63
66
|
async download(novel, tag, target) {
|
|
64
67
|
// Metadata cache: language filtering otherwise pulls FULL text per candidate
|
|
@@ -405,30 +408,59 @@ class NovelDownloader {
|
|
|
405
408
|
* Pixiv's "author set no cover" case is invisible in the API: the design it
|
|
406
409
|
* renders (title typeset on a template) is served from the same CDN path with
|
|
407
410
|
* a unique hash as a real cover, so the URL cannot decide. The candidate cover
|
|
408
|
-
* is therefore fetched once and its header
|
|
409
|
-
* exactly 640x900.
|
|
410
|
-
*
|
|
411
|
+
* is therefore fetched once and classified from its header; Pixiv's design
|
|
412
|
+
* canvas is exactly 640x900.
|
|
413
|
+
*
|
|
414
|
+
* Policy (§media-asset-pipeline):
|
|
415
|
+
* - custom → deliver the cover
|
|
416
|
+
* - pixiv_generated → never deliver (a design cover is not content and must
|
|
417
|
+
* not become Telegram media)
|
|
418
|
+
* - unknown → policy decision, default 'skip' (safe mode), so a future
|
|
419
|
+
* Pixiv cover-format change surfaces as a loud
|
|
420
|
+
* `coverType=unknown` log instead of leaking silently
|
|
421
|
+
* A FAILED probe (network / auth / rate limit) is a separate case and always
|
|
422
|
+
* keeps the cover: a transient fetch error must never cost a real one.
|
|
411
423
|
*/
|
|
412
424
|
async resolveCoverUrl(novelId, coverUrl) {
|
|
413
425
|
const normalized = (0, novelCover_1.normalizeNovelCoverUrl)(coverUrl);
|
|
414
426
|
if (!normalized)
|
|
415
427
|
return null;
|
|
428
|
+
let cover;
|
|
416
429
|
try {
|
|
417
|
-
|
|
418
|
-
if ((0, novelCover_1.isPixivDesignCoverImage)(cover)) {
|
|
419
|
-
logger_1.logger.info(`Novel ${novelId} cover is a Pixiv design cover (${novelCover_1.PIXIV_DESIGN_COVER_WIDTH}x${novelCover_1.PIXIV_DESIGN_COVER_HEIGHT}); delivering without a cover`, { novelId, coverUrl: normalized, designCover: true });
|
|
420
|
-
return null;
|
|
421
|
-
}
|
|
422
|
-
return normalized;
|
|
430
|
+
cover = await this.client.downloadImage(normalized);
|
|
423
431
|
}
|
|
424
432
|
catch (error) {
|
|
425
|
-
logger_1.logger.warn(`Novel ${novelId} cover probe failed; keeping the cover`, {
|
|
433
|
+
logger_1.logger.warn(`Novel ${novelId} cover probe failed; keeping the cover (coverType=probe_failed)`, {
|
|
426
434
|
novelId,
|
|
427
435
|
coverUrl: normalized,
|
|
436
|
+
coverType: 'probe_failed',
|
|
428
437
|
reason: error instanceof Error ? error.message : String(error),
|
|
429
438
|
});
|
|
430
439
|
return normalized;
|
|
431
440
|
}
|
|
441
|
+
const coverType = (0, NovelCoverPolicy_1.classifyNovelCover)(cover);
|
|
442
|
+
const decision = (0, NovelCoverPolicy_1.coverDeliveryDecision)(this.novelCoverPolicy, coverType);
|
|
443
|
+
const canvas = `${NovelCoverPolicy_1.PIXIV_GENERATED_COVER_WIDTH}x${NovelCoverPolicy_1.PIXIV_GENERATED_COVER_HEIGHT}`;
|
|
444
|
+
if (decision === 'skip') {
|
|
445
|
+
if (coverType === 'pixiv_generated') {
|
|
446
|
+
logger_1.logger.info(`Novel ${novelId} cover is a Pixiv generated design (${canvas}); delivering without a cover (coverType=${coverType})`, { novelId, coverUrl: normalized, coverType, canvas });
|
|
447
|
+
}
|
|
448
|
+
else {
|
|
449
|
+
logger_1.logger.warn(`Novel ${novelId} cover could not be classified; skipping it per novelCover.unknown=skip (coverType=${coverType})`, { novelId, coverUrl: normalized, coverType, policy: this.novelCoverPolicy.unknownCover });
|
|
450
|
+
}
|
|
451
|
+
return null;
|
|
452
|
+
}
|
|
453
|
+
if (coverType === 'unknown') {
|
|
454
|
+
logger_1.logger.warn(`Novel ${novelId} cover could not be classified; keeping it per novelCover.unknown=keep (coverType=${coverType})`, { novelId, coverUrl: normalized, coverType, policy: this.novelCoverPolicy.unknownCover });
|
|
455
|
+
}
|
|
456
|
+
else {
|
|
457
|
+
logger_1.logger.debug(`Novel ${novelId} cover classified (coverType=${coverType})`, {
|
|
458
|
+
novelId,
|
|
459
|
+
coverUrl: normalized,
|
|
460
|
+
coverType,
|
|
461
|
+
});
|
|
462
|
+
}
|
|
463
|
+
return normalized;
|
|
432
464
|
}
|
|
433
465
|
}
|
|
434
466
|
exports.NovelDownloader = NovelDownloader;
|
|
@@ -27,17 +27,16 @@
|
|
|
27
27
|
*/
|
|
28
28
|
import { MediaAsset } from '../domain/media/MediaAsset';
|
|
29
29
|
/**
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
* is delivered at exactly this size, while author covers keep their own
|
|
33
|
-
* dimensions (512x512, 768x768, 800x1200, 822x1200, 826x1169, 1024x1024 …).
|
|
30
|
+
* Historical names for the generated-cover canvas; the classification and the
|
|
31
|
+
* policy live in `src/domain/media/NovelCoverPolicy.ts` (§media-asset-pipeline).
|
|
34
32
|
*/
|
|
35
33
|
export declare const PIXIV_DESIGN_COVER_WIDTH = 640;
|
|
36
34
|
export declare const PIXIV_DESIGN_COVER_HEIGHT = 900;
|
|
37
35
|
/**
|
|
38
36
|
* True when the fetched cover bytes are one of Pixiv's generated designs.
|
|
39
|
-
* Unknown formats / unreadable payloads return false
|
|
40
|
-
*
|
|
37
|
+
* Unknown formats / unreadable payloads return false — this predicate is the
|
|
38
|
+
* raw classification only; whether an unclassifiable cover is delivered is the
|
|
39
|
+
* caller's policy (`coverDeliveryDecision`, default: skipped).
|
|
41
40
|
*/
|
|
42
41
|
export declare function isPixivDesignCoverImage(cover: ArrayBuffer | Uint8Array | null | undefined): boolean;
|
|
43
42
|
export declare function normalizeNovelCoverUrl(coverUrl?: string | null): string | null;
|
|
@@ -32,27 +32,24 @@ exports.novelCoverAsset = novelCoverAsset;
|
|
|
32
32
|
* positional guessing.
|
|
33
33
|
*/
|
|
34
34
|
const MediaAsset_1 = require("../domain/media/MediaAsset");
|
|
35
|
-
const
|
|
35
|
+
const NovelCoverPolicy_1 = require("../domain/media/NovelCoverPolicy");
|
|
36
36
|
const DEFAULT_COVER_PATTERN = /novel-cover-(master-)?default/i;
|
|
37
37
|
/** Strip the `/c/<spec>/` resizer segment to recover the original-size URL. */
|
|
38
38
|
const RESIZED_COVER_PATTERN = /\/c\/[^/]+\/(novel-cover-master\/)/;
|
|
39
39
|
/**
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
* is delivered at exactly this size, while author covers keep their own
|
|
43
|
-
* dimensions (512x512, 768x768, 800x1200, 822x1200, 826x1169, 1024x1024 …).
|
|
40
|
+
* Historical names for the generated-cover canvas; the classification and the
|
|
41
|
+
* policy live in `src/domain/media/NovelCoverPolicy.ts` (§media-asset-pipeline).
|
|
44
42
|
*/
|
|
45
|
-
exports.PIXIV_DESIGN_COVER_WIDTH =
|
|
46
|
-
exports.PIXIV_DESIGN_COVER_HEIGHT =
|
|
43
|
+
exports.PIXIV_DESIGN_COVER_WIDTH = NovelCoverPolicy_1.PIXIV_GENERATED_COVER_WIDTH;
|
|
44
|
+
exports.PIXIV_DESIGN_COVER_HEIGHT = NovelCoverPolicy_1.PIXIV_GENERATED_COVER_HEIGHT;
|
|
47
45
|
/**
|
|
48
46
|
* True when the fetched cover bytes are one of Pixiv's generated designs.
|
|
49
|
-
* Unknown formats / unreadable payloads return false
|
|
50
|
-
*
|
|
47
|
+
* Unknown formats / unreadable payloads return false — this predicate is the
|
|
48
|
+
* raw classification only; whether an unclassifiable cover is delivered is the
|
|
49
|
+
* caller's policy (`coverDeliveryDecision`, default: skipped).
|
|
51
50
|
*/
|
|
52
51
|
function isPixivDesignCoverImage(cover) {
|
|
53
|
-
|
|
54
|
-
return (dimensions?.width === exports.PIXIV_DESIGN_COVER_WIDTH &&
|
|
55
|
-
dimensions?.height === exports.PIXIV_DESIGN_COVER_HEIGHT);
|
|
52
|
+
return (0, NovelCoverPolicy_1.classifyNovelCover)(cover) === 'pixiv_generated';
|
|
56
53
|
}
|
|
57
54
|
function normalizeNovelCoverUrl(coverUrl) {
|
|
58
55
|
const url = typeof coverUrl === 'string' ? coverUrl.trim() : '';
|
package/dist/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { TargetConfig } from '../config';
|
|
2
2
|
import type { TopicResolver } from './TopicResolver';
|
|
3
|
-
import type { TopicCandidate, TopicClient, TopicCollectOptions, TopicContentType, TopicDiscoveryOptions, WorkLike } from './types';
|
|
3
|
+
import type { RelatedTagMode, ResolvedTag, TopicCandidate, TopicClient, TopicCollectOptions, TopicContentType, TopicDiscoveryOptions, TopicRelationsOptions, WorkLike } from './types';
|
|
4
4
|
export interface TopicSelection {
|
|
5
5
|
candidates: TopicCandidate[];
|
|
6
6
|
selected: TopicCandidate[];
|
|
@@ -18,6 +18,26 @@ export interface TopicSelection {
|
|
|
18
18
|
/** Works seen more than once across the topic tag space (recorded, dropped). */
|
|
19
19
|
duplicateRemovedCount: number;
|
|
20
20
|
}
|
|
21
|
+
/**
|
|
22
|
+
* The tag the operator asked for, even when `tagRelations` filters it out.
|
|
23
|
+
*
|
|
24
|
+
* `deny` may drop the seed, but a space filtered down to nothing must still be
|
|
25
|
+
* usable: the pipeline then falls back to the seed tag alone, exactly like the
|
|
26
|
+
* resolver's `seedOnly()` degradation. `weight` defaults to the tag's `score`
|
|
27
|
+
* for spaces persisted before provenance existed.
|
|
28
|
+
*/
|
|
29
|
+
export declare function seedFallbackTag(seedTag: ResolvedTag): ResolvedTag;
|
|
30
|
+
/**
|
|
31
|
+
* Applies `topicDiscovery.tagRelations` to a resolved space (§tag-provenance).
|
|
32
|
+
*
|
|
33
|
+
* Deny beats allow, and deny is the only thing that can drop the seed tag. A
|
|
34
|
+
* tag whose provenance is unknown (a space persisted before `source` existed, or
|
|
35
|
+
* a hand-written space) is treated as `cooccurrence` so that `allowSources` can
|
|
36
|
+
* still narrow it while the default (all sources) keeps walking everything.
|
|
37
|
+
*/
|
|
38
|
+
export declare function selectWalkedTags(tags: readonly ResolvedTag[], seedKey: string, relations?: TopicRelationsOptions): ResolvedTag[];
|
|
39
|
+
/** Which of the walked tags a `relatedTags` mode will actually search today. */
|
|
40
|
+
export declare function recallChannels(walked: readonly ResolvedTag[], seedKey: string, relatedMode: RelatedTagMode): ResolvedTag[];
|
|
21
41
|
/**
|
|
22
42
|
* Resolves a topic to a tag space, collects that day's works across the tags,
|
|
23
43
|
* filters by lightweight metadata relevance and ranks by local popularity, then
|
|
@@ -63,6 +83,16 @@ export declare class TopicPipeline {
|
|
|
63
83
|
* Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
|
|
64
84
|
* title/caption add smaller boosts. The seed tag is strong evidence.
|
|
65
85
|
* No text model — case/symbol-insensitive substring matching only.
|
|
86
|
+
*
|
|
87
|
+
* `matchTranslatedNames` (default false) additionally counts a work's
|
|
88
|
+
* `translated_name` as a hit for the resolved tag it translates to, so a
|
|
89
|
+
* topic written in one language can still match works tagged in another. The
|
|
90
|
+
* translated name is matched against the resolved space — never added to
|
|
91
|
+
* `candidate.tags` — so the work is not reported as carrying a tag it lacks.
|
|
92
|
+
*
|
|
93
|
+
* A resolved tag counts at most ONCE per work, whether it matched through the
|
|
94
|
+
* tag name or through its translation (and a work that repeats a tag, as Pixiv
|
|
95
|
+
* payloads do for a tag plus its translation, does not double its weight).
|
|
66
96
|
*/
|
|
67
97
|
private metadataScore;
|
|
68
98
|
/**
|
|
@@ -72,12 +102,13 @@ export declare class TopicPipeline {
|
|
|
72
102
|
* alone. A work with a higher metadata score does NOT outrank a more popular
|
|
73
103
|
* accepted work.
|
|
74
104
|
*
|
|
75
|
-
* `seedKey` adds a single tier in front of that popularity order and is
|
|
76
|
-
* passed by the seed-first recall modes (§topic-recall)
|
|
77
|
-
*
|
|
78
|
-
*
|
|
79
|
-
*
|
|
80
|
-
* is
|
|
105
|
+
* `seedKey` adds a single tier in front of that popularity order and is
|
|
106
|
+
* passed by the seed-first recall modes (§topic-recall) and by
|
|
107
|
+
* `topicDiscovery.seedTier: 'on'` (§tag-provenance): a work that actually
|
|
108
|
+
* carries the topic tag outranks a related-only work, however popular the
|
|
109
|
+
* latter is, and popularity decides within each tier. With `seedTier: 'off'`
|
|
110
|
+
* and the default `relatedTags: 'always'` no `seedKey` is passed, so the
|
|
111
|
+
* documented popularity-only ranking is unchanged.
|
|
81
112
|
*/
|
|
82
113
|
private popCompare;
|
|
83
114
|
private rankCompare;
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.TopicPipeline = void 0;
|
|
4
|
+
exports.seedFallbackTag = seedFallbackTag;
|
|
5
|
+
exports.selectWalkedTags = selectWalkedTags;
|
|
6
|
+
exports.recallChannels = recallChannels;
|
|
4
7
|
const promises_1 = require("node:timers/promises");
|
|
5
8
|
const logger_1 = require("../logger");
|
|
6
9
|
const pixiv_utils_1 = require("../utils/pixiv-utils");
|
|
@@ -14,6 +17,105 @@ const STOP_TAGS = new Set([
|
|
|
14
17
|
'1000users入り', '5000users入り', '10000users入り', '500users入り', '100users入り',
|
|
15
18
|
'pixiv', 'commission', 'skeb', '依頼絵', '仕事絵',
|
|
16
19
|
]);
|
|
20
|
+
/** Normalization shared by the tag-relation filter and the pipeline itself. */
|
|
21
|
+
function normalizeKey(value) {
|
|
22
|
+
return value.trim().normalize('NFKC').toLocaleLowerCase();
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* The tag the operator asked for, even when `tagRelations` filters it out.
|
|
26
|
+
*
|
|
27
|
+
* `deny` may drop the seed, but a space filtered down to nothing must still be
|
|
28
|
+
* usable: the pipeline then falls back to the seed tag alone, exactly like the
|
|
29
|
+
* resolver's `seedOnly()` degradation. `weight` defaults to the tag's `score`
|
|
30
|
+
* for spaces persisted before provenance existed.
|
|
31
|
+
*/
|
|
32
|
+
function seedFallbackTag(seedTag) {
|
|
33
|
+
if (seedTag.source === 'seed' && seedTag.weight !== undefined)
|
|
34
|
+
return seedTag;
|
|
35
|
+
return { ...seedTag, source: 'seed', weight: seedTag.weight ?? seedTag.score, seed: true };
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Applies `topicDiscovery.tagRelations` to a resolved space (§tag-provenance).
|
|
39
|
+
*
|
|
40
|
+
* Deny beats allow, and deny is the only thing that can drop the seed tag. A
|
|
41
|
+
* tag whose provenance is unknown (a space persisted before `source` existed, or
|
|
42
|
+
* a hand-written space) is treated as `cooccurrence` so that `allowSources` can
|
|
43
|
+
* still narrow it while the default (all sources) keeps walking everything.
|
|
44
|
+
*/
|
|
45
|
+
function selectWalkedTags(tags, seedKey, relations = {}) {
|
|
46
|
+
const seedTag = tags.find((tag) => normalizeKey(tag.name) === seedKey);
|
|
47
|
+
const allowed = tags.filter((tag) => normalizeKey(tag.name) !== seedKey);
|
|
48
|
+
const deny = (relations.deny ?? []).map(normalizeKey);
|
|
49
|
+
const deniedKeys = new Set(deny);
|
|
50
|
+
const seedDenied = seedTag !== undefined && deniedKeys.has(seedKey);
|
|
51
|
+
if (seedDenied)
|
|
52
|
+
deniedKeys.clear(); // the seed fallback ignores deny
|
|
53
|
+
const allow = (relations.allow ?? []).map(normalizeKey);
|
|
54
|
+
const allowKeys = new Set(allow);
|
|
55
|
+
const open = allowKeys.size === 0;
|
|
56
|
+
const sources = relations.allowSources;
|
|
57
|
+
const sourceAllowed = (tag) => {
|
|
58
|
+
if (!sources || sources.length === 0)
|
|
59
|
+
return true;
|
|
60
|
+
const name = normalizeKey(tag.name);
|
|
61
|
+
if (allowKeys.has(name))
|
|
62
|
+
return true; // an explicit allow beats allowSources
|
|
63
|
+
const provenance = tag.source;
|
|
64
|
+
if (provenance === undefined)
|
|
65
|
+
return sources.includes('cooccurrence');
|
|
66
|
+
if (provenance === 'cooccurrence+autocomplete') {
|
|
67
|
+
return sources.includes('cooccurrence') || sources.includes('autocomplete');
|
|
68
|
+
}
|
|
69
|
+
return sources.includes(provenance);
|
|
70
|
+
};
|
|
71
|
+
const walked = allowed.filter((tag) => {
|
|
72
|
+
const name = normalizeKey(tag.name);
|
|
73
|
+
if (deniedKeys.has(name))
|
|
74
|
+
return false;
|
|
75
|
+
if (!open && !allowKeys.has(name))
|
|
76
|
+
return false;
|
|
77
|
+
return sourceAllowed(tag);
|
|
78
|
+
});
|
|
79
|
+
// No seed tag in the space at all: there is nothing to dominate, so fall back
|
|
80
|
+
// to walking the whole (filtered) space — the pre-provenance behaviour.
|
|
81
|
+
if (!seedTag)
|
|
82
|
+
return walked;
|
|
83
|
+
// A denied seed empties the space, exactly like the resolver's seedOnly():
|
|
84
|
+
// the operator still gets their own topic, and no related tag is walked.
|
|
85
|
+
if (seedDenied)
|
|
86
|
+
return [seedFallbackTag(seedTag)];
|
|
87
|
+
return [seedTag, ...walked];
|
|
88
|
+
}
|
|
89
|
+
/** Every spelling under which a set of resolved tags can be recognized. */
|
|
90
|
+
function buildTagAliases(tags, matchTranslatedNames) {
|
|
91
|
+
const aliases = new Map();
|
|
92
|
+
for (const tag of tags) {
|
|
93
|
+
const primary = normalizeKey(tag.name);
|
|
94
|
+
const entry = { primary, weight: tag.weight ?? tag.score };
|
|
95
|
+
if (!aliases.has(primary))
|
|
96
|
+
aliases.set(primary, entry);
|
|
97
|
+
if (!matchTranslatedNames)
|
|
98
|
+
continue;
|
|
99
|
+
// The work-side `translated_name` for this tag is the same resolved tag: the
|
|
100
|
+
// space still exposes one tag, it is simply recognized under two keys.
|
|
101
|
+
const translated = tag.translatedName ? normalizeKey(tag.translatedName) : '';
|
|
102
|
+
if (translated && !aliases.has(translated))
|
|
103
|
+
aliases.set(translated, entry);
|
|
104
|
+
}
|
|
105
|
+
return aliases;
|
|
106
|
+
}
|
|
107
|
+
/** Which of the walked tags a `relatedTags` mode will actually search today. */
|
|
108
|
+
function recallChannels(walked, seedKey, relatedMode) {
|
|
109
|
+
const seedTags = walked.filter((tag) => normalizeKey(tag.name) === seedKey);
|
|
110
|
+
// A space with no seed tag has nothing to protect: every remaining tag is the
|
|
111
|
+
// only channel there is, so the whole space is walked whatever the mode
|
|
112
|
+
// (pre-provenance fallback for a hand-written space).
|
|
113
|
+
if (seedTags.length === 0)
|
|
114
|
+
return [...walked];
|
|
115
|
+
if (relatedMode === 'never' || relatedMode === 'when_seed_insufficient')
|
|
116
|
+
return seedTags;
|
|
117
|
+
return [...seedTags, ...walked.filter((tag) => normalizeKey(tag.name) !== seedKey)];
|
|
118
|
+
}
|
|
17
119
|
/**
|
|
18
120
|
* Resolves a topic to a tag space, collects that day's works across the tags,
|
|
19
121
|
* filters by lightweight metadata relevance and ranks by local popularity, then
|
|
@@ -42,7 +144,6 @@ class TopicPipeline {
|
|
|
42
144
|
async selectWorks(target, contentType, day, limit, discovery, collect) {
|
|
43
145
|
const topic = (target.topic ?? '').trim();
|
|
44
146
|
const { space } = await this.resolver.resolve(topic, contentType, discovery);
|
|
45
|
-
const tagScores = new Map(space.tags.map((t) => [this.key(t.name), t.score]));
|
|
46
147
|
const maxPerTag = this.bound(collect.maxPerTag, COLLECT_DEFAULTS.maxPerTag, 5, 100);
|
|
47
148
|
const maxCandidates = this.bound(collect.maxCandidates, COLLECT_DEFAULTS.maxCandidates, 20, 500);
|
|
48
149
|
const minMetadataScore = collect.minMetadataScore ?? COLLECT_DEFAULTS.minMetadataScore;
|
|
@@ -55,11 +156,22 @@ class TopicPipeline {
|
|
|
55
156
|
let rawCount = 0;
|
|
56
157
|
let aiExcludedCount = 0;
|
|
57
158
|
let duplicateRemovedCount = 0;
|
|
58
|
-
const tagNames = space.tags.map((t) => t.name);
|
|
59
159
|
const seedKey = this.key(topic);
|
|
160
|
+
// §tag-provenance: apply tagRelations where the walked list is built, so the
|
|
161
|
+
// filter governs seedTags/relatedTags BEFORE any search is issued. The seed
|
|
162
|
+
// tag survives allow/allowSources; a deny entry that empties the space falls
|
|
163
|
+
// back to the seed tag alone, exactly like the resolver's seedOnly().
|
|
164
|
+
const matchTranslatedNames = discovery.matchTranslatedNames === true;
|
|
165
|
+
const walked = selectWalkedTags(space.tags, seedKey, discovery.tagRelations);
|
|
166
|
+
const tagNames = walked.map((t) => t.name);
|
|
60
167
|
const seedTags = tagNames.filter((name) => this.key(name) === seedKey);
|
|
61
168
|
const relatedTags = tagNames.filter((name) => this.key(name) !== seedKey);
|
|
169
|
+
// Ranking reads the documented semantic weight, falling back to `score` for
|
|
170
|
+
// spaces persisted before provenance existed. The two are always equal.
|
|
171
|
+
const tagScores = buildTagAliases(walked, matchTranslatedNames);
|
|
62
172
|
const searchedTags = [];
|
|
173
|
+
// A hard seed tier: 'on' always, and the seed-first modes keep their layer.
|
|
174
|
+
const hardSeedTier = discovery.seedTier === 'on' || relatedMode !== 'always';
|
|
63
175
|
const collectTag = async (tag) => {
|
|
64
176
|
// Cancellation is checked between tags, so a cancelled run stops issuing
|
|
65
177
|
// new searches even when the aborted request itself had already returned.
|
|
@@ -100,11 +212,11 @@ class TopicPipeline {
|
|
|
100
212
|
if (seedTags.length === 0 || relatedMode === 'always') {
|
|
101
213
|
// No seed tag in the space (hand-written space): keep walking everything
|
|
102
214
|
// rather than returning nothing.
|
|
103
|
-
await runTags(
|
|
215
|
+
await runTags(recallChannels(walked, seedKey, relatedMode).map((t) => t.name));
|
|
104
216
|
}
|
|
105
217
|
else {
|
|
106
218
|
await runTags(seedTags);
|
|
107
|
-
const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType).length;
|
|
219
|
+
const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames).length;
|
|
108
220
|
if (relatedMode === 'never') {
|
|
109
221
|
logger_1.logger.info('[TopicRecall] mode=never tag=' + topic + ' day=' + day + ' accepted=' + seedAccepted);
|
|
110
222
|
}
|
|
@@ -118,19 +230,19 @@ class TopicPipeline {
|
|
|
118
230
|
}
|
|
119
231
|
const dedupedCount = byId.size;
|
|
120
232
|
logger_1.logger.info('[TopicCollector] type=' + contentType + ' raw=' + rawCount + ' deduplicated=' + dedupedCount + ' aiExcluded=' + aiExcludedCount + ' searchedTags=' + searchedTags.length);
|
|
121
|
-
const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType);
|
|
122
|
-
const chosen = this.topByPopularity(accepted, limit,
|
|
233
|
+
const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames);
|
|
234
|
+
const chosen = this.topByPopularity(accepted, limit, hardSeedTier ? seedKey : undefined);
|
|
123
235
|
const selected = chosen.map((e) => e.candidate);
|
|
124
236
|
logger_1.logger.info('[MetadataTopicFilter] accepted=' + accepted.length);
|
|
125
237
|
if (selected[0]) {
|
|
126
|
-
logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + ' title=' + selected[0].title);
|
|
238
|
+
logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + (hardSeedTier ? ' seedTier=on' : '') + ' title=' + selected[0].title);
|
|
127
239
|
}
|
|
128
240
|
return {
|
|
129
241
|
works: chosen.map((e) => e.work),
|
|
130
242
|
selection: {
|
|
131
243
|
candidates: [...byId.values()].map((e) => e.candidate),
|
|
132
244
|
selected,
|
|
133
|
-
resolvedTagCount:
|
|
245
|
+
resolvedTagCount: walked.length,
|
|
134
246
|
searchedTags,
|
|
135
247
|
rawCount,
|
|
136
248
|
dedupedCount,
|
|
@@ -163,12 +275,16 @@ class TopicPipeline {
|
|
|
163
275
|
}
|
|
164
276
|
toCandidate(work, type) {
|
|
165
277
|
const popularity = (0, pixiv_utils_1.calculatePopularityScore)(work);
|
|
278
|
+
const translatedTags = (work.tags ?? [])
|
|
279
|
+
.map((t) => t.translated_name)
|
|
280
|
+
.filter((name) => Boolean(name));
|
|
166
281
|
return {
|
|
167
282
|
id: work.id,
|
|
168
283
|
type,
|
|
169
284
|
title: work.title ?? '',
|
|
170
285
|
caption: work.caption ?? '',
|
|
171
286
|
tags: (work.tags ?? []).map((t) => t.name).filter(Boolean),
|
|
287
|
+
...(translatedTags.length > 0 ? { translatedTags } : {}),
|
|
172
288
|
bookmarks: Number(work.total_bookmarks ?? work.bookmark_count ?? 0) || 0,
|
|
173
289
|
views: Number(work.total_view ?? work.view_count ?? 0) || 0,
|
|
174
290
|
popularity,
|
|
@@ -184,16 +300,23 @@ class TopicPipeline {
|
|
|
184
300
|
* the seed pass can be evaluated before deciding whether the related channel
|
|
185
301
|
* is needed at all (§topic-recall).
|
|
186
302
|
*/
|
|
187
|
-
acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType) {
|
|
303
|
+
acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames) {
|
|
188
304
|
const accepted = [];
|
|
305
|
+
const hitsSeed = (candidate) => {
|
|
306
|
+
if (candidate.tags.some((t) => this.key(t) === seedKey))
|
|
307
|
+
return true;
|
|
308
|
+
if (!matchTranslatedNames)
|
|
309
|
+
return false;
|
|
310
|
+
return (candidate.translatedTags ?? []).some((t) => this.key(t) === seedKey);
|
|
311
|
+
};
|
|
189
312
|
for (const entry of byId.values()) {
|
|
190
|
-
entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores);
|
|
313
|
+
entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores, matchTranslatedNames);
|
|
191
314
|
if (entry.candidate.metadataScore >= minMetadataScore)
|
|
192
315
|
accepted.push(entry);
|
|
193
316
|
}
|
|
194
317
|
if (accepted.length === 0 && byId.size > 0) {
|
|
195
318
|
const fallback = [...byId.values()]
|
|
196
|
-
.filter((e) => e.candidate
|
|
319
|
+
.filter((e) => hitsSeed(e.candidate))
|
|
197
320
|
.sort((a, b) => b.candidate.popularity - a.candidate.popularity);
|
|
198
321
|
accepted.push(...fallback.slice(0, Math.max(limit, 1)));
|
|
199
322
|
logger_1.logger.warn('[MetadataTopicFilter] type=' + contentType + ' none above threshold ' + minMetadataScore + '; kept ' + accepted.length + ' seed-tag fallback');
|
|
@@ -204,25 +327,49 @@ class TopicPipeline {
|
|
|
204
327
|
* Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
|
|
205
328
|
* title/caption add smaller boosts. The seed tag is strong evidence.
|
|
206
329
|
* No text model — case/symbol-insensitive substring matching only.
|
|
330
|
+
*
|
|
331
|
+
* `matchTranslatedNames` (default false) additionally counts a work's
|
|
332
|
+
* `translated_name` as a hit for the resolved tag it translates to, so a
|
|
333
|
+
* topic written in one language can still match works tagged in another. The
|
|
334
|
+
* translated name is matched against the resolved space — never added to
|
|
335
|
+
* `candidate.tags` — so the work is not reported as carrying a tag it lacks.
|
|
336
|
+
*
|
|
337
|
+
* A resolved tag counts at most ONCE per work, whether it matched through the
|
|
338
|
+
* tag name or through its translation (and a work that repeats a tag, as Pixiv
|
|
339
|
+
* payloads do for a tag plus its translation, does not double its weight).
|
|
207
340
|
*/
|
|
208
|
-
metadataScore(candidate, seedKey, tagScores) {
|
|
341
|
+
metadataScore(candidate, seedKey, tagScores, matchTranslatedNames) {
|
|
209
342
|
let seedHit = false;
|
|
210
343
|
let relatedSum = 0;
|
|
211
344
|
let relatedHits = 0;
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
345
|
+
let strongRelated = false;
|
|
346
|
+
// A resolved tag is matched through its own name, and — when the caller
|
|
347
|
+
// opted in — through the work-side translation of that tag. Either spelling
|
|
348
|
+
// maps back to the SAME resolved tag, so a work carrying both spellings
|
|
349
|
+
// counts once and never gets a double weight.
|
|
350
|
+
const countedTagKeys = new Set();
|
|
351
|
+
const consider = (candidateKey) => {
|
|
352
|
+
if (candidateKey === '' || STOP_TAGS.has(candidateKey))
|
|
353
|
+
return;
|
|
354
|
+
if (candidateKey === seedKey) {
|
|
217
355
|
seedHit = true;
|
|
218
|
-
|
|
219
|
-
}
|
|
220
|
-
const related = tagScores.get(k);
|
|
221
|
-
if (related !== undefined) {
|
|
222
|
-
relatedSum += Math.min(related, 0.6);
|
|
223
|
-
relatedHits += 1;
|
|
356
|
+
return;
|
|
224
357
|
}
|
|
358
|
+
const aliases = tagScores.get(candidateKey);
|
|
359
|
+
if (!aliases || countedTagKeys.has(aliases.primary))
|
|
360
|
+
return;
|
|
361
|
+
countedTagKeys.add(aliases.primary);
|
|
362
|
+
relatedSum += Math.min(aliases.weight, 0.6);
|
|
363
|
+
relatedHits += 1;
|
|
364
|
+
if (aliases.weight >= 0.6)
|
|
365
|
+
strongRelated = true;
|
|
366
|
+
};
|
|
367
|
+
if (matchTranslatedNames) {
|
|
368
|
+
for (const tag of candidate.translatedTags ?? [])
|
|
369
|
+
consider(this.key(tag));
|
|
225
370
|
}
|
|
371
|
+
for (const tag of candidate.tags)
|
|
372
|
+
consider(this.key(tag));
|
|
226
373
|
const hayTitle = this.normalize(candidate.title);
|
|
227
374
|
const hayCaption = this.normalize(candidate.caption);
|
|
228
375
|
const titleSeed = !!seedKey && hayTitle.includes(seedKey);
|
|
@@ -239,7 +386,6 @@ class TopicPipeline {
|
|
|
239
386
|
// deliberately does NOT clear the bar, so a hugely popular tangential work
|
|
240
387
|
// cannot crowd out core topic works.
|
|
241
388
|
let score = 0;
|
|
242
|
-
const strongRelated = [...tagScores.entries()].some(([k, w]) => w >= 0.6 && candidate.tags.some((t) => this.key(t) === k));
|
|
243
389
|
if (titleSeed)
|
|
244
390
|
score += 0.8;
|
|
245
391
|
if (captionSeed)
|
|
@@ -256,12 +402,13 @@ class TopicPipeline {
|
|
|
256
402
|
* alone. A work with a higher metadata score does NOT outrank a more popular
|
|
257
403
|
* accepted work.
|
|
258
404
|
*
|
|
259
|
-
* `seedKey` adds a single tier in front of that popularity order and is
|
|
260
|
-
* passed by the seed-first recall modes (§topic-recall)
|
|
261
|
-
*
|
|
262
|
-
*
|
|
263
|
-
*
|
|
264
|
-
* is
|
|
405
|
+
* `seedKey` adds a single tier in front of that popularity order and is
|
|
406
|
+
* passed by the seed-first recall modes (§topic-recall) and by
|
|
407
|
+
* `topicDiscovery.seedTier: 'on'` (§tag-provenance): a work that actually
|
|
408
|
+
* carries the topic tag outranks a related-only work, however popular the
|
|
409
|
+
* latter is, and popularity decides within each tier. With `seedTier: 'off'`
|
|
410
|
+
* and the default `relatedTags: 'always'` no `seedKey` is passed, so the
|
|
411
|
+
* documented popularity-only ranking is unchanged.
|
|
265
412
|
*/
|
|
266
413
|
popCompare(a, b) {
|
|
267
414
|
return b.popularity - a.popularity;
|
|
@@ -97,6 +97,9 @@ class TopicResolver {
|
|
|
97
97
|
name: seed,
|
|
98
98
|
translatedName: suggested.find((t) => this.scorer.key(t.name) === this.scorer.key(seed))?.translated_name,
|
|
99
99
|
score: 1,
|
|
100
|
+
weight: 1,
|
|
101
|
+
// The tag the operator asked for: the strongest provenance, never a hint.
|
|
102
|
+
source: 'seed',
|
|
100
103
|
occurrences: topicWorks.length,
|
|
101
104
|
coverage: 1,
|
|
102
105
|
specificity: 1,
|
|
@@ -168,7 +171,7 @@ class TopicResolver {
|
|
|
168
171
|
expiresAt: new Date(now + cacheDays * 24 * 60 * 60_000).toISOString(),
|
|
169
172
|
sampleSize: 0,
|
|
170
173
|
sampledWorks: 0,
|
|
171
|
-
tags: [{ name: seed, score: 1, occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
|
|
174
|
+
tags: [{ name: seed, score: 1, weight: 1, source: 'seed', occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
|
|
172
175
|
};
|
|
173
176
|
}
|
|
174
177
|
ageDays(space) {
|
|
@@ -99,10 +99,17 @@ class TopicTagScorer {
|
|
|
99
99
|
const suggestionWeight = stat.suggested ? 1.1 : 1.0;
|
|
100
100
|
const genericPenalty = GENERIC_TAG_PENALTY.has(k) ? 0.4 : 1.0;
|
|
101
101
|
const raw = recall * specificity * suggestionWeight * genericPenalty;
|
|
102
|
+
const score = Number(raw.toFixed(4));
|
|
102
103
|
resolved.push({
|
|
103
104
|
name: stat.name,
|
|
104
105
|
translatedName: stat.translatedName,
|
|
105
|
-
score
|
|
106
|
+
score,
|
|
107
|
+
// `weight` is the same semantic number; ranking and diagnostics read it
|
|
108
|
+
// while `score` stays for backward compatibility (§tag-provenance).
|
|
109
|
+
weight: score,
|
|
110
|
+
// Both channels may agree on a tag; the combined provenance records
|
|
111
|
+
// that, which is strictly more informative than either alone.
|
|
112
|
+
source: stat.suggested ? 'cooccurrence+autocomplete' : 'cooccurrence',
|
|
106
113
|
occurrences: stat.topicDocs,
|
|
107
114
|
coverage: Number(coverage.toFixed(4)),
|
|
108
115
|
specificity: Number(specificity.toFixed(4)),
|
|
@@ -127,6 +134,10 @@ class TopicTagScorer {
|
|
|
127
134
|
name: name,
|
|
128
135
|
translatedName: sug.translated_name?.trim() || undefined,
|
|
129
136
|
score: AUTOCOMPLETE_ONLY_SCORE,
|
|
137
|
+
weight: AUTOCOMPLETE_ONLY_SCORE,
|
|
138
|
+
// Pixiv autocomplete is the only evidence for this tag: it never
|
|
139
|
+
// co-occurred in the bounded sample.
|
|
140
|
+
source: 'autocomplete',
|
|
130
141
|
occurrences: 0,
|
|
131
142
|
coverage: 0,
|
|
132
143
|
specificity: 1.0, // Pixiv-endorsed related; treated as specific but unobserved
|
package/dist/topic/types.d.ts
CHANGED
|
@@ -4,12 +4,34 @@
|
|
|
4
4
|
* local models are used anywhere in this module.
|
|
5
5
|
*/
|
|
6
6
|
export type TopicContentType = 'illustration' | 'novel';
|
|
7
|
+
/**
|
|
8
|
+
* Where a resolved tag came from. Provenance is what lets a caller treat a weak
|
|
9
|
+
* expansion differently from the topic it was asked for (§tag-provenance):
|
|
10
|
+
*
|
|
11
|
+
* - `'seed'`: the tag the operator asked for. Always the strongest key.
|
|
12
|
+
* - `'cooccurrence'`: sampled together with the seed, but Pixiv autocomplete
|
|
13
|
+
* does not relate it to the seed. Co-occurrence evidence only.
|
|
14
|
+
* - `'autocomplete'`: Pixiv autocomplete relates it to the seed, but it never
|
|
15
|
+
* appeared in the sample. No co-occurrence evidence.
|
|
16
|
+
* - `'cooccurrence+autocomplete'`: both channels agree — the strongest related
|
|
17
|
+
* provenance available.
|
|
18
|
+
*/
|
|
19
|
+
export type TagSource = 'seed' | 'cooccurrence' | 'autocomplete' | 'cooccurrence+autocomplete';
|
|
7
20
|
/** A single related tag with a 0..1 relatedness score and provenance. */
|
|
8
21
|
export interface ResolvedTag {
|
|
9
22
|
name: string;
|
|
10
23
|
translatedName?: string;
|
|
11
24
|
/** Combined relatedness score (co-occurrence * specificity * suggestion). */
|
|
12
25
|
score: number;
|
|
26
|
+
/**
|
|
27
|
+
* Semantic weight used for ranking, filtering and diagnostics. Always the
|
|
28
|
+
* same number as `score`; kept as a separate, documented field so ranking can
|
|
29
|
+
* be explained (and, later, adjusted) without redefining `score`.
|
|
30
|
+
* Optional: spaces persisted before provenance existed have neither field.
|
|
31
|
+
*/
|
|
32
|
+
weight?: number;
|
|
33
|
+
/** Provenance of the tag. Optional for the same reason as `weight`. */
|
|
34
|
+
source?: TagSource;
|
|
13
35
|
/** How many sampled works (of the seed search) carried this tag. */
|
|
14
36
|
occurrences: number;
|
|
15
37
|
/** Coverage of the sampled seed works (occurrences / sample size). */
|
|
@@ -57,6 +79,26 @@ export interface TopicDiscoveryOptions {
|
|
|
57
79
|
* `'never'` searches the seed tag alone.
|
|
58
80
|
*/
|
|
59
81
|
relatedTags?: RelatedTagMode;
|
|
82
|
+
/**
|
|
83
|
+
* Which resolved tags may become recall channels (§tag-provenance). Deny wins
|
|
84
|
+
* over allow; the seed tag is never dropped by `allow`/`allowSources`.
|
|
85
|
+
*/
|
|
86
|
+
tagRelations?: TopicRelationsOptions;
|
|
87
|
+
/** Make the seed tag a hard ranking tier (default `'off'`). */
|
|
88
|
+
seedTier?: 'off' | 'on';
|
|
89
|
+
/** Count a work's translated tag names as tag hits (default false). */
|
|
90
|
+
matchTranslatedNames?: boolean;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Runtime shape of `TopicDiscoveryConfig.tagRelations`, declared here so the
|
|
94
|
+
* topic module does not depend on the config layer (the topic pipeline is also
|
|
95
|
+
* driven by hand-written targets in tests and by callers that never load a
|
|
96
|
+
* config file).
|
|
97
|
+
*/
|
|
98
|
+
export interface TopicRelationsOptions {
|
|
99
|
+
allowSources?: TagSource[];
|
|
100
|
+
allow?: string[];
|
|
101
|
+
deny?: string[];
|
|
60
102
|
}
|
|
61
103
|
export interface TopicCollectOptions {
|
|
62
104
|
maxPerTag?: number;
|
|
@@ -76,6 +118,12 @@ export interface TopicCandidate {
|
|
|
76
118
|
popularity: number;
|
|
77
119
|
/** Metadata topic-relevance score computed by the filter stage. */
|
|
78
120
|
metadataScore: number;
|
|
121
|
+
/**
|
|
122
|
+
* Translated tag names carried by the work, kept beside `tags` so
|
|
123
|
+
* `matchTranslatedNames` can compare them WITHOUT claiming the work itself
|
|
124
|
+
* carries a tag it does not (§tag-provenance). Default-off.
|
|
125
|
+
*/
|
|
126
|
+
translatedTags?: string[];
|
|
79
127
|
/** Pixiv AI classification copied from illustration search metadata. */
|
|
80
128
|
aiType?: number;
|
|
81
129
|
}
|
package/dist/version.js
CHANGED
|
@@ -2,5 +2,5 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.BUILD = void 0;
|
|
4
4
|
// GENERATED by scripts/write-version.js — do not edit manually.
|
|
5
|
-
exports.BUILD = { version: '3.0
|
|
5
|
+
exports.BUILD = { version: '3.1.0', commit: '583a74c98ef7' };
|
|
6
6
|
//# sourceMappingURL=version.js.map
|
package/dist/webui/package.json
CHANGED
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pixivflow",
|
|
3
|
-
"version": "3.0
|
|
3
|
+
"version": "3.1.0",
|
|
4
4
|
"description": "🎨 Pixiv 下载、筛选与自动收集工具 - 批量下载插画和小说、按标签/热度/日期筛选、定时任务与可靠 HTTP 交付 | Pixiv downloader and automation toolkit with filtering, scheduling and reliable HTTP delivery",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|