pixivflow 3.0.2 → 3.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/types.d.ts +11 -1
- package/dist/config/validation.js +3 -0
- package/dist/download/NovelDownloader.d.ts +11 -0
- package/dist/download/NovelDownloader.js +36 -4
- package/dist/download/novelCover.d.ts +30 -5
- package/dist/download/novelCover.js +37 -5
- package/dist/package.json +1 -1
- package/dist/topic/TopicPipeline.d.ts +23 -0
- package/dist/topic/TopicPipeline.js +92 -24
- package/dist/topic/types.d.ts +14 -0
- package/dist/utils/imageDimensions.d.ts +19 -0
- package/dist/utils/imageDimensions.js +95 -0
- package/dist/version.js +1 -1
- package/dist/webui/package.json +1 -1
- package/package.json +1 -1
package/dist/config/types.d.ts
CHANGED
|
@@ -12,12 +12,22 @@ export interface TopicDiscoveryConfig {
|
|
|
12
12
|
sampleWorks?: number;
|
|
13
13
|
/** Cache lifetime in days for the resolved tag space (default 7). */
|
|
14
14
|
cacheDays?: number;
|
|
15
|
-
/** Minimum relatedness score for a tag to enter the space (default 0.
|
|
15
|
+
/** Minimum relatedness score for a tag to enter the space (default 0.22). */
|
|
16
16
|
minScore?: number;
|
|
17
17
|
/** Ignore a fresh cache and re-discover now (default false). */
|
|
18
18
|
refresh?: boolean;
|
|
19
19
|
/** Include R-18 works in topic sampling and collection (default false). */
|
|
20
20
|
includeR18?: boolean;
|
|
21
|
+
/**
|
|
22
|
+
* When related tags may be searched as their own recall channel (§topic-recall).
|
|
23
|
+
*
|
|
24
|
+
* - `'always'` (default): walk the whole resolved tag space every day.
|
|
25
|
+
* - `'when_seed_insufficient'`: search the topic tag first and only fall back
|
|
26
|
+
* to related tags when it cannot fill `limit` for that day — a related tag
|
|
27
|
+
* is a hint, never a substitute for the topic the operator asked for.
|
|
28
|
+
* - `'never'`: search the topic tag alone.
|
|
29
|
+
*/
|
|
30
|
+
relatedTags?: 'always' | 'when_seed_insufficient' | 'never';
|
|
21
31
|
}
|
|
22
32
|
/** Candidate collection tuning for mode='topic'. */
|
|
23
33
|
export interface CandidateCollectionConfig {
|
|
@@ -191,6 +191,9 @@ function validateConfig(config, location, databasePath) {
|
|
|
191
191
|
if (td.includeR18 !== undefined && typeof td.includeR18 !== 'boolean') {
|
|
192
192
|
errors.push(`targets[${index}].topicDiscovery.includeR18: Must be a boolean (got ${typeof td.includeR18})`);
|
|
193
193
|
}
|
|
194
|
+
if (td.relatedTags !== undefined && !['always', 'when_seed_insufficient', 'never'].includes(td.relatedTags)) {
|
|
195
|
+
errors.push(`targets[${index}].topicDiscovery.relatedTags: Must be "always", "when_seed_insufficient" or "never" (got ${String(td.relatedTags)})`);
|
|
196
|
+
}
|
|
194
197
|
}
|
|
195
198
|
const cc = target.candidateCollection;
|
|
196
199
|
if (cc) {
|
|
@@ -16,5 +16,16 @@ export declare class NovelDownloader {
|
|
|
16
16
|
private readonly materializationPolicy;
|
|
17
17
|
constructor(client: IPixivClient, database: IDatabase, fileService: IFileService, metadataDb?: Database | undefined, materializer?: MediaMaterializer, materializationPolicy?: MaterializationPolicy);
|
|
18
18
|
download(novel: PixivNovel, tag: string, target: TargetConfig): Promise<DownloadedArtifact | undefined>;
|
|
19
|
+
/**
|
|
20
|
+
* Resolves the cover URL a novel should ship with (§novel-cover).
|
|
21
|
+
*
|
|
22
|
+
* Pixiv's "author set no cover" case is invisible in the API: the design it
|
|
23
|
+
* renders (title typeset on a template) is served from the same CDN path with
|
|
24
|
+
* a unique hash as a real cover, so the URL cannot decide. The candidate cover
|
|
25
|
+
* is therefore fetched once and its header inspected; Pixiv's design canvas is
|
|
26
|
+
* exactly 640x900. Anything that cannot be classified keeps the cover — a
|
|
27
|
+
* failed probe must never cost a real cover.
|
|
28
|
+
*/
|
|
29
|
+
private resolveCoverUrl;
|
|
19
30
|
}
|
|
20
31
|
//# sourceMappingURL=NovelDownloader.d.ts.map
|
|
@@ -226,10 +226,11 @@ class NovelDownloader {
|
|
|
226
226
|
}
|
|
227
227
|
}
|
|
228
228
|
// Novel cover (§novel-cover): normalized to null for Pixiv's default
|
|
229
|
-
// placeholder
|
|
230
|
-
//
|
|
231
|
-
// cover
|
|
232
|
-
|
|
229
|
+
// placeholder AND for Pixiv's generated design covers, carried as a
|
|
230
|
+
// dedicated `novelcover` MediaAsset so TelePost can build
|
|
231
|
+
// `cover root + TXT reply` without positional guessing. The cover is never
|
|
232
|
+
// materialized locally — Telegram fetches it (via proxy).
|
|
233
|
+
const coverUrl = await this.resolveCoverUrl(detail.id, typeof textResponse === 'string' ? undefined : textResponse.coverUrl);
|
|
233
234
|
const coverAsset = coverUrl ? (0, novelCover_1.novelCoverAsset)(String(detail.id), coverUrl) : undefined;
|
|
234
235
|
const downloadByAssetKey = new Map(assets.filter((a) => a.status === 'downloaded' && a.localPath)
|
|
235
236
|
.map((a) => [`${a.kind}:${a.sourceId}`, a]));
|
|
@@ -398,6 +399,37 @@ class NovelDownloader {
|
|
|
398
399
|
language: detectedLang ? `${detectedLang.name} (${detectedLang.code})` : undefined,
|
|
399
400
|
};
|
|
400
401
|
}
|
|
402
|
+
/**
|
|
403
|
+
* Resolves the cover URL a novel should ship with (§novel-cover).
|
|
404
|
+
*
|
|
405
|
+
* Pixiv's "author set no cover" case is invisible in the API: the design it
|
|
406
|
+
* renders (title typeset on a template) is served from the same CDN path with
|
|
407
|
+
* a unique hash as a real cover, so the URL cannot decide. The candidate cover
|
|
408
|
+
* is therefore fetched once and its header inspected; Pixiv's design canvas is
|
|
409
|
+
* exactly 640x900. Anything that cannot be classified keeps the cover — a
|
|
410
|
+
* failed probe must never cost a real cover.
|
|
411
|
+
*/
|
|
412
|
+
async resolveCoverUrl(novelId, coverUrl) {
|
|
413
|
+
const normalized = (0, novelCover_1.normalizeNovelCoverUrl)(coverUrl);
|
|
414
|
+
if (!normalized)
|
|
415
|
+
return null;
|
|
416
|
+
try {
|
|
417
|
+
const cover = await this.client.downloadImage(normalized);
|
|
418
|
+
if ((0, novelCover_1.isPixivDesignCoverImage)(cover)) {
|
|
419
|
+
logger_1.logger.info(`Novel ${novelId} cover is a Pixiv design cover (${novelCover_1.PIXIV_DESIGN_COVER_WIDTH}x${novelCover_1.PIXIV_DESIGN_COVER_HEIGHT}); delivering without a cover`, { novelId, coverUrl: normalized, designCover: true });
|
|
420
|
+
return null;
|
|
421
|
+
}
|
|
422
|
+
return normalized;
|
|
423
|
+
}
|
|
424
|
+
catch (error) {
|
|
425
|
+
logger_1.logger.warn(`Novel ${novelId} cover probe failed; keeping the cover`, {
|
|
426
|
+
novelId,
|
|
427
|
+
coverUrl: normalized,
|
|
428
|
+
reason: error instanceof Error ? error.message : String(error),
|
|
429
|
+
});
|
|
430
|
+
return normalized;
|
|
431
|
+
}
|
|
432
|
+
}
|
|
401
433
|
}
|
|
402
434
|
exports.NovelDownloader = NovelDownloader;
|
|
403
435
|
function buildMediaAssetById(mediaAsset, artifactIdValue) {
|
|
@@ -2,12 +2,23 @@
|
|
|
2
2
|
* Novel cover normalization (§novel-cover).
|
|
3
3
|
*
|
|
4
4
|
* Pixiv exposes the novel cover as `coverUrl` on the webview v2 text response.
|
|
5
|
-
* Novels WITHOUT a custom cover return the stock
|
|
6
|
-
* placeholder
|
|
7
|
-
*
|
|
5
|
+
* Novels WITHOUT a custom cover used to return the stock
|
|
6
|
+
* "novel-cover-master-default" placeholder; today Pixiv instead RENDERS a
|
|
7
|
+
* per-novel design (floral / seasonal / genre template with the novel title
|
|
8
|
+
* typeset) and serves it from `novel-cover-master/img/...` with a unique hash,
|
|
9
|
+
* so it is URL-indistinguishable from an author cover. Those design covers are
|
|
10
|
+
* NOT real covers and must never ship as Telegram media, and no API field
|
|
11
|
+
* (app-api `novel/detail`, webview v2) marks them: the one reliable
|
|
12
|
+
* discriminator is the canvas Pixiv renders them on, always exactly 640x900
|
|
13
|
+
* (`PIXIV_DESIGN_COVER_*` below). The normalized model is therefore
|
|
14
|
+
* `string | null`:
|
|
8
15
|
*
|
|
9
|
-
* real cover
|
|
10
|
-
* default / absent → null
|
|
16
|
+
* real cover → the Pixiv CDN URL (original size when derivable)
|
|
17
|
+
* default / design / absent → null
|
|
18
|
+
*
|
|
19
|
+
* Callers cannot tell a design cover from the URL alone, so the fetch-and-check
|
|
20
|
+
* step (`isPixivDesignCoverImage`) lives in the downloader; this module stays a
|
|
21
|
+
* pure URL/bytes helper.
|
|
11
22
|
*
|
|
12
23
|
* The cover rides the canonical MediaAsset contract with a dedicated
|
|
13
24
|
* `novelcover` pixivKind, so consumers (TelePost) can distinguish it from
|
|
@@ -15,6 +26,20 @@
|
|
|
15
26
|
* positional guessing.
|
|
16
27
|
*/
|
|
17
28
|
import { MediaAsset } from '../domain/media/MediaAsset';
|
|
29
|
+
/**
|
|
30
|
+
* The canvas Pixiv renders its built-in novel cover designs on. Every design
|
|
31
|
+
* observed in production (floral, seasonal sweets, treasure map, genre label)
|
|
32
|
+
* is delivered at exactly this size, while author covers keep their own
|
|
33
|
+
* dimensions (512x512, 768x768, 800x1200, 822x1200, 826x1169, 1024x1024 …).
|
|
34
|
+
*/
|
|
35
|
+
export declare const PIXIV_DESIGN_COVER_WIDTH = 640;
|
|
36
|
+
export declare const PIXIV_DESIGN_COVER_HEIGHT = 900;
|
|
37
|
+
/**
|
|
38
|
+
* True when the fetched cover bytes are one of Pixiv's generated designs.
|
|
39
|
+
* Unknown formats / unreadable payloads return false so callers fail open and
|
|
40
|
+
* keep the cover rather than dropping a real one.
|
|
41
|
+
*/
|
|
42
|
+
export declare function isPixivDesignCoverImage(cover: ArrayBuffer | Uint8Array | null | undefined): boolean;
|
|
18
43
|
export declare function normalizeNovelCoverUrl(coverUrl?: string | null): string | null;
|
|
19
44
|
export declare function novelCoverAsset(workId: string, coverUrl: string): MediaAsset;
|
|
20
45
|
//# sourceMappingURL=novelCover.d.ts.map
|
|
@@ -1,17 +1,30 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.PIXIV_DESIGN_COVER_HEIGHT = exports.PIXIV_DESIGN_COVER_WIDTH = void 0;
|
|
4
|
+
exports.isPixivDesignCoverImage = isPixivDesignCoverImage;
|
|
3
5
|
exports.normalizeNovelCoverUrl = normalizeNovelCoverUrl;
|
|
4
6
|
exports.novelCoverAsset = novelCoverAsset;
|
|
5
7
|
/**
|
|
6
8
|
* Novel cover normalization (§novel-cover).
|
|
7
9
|
*
|
|
8
10
|
* Pixiv exposes the novel cover as `coverUrl` on the webview v2 text response.
|
|
9
|
-
* Novels WITHOUT a custom cover return the stock
|
|
10
|
-
* placeholder
|
|
11
|
-
*
|
|
11
|
+
* Novels WITHOUT a custom cover used to return the stock
|
|
12
|
+
* "novel-cover-master-default" placeholder; today Pixiv instead RENDERS a
|
|
13
|
+
* per-novel design (floral / seasonal / genre template with the novel title
|
|
14
|
+
* typeset) and serves it from `novel-cover-master/img/...` with a unique hash,
|
|
15
|
+
* so it is URL-indistinguishable from an author cover. Those design covers are
|
|
16
|
+
* NOT real covers and must never ship as Telegram media, and no API field
|
|
17
|
+
* (app-api `novel/detail`, webview v2) marks them: the one reliable
|
|
18
|
+
* discriminator is the canvas Pixiv renders them on, always exactly 640x900
|
|
19
|
+
* (`PIXIV_DESIGN_COVER_*` below). The normalized model is therefore
|
|
20
|
+
* `string | null`:
|
|
12
21
|
*
|
|
13
|
-
* real cover
|
|
14
|
-
* default / absent → null
|
|
22
|
+
* real cover → the Pixiv CDN URL (original size when derivable)
|
|
23
|
+
* default / design / absent → null
|
|
24
|
+
*
|
|
25
|
+
* Callers cannot tell a design cover from the URL alone, so the fetch-and-check
|
|
26
|
+
* step (`isPixivDesignCoverImage`) lives in the downloader; this module stays a
|
|
27
|
+
* pure URL/bytes helper.
|
|
15
28
|
*
|
|
16
29
|
* The cover rides the canonical MediaAsset contract with a dedicated
|
|
17
30
|
* `novelcover` pixivKind, so consumers (TelePost) can distinguish it from
|
|
@@ -19,9 +32,28 @@ exports.novelCoverAsset = novelCoverAsset;
|
|
|
19
32
|
* positional guessing.
|
|
20
33
|
*/
|
|
21
34
|
const MediaAsset_1 = require("../domain/media/MediaAsset");
|
|
35
|
+
const imageDimensions_1 = require("../utils/imageDimensions");
|
|
22
36
|
const DEFAULT_COVER_PATTERN = /novel-cover-(master-)?default/i;
|
|
23
37
|
/** Strip the `/c/<spec>/` resizer segment to recover the original-size URL. */
|
|
24
38
|
const RESIZED_COVER_PATTERN = /\/c\/[^/]+\/(novel-cover-master\/)/;
|
|
39
|
+
/**
|
|
40
|
+
* The canvas Pixiv renders its built-in novel cover designs on. Every design
|
|
41
|
+
* observed in production (floral, seasonal sweets, treasure map, genre label)
|
|
42
|
+
* is delivered at exactly this size, while author covers keep their own
|
|
43
|
+
* dimensions (512x512, 768x768, 800x1200, 822x1200, 826x1169, 1024x1024 …).
|
|
44
|
+
*/
|
|
45
|
+
exports.PIXIV_DESIGN_COVER_WIDTH = 640;
|
|
46
|
+
exports.PIXIV_DESIGN_COVER_HEIGHT = 900;
|
|
47
|
+
/**
|
|
48
|
+
* True when the fetched cover bytes are one of Pixiv's generated designs.
|
|
49
|
+
* Unknown formats / unreadable payloads return false so callers fail open and
|
|
50
|
+
* keep the cover rather than dropping a real one.
|
|
51
|
+
*/
|
|
52
|
+
function isPixivDesignCoverImage(cover) {
|
|
53
|
+
const dimensions = (0, imageDimensions_1.readImageDimensions)(cover);
|
|
54
|
+
return (dimensions?.width === exports.PIXIV_DESIGN_COVER_WIDTH &&
|
|
55
|
+
dimensions?.height === exports.PIXIV_DESIGN_COVER_HEIGHT);
|
|
56
|
+
}
|
|
25
57
|
function normalizeNovelCoverUrl(coverUrl) {
|
|
26
58
|
const url = typeof coverUrl === 'string' ? coverUrl.trim() : '';
|
|
27
59
|
if (!url)
|
package/dist/package.json
CHANGED
|
@@ -5,6 +5,12 @@ export interface TopicSelection {
|
|
|
5
5
|
candidates: TopicCandidate[];
|
|
6
6
|
selected: TopicCandidate[];
|
|
7
7
|
resolvedTagCount: number;
|
|
8
|
+
/**
|
|
9
|
+
* Tags actually searched for the day. Under `relatedTags: 'always'` this is
|
|
10
|
+
* the whole resolved space; under the seed-first modes it is the seed tag and
|
|
11
|
+
* only the related tags that were really needed (§topic-recall).
|
|
12
|
+
*/
|
|
13
|
+
searchedTags?: string[];
|
|
8
14
|
rawCount: number;
|
|
9
15
|
dedupedCount: number;
|
|
10
16
|
acceptedCount: number;
|
|
@@ -44,6 +50,15 @@ export declare class TopicPipeline {
|
|
|
44
50
|
}>;
|
|
45
51
|
private searchDay;
|
|
46
52
|
private toCandidate;
|
|
53
|
+
/**
|
|
54
|
+
* Applies the metadata gate to everything collected so far. When nothing at
|
|
55
|
+
* all clears the threshold but the seed tag is present, the seed-tag works are
|
|
56
|
+
* kept anyway: a sparse day must stay usable instead of reporting "no
|
|
57
|
+
* candidates" for a topic that visibly has works. Extracted from selection so
|
|
58
|
+
* the seed pass can be evaluated before deciding whether the related channel
|
|
59
|
+
* is needed at all (§topic-recall).
|
|
60
|
+
*/
|
|
61
|
+
private acceptedWorks;
|
|
47
62
|
/**
|
|
48
63
|
* Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
|
|
49
64
|
* title/caption add smaller boosts. The seed tag is strong evidence.
|
|
@@ -56,8 +71,16 @@ export declare class TopicPipeline {
|
|
|
56
71
|
* as on-topic, and the choice between accepted works is decided by popularity
|
|
57
72
|
* alone. A work with a higher metadata score does NOT outrank a more popular
|
|
58
73
|
* accepted work.
|
|
74
|
+
*
|
|
75
|
+
* `seedKey` adds a single tier in front of that popularity order and is only
|
|
76
|
+
* passed by the seed-first recall modes (§topic-recall): when related tags
|
|
77
|
+
* were reached as a fallback, a work that actually carries the topic tag must
|
|
78
|
+
* outrank a related-only work, and popularity decides within each tier. The
|
|
79
|
+
* default mode passes no `seedKey`, so the documented popularity-only ranking
|
|
80
|
+
* is unchanged.
|
|
59
81
|
*/
|
|
60
82
|
private popCompare;
|
|
83
|
+
private rankCompare;
|
|
61
84
|
private topByPopularity;
|
|
62
85
|
private onDay;
|
|
63
86
|
private normalize;
|
|
@@ -47,18 +47,24 @@ class TopicPipeline {
|
|
|
47
47
|
const maxCandidates = this.bound(collect.maxCandidates, COLLECT_DEFAULTS.maxCandidates, 20, 500);
|
|
48
48
|
const minMetadataScore = collect.minMetadataScore ?? COLLECT_DEFAULTS.minMetadataScore;
|
|
49
49
|
const includeR18 = discovery.includeR18 === true;
|
|
50
|
+
// An unknown mode (hand-written config bypassing validation) falls back to
|
|
51
|
+
// the historical behaviour rather than silently narrowing recall.
|
|
52
|
+
const requestedMode = discovery.relatedTags;
|
|
53
|
+
const relatedMode = requestedMode === 'when_seed_insufficient' || requestedMode === 'never' ? requestedMode : 'always';
|
|
50
54
|
const byId = new Map();
|
|
51
55
|
let rawCount = 0;
|
|
52
56
|
let aiExcludedCount = 0;
|
|
53
57
|
let duplicateRemovedCount = 0;
|
|
54
58
|
const tagNames = space.tags.map((t) => t.name);
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
59
|
+
const seedKey = this.key(topic);
|
|
60
|
+
const seedTags = tagNames.filter((name) => this.key(name) === seedKey);
|
|
61
|
+
const relatedTags = tagNames.filter((name) => this.key(name) !== seedKey);
|
|
62
|
+
const searchedTags = [];
|
|
63
|
+
const collectTag = async (tag) => {
|
|
58
64
|
// Cancellation is checked between tags, so a cancelled run stops issuing
|
|
59
65
|
// new searches even when the aborted request itself had already returned.
|
|
60
66
|
(0, errors_1.throwIfAborted)(this.signal, 'topic collection cancelled');
|
|
61
|
-
|
|
67
|
+
searchedTags.push(tag);
|
|
62
68
|
const works = await this.searchDay(contentType, tag, day, maxPerTag, includeR18);
|
|
63
69
|
rawCount += works.length;
|
|
64
70
|
for (const work of works) {
|
|
@@ -75,26 +81,45 @@ class TopicPipeline {
|
|
|
75
81
|
break;
|
|
76
82
|
}
|
|
77
83
|
logger_1.logger.debug('[TopicCollector] type=' + contentType + ' tag=' + tag + ' day=' + day + ' fetched=' + works.length + ' pool=' + byId.size);
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
84
|
+
};
|
|
85
|
+
const runTags = async (names) => {
|
|
86
|
+
for (let i = 0; i < names.length; i++) {
|
|
87
|
+
if (byId.size >= maxCandidates)
|
|
88
|
+
break;
|
|
89
|
+
await collectTag(names[i]);
|
|
90
|
+
if (i < names.length - 1 && this.requestDelayMs > 0)
|
|
91
|
+
await (0, promises_1.setTimeout)(this.requestDelayMs);
|
|
92
|
+
}
|
|
93
|
+
};
|
|
94
|
+
// §topic-recall: a resolved tag space is a hierarchy, not a bag of
|
|
95
|
+
// interchangeable tags. 'always' keeps the historical behaviour — every
|
|
96
|
+
// resolved tag is a recall channel for the day. The seed-first modes search
|
|
97
|
+
// the topic tag the operator actually asked for and only walk the related
|
|
98
|
+
// channel when that cannot fill the target, so a second high-weight tag
|
|
99
|
+
// (丸吞) cannot take the only slot of a 西瓜肚 target.
|
|
100
|
+
if (seedTags.length === 0 || relatedMode === 'always') {
|
|
101
|
+
// No seed tag in the space (hand-written space): keep walking everything
|
|
102
|
+
// rather than returning nothing.
|
|
103
|
+
await runTags(seedTags.length === 0 ? tagNames : [...seedTags, ...relatedTags]);
|
|
89
104
|
}
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
105
|
+
else {
|
|
106
|
+
await runTags(seedTags);
|
|
107
|
+
const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType).length;
|
|
108
|
+
if (relatedMode === 'never') {
|
|
109
|
+
logger_1.logger.info('[TopicRecall] mode=never tag=' + topic + ' day=' + day + ' accepted=' + seedAccepted);
|
|
110
|
+
}
|
|
111
|
+
else if (seedAccepted < limit) {
|
|
112
|
+
logger_1.logger.info('[TopicRecall] mode=when_seed_insufficient tag=' + topic + ' seedAccepted=' + seedAccepted + '/' + limit + ' relatedTags=' + relatedTags.length + '; expanding');
|
|
113
|
+
await runTags(relatedTags);
|
|
114
|
+
}
|
|
115
|
+
else {
|
|
116
|
+
logger_1.logger.info('[TopicRecall] mode=when_seed_insufficient tag=' + topic + ' seedAccepted=' + seedAccepted + '/' + limit + '; related tags not searched');
|
|
117
|
+
}
|
|
96
118
|
}
|
|
97
|
-
const
|
|
119
|
+
const dedupedCount = byId.size;
|
|
120
|
+
logger_1.logger.info('[TopicCollector] type=' + contentType + ' raw=' + rawCount + ' deduplicated=' + dedupedCount + ' aiExcluded=' + aiExcludedCount + ' searchedTags=' + searchedTags.length);
|
|
121
|
+
const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType);
|
|
122
|
+
const chosen = this.topByPopularity(accepted, limit, relatedMode === 'always' ? undefined : seedKey);
|
|
98
123
|
const selected = chosen.map((e) => e.candidate);
|
|
99
124
|
logger_1.logger.info('[MetadataTopicFilter] accepted=' + accepted.length);
|
|
100
125
|
if (selected[0]) {
|
|
@@ -106,6 +131,7 @@ class TopicPipeline {
|
|
|
106
131
|
candidates: [...byId.values()].map((e) => e.candidate),
|
|
107
132
|
selected,
|
|
108
133
|
resolvedTagCount: space.tags.length,
|
|
134
|
+
searchedTags,
|
|
109
135
|
rawCount,
|
|
110
136
|
dedupedCount,
|
|
111
137
|
acceptedCount: accepted.length,
|
|
@@ -150,6 +176,30 @@ class TopicPipeline {
|
|
|
150
176
|
...(work.illust_ai_type !== undefined ? { aiType: work.illust_ai_type } : {}),
|
|
151
177
|
};
|
|
152
178
|
}
|
|
179
|
+
/**
|
|
180
|
+
* Applies the metadata gate to everything collected so far. When nothing at
|
|
181
|
+
* all clears the threshold but the seed tag is present, the seed-tag works are
|
|
182
|
+
* kept anyway: a sparse day must stay usable instead of reporting "no
|
|
183
|
+
* candidates" for a topic that visibly has works. Extracted from selection so
|
|
184
|
+
* the seed pass can be evaluated before deciding whether the related channel
|
|
185
|
+
* is needed at all (§topic-recall).
|
|
186
|
+
*/
|
|
187
|
+
acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType) {
|
|
188
|
+
const accepted = [];
|
|
189
|
+
for (const entry of byId.values()) {
|
|
190
|
+
entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores);
|
|
191
|
+
if (entry.candidate.metadataScore >= minMetadataScore)
|
|
192
|
+
accepted.push(entry);
|
|
193
|
+
}
|
|
194
|
+
if (accepted.length === 0 && byId.size > 0) {
|
|
195
|
+
const fallback = [...byId.values()]
|
|
196
|
+
.filter((e) => e.candidate.tags.some((t) => this.key(t) === seedKey))
|
|
197
|
+
.sort((a, b) => b.candidate.popularity - a.candidate.popularity);
|
|
198
|
+
accepted.push(...fallback.slice(0, Math.max(limit, 1)));
|
|
199
|
+
logger_1.logger.warn('[MetadataTopicFilter] type=' + contentType + ' none above threshold ' + minMetadataScore + '; kept ' + accepted.length + ' seed-tag fallback');
|
|
200
|
+
}
|
|
201
|
+
return accepted;
|
|
202
|
+
}
|
|
153
203
|
/**
|
|
154
204
|
* Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
|
|
155
205
|
* title/caption add smaller boosts. The seed tag is strong evidence.
|
|
@@ -205,13 +255,31 @@ class TopicPipeline {
|
|
|
205
255
|
* as on-topic, and the choice between accepted works is decided by popularity
|
|
206
256
|
* alone. A work with a higher metadata score does NOT outrank a more popular
|
|
207
257
|
* accepted work.
|
|
258
|
+
*
|
|
259
|
+
* `seedKey` adds a single tier in front of that popularity order and is only
|
|
260
|
+
* passed by the seed-first recall modes (§topic-recall): when related tags
|
|
261
|
+
* were reached as a fallback, a work that actually carries the topic tag must
|
|
262
|
+
* outrank a related-only work, and popularity decides within each tier. The
|
|
263
|
+
* default mode passes no `seedKey`, so the documented popularity-only ranking
|
|
264
|
+
* is unchanged.
|
|
208
265
|
*/
|
|
209
266
|
popCompare(a, b) {
|
|
210
267
|
return b.popularity - a.popularity;
|
|
211
268
|
}
|
|
212
|
-
|
|
269
|
+
rankCompare(seedKey) {
|
|
270
|
+
if (!seedKey)
|
|
271
|
+
return (a, b) => this.popCompare(a, b);
|
|
272
|
+
const tier = (c) => (c.tags.some((t) => this.key(t) === seedKey) ? 0 : 1);
|
|
273
|
+
return (a, b) => {
|
|
274
|
+
const diff = tier(a) - tier(b);
|
|
275
|
+
return diff !== 0 ? diff : this.popCompare(a, b);
|
|
276
|
+
};
|
|
277
|
+
}
|
|
278
|
+
topByPopularity(items, limit, seedKey) {
|
|
213
279
|
if (items.length <= limit)
|
|
214
|
-
return items.sort((a, b) => this.
|
|
280
|
+
return items.sort((a, b) => this.rankCompare(seedKey)(a.candidate, b.candidate));
|
|
281
|
+
if (seedKey)
|
|
282
|
+
return items.sort((a, b) => this.rankCompare(seedKey)(a.candidate, b.candidate)).slice(0, limit);
|
|
215
283
|
// O(n) top-`limit` selection (limit is tiny, e.g. 1); avoids a full sort.
|
|
216
284
|
const top = [];
|
|
217
285
|
for (const item of items) {
|
package/dist/topic/types.d.ts
CHANGED
|
@@ -35,6 +35,8 @@ export interface TopicSpace {
|
|
|
35
35
|
sampledWorks: number;
|
|
36
36
|
tags: ResolvedTag[];
|
|
37
37
|
}
|
|
38
|
+
/** When related tags may be used as their own recall channel (§topic-recall). */
|
|
39
|
+
export type RelatedTagMode = 'always' | 'when_seed_insufficient' | 'never';
|
|
38
40
|
export interface TopicDiscoveryOptions {
|
|
39
41
|
/** Include R-18 works in topic sampling and collection (default false). */
|
|
40
42
|
includeR18?: boolean;
|
|
@@ -43,6 +45,18 @@ export interface TopicDiscoveryOptions {
|
|
|
43
45
|
cacheDays?: number;
|
|
44
46
|
minScore?: number;
|
|
45
47
|
refresh?: boolean;
|
|
48
|
+
/**
|
|
49
|
+
* Related-tag recall mode (default `'always'`).
|
|
50
|
+
*
|
|
51
|
+
* A resolved tag space is a hierarchy, not a bag of interchangeable tags: the
|
|
52
|
+
* seed tag is the topic the operator asked for and every other tag is a hint.
|
|
53
|
+
* Under `'always'` each resolved tag is searched for the day's works, so a
|
|
54
|
+
* second high-weight tag (丸吞) can occupy the only slot of a 西瓜肚 target.
|
|
55
|
+
* `'when_seed_insufficient'` searches the seed tag first and only walks the
|
|
56
|
+
* related channel when the seed cannot fill the limit for that day;
|
|
57
|
+
* `'never'` searches the seed tag alone.
|
|
58
|
+
*/
|
|
59
|
+
relatedTags?: RelatedTagMode;
|
|
46
60
|
}
|
|
47
61
|
export interface TopicCollectOptions {
|
|
48
62
|
maxPerTag?: number;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Intrinsic image dimensions read straight from the container header.
|
|
3
|
+
*
|
|
4
|
+
* PixivFlow ships no image decoder on purpose (§novel-cover): the only question
|
|
5
|
+
* it ever asks about a remote image is "what canvas is this?", and a handful of
|
|
6
|
+
* header bytes answer it for JPEG/PNG/GIF without decoding any pixels. Unknown
|
|
7
|
+
* containers, truncated input and non-image payloads return `undefined` so
|
|
8
|
+
* callers can fail open instead of guessing.
|
|
9
|
+
*/
|
|
10
|
+
export interface ImageDimensions {
|
|
11
|
+
width: number;
|
|
12
|
+
height: number;
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Reads the intrinsic canvas of a JPEG, PNG or GIF payload.
|
|
16
|
+
* Returns `undefined` when the format is unknown or the header is incomplete.
|
|
17
|
+
*/
|
|
18
|
+
export declare function readImageDimensions(input: ArrayBuffer | Uint8Array | null | undefined): ImageDimensions | undefined;
|
|
19
|
+
//# sourceMappingURL=imageDimensions.d.ts.map
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.readImageDimensions = readImageDimensions;
|
|
4
|
+
/** JPEG frame markers that carry the sample dimensions (SOF0..SOF15 minus DHT/JPG/DAC). */
|
|
5
|
+
const JPEG_SOF_MARKERS = new Set([
|
|
6
|
+
0xc0, 0xc1, 0xc2, 0xc3, 0xc5, 0xc6, 0xc7, 0xc9, 0xca, 0xcb, 0xcd, 0xce, 0xcf,
|
|
7
|
+
]);
|
|
8
|
+
const PNG_SIGNATURE = [0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a];
|
|
9
|
+
function toBytes(input) {
|
|
10
|
+
if (!input)
|
|
11
|
+
return undefined;
|
|
12
|
+
if (input instanceof Uint8Array)
|
|
13
|
+
return input;
|
|
14
|
+
if (input instanceof ArrayBuffer)
|
|
15
|
+
return new Uint8Array(input);
|
|
16
|
+
return undefined;
|
|
17
|
+
}
|
|
18
|
+
function readPngDimensions(bytes) {
|
|
19
|
+
// signature (8) + chunk length (4) + "IHDR" (4) + width (4) + height (4)
|
|
20
|
+
if (bytes.length < 24)
|
|
21
|
+
return undefined;
|
|
22
|
+
for (let i = 0; i < PNG_SIGNATURE.length; i += 1) {
|
|
23
|
+
if (bytes[i] !== PNG_SIGNATURE[i])
|
|
24
|
+
return undefined;
|
|
25
|
+
}
|
|
26
|
+
if (!(bytes[12] === 0x49 && bytes[13] === 0x48 && bytes[14] === 0x44 && bytes[15] === 0x52)) {
|
|
27
|
+
return undefined;
|
|
28
|
+
}
|
|
29
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
30
|
+
return { width: view.getUint32(16), height: view.getUint32(20) };
|
|
31
|
+
}
|
|
32
|
+
function readGifDimensions(bytes) {
|
|
33
|
+
if (bytes.length < 10)
|
|
34
|
+
return undefined;
|
|
35
|
+
const magic = String.fromCharCode(bytes[0], bytes[1], bytes[2]);
|
|
36
|
+
if (magic !== 'GIF')
|
|
37
|
+
return undefined;
|
|
38
|
+
const width = bytes[6] | (bytes[7] << 8);
|
|
39
|
+
const height = bytes[8] | (bytes[9] << 8);
|
|
40
|
+
if (!width || !height)
|
|
41
|
+
return undefined;
|
|
42
|
+
return { width, height };
|
|
43
|
+
}
|
|
44
|
+
function readJpegDimensions(bytes) {
|
|
45
|
+
if (bytes.length < 4 || bytes[0] !== 0xff || bytes[1] !== 0xd8)
|
|
46
|
+
return undefined;
|
|
47
|
+
let offset = 2;
|
|
48
|
+
while (offset + 9 < bytes.length) {
|
|
49
|
+
if (bytes[offset] !== 0xff) {
|
|
50
|
+
offset += 1;
|
|
51
|
+
continue;
|
|
52
|
+
}
|
|
53
|
+
const marker = bytes[offset + 1];
|
|
54
|
+
// Fill bytes / standalone markers carry no payload.
|
|
55
|
+
if (marker === 0xff) {
|
|
56
|
+
offset += 1;
|
|
57
|
+
continue;
|
|
58
|
+
}
|
|
59
|
+
if (marker === 0x01 || (marker >= 0xd0 && marker <= 0xd8)) {
|
|
60
|
+
offset += 2;
|
|
61
|
+
continue;
|
|
62
|
+
}
|
|
63
|
+
if (marker === 0xda)
|
|
64
|
+
return undefined; // entropy-coded data: no frame header left
|
|
65
|
+
const length = (bytes[offset + 2] << 8) | bytes[offset + 3];
|
|
66
|
+
if (length < 2)
|
|
67
|
+
return undefined;
|
|
68
|
+
if (JPEG_SOF_MARKERS.has(marker)) {
|
|
69
|
+
const height = (bytes[offset + 5] << 8) | bytes[offset + 6];
|
|
70
|
+
const width = (bytes[offset + 7] << 8) | bytes[offset + 8];
|
|
71
|
+
if (!width || !height)
|
|
72
|
+
return undefined;
|
|
73
|
+
return { width, height };
|
|
74
|
+
}
|
|
75
|
+
offset += 2 + length;
|
|
76
|
+
}
|
|
77
|
+
return undefined;
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* Reads the intrinsic canvas of a JPEG, PNG or GIF payload.
|
|
81
|
+
* Returns `undefined` when the format is unknown or the header is incomplete.
|
|
82
|
+
*/
|
|
83
|
+
function readImageDimensions(input) {
|
|
84
|
+
const bytes = toBytes(input);
|
|
85
|
+
if (!bytes || bytes.length < 10)
|
|
86
|
+
return undefined;
|
|
87
|
+
if (bytes[0] === 0xff && bytes[1] === 0xd8)
|
|
88
|
+
return readJpegDimensions(bytes);
|
|
89
|
+
if (bytes[0] === 0x89 && bytes[1] === 0x50)
|
|
90
|
+
return readPngDimensions(bytes);
|
|
91
|
+
if (bytes[0] === 0x47 && bytes[1] === 0x49 && bytes[2] === 0x46)
|
|
92
|
+
return readGifDimensions(bytes);
|
|
93
|
+
return undefined;
|
|
94
|
+
}
|
|
95
|
+
//# sourceMappingURL=imageDimensions.js.map
|
package/dist/version.js
CHANGED
|
@@ -2,5 +2,5 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.BUILD = void 0;
|
|
4
4
|
// GENERATED by scripts/write-version.js — do not edit manually.
|
|
5
|
-
exports.BUILD = { version: '3.0.
|
|
5
|
+
exports.BUILD = { version: '3.0.3', commit: 'b01a93d49e63' };
|
|
6
6
|
//# sourceMappingURL=version.js.map
|
package/dist/webui/package.json
CHANGED
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pixivflow",
|
|
3
|
-
"version": "3.0.
|
|
3
|
+
"version": "3.0.3",
|
|
4
4
|
"description": "🎨 Pixiv 下载、筛选与自动收集工具 - 批量下载插画和小说、按标签/热度/日期筛选、定时任务与可靠 HTTP 交付 | Pixiv downloader and automation toolkit with filtering, scheduling and reliable HTTP delivery",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|