pixivflow 3.0.2 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "type": "commonjs",
3
3
  "name": "pixivflow",
4
- "version": "3.0.2",
4
+ "version": "3.1.0",
5
5
  "private": true
6
6
  }
@@ -1,10 +1,16 @@
1
1
  import type { TargetConfig } from '../config';
2
2
  import type { TopicResolver } from './TopicResolver';
3
- import type { TopicCandidate, TopicClient, TopicCollectOptions, TopicContentType, TopicDiscoveryOptions, WorkLike } from './types';
3
+ import type { RelatedTagMode, ResolvedTag, TopicCandidate, TopicClient, TopicCollectOptions, TopicContentType, TopicDiscoveryOptions, TopicRelationsOptions, WorkLike } from './types';
4
4
  export interface TopicSelection {
5
5
  candidates: TopicCandidate[];
6
6
  selected: TopicCandidate[];
7
7
  resolvedTagCount: number;
8
+ /**
9
+ * Tags actually searched for the day. Under `relatedTags: 'always'` this is
10
+ * the whole resolved space; under the seed-first modes it is the seed tag and
11
+ * only the related tags that were really needed (§topic-recall).
12
+ */
13
+ searchedTags?: string[];
8
14
  rawCount: number;
9
15
  dedupedCount: number;
10
16
  acceptedCount: number;
@@ -12,6 +18,26 @@ export interface TopicSelection {
12
18
  /** Works seen more than once across the topic tag space (recorded, dropped). */
13
19
  duplicateRemovedCount: number;
14
20
  }
21
+ /**
22
+ * The tag the operator asked for, even when `tagRelations` filters it out.
23
+ *
24
+ * `deny` may drop the seed, but a space filtered down to nothing must still be
25
+ * usable: the pipeline then falls back to the seed tag alone, exactly like the
26
+ * resolver's `seedOnly()` degradation. `weight` defaults to the tag's `score`
27
+ * for spaces persisted before provenance existed.
28
+ */
29
+ export declare function seedFallbackTag(seedTag: ResolvedTag): ResolvedTag;
30
+ /**
31
+ * Applies `topicDiscovery.tagRelations` to a resolved space (§tag-provenance).
32
+ *
33
+ * Deny beats allow, and deny is the only thing that can drop the seed tag. A
34
+ * tag whose provenance is unknown (a space persisted before `source` existed, or
35
+ * a hand-written space) is treated as `cooccurrence` so that `allowSources` can
36
+ * still narrow it while the default (all sources) keeps walking everything.
37
+ */
38
+ export declare function selectWalkedTags(tags: readonly ResolvedTag[], seedKey: string, relations?: TopicRelationsOptions): ResolvedTag[];
39
+ /** Which of the walked tags a `relatedTags` mode will actually search today. */
40
+ export declare function recallChannels(walked: readonly ResolvedTag[], seedKey: string, relatedMode: RelatedTagMode): ResolvedTag[];
15
41
  /**
16
42
  * Resolves a topic to a tag space, collects that day's works across the tags,
17
43
  * filters by lightweight metadata relevance and ranks by local popularity, then
@@ -44,10 +70,29 @@ export declare class TopicPipeline {
44
70
  }>;
45
71
  private searchDay;
46
72
  private toCandidate;
73
+ /**
74
+ * Applies the metadata gate to everything collected so far. When nothing at
75
+ * all clears the threshold but the seed tag is present, the seed-tag works are
76
+ * kept anyway: a sparse day must stay usable instead of reporting "no
77
+ * candidates" for a topic that visibly has works. Extracted from selection so
78
+ * the seed pass can be evaluated before deciding whether the related channel
79
+ * is needed at all (§topic-recall).
80
+ */
81
+ private acceptedWorks;
47
82
  /**
48
83
  * Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
49
84
  * title/caption add smaller boosts. The seed tag is strong evidence.
50
85
  * No text model — case/symbol-insensitive substring matching only.
86
+ *
87
+ * `matchTranslatedNames` (default false) additionally counts a work's
88
+ * `translated_name` as a hit for the resolved tag it translates to, so a
89
+ * topic written in one language can still match works tagged in another. The
90
+ * translated name is matched against the resolved space — never added to
91
+ * `candidate.tags` — so the work is not reported as carrying a tag it lacks.
92
+ *
93
+ * A resolved tag counts at most ONCE per work, whether it matched through the
94
+ * tag name or through its translation (and a work that repeats a tag, as Pixiv
95
+ * payloads do for a tag plus its translation, does not double its weight).
51
96
  */
52
97
  private metadataScore;
53
98
  /**
@@ -56,8 +101,17 @@ export declare class TopicPipeline {
56
101
  * as on-topic, and the choice between accepted works is decided by popularity
57
102
  * alone. A work with a higher metadata score does NOT outrank a more popular
58
103
  * accepted work.
104
+ *
105
+ * `seedKey` adds a single tier in front of that popularity order and is
106
+ * passed by the seed-first recall modes (§topic-recall) and by
107
+ * `topicDiscovery.seedTier: 'on'` (§tag-provenance): a work that actually
108
+ * carries the topic tag outranks a related-only work, however popular the
109
+ * latter is, and popularity decides within each tier. With `seedTier: 'off'`
110
+ * and the default `relatedTags: 'always'` no `seedKey` is passed, so the
111
+ * documented popularity-only ranking is unchanged.
59
112
  */
60
113
  private popCompare;
114
+ private rankCompare;
61
115
  private topByPopularity;
62
116
  private onDay;
63
117
  private normalize;
@@ -1,6 +1,9 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.TopicPipeline = void 0;
4
+ exports.seedFallbackTag = seedFallbackTag;
5
+ exports.selectWalkedTags = selectWalkedTags;
6
+ exports.recallChannels = recallChannels;
4
7
  const promises_1 = require("node:timers/promises");
5
8
  const logger_1 = require("../logger");
6
9
  const pixiv_utils_1 = require("../utils/pixiv-utils");
@@ -14,6 +17,105 @@ const STOP_TAGS = new Set([
14
17
  '1000users入り', '5000users入り', '10000users入り', '500users入り', '100users入り',
15
18
  'pixiv', 'commission', 'skeb', '依頼絵', '仕事絵',
16
19
  ]);
20
+ /** Normalization shared by the tag-relation filter and the pipeline itself. */
21
+ function normalizeKey(value) {
22
+ return value.trim().normalize('NFKC').toLocaleLowerCase();
23
+ }
24
+ /**
25
+ * The tag the operator asked for, even when `tagRelations` filters it out.
26
+ *
27
+ * `deny` may drop the seed, but a space filtered down to nothing must still be
28
+ * usable: the pipeline then falls back to the seed tag alone, exactly like the
29
+ * resolver's `seedOnly()` degradation. `weight` defaults to the tag's `score`
30
+ * for spaces persisted before provenance existed.
31
+ */
32
+ function seedFallbackTag(seedTag) {
33
+ if (seedTag.source === 'seed' && seedTag.weight !== undefined)
34
+ return seedTag;
35
+ return { ...seedTag, source: 'seed', weight: seedTag.weight ?? seedTag.score, seed: true };
36
+ }
37
+ /**
38
+ * Applies `topicDiscovery.tagRelations` to a resolved space (§tag-provenance).
39
+ *
40
+ * Deny beats allow, and deny is the only thing that can drop the seed tag. A
41
+ * tag whose provenance is unknown (a space persisted before `source` existed, or
42
+ * a hand-written space) is treated as `cooccurrence` so that `allowSources` can
43
+ * still narrow it while the default (all sources) keeps walking everything.
44
+ */
45
+ function selectWalkedTags(tags, seedKey, relations = {}) {
46
+ const seedTag = tags.find((tag) => normalizeKey(tag.name) === seedKey);
47
+ const allowed = tags.filter((tag) => normalizeKey(tag.name) !== seedKey);
48
+ const deny = (relations.deny ?? []).map(normalizeKey);
49
+ const deniedKeys = new Set(deny);
50
+ const seedDenied = seedTag !== undefined && deniedKeys.has(seedKey);
51
+ if (seedDenied)
52
+ deniedKeys.clear(); // the seed fallback ignores deny
53
+ const allow = (relations.allow ?? []).map(normalizeKey);
54
+ const allowKeys = new Set(allow);
55
+ const open = allowKeys.size === 0;
56
+ const sources = relations.allowSources;
57
+ const sourceAllowed = (tag) => {
58
+ if (!sources || sources.length === 0)
59
+ return true;
60
+ const name = normalizeKey(tag.name);
61
+ if (allowKeys.has(name))
62
+ return true; // an explicit allow beats allowSources
63
+ const provenance = tag.source;
64
+ if (provenance === undefined)
65
+ return sources.includes('cooccurrence');
66
+ if (provenance === 'cooccurrence+autocomplete') {
67
+ return sources.includes('cooccurrence') || sources.includes('autocomplete');
68
+ }
69
+ return sources.includes(provenance);
70
+ };
71
+ const walked = allowed.filter((tag) => {
72
+ const name = normalizeKey(tag.name);
73
+ if (deniedKeys.has(name))
74
+ return false;
75
+ if (!open && !allowKeys.has(name))
76
+ return false;
77
+ return sourceAllowed(tag);
78
+ });
79
+ // No seed tag in the space at all: there is nothing to dominate, so fall back
80
+ // to walking the whole (filtered) space — the pre-provenance behaviour.
81
+ if (!seedTag)
82
+ return walked;
83
+ // A denied seed empties the space, exactly like the resolver's seedOnly():
84
+ // the operator still gets their own topic, and no related tag is walked.
85
+ if (seedDenied)
86
+ return [seedFallbackTag(seedTag)];
87
+ return [seedTag, ...walked];
88
+ }
89
+ /** Every spelling under which a set of resolved tags can be recognized. */
90
+ function buildTagAliases(tags, matchTranslatedNames) {
91
+ const aliases = new Map();
92
+ for (const tag of tags) {
93
+ const primary = normalizeKey(tag.name);
94
+ const entry = { primary, weight: tag.weight ?? tag.score };
95
+ if (!aliases.has(primary))
96
+ aliases.set(primary, entry);
97
+ if (!matchTranslatedNames)
98
+ continue;
99
+ // The work-side `translated_name` for this tag is the same resolved tag: the
100
+ // space still exposes one tag, it is simply recognized under two keys.
101
+ const translated = tag.translatedName ? normalizeKey(tag.translatedName) : '';
102
+ if (translated && !aliases.has(translated))
103
+ aliases.set(translated, entry);
104
+ }
105
+ return aliases;
106
+ }
107
+ /** Which of the walked tags a `relatedTags` mode will actually search today. */
108
+ function recallChannels(walked, seedKey, relatedMode) {
109
+ const seedTags = walked.filter((tag) => normalizeKey(tag.name) === seedKey);
110
+ // A space with no seed tag has nothing to protect: every remaining tag is the
111
+ // only channel there is, so the whole space is walked whatever the mode
112
+ // (pre-provenance fallback for a hand-written space).
113
+ if (seedTags.length === 0)
114
+ return [...walked];
115
+ if (relatedMode === 'never' || relatedMode === 'when_seed_insufficient')
116
+ return seedTags;
117
+ return [...seedTags, ...walked.filter((tag) => normalizeKey(tag.name) !== seedKey)];
118
+ }
17
119
  /**
18
120
  * Resolves a topic to a tag space, collects that day's works across the tags,
19
121
  * filters by lightweight metadata relevance and ranks by local popularity, then
@@ -42,23 +144,39 @@ class TopicPipeline {
42
144
  async selectWorks(target, contentType, day, limit, discovery, collect) {
43
145
  const topic = (target.topic ?? '').trim();
44
146
  const { space } = await this.resolver.resolve(topic, contentType, discovery);
45
- const tagScores = new Map(space.tags.map((t) => [this.key(t.name), t.score]));
46
147
  const maxPerTag = this.bound(collect.maxPerTag, COLLECT_DEFAULTS.maxPerTag, 5, 100);
47
148
  const maxCandidates = this.bound(collect.maxCandidates, COLLECT_DEFAULTS.maxCandidates, 20, 500);
48
149
  const minMetadataScore = collect.minMetadataScore ?? COLLECT_DEFAULTS.minMetadataScore;
49
150
  const includeR18 = discovery.includeR18 === true;
151
+ // An unknown mode (hand-written config bypassing validation) falls back to
152
+ // the historical behaviour rather than silently narrowing recall.
153
+ const requestedMode = discovery.relatedTags;
154
+ const relatedMode = requestedMode === 'when_seed_insufficient' || requestedMode === 'never' ? requestedMode : 'always';
50
155
  const byId = new Map();
51
156
  let rawCount = 0;
52
157
  let aiExcludedCount = 0;
53
158
  let duplicateRemovedCount = 0;
54
- const tagNames = space.tags.map((t) => t.name);
55
- for (let i = 0; i < tagNames.length; i++) {
56
- if (byId.size >= maxCandidates)
57
- break;
159
+ const seedKey = this.key(topic);
160
+ // §tag-provenance: apply tagRelations where the walked list is built, so the
161
+ // filter governs seedTags/relatedTags BEFORE any search is issued. The seed
162
+ // tag survives allow/allowSources; a deny entry that empties the space falls
163
+ // back to the seed tag alone, exactly like the resolver's seedOnly().
164
+ const matchTranslatedNames = discovery.matchTranslatedNames === true;
165
+ const walked = selectWalkedTags(space.tags, seedKey, discovery.tagRelations);
166
+ const tagNames = walked.map((t) => t.name);
167
+ const seedTags = tagNames.filter((name) => this.key(name) === seedKey);
168
+ const relatedTags = tagNames.filter((name) => this.key(name) !== seedKey);
169
+ // Ranking reads the documented semantic weight, falling back to `score` for
170
+ // spaces persisted before provenance existed. The two are always equal.
171
+ const tagScores = buildTagAliases(walked, matchTranslatedNames);
172
+ const searchedTags = [];
173
+ // A hard seed tier: 'on' always, and the seed-first modes keep their layer.
174
+ const hardSeedTier = discovery.seedTier === 'on' || relatedMode !== 'always';
175
+ const collectTag = async (tag) => {
58
176
  // Cancellation is checked between tags, so a cancelled run stops issuing
59
177
  // new searches even when the aborted request itself had already returned.
60
178
  (0, errors_1.throwIfAborted)(this.signal, 'topic collection cancelled');
61
- const tag = tagNames[i];
179
+ searchedTags.push(tag);
62
180
  const works = await this.searchDay(contentType, tag, day, maxPerTag, includeR18);
63
181
  rawCount += works.length;
64
182
  for (const work of works) {
@@ -75,37 +193,57 @@ class TopicPipeline {
75
193
  break;
76
194
  }
77
195
  logger_1.logger.debug('[TopicCollector] type=' + contentType + ' tag=' + tag + ' day=' + day + ' fetched=' + works.length + ' pool=' + byId.size);
78
- if (i < tagNames.length - 1 && this.requestDelayMs > 0)
79
- await (0, promises_1.setTimeout)(this.requestDelayMs);
80
- }
81
- const dedupedCount = byId.size;
82
- logger_1.logger.info('[TopicCollector] type=' + contentType + ' raw=' + rawCount + ' deduplicated=' + dedupedCount + ' aiExcluded=' + aiExcludedCount);
83
- const seedKey = this.key(topic);
84
- const accepted = [];
85
- for (const entry of byId.values()) {
86
- entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores);
87
- if (entry.candidate.metadataScore >= minMetadataScore)
88
- accepted.push(entry);
196
+ };
197
+ const runTags = async (names) => {
198
+ for (let i = 0; i < names.length; i++) {
199
+ if (byId.size >= maxCandidates)
200
+ break;
201
+ await collectTag(names[i]);
202
+ if (i < names.length - 1 && this.requestDelayMs > 0)
203
+ await (0, promises_1.setTimeout)(this.requestDelayMs);
204
+ }
205
+ };
206
+ // §topic-recall: a resolved tag space is a hierarchy, not a bag of
207
+ // interchangeable tags. 'always' keeps the historical behaviour — every
208
+ // resolved tag is a recall channel for the day. The seed-first modes search
209
+ // the topic tag the operator actually asked for and only walk the related
210
+ // channel when that cannot fill the target, so a second high-weight tag
211
+ // (丸吞) cannot take the only slot of a 西瓜肚 target.
212
+ if (seedTags.length === 0 || relatedMode === 'always') {
213
+ // No seed tag in the space (hand-written space): keep walking everything
214
+ // rather than returning nothing.
215
+ await runTags(recallChannels(walked, seedKey, relatedMode).map((t) => t.name));
89
216
  }
90
- if (accepted.length === 0 && byId.size > 0) {
91
- const fallback = [...byId.values()]
92
- .filter((e) => e.candidate.tags.some((t) => this.key(t) === seedKey))
93
- .sort((a, b) => b.candidate.popularity - a.candidate.popularity);
94
- accepted.push(...fallback.slice(0, Math.max(limit, 1)));
95
- logger_1.logger.warn('[MetadataTopicFilter] type=' + contentType + ' none above threshold ' + minMetadataScore + '; kept ' + accepted.length + ' seed-tag fallback');
217
+ else {
218
+ await runTags(seedTags);
219
+ const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames).length;
220
+ if (relatedMode === 'never') {
221
+ logger_1.logger.info('[TopicRecall] mode=never tag=' + topic + ' day=' + day + ' accepted=' + seedAccepted);
222
+ }
223
+ else if (seedAccepted < limit) {
224
+ logger_1.logger.info('[TopicRecall] mode=when_seed_insufficient tag=' + topic + ' seedAccepted=' + seedAccepted + '/' + limit + ' relatedTags=' + relatedTags.length + '; expanding');
225
+ await runTags(relatedTags);
226
+ }
227
+ else {
228
+ logger_1.logger.info('[TopicRecall] mode=when_seed_insufficient tag=' + topic + ' seedAccepted=' + seedAccepted + '/' + limit + '; related tags not searched');
229
+ }
96
230
  }
97
- const chosen = this.topByPopularity(accepted, limit);
231
+ const dedupedCount = byId.size;
232
+ logger_1.logger.info('[TopicCollector] type=' + contentType + ' raw=' + rawCount + ' deduplicated=' + dedupedCount + ' aiExcluded=' + aiExcludedCount + ' searchedTags=' + searchedTags.length);
233
+ const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames);
234
+ const chosen = this.topByPopularity(accepted, limit, hardSeedTier ? seedKey : undefined);
98
235
  const selected = chosen.map((e) => e.candidate);
99
236
  logger_1.logger.info('[MetadataTopicFilter] accepted=' + accepted.length);
100
237
  if (selected[0]) {
101
- logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + ' title=' + selected[0].title);
238
+ logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + (hardSeedTier ? ' seedTier=on' : '') + ' title=' + selected[0].title);
102
239
  }
103
240
  return {
104
241
  works: chosen.map((e) => e.work),
105
242
  selection: {
106
243
  candidates: [...byId.values()].map((e) => e.candidate),
107
244
  selected,
108
- resolvedTagCount: space.tags.length,
245
+ resolvedTagCount: walked.length,
246
+ searchedTags,
109
247
  rawCount,
110
248
  dedupedCount,
111
249
  acceptedCount: accepted.length,
@@ -137,12 +275,16 @@ class TopicPipeline {
137
275
  }
138
276
  toCandidate(work, type) {
139
277
  const popularity = (0, pixiv_utils_1.calculatePopularityScore)(work);
278
+ const translatedTags = (work.tags ?? [])
279
+ .map((t) => t.translated_name)
280
+ .filter((name) => Boolean(name));
140
281
  return {
141
282
  id: work.id,
142
283
  type,
143
284
  title: work.title ?? '',
144
285
  caption: work.caption ?? '',
145
286
  tags: (work.tags ?? []).map((t) => t.name).filter(Boolean),
287
+ ...(translatedTags.length > 0 ? { translatedTags } : {}),
146
288
  bookmarks: Number(work.total_bookmarks ?? work.bookmark_count ?? 0) || 0,
147
289
  views: Number(work.total_view ?? work.view_count ?? 0) || 0,
148
290
  popularity,
@@ -150,29 +292,84 @@ class TopicPipeline {
150
292
  ...(work.illust_ai_type !== undefined ? { aiType: work.illust_ai_type } : {}),
151
293
  };
152
294
  }
295
+ /**
296
+ * Applies the metadata gate to everything collected so far. When nothing at
297
+ * all clears the threshold but the seed tag is present, the seed-tag works are
298
+ * kept anyway: a sparse day must stay usable instead of reporting "no
299
+ * candidates" for a topic that visibly has works. Extracted from selection so
300
+ * the seed pass can be evaluated before deciding whether the related channel
301
+ * is needed at all (§topic-recall).
302
+ */
303
+ acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames) {
304
+ const accepted = [];
305
+ const hitsSeed = (candidate) => {
306
+ if (candidate.tags.some((t) => this.key(t) === seedKey))
307
+ return true;
308
+ if (!matchTranslatedNames)
309
+ return false;
310
+ return (candidate.translatedTags ?? []).some((t) => this.key(t) === seedKey);
311
+ };
312
+ for (const entry of byId.values()) {
313
+ entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores, matchTranslatedNames);
314
+ if (entry.candidate.metadataScore >= minMetadataScore)
315
+ accepted.push(entry);
316
+ }
317
+ if (accepted.length === 0 && byId.size > 0) {
318
+ const fallback = [...byId.values()]
319
+ .filter((e) => hitsSeed(e.candidate))
320
+ .sort((a, b) => b.candidate.popularity - a.candidate.popularity);
321
+ accepted.push(...fallback.slice(0, Math.max(limit, 1)));
322
+ logger_1.logger.warn('[MetadataTopicFilter] type=' + contentType + ' none above threshold ' + minMetadataScore + '; kept ' + accepted.length + ' seed-tag fallback');
323
+ }
324
+ return accepted;
325
+ }
153
326
  /**
154
327
  * Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
155
328
  * title/caption add smaller boosts. The seed tag is strong evidence.
156
329
  * No text model — case/symbol-insensitive substring matching only.
330
+ *
331
+ * `matchTranslatedNames` (default false) additionally counts a work's
332
+ * `translated_name` as a hit for the resolved tag it translates to, so a
333
+ * topic written in one language can still match works tagged in another. The
334
+ * translated name is matched against the resolved space — never added to
335
+ * `candidate.tags` — so the work is not reported as carrying a tag it lacks.
336
+ *
337
+ * A resolved tag counts at most ONCE per work, whether it matched through the
338
+ * tag name or through its translation (and a work that repeats a tag, as Pixiv
339
+ * payloads do for a tag plus its translation, does not double its weight).
157
340
  */
158
- metadataScore(candidate, seedKey, tagScores) {
341
+ metadataScore(candidate, seedKey, tagScores, matchTranslatedNames) {
159
342
  let seedHit = false;
160
343
  let relatedSum = 0;
161
344
  let relatedHits = 0;
162
- for (const tag of candidate.tags) {
163
- const k = this.key(tag);
164
- if (STOP_TAGS.has(k))
165
- continue;
166
- if (k === seedKey) {
345
+ let strongRelated = false;
346
+ // A resolved tag is matched through its own name, and — when the caller
347
+ // opted in — through the work-side translation of that tag. Either spelling
348
+ // maps back to the SAME resolved tag, so a work carrying both spellings
349
+ // counts once and never gets a double weight.
350
+ const countedTagKeys = new Set();
351
+ const consider = (candidateKey) => {
352
+ if (candidateKey === '' || STOP_TAGS.has(candidateKey))
353
+ return;
354
+ if (candidateKey === seedKey) {
167
355
  seedHit = true;
168
- continue;
169
- }
170
- const related = tagScores.get(k);
171
- if (related !== undefined) {
172
- relatedSum += Math.min(related, 0.6);
173
- relatedHits += 1;
356
+ return;
174
357
  }
358
+ const aliases = tagScores.get(candidateKey);
359
+ if (!aliases || countedTagKeys.has(aliases.primary))
360
+ return;
361
+ countedTagKeys.add(aliases.primary);
362
+ relatedSum += Math.min(aliases.weight, 0.6);
363
+ relatedHits += 1;
364
+ if (aliases.weight >= 0.6)
365
+ strongRelated = true;
366
+ };
367
+ if (matchTranslatedNames) {
368
+ for (const tag of candidate.translatedTags ?? [])
369
+ consider(this.key(tag));
175
370
  }
371
+ for (const tag of candidate.tags)
372
+ consider(this.key(tag));
176
373
  const hayTitle = this.normalize(candidate.title);
177
374
  const hayCaption = this.normalize(candidate.caption);
178
375
  const titleSeed = !!seedKey && hayTitle.includes(seedKey);
@@ -189,7 +386,6 @@ class TopicPipeline {
189
386
  // deliberately does NOT clear the bar, so a hugely popular tangential work
190
387
  // cannot crowd out core topic works.
191
388
  let score = 0;
192
- const strongRelated = [...tagScores.entries()].some(([k, w]) => w >= 0.6 && candidate.tags.some((t) => this.key(t) === k));
193
389
  if (titleSeed)
194
390
  score += 0.8;
195
391
  if (captionSeed)
@@ -205,13 +401,32 @@ class TopicPipeline {
205
401
  * as on-topic, and the choice between accepted works is decided by popularity
206
402
  * alone. A work with a higher metadata score does NOT outrank a more popular
207
403
  * accepted work.
404
+ *
405
+ * `seedKey` adds a single tier in front of that popularity order and is
406
+ * passed by the seed-first recall modes (§topic-recall) and by
407
+ * `topicDiscovery.seedTier: 'on'` (§tag-provenance): a work that actually
408
+ * carries the topic tag outranks a related-only work, however popular the
409
+ * latter is, and popularity decides within each tier. With `seedTier: 'off'`
410
+ * and the default `relatedTags: 'always'` no `seedKey` is passed, so the
411
+ * documented popularity-only ranking is unchanged.
208
412
  */
209
413
  popCompare(a, b) {
210
414
  return b.popularity - a.popularity;
211
415
  }
212
- topByPopularity(items, limit) {
416
+ rankCompare(seedKey) {
417
+ if (!seedKey)
418
+ return (a, b) => this.popCompare(a, b);
419
+ const tier = (c) => (c.tags.some((t) => this.key(t) === seedKey) ? 0 : 1);
420
+ return (a, b) => {
421
+ const diff = tier(a) - tier(b);
422
+ return diff !== 0 ? diff : this.popCompare(a, b);
423
+ };
424
+ }
425
+ topByPopularity(items, limit, seedKey) {
213
426
  if (items.length <= limit)
214
- return items.sort((a, b) => this.popCompare(a.candidate, b.candidate));
427
+ return items.sort((a, b) => this.rankCompare(seedKey)(a.candidate, b.candidate));
428
+ if (seedKey)
429
+ return items.sort((a, b) => this.rankCompare(seedKey)(a.candidate, b.candidate)).slice(0, limit);
215
430
  // O(n) top-`limit` selection (limit is tiny, e.g. 1); avoids a full sort.
216
431
  const top = [];
217
432
  for (const item of items) {
@@ -97,6 +97,9 @@ class TopicResolver {
97
97
  name: seed,
98
98
  translatedName: suggested.find((t) => this.scorer.key(t.name) === this.scorer.key(seed))?.translated_name,
99
99
  score: 1,
100
+ weight: 1,
101
+ // The tag the operator asked for: the strongest provenance, never a hint.
102
+ source: 'seed',
100
103
  occurrences: topicWorks.length,
101
104
  coverage: 1,
102
105
  specificity: 1,
@@ -168,7 +171,7 @@ class TopicResolver {
168
171
  expiresAt: new Date(now + cacheDays * 24 * 60 * 60_000).toISOString(),
169
172
  sampleSize: 0,
170
173
  sampledWorks: 0,
171
- tags: [{ name: seed, score: 1, occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
174
+ tags: [{ name: seed, score: 1, weight: 1, source: 'seed', occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
172
175
  };
173
176
  }
174
177
  ageDays(space) {
@@ -99,10 +99,17 @@ class TopicTagScorer {
99
99
  const suggestionWeight = stat.suggested ? 1.1 : 1.0;
100
100
  const genericPenalty = GENERIC_TAG_PENALTY.has(k) ? 0.4 : 1.0;
101
101
  const raw = recall * specificity * suggestionWeight * genericPenalty;
102
+ const score = Number(raw.toFixed(4));
102
103
  resolved.push({
103
104
  name: stat.name,
104
105
  translatedName: stat.translatedName,
105
- score: Number(raw.toFixed(4)),
106
+ score,
107
+ // `weight` is the same semantic number; ranking and diagnostics read it
108
+ // while `score` stays for backward compatibility (§tag-provenance).
109
+ weight: score,
110
+ // Both channels may agree on a tag; the combined provenance records
111
+ // that, which is strictly more informative than either alone.
112
+ source: stat.suggested ? 'cooccurrence+autocomplete' : 'cooccurrence',
106
113
  occurrences: stat.topicDocs,
107
114
  coverage: Number(coverage.toFixed(4)),
108
115
  specificity: Number(specificity.toFixed(4)),
@@ -127,6 +134,10 @@ class TopicTagScorer {
127
134
  name: name,
128
135
  translatedName: sug.translated_name?.trim() || undefined,
129
136
  score: AUTOCOMPLETE_ONLY_SCORE,
137
+ weight: AUTOCOMPLETE_ONLY_SCORE,
138
+ // Pixiv autocomplete is the only evidence for this tag: it never
139
+ // co-occurred in the bounded sample.
140
+ source: 'autocomplete',
130
141
  occurrences: 0,
131
142
  coverage: 0,
132
143
  specificity: 1.0, // Pixiv-endorsed related; treated as specific but unobserved
@@ -4,12 +4,34 @@
4
4
  * local models are used anywhere in this module.
5
5
  */
6
6
  export type TopicContentType = 'illustration' | 'novel';
7
+ /**
8
+ * Where a resolved tag came from. Provenance is what lets a caller treat a weak
9
+ * expansion differently from the topic it was asked for (§tag-provenance):
10
+ *
11
+ * - `'seed'`: the tag the operator asked for. Always the strongest key.
12
+ * - `'cooccurrence'`: sampled together with the seed, but Pixiv autocomplete
13
+ * does not relate it to the seed. Co-occurrence evidence only.
14
+ * - `'autocomplete'`: Pixiv autocomplete relates it to the seed, but it never
15
+ * appeared in the sample. No co-occurrence evidence.
16
+ * - `'cooccurrence+autocomplete'`: both channels agree — the strongest related
17
+ * provenance available.
18
+ */
19
+ export type TagSource = 'seed' | 'cooccurrence' | 'autocomplete' | 'cooccurrence+autocomplete';
7
20
  /** A single related tag with a 0..1 relatedness score and provenance. */
8
21
  export interface ResolvedTag {
9
22
  name: string;
10
23
  translatedName?: string;
11
24
  /** Combined relatedness score (co-occurrence * specificity * suggestion). */
12
25
  score: number;
26
+ /**
27
+ * Semantic weight used for ranking, filtering and diagnostics. Always the
28
+ * same number as `score`; kept as a separate, documented field so ranking can
29
+ * be explained (and, later, adjusted) without redefining `score`.
30
+ * Optional: spaces persisted before provenance existed have neither field.
31
+ */
32
+ weight?: number;
33
+ /** Provenance of the tag. Optional for the same reason as `weight`. */
34
+ source?: TagSource;
13
35
  /** How many sampled works (of the seed search) carried this tag. */
14
36
  occurrences: number;
15
37
  /** Coverage of the sampled seed works (occurrences / sample size). */
@@ -35,6 +57,8 @@ export interface TopicSpace {
35
57
  sampledWorks: number;
36
58
  tags: ResolvedTag[];
37
59
  }
60
+ /** When related tags may be used as their own recall channel (§topic-recall). */
61
+ export type RelatedTagMode = 'always' | 'when_seed_insufficient' | 'never';
38
62
  export interface TopicDiscoveryOptions {
39
63
  /** Include R-18 works in topic sampling and collection (default false). */
40
64
  includeR18?: boolean;
@@ -43,6 +67,38 @@ export interface TopicDiscoveryOptions {
43
67
  cacheDays?: number;
44
68
  minScore?: number;
45
69
  refresh?: boolean;
70
+ /**
71
+ * Related-tag recall mode (default `'always'`).
72
+ *
73
+ * A resolved tag space is a hierarchy, not a bag of interchangeable tags: the
74
+ * seed tag is the topic the operator asked for and every other tag is a hint.
75
+ * Under `'always'` each resolved tag is searched for the day's works, so a
76
+ * second high-weight tag (丸吞) can occupy the only slot of a 西瓜肚 target.
77
+ * `'when_seed_insufficient'` searches the seed tag first and only walks the
78
+ * related channel when the seed cannot fill the limit for that day;
79
+ * `'never'` searches the seed tag alone.
80
+ */
81
+ relatedTags?: RelatedTagMode;
82
+ /**
83
+ * Which resolved tags may become recall channels (§tag-provenance). Deny wins
84
+ * over allow; the seed tag is never dropped by `allow`/`allowSources`.
85
+ */
86
+ tagRelations?: TopicRelationsOptions;
87
+ /** Make the seed tag a hard ranking tier (default `'off'`). */
88
+ seedTier?: 'off' | 'on';
89
+ /** Count a work's translated tag names as tag hits (default false). */
90
+ matchTranslatedNames?: boolean;
91
+ }
92
+ /**
93
+ * Runtime shape of `TopicDiscoveryConfig.tagRelations`, declared here so the
94
+ * topic module does not depend on the config layer (the topic pipeline is also
95
+ * driven by hand-written targets in tests and by callers that never load a
96
+ * config file).
97
+ */
98
+ export interface TopicRelationsOptions {
99
+ allowSources?: TagSource[];
100
+ allow?: string[];
101
+ deny?: string[];
46
102
  }
47
103
  export interface TopicCollectOptions {
48
104
  maxPerTag?: number;
@@ -62,6 +118,12 @@ export interface TopicCandidate {
62
118
  popularity: number;
63
119
  /** Metadata topic-relevance score computed by the filter stage. */
64
120
  metadataScore: number;
121
+ /**
122
+ * Translated tag names carried by the work, kept beside `tags` so
123
+ * `matchTranslatedNames` can compare them WITHOUT claiming the work itself
124
+ * carries a tag it does not (§tag-provenance). Default-off.
125
+ */
126
+ translatedTags?: string[];
65
127
  /** Pixiv AI classification copied from illustration search metadata. */
66
128
  aiType?: number;
67
129
  }
@@ -0,0 +1,19 @@
1
+ /**
2
+ * Intrinsic image dimensions read straight from the container header.
3
+ *
4
+ * PixivFlow ships no image decoder on purpose (§novel-cover): the only question
5
+ * it ever asks about a remote image is "what canvas is this?", and a handful of
6
+ * header bytes answer it for JPEG/PNG/GIF without decoding any pixels. Unknown
7
+ * containers, truncated input and non-image payloads return `undefined` so
8
+ * callers can fail open instead of guessing.
9
+ */
10
+ export interface ImageDimensions {
11
+ width: number;
12
+ height: number;
13
+ }
14
+ /**
15
+ * Reads the intrinsic canvas of a JPEG, PNG or GIF payload.
16
+ * Returns `undefined` when the format is unknown or the header is incomplete.
17
+ */
18
+ export declare function readImageDimensions(input: ArrayBuffer | Uint8Array | null | undefined): ImageDimensions | undefined;
19
+ //# sourceMappingURL=imageDimensions.d.ts.map