pixivflow 3.0.3 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,9 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.TopicPipeline = void 0;
4
+ exports.seedFallbackTag = seedFallbackTag;
5
+ exports.selectWalkedTags = selectWalkedTags;
6
+ exports.recallChannels = recallChannels;
4
7
  const promises_1 = require("node:timers/promises");
5
8
  const logger_1 = require("../logger");
6
9
  const pixiv_utils_1 = require("../utils/pixiv-utils");
@@ -14,6 +17,105 @@ const STOP_TAGS = new Set([
14
17
  '1000users入り', '5000users入り', '10000users入り', '500users入り', '100users入り',
15
18
  'pixiv', 'commission', 'skeb', '依頼絵', '仕事絵',
16
19
  ]);
20
+ /** Normalization shared by the tag-relation filter and the pipeline itself. */
21
+ function normalizeKey(value) {
22
+ return value.trim().normalize('NFKC').toLocaleLowerCase();
23
+ }
24
+ /**
25
+ * The tag the operator asked for, even when `tagRelations` filters it out.
26
+ *
27
+ * `deny` may drop the seed, but a space filtered down to nothing must still be
28
+ * usable: the pipeline then falls back to the seed tag alone, exactly like the
29
+ * resolver's `seedOnly()` degradation. `weight` defaults to the tag's `score`
30
+ * for spaces persisted before provenance existed.
31
+ */
32
+ function seedFallbackTag(seedTag) {
33
+ if (seedTag.source === 'seed' && seedTag.weight !== undefined)
34
+ return seedTag;
35
+ return { ...seedTag, source: 'seed', weight: seedTag.weight ?? seedTag.score, seed: true };
36
+ }
37
+ /**
38
+ * Applies `topicDiscovery.tagRelations` to a resolved space (§tag-provenance).
39
+ *
40
+ * Deny beats allow, and deny is the only thing that can drop the seed tag. A
41
+ * tag whose provenance is unknown (a space persisted before `source` existed, or
42
+ * a hand-written space) is treated as `cooccurrence` so that `allowSources` can
43
+ * still narrow it while the default (all sources) keeps walking everything.
44
+ */
45
+ function selectWalkedTags(tags, seedKey, relations = {}) {
46
+ const seedTag = tags.find((tag) => normalizeKey(tag.name) === seedKey);
47
+ const allowed = tags.filter((tag) => normalizeKey(tag.name) !== seedKey);
48
+ const deny = (relations.deny ?? []).map(normalizeKey);
49
+ const deniedKeys = new Set(deny);
50
+ const seedDenied = seedTag !== undefined && deniedKeys.has(seedKey);
51
+ if (seedDenied)
52
+ deniedKeys.clear(); // the seed fallback ignores deny
53
+ const allow = (relations.allow ?? []).map(normalizeKey);
54
+ const allowKeys = new Set(allow);
55
+ const open = allowKeys.size === 0;
56
+ const sources = relations.allowSources;
57
+ const sourceAllowed = (tag) => {
58
+ if (!sources || sources.length === 0)
59
+ return true;
60
+ const name = normalizeKey(tag.name);
61
+ if (allowKeys.has(name))
62
+ return true; // an explicit allow beats allowSources
63
+ const provenance = tag.source;
64
+ if (provenance === undefined)
65
+ return sources.includes('cooccurrence');
66
+ if (provenance === 'cooccurrence+autocomplete') {
67
+ return sources.includes('cooccurrence') || sources.includes('autocomplete');
68
+ }
69
+ return sources.includes(provenance);
70
+ };
71
+ const walked = allowed.filter((tag) => {
72
+ const name = normalizeKey(tag.name);
73
+ if (deniedKeys.has(name))
74
+ return false;
75
+ if (!open && !allowKeys.has(name))
76
+ return false;
77
+ return sourceAllowed(tag);
78
+ });
79
+ // No seed tag in the space at all: there is nothing to dominate, so fall back
80
+ // to walking the whole (filtered) space — the pre-provenance behaviour.
81
+ if (!seedTag)
82
+ return walked;
83
+ // A denied seed empties the space, exactly like the resolver's seedOnly():
84
+ // the operator still gets their own topic, and no related tag is walked.
85
+ if (seedDenied)
86
+ return [seedFallbackTag(seedTag)];
87
+ return [seedTag, ...walked];
88
+ }
89
+ /** Every spelling under which a set of resolved tags can be recognized. */
90
+ function buildTagAliases(tags, matchTranslatedNames) {
91
+ const aliases = new Map();
92
+ for (const tag of tags) {
93
+ const primary = normalizeKey(tag.name);
94
+ const entry = { primary, weight: tag.weight ?? tag.score };
95
+ if (!aliases.has(primary))
96
+ aliases.set(primary, entry);
97
+ if (!matchTranslatedNames)
98
+ continue;
99
+ // The work-side `translated_name` for this tag is the same resolved tag: the
100
+ // space still exposes one tag, it is simply recognized under two keys.
101
+ const translated = tag.translatedName ? normalizeKey(tag.translatedName) : '';
102
+ if (translated && !aliases.has(translated))
103
+ aliases.set(translated, entry);
104
+ }
105
+ return aliases;
106
+ }
107
+ /** Which of the walked tags a `relatedTags` mode will actually search today. */
108
+ function recallChannels(walked, seedKey, relatedMode) {
109
+ const seedTags = walked.filter((tag) => normalizeKey(tag.name) === seedKey);
110
+ // A space with no seed tag has nothing to protect: every remaining tag is the
111
+ // only channel there is, so the whole space is walked whatever the mode
112
+ // (pre-provenance fallback for a hand-written space).
113
+ if (seedTags.length === 0)
114
+ return [...walked];
115
+ if (relatedMode === 'never' || relatedMode === 'when_seed_insufficient')
116
+ return seedTags;
117
+ return [...seedTags, ...walked.filter((tag) => normalizeKey(tag.name) !== seedKey)];
118
+ }
17
119
  /**
18
120
  * Resolves a topic to a tag space, collects that day's works across the tags,
19
121
  * filters by lightweight metadata relevance and ranks by local popularity, then
@@ -42,7 +144,6 @@ class TopicPipeline {
42
144
  async selectWorks(target, contentType, day, limit, discovery, collect) {
43
145
  const topic = (target.topic ?? '').trim();
44
146
  const { space } = await this.resolver.resolve(topic, contentType, discovery);
45
- const tagScores = new Map(space.tags.map((t) => [this.key(t.name), t.score]));
46
147
  const maxPerTag = this.bound(collect.maxPerTag, COLLECT_DEFAULTS.maxPerTag, 5, 100);
47
148
  const maxCandidates = this.bound(collect.maxCandidates, COLLECT_DEFAULTS.maxCandidates, 20, 500);
48
149
  const minMetadataScore = collect.minMetadataScore ?? COLLECT_DEFAULTS.minMetadataScore;
@@ -55,11 +156,22 @@ class TopicPipeline {
55
156
  let rawCount = 0;
56
157
  let aiExcludedCount = 0;
57
158
  let duplicateRemovedCount = 0;
58
- const tagNames = space.tags.map((t) => t.name);
59
159
  const seedKey = this.key(topic);
160
+ // §tag-provenance: apply tagRelations where the walked list is built, so the
161
+ // filter governs seedTags/relatedTags BEFORE any search is issued. The seed
162
+ // tag survives allow/allowSources; a deny entry that empties the space falls
163
+ // back to the seed tag alone, exactly like the resolver's seedOnly().
164
+ const matchTranslatedNames = discovery.matchTranslatedNames === true;
165
+ const walked = selectWalkedTags(space.tags, seedKey, discovery.tagRelations);
166
+ const tagNames = walked.map((t) => t.name);
60
167
  const seedTags = tagNames.filter((name) => this.key(name) === seedKey);
61
168
  const relatedTags = tagNames.filter((name) => this.key(name) !== seedKey);
169
+ // Ranking reads the documented semantic weight, falling back to `score` for
170
+ // spaces persisted before provenance existed. The two are always equal.
171
+ const tagScores = buildTagAliases(walked, matchTranslatedNames);
62
172
  const searchedTags = [];
173
+ // A hard seed tier: 'on' always, and the seed-first modes keep their layer.
174
+ const hardSeedTier = discovery.seedTier === 'on' || relatedMode !== 'always';
63
175
  const collectTag = async (tag) => {
64
176
  // Cancellation is checked between tags, so a cancelled run stops issuing
65
177
  // new searches even when the aborted request itself had already returned.
@@ -100,11 +212,11 @@ class TopicPipeline {
100
212
  if (seedTags.length === 0 || relatedMode === 'always') {
101
213
  // No seed tag in the space (hand-written space): keep walking everything
102
214
  // rather than returning nothing.
103
- await runTags(seedTags.length === 0 ? tagNames : [...seedTags, ...relatedTags]);
215
+ await runTags(recallChannels(walked, seedKey, relatedMode).map((t) => t.name));
104
216
  }
105
217
  else {
106
218
  await runTags(seedTags);
107
- const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType).length;
219
+ const seedAccepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames).length;
108
220
  if (relatedMode === 'never') {
109
221
  logger_1.logger.info('[TopicRecall] mode=never tag=' + topic + ' day=' + day + ' accepted=' + seedAccepted);
110
222
  }
@@ -118,19 +230,19 @@ class TopicPipeline {
118
230
  }
119
231
  const dedupedCount = byId.size;
120
232
  logger_1.logger.info('[TopicCollector] type=' + contentType + ' raw=' + rawCount + ' deduplicated=' + dedupedCount + ' aiExcluded=' + aiExcludedCount + ' searchedTags=' + searchedTags.length);
121
- const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType);
122
- const chosen = this.topByPopularity(accepted, limit, relatedMode === 'always' ? undefined : seedKey);
233
+ const accepted = this.acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames);
234
+ const chosen = this.topByPopularity(accepted, limit, hardSeedTier ? seedKey : undefined);
123
235
  const selected = chosen.map((e) => e.candidate);
124
236
  logger_1.logger.info('[MetadataTopicFilter] accepted=' + accepted.length);
125
237
  if (selected[0]) {
126
- logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + ' title=' + selected[0].title);
238
+ logger_1.logger.info('[PopularityRanker] selected=' + selected[0].id + ' popularity=' + selected[0].popularity.toFixed(1) + ' meta=' + selected[0].metadataScore.toFixed(2) + (hardSeedTier ? ' seedTier=on' : '') + ' title=' + selected[0].title);
127
239
  }
128
240
  return {
129
241
  works: chosen.map((e) => e.work),
130
242
  selection: {
131
243
  candidates: [...byId.values()].map((e) => e.candidate),
132
244
  selected,
133
- resolvedTagCount: space.tags.length,
245
+ resolvedTagCount: walked.length,
134
246
  searchedTags,
135
247
  rawCount,
136
248
  dedupedCount,
@@ -163,12 +275,16 @@ class TopicPipeline {
163
275
  }
164
276
  toCandidate(work, type) {
165
277
  const popularity = (0, pixiv_utils_1.calculatePopularityScore)(work);
278
+ const translatedTags = (work.tags ?? [])
279
+ .map((t) => t.translated_name)
280
+ .filter((name) => Boolean(name));
166
281
  return {
167
282
  id: work.id,
168
283
  type,
169
284
  title: work.title ?? '',
170
285
  caption: work.caption ?? '',
171
286
  tags: (work.tags ?? []).map((t) => t.name).filter(Boolean),
287
+ ...(translatedTags.length > 0 ? { translatedTags } : {}),
172
288
  bookmarks: Number(work.total_bookmarks ?? work.bookmark_count ?? 0) || 0,
173
289
  views: Number(work.total_view ?? work.view_count ?? 0) || 0,
174
290
  popularity,
@@ -184,16 +300,23 @@ class TopicPipeline {
184
300
  * the seed pass can be evaluated before deciding whether the related channel
185
301
  * is needed at all (§topic-recall).
186
302
  */
187
- acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType) {
303
+ acceptedWorks(byId, seedKey, tagScores, minMetadataScore, limit, contentType, matchTranslatedNames) {
188
304
  const accepted = [];
305
+ const hitsSeed = (candidate) => {
306
+ if (candidate.tags.some((t) => this.key(t) === seedKey))
307
+ return true;
308
+ if (!matchTranslatedNames)
309
+ return false;
310
+ return (candidate.translatedTags ?? []).some((t) => this.key(t) === seedKey);
311
+ };
189
312
  for (const entry of byId.values()) {
190
- entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores);
313
+ entry.candidate.metadataScore = this.metadataScore(entry.candidate, seedKey, tagScores, matchTranslatedNames);
191
314
  if (entry.candidate.metadataScore >= minMetadataScore)
192
315
  accepted.push(entry);
193
316
  }
194
317
  if (accepted.length === 0 && byId.size > 0) {
195
318
  const fallback = [...byId.values()]
196
- .filter((e) => e.candidate.tags.some((t) => this.key(t) === seedKey))
319
+ .filter((e) => hitsSeed(e.candidate))
197
320
  .sort((a, b) => b.candidate.popularity - a.candidate.popularity);
198
321
  accepted.push(...fallback.slice(0, Math.max(limit, 1)));
199
322
  logger_1.logger.warn('[MetadataTopicFilter] type=' + contentType + ' none above threshold ' + minMetadataScore + '; kept ' + accepted.length + ' seed-tag fallback');
@@ -204,25 +327,49 @@ class TopicPipeline {
204
327
  * Lightweight metadata relevance. Tags dominate (Pixiv's own taxonomy);
205
328
  * title/caption add smaller boosts. The seed tag is strong evidence.
206
329
  * No text model — case/symbol-insensitive substring matching only.
330
+ *
331
+ * `matchTranslatedNames` (default false) additionally counts a work's
332
+ * `translated_name` as a hit for the resolved tag it translates to, so a
333
+ * topic written in one language can still match works tagged in another. The
334
+ * translated name is matched against the resolved space — never added to
335
+ * `candidate.tags` — so the work is not reported as carrying a tag it lacks.
336
+ *
337
+ * A resolved tag counts at most ONCE per work, whether it matched through the
338
+ * tag name or through its translation (and a work that repeats a tag, as Pixiv
339
+ * payloads do for a tag plus its translation, does not double its weight).
207
340
  */
208
- metadataScore(candidate, seedKey, tagScores) {
341
+ metadataScore(candidate, seedKey, tagScores, matchTranslatedNames) {
209
342
  let seedHit = false;
210
343
  let relatedSum = 0;
211
344
  let relatedHits = 0;
212
- for (const tag of candidate.tags) {
213
- const k = this.key(tag);
214
- if (STOP_TAGS.has(k))
215
- continue;
216
- if (k === seedKey) {
345
+ let strongRelated = false;
346
+ // A resolved tag is matched through its own name, and — when the caller
347
+ // opted in — through the work-side translation of that tag. Either spelling
348
+ // maps back to the SAME resolved tag, so a work carrying both spellings
349
+ // counts once and never gets a double weight.
350
+ const countedTagKeys = new Set();
351
+ const consider = (candidateKey) => {
352
+ if (candidateKey === '' || STOP_TAGS.has(candidateKey))
353
+ return;
354
+ if (candidateKey === seedKey) {
217
355
  seedHit = true;
218
- continue;
219
- }
220
- const related = tagScores.get(k);
221
- if (related !== undefined) {
222
- relatedSum += Math.min(related, 0.6);
223
- relatedHits += 1;
356
+ return;
224
357
  }
358
+ const aliases = tagScores.get(candidateKey);
359
+ if (!aliases || countedTagKeys.has(aliases.primary))
360
+ return;
361
+ countedTagKeys.add(aliases.primary);
362
+ relatedSum += Math.min(aliases.weight, 0.6);
363
+ relatedHits += 1;
364
+ if (aliases.weight >= 0.6)
365
+ strongRelated = true;
366
+ };
367
+ if (matchTranslatedNames) {
368
+ for (const tag of candidate.translatedTags ?? [])
369
+ consider(this.key(tag));
225
370
  }
371
+ for (const tag of candidate.tags)
372
+ consider(this.key(tag));
226
373
  const hayTitle = this.normalize(candidate.title);
227
374
  const hayCaption = this.normalize(candidate.caption);
228
375
  const titleSeed = !!seedKey && hayTitle.includes(seedKey);
@@ -239,7 +386,6 @@ class TopicPipeline {
239
386
  // deliberately does NOT clear the bar, so a hugely popular tangential work
240
387
  // cannot crowd out core topic works.
241
388
  let score = 0;
242
- const strongRelated = [...tagScores.entries()].some(([k, w]) => w >= 0.6 && candidate.tags.some((t) => this.key(t) === k));
243
389
  if (titleSeed)
244
390
  score += 0.8;
245
391
  if (captionSeed)
@@ -256,12 +402,13 @@ class TopicPipeline {
256
402
  * alone. A work with a higher metadata score does NOT outrank a more popular
257
403
  * accepted work.
258
404
  *
259
- * `seedKey` adds a single tier in front of that popularity order and is only
260
- * passed by the seed-first recall modes (§topic-recall): when related tags
261
- * were reached as a fallback, a work that actually carries the topic tag must
262
- * outrank a related-only work, and popularity decides within each tier. The
263
- * default mode passes no `seedKey`, so the documented popularity-only ranking
264
- * is unchanged.
405
+ * `seedKey` adds a single tier in front of that popularity order and is
406
+ * passed by the seed-first recall modes (§topic-recall) and by
407
+ * `topicDiscovery.seedTier: 'on'` (§tag-provenance): a work that actually
408
+ * carries the topic tag outranks a related-only work, however popular the
409
+ * latter is, and popularity decides within each tier. With `seedTier: 'off'`
410
+ * and the default `relatedTags: 'always'` no `seedKey` is passed, so the
411
+ * documented popularity-only ranking is unchanged.
265
412
  */
266
413
  popCompare(a, b) {
267
414
  return b.popularity - a.popularity;
@@ -97,6 +97,9 @@ class TopicResolver {
97
97
  name: seed,
98
98
  translatedName: suggested.find((t) => this.scorer.key(t.name) === this.scorer.key(seed))?.translated_name,
99
99
  score: 1,
100
+ weight: 1,
101
+ // The tag the operator asked for: the strongest provenance, never a hint.
102
+ source: 'seed',
100
103
  occurrences: topicWorks.length,
101
104
  coverage: 1,
102
105
  specificity: 1,
@@ -168,7 +171,7 @@ class TopicResolver {
168
171
  expiresAt: new Date(now + cacheDays * 24 * 60 * 60_000).toISOString(),
169
172
  sampleSize: 0,
170
173
  sampledWorks: 0,
171
- tags: [{ name: seed, score: 1, occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
174
+ tags: [{ name: seed, score: 1, weight: 1, source: 'seed', occurrences: 0, coverage: 1, specificity: 1, suggested: false, seed: true }],
172
175
  };
173
176
  }
174
177
  ageDays(space) {
@@ -99,10 +99,17 @@ class TopicTagScorer {
99
99
  const suggestionWeight = stat.suggested ? 1.1 : 1.0;
100
100
  const genericPenalty = GENERIC_TAG_PENALTY.has(k) ? 0.4 : 1.0;
101
101
  const raw = recall * specificity * suggestionWeight * genericPenalty;
102
+ const score = Number(raw.toFixed(4));
102
103
  resolved.push({
103
104
  name: stat.name,
104
105
  translatedName: stat.translatedName,
105
- score: Number(raw.toFixed(4)),
106
+ score,
107
+ // `weight` is the same semantic number; ranking and diagnostics read it
108
+ // while `score` stays for backward compatibility (§tag-provenance).
109
+ weight: score,
110
+ // Both channels may agree on a tag; the combined provenance records
111
+ // that, which is strictly more informative than either alone.
112
+ source: stat.suggested ? 'cooccurrence+autocomplete' : 'cooccurrence',
106
113
  occurrences: stat.topicDocs,
107
114
  coverage: Number(coverage.toFixed(4)),
108
115
  specificity: Number(specificity.toFixed(4)),
@@ -127,6 +134,10 @@ class TopicTagScorer {
127
134
  name: name,
128
135
  translatedName: sug.translated_name?.trim() || undefined,
129
136
  score: AUTOCOMPLETE_ONLY_SCORE,
137
+ weight: AUTOCOMPLETE_ONLY_SCORE,
138
+ // Pixiv autocomplete is the only evidence for this tag: it never
139
+ // co-occurred in the bounded sample.
140
+ source: 'autocomplete',
130
141
  occurrences: 0,
131
142
  coverage: 0,
132
143
  specificity: 1.0, // Pixiv-endorsed related; treated as specific but unobserved
@@ -4,12 +4,34 @@
4
4
  * local models are used anywhere in this module.
5
5
  */
6
6
  export type TopicContentType = 'illustration' | 'novel';
7
+ /**
8
+ * Where a resolved tag came from. Provenance is what lets a caller treat a weak
9
+ * expansion differently from the topic it was asked for (§tag-provenance):
10
+ *
11
+ * - `'seed'`: the tag the operator asked for. Always the strongest key.
12
+ * - `'cooccurrence'`: sampled together with the seed, but Pixiv autocomplete
13
+ * does not relate it to the seed. Co-occurrence evidence only.
14
+ * - `'autocomplete'`: Pixiv autocomplete relates it to the seed, but it never
15
+ * appeared in the sample. No co-occurrence evidence.
16
+ * - `'cooccurrence+autocomplete'`: both channels agree — the strongest related
17
+ * provenance available.
18
+ */
19
+ export type TagSource = 'seed' | 'cooccurrence' | 'autocomplete' | 'cooccurrence+autocomplete';
7
20
  /** A single related tag with a 0..1 relatedness score and provenance. */
8
21
  export interface ResolvedTag {
9
22
  name: string;
10
23
  translatedName?: string;
11
24
  /** Combined relatedness score (co-occurrence * specificity * suggestion). */
12
25
  score: number;
26
+ /**
27
+ * Semantic weight used for ranking, filtering and diagnostics. Always the
28
+ * same number as `score`; kept as a separate, documented field so ranking can
29
+ * be explained (and, later, adjusted) without redefining `score`.
30
+ * Optional: spaces persisted before provenance existed have neither field.
31
+ */
32
+ weight?: number;
33
+ /** Provenance of the tag. Optional for the same reason as `weight`. */
34
+ source?: TagSource;
13
35
  /** How many sampled works (of the seed search) carried this tag. */
14
36
  occurrences: number;
15
37
  /** Coverage of the sampled seed works (occurrences / sample size). */
@@ -57,6 +79,26 @@ export interface TopicDiscoveryOptions {
57
79
  * `'never'` searches the seed tag alone.
58
80
  */
59
81
  relatedTags?: RelatedTagMode;
82
+ /**
83
+ * Which resolved tags may become recall channels (§tag-provenance). Deny wins
84
+ * over allow; the seed tag is never dropped by `allow`/`allowSources`.
85
+ */
86
+ tagRelations?: TopicRelationsOptions;
87
+ /** Make the seed tag a hard ranking tier (default `'off'`). */
88
+ seedTier?: 'off' | 'on';
89
+ /** Count a work's translated tag names as tag hits (default false). */
90
+ matchTranslatedNames?: boolean;
91
+ }
92
+ /**
93
+ * Runtime shape of `TopicDiscoveryConfig.tagRelations`, declared here so the
94
+ * topic module does not depend on the config layer (the topic pipeline is also
95
+ * driven by hand-written targets in tests and by callers that never load a
96
+ * config file).
97
+ */
98
+ export interface TopicRelationsOptions {
99
+ allowSources?: TagSource[];
100
+ allow?: string[];
101
+ deny?: string[];
60
102
  }
61
103
  export interface TopicCollectOptions {
62
104
  maxPerTag?: number;
@@ -76,6 +118,12 @@ export interface TopicCandidate {
76
118
  popularity: number;
77
119
  /** Metadata topic-relevance score computed by the filter stage. */
78
120
  metadataScore: number;
121
+ /**
122
+ * Translated tag names carried by the work, kept beside `tags` so
123
+ * `matchTranslatedNames` can compare them WITHOUT claiming the work itself
124
+ * carries a tag it does not (§tag-provenance). Default-off.
125
+ */
126
+ translatedTags?: string[];
79
127
  /** Pixiv AI classification copied from illustration search metadata. */
80
128
  aiType?: number;
81
129
  }
package/dist/version.js CHANGED
@@ -2,5 +2,5 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.BUILD = void 0;
4
4
  // GENERATED by scripts/write-version.js — do not edit manually.
5
- exports.BUILD = { version: '3.0.3', commit: 'b01a93d49e63' };
5
+ exports.BUILD = { version: '3.2.0', commit: 'c195b909063c' };
6
6
  //# sourceMappingURL=version.js.map
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "type": "commonjs",
3
3
  "name": "pixivflow-webui-backend",
4
- "version": "3.0.3",
4
+ "version": "3.2.0",
5
5
  "description": "PixivFlow WebUI Backend - CommonJS module"
6
6
  }
@@ -85,6 +85,10 @@ npx jest src/__tests__/delivery/gateway-reference-e2e.test.ts
85
85
  重启后带同一个 `idempotencyKey` 收敛;一条坏路由的失败不会影响另一条。写自己的网关时,
86
86
  让这个文件继续通过就是「你接对了」的最强证据。
87
87
 
88
+ 接平台那一步请用 [`examples/onebot-adapter/`](../onebot-adapter/README.md):它是本节所述
89
+ 「最小转换进程」的可运行实例(QQ / OneBot v11),同样零依赖、同样不实现 QQ 协议。本参考网关
90
+ 与它是**互补**关系:一个证明「契约本身通不通」,另一个证明「平台映射写对了没有」。
91
+
88
92
  ## 它有意不做什么
89
93
 
90
94
  - 不保存任何东西到磁盘(去重表在内存里,进程重启即丢)—— 生产网关必须持久化,
@@ -0,0 +1,139 @@
1
+ # Example OneBot v11 adapter(QQ)
2
+
3
+ 一个**零依赖**的 OneBot v11 投递适配器,把 [Gateway Contract v1](../../docs/GATEWAY_CONTRACT.md)
4
+ 的三个端点翻译成 OneBot v11 的 HTTP API 调用。它是 [docs/GATEWAY.md](../../docs/GATEWAY.md) §5.3
5
+ 所描述的「最小转换进程」的**可运行实例**。
6
+
7
+ ```
8
+ PixivFlow ──POST /deliver(契约 v1,带签名/幂等键)──▶ 本进程
9
+ │ POST /send_group_msg
10
+ ▼
11
+ NapCat / Lagrange / LLOneBot ──▶ QQ
12
+ ```
13
+
14
+ ## 它不是什么
15
+
16
+ - **不实现 QQ 协议**,也不实现 OneBot 本身:QQ 会话属于你已经在跑的 OneBot 实现。
17
+ - **不做扫码登录**:二维码由 NapCat 自己的面板显示,凭据不会经过本进程,也不会经过 PixivFlow
18
+ (见 GATEWAY.md §5.1/§5.2)。
19
+ - **不做重试**:重试由 PixivFlow 的 outbox 负责;本进程只负责把一次请求翻译成一次 OneBot 调用,
20
+ 并把「成功 / 待定 / 永久失败」如实翻译回契约词汇。
21
+ - **不按平台长分支**:这就是适配器独立成进程的原因 —— PixivFlow 侧的契约里没有 QQ。
22
+
23
+ 只想先验证「契约本身通不通」、还不想碰 QQ,请先用
24
+ [`examples/gateway/`](../gateway/README.md)(它只打印消息,不接平台)。
25
+
26
+ ## 跑起来
27
+
28
+ ```bash
29
+ # 1. 先让 NapCat(或其它 OneBot v11 实现)的 HTTP API 在 3000 端口可用,并记下它的 token
30
+ # 2. 起适配器
31
+ ONEBOT_URL=http://127.0.0.1:3000 \
32
+ ONEBOT_TOKEN=<napcat-token> \
33
+ ONEBOT_TARGET=group:987654 \
34
+ ADAPTER_TOKEN=<给 PixivFlow 用的 token> \
35
+ node examples/onebot-adapter/server.mjs
36
+
37
+ # 自检:起一个假 OneBot,跑完三个端点与去重逻辑,打印结果并退出(0 = 通过)
38
+ node examples/onebot-adapter/server.mjs --selftest
39
+ ```
40
+
41
+ 启动时会拒绝「半配置」:`ONEBOT_URL` 缺失或 `ONEBOT_TARGET` 还是占位值 `group:0` 时,
42
+ 进程会打印 `config.problem` 并以退出码 2 结束 —— 发错群比不启动更糟。
43
+
44
+ ## 环境变量
45
+
46
+ | 变量 | 作用 |
47
+ | --- | --- |
48
+ | `PORT` | 监听端口(默认 `8791`) |
49
+ | `HOST` | 绑定地址(默认 `127.0.0.1`) |
50
+ | `ADAPTER_TOKEN` | **给 PixivFlow 用的** token,校验 `Authorization: Bearer <token>`;未设置只告警(端点无鉴权) |
51
+ | `ADAPTER_SECRET` | 设置后强制校验 `X-Webhook-Signature`(对**原始字节**做 HMAC,5 分钟时间窗) |
52
+ | `ONEBOT_URL` | OneBot HTTP API 基址,如 `http://127.0.0.1:3000` |
53
+ | `ONEBOT_TOKEN` | **给 OneBot 用的** token(与 `ADAPTER_TOKEN` 不要复用同一个值) |
54
+ | `ONEBOT_TARGET` | 投递目标:`group:987654`(默认)或 `private:987654` |
55
+ | `ONEBOT_TIMEOUT_MS` | 单次 OneBot 调用超时(默认 `15000`) |
56
+ | `ONEBOT_MIN_SEND_INTERVAL_MS` | 两次 OneBot 调用之间的最小间隔(默认 `500`),用于限速 |
57
+ | `ONEBOT_STATE_FILE` | 幂等账本(默认 `./.onebot-adapter-state.jsonl`,追加写 JSONL) |
58
+
59
+ ## 端点
60
+
61
+ | 端点 | 行为 |
62
+ | --- | --- |
63
+ | `POST /deliver` | 验签 → 校验 Bearer → 解析 JSON → 校验 `schemaVersion` → 按 `idempotencyKey` 去重 → 发送消息段 → 上传附件 → 回契约 ACK 词 |
64
+ | `GET /pairing` | 调 `get_login_info`:成功 `{status:"connected", account, nickname}`;可达但未登录 `{status:"waiting", reason}`(HTTP 200);不可达 HTTP 503 `{status:"unreachable"}` |
65
+ | `GET /health` | 调 `get_status`:`data.online !== false && data.good === true` 时 `{status:"connected", contractVersion:1, gateway:"pixivflow-onebot-adapter", onebot:{…}}`,否则 HTTP 503。**供运维用**,PixivFlow 的投递链路从不调用它 |
66
+
67
+ 未知路由回 `404 {status:"invalid"}`。
68
+
69
+ ## 翻译规则(这是整个文件的重点)
70
+
71
+ **消息段**(契约 `message.parts` → OneBot `message` 数组,顺序保留):
72
+
73
+ | 契约 part | OneBot 段 |
74
+ | --- | --- |
75
+ | `{kind:"text"}` | `{type:"text", data:{text}}` —— 文案取自 `message.text`(契约里正文只出现在 `message.text`,`parts` 里只是一个位置标记),只消费一次 |
76
+ | `{kind:"image", media}` | `{type:"image", data:{file}}` |
77
+ | `{kind:"video", media}` | `{type:"video", data:{file}}` |
78
+ | `{kind:"album"}` | PixivFlow 会展开成 N 个各自带 `media` 的 part,因此这里就是 N 个 `image`/`video` 段 |
79
+ | `{kind:"file", media}` | **不是消息段**:走 `upload_group_file {group_id, file, name}`,再补发一条 `📎 附件:<name>` 提示消息 |
80
+
81
+ `media.file` 的取值:`base64://<...>`(`base64` 传输,永远可用)或 `file://<绝对路径>`
82
+ (`reference` 传输,要求 OneBot 能读到 PixivFlow 的磁盘 —— 这正是该传输的取舍,不做静默降级)。
83
+
84
+ **ACK 映射**(OneBot 的 HTTP 状态码几乎永远是 200,成败在 `retcode`;契约的规则是「先看词,再看码」):
85
+
86
+ | OneBot 回答 | 本适配器回给 PixivFlow | 结果 |
87
+ | --- | --- | --- |
88
+ | `retcode 0` | `200 {status:"accepted", id:<message_id>}` | `delivered`,`remote_id` 就是平台消息号 |
89
+ | `status:"async"` 或 `retcode 1` | `200 {status:"pending", reason}` | 保持待投递,**绝不报成功** |
90
+ | `retcode 100/102/103/104/105/1400/1404` | `200 {status:"failed", reason}` | 永久失败,进死信,不无限重试 |
91
+ | 未知 `retcode` | `502 {reason}`(**无状态词**) | 可重试,不猜 |
92
+ | 网络不可达 / 超时 / HTTP ≥ 500 | `502 {reason}`(无状态词) | 可重试 |
93
+ | HTTP `401/403/404` | 原样 4xx(无状态词) | 令牌/地址写错,修好即可重试 |
94
+ | HTTP `429` | `429`(无状态词) | 限速,可重试 |
95
+ | 非 JSON 响应体 | `502 {reason}`(无状态词) | 可重试 |
96
+
97
+ **「无状态词」是刻意的**:契约里状态词优先于 HTTP 码,一旦回了 `failed` 就会进死信;
98
+ 所以凡是「请求本身有问题、但改配置后能成功」的情形,都只回一个裸 HTTP 码,让 PixivFlow 继续重试。
99
+
100
+ ## PixivFlow 侧配置
101
+
102
+ ```json
103
+ {
104
+ "delivery": {
105
+ "targets": {
106
+ "qq-main": {
107
+ "type": "webhook",
108
+ "url": "http://127.0.0.1:8791/deliver",
109
+ "token": "${QQ_ADAPTER_TOKEN}",
110
+ "pairingUrl": "http://127.0.0.1:8791/pairing",
111
+ "capabilities": {
112
+ "maxTextLength": 4000,
113
+ "maxAttachmentsPerMessage": 9,
114
+ "album": true,
115
+ "albumMin": 2,
116
+ "albumMax": 9,
117
+ "minSendIntervalMs": 500
118
+ }
119
+ }
120
+ }
121
+ }
122
+ }
123
+ ```
124
+
125
+ 跨机部署时把 `url`/`pairingUrl` 换成适配器所在主机的地址;若两者不在同一台机器上,
126
+ `reference` 传输的本地路径对端读不到,请改用 `base64`
127
+ (见 GATEWAY.md §6「`reference` 传输的文件可见性」)。
128
+
129
+ ## 验证
130
+
131
+ | 层 | 命令/动作 | 证明的是什么 |
132
+ | --- | --- | --- |
133
+ | 适配器自身 | `node examples/onebot-adapter/server.mjs --selftest` | 段构造、ACK 映射、去重、`/pairing`、`/health`(假 OneBot,不接 QQ) |
134
+ | 契约到底 | `npx jest src/__tests__/delivery/onebot-adapter-e2e.test.ts` | **真**投递运行时 → outbox → 适配器进程 → OneBot HTTP:`remote_id`、`pending` 不落地、`failed` 进死信、重放 `duplicate_existing` |
135
+ | QQ 登录态 | NapCat 自己的面板 | 账号在线;**不证明** PixivFlow 能投递 |
136
+ | 真实投递 | 跑一次下载 + `pixivflow delivery status` | 账本上的 `delivered` 与群里的那条消息 |
137
+
138
+ `--selftest` 与上面的 jest 套件都不需要 QQ、不需要 NapCat:它们证明的是**翻译**正确,
139
+ 不是「QQ 已经通了」。