flowocr 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. flowocr/__init__.py +15 -0
  2. flowocr/analyze/__init__.py +8 -0
  3. flowocr/analyze/align.py +151 -0
  4. flowocr/analyze/build_tracks.py +2886 -0
  5. flowocr/analyze/cluster_layers.py +155 -0
  6. flowocr/analyze/game_align.py +652 -0
  7. flowocr/analyze/gamescript.py +1220 -0
  8. flowocr/analyze/gtdbundle.py +306 -0
  9. flowocr/analyze/match.py +67 -0
  10. flowocr/analyze/matchers/__init__.py +5 -0
  11. flowocr/analyze/matchers/gametext.py +70 -0
  12. flowocr/analyze/merge_nameplate.py +178 -0
  13. flowocr/analyze/models/slot_pair.json +148 -0
  14. flowocr/analyze/nameplate.py +148 -0
  15. flowocr/analyze/pair_features.py +168 -0
  16. flowocr/analyze/pair_model.py +127 -0
  17. flowocr/analyze/refine_boundaries.py +409 -0
  18. flowocr/analyze/script_align.py +641 -0
  19. flowocr/analyze/scriptmatch.py +1018 -0
  20. flowocr/analyze/slot_learned.py +306 -0
  21. flowocr/analyze/slot_lines.py +251 -0
  22. flowocr/analyze/slot_modes.py +143 -0
  23. flowocr/analyze/slot_pairs.py +562 -0
  24. flowocr/analyze/slot_veto.py +104 -0
  25. flowocr/analyze/uigate.py +351 -0
  26. flowocr/artifacts/__init__.py +5 -0
  27. flowocr/artifacts/evalkit.py +123 -0
  28. flowocr/artifacts/matchedio.py +75 -0
  29. flowocr/artifacts/srtio.py +166 -0
  30. flowocr/artifacts/tracksio.py +345 -0
  31. flowocr/extensions.py +77 -0
  32. flowocr/extract/__init__.py +8 -0
  33. flowocr/extract/childproc.py +118 -0
  34. flowocr/extract/decode_proc.py +511 -0
  35. flowocr/extract/decode_shards.py +574 -0
  36. flowocr/extract/detpost.py +215 -0
  37. flowocr/extract/edge_proc.py +314 -0
  38. flowocr/extract/edge_refine.py +598 -0
  39. flowocr/extract/fast_det.py +214 -0
  40. flowocr/extract/ffcheck.py +87 -0
  41. flowocr/extract/framegrid.py +265 -0
  42. flowocr/extract/framesource.py +1264 -0
  43. flowocr/extract/ocr_args.py +754 -0
  44. flowocr/extract/ocr_complete.py +231 -0
  45. flowocr/extract/ocr_parallel.py +208 -0
  46. flowocr/extract/ort_server.py +632 -0
  47. flowocr/extract/ortclient.py +363 -0
  48. flowocr/extract/ptsclock.py +150 -0
  49. flowocr/extract/recdecode.py +78 -0
  50. flowocr/extract/recort.py +162 -0
  51. flowocr/extract/recpack.py +94 -0
  52. flowocr/extract/recpool.py +206 -0
  53. flowocr/extract/recprep.py +71 -0
  54. flowocr/extract/refine_video.py +271 -0
  55. flowocr/extract/regions.py +387 -0
  56. flowocr/extract/reuse_v2.py +593 -0
  57. flowocr/extract/run_groups.py +247 -0
  58. flowocr/extract/run_ocr2.py +1605 -0
  59. flowocr/extract/supervisor.py +261 -0
  60. flowocr/extract/timeline.py +84 -0
  61. flowocr/extract/typewriter_fuse.py +394 -0
  62. flowocr/models.py +263 -0
  63. flowocr/output/__init__.py +3 -0
  64. flowocr/output/export.py +205 -0
  65. flowocr/output/layout.py +210 -0
  66. flowocr/output/presets/__init__.py +6 -0
  67. flowocr/output/presets/_overlay.py +40 -0
  68. flowocr/output/presets/default.py +21 -0
  69. flowocr/output/presets/default_all.py +20 -0
  70. flowocr/output/presets/dev.py +21 -0
  71. flowocr/output/presets/matched_srt.py +32 -0
  72. flowocr/output/presets/script.py +20 -0
  73. flowocr/output/presets/srt_main.py +33 -0
  74. flowocr/output/render.py +82 -0
  75. flowocr/output/run_srt.py +83 -0
  76. flowocr/output/script.py +541 -0
  77. flowocr/paths.py +168 -0
  78. flowocr/provenance.py +119 -0
  79. flowocr/typeset/__init__.py +9 -0
  80. flowocr/typeset/__main__.py +38 -0
  81. flowocr/typeset/assfile.py +216 -0
  82. flowocr/typeset/core.py +821 -0
  83. flowocr/typeset/fx/__init__.py +11 -0
  84. flowocr-0.1.0.dist-info/METADATA +109 -0
  85. flowocr-0.1.0.dist-info/RECORD +91 -0
  86. flowocr-0.1.0.dist-info/WHEEL +5 -0
  87. flowocr-0.1.0.dist-info/entry_points.txt +10 -0
  88. flowocr-0.1.0.dist-info/licenses/LICENSE +674 -0
  89. flowocr-0.1.0.dist-info/licenses/LICENSES/Apache-2.0.txt +201 -0
  90. flowocr-0.1.0.dist-info/licenses/LICENSES/PP-OCRv6-NOTICE.md +18 -0
  91. flowocr-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1220 @@
1
+ """游戏文本包(`gtd-bundle/1`,`flowocr.analyze.gtdbundle`)-> 一段视频的「剧本」。
2
+
3
+ 存在的理由:游戏直播那批素材(`data/index.md` F 节)一直**没有原文剧本**,用途 2 的尺
4
+ (命中 / 三档漏条 / 重复认领)在那批上一个都用不了(gamestream-first-look 报告)。
5
+ 文本包(`../game-text-data` 的 `gametext <游戏> bundle` 出的,一款游戏一个)收了原神 / 星铁 / 绝区零三款的解包文本,但它是**整个游戏**
6
+ (原神 34.5 万个对话节点),不是"这段视频播了什么"。所以要先圈出这段视频打过的对话,
7
+ 这就是本工具做的事;量轨在 `flowocr.analyze.game_align`。
8
+
9
+ ## 圈对话单元:只用 obs,不用任何一条臂的轨
10
+
11
+ 单元 = 一段连续对话(原神 talk / 星铁一个 script / 绝区零一个 scene;原神语音字幕一条链一个);
12
+ 另有两类库里没有结构的:星铁全量台词表里没人引用的台词按 id 百位块成组,绝区零散条目按 key 里的
13
+ 事件号成组(字幕带上的 `CN` / `TL` 按连号成组,`band_loose_slot`)、否则一条一个。
14
+ 拿 obs 里**全部**的 OCR 文本(每个框、每一帧,去重)去整库做 n-gram 检索 + 包含度校验,
15
+ 锚住的单元就是这段视频打过的对话;对齐顺序按锚点时间排到**行一级**(`sequence`),
16
+ 因为一个 talk 可以拖一个小时(玩家中途去做别的)。
17
+
18
+ **为什么用 obs 而不是主轨**:`build_tracks` 的所有 A/B 臂共用同一份 obs,
19
+ 从 obs 圈出来的分母对各臂**一模一样**;从某条臂的主轨圈,分母就偏向那条臂。
20
+
21
+ **这一步有选择偏差,报数时要连它一起报**:一段对话若在**任何**区域、任何一帧都没被读到,
22
+ 它就不进分母——这种"整段漏掉"量不出来。它和 `script_align --span-ref`(由参照字幕圈区间)
23
+ 是同一类代价;好在圈的判据是**全部区域的 OCR**,而被量的是**主轨**,两者差一层。
24
+
25
+ ## 分支:用图结构判"没走到",别让 gap 分桶去猜
26
+
27
+ 原神 1.4 万个分支点,子节点 99% 是玩家选项(`TALK_ROLE_PLAYER`);星铁选项之后是
28
+ `wait TalkSentence_<选项>` 引出的一段。选项文本走**选项 UI**,不进字幕带(类比《魔裁》的
29
+ `@choice`),记成 `Choice`;而**没选的那一支的 NPC 回复**如果留在分母里,会以 1–3 条的
30
+ 短 gap 出现,全被算成"疑似真漏"。`script_align` 的 gap 分桶(≥5 条才算分支)接不住它。
31
+
32
+ 所以按图算:分支点的每个子节点取**独占可达集**(只从它能到、兄弟到不了的节点),
33
+ 独占集里有 obs 锚点 = 走过;同组有兄弟走过而它没有 = `unwalked`;同组谁都没锚点 =
34
+ `ambiguous`(多半是回复都很短、锚不住)。两类都**排出分母、单独报数**。
35
+ 这同样依赖 obs,所以同样是各臂共用的分母。绝区零上游没有剧情分支(它自己的 README
36
+ "上游边界"),两位主角的两种说法在库里是相邻两行,按相邻相似度连成**变体组**、一组只算一个
37
+ 分母条目(`load_zzz` / `fold_variants`);它的对话选项库里判不出来,靠下面的位置判据。
38
+
39
+ ## 字幕带:用位置判库里没有结构的
40
+
41
+ 用途 2 的分母是**字幕带**上的条目。库的结构能判的(选项、黑屏旁白、横幅)上面已经排掉;判不了的
42
+ (绝区零的选项按钮、居中旁白、头顶气泡——上游既没分支也没说话人)靠 obs 框的位置:先从锚住的行
43
+ 估出这段视频的字幕带(`band_of`,众数窗口),读到过而框**全在带外**的行记 `OffBand`(`mark_offband`)。
44
+ 同样只用 obs、各臂共用。2026-09-11 整场绝区零的未命中里 208 条是这一类。
45
+
46
+ 短行(内容不到 `--qmin` 字)本来永远没有锚点、也就没有位置证据;**单元圈中之后候选只剩几十行**,
47
+ 于是在本单元的时段内按逐字相等补一遍(`short_anchors`),整场绝区零又有 44 条选项按钮因此判出带外。
48
+
49
+ ## 文本清洗(只影响匹配,不改语料)
50
+
51
+ `{M#…}{F#…}` 按视频的主角性别二选一(**从 obs 投票定**,两种写法各算一遍、看哪种读得上);
52
+ `{NICKNAME}` / `{REALNAME[…]}` / `{PLAYERAVATAR#…}` 这类玩家相关的占位删掉(屏幕上是
53
+ 玩家起的名字,库里没有);`<color>` 等标签删掉;`{RUBY#[S]…}` 是注音(小字在正文上方),
54
+ 删掉;`\\n` 是条内换行,写成 `\\N`(和《魔裁》剧本同一口径,`script_align` 认它)。
55
+
56
+ 用法:
57
+ python -m flowocr.analyze.gamescript out/gamestream/gi-s2.jsonl --out out/gametext/gi-s2-ref.json
58
+ # 游戏按文件名前缀猜(gi/hsr/zzz),猜不出就要 --game;包默认在 paths.gametext_root(),--gametext 可指
59
+ """
60
+ from __future__ import annotations
61
+
62
+ import argparse
63
+ import json
64
+ import re
65
+ import sys
66
+ import time
67
+ import unicodedata
68
+ from bisect import bisect_right
69
+ from collections import Counter, defaultdict
70
+ from dataclasses import dataclass, field
71
+ from difflib import SequenceMatcher
72
+ from itertools import zip_longest
73
+ from pathlib import Path
74
+
75
+ import numpy as np
76
+
77
+ from flowocr import paths # noqa: E402
78
+ from flowocr.analyze import gtdbundle # noqa: E402
79
+
80
+ CODE_FP_FILES = ("src/flowocr/analyze/gamescript.py", "src/flowocr/analyze/gtdbundle.py", "src/flowocr/paths.py")
81
+ """产剧本的代码(本文件 + 它 import 的本地模块,仓库相对路径;`paths` 是 shim,记真实现),
82
+ `build_tracks.code_fp` 按内容算指纹。"""
83
+ SCHEMA = "flowocr-gameref/4"
84
+ """/2(2026-09-11,audit-6):锚点带 `unique`、绝区零变体组折成 `Variant` + `variant_of`、
85
+ provenance 带 `integrity_warnings`。/3(同日):锚点带框位置 `y` / `in_band`、新类型 `OffBand`、
86
+ `stats.band`。/4(2026-09-25,发布前):库改成读文本包,provenance 的 `db` / `upstream` / `extractor` /
87
+ `integrity_warnings` 合成一个 `gametext`(带内容指纹,`gtdbundle.Bundle.provenance`)。
88
+ 旧剧本不做兼容,`game_align` 见到就让你重跑。
89
+ /4 之内加的**可选**顶层 `terms`(2026-09-26,`term_table`):文本包有短名词表时才有,读的一方缺了照常工作,所以没升版本。"""
90
+
91
+ GAMES = {
92
+ # 游戏 -> {loader 要读的表(文本包清单里的 `table`): 认得的内容契约主版本}。语种按标准码取 `ja` / `zh-Hans`,
93
+ # 文件后缀由清单给。主版本对不上的包 `gtdbundle.Bundle.require` 拒绝:那是字段含义变了,这里的读法要先跟上。
94
+ "genshin": {"quests": 1, "reminders": 1},
95
+ "starrail": {"missions": 1, "dialogues": 1, "sentences": 1},
96
+ "zzz": {"scenes": 1, "loose": 1},
97
+ }
98
+ KNOWN_VALUES = {
99
+ # 据以分类的枚举字段 -> 这里认得的取值(抄自 game-text-data 那边契约声明在上面主版本时的取值)。
100
+ # 上游加一种取值不改契约主版本(开放枚举),所以读库时把没见过的计数、打一行警告(`warn_unseen`):
101
+ # 新值照现有规则归类(style 不是 Normal 一律不进分母),要不要改归类人看过再定,定了加进这里。
102
+ "genshin.quests.dialog[].role.type": frozenset({
103
+ "TALK_ROLE_NPC", "TALK_ROLE_PLAYER", "TALK_ROLE_BLACK_SCREEN", "TALK_ROLE_NONE",
104
+ "TALK_ROLE_NEED_CLICK_BLACK_SCREEN", "TALK_ROLE_MATE_AVATAR", "TALK_ROLE_GADGET",
105
+ "TALK_ROLE_CONSEQUENT_NEED_CLICK_BLACK_SCREEN", "TALK_ROLE_CONSEQUENT_BLACK_SCREEN"}),
106
+ "genshin.reminders.lines[].style": frozenset({
107
+ "Normal", "Banner", "CombatChat", "MoonInfoReminder", "DialogueWithPortrait", "WhiteMessage",
108
+ "SpecialReminderOnTop", "TpsGlassesChat", "PenumbraStory", "EventPromptDown", "AbyssWarReminder",
109
+ "SpecialReminderOnBottom", "RoleCombatBanner", "BottomLine", "WxycReminder", "DrillBattleReminder",
110
+ "AbyssVsLunarisOnTop", "NoText", "InteractionPromptUI", "InfoTextDialog", "InstituteHomeInfo",
111
+ "WitchNicoleReminder", "TradeShowBonus", "SpecialReminderV2OnTop", "VarkaReminderBottom",
112
+ "InstituteHomeWarning", "VarkaReminder", "HerculesBattle", "EscoffierCookingChat", "AbyssWhisper",
113
+ "VaultDungeonRadio", "PenumbraMiniStory", "SpecialReminderV2OnBottom", "WeaponQuestSnezhnayaTop",
114
+ "InLevelTowerOfHanoiReminder", "SpecialFocusAttackReminder", "AbyssVsLunarisOnBottom",
115
+ "RobotAbyssalNarrate", "RobotAbyssalNarrateBroken", "CommonSpecialReminderNormal",
116
+ "MarionetteTeaTimeReminder", "RankedMatchFeverReminder", "PenumbraTarget", "PenumbraInfo",
117
+ "CommonSpecialReminderWarning", "RobotAbyssalWarning", "VarkaReminderLocked", "SneakLEReminder",
118
+ "NarrationChat", "RedBanner"}),
119
+ "starrail.events[].kind": frozenset({"talk", "option", "trigger", "wait"}),
120
+ # 绝区零不在这里:散条目按 `head` 分类,`head` 在契约里是自由取值(种类上千);`loose.reason` 这里不读
121
+ }
122
+ UNSEEN: Counter = Counter()
123
+ """(字段, 取值) -> 这次读库见到几次;`KNOWN_VALUES` 里没有的才记。"""
124
+ GAMETEXT: str | None = None
125
+ """`--gametext` 给的包 / 包目录;None 走 `paths.gametext_root()`。"""
126
+ _BUNDLES: dict[str, "gtdbundle.Bundle"] = {}
127
+ TAG_GAME = {"gi": "genshin", "hsr": "starrail", "zzz": "zzz"}
128
+ """视频标签前缀 -> 游戏(`data/index.md` F 节的命名)。鸣潮没有库。"""
129
+
130
+ NON_BAND_KINDS = {"Choice", "BlackScreen", "Unwalked", "Ambiguous", "OutOfSpan", "Duplicate",
131
+ "Loose", "ReminderUI", "Variant", "OffBand"}
132
+ """不进用途 2 分母的条目类型(**用途 2 = 字幕带**;前两类在屏幕上,是用途 1 的分母):
133
+
134
+ * `Choice`:走选项 UI。原神里**玩家的非独白台词一律是选项按钮**,只有一个选项时也是
135
+ (摆帧确认:gi1 148 s 的 `ナタ人って、見たことないしね。` 在右侧对话按钮上);
136
+ 括号独白 `(…)` 才进字幕带。gi1 实测玩家行被主轨命中 10 条,9 条是括号独白;
137
+ 剩下那条 `カチーナ!ムアラニ!` 是例外(非独白却上了字幕带),这条判据会把它排出分母。
138
+ * `BlackScreen`:原神黑屏旁白(屏幕正中),不在字幕带。
139
+ * `Unwalked` / `Ambiguous`:分支没走到 / 判不了(见模块说明)。
140
+ * `OutOfSpan`:单元里没在这段视频里播的那几截——**第一条之前、最后一条之后**没被 obs
141
+ 锚住的行(切片从对话中间开始、玩家中途离开,同 `script_align --span-ref` 的道理),
142
+ 以及中间一长段零锚点、两头锚点隔了几个小时的(`unplayed`)。
143
+ * `Duplicate`:库里同一段对话常有几份拷贝(换条件的重演版本,gi2 的 402513 / 402701
144
+ 共 59 行相同)。后一份里和前面单元重复的行,若 obs 里这段文字只上屏过一次,就不计,
145
+ 而且**连对齐序列都不进**(`sequence`)——只排出分母的话,对齐会去认领后面那份拷贝
146
+ (gi1 实测 41/49 被命中),原件反被记成漏。
147
+ * `Loose`:绝区零的散条目(`loose_*`:漫画过场、AutoEvent、教程、UI…)。上屏形态不知道
148
+ (漫画气泡、弹窗都不是字幕带),所以只参与检索和对齐——cue 读到它就不算"对不上剧本"——
149
+ 不进分母。例外是摆帧确认在字幕带的 `CN` / `TL` 两类(`LOOSE_BAND_HEADS`),按连号成组、进分母。
150
+ * `ReminderUI`:原神语音字幕(`reminders_*`)里 style 不是 `Normal` 的(`Banner` / `WhiteMessage` /
151
+ `CombatChat`…)。`Normal` 的记 `Reminder`、**进分母**:摆帧 gi1 6678 s 的
152
+ `カチーナ / ありがとう、ムアラニちゃん。` 就在底部字幕带、带名牌,主轨命中 gi1 57/64、gi2 14/20;
153
+ `Banner` 摆帧 gi1 7456 s 在画面上方横幅,gi1 / gi2 主轨命中 0/4、0/8(obs 全读到过),
154
+ `WhiteMessage` 0/4。其余 style 没见过上屏,保守地不进分母。
155
+ * `Variant`:绝区零变体组里代表行以外的那种说法(`fold_variants`)。它参与对齐,被认领时
156
+ **记在代表行上**——一组只算一个分母条目,任一种说法命中就算命中。
157
+ * `OffBand`:上面这些都没排掉、obs 也读到过(唯一查询),但锚住它的框**没有一个**落在字幕带里
158
+ (`band_of` / `mark_offband`)。上面几类是库的结构判的,这一类是**位置**判的,判得出库里没有结构的:
159
+ 绝区零的选项按钮、居中旁白、头顶气泡(上游没有分支和说话人),原神 / 星铁结构判据漏掉的。
160
+ 拿已有的结构判据核过位置判据(2026-09-11 四场整片):原神 / 星铁的 `Choice` 146/158、75/76、61/67
161
+ 落在带外,`BlackScreen` 21/29、6/6,`ReminderUI` 9/9、4/5;主轨命中的行被判出带外的 4 / 0 / 0 / 2 条
162
+ (绝区零居中的选项、原神一条居中的独白,都进了主轨)。没锚点的行没有位置证据,照旧留在分母里。"""
163
+
164
+
165
+ # ---------------------------------------------------------------- 读库
166
+
167
+ def note_value(field: str, value) -> None:
168
+ if value is not None and value not in KNOWN_VALUES[field]:
169
+ UNSEEN[(field, value)] += 1
170
+
171
+
172
+ def warn_unseen() -> None:
173
+ """读完一款游戏的库:没见过的枚举取值各打一行(只在建库时读库,缓存命中时不重报——换了包缓存就不命中)。"""
174
+ for (fld, v), n in sorted(UNSEEN.items()):
175
+ print(f" ⚠ 文本包:{fld} 出现没见过的取值 {v!r}({n} 处),照现有规则归类;确认含义后加进 KNOWN_VALUES",
176
+ file=sys.stderr, flush=True)
177
+ UNSEEN.clear()
178
+
179
+
180
+ def bundle_of(game: str) -> "gtdbundle.Bundle":
181
+ """这款游戏的文本包(进程内只找一次):找到、核清单、确认 loader 要的表和语种都在。"""
182
+ if game not in _BUNDLES:
183
+ b = gtdbundle.locate(game, GAMETEXT)
184
+ b.require(GAMES[game])
185
+ _BUNDLES[game] = b
186
+ return _BUNDLES[game]
187
+
188
+
189
+ def paired(game: str, stem: str):
190
+ """同一张表的日文 / 中文两份,**逐行 id 相同**(game-text-data 的不变量),这里再核一遍。
191
+ 两份都读到底,`iter_rows` 读完各自核一遍哈希。"""
192
+ bd = bundle_of(game)
193
+ # 按最长的配:只核 id 相等核不出"一份少了尾巴",裸 zip 会静默截断
194
+ for a, b in zip_longest(bd.iter_rows(stem, "ja"), bd.iter_rows(stem, "zh-Hans")):
195
+ if a is None or b is None:
196
+ raise SystemExit(f"{game}/{stem} 日中两份行数不一样({'日文' if a is None else '中文'}那份先完)")
197
+ if a["id"] != b["id"]:
198
+ raise SystemExit(f"{game}/{stem} 日中两份逐行 id 对不上:{a['id']} vs {b['id']}")
199
+ yield a, b
200
+
201
+
202
+ @dataclass
203
+ class Line:
204
+ key: str
205
+ kind: str # 本工具的分类(决定进不进分母)
206
+ role: str # 上游的角色 / 载体(原样,用来校准分母)
207
+ speaker: str | None
208
+ raw: str # 上游原文(含占位 / 标签)
209
+ cn: str | None
210
+ memberships: list[tuple[str, str]] = field(default_factory=list)
211
+ """独占所属的分支:(分支组, 选项)。嵌套分支会有多条。空 = 不在任何分支的独占集里。"""
212
+ variant: str | None = None
213
+ """绝区零变体组(同一句话按主角给的两种说法,`load_zzz`)。同组只算一个分母条目(`fold_variants`)。"""
214
+ speaker_cn: str | None = None
215
+ """说话人的中文名(上游中文那份的 `speaker`;原神任务对话多半为空、绝区零 scene 没有这个字段)。
216
+ 叠加 ASS 的名牌层译文从这里来(`scriptmatch.speaker_names`)。"""
217
+
218
+
219
+ @dataclass
220
+ class Unit:
221
+ uid: str
222
+ title: str | None
223
+ lines: list[Line]
224
+ solo: bool = False
225
+ """一行被锚住就算这段视频有它(单条的散条目:没有同组的其他行可以凑够两个锚点)。"""
226
+
227
+
228
+ def is_monologue(raw: str | None) -> bool:
229
+ """原神主角的心里话:整条用括号包着,进字幕带;其余玩家台词是选项按钮。"""
230
+ t = clean(raw or "", "F")
231
+ return t[:1] in ("(", "(")
232
+
233
+
234
+ def exclusive_members(nexts: dict[int, list[int]]) -> dict[int, list[tuple[int, int]]]:
235
+ """对话图里每个分支点(出边 ≥2),每个子节点的**独占可达集**——只从它能到、
236
+ 兄弟到不了的那部分。返回 节点 -> [(分支点, 子节点)…](嵌套分支会有多条)。
237
+ 可达性**不穿过分支点本身**:选项绕回分支点是原神的常态("把几个选项都问一遍"),
238
+ 穿过去的话,分支点和它之后的兄弟分支都会被算进这一支。"""
239
+ def reach(start: int, block: int) -> set[int]:
240
+ seen, st = set(), [start]
241
+ while st:
242
+ x = st.pop()
243
+ if x in seen or x == block or x not in nexts:
244
+ continue
245
+ seen.add(x)
246
+ st.extend(nexts[x])
247
+ return seen
248
+ out: dict[int, list[tuple[int, int]]] = defaultdict(list)
249
+ for g, nx in nexts.items():
250
+ if len(nx) < 2:
251
+ continue
252
+ rs = {n: reach(n, g) for n in nx}
253
+ for n in nx:
254
+ others = set().union(*(rs[m] for m in nx if m != n))
255
+ for x in rs[n] - others:
256
+ out[x].append((g, n))
257
+ return out
258
+
259
+
260
+ def branch_kind(memberships: list[tuple[str, str]], walked: dict[str, set]) -> str | None:
261
+ """这一行所属的分支走没走:同组有兄弟走过而它没有 -> Unwalked;同组谁都没走过 -> Ambiguous。"""
262
+ st = ["unwalked" if walked.get(g) and o not in walked[g] else
263
+ "ambiguous" if not walked.get(g) else "walked" for g, o in memberships]
264
+ return "Unwalked" if "unwalked" in st else "Ambiguous" if "ambiguous" in st else None
265
+
266
+
267
+ def load_genshin() -> list[Unit]:
268
+ units: dict[int, Unit] = {}
269
+ for q, qc in paired("genshin", "quests"):
270
+ cn_of = {d["dialog_id"]: d["text"] for d in qc["dialog"]}
271
+ spk_of = {d["dialog_id"]: d.get("speaker") for d in qc["dialog"]}
272
+ talks: dict[int, list[dict]] = defaultdict(list)
273
+ for d in q["dialog"]:
274
+ talks[d["talk_id"]].append(d)
275
+ for tid, ds in talks.items():
276
+ if tid in units: # 同一个 talk 挂在几个任务下(上游 9,597 个节点如此),留第一个
277
+ continue
278
+ ds = sorted({d["dialog_id"]: d for d in ds}.values(), key=lambda d: d["dialog_id"])
279
+ by = {d["dialog_id"]: d for d in ds}
280
+ choice = set()
281
+ for d in ds:
282
+ note_value("genshin.quests.dialog[].role.type", (d.get("role") or {}).get("type"))
283
+ nx = [n for n in (d.get("next") or []) if n in by]
284
+ if len(nx) > 1:
285
+ choice.update(n for n in nx
286
+ if (by[n].get("role") or {}).get("type") == "TALK_ROLE_PLAYER")
287
+ lines = {}
288
+ for d in ds:
289
+ role = ((d.get("role") or {}).get("type") or "NONE").removeprefix("TALK_ROLE_")
290
+ if d["dialog_id"] in choice or (role == "PLAYER" and not is_monologue(d["text"])):
291
+ kind = "Choice"
292
+ elif "BLACK_SCREEN" in role:
293
+ kind = "BlackScreen"
294
+ else:
295
+ kind = role
296
+ lines[d["dialog_id"]] = Line(key=f"gi:{d['dialog_id']}", kind=kind, role=role,
297
+ speaker=d.get("speaker"), raw=d["text"] or "",
298
+ cn=cn_of.get(d["dialog_id"]), speaker_cn=spk_of.get(d["dialog_id"]))
299
+ nexts = {i: [n for n in (d.get("next") or []) if n in by] for i, d in by.items()}
300
+ for x, ms in exclusive_members(nexts).items():
301
+ lines[x].memberships.extend((f"gi:{g}", str(o)) for g, o in ms)
302
+ units[tid] = Unit(uid=f"gi:talk:{tid}", title=q.get("name"), lines=list(lines.values()))
303
+ out = list(units.values())
304
+ # 语音字幕(`reminders_*`:边玩边说的对白,不在任何 talk 里):一条链一个单元,
305
+ # 单句链一条锚点就算。role 记上游的 style;只有 `Normal` 在字幕带(见 NON_BAND_KINDS)
306
+ for r, rc in paired("genshin", "reminders"):
307
+ for l in r["lines"]:
308
+ note_value("genshin.reminders.lines[].style", l["style"])
309
+ lines = [Line(key=f"gi:rem:{l['reminder_id']}",
310
+ kind="Reminder" if l["style"] == "Normal" else "ReminderUI",
311
+ role=l["style"] or "?", speaker=l["speaker"], raw=l["text"], cn=c["text"],
312
+ speaker_cn=c.get("speaker"))
313
+ for l, c in zip(r["lines"], rc["lines"], strict=True) if l["text"]]
314
+ if lines:
315
+ out.append(Unit(uid=f"gi:rem:{r['id']}", title=None, lines=lines, solo=len(lines) == 1))
316
+ warn_unseen()
317
+ return out
318
+
319
+
320
+ def _hsr_script(uid: str, title: str | None, ev: list[dict], evc: list[dict]) -> Unit:
321
+ """星铁一个脚本:`talk`/`trigger` 是上屏的台词,`option` 是选项(它的 `reply` 是 NPC 的回答,
322
+ 上屏、属于这个选项的分支),`wait <选项的 trigger>` 之后到下一个 `wait` 之间是那个选项的分支段。
323
+
324
+ 两处第一版都做错过(2026-09-11 整场 hsr 反查出来的):①没读 `reply`,选项对话里只剩玩家的选项、
325
+ 没有 NPC 的回答(上游 27,476 个选项带回答,game-text-data 那边第一版也这么丢过);②分支段按
326
+ `wait TalkSentence_<选项 id>` 认,而 wait 引用的多是选项的 `trigger`(= 回答那句,27,141 次对 2,054 次),
327
+ 九成以上的分支段没认出来、全当了主干。"""
328
+ # wait 引用的是选项的 trigger;少数脚本 trigger 就是选项自己。一个 trigger 挂在几个选项上 = 不独占
329
+ opts_of: dict[str, set] = defaultdict(set)
330
+ group_of, g, prev_opt = {}, 0, False
331
+ for e in ev:
332
+ note_value("starrail.events[].kind", e["kind"])
333
+ if e["kind"] == "option":
334
+ if not prev_opt:
335
+ g += 1
336
+ group_of[e["sentence_id"]] = g
337
+ prev_opt = True
338
+ for t in {e.get("trigger"), f"TalkSentence_{e['sentence_id']}"} - {None}:
339
+ opts_of[t].add(e["sentence_id"])
340
+ else:
341
+ prev_opt = False
342
+ lines: dict[int, Line] = {}
343
+ where: dict[int, set] = defaultdict(set) # sentence -> {选项 id 或 None(主干)}
344
+
345
+ def add(sid, kind, role, src, src_c):
346
+ if sid not in lines:
347
+ lines[sid] = Line(key=f"hsr:{sid}", kind=kind, role=role, speaker=src.get("speaker"),
348
+ raw=src["text"], cn=src_c.get("text"), speaker_cn=src_c.get("speaker"))
349
+
350
+ cur = None
351
+ for e, c in zip(ev, evc, strict=True):
352
+ k = e["kind"]
353
+ sid = e.get("sentence_id")
354
+ if k == "wait":
355
+ os_ = opts_of.get(e.get("custom_string") or "", set())
356
+ cur = next(iter(os_)) if len(os_) == 1 else None
357
+ continue
358
+ if sid is None or not e.get("text"):
359
+ continue
360
+ add(sid, "Choice" if k == "option" else "Talk", e.get("carrier") or k, e, c)
361
+ if k != "option":
362
+ where[sid].add(cur)
363
+ continue
364
+ r = e.get("reply") or {}
365
+ if r.get("sentence_id") is not None and r.get("text"):
366
+ add(r["sentence_id"], "Talk", "OptionReply", r, c.get("reply") or {})
367
+ where[r["sentence_id"]].add(sid)
368
+ for sid, ws in where.items():
369
+ if len(ws) == 1 and None not in ws:
370
+ (o,) = ws
371
+ lines[sid].memberships.append((f"{uid}#g{group_of[o]}", str(o)))
372
+ return Unit(uid=uid, title=title, lines=list(lines.values()))
373
+
374
+
375
+ def load_starrail() -> list[Unit]:
376
+ out = []
377
+ for m, mc in paired("starrail", "missions"):
378
+ for s, sc in zip(m["scripts"], mc["scripts"], strict=True):
379
+ out.append(_hsr_script(f"hsr:{s['source']}", m.get("name"), s["events"], sc["events"]))
380
+ for d, dc in paired("starrail", "dialogues"):
381
+ out.append(_hsr_script(f"hsr:{d['id']}", d.get("interact_title") or d.get("location"),
382
+ d["events"], dc["events"]))
383
+ # 全量台词表里**没有脚本引用**的那些(过场时间轴里念的,`.playable` 不在上游仓库)。
384
+ # 按 id 的百位块成组:同一脚本的台词 id 连号(game-text-data 的 docs/starrail),
385
+ # 块内按 id 排。这是本工具的归组,库那边刻意不做。
386
+ blocks: dict[int, list[Line]] = defaultdict(list)
387
+ for s, sc in paired("starrail", "sentences"):
388
+ if s["sources"] or not s["text"]:
389
+ continue
390
+ blocks[s["id"] // 100].append(Line(key=f"hsr:{s['id']}", kind="Talk", role="Timeline",
391
+ speaker=s["speaker"], raw=s["text"], cn=sc["text"],
392
+ speaker_cn=sc.get("speaker")))
393
+ out.extend(Unit(uid=f"hsr:block:{b}", title=None, lines=ls) for b, ls in sorted(blocks.items()))
394
+ warn_unseen()
395
+ return out
396
+
397
+
398
+ VARIANT_SIM = 0.5
399
+ """绝区零相邻两行的内容相似度达到这么多,就当成主角的两种说法(见 `load_zzz`)。"""
400
+
401
+
402
+ def variant_groups(texts: list[str], sim: float = VARIANT_SIM) -> list[list[int]]:
403
+ """相邻且相似的行连成组(长度 ≥2 的才返回)。只看相邻:两种说法在库里总是挨着放的。"""
404
+ groups, cur = [], [0] if texts else []
405
+ for i in range(1, len(texts)):
406
+ a, b = cnorm(texts[i - 1]), cnorm(texts[i])
407
+ if len(a) >= 3 and len(b) >= 3 and SequenceMatcher(None, a, b, autojunk=False).ratio() >= sim:
408
+ cur.append(i)
409
+ else:
410
+ if len(cur) > 1:
411
+ groups.append(cur)
412
+ cur = [i]
413
+ if len(cur) > 1:
414
+ groups.append(cur)
415
+ return groups
416
+
417
+
418
+ def load_zzz() -> list[Unit]:
419
+ """绝区零两位主角(リン / アキラ)的台词**在库里是相邻的两行**,玩家只会看到其中一行:
420
+
421
+ _001 リンさん、ビリーさん、聞こえるかしら?…
422
+ _002 アキラさん、ビリーさん、聞こえるかしら?…
423
+
424
+ 上游没有分支信息(它自己 README 的"上游边界"),所以按相邻相似度把它们连成一组(`variant`),
425
+ 同组**只算一个分母条目**、任一种说法被认领就算命中(`fold_variants`)。
426
+
427
+ 第一版把组交给 `branch_kind` 当选项判(锚住的那行走过、另一行 Unwalked),两处都错
428
+ (audit-6,2026-09-11 数的):①两行都没锚点就整组记 Ambiguous 排出分母,而整场 zzz
429
+ 这 47 行被主轨命中了 10 行;②obs 一个框的文字常常同时"包含于"两种说法,两行都被锚住、
430
+ 都留在分母里,于是必有一行记成漏。"""
431
+ out = []
432
+ for s, sc in paired("zzz", "scenes"):
433
+ cn = {l["key"]: l["text"] for l in sc["lines"]}
434
+ lines = [Line(key=f"zzz:{l['key']}", kind="Talk", role=s.get("kind") or "?",
435
+ speaker=l.get("speaker"), raw=l["text"] or "", cn=cn.get(l["key"]))
436
+ for l in s["lines"]]
437
+ for g in variant_groups([clean(ln.raw, "F") for ln in lines]):
438
+ for i in g:
439
+ lines[i].variant = f"zzz:{s['id']}#v{g[0]}"
440
+ out.append(Unit(uid=f"zzz:{s['id']}", title=s.get("title"), lines=lines))
441
+ # 散条目:在字幕带上的两类(CN / TL)按连号成组进分母;其余 key 里读得出分组的
442
+ # (AutoEvent<事件>_<序号>)成组按序号排,读不出的一条一个单元,都只参与检索与对齐
443
+ groups: dict[str, list] = defaultdict(list)
444
+ band: dict[str, list] = defaultdict(list)
445
+ for r, rc in paired("zzz", "loose"):
446
+ if not r["text"]:
447
+ continue
448
+ slot = band_loose_slot(r["head"], r["id"])
449
+ ln = Line(key=f"zzz:{r['id']}", kind="Talk" if slot else "Loose", role=r["head"],
450
+ speaker=None, raw=r["text"], cn=rc["text"])
451
+ if slot:
452
+ base, seq, pc = slot
453
+ if pc:
454
+ ln.variant = f"zzz:loose:{r['head']}:{base}#{seq}"
455
+ band[f"{r['head']}:{base}"].append(((seq, pc), ln))
456
+ elif r["group"]:
457
+ groups[f"{r['head']}:{r['group']}"].append((r["seq"], ln))
458
+ else:
459
+ out.append(Unit(uid=f"zzz:loose:{r['id']}", title=None, lines=[ln], solo=True))
460
+ for g, items in sorted(groups.items()) + sorted(band.items()):
461
+ out.append(Unit(uid=f"zzz:loose:{g}", title=None,
462
+ lines=[ln for _, ln in sorted(items, key=lambda x: x[0])],
463
+ solo=len(items) == 1))
464
+ return out
465
+
466
+
467
+ LOOSE_BAND_HEADS = ("CN", "TL")
468
+ """绝区零散条目里**在字幕带上**的两类,进分母(audit-6,摆帧:`CN` 7,972 s 是带名牌 `リン` 的
469
+ 对话框,`TL` 1,402 s 是过场字幕;整场 zzz 主轨命中 `CN` 6/8、`TL` 2/2)。同一批被命中的
470
+ `Comic`(漫画格里的气泡,位置随格子变)、技能说明、线索描述、按键提示不是字幕带,仍是 `Loose`。"""
471
+ CN_ID_RE = re.compile(r"^(?P<base>.*?_EP\d+)(?P<pc>[BG]?)_(?P<seq>\d+)$")
472
+ SEQ_ID_RE = re.compile(r"^(?P<base>.*)_(?P<seq>\d+)$")
473
+
474
+
475
+ def band_loose_slot(head: str, id_: str) -> tuple[str, int, str] | None:
476
+ """字幕带散条目在它那段过场里的位置:(组, 序号, 主角写法 'B' / 'G' / '');不进分母的给 None。
477
+
478
+ 库那边这两类**没有分组**(`group` 为空),每条一个单元的话,只有被 OCR 读到的行才进得了
479
+ 分母、漏量不出来,所以按连号成组(本工具的归组,同星铁的百位块)。`CN` 的 key 是
480
+ `C280_EP020_010`:基底 `C280_EP020` 放两位主角共有的行,主角相关的几句在 `C280_EP020G_030` /
481
+ `C280_EP020B_030` 两份里、**同一个序号各一种说法**(`私が…みんなのプロキシだよ!` /
482
+ `僕が…みんなのプロキシだ!`)——并进基底那一组、同序号的两行记成变体组。
483
+ 名牌(`_Name_`)与读不出序号的(`_Lyric` 歌词)不进。"""
484
+ if head not in LOOSE_BAND_HEADS or "_Name_" in id_:
485
+ return None
486
+ m = CN_ID_RE.match(id_) if head == "CN" else None
487
+ if m:
488
+ return m["base"], int(m["seq"]), m["pc"]
489
+ m = SEQ_ID_RE.match(id_)
490
+ return (m["base"], int(m["seq"]), "") if m else None
491
+
492
+
493
+ LOADERS = {"genshin": load_genshin, "starrail": load_starrail, "zzz": load_zzz}
494
+
495
+
496
+ # ---------------------------------------------------------------- 清洗与归一
497
+
498
+ GENDER_RE = re.compile(r"\{([MF])#([^{}]*)\}")
499
+ RUBY_RE = re.compile(r"\{RUBY#\[[^\]]*\][^{}]*\}")
500
+ LAYOUT_RE = re.compile(r"\{LAYOUT_([A-Z]+)#([^{}]*)\}")
501
+ PC_LAYOUTS = ("KEYBOARD", "PC")
502
+ """绝区零教程提示按平台给几种写法;这批直播都是 PC,留键盘那一种(没有就留 FALLBACK)。"""
503
+ PLACEHOLDER_RE = re.compile(r"\{[^{}]*\}")
504
+ """性别写法 / 注音 / 平台写法先处理掉之后,剩下的花括号全删:多数是玩家相关的占位
505
+ (`{NICKNAME}` / `{REALNAME[…]}` / `{PLAYERAVATAR#SEXPRO[…]}`),星铁的
506
+ `{RUBY_B#注音}正文{RUBY_E#}` 删掉两个花括号正好剩正文。**会丢字的**是少数运行时取值:
507
+ 原神 `{ABYSSWAR#…}`(21 处以内)、星铁 `{TEXTJOIN#…}`(142)、绝区零 `{NPC_…}`(25)。"""
508
+ TAG_RE = re.compile(r"<[^<>]*>")
509
+
510
+
511
+ def _layout(t: str) -> str:
512
+ keep = next((p for p in PC_LAYOUTS if f"{{LAYOUT_{p}#" in t), "FALLBACK")
513
+ return LAYOUT_RE.sub(lambda m: m.group(2) if m.group(1) == keep else "", t)
514
+
515
+
516
+ def clean(raw: str, gender: str) -> str:
517
+ """上游原文 -> 屏幕上会出现的文字(`\\N` 表示条内换行)。"""
518
+ t = raw.lstrip("#")
519
+ t = t.replace("\\n", "\n")
520
+ t = GENDER_RE.sub(lambda m: m.group(2) if m.group(1) == gender else "", t)
521
+ t = RUBY_RE.sub("", t)
522
+ t = _layout(t)
523
+ t = PLACEHOLDER_RE.sub("", t)
524
+ t = TAG_RE.sub("", t)
525
+ return "\\N".join(x.strip() for x in t.split("\n") if x.strip())
526
+
527
+
528
+ SMALL_KANA = str.maketrans("ぁぃぅぇぉっゃゅょゎァィゥェォッャュョヮヵヶ",
529
+ "あいうえおつやゆよわアイウエオツヤユヨワカケ")
530
+ NONWORD = re.compile(r"[\W_]+")
531
+
532
+
533
+ def cnorm(t: str) -> str:
534
+ """检索用的**内容**归一:NFKC、小书写假名并成大写、去掉一切标点空白。
535
+ OCR 最常见的差异是省略号 / 小书写假名(script-corpus 报告文本质量那节),
536
+ 检索这一步不该被它们挡住;量文本质量是 game_align 的事,不在这里。"""
537
+ t = unicodedata.normalize("NFKC", t.replace("\\N", ""))
538
+ return NONWORD.sub("", t.translate(SMALL_KANA))
539
+
540
+
541
+ # ---------------------------------------------------------------- 短名词表(可选)
542
+
543
+ TERMS_TABLE = "terms"
544
+ """文本包里的短名词对照表(game-text-data 的 `terms_*`:TextMap 里名词形的短串,一个 hash 一行,日中逐行 id 相同)。
545
+ **可选**:旧包没有它照读,剧本里就没有 `terms`,名牌层只查说话人表。"""
546
+ TERMS_CONTRACT = 1
547
+ """短名词表认得的内容契约主版本(表在就要对上,见 `GAMES`)。"""
548
+ TERM_MIN_CONTENT = 2
549
+ """查询键至少几个内容字(`cnorm` 之后)。名牌层会把噪音读数(`口`、`E`、`9`)当开头行,
550
+ 一个字的键在 TextMap 里几乎总查得到,收进来就会把噪音"译"出来、在默认预设里画上板。"""
551
+ _KANA_HAN = re.compile(r"[぀-ヿ㐀-鿿]")
552
+
553
+
554
+ def term_key(t: str) -> str:
555
+ """短名词表的查询键:NFKC、去空白——和 `script_align.norm` 对纯文本的结果相同,
556
+ 名牌层(`scriptmatch.term_names`)按那个归一查,两边键一致。"""
557
+ return "".join(unicodedata.normalize("NFKC", t.replace("\\N", "")).split())
558
+
559
+
560
+ def term_table(game: str, obs_raw, gender: str) -> tuple[dict | None, dict]:
561
+ """这段视频用得上的短名词:日文键(`term_key`)-> {`cn`, `n`(取中的中文对应几个 hash), `of`(这个键共几个 hash)}。
562
+
563
+ * 只收 obs 里**逐字出现过**的键(归一后相等,内容 ≥`TERM_MIN_CONTENT` 字、含假名或汉字):整张表二十多万条,
564
+ 名牌层只按屏幕上的字查,别的键写进剧本也用不上。
565
+ * **消歧**:同一个日文键对几个中文时按 **hash 条数**取最多的那个(同一个中文串在 TextMap 里常有好几条,
566
+ 条数就是游戏里这种译法用得多不多);打平取短的、再按字典序(确定性)。原神 7.0.0 短串的日文键里
567
+ 对多个中文的约 2.6%,抽样多是近义写法(`木製の橋` → 木质桥梁 13 / 木桥 6)。`n / of` 记下来,
568
+ 多数派占比低的那些下游可以另眼看。
569
+ * 文本照剧本行一样 `clean`(性别写法、占位、标签)。
570
+ 返回 (表或 None——包里没有这张表, 统计)。"""
571
+ if not bundle_of(game).has([TERMS_TABLE]):
572
+ return None, {"table": False}
573
+ bundle_of(game).require({TERMS_TABLE: TERMS_CONTRACT})
574
+ want = term_wanted(obs_raw)
575
+ out = pick_terms(paired(game, TERMS_TABLE), want, gender)
576
+ amb = [v for v in out.values() if v["n"] < v["of"]]
577
+ return out, {"table": True, "obs_keys": len(want), "keys": len(out), "ambiguous": len(amb),
578
+ "majority_below_60": sum(1 for v in amb if v["n"] < 0.6 * v["of"])}
579
+
580
+
581
+ def term_wanted(obs_raw) -> set[str]:
582
+ """obs 文本里够格去查短名词表的键:内容 ≥`TERM_MIN_CONTENT` 字、含假名或汉字。"""
583
+ want = set()
584
+ for raw in obs_raw:
585
+ k = term_key(raw)
586
+ if len(cnorm(k)) >= TERM_MIN_CONTENT and _KANA_HAN.search(k):
587
+ want.add(k)
588
+ return want
589
+
590
+
591
+ def pick_terms(pairs, want: set[str], gender: str) -> dict[str, dict]:
592
+ """(日文行, 中文行) 逐对 -> 键在 `want` 里的那些:按 hash 条数取最多的中文,打平取短的、再按字典序。
593
+
594
+ ⚠ 观察项(09-26 审计定级:中度,只在匹配器这一层):"短"的上限是各游戏说话人名的 P99(15 / 13 / 9),查询门、消歧规则都是手定;
595
+ 五段切片新译出的 54 块里真名牌 24、任务目标 / 奖励提示 30(game-text-corpus 报告 7.8)。看 `overlay.stats.name_from_terms` 和 `name_src = terms` 的块。"""
596
+ cnt: dict[str, Counter] = defaultdict(Counter)
597
+ for a, b in pairs:
598
+ if not a["text"] or not b["text"]:
599
+ continue
600
+ k = term_key(clean(a["text"], gender))
601
+ if k in want:
602
+ cn = clean(b["text"], gender)
603
+ if cn:
604
+ cnt[k][cn] += 1
605
+ out = {}
606
+ for k, c in sorted(cnt.items()):
607
+ cn, n = min(c.items(), key=lambda kv: (-kv[1], len(kv[0]), kv[0]))
608
+ out[k] = {"cn": cn, "n": n, "of": sum(c.values())}
609
+ return out
610
+
611
+
612
+ # ---------------------------------------------------------------- 检索
613
+
614
+ K = 5
615
+ """n-gram 长度。日文 5 字的串在 30 万行里已经足够稀有;库侧隔一位取、查询侧逐位取,
616
+ 对齐时至少有一半的查询 gram 落在库侧取过的位置上。"""
617
+ QMIN_FLOOR = K + 3
618
+ """`--qmin` 的下限。查询 L 字有 L−K+1 个 gram、至少要中 2 个(`Index.query` 的 `need`),
619
+ 而库侧只存隔一位的 gram:L = K+1 时两个 gram 相邻、库里至多存了一个,**永远命不中**;
620
+ L = K+2 时要看查询落在库行的奇偶位,一半的概率命不中。L ≥ K+3 才保证整行被包含时一定找得到。"""
621
+
622
+ _GRAM_BASE = np.uint64(0x9E3779B97F4A7C15)
623
+
624
+
625
+ def gram_hashes(s: str) -> np.ndarray:
626
+ """s 里每个起点的 K-gram -> 64 位多项式哈希(按码位,uint64 自然回绕)。
627
+
628
+ **跨进程稳定**:第一版用内置 `hash()`,它对 str 每个进程随机加盐——结果只在 gram 撞哈希时
629
+ 才依赖它,"重跑逐格相同"是实测出来的;换成确定的哈希,确定性就是可证的(audit-6)。"""
630
+ c = np.frombuffer(s.encode("utf-32-le"), dtype=np.uint32).astype(np.uint64)
631
+ n = len(c) - K + 1
632
+ if n <= 0:
633
+ return np.empty(0, dtype=np.uint64)
634
+ h = np.zeros(n, dtype=np.uint64)
635
+ for k in range(K):
636
+ h = h * _GRAM_BASE + c[k:k + n]
637
+ return h
638
+
639
+
640
+ class Index:
641
+ def __init__(self, texts: list[str]):
642
+ hs, ids = [], []
643
+ for i, c in enumerate(texts):
644
+ n = len(c)
645
+ if n < K:
646
+ continue
647
+ pos = list(range(0, n - K + 1, 2))
648
+ if pos[-1] != n - K:
649
+ pos.append(n - K)
650
+ hs.append(gram_hashes(c)[pos])
651
+ ids.append(np.full(len(pos), i, dtype=np.int32))
652
+ h = np.concatenate(hs) if hs else np.empty(0, dtype=np.uint64)
653
+ order = np.argsort(h, kind="stable")
654
+ self.h = h[order]
655
+ self.ids = (np.concatenate(ids) if ids else np.empty(0, dtype=np.int32))[order]
656
+ self.texts = texts
657
+
658
+ def query(self, q: str, top: int = 6) -> list[int]:
659
+ qh = gram_hashes(q)
660
+ lo = np.searchsorted(self.h, qh, "left")
661
+ hi = np.searchsorted(self.h, qh, "right")
662
+ parts = [self.ids[a:b] for a, b in zip(lo, hi) if b > a]
663
+ if not parts:
664
+ return []
665
+ # 同一个 gram 在同一行里出现多次只算一次,免得长重复行靠刷票上榜
666
+ ids, cnt = np.unique(np.concatenate([np.unique(p) for p in parts]), return_counts=True)
667
+ need = max(2, int(0.25 * len(qh)))
668
+ ok = cnt >= need
669
+ ids, cnt = ids[ok], cnt[ok]
670
+ return [int(ids[j]) for j in np.argsort(-cnt, kind="stable")[:top]]
671
+
672
+
673
+ def contained(q: str, c: str) -> float:
674
+ """q(一个 OCR 框 = 屏幕上的一行)有多少落在 c(库里一整条,可能跨两行)里。"""
675
+ sm = SequenceMatcher(None, q, c, autojunk=False)
676
+ return sum(b.size for b in sm.get_matching_blocks()) / max(1, len(q))
677
+
678
+
679
+ def obs_texts(obs: Path) -> dict[str, list[tuple[float, float]]]:
680
+ """obs 里每条去重文本 -> 它出现过的全部 (时刻秒, 框中心 y / 画面高)。"""
681
+ out: dict[str, list[tuple[float, float]]] = defaultdict(list)
682
+ h = None
683
+ with open(obs, encoding="utf-8") as f:
684
+ for line in f:
685
+ r = json.loads(line)
686
+ if "_meta" in r:
687
+ h = r["_meta"].get("height")
688
+ continue
689
+ if not h:
690
+ raise SystemExit(f"{obs} 的 _meta 没有画面高度(height),判不了字幕带(见 band_of)")
691
+ b = r["box"]
692
+ out[r["text"]].append((r["t_us"] / 1e6, (b[1] + b[3]) / 2 / h))
693
+ return out
694
+
695
+
696
+ EPISODE_GAP = 3.0
697
+ """相邻两次出现隔得不超过这么多秒,算同一次上屏(2 fps 采样,允许丢几帧)。"""
698
+
699
+
700
+ def episodes(times: list[float], gap: float = EPISODE_GAP) -> list[list[float]]:
701
+ """一串时刻 -> 若干次上屏 [[起, 止]…]。同一行台词可能隔一个小时又出现一次(回想、
702
+ 重看),只记首末时刻会把两次当成一次。"""
703
+ out: list[list[float]] = []
704
+ for t in sorted(times):
705
+ if out and t - out[-1][1] <= gap:
706
+ out[-1][1] = t
707
+ else:
708
+ out.append([t, t])
709
+ return out
710
+
711
+
712
+ BAND_WIN, BAND_PAD = 0.12, 0.03
713
+ """字幕带的高(众数窗口)与两头的放宽,都按画面高算。原神 / 星铁两行字、绝区零四行字的对话框,
714
+ 每行框中心都落在 0.12 以内(2026-09-11 四场整片:带内装下 822/1044、456/558、1667/1708、978/994 行)。"""
715
+ BAND_MIN_LINES = 5
716
+ """锚住的行少于这么多就不估字幕带、不判 OffBand(众数窗口没有意义)。"""
717
+
718
+
719
+ def band_of(ys_per_line: list[list[float]]) -> tuple[float, float] | None:
720
+ """这段视频的字幕带(画面高的比例,上 -> 下):每行取锚点框 y 中心的中位数,
721
+ 找一个高 `BAND_WIN` 的窗口装下最多的行,两头各放宽 `BAND_PAD`。
722
+
723
+ 只用 obs,所以各臂共用;**不看任何一条臂的主轨在哪**(那样分母就偏向那条臂)。"""
724
+ meds = sorted(float(np.median(ys)) for ys in ys_per_line if ys)
725
+ if len(meds) < BAND_MIN_LINES:
726
+ return None
727
+ fit = [bisect_right(meds, m + BAND_WIN) - i for i, m in enumerate(meds)]
728
+ i = fit.index(max(fit))
729
+ return meds[i] - BAND_PAD, meds[i] + BAND_WIN + BAND_PAD
730
+
731
+
732
+ # ---------------------------------------------------------------- 主流程
733
+
734
+ def shown_twice(anchor: dict | None, span: float) -> bool:
735
+ """obs 里这行确实上屏过两次:两次上屏之间隔了 ≥span 秒。只看首末时刻不行——
736
+ 一条挂在屏幕上一分多钟的台词会被当成"出现了两次"。"""
737
+ eps = (anchor or {}).get("episodes") or []
738
+ return any(b[0] - a[1] >= span for a, b in zip(eps, eps[1:]))
739
+
740
+
741
+ def mark_duplicates(out_units: list[dict], share: float, span: float) -> int:
742
+ """后出现的单元若大半内容行和前面的单元重复,它多半是同一段对话的另一份拷贝
743
+ (库里换条件的重演版本)——重复的行只在 obs 里确实出现过两次时才算。"""
744
+ seen: set[str] = set()
745
+ n = 0
746
+ # 先按节点:原神同一个对话节点可以挂在两个 talk 下(gi1 的 500001 / 500017 共用 49 个
747
+ # dialog_id),那就是同一行,后一份无条件记 Duplicate——否则 key 重复,后面按 key 查表会互相覆盖。
748
+ node_keys: set[str] = set()
749
+ for u in out_units:
750
+ for ln in u["lines"]:
751
+ if ln["key"] in node_keys:
752
+ ln["kind"] = "Duplicate"
753
+ n += 1
754
+ node_keys.add(ln["key"])
755
+ for u in out_units:
756
+ band = [ln for ln in u["lines"] if ln["kind"] not in NON_BAND_KINDS]
757
+ texts = [cnorm(ln["text"]) for ln in band]
758
+ # "是不是拷贝"只按 ≥3 字的行判(`うん` 这种短句哪个单元都有,拿它投票会误判);
759
+ # 判定是拷贝之后,短行也照样按"前面出现过"标——同一份拷贝里的短句同样是重复的
760
+ long = [k for k in texts if len(k) >= 3]
761
+ if long and sum(k in seen for k in long) / len(long) >= share:
762
+ for ln, k in zip(band, texts):
763
+ if k in seen and not shown_twice(ln["anchor"], span):
764
+ ln["kind"] = "Duplicate"
765
+ n += 1
766
+ seen.update(k for k in texts if k)
767
+ return n
768
+
769
+
770
+ def fold_variants(groups: list[list[dict]]) -> int:
771
+ """每个变体组(输出行 dict 的列表,已去掉 Choice / OutOfSpan)留**一个**代表行进分母,
772
+ 其余记 `Variant`、带 `variant_of`(代表行的 key);返回记了几行。
773
+
774
+ 代表行挑"被唯一查询锚住的 > 被锚住的 > 组里第一个"——只影响报表里显示哪一种写法,
775
+ 不影响命中:`game_align` 那边任一种说法被认领都记在代表行上。"""
776
+ n = 0
777
+ for members in groups:
778
+ if len(members) < 2:
779
+ continue
780
+ rep = min(members, key=lambda d: (not (d["anchor"] and d["anchor"]["unique"]),
781
+ not d["anchor"], members.index(d)))
782
+ for d in members:
783
+ if d is not rep:
784
+ d["kind"], d["variant_of"] = "Variant", rep["key"]
785
+ n += 1
786
+ return n
787
+
788
+
789
+ SHORT_ANCHOR_PAD = 60.0
790
+ """短行锚点:只认落在本单元锚点时段两头各放宽这么多秒之内的出现。"""
791
+
792
+
793
+ def short_anchors(units: list[Unit], picked: list[int], anchors: dict, qs_short: dict,
794
+ max_units: int, gender: str = "F", pad: float = SHORT_ANCHOR_PAD) -> int:
795
+ """圈中单元里**没有锚点的短行**(内容不到 `--qmin` 字),在本单元的时段内按归一后**逐字相等**
796
+ 去 obs 里找;就地写进 `anchors`,返回找到几行。
797
+
798
+ 为什么这样才安全:短串在 30 万行的库里到处都是,所以它们不参与检索、不参与圈单元、
799
+ 也不参与估字幕带(`band_of` 在这之前就算完了)。但**单元一旦圈中,候选就只剩这几十行**,
800
+ 歧义没了——再排掉"同一串在 >`max_units` 个圈中单元里都有"的,剩下的是这一处的证据。
801
+
802
+ 收益(2026-09-11,整场绝区零):现在记成漏的短行里 44 条能在**带外**找到(选项按钮:
803
+ GalGame 居中、Chat 右侧),带内只有 3 条;这 44 条随后被 `mark_offband` 排出分母。
804
+ 没有这一步,它们只能以"太短判不了"挂在疑似真漏里(corpus)。"""
805
+ where: dict[str, list[tuple[int, int]]] = defaultdict(list)
806
+ span: dict[int, tuple[float, float]] = {}
807
+ for ui in picked:
808
+ eps = [e for li in range(len(units[ui].lines)) if (ui, li) in anchors
809
+ for e in anchors[(ui, li)][0]]
810
+ if not eps:
811
+ continue
812
+ span[ui] = (min(e[0] for e in eps) - pad, max(e[1] for e in eps) + pad)
813
+ for li, ln in enumerate(units[ui].lines):
814
+ if (ui, li) not in anchors:
815
+ where[cnorm(clean(ln.raw, gender))].append((ui, li))
816
+ n = 0
817
+ for q, places in where.items():
818
+ occ = qs_short.get(q)
819
+ if not occ or len({ui for ui, _ in places}) > max_units:
820
+ continue
821
+ for ui, li in places:
822
+ lo, hi = span[ui]
823
+ got = [(t, y) for t, y in occ if lo <= t <= hi]
824
+ if got:
825
+ anchors[(ui, li)] = (episodes([t for t, _ in got]), 1, 1, [y for _, y in got])
826
+ n += 1
827
+ return n
828
+
829
+
830
+ def mark_offband(out_units: list[dict]) -> int:
831
+ """还在分母里的行,若 obs 读到过它(唯一查询)而那些框**没有一个**落在字幕带里,记 `OffBand`;
832
+ 返回记了几行。变体组看整组:代表行没有位置证据时,拿同组另一种说法的。
833
+
834
+ 放在重演拷贝之后:拷贝的判定只看字幕带里的行,先判 OffBand 会让它少看几行、改了拷贝的结果。"""
835
+ n = 0
836
+ for u in out_units:
837
+ members: dict[str, list[dict]] = defaultdict(list)
838
+ for ln in u["lines"]:
839
+ if ln.get("variant_of"):
840
+ members[ln["variant_of"]].append(ln)
841
+ for ln in u["lines"]:
842
+ if ln["kind"] in NON_BAND_KINDS:
843
+ continue
844
+ seen = [m["anchor"]["in_band"] for m in (ln, *members[ln["key"]])
845
+ if m["anchor"] and m["anchor"]["in_band"] is not None]
846
+ if seen and not any(seen):
847
+ ln["kind"] = "OffBand"
848
+ n += 1
849
+ return n
850
+
851
+
852
+ SPAN_RUN = 5
853
+ SPAN_GAP_MIN, SPAN_PER_LINE = 300.0, 20.0
854
+
855
+
856
+ def unplayed(on: list[int], eps: dict[int, list], n: int) -> set[int]:
857
+ """单元里**没在这段视频里播**的行号(记 OutOfSpan):
858
+
859
+ * 第一个锚点之前、最后一个之后(切片从对话中间开始 / 玩家中途离开);
860
+ * 中间一段:连续 ≥`SPAN_RUN` 行一个锚点都没有,而夹着它的两个锚点在时间上隔得比
861
+ "这几行正常播完"远得多(> max(`SPAN_GAP_MIN`, `SPAN_PER_LINE` × 行数)),或者前后颠倒。
862
+
863
+ 第二条是 gi2 逼出来的:talk 400004 在 11,386 s 锚住开头几行,然后 58 行一个锚点都没有,
864
+ 下一个锚点在 16,479 s——中间 43 条内容行 OCR 一条都没读到过。一段真播过的对话,
865
+ OCR 在任何区域都读不到其中任何一行,是不可能的;这是库里同一个 talk 的两截分别出现在
866
+ 视频里(或者其中一截是别处重复的句子锚上的),中间那截根本没播。
867
+ 只看 obs 锚点,所以各臂共用。`on` 是有锚点的行号(升序),`eps` 是它们的上屏时段。"""
868
+ if not on:
869
+ return set(range(n))
870
+ out = set(range(on[0])) | set(range(on[-1] + 1, n))
871
+ for i, j in zip(on, on[1:]):
872
+ k = j - i - 1
873
+ if k >= SPAN_RUN:
874
+ # 两行各自可能上屏过几次:取"i 的某次结束 -> j 的某次开始"里最近的那一对
875
+ fwd = [s - e for s, _ in eps[j] for _, e in eps[i] if s >= e]
876
+ if not fwd or min(fwd) > max(SPAN_GAP_MIN, SPAN_PER_LINE * k):
877
+ out.update(range(i + 1, j))
878
+ return out
879
+
880
+
881
+ def sequence(out_units: list[dict], back: float, fwd: float) -> list[list]:
882
+ """剧本行的**对齐序列**:[[key, 排序时刻, [[可认领起, 止]…]]…],按 obs 时间排到行一级。
883
+
884
+ * 有锚点的行:它的每一次上屏各外扩 `fwd` 秒,就是 cue 可以认领它的时段;
885
+ * 没锚点的行(短句从来不拿去检索):借本单元里**前面**最近那个有锚点的行的上屏时段,
886
+ 往后放宽 `back` 秒(它真正上屏只会更晚);单元开头那几行借后面第一个,往前放宽。
887
+
888
+ 为什么不按单元排、也不只记一个时刻:一个 talk 可以拖很久(gi1 的 500010 锚点从 974 s
889
+ 一直到 3937 s,玩家中途去做别的),同一行也会隔很久再出现一次。这些时段给 `game_align`
890
+ 圈每条 cue 的候选(**只圈范围,命中与否仍看文本**)。只用 obs,所以各臂共用。
891
+ 重演拷贝(Duplicate)不进序列。"""
892
+ items = []
893
+ for ui, u in enumerate(out_units):
894
+ eps = [ln["anchor"]["episodes"] if ln["anchor"] else None for ln in u["lines"]]
895
+ on = [li for li, e in enumerate(eps) if e]
896
+ for li, ln in enumerate(u["lines"]):
897
+ if ln["kind"] == "Duplicate":
898
+ continue
899
+ if eps[li]:
900
+ src, win = eps[li], [[s - fwd, e + fwd] for s, e in eps[li]]
901
+ elif on and on[0] < li:
902
+ src = eps[max(j for j in on if j < li)]
903
+ win = [[s, e + back] for s, e in src]
904
+ elif on:
905
+ src = eps[on[0]]
906
+ win = [[s - back, e] for s, e in src]
907
+ else:
908
+ src, win = [[u["t0"], u["t0"]]], [[u["t0"] - back, u["t0"] + back]]
909
+ items.append((src[0][0], ui, li, ln["key"], win))
910
+ items.sort(key=lambda x: x[:3])
911
+ return [[k, round(t, 2), [[round(a, 2), round(b, 2)] for a, b in w]] for t, _, _, k, w in items]
912
+
913
+
914
+ def library(game: str) -> tuple[list[Unit], list[tuple[int, int, str]], list[str], Index, str]:
915
+ """读库 + 两种性别写法各归一一遍 + 建检索索引:(单元, flat, texts, 索引, 'cache hit|miss')。
916
+
917
+ 这三步只依赖**库文件和产剧本的代码**,不依赖 obs——原神一次 29 s 里占 ~28 s
918
+ (2026-09-14 cProfile:读库 9.6 s、`clean`/`cnorm` ~6 s、`Index` 6.8 s),而同一次评测里库是常数。
919
+ 所以落盘到 `out/_cache/gamescript/`。键 = 游戏 + `code_fp(CODE_FP_FILES)` + 文本包的**内容指纹**:
920
+ 代码改了、库换了都换键;拷贝 / 解压包不改指纹,缓存照样命中。命中时照样核包里用到的文件的存储哈希
921
+ (`verify_stored`,不解压,远比建库便宜)——否则清单完好、成员坏了的包在有缓存的机器上会静默通过。
922
+ 缓存只是加速,产物必须和不缓存时逐格相同(`--no-cache` 可对账);整目录删掉就等于没有。"""
923
+ import hashlib
924
+ import os
925
+ import pickle
926
+ from flowocr.analyze import build_tracks as bt
927
+ use_cache = USE_CACHE and game in GAMES # 守卫里的合成库(`LOADERS["syn"]`)没有库文件,不缓存
928
+ if not use_cache:
929
+ return (*flatten(LOADERS[game]()), "no cache")
930
+ b = bundle_of(game)
931
+ key = hashlib.sha256(json.dumps([game, K, bt.code_fp(CODE_FP_FILES), b.fingerprint])
932
+ .encode()).hexdigest()[:16]
933
+ cp = paths.data_root() / "out" / "_cache" / "gamescript" / f"{game}-{key}.pkl"
934
+ if cp.exists():
935
+ b.verify_stored(GAMES[game])
936
+ with open(cp, "rb") as f:
937
+ units, flat, texts, h, ids = pickle.load(f)
938
+ idx = Index.__new__(Index)
939
+ idx.h, idx.ids, idx.texts = h, ids, texts
940
+ return units, flat, texts, idx, "cache hit"
941
+ units, flat, texts, idx = flatten(LOADERS[game]())
942
+ cp.parent.mkdir(parents=True, exist_ok=True)
943
+ for old in cp.parent.glob(f"{game}-*.pkl"): # 旧键永远不会再命中,留着只占盘
944
+ old.unlink()
945
+ tmp = cp.with_suffix(f".{os.getpid()}.tmp") # 带 pid:两个进程同时建同一款游戏的缓存时不共用一个临时文件
946
+ with open(tmp, "wb") as f:
947
+ pickle.dump((units, flat, texts, idx.h, idx.ids), f, protocol=pickle.HIGHEST_PROTOCOL)
948
+ tmp.replace(cp)
949
+ return units, flat, texts, idx, "cache miss"
950
+
951
+
952
+ def flatten(units: list[Unit]) -> tuple[list[Unit], list[tuple[int, int, str]], list[str], Index]:
953
+ """每行两种性别写法各归一一遍(写法不同才各占一格),再建检索索引。"""
954
+ flat: list[tuple[int, int, str]] = [] # (unit 下标, line 下标, 性别)
955
+ texts: list[str] = []
956
+ for ui, u in enumerate(units):
957
+ for li, ln in enumerate(u.lines):
958
+ f, m = cnorm(clean(ln.raw, "F")), cnorm(clean(ln.raw, "M"))
959
+ flat.append((ui, li, "F" if f != m else ""))
960
+ texts.append(f)
961
+ if f != m:
962
+ flat.append((ui, li, "M"))
963
+ texts.append(m)
964
+ return units, flat, texts, Index(texts)
965
+
966
+
967
+ USE_CACHE = True
968
+ """`library()` 用不用盘上缓存(`--no-cache` 关掉)。"""
969
+
970
+
971
+ def build(game: str, obs: Path, qmin: int, min_contain: float, max_units: int,
972
+ min_anchors: int, solo_len: int, dup_share: float = 0.5, dup_span: float = 60.0,
973
+ claim_back: float = 120.0, claim_fwd: float = 30.0, short_min: int = 4) -> dict:
974
+ t0 = time.time()
975
+ units, flat, texts, idx, cache = library(game)
976
+ n_lines = sum(len(u.lines) for u in units)
977
+ print(f"[{game}] 库:{len(units):,} 个对话单元 / {n_lines:,} 行;"
978
+ f"索引 {len(idx.h):,} 个 {K}-gram({cache},{time.time()-t0:.0f}s)")
979
+
980
+ ot = obs_texts(obs)
981
+ qs: dict[str, list[tuple[float, float]]] = defaultdict(list) # 归一后合并同一串
982
+ qs_short: dict[str, list[tuple[float, float]]] = defaultdict(list)
983
+ for raw, occ in ot.items():
984
+ q = cnorm(raw)
985
+ if len(q) >= qmin:
986
+ qs[q].extend(occ)
987
+ elif len(q) >= short_min:
988
+ qs_short[q].extend(occ)
989
+ print(f" obs:{len(ot):,} 条去重文本,其中内容 ≥{qmin} 字的 {len(qs):,} 条、"
990
+ f"{short_min}–{qmin-1} 字的 {len(qs_short):,} 条(后者只给圈中单元里的短行当锚点)")
991
+
992
+ hits: list[tuple[str, list[int], list]] = [] # (查询串, 命中的 flat 下标, 出现的 (时刻, y))
993
+ n_hit = 0
994
+ for q, tm in qs.items():
995
+ sc = {j: contained(q, texts[j]) for j in idx.query(q)}
996
+ ok = [j for j, c in sc.items() if c >= min_contain]
997
+ if ok:
998
+ n_hit += 1
999
+ # 同一个单元里只留最像的那几行:一个框只是一行的一部分,公共部分会同时"包含于"
1000
+ # 两种说法(绝区零的リン / アキラ两行、原神的重演拷贝),都算锚点的话
1001
+ # 分支就判不出谁走过。
1002
+ best: dict[int, float] = {}
1003
+ for j in ok:
1004
+ best[flat[j][0]] = max(best.get(flat[j][0], 0.0), sc[j])
1005
+ ok = [j for j in ok if sc[j] >= best[flat[j][0]]]
1006
+ hits.append((q, ok, tm))
1007
+ print(f" obs 文本在库里找得到的:{n_hit:,}/{len(qs):,}({n_hit/max(1,len(qs)):.1%})"
1008
+ f"({time.time()-t0:.0f}s)")
1009
+
1010
+ # 性别:只看两种写法读法不同的行,两种各算一次包含度,**高的那种**投一票、打平不投。
1011
+ # 第一版是"哪种写法进了候选就投哪种",而性别差异通常只占一两个字,两种都过包含度门,
1012
+ # 于是两边都得票——gi1 投出 F 29 / M 18 这种没有意义的数。
1013
+ pair: dict[tuple[int, int], dict[str, int]] = defaultdict(dict)
1014
+ for j, (ui, li, g) in enumerate(flat):
1015
+ if g:
1016
+ pair[(ui, li)][g] = j
1017
+ votes = Counter()
1018
+ for q, ok, _ in hits:
1019
+ for key in {flat[j][:2] for j in ok if flat[j][2]}:
1020
+ cf, cm = (contained(q, texts[pair[key][g]]) for g in "FM")
1021
+ if cf != cm:
1022
+ votes["F" if cf > cm else "M"] += 1
1023
+ gender = "M" if votes["M"] > votes["F"] else "F"
1024
+ print(f" 主角性别写法投票:F {votes['F']} / M {votes['M']} -> 取 {gender}")
1025
+
1026
+ # 锚点:每行记读到它的查询出现过的全部时刻与查询数;只挂在"唯一性够"的查询上的才用来圈单元。
1027
+ # 撞了 >max_units 个单元的常用句仍然记成锚点(时段、走没走分支、播没播都用它),
1028
+ # 但锚点上记下有没有唯一查询:只靠常用句锚住的行,"obs 读到过"说的是别处(audit-6)。
1029
+ # 框的位置(判字幕带)同理只记唯一查询的——常用句的框可能在画面任何地方
1030
+ anchors: dict[tuple[int, int], list] = {}
1031
+ unit_anchor_lines: dict[int, set] = defaultdict(set)
1032
+ unit_solo: set[int] = set()
1033
+ n_ambig = 0
1034
+ for q, ok, occ in hits:
1035
+ us = {flat[j][0] for j in ok}
1036
+ uniq = len(us) <= max_units
1037
+ n_ambig += not uniq
1038
+ for j in {flat[j][:2]: j for j in ok}.values(): # 两种性别写法算一行
1039
+ ui, li, _ = flat[j]
1040
+ v = anchors.setdefault((ui, li), [[], 0, 0, []])
1041
+ v[0].extend(t for t, _ in occ)
1042
+ v[1] += 1
1043
+ v[2] += uniq
1044
+ if uniq:
1045
+ v[3].extend(y for _, y in occ)
1046
+ unit_anchor_lines[ui].add(li)
1047
+ if len(q) >= solo_len:
1048
+ unit_solo.add(ui)
1049
+ picked = [ui for ui, ls in unit_anchor_lines.items()
1050
+ if len(ls) >= min_anchors or ui in unit_solo or units[ui].solo]
1051
+ print(f" 圈中 {len(picked)} 个单元(≥{min_anchors} 行锚点,或一条 ≥{solo_len} 字的独占锚点);"
1052
+ f"撞了 >{max_units} 个单元而不参与圈选的查询 {n_ambig} 条")
1053
+
1054
+ anchors = {k: (episodes(ts), n, nu, ys) for k, (ts, n, nu, ys) in anchors.items()}
1055
+ # 字幕带只拿"本来就该在带里"的行估:库里已经知道不在带里的(选项、黑屏…)不投票
1056
+ band = band_of([anchors[(ui, li)][3] for ui in picked for li, ln in enumerate(units[ui].lines)
1057
+ if (ui, li) in anchors and ln.kind not in NON_BAND_KINDS])
1058
+ print(" 字幕带(obs 锚点框 y 中心的众数窗口,画面高的比例):"
1059
+ + (f"{band[0]:.3f}–{band[1]:.3f}" if band else f"锚住的行不到 {BAND_MIN_LINES} 条,不判"))
1060
+ n_short = short_anchors(units, picked, anchors, qs_short, max_units, gender)
1061
+ print(f" 圈中单元里的短行({short_min}–{qmin-1} 字):{n_short} 行在本单元的时段内按逐字相等找到了锚点"
1062
+ f"(不参与圈单元、也不参与估字幕带;作用是给它们位置证据与时段)")
1063
+
1064
+ def first_t(ui):
1065
+ return min(anchors[(ui, li)][0][0][0] for li in range(len(units[ui].lines))
1066
+ if (ui, li) in anchors)
1067
+ picked.sort(key=first_t)
1068
+
1069
+ out_units = []
1070
+ n_variant = 0
1071
+ for ui in picked:
1072
+ u = units[ui]
1073
+ # 分支:独占集里有锚点的选项算走过
1074
+ walked: dict[str, set] = defaultdict(set)
1075
+ for li, ln in enumerate(u.lines):
1076
+ for g, o in ln.memberships:
1077
+ if ln.kind != "Choice" and (ui, li) in anchors:
1078
+ walked[g].add(o)
1079
+ on = [li for li, ln in enumerate(u.lines) if ln.kind != "Choice" and (ui, li) in anchors]
1080
+ out_span = unplayed(on, {li: anchors[(ui, li)][0] for li in on}, len(u.lines))
1081
+ lines = []
1082
+ vgroups: dict[str, list[dict]] = defaultdict(list)
1083
+ for li, ln in enumerate(u.lines):
1084
+ kind = ln.kind
1085
+ if kind != "Choice" and li in out_span:
1086
+ kind = "OutOfSpan"
1087
+ elif kind != "Choice" and ln.memberships:
1088
+ kind = branch_kind(ln.memberships, walked) or kind
1089
+ text = clean(ln.raw, gender)
1090
+ if not text:
1091
+ continue
1092
+ a = anchors.get((ui, li))
1093
+ lines.append({"key": ln.key, "kind": kind, "role": ln.role, "speaker": ln.speaker,
1094
+ "speaker_cn": ln.speaker_cn,
1095
+ "text": text, "cn": clean(ln.cn or "", gender) or None,
1096
+ "anchor": None if a is None else
1097
+ {"t0": a[0][0][0], "t1": a[0][-1][1], "queries": a[1],
1098
+ "unique": a[2] > 0,
1099
+ "y": round(float(np.median(a[3])), 3) if a[3] else None,
1100
+ "in_band": None if not (a[3] and band) else
1101
+ any(band[0] <= y <= band[1] for y in a[3]),
1102
+ "episodes": [[round(s, 2), round(e, 2)] for s, e in a[0]]}})
1103
+ if ln.variant and kind not in ("Choice", "OutOfSpan"):
1104
+ vgroups[ln.variant].append(lines[-1])
1105
+ n_variant += fold_variants(list(vgroups.values()))
1106
+ out_units.append({"uid": u.uid, "title": u.title,
1107
+ "t0": round(first_t(ui), 2), "lines": lines})
1108
+ n_dup = mark_duplicates(out_units, dup_share, dup_span)
1109
+ for u in out_units:
1110
+ # 代表行被判成重演拷贝时,同组的另一种说法也是拷贝——否则它指向一条不在对齐序列里的行
1111
+ dup = {ln["key"] for ln in u["lines"] if ln["kind"] == "Duplicate"}
1112
+ for ln in u["lines"]:
1113
+ if ln.get("variant_of") in dup:
1114
+ ln["kind"] = "Duplicate"
1115
+ n_dup += 1
1116
+ n_off = mark_offband(out_units)
1117
+ kinds = Counter(ln["kind"] for u in out_units for ln in u["lines"])
1118
+ if n_variant:
1119
+ print(f" 变体组:{n_variant} 行记成 Variant(同一句话的另一种说法,一组只算一个分母条目)")
1120
+ print(f" 重演拷贝:{n_dup} 行记成 Duplicate(和前面单元重复的内容行占本单元 ≥{dup_share:.0%},"
1121
+ f"且 obs 里只上屏过一次)")
1122
+ print(f" 不在字幕带:{n_off} 行记成 OffBand(锚住它的框没有一个落在字幕带里)")
1123
+ print(" 圈出的剧本按类型:" + " ".join(f"{k} {v}" for k, v in kinds.most_common()))
1124
+ # 整库的说话人 日文 -> 中文(同名取出现最多的写法):叠加 ASS 名牌层的译文从这里取。只看圈中的剧本行不够——
1125
+ # 原神任务对话的剧本行多半没有说话人,gi-s2 的 4 个名牌只看剧本行一个都翻不出,整库查得到 3 个(2026-09-14)
1126
+ spk: dict[str, Counter] = defaultdict(Counter)
1127
+ for u in units:
1128
+ for ln in u.lines:
1129
+ if ln.speaker and ln.speaker_cn:
1130
+ spk[ln.speaker][ln.speaker_cn] += 1
1131
+ # 说话人表查不到的名牌(原神隐藏名字的 NPC、星铁 / 绝区零没有说话人的场景)再查短名词表
1132
+ terms, tstat = term_table(game, ot, gender) if game in GAMES else (None, {"table": False})
1133
+ print(" 短名词表:" + (f"obs 里 {tstat['obs_keys']:,} 个候选键、查到 {tstat['keys']:,} 个,"
1134
+ f"其中对多个中文的 {tstat['ambiguous']}(多数派 <60% 的 {tstat['majority_below_60']})"
1135
+ if tstat["table"] else "文本包里没有(旧包),名牌层只查说话人表"))
1136
+ return {"units": out_units, "sequence": sequence(out_units, claim_back, claim_fwd),
1137
+ "speakers": {k: c.most_common(1)[0][0] for k, c in sorted(spk.items())},
1138
+ **({"terms": terms} if terms is not None else {}),
1139
+ "gender": gender, "gender_votes": dict(votes),
1140
+ "stats": {"db_units": len(units), "db_lines": n_lines,
1141
+ "obs_texts": len(ot), "queries": len(qs), "queries_hit": n_hit,
1142
+ "queries_ambiguous": n_ambig, "units_picked": len(picked),
1143
+ "short_anchors": n_short, "terms": tstat,
1144
+ "band": [round(band[0], 3), round(band[1], 3)] if band else None,
1145
+ "kinds": dict(kinds)}}
1146
+
1147
+
1148
+ def main() -> int:
1149
+ ap = argparse.ArgumentParser(prog="flowocr-gamescript", description="从游戏文本包里圈出这段视频打过的对话,写成剧本 JSON(匹配原文用)。"
1150
+ "设计与判据见本模块的文档串。")
1151
+ ap.add_argument("obs", help="第 1 步的观测 obs.jsonl(圈对话单元只用它)")
1152
+ ap.add_argument("--out", required=True, help="剧本 JSON")
1153
+ ap.add_argument("--game", choices=sorted(GAMES), default=None,
1154
+ help="默认按 obs 文件名前缀猜(gi/hsr/zzz)")
1155
+ ap.add_argument("--qmin", type=int, default=8,
1156
+ help="obs 文本内容字数达到这么多才拿去检索。短串在 30 万行里到处都是")
1157
+ ap.add_argument("--min-contain", type=float, default=0.85,
1158
+ help="obs 文本有多大比例落在库里那一行里才算读到(一个框只是一行的一部分)")
1159
+ ap.add_argument("--max-units", type=int, default=3,
1160
+ help="一条查询撞中超过这么多个单元,就不拿它圈单元(常用句)")
1161
+ ap.add_argument("--min-anchors", type=int, default=2,
1162
+ help="一个单元至少有几行被锚住才算这段视频打过它")
1163
+ ap.add_argument("--solo-len", type=int, default=20,
1164
+ help="或者:只有一行锚住,但那条查询内容 ≥ 这么多字且独占")
1165
+ ap.add_argument("--short-anchor-min", type=int, default=4,
1166
+ help="圈中单元里的短行:内容达到这么多字才拿去和 obs 逐字比(见 short_anchors)")
1167
+ ap.add_argument("--claim-back", type=float, default=120.0,
1168
+ help="没锚点的行借前一个锚点的上屏时段,往后放宽这么多秒(见 sequence)")
1169
+ ap.add_argument("--claim-fwd", type=float, default=30.0,
1170
+ help="有锚点的行的每次上屏两头各外扩这么多秒(主轨 cue 可能早于 obs 锚点)")
1171
+ ap.add_argument("--no-cache", action="store_true",
1172
+ help="读库和建索引不走 out/_cache/gamescript(对账用,见 library)")
1173
+ ap.add_argument("--gametext", default=None,
1174
+ help="文本包:一个 .zip、带 manifest.json 的目录、或装着若干包的目录(默认 paths.gametext_root(),"
1175
+ "即 FLOWOCR_GAMETEXT 或数据根的 gametext/)")
1176
+ a = ap.parse_args()
1177
+ global USE_CACHE, GAMETEXT
1178
+ USE_CACHE = not a.no_cache
1179
+ GAMETEXT = a.gametext
1180
+ if a.qmin < QMIN_FLOOR:
1181
+ ap.error(f"--qmin 不能小于 {QMIN_FLOOR}:更短的查询在检索这一步命不中或只命中一半(见 QMIN_FLOOR)")
1182
+
1183
+ obs = Path(a.obs)
1184
+ game = a.game or TAG_GAME.get(obs.stem.split("-")[0].rstrip("0123456789"))
1185
+ if not game:
1186
+ raise SystemExit(f"从 {obs.name} 猜不出是哪款游戏,给 --game")
1187
+ doc = build(game, obs, a.qmin, a.min_contain, a.max_units, a.min_anchors, a.solo_len,
1188
+ claim_back=a.claim_back, claim_fwd=a.claim_fwd, short_min=a.short_anchor_min)
1189
+ from flowocr.analyze import build_tracks as bt
1190
+ gt = bundle_of(game).provenance()
1191
+ print(f" 文本包:{gt['source']}({game} {gt['version']},指纹 {gt['fingerprint'][7:19]})")
1192
+ if not gt["integrity_checked"]:
1193
+ print(" ⚠ 文本包:上游快照的完整性检查没过或没做(工作区是否等于所记 commit,见包里 report.json 的 upstream.integrity_warnings)")
1194
+ doc = {"schema": SCHEMA,
1195
+ "provenance": {"tool": "gamescript.py", "version": bt.version(), "git_head": bt.git_head(),
1196
+ "code_fp": bt.code_fp(CODE_FP_FILES),
1197
+ "argv": [bt.portable(x) for x in sys.argv[1:]], "obs": bt.portable(obs),
1198
+ "obs_mtime": obs.stat().st_mtime, "game": game,
1199
+ "gametext": gt,
1200
+ "params": {"qmin": a.qmin, "min_contain": a.min_contain,
1201
+ "max_units": a.max_units, "min_anchors": a.min_anchors,
1202
+ "solo_len": a.solo_len, "K": K, "episode_gap": EPISODE_GAP,
1203
+ "band_win": BAND_WIN, "band_pad": BAND_PAD,
1204
+ "short_anchor_min": a.short_anchor_min,
1205
+ "short_anchor_pad": SHORT_ANCHOR_PAD,
1206
+ "claim_back": a.claim_back, "claim_fwd": a.claim_fwd}},
1207
+ **doc}
1208
+ p = Path(a.out)
1209
+ p.parent.mkdir(parents=True, exist_ok=True)
1210
+ p.write_text(json.dumps(doc, ensure_ascii=False, indent=1), encoding="utf-8")
1211
+ print(f"-> {p}")
1212
+ return 0
1213
+
1214
+
1215
+ if __name__ == "__main__":
1216
+ # **先按模块名导入自己再跑**:直接跑 main() 的话 `Unit` / `Line` 挂在 `__main__` 下,
1217
+ # `library()` pickle 进缓存的类名就是 `__main__.Unit`——之后别的工具 `import gamescript` 再读同一份缓存,
1218
+ # 按 `__main__` 找不到类,当场 AttributeError(2026-09-14 复查 try-list 时实测出来的)
1219
+ from flowocr.analyze import gamescript
1220
+ raise SystemExit(gamescript.main())