flowocr 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- flowocr/__init__.py +15 -0
- flowocr/analyze/__init__.py +8 -0
- flowocr/analyze/align.py +151 -0
- flowocr/analyze/build_tracks.py +2886 -0
- flowocr/analyze/cluster_layers.py +155 -0
- flowocr/analyze/game_align.py +652 -0
- flowocr/analyze/gamescript.py +1220 -0
- flowocr/analyze/gtdbundle.py +306 -0
- flowocr/analyze/match.py +67 -0
- flowocr/analyze/matchers/__init__.py +5 -0
- flowocr/analyze/matchers/gametext.py +70 -0
- flowocr/analyze/merge_nameplate.py +178 -0
- flowocr/analyze/models/slot_pair.json +148 -0
- flowocr/analyze/nameplate.py +148 -0
- flowocr/analyze/pair_features.py +168 -0
- flowocr/analyze/pair_model.py +127 -0
- flowocr/analyze/refine_boundaries.py +409 -0
- flowocr/analyze/script_align.py +641 -0
- flowocr/analyze/scriptmatch.py +1018 -0
- flowocr/analyze/slot_learned.py +306 -0
- flowocr/analyze/slot_lines.py +251 -0
- flowocr/analyze/slot_modes.py +143 -0
- flowocr/analyze/slot_pairs.py +562 -0
- flowocr/analyze/slot_veto.py +104 -0
- flowocr/analyze/uigate.py +351 -0
- flowocr/artifacts/__init__.py +5 -0
- flowocr/artifacts/evalkit.py +123 -0
- flowocr/artifacts/matchedio.py +75 -0
- flowocr/artifacts/srtio.py +166 -0
- flowocr/artifacts/tracksio.py +345 -0
- flowocr/extensions.py +77 -0
- flowocr/extract/__init__.py +8 -0
- flowocr/extract/childproc.py +118 -0
- flowocr/extract/decode_proc.py +511 -0
- flowocr/extract/decode_shards.py +574 -0
- flowocr/extract/detpost.py +215 -0
- flowocr/extract/edge_proc.py +314 -0
- flowocr/extract/edge_refine.py +598 -0
- flowocr/extract/fast_det.py +214 -0
- flowocr/extract/ffcheck.py +87 -0
- flowocr/extract/framegrid.py +265 -0
- flowocr/extract/framesource.py +1264 -0
- flowocr/extract/ocr_args.py +754 -0
- flowocr/extract/ocr_complete.py +231 -0
- flowocr/extract/ocr_parallel.py +208 -0
- flowocr/extract/ort_server.py +632 -0
- flowocr/extract/ortclient.py +363 -0
- flowocr/extract/ptsclock.py +150 -0
- flowocr/extract/recdecode.py +78 -0
- flowocr/extract/recort.py +162 -0
- flowocr/extract/recpack.py +94 -0
- flowocr/extract/recpool.py +206 -0
- flowocr/extract/recprep.py +71 -0
- flowocr/extract/refine_video.py +271 -0
- flowocr/extract/regions.py +387 -0
- flowocr/extract/reuse_v2.py +593 -0
- flowocr/extract/run_groups.py +247 -0
- flowocr/extract/run_ocr2.py +1605 -0
- flowocr/extract/supervisor.py +261 -0
- flowocr/extract/timeline.py +84 -0
- flowocr/extract/typewriter_fuse.py +394 -0
- flowocr/models.py +263 -0
- flowocr/output/__init__.py +3 -0
- flowocr/output/export.py +205 -0
- flowocr/output/layout.py +210 -0
- flowocr/output/presets/__init__.py +6 -0
- flowocr/output/presets/_overlay.py +40 -0
- flowocr/output/presets/default.py +21 -0
- flowocr/output/presets/default_all.py +20 -0
- flowocr/output/presets/dev.py +21 -0
- flowocr/output/presets/matched_srt.py +32 -0
- flowocr/output/presets/script.py +20 -0
- flowocr/output/presets/srt_main.py +33 -0
- flowocr/output/render.py +82 -0
- flowocr/output/run_srt.py +83 -0
- flowocr/output/script.py +541 -0
- flowocr/paths.py +168 -0
- flowocr/provenance.py +119 -0
- flowocr/typeset/__init__.py +9 -0
- flowocr/typeset/__main__.py +38 -0
- flowocr/typeset/assfile.py +216 -0
- flowocr/typeset/core.py +821 -0
- flowocr/typeset/fx/__init__.py +11 -0
- flowocr-0.1.0.dist-info/METADATA +109 -0
- flowocr-0.1.0.dist-info/RECORD +91 -0
- flowocr-0.1.0.dist-info/WHEEL +5 -0
- flowocr-0.1.0.dist-info/entry_points.txt +10 -0
- flowocr-0.1.0.dist-info/licenses/LICENSE +674 -0
- flowocr-0.1.0.dist-info/licenses/LICENSES/Apache-2.0.txt +201 -0
- flowocr-0.1.0.dist-info/licenses/LICENSES/PP-OCRv6-NOTICE.md +18 -0
- flowocr-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,652 @@
|
|
|
1
|
+
"""拿 `flowocr.analyze.gamescript` 圈出来的剧本量一条轨:命中 / 三档漏条 / 重复认领 / 行首行尾缺字。
|
|
2
|
+
|
|
3
|
+
分桶与三档复用 `script_align.gap_report` 和 `evalkit.triviality`,**不重写第二份**。
|
|
4
|
+
和《魔裁》那套差三处,都是游戏直播的形态逼出来的:
|
|
5
|
+
|
|
6
|
+
1. **一条 cue 有几种读法**。原神 / 星铁的主轨 cue 是 `名牌 \\n 称号 \\n 正文…`
|
|
7
|
+
粘在一起的(gamestream-first-look 报告),剧本行只有正文;
|
|
8
|
+
末尾还常挂一个被吸进来的键位提示(`D`)。所以每条 cue 试"去掉开头 0–3 行 ×
|
|
9
|
+
去不去掉最后一行"几种读法,取和剧本最像的那个。不靠名字表——库里原神的说话人
|
|
10
|
+
只覆盖 4% 的节点,而且屏幕上的名牌和库里的 `speaker` 字段不是一回事。
|
|
11
|
+
2. **对齐按时间圈候选,不走游标**(`run` 的说明)。剧本行带 gamescript 从 obs 锚来的时刻,
|
|
12
|
+
各臂共用;时间只圈范围,命中仍看文本。主认领之后还有**第二遍**(`claim_contained`):
|
|
13
|
+
一条 cue 常常装着两行剧本(选项按钮 + 正文、并带把两个说话人并成一条),整行落在 cue 里
|
|
14
|
+
就算认领,但**不许和主认领占同一段字**——否则库里近似重复的另一行会被白送一个命中。
|
|
15
|
+
3. **分母里排掉 `gamescript.NON_BAND_KINDS`**(选项 / 黑屏 / 没走的分支 / 单元外 / 重演拷贝 /
|
|
16
|
+
读到过但不在字幕带,判据见那边)。排掉的条目仍参与对齐、只是不计分母;它们若被主轨命中会单独报出来——
|
|
17
|
+
那说明分支判错了,或者选项文本真的进了这条轨。重演拷贝例外:它连序列都不进。
|
|
18
|
+
绝区零的变体组一组只算一个分母条目,任一种说法被认领都记在代表行上(`score`)。
|
|
19
|
+
4. **长 gap 照样算疑似真漏**(`gap_report(long_is_branch=False)`,audit-6):《魔裁》那边
|
|
20
|
+
长 gap 被解释成"分支没走到",而这里分支和没播的一截已经排出分母,长 gap 没有分支解释。
|
|
21
|
+
于是疑似真漏 = 全部未命中 = 分母 − 命中,逐档不经过 gap 分桶,分母各臂相同,三档可以跨臂比。
|
|
22
|
+
|
|
23
|
+
**这里量的是原始轨,不过匹配器。** 所以"对不上剧本"那一档是**幽灵条目的上限**,
|
|
24
|
+
而且比《魔裁》那边更松:库收什么因游戏而异(原神只有任务对话;星铁含全量台词;
|
|
25
|
+
绝区零含散条目),库外的文字——UI、路人、主播自己压的字幕——都落在这一档。
|
|
26
|
+
**重复认领**这里是原始轨口径:一条剧本行被几条 cue 认领,多出来的次数之和
|
|
27
|
+
(hit_delta 那个是匹配器输出口径,两者不可混比)。
|
|
28
|
+
|
|
29
|
+
用法:见 `CLI_DESCRIPTION`。
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import argparse
|
|
35
|
+
import copy
|
|
36
|
+
import json
|
|
37
|
+
import re
|
|
38
|
+
from collections import Counter
|
|
39
|
+
from dataclasses import dataclass
|
|
40
|
+
from difflib import SequenceMatcher
|
|
41
|
+
from pathlib import Path
|
|
42
|
+
|
|
43
|
+
from flowocr.artifacts import evalkit # noqa: E402
|
|
44
|
+
from flowocr.analyze import gamescript as GS # noqa: E402
|
|
45
|
+
from flowocr.analyze import script_align as SA # noqa: E402
|
|
46
|
+
from flowocr.artifacts import srtio # noqa: E402
|
|
47
|
+
|
|
48
|
+
CLI_DESCRIPTION = """拿 gamescript 圈出来的剧本量一条轨(用途 2 的尺):命中 / 三档疑似真漏 / 重复认领 / 行首行尾缺字。
|
|
49
|
+
|
|
50
|
+
分母排掉选项、黑屏旁白、没走到的分支、没播的那截、重演拷贝和不在字幕带的行;量的是原始轨、不过匹配器,
|
|
51
|
+
所以"对不上剧本"那一档是幽灵条目的上限(UI、路人、主播自己的字幕都落在这里)。
|
|
52
|
+
|
|
53
|
+
python -m flowocr.analyze.game_align out/gametext/gi-s2-ref.json --subs out/gs-gi-s2/gi-s2-main.srt
|
|
54
|
+
# 两条臂比命中集合(按归一化文本的多重集):
|
|
55
|
+
python -m flowocr.analyze.game_align <ref.json> --subs <A 主轨> --vs <B 主轨>"""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class Ref:
|
|
60
|
+
doc: dict
|
|
61
|
+
script: list[SA.Entry]
|
|
62
|
+
"""对齐序列(gamescript 排好的 `sequence`:按 obs 时间排到行一级,重演拷贝不在里面)。"""
|
|
63
|
+
windows: list[list]
|
|
64
|
+
"""每行可以被认领的时段——理由见 gamescript.sequence。"""
|
|
65
|
+
role: dict[str, str]
|
|
66
|
+
variant_of: dict[str, str]
|
|
67
|
+
"""绝区零变体行 -> 它那组的代表行(gamescript `fold_variants`)。一组只算一个分母条目。"""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def load_ref(p: Path) -> Ref:
|
|
71
|
+
doc = json.loads(p.read_text(encoding="utf-8"))
|
|
72
|
+
if doc.get("schema") != GS.SCHEMA:
|
|
73
|
+
raise SystemExit(f"{p} 的 schema 不是 {GS.SCHEMA}:{doc.get('schema')!r}(重跑 gamescript)")
|
|
74
|
+
by_key, role, variant_of = {}, {}, {}
|
|
75
|
+
for u in doc["units"]:
|
|
76
|
+
for ln in u["lines"]:
|
|
77
|
+
if ln["kind"] == "Duplicate": # 同一个节点的第二份 key 相同,别覆盖原件
|
|
78
|
+
continue
|
|
79
|
+
by_key[ln["key"]] = SA.Entry(key=ln["key"], kind=ln["kind"], cmd="", text=ln["text"],
|
|
80
|
+
src=u["uid"])
|
|
81
|
+
role[ln["key"]] = ln["role"]
|
|
82
|
+
if ln.get("variant_of"):
|
|
83
|
+
variant_of[ln["key"]] = ln["variant_of"]
|
|
84
|
+
return Ref(doc=doc, script=[by_key[k] for k, _, _ in doc["sequence"]],
|
|
85
|
+
windows=[w for _, _, w in doc["sequence"]], role=role, variant_of=variant_of)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
OBS_READ, OBS_COMMON, OBS_NONE, OBS_SHORT = "obs 读到过", "只撞常用句", "obs 没读到", "太短判不了"
|
|
89
|
+
OBS_SIDES = (OBS_READ, OBS_COMMON, OBS_NONE, OBS_SHORT)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def obs_side(doc: dict) -> dict[str, str]:
|
|
93
|
+
"""每行剧本在 obs 里读到过没有(gamescript 记的锚点,任何区域、任何一帧)。
|
|
94
|
+
|
|
95
|
+
这是**这一处**的判据,不是全局搜:锚点是 obs 里某个框的文字落在**这一行**里
|
|
96
|
+
(同一单元内只锚最像的那行),而单元本身是按时间圈出来的。内容不到
|
|
97
|
+
`qmin` 字的行从来不拿去检索,没有锚点不说明什么,单列。
|
|
98
|
+
* 读到过而主轨没命中 = 丢在**轨这一层**(没进主轨 / cue 读法对不上);
|
|
99
|
+
* 只撞常用句 = 锚住它的查询**全都**撞了 >`max_units` 个单元(`言ってみるといい。` 这种到处都有的
|
|
100
|
+
句子,别处的同句也会锚上来)——算不得"这一处读到过",单列(audit-6);
|
|
101
|
+
* 没读到 = 丢在 **OCR 之前**(没出框、被遮挡、显示太短没采到……分不开,要摆帧)。"""
|
|
102
|
+
qmin = doc["provenance"]["params"]["qmin"]
|
|
103
|
+
out = {}
|
|
104
|
+
for u in doc["units"]:
|
|
105
|
+
for ln in u["lines"]:
|
|
106
|
+
if ln["kind"] == "Duplicate": # 同 load_ref:同一节点的第二份别覆盖原件
|
|
107
|
+
continue
|
|
108
|
+
a = ln["anchor"]
|
|
109
|
+
out[ln["key"]] = (OBS_READ if a and a["unique"] else OBS_COMMON if a else
|
|
110
|
+
OBS_NONE if len(GS.cnorm(ln["text"])) >= qmin else OBS_SHORT)
|
|
111
|
+
return out
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def resolve(p: str) -> tuple[Path, dict]:
|
|
115
|
+
"""轨可以直接给 SRT,也可以给 `*-tracks.json`——后者从 provenance 取 `main_srt`
|
|
116
|
+
(**不许 glob 猜**,字典序事故见 docs/dev-guide/verification.md「追溯」),并带回产它的 `git_head` / `code_fp`。"""
|
|
117
|
+
path = Path(p)
|
|
118
|
+
if not path.name.endswith("-tracks.json"):
|
|
119
|
+
return path, {}
|
|
120
|
+
from flowocr.artifacts import tracksio
|
|
121
|
+
prov = tracksio.load(path)["provenance"]
|
|
122
|
+
if not prov.get("main_srt"):
|
|
123
|
+
raise SystemExit(f"{path} 没有主轨(provenance.main_srt 为空)")
|
|
124
|
+
return path.parent / prov["main_srt"], {k: prov.get(k) for k in ("git_head", "code_fp")}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def code_same(a: dict, b: dict) -> bool | None:
|
|
128
|
+
"""两臂的轨是不是同一份 build_tracks 产的:比**代码指纹**不比 HEAD(audit-6——HEAD 移动、
|
|
129
|
+
代码没变时 git_head 会不同;反过来 `+dirty` 相同也证明不了代码相同)。没指纹的算不知道。"""
|
|
130
|
+
fa, fb = (a or {}).get("code_fp"), (b or {}).get("code_fp")
|
|
131
|
+
return None if not (fa and fb) else fa == fb
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def read_cues(p: Path) -> list[tuple[float, str, str]]:
|
|
135
|
+
"""(起点秒, 各行用换行拼起来的原文, 文件名)。**保留行结构**,读法要按行切。"""
|
|
136
|
+
return [(c.start, "\n".join(c.lines), p.name) for c in srtio.read_subs([p])]
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
HEAD_MIN_BODIES = 3
|
|
140
|
+
"""一行在整条轨里作为**非末行**、配过至少这么多种不同的末行,就是"开头行"(`head_lines`)。"""
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def head_lines(subs) -> frozenset[str]:
|
|
144
|
+
"""整条轨里的**开头行**(归一化后):名牌、称号、常驻在正文上面的角标。
|
|
145
|
+
|
|
146
|
+
判据只看这条轨自己:一行作为非末行出现、而且配过 ≥ `HEAD_MIN_BODIES` 种不同的末行(正文)。
|
|
147
|
+
名牌天然如此——一个说话人要说很多句(gi1 `カチーナ` 配过 271 种正文,zzz `シーシイア` 217 种);
|
|
148
|
+
真台词作为非末行出现时,下面跟的几乎总是同一句的续行。不靠名字表(库里原神的说话人只覆盖 4% 的节点)。
|
|
149
|
+
|
|
150
|
+
为什么需要(2026-09-12,matcher 计划那 12 条"只由 `extra` 认领"的行逐条看出来的):
|
|
151
|
+
①`カチーナ / はあ、はあ` 这种**正文很短**的 cue,去掉末行的那种读法只剩名牌,
|
|
152
|
+
和剧本里恰好存在的喊名字的台词 `カチーナ!` 相似度 0.89,比短正文本身还高,于是主认领被名牌抢走、
|
|
153
|
+
输出成"卡齐娜!"——认错,不只是丢;②第二遍包含认领把名牌 `シーシイア` 整行认成 `シーシィア…?`。"""
|
|
154
|
+
from collections import defaultdict
|
|
155
|
+
bodies: dict[str, set[str]] = defaultdict(set)
|
|
156
|
+
span: dict[str, list[float]] = {}
|
|
157
|
+
for t, raw, _ in subs:
|
|
158
|
+
ls = [x for x in raw.split("\n") if x.strip()]
|
|
159
|
+
for x in ls[:-1]:
|
|
160
|
+
k = SA.norm(x)
|
|
161
|
+
if k and len(k) <= HEAD_MAX_CHARS:
|
|
162
|
+
bodies[k].add(SA.norm(ls[-1]))
|
|
163
|
+
lo, hi = span.get(k, (t, t))
|
|
164
|
+
span[k] = [min(lo, t), max(hi, t)]
|
|
165
|
+
return frozenset(k for k, v in bodies.items()
|
|
166
|
+
if span[k][1] - span[k][0] >= HEAD_MIN_SPAN
|
|
167
|
+
and distinct_bodies(v, HEAD_MIN_BODIES) >= HEAD_MIN_BODIES)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
HEAD_MAX_CHARS = 16
|
|
171
|
+
"""开头行最多这么多字(归一化后)。名牌、称号都短(最长见过 `「プラックウルフ」ロームル` 12 字);
|
|
172
|
+
两行正文的第一行几乎总比这长——长度门是 `distinct_bodies` 之外的第二道保险。"""
|
|
173
|
+
|
|
174
|
+
HEAD_MIN_SPAN = 30.0
|
|
175
|
+
"""开头行在轨里首末两次出现至少隔这么多秒。**第三道保险**:名牌跟着说话人出现在一场又一场对话里;
|
|
176
|
+
正文的第一行只在它自己那几秒里出现。zzz-s2 `待った待った!刀を納めろ! / ちゃんと言うから!` 这条
|
|
177
|
+
(11 字、第二行被打字机和 OCR 抖出几种互不相似的末行)前两道都拦不住,被当成开头行剥掉、命中 −1。"""
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def distinct_bodies(bodies: set[str], enough: int) -> int:
|
|
181
|
+
"""数"真正不同"的末行有几种:**互为前缀、或相似度 ≥ 0.6 的算同一种**,数到 `enough` 就停。
|
|
182
|
+
|
|
183
|
+
不这么数的话,两行正文的**第一行**会被当成开头行:打字机把第二行打成 `一緒に` / `一緒にテレ` /
|
|
184
|
+
`一緒にテレビでも…` 好几种半截、OCR 再抖几个字,它就"配过 ≥3 种末行"了——剥掉之后整条 cue 对不上
|
|
185
|
+
(2026-09-12 第一版剥开头行时实测:gi-s1 命中 30 → 23、zzz-s1 46 → 38,逐条看都是这个形状)。
|
|
186
|
+
|
|
187
|
+
代表是贪心挑的,结果依赖遍历顺序,所以**等长的按字符串排**:只按长度排时等长的几条按 set 的迭代顺序走,
|
|
188
|
+
而 str 的哈希每个进程随机加盐——同一份轨两次跑出的开头行集合不同(2026-09-26 gi-s1 名牌层 127 / 132 / 133 块)。"""
|
|
189
|
+
reps: list[str] = []
|
|
190
|
+
for b in sorted(bodies, key=lambda s: (-len(s), s)):
|
|
191
|
+
if any(r.startswith(b) or SequenceMatcher(None, b, r, autojunk=False).ratio() >= 0.6 for r in reps):
|
|
192
|
+
continue
|
|
193
|
+
reps.append(b)
|
|
194
|
+
if len(reps) >= enough:
|
|
195
|
+
break
|
|
196
|
+
return len(reps)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
INLINE_NAME = re.compile(r"^\s*([^::]{1,16}?)\s*[::]\s*(\S.*)$")
|
|
200
|
+
"""过场字幕的"名字: 台词"同一行格式(gi1 `カチーナ: ハッ!`、`パイモン: 気持ちいいな。…`)。"""
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def strip_heads(raw: str, heads: frozenset[str]) -> tuple[list[str], bool]:
|
|
204
|
+
"""剥掉 cue 里的开头行,返回 (剩下的行, 剥掉过没有)。
|
|
205
|
+
|
|
206
|
+
* 非末行里的开头行整行剥掉;末行是正文,不剥(`パイモン / カチーナ!` 的末行碰巧和名牌同字);
|
|
207
|
+
* **只剩一行、而它就是开头行**时整条不留:打字机起步那半秒只有名牌在屏上;
|
|
208
|
+
* 任一行是 `开头行: 台词` 的同一行格式,剥掉前缀。"""
|
|
209
|
+
ls = [x for x in raw.split("\n") if x.strip()]
|
|
210
|
+
if not heads or not ls:
|
|
211
|
+
return ls, False
|
|
212
|
+
out, hit = [], False
|
|
213
|
+
for k, x in enumerate(ls):
|
|
214
|
+
m = INLINE_NAME.match(x)
|
|
215
|
+
if m and SA.norm(m.group(1)) in heads:
|
|
216
|
+
x, hit = m.group(2), True
|
|
217
|
+
elif SA.norm(x) in heads and (k < len(ls) - 1 or len(ls) == 1):
|
|
218
|
+
hit = True
|
|
219
|
+
continue
|
|
220
|
+
out.append(x)
|
|
221
|
+
return out, hit
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def variants(raw: str, heads: frozenset[str] = frozenset()) -> list[str]:
|
|
225
|
+
"""去掉开头 0–3 行(名牌 / 称号 / 偶尔吸进来的一个角标)× 去不去掉最后一行(键位提示)。
|
|
226
|
+
|
|
227
|
+
**开头行(`head_lines`)先剥掉,不进任何读法**(`strip_heads`):留着它,名牌就会和剧本里喊名字的台词
|
|
228
|
+
配上——去掉末行只剩名牌(gi1 `カチーナ / はあ、はあ` → `カチーナ!`),或者名牌拼上打字机打出的半截
|
|
229
|
+
(gi2 `ナヴィア / (…すご` → `ナヴィア…`,0.82)。两例都是 2026-09-12 摆帧抓到的。
|
|
230
|
+
剥过名牌、末行又是有内容的正文(≥3 字)时,去掉末行后只剩 1–2 字碎片的读法也不要
|
|
231
|
+
(gi1 `カチーナ / L / はあ、はあ` 里夹的 OCR 垃圾行 `L`);没有名牌的 cue 不受这条影响。"""
|
|
232
|
+
ls, stripped = strip_heads(raw, heads)
|
|
233
|
+
content = evalkit.CLASSES[2]
|
|
234
|
+
body_last = stripped and bool(ls) and evalkit.triviality(ls[-1]) == content
|
|
235
|
+
out = []
|
|
236
|
+
for i in range(min(4, len(ls))):
|
|
237
|
+
for j in {len(ls), len(ls) - 1}:
|
|
238
|
+
if j <= i or (j < len(ls) and body_last
|
|
239
|
+
and all(evalkit.triviality(x) != content for x in ls[i:j])):
|
|
240
|
+
continue
|
|
241
|
+
v = SA.norm(" ".join(ls[i:j]))
|
|
242
|
+
if v and v not in out:
|
|
243
|
+
out.append(v)
|
|
244
|
+
return out
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def band(e: SA.Entry) -> bool:
|
|
248
|
+
return e.kind not in GS.NON_BAND_KINDS
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
ROLE_ROWS = 16
|
|
252
|
+
"""『按上游角色拆』那张表最多列几行(进分母的类总是全列)。"""
|
|
253
|
+
|
|
254
|
+
BIN = 10.0
|
|
255
|
+
"""按时段找候选时的分桶宽度(秒)。只影响速度,不影响结果。"""
|
|
256
|
+
|
|
257
|
+
CONTAIN_MIN, CONTAIN_MIN_CHARS = 0.85, 3
|
|
258
|
+
"""第二遍认领(`claim_contained`):剧本行有这么多内容字**整行**落在 cue 剩下的字里才算被认领,
|
|
259
|
+
且这一行的内容不少于这么多字(1–2 字没有判别力,同 owner 的三档)。包含度的门和
|
|
260
|
+
`gamescript` 检索那道一样。"""
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def rest_of(v: str, used: str) -> str:
|
|
264
|
+
"""cue 的一种读法 `v` 里,挖掉和 `used`(主认领那一行)对上的字之后剩下的。"""
|
|
265
|
+
if not used:
|
|
266
|
+
return v
|
|
267
|
+
keep = [True] * len(v)
|
|
268
|
+
for b in SequenceMatcher(None, v, used, autojunk=False).get_matching_blocks():
|
|
269
|
+
for k in range(b.a, b.a + b.size):
|
|
270
|
+
keep[k] = False
|
|
271
|
+
return "".join(c for c, k in zip(v, keep) if k)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def claim_contained(script: list[SA.Entry], subs, cand, sub_hit: list[int],
|
|
275
|
+
heads: frozenset[str] = frozenset()) -> list[list[int]]:
|
|
276
|
+
"""第二遍:一条 cue 认领**整行落在它里面**的其他剧本行;返回每条 cue 多认领了哪些行。
|
|
277
|
+
|
|
278
|
+
为什么需要(2026-09-11 数的):主认领是一对一的,而屏幕上一条 cue 常常装着两行剧本——
|
|
279
|
+
绝区零的选项按钮和正文同屏(`また後で来るね` + 对白)、星铁并带把两个说话人并进一条 cue;
|
|
280
|
+
还有一类是 cue 里混进了 OCR 垃圾行(`ユーちゃん / キュン / 1`),几种读法都够不着模糊阈值,
|
|
281
|
+
但那一行的字**整行都在**。这两种主轨都确实读到了,算漏是错的。
|
|
282
|
+
|
|
283
|
+
**不许和主认领占同一段字**(`rest_of`):不加这条,库里近似重复的另一行会被白送一个命中——
|
|
284
|
+
绝区零两位主角的两种说法、原神的重演拷贝、星铁改过字的同一句,都只上屏了一种
|
|
285
|
+
(实测:不挖掉主认领的字是 +57/+14/+9/+10,挖掉之后 +15/+7/+4/+8,差的那些逐条看全是这类)。
|
|
286
|
+
**认领了一行就把它的字也挖掉**:同一条 cue 里认两行时,第二行不许再用第一行的字——
|
|
287
|
+
否则库里同文本、不同编号的两行会被同一段字各认一次(zzz `いやああぁぁ————!`,2026-09-12)。
|
|
288
|
+
**开头行(名牌)先剥掉再找**(`head_lines`):名牌整行落在 cue 里,不等于有人喊了这个名字。"""
|
|
289
|
+
out: list[list[int]] = [[] for _ in subs]
|
|
290
|
+
cn = [GS.cnorm(e.text) for e in script]
|
|
291
|
+
for si, (t, raw, _) in enumerate(subs):
|
|
292
|
+
vs = [GS.cnorm(v) for v in variants(raw, heads)]
|
|
293
|
+
if not vs:
|
|
294
|
+
continue
|
|
295
|
+
used = cn[sub_hit[si]] if sub_hit[si] >= 0 else ""
|
|
296
|
+
rests = [rest_of(v, used) for v in vs]
|
|
297
|
+
for i in cand(t):
|
|
298
|
+
if script[i].matched >= 0 or len(cn[i]) < CONTAIN_MIN_CHARS:
|
|
299
|
+
continue
|
|
300
|
+
if max(GS.contained(cn[i], r) for r in rests) >= CONTAIN_MIN:
|
|
301
|
+
script[i].matched = si
|
|
302
|
+
out[si].append(i)
|
|
303
|
+
rests = [rest_of(r, cn[i]) for r in rests]
|
|
304
|
+
return out
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def run(script: list[SA.Entry], windows: list[list], subs, fuzzy: float) -> dict:
|
|
308
|
+
"""**按时间圈候选**的对齐:cue 在 t 秒,只和"t 落在它可认领时段里"的剧本行比
|
|
309
|
+
(时段由 gamescript 从 obs 定,见 `gamescript.sequence`);精确相等优先,否则取相似度
|
|
310
|
+
最高的(≥ fuzzy);平手时先给还没被认领的行、再按序列先后。
|
|
311
|
+
返回和 `script_align.align` 同形的统计,并就地填 `Entry.matched`(第一条认领它的 cue)。
|
|
312
|
+
|
|
313
|
+
为什么不用 `script_align.align` 的游标:那套在一条字幕带 + 有序剧本上是对的,
|
|
314
|
+
在这里会被短句带跑——一条 `うん` 的精确匹配把游标甩到很远的后面,之后读得几乎一样的
|
|
315
|
+
cue 也够不着(gi1 实测 `カチ一ナちゃんを助けて…` 在主轨里只差一个 `ー`,被记成漏)。
|
|
316
|
+
这里的剧本行本来就带 obs 时段,用它圈范围更稳;**时间只圈范围,命中仍看文本**。"""
|
|
317
|
+
from collections import defaultdict
|
|
318
|
+
for e in script:
|
|
319
|
+
e.matched, e.sim = -1, 0.0
|
|
320
|
+
bins: dict[int, list[int]] = defaultdict(list)
|
|
321
|
+
for i, ws in enumerate(windows):
|
|
322
|
+
for lo, hi in ws:
|
|
323
|
+
for b in range(int(lo // BIN), int(hi // BIN) + 1):
|
|
324
|
+
bins[b].append(i)
|
|
325
|
+
|
|
326
|
+
def cand(t: float) -> list[int]:
|
|
327
|
+
return sorted({i for i in bins.get(int(t // BIN), ())
|
|
328
|
+
if any(lo <= t <= hi for lo, hi in windows[i])})
|
|
329
|
+
sn = [SA.norm(e.text) for e in script]
|
|
330
|
+
sm = SequenceMatcher(None)
|
|
331
|
+
heads = head_lines(subs)
|
|
332
|
+
n_exact = n_fuzzy = n_none = 0
|
|
333
|
+
sub_hit: list[int] = []
|
|
334
|
+
for si, (t, raw, _) in enumerate(subs):
|
|
335
|
+
vs = variants(raw, heads)
|
|
336
|
+
if not vs:
|
|
337
|
+
sub_hit.append(-1)
|
|
338
|
+
continue
|
|
339
|
+
cnd = cand(t)
|
|
340
|
+
exact = [i for i in cnd if sn[i] in vs]
|
|
341
|
+
# 逐字相等的全是带外行(选项按钮)时,把**只差标点**的带内行也拉进来一起挑:OCR 读漏了省略号,
|
|
342
|
+
# 台词 `「未着」……` 就只剩按钮 `「未着」` 逐字相等;主轨是字幕带,认成选项就把正文换成了按钮的译文
|
|
343
|
+
# (hsr 2:10,2026-09-12 owner 看预览指出)。只在这种情况触发,别的打平照旧
|
|
344
|
+
if exact and not any(band(script[i]) for i in exact):
|
|
345
|
+
cvs = {GS.cnorm(v) for v in vs}
|
|
346
|
+
exact += [i for i in cnd if i not in exact and band(script[i]) and GS.cnorm(script[i].text) in cvs]
|
|
347
|
+
if exact:
|
|
348
|
+
# 先给还没被认领的:同文本两行同时在窗口里时,打字机拆出的后半条 cue 会去认领
|
|
349
|
+
# 第二行、把一条漏抹平。这和 hit_delta 的多重集口径一致(同文本行按条数算),接受;
|
|
350
|
+
# 再先给该在字幕带的行(见上)
|
|
351
|
+
hit = min(exact, key=lambda i: (script[i].matched >= 0, not band(script[i]), i))
|
|
352
|
+
n_exact += 1
|
|
353
|
+
else:
|
|
354
|
+
# 剪枝不改结果:两个 quick 比值都是 ratio 的上界(同 script_align)。
|
|
355
|
+
# 这里剪的是 `< best_s`,script_align 是 `<= best_s`——**别"统一"**:那边平手不换,
|
|
356
|
+
# 这边平手要让给还没被认领的行(下面那个 `s == best_s` 分支),剪掉等分的就换不成了
|
|
357
|
+
hit, best_s = -1, fuzzy
|
|
358
|
+
for i in cnd:
|
|
359
|
+
sm.set_seq2(sn[i])
|
|
360
|
+
s = 0.0
|
|
361
|
+
for v in vs:
|
|
362
|
+
sm.set_seq1(v)
|
|
363
|
+
if sm.real_quick_ratio() < best_s or sm.quick_ratio() < best_s:
|
|
364
|
+
continue
|
|
365
|
+
s = max(s, sm.ratio())
|
|
366
|
+
if s > best_s or (s == best_s and hit >= 0
|
|
367
|
+
and script[hit].matched >= 0 > script[i].matched):
|
|
368
|
+
hit, best_s = i, s
|
|
369
|
+
if hit >= 0:
|
|
370
|
+
n_fuzzy += 1
|
|
371
|
+
script[hit].sim = round(best_s, 3)
|
|
372
|
+
else:
|
|
373
|
+
n_none += 1
|
|
374
|
+
sub_hit.append(hit)
|
|
375
|
+
if hit >= 0 and script[hit].matched < 0:
|
|
376
|
+
script[hit].matched = si
|
|
377
|
+
sub_extra = claim_contained(script, subs, cand, sub_hit, heads)
|
|
378
|
+
n_contain = sum(1 for h, ex in zip(sub_hit, sub_extra) if h < 0 and ex)
|
|
379
|
+
return {"subs": len(subs), "exact": n_exact, "fuzzy": n_fuzzy, "contain": n_contain,
|
|
380
|
+
"unmatched_subs": n_none - n_contain, "jumps": 0,
|
|
381
|
+
"sub_hit": sub_hit, "sub_extra": sub_extra, "heads": heads}
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def score(ref: Ref, script: list[SA.Entry], subs, fuzzy: float) -> dict:
|
|
385
|
+
"""`run` + 把变体组折成一个条目:组里任何一种说法被认领,就记在代表行上
|
|
386
|
+
(`via`:代表行 key -> 实际被认领的那一行);`sub_hit` 里指向变体的也改指代表行,
|
|
387
|
+
免得同一句话的两种说法各被一条 cue 认领时,重复认领数不出来。
|
|
388
|
+
`script` 可以是 `ref.script` 的深拷贝(两臂各量各的,不互相覆盖 `matched`)。"""
|
|
389
|
+
st = run(script, ref.windows, subs, fuzzy)
|
|
390
|
+
idx = {e.key: i for i, e in enumerate(script)}
|
|
391
|
+
via: dict[str, SA.Entry] = {}
|
|
392
|
+
for e in script:
|
|
393
|
+
rep = ref.variant_of.get(e.key)
|
|
394
|
+
if rep and e.matched >= 0 and script[idx[rep]].matched < 0:
|
|
395
|
+
script[idx[rep]].matched = e.matched
|
|
396
|
+
via[rep] = e
|
|
397
|
+
def rep_of(h): # 变体被认领时记在代表行上,重复认领才数得出来
|
|
398
|
+
return idx[ref.variant_of[script[h].key]] if script[h].key in ref.variant_of else h
|
|
399
|
+
st["sub_hit"] = [rep_of(h) if h >= 0 else h for h in st["sub_hit"]]
|
|
400
|
+
st["sub_extra"] = [[rep_of(i) for i in ex] for ex in st["sub_extra"]]
|
|
401
|
+
st["via"] = via
|
|
402
|
+
return st
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def claims(st: dict) -> list[int]:
|
|
406
|
+
"""每一次认领指向的剧本行:主认领 + 第二遍的包含认领(`claim_contained`)。"""
|
|
407
|
+
return [h for h in st["sub_hit"] if h >= 0] + [i for ex in st["sub_extra"] for i in ex]
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def repeat_claims(sub_hit: list[int]) -> tuple[int, int]:
|
|
411
|
+
"""(被 ≥2 条 cue 认领的剧本行数, 多出来的认领次数之和)。原始轨口径,见模块说明。"""
|
|
412
|
+
c = Counter(h for h in sub_hit if h >= 0)
|
|
413
|
+
return sum(1 for v in c.values() if v > 1), sum(v - 1 for v in c.values())
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def head_tail(script: list[SA.Entry], subs, via: dict | None = None,
|
|
417
|
+
skip: set[int] | None = None,
|
|
418
|
+
heads: frozenset[str] = frozenset()) -> list[tuple[int, int, int]]:
|
|
419
|
+
"""配上的每一对:(剧本内容字数, 行首缺几个内容字, 行尾缺几个内容字)。
|
|
420
|
+
|
|
421
|
+
用内容归一(去标点、小书写假名并大写)比,免得省略号读没了被算成"缺字"——
|
|
422
|
+
那是另一件事(script-corpus 报告文本质量那节)。一条剧本行只取第一条认领它的 cue;
|
|
423
|
+
变体组由另一种说法命中的,拿**被认领的那种说法**比(`score` 的 `via`)。
|
|
424
|
+
|
|
425
|
+
判据的洞:取第一个匹配块的起点当"行首缺字"——cue 的读法带着没剥干净的名牌时,第一块
|
|
426
|
+
可能对上剧本中段,行首缺字会被高估。gi-s2 那 61.8% 摆帧证实是立绘遮挡,那次没出错。
|
|
427
|
+
**2026-09-18 起 `heads` 传进来了**(审计 P2 的同一族:报数用的读法要和匹配用的一致),
|
|
428
|
+
名牌不再留在读法里——这个数因此比以前小一点,和以前的读数不能直接比。"""
|
|
429
|
+
out = []
|
|
430
|
+
for k, e in enumerate(script):
|
|
431
|
+
# 包含认领的行按定义整行都在 cue 里,缺字恒为 0,算进去会把这个数压低
|
|
432
|
+
if e.matched < 0 or not band(e) or k in (skip or ()):
|
|
433
|
+
continue
|
|
434
|
+
s = GS.cnorm((via or {}).get(e.key, e).text)
|
|
435
|
+
if not s:
|
|
436
|
+
continue
|
|
437
|
+
best = None
|
|
438
|
+
for v in variants(subs[e.matched][1], heads):
|
|
439
|
+
v = GS.cnorm(v)
|
|
440
|
+
sm = SequenceMatcher(None, v, s, autojunk=False)
|
|
441
|
+
blocks = [b for b in sm.get_matching_blocks() if b.size]
|
|
442
|
+
if not blocks:
|
|
443
|
+
continue
|
|
444
|
+
r = sm.ratio()
|
|
445
|
+
if best is None or r > best[0]:
|
|
446
|
+
best = (r, blocks[0].b, len(s) - (blocks[-1].b + blocks[-1].size))
|
|
447
|
+
if best:
|
|
448
|
+
out.append((len(s), best[1], best[2]))
|
|
449
|
+
return out
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def hitset(script: list[SA.Entry]) -> Counter:
|
|
453
|
+
return Counter(SA.norm(e.text) for e in script if band(e) and e.matched >= 0)
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def main() -> int:
|
|
457
|
+
# --help 只放用法;和《魔裁》那套的三处差别写在模块文档串里,给读代码的人(发布前清理,Opus 审查 O6)
|
|
458
|
+
ap = argparse.ArgumentParser(description=CLI_DESCRIPTION,
|
|
459
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
460
|
+
ap.add_argument("ref", help="gamescript.py 的剧本 JSON")
|
|
461
|
+
ap.add_argument("--subs", required=True,
|
|
462
|
+
help="被量的轨:SRT,或 `*-tracks.json`(取 provenance 里的 main_srt)")
|
|
463
|
+
ap.add_argument("--vs", default=None,
|
|
464
|
+
help="对比臂的轨(同上两种写法):报命中集合差(三档)与重复认领的变化")
|
|
465
|
+
ap.add_argument("--fuzzy", type=float, default=0.75, help="同 script_align")
|
|
466
|
+
ap.add_argument("--branch-run", type=int, default=5,
|
|
467
|
+
help="连续这么多条未命中算『长 gap』。这里**只用来分开报**,长 gap 照样算疑似真漏"
|
|
468
|
+
"(分支已按对话图排出分母,见 script_align.gap_report 的 long_is_branch)")
|
|
469
|
+
ap.add_argument("--show", type=int, default=8, help="列几条疑似真漏")
|
|
470
|
+
ap.add_argument("--show-unmatched", type=int, default=0,
|
|
471
|
+
help="列几条对不上剧本的 cue(看幽灵上限里装的是什么)")
|
|
472
|
+
ap.add_argument("--out", default=None, help="数和逐条清单写到这个 JSON")
|
|
473
|
+
a = ap.parse_args()
|
|
474
|
+
|
|
475
|
+
ref = load_ref(Path(a.ref))
|
|
476
|
+
doc, script, role = ref.doc, ref.script, ref.role
|
|
477
|
+
pv = doc["provenance"]
|
|
478
|
+
gt = pv["gametext"]
|
|
479
|
+
print(f"剧本:{pv['game']} {gt.get('version') or '?'}(文本包指纹 {gt['fingerprint'][7:19]}),"
|
|
480
|
+
f"{len(doc['units'])} 个对话单元 / {len(script)} 行,由 {Path(pv['obs']).name} 圈出"
|
|
481
|
+
f"(gamescript git_head {pv['git_head']})")
|
|
482
|
+
if not gt.get("integrity_checked"):
|
|
483
|
+
print(" ⚠ 文本包:上游快照的完整性检查没过或没做")
|
|
484
|
+
subs_path, tp = resolve(a.subs)
|
|
485
|
+
subs = read_cues(subs_path)
|
|
486
|
+
st = score(ref, script, subs, a.fuzzy)
|
|
487
|
+
n_empty = st["subs"] - st["exact"] - st["fuzzy"] - st["contain"] - st["unmatched_subs"]
|
|
488
|
+
print(f"轨 {subs_path.name}"
|
|
489
|
+
+ (f"(build_tracks git_head {tp['git_head']},代码指纹 {tp['code_fp']})" if tp else "")
|
|
490
|
+
+ f":{st['subs']} 条 cue;精确 {st['exact']}、模糊 {st['fuzzy']}、"
|
|
491
|
+
f"只靠包含认上 {st['contain']}、"
|
|
492
|
+
f"**对不上剧本 {st['unmatched_subs']}({st['unmatched_subs']/max(1,st['subs']):.1%},"
|
|
493
|
+
f"幽灵上限——库外的文字:UI / 路人 / 主播自己的字幕…都会落在这里)**"
|
|
494
|
+
+ (f"、归一化后为空 {n_empty}" if n_empty else ""))
|
|
495
|
+
|
|
496
|
+
dial = [e for e in script if band(e)]
|
|
497
|
+
excl = Counter(e.kind for e in script if not band(e))
|
|
498
|
+
excl["Duplicate"] = sum(1 for u in doc["units"] for ln in u["lines"] if ln["kind"] == "Duplicate")
|
|
499
|
+
excl = +excl
|
|
500
|
+
# 变体被认领是预期内的(记在代表行上),不算"排出分母的却被命中"
|
|
501
|
+
excl_hit = Counter(e.kind for e in script if not band(e) and e.matched >= 0 and e.kind != "Variant")
|
|
502
|
+
hit = [e for e in dial if e.matched >= 0]
|
|
503
|
+
print()
|
|
504
|
+
print(f"剧本侧:**该上屏的 {len(dial)} 条**(另有 "
|
|
505
|
+
+ "、".join(f"{k} {v}" for k, v in excl.most_common()) + " 不计),"
|
|
506
|
+
f"命中 {len(hit)}({len(hit)/max(1,len(dial)):.1%})")
|
|
507
|
+
if ref.variant_of:
|
|
508
|
+
n_groups = len(set(ref.variant_of.values()))
|
|
509
|
+
print(f" 变体组(同一句话的两种说法,一组算一条):{n_groups} 组,其中 {len(st['via'])} 组"
|
|
510
|
+
f"由代表行以外的那种说法命中")
|
|
511
|
+
if excl_hit:
|
|
512
|
+
print(" ⚠ 排出分母的条目被主轨命中了:"
|
|
513
|
+
+ "、".join(f"{k} {v}/{excl[k]}" for k, v in excl_hit.most_common())
|
|
514
|
+
+ "——Unwalked 被命中 = 分支判错;Choice / OffBand 被命中 = 字幕带以外的字进了这条轨")
|
|
515
|
+
gr = SA.gap_report(dial, a.branch_run, long_is_branch=False)
|
|
516
|
+
side = obs_side(doc)
|
|
517
|
+
C = evalkit.CLASSES[2]
|
|
518
|
+
missed_all = [(k, "short" if k in gr["short_idx"] else "long") for k, e in enumerate(dial)
|
|
519
|
+
if e.matched < 0 and evalkit.triviality(e.text) == C]
|
|
520
|
+
attr = Counter((g, side[dial[k].key]) for k, g in missed_all)
|
|
521
|
+
print()
|
|
522
|
+
print(f"未命中的『{C}』按 **obs 里读到过没有** 拆(gamescript 的锚点,只看这一行、这一段):")
|
|
523
|
+
print(f" {'':<10}" + "".join(f"{x:>12}" for x in OBS_SIDES))
|
|
524
|
+
for g, name in (("short", "短 gap"), ("long", "长 gap")):
|
|
525
|
+
print(f" {name:<10}" + "".join(f"{attr[(g, x)]:>14}" for x in OBS_SIDES))
|
|
526
|
+
print(f" 读到过 = 丢在轨这一层;只撞常用句 = 别处的同句也能锚上来,算不得这一处读到过;"
|
|
527
|
+
f"没读到 = 丢在 OCR 之前(没出框 / 遮挡 / 显示太短没采到,要摆帧)")
|
|
528
|
+
|
|
529
|
+
print()
|
|
530
|
+
print("按上游角色拆(**用来校准分母**:分母内命中率接近 0 说明它不走字幕带;"
|
|
531
|
+
"『全部』连排出分母的也算,别拿它当分母占比读):")
|
|
532
|
+
print(f" {'':<34}{'分母内 命中/条':>18}{'全部 命中/条':>18}")
|
|
533
|
+
rb = Counter(role[e.key] for e in dial)
|
|
534
|
+
rbh = Counter(role[e.key] for e in hit)
|
|
535
|
+
rc = Counter(role[e.key] for e in script)
|
|
536
|
+
rh = Counter(role[e.key] for e in script if e.matched >= 0)
|
|
537
|
+
# 进分母的类全列;不进的只列前几类(绝区零散条目的 head 有上百种),其余并成一行
|
|
538
|
+
shown = [r for r, _ in rc.most_common() if rb[r]]
|
|
539
|
+
shown += [r for r, _ in rc.most_common() if not rb[r]][:max(0, ROLE_ROWS - len(shown))]
|
|
540
|
+
for r in shown:
|
|
541
|
+
print(f" {r:<34}{evalkit.denom(rbh[r], rb[r]) if rb[r] else '—':>18}"
|
|
542
|
+
f"{evalkit.denom(rh[r], rc[r]):>18}")
|
|
543
|
+
rest = [r for r in rc if r not in shown]
|
|
544
|
+
if rest:
|
|
545
|
+
print(f" {f'其余 {len(rest)} 类(都不进分母)':<30}{'—':>18}"
|
|
546
|
+
f"{evalkit.denom(sum(rh[r] for r in rest), sum(rc[r] for r in rest)):>18}")
|
|
547
|
+
|
|
548
|
+
groups, extra = repeat_claims(claims(st))
|
|
549
|
+
n_extra_lines = sum(len(ex) for ex in st["sub_extra"])
|
|
550
|
+
print(f"\n重复认领(原始轨口径):{groups} 条剧本行被 ≥2 条 cue 认领,多出 {extra} 次"
|
|
551
|
+
f"(**报警不是错误数**:真实重播 / 打字机拆成两条 cue 都会让它涨)")
|
|
552
|
+
print(f"第二遍『整行落在 cue 里』多认领 {n_extra_lines} 行(一条 cue 装着两行剧本、"
|
|
553
|
+
f"或 cue 混进垃圾行够不着模糊阈值;不许和主认领占同一段字,见 claim_contained)")
|
|
554
|
+
|
|
555
|
+
ht = head_tail(script, subs, st["via"], {i for ex in st["sub_extra"] for i in ex},
|
|
556
|
+
st.get("heads") or frozenset())
|
|
557
|
+
if ht:
|
|
558
|
+
n = len(ht)
|
|
559
|
+
h1 = sum(1 for _, h, _ in ht if h >= 1)
|
|
560
|
+
h2 = sum(1 for _, h, _ in ht if h >= 2)
|
|
561
|
+
t1 = sum(1 for _, _, t in ht if t >= 1)
|
|
562
|
+
ex = sum(1 for _, h, t in ht if h == 0 and t == 0)
|
|
563
|
+
print(f"行首 / 行尾缺字(配上的 {n} 对,按内容字比,不计标点):首尾都不缺 {evalkit.denom(ex, n)};"
|
|
564
|
+
f"**行首缺 ≥1 字 {evalkit.denom(h1, n)}**(≥2 字 {evalkit.denom(h2, n)});"
|
|
565
|
+
f"行尾缺 ≥1 字 {evalkit.denom(t1, n)}")
|
|
566
|
+
|
|
567
|
+
if a.show_unmatched:
|
|
568
|
+
un = [(t, raw) for (t, raw, _), h, ex in zip(subs, st["sub_hit"], st["sub_extra"])
|
|
569
|
+
if h < 0 and not ex]
|
|
570
|
+
print(f"\n对不上剧本的 cue(前 {a.show_unmatched} 条 / 共 {len(un)}):")
|
|
571
|
+
for t, raw in un[:a.show_unmatched]:
|
|
572
|
+
print(f" {t:8.1f}s {raw.replace(chr(10), ' / ')[:80]}")
|
|
573
|
+
|
|
574
|
+
short = gr["short"]
|
|
575
|
+
if short and a.show:
|
|
576
|
+
print(f"\n疑似真漏抽样(前 {a.show} 处):")
|
|
577
|
+
for i, k in short[:a.show]:
|
|
578
|
+
for e in dial[i:i + k]:
|
|
579
|
+
print(f" [{role[e.key]:<10}] {e.src:<26} {e.text[:56]}")
|
|
580
|
+
|
|
581
|
+
delta = None
|
|
582
|
+
if a.vs:
|
|
583
|
+
ha = hitset(script)
|
|
584
|
+
vs_path, vp = resolve(a.vs)
|
|
585
|
+
subs_b = read_cues(vs_path)
|
|
586
|
+
script_b = copy.deepcopy(script) # B 臂量自己那份,A 的 matched 原样留给下面写 JSON
|
|
587
|
+
st_b = score(ref, script_b, subs_b, a.fuzzy)
|
|
588
|
+
hb = hitset(script_b)
|
|
589
|
+
gained, lost = hb - ha, ha - hb
|
|
590
|
+
g = Counter(evalkit.triviality(t) for t in gained.elements())
|
|
591
|
+
l = Counter(evalkit.triviality(t) for t in lost.elements())
|
|
592
|
+
_, eb = repeat_claims(claims(st_b))
|
|
593
|
+
net = sum(hb.values()) - sum(ha.values())
|
|
594
|
+
print(f"\n命中集合差(归一化文本多重集):**{a.subs} -> {a.vs}**")
|
|
595
|
+
same = code_same(tp, vp)
|
|
596
|
+
if same is False:
|
|
597
|
+
print(f" ⚠ 两臂的 build_tracks 代码指纹不同({tp['code_fp']} / {vp['code_fp']})——"
|
|
598
|
+
f"差里混着代码版本的变化")
|
|
599
|
+
elif same is None and tp and vp:
|
|
600
|
+
print(" ⚠ 有一臂的产物没有代码指纹(早于该字段),证明不了两臂是同一份 build_tracks")
|
|
601
|
+
print(f" 命中 {sum(ha.values())} -> {sum(hb.values())}(净 {net:+d}:新增 +{sum(gained.values())}"
|
|
602
|
+
f" / 丢失 -{sum(lost.values())});按三档:"
|
|
603
|
+
+ " ".join(f"{c} {g[c]-l[c]:+d}" for c in evalkit.CLASSES))
|
|
604
|
+
print(f" cue {st['subs']} -> {st_b['subs']};对不上剧本 {st['unmatched_subs']} -> "
|
|
605
|
+
f"{st_b['unmatched_subs']};重复认领 {extra} -> {eb}")
|
|
606
|
+
ex_l = [t for t in lost if evalkit.triviality(t) == C][:a.show]
|
|
607
|
+
ex_g = [t for t in gained if evalkit.triviality(t) == C][:a.show]
|
|
608
|
+
if ex_l:
|
|
609
|
+
print(f" 丢失的『{C}』:" + " | ".join(t[:24] for t in ex_l))
|
|
610
|
+
if ex_g:
|
|
611
|
+
print(f" 新增的『{C}』:" + " | ".join(t[:24] for t in ex_g))
|
|
612
|
+
delta = {"vs": a.vs, "vs_git_head": vp.get("git_head"), "vs_code_fp": vp.get("code_fp"),
|
|
613
|
+
"hit_a": sum(ha.values()), "hit_b": sum(hb.values()),
|
|
614
|
+
"by_class": {c: g[c] - l[c] for c in evalkit.CLASSES},
|
|
615
|
+
"repeat_extra_b": eb, "cues_b": st_b["subs"],
|
|
616
|
+
"unmatched_b": st_b["unmatched_subs"]}
|
|
617
|
+
|
|
618
|
+
if a.out:
|
|
619
|
+
p = Path(a.out)
|
|
620
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
621
|
+
p.write_text(json.dumps({
|
|
622
|
+
"ref": a.ref, "subs": str(subs_path), "ref_git_head": pv["git_head"],
|
|
623
|
+
"ref_code_fp": pv.get("code_fp"),
|
|
624
|
+
"tracks_git_head": tp.get("git_head"), "tracks_code_fp": tp.get("code_fp"),
|
|
625
|
+
"gametext_fingerprint": pv["gametext"]["fingerprint"],
|
|
626
|
+
"gametext_integrity_checked": bool(pv["gametext"].get("integrity_checked")),
|
|
627
|
+
"cues": st["subs"], "exact": st["exact"], "fuzzy": st["fuzzy"],
|
|
628
|
+
"contain_cues": st["contain"], "contain_lines": n_extra_lines,
|
|
629
|
+
"unmatched_cues": st["unmatched_subs"],
|
|
630
|
+
"band_entries": len(dial), "band_hit": len(hit),
|
|
631
|
+
"excluded": dict(excl), "excluded_hit": dict(excl_hit),
|
|
632
|
+
"variant_groups": len(set(ref.variant_of.values())), "variant_via": len(st["via"]),
|
|
633
|
+
"gap_long_entries": gr["n_long"], "gap_short_entries": gr["n_short"],
|
|
634
|
+
"miss_by_class": gr["by_class"],
|
|
635
|
+
"repeat_groups": groups, "repeat_extra": extra,
|
|
636
|
+
"head_tail": {"pairs": len(ht), "head_ge1": sum(1 for _, h, _ in ht if h >= 1),
|
|
637
|
+
"head_ge2": sum(1 for _, h, _ in ht if h >= 2),
|
|
638
|
+
"tail_ge1": sum(1 for _, _, t in ht if t >= 1)},
|
|
639
|
+
"delta": delta,
|
|
640
|
+
"missed": [{"key": e.key, "role": role[e.key], "unit": e.src, "text": e.text,
|
|
641
|
+
"gap": "short" if k in gr["short_idx"] else "long"}
|
|
642
|
+
for k, e in enumerate(dial) if e.matched < 0],
|
|
643
|
+
"missed_content": [{"key": dial[k].key, "gap": g, "obs": side[dial[k].key],
|
|
644
|
+
"role": role[dial[k].key], "unit": dial[k].src,
|
|
645
|
+
"text": dial[k].text} for k, g in missed_all],
|
|
646
|
+
}, ensure_ascii=False, indent=1), encoding="utf-8")
|
|
647
|
+
print(f"\n-> {p}")
|
|
648
|
+
return 0
|
|
649
|
+
|
|
650
|
+
|
|
651
|
+
if __name__ == "__main__":
|
|
652
|
+
raise SystemExit(main())
|