flowocr 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- flowocr/__init__.py +15 -0
- flowocr/analyze/__init__.py +8 -0
- flowocr/analyze/align.py +151 -0
- flowocr/analyze/build_tracks.py +2886 -0
- flowocr/analyze/cluster_layers.py +155 -0
- flowocr/analyze/game_align.py +652 -0
- flowocr/analyze/gamescript.py +1220 -0
- flowocr/analyze/gtdbundle.py +306 -0
- flowocr/analyze/match.py +67 -0
- flowocr/analyze/matchers/__init__.py +5 -0
- flowocr/analyze/matchers/gametext.py +70 -0
- flowocr/analyze/merge_nameplate.py +178 -0
- flowocr/analyze/models/slot_pair.json +148 -0
- flowocr/analyze/nameplate.py +148 -0
- flowocr/analyze/pair_features.py +168 -0
- flowocr/analyze/pair_model.py +127 -0
- flowocr/analyze/refine_boundaries.py +409 -0
- flowocr/analyze/script_align.py +641 -0
- flowocr/analyze/scriptmatch.py +1018 -0
- flowocr/analyze/slot_learned.py +306 -0
- flowocr/analyze/slot_lines.py +251 -0
- flowocr/analyze/slot_modes.py +143 -0
- flowocr/analyze/slot_pairs.py +562 -0
- flowocr/analyze/slot_veto.py +104 -0
- flowocr/analyze/uigate.py +351 -0
- flowocr/artifacts/__init__.py +5 -0
- flowocr/artifacts/evalkit.py +123 -0
- flowocr/artifacts/matchedio.py +75 -0
- flowocr/artifacts/srtio.py +166 -0
- flowocr/artifacts/tracksio.py +345 -0
- flowocr/extensions.py +77 -0
- flowocr/extract/__init__.py +8 -0
- flowocr/extract/childproc.py +118 -0
- flowocr/extract/decode_proc.py +511 -0
- flowocr/extract/decode_shards.py +574 -0
- flowocr/extract/detpost.py +215 -0
- flowocr/extract/edge_proc.py +314 -0
- flowocr/extract/edge_refine.py +598 -0
- flowocr/extract/fast_det.py +214 -0
- flowocr/extract/ffcheck.py +87 -0
- flowocr/extract/framegrid.py +265 -0
- flowocr/extract/framesource.py +1264 -0
- flowocr/extract/ocr_args.py +754 -0
- flowocr/extract/ocr_complete.py +231 -0
- flowocr/extract/ocr_parallel.py +208 -0
- flowocr/extract/ort_server.py +632 -0
- flowocr/extract/ortclient.py +363 -0
- flowocr/extract/ptsclock.py +150 -0
- flowocr/extract/recdecode.py +78 -0
- flowocr/extract/recort.py +162 -0
- flowocr/extract/recpack.py +94 -0
- flowocr/extract/recpool.py +206 -0
- flowocr/extract/recprep.py +71 -0
- flowocr/extract/refine_video.py +271 -0
- flowocr/extract/regions.py +387 -0
- flowocr/extract/reuse_v2.py +593 -0
- flowocr/extract/run_groups.py +247 -0
- flowocr/extract/run_ocr2.py +1605 -0
- flowocr/extract/supervisor.py +261 -0
- flowocr/extract/timeline.py +84 -0
- flowocr/extract/typewriter_fuse.py +394 -0
- flowocr/models.py +263 -0
- flowocr/output/__init__.py +3 -0
- flowocr/output/export.py +205 -0
- flowocr/output/layout.py +210 -0
- flowocr/output/presets/__init__.py +6 -0
- flowocr/output/presets/_overlay.py +40 -0
- flowocr/output/presets/default.py +21 -0
- flowocr/output/presets/default_all.py +20 -0
- flowocr/output/presets/dev.py +21 -0
- flowocr/output/presets/matched_srt.py +32 -0
- flowocr/output/presets/script.py +20 -0
- flowocr/output/presets/srt_main.py +33 -0
- flowocr/output/render.py +82 -0
- flowocr/output/run_srt.py +83 -0
- flowocr/output/script.py +541 -0
- flowocr/paths.py +168 -0
- flowocr/provenance.py +119 -0
- flowocr/typeset/__init__.py +9 -0
- flowocr/typeset/__main__.py +38 -0
- flowocr/typeset/assfile.py +216 -0
- flowocr/typeset/core.py +821 -0
- flowocr/typeset/fx/__init__.py +11 -0
- flowocr-0.1.0.dist-info/METADATA +109 -0
- flowocr-0.1.0.dist-info/RECORD +91 -0
- flowocr-0.1.0.dist-info/WHEEL +5 -0
- flowocr-0.1.0.dist-info/entry_points.txt +10 -0
- flowocr-0.1.0.dist-info/licenses/LICENSE +674 -0
- flowocr-0.1.0.dist-info/licenses/LICENSES/Apache-2.0.txt +201 -0
- flowocr-0.1.0.dist-info/licenses/LICENSES/PP-OCRv6-NOTICE.md +18 -0
- flowocr-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,641 @@
|
|
|
1
|
+
"""把一份字幕对回游戏原文剧本,量"剧本里的话有多少条被抓到了"。
|
|
2
|
+
|
|
3
|
+
为什么需要它(script-corpus 报告):这批语料**没有人工定稿**,唯一权威的文本是
|
|
4
|
+
游戏剧本本身。所以"准不准"不能按 CER 算,只能问:**剧本里该上屏的那些行,
|
|
5
|
+
有多少条在字幕里出现了**。这正是 owner 定的准确性重心(漏条 / 幽灵 / 切分 / 时间)。
|
|
6
|
+
|
|
7
|
+
## 三件必须先分开的事,否则"漏条率"是假的
|
|
8
|
+
|
|
9
|
+
owner 2026-09-04:「几乎全部覆盖。可能有些游戏分支没走到,另外剧本里的一些条目
|
|
10
|
+
可能不是对话所以不会被 ocr。」——所以"剧本行没有对应字幕"有三个原因,只有第三个是缺陷:
|
|
11
|
+
|
|
12
|
+
1. **不是对话**。Naninovel 的 `# key` 直接给出类型:`_Narrative###`(旁白)、
|
|
13
|
+
`_<角色名>###`(台词)是会上屏到字幕带的;`_Choice###`(选项)、`_Toast###`、
|
|
14
|
+
`_Return###` 以及 `@choice` / `@toast` 这些命令条目走的是别的 UI。默认只统计前两类。
|
|
15
|
+
2. **分支没走到**。表现是**连续一整段**都没命中。孤立的一两条落在已命中的上下文里,
|
|
16
|
+
才像真漏。所以报的时候按"未命中连续段的长度"分档,不合成一个数。
|
|
17
|
+
3. **真漏**。剩下的才是。
|
|
18
|
+
|
|
19
|
+
## 对齐怎么做
|
|
20
|
+
|
|
21
|
+
剧本是**有序**的,字幕也是有序的,所以不做全局最优匹配,而是走游标:
|
|
22
|
+
每条字幕在游标前方的窗口里找,找不到再全局找最近的一个(分支跳转)。
|
|
23
|
+
这与 `dev_tools/reference/match.py` 的锚点思路一致,但这里**只做统计、不改文本**。
|
|
24
|
+
|
|
25
|
+
用法:
|
|
26
|
+
python -m flowocr.analyze.script_align --script data/corpus/mocai/script-raw/Scripts \
|
|
27
|
+
--text data/corpus/mocai/script-jp/Scripts \
|
|
28
|
+
--subs data/corpus/mocai/game_text/lower1_jp.ass ... \
|
|
29
|
+
--out out/yuka/align-lower.json
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import argparse
|
|
34
|
+
import json
|
|
35
|
+
import re
|
|
36
|
+
import unicodedata
|
|
37
|
+
from dataclasses import dataclass, field
|
|
38
|
+
from difflib import SequenceMatcher
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
|
|
41
|
+
from flowocr.artifacts import evalkit # noqa: E402
|
|
42
|
+
from flowocr.artifacts import srtio # noqa: E402
|
|
43
|
+
|
|
44
|
+
# 会上屏到字幕带的条目类型。`Choice`/`Toast`/`Return`/`Unknown` 走别的 UI,默认不计。
|
|
45
|
+
DIALOGUE_KINDS_DEFAULT = "Narrative,CHARACTER"
|
|
46
|
+
"""CHARACTER 是占位:凡是 key 后缀不在已知非对话表里的,都当成角色名(台词)。"""
|
|
47
|
+
NON_DIALOGUE_KINDS = {"Choice", "Toast", "Return", "Unknown"}
|
|
48
|
+
NON_DIALOGUE_CMDS = {"@choice", "@toast", "@printDebate", "@print", "@char"}
|
|
49
|
+
"""`@printDebate` 是被实测逼出来的:它有 1999 条,**命中率 0%**——不是漏,
|
|
50
|
+
是反驳(法庭辩论)那一屏走的是另一套 UI,根本不出现在字幕带里。
|
|
51
|
+
把它算进分母,字幕带的"漏条率"会凭空多 6 个百分点。
|
|
52
|
+
|
|
53
|
+
⚠ **用途 1(画面所有文字)要把这些加回来**:它们确实在屏幕上,只是不在字幕带。
|
|
54
|
+
这个开关就是两个用途口径不同的地方,`--include-cmds` 可以放回来。"""
|
|
55
|
+
|
|
56
|
+
KEY_RE = re.compile(r"^#\s*(.+?)\s*$")
|
|
57
|
+
CMD_RE = re.compile(r"^;\s*>\s*(@\w+)")
|
|
58
|
+
KIND_RE = re.compile(r"_([A-Za-z]+?)_?\d*$")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass
|
|
62
|
+
class Entry:
|
|
63
|
+
key: str
|
|
64
|
+
kind: str
|
|
65
|
+
cmd: str
|
|
66
|
+
text: str
|
|
67
|
+
src: str
|
|
68
|
+
matched: int = -1 # 对上的字幕下标,-1 = 没对上
|
|
69
|
+
sim: float = 0.0
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
ASS_TAG = re.compile(r"{[^}]*}")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def norm(t: str) -> str:
|
|
76
|
+
r"""归一到可比形态。与 paddleocr 实验里的 second_opinion.py 同一个理由:
|
|
77
|
+
括号字形、全半角这类装饰性差异不是"对不上"。
|
|
78
|
+
|
|
79
|
+
**必须先剥掉 ASS 样式标签**:`pre_process.py` 把游戏的 `<color>` / `<b>` 转成了
|
|
80
|
+
`{\c&Hff8e9c&}{\b1}…{\r}`,这些标签留在剧本文本里。老算法的 .ass 是把剧本原文
|
|
81
|
+
整条抄过去的,标签也一起抄,所以它自己跟自己精确相等;我们的 OCR 读到的是屏幕上的
|
|
82
|
+
纯文字,永远配不上。不剥标签,这个对比就是偏向基线的。"""
|
|
83
|
+
t = ASS_TAG.sub("", t)
|
|
84
|
+
t = unicodedata.normalize("NFKC", t)
|
|
85
|
+
t = t.replace("\\N", "").replace("\\n", "")
|
|
86
|
+
return "".join(t.split())
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def load_script(raw_root: Path, text_root: Path) -> list[Entry]:
|
|
90
|
+
"""raw 提供 key/类型,text 提供已经去过标签的干净文本。
|
|
91
|
+
|
|
92
|
+
两边**逐条对应**(raw 的 `# key` 条目数 == text 的非空行数),
|
|
93
|
+
不对应就直接报错——静默错位会让后面所有数都错。
|
|
94
|
+
"""
|
|
95
|
+
out: list[Entry] = []
|
|
96
|
+
for f in sorted(raw_root.rglob("*.bytes")):
|
|
97
|
+
keys, cmds = [], []
|
|
98
|
+
cur_cmd = ""
|
|
99
|
+
for line in f.read_text(encoding="utf-8").splitlines():
|
|
100
|
+
m = KEY_RE.match(line)
|
|
101
|
+
if m:
|
|
102
|
+
keys.append(m.group(1))
|
|
103
|
+
cmds.append("")
|
|
104
|
+
cur_cmd = ""
|
|
105
|
+
continue
|
|
106
|
+
c = CMD_RE.match(line)
|
|
107
|
+
if c and keys:
|
|
108
|
+
cur_cmd = c.group(1)
|
|
109
|
+
cmds[-1] = cur_cmd
|
|
110
|
+
if not keys:
|
|
111
|
+
continue
|
|
112
|
+
g = text_root / f.relative_to(raw_root)
|
|
113
|
+
if not g.exists():
|
|
114
|
+
raise SystemExit(f"{f} 有 {len(keys)} 个条目,但 {g} 不存在")
|
|
115
|
+
lines = [l for l in g.read_text(encoding="utf-8").splitlines() if l.strip()]
|
|
116
|
+
if len(lines) != len(keys):
|
|
117
|
+
raise SystemExit(f"逐条对应断了:{f.name} raw {len(keys)} 条 / text {len(lines)} 行")
|
|
118
|
+
rel = str(f.relative_to(raw_root)).replace("\\", "/")
|
|
119
|
+
for k, c, t in zip(keys, cmds, lines):
|
|
120
|
+
m = KIND_RE.search(k)
|
|
121
|
+
out.append(Entry(key=k, kind=m.group(1) if m else "?", cmd=c, text=t, src=rel))
|
|
122
|
+
return out
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
ALL_ENTRIES = False
|
|
126
|
+
"""用途 1 开关:True 时所有剧本条目都算"该上屏",见 --usage1。"""
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def is_dialogue(e: Entry) -> bool:
|
|
130
|
+
if ALL_ENTRIES:
|
|
131
|
+
return True
|
|
132
|
+
return e.kind not in NON_DIALOGUE_KINDS and e.cmd not in NON_DIALOGUE_CMDS
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def read_subs(paths: list[Path], sort_by_time: bool = False) -> list[tuple[float, str, str]]:
|
|
136
|
+
"""按给定顺序读 .ass / .srt,返回 (起点秒, 文本, 来源文件)。
|
|
137
|
+
|
|
138
|
+
默认**按命令行给的顺序首尾相接**,不重排:input1..5 是一次通关的先后,
|
|
139
|
+
每个文件的时间戳都从 0 重新开始,重排会把游标逻辑毁掉。
|
|
140
|
+
|
|
141
|
+
但用途 1 要把**同一段视频的多条轨**合起来看(字幕带 + 辩论屏 + UI…),
|
|
142
|
+
那些文件共享同一条时间轴,必须 `sort_by_time=True` 按时间归并——
|
|
143
|
+
否则一条轨读完再读下一条,游标会被拽回开头,对齐全乱。
|
|
144
|
+
**这两种用法必须显式区分,猜错了不会报错、只会出一堆假数。**
|
|
145
|
+
"""
|
|
146
|
+
out = []
|
|
147
|
+
for p in paths:
|
|
148
|
+
# 解析与"一条 = 一个 cue"的口径统一在 flowocr.artifacts.srtio。
|
|
149
|
+
# 拼行用 `" "` 是本工具的历史口径——分隔符会改变文本相似度读数,不能随手改。
|
|
150
|
+
for c in srtio.read_subs([p]):
|
|
151
|
+
out.append((c.start, c.text(" "), p.name))
|
|
152
|
+
if sort_by_time:
|
|
153
|
+
out.sort(key=lambda x: x[0])
|
|
154
|
+
return out
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def align(script: list[Entry], subs: list[tuple[float, str, str]],
|
|
158
|
+
window: int, fuzzy: float) -> dict:
|
|
159
|
+
"""游标对齐。返回统计,同时就地把 Entry.matched 填上。"""
|
|
160
|
+
sn = [norm(e.text) for e in script]
|
|
161
|
+
by_text: dict[str, list[int]] = {}
|
|
162
|
+
for i, t in enumerate(sn):
|
|
163
|
+
by_text.setdefault(t, []).append(i)
|
|
164
|
+
sm = SequenceMatcher(None)
|
|
165
|
+
|
|
166
|
+
cursor = 0
|
|
167
|
+
n_exact = n_fuzzy = n_none = n_jump = 0
|
|
168
|
+
sub_hit: list[int] = []
|
|
169
|
+
for si, (_, raw_text, _) in enumerate(subs):
|
|
170
|
+
t = norm(raw_text)
|
|
171
|
+
if not t:
|
|
172
|
+
sub_hit.append(-1)
|
|
173
|
+
continue
|
|
174
|
+
hit = -1
|
|
175
|
+
cand = by_text.get(t)
|
|
176
|
+
if cand:
|
|
177
|
+
# 优先游标前方最近的一个;都在后方说明是分支跳转/回读
|
|
178
|
+
ahead = [i for i in cand if i >= cursor]
|
|
179
|
+
hit = ahead[0] if ahead else cand[-1]
|
|
180
|
+
if not ahead or hit - cursor > window:
|
|
181
|
+
n_jump += 1
|
|
182
|
+
n_exact += 1
|
|
183
|
+
else:
|
|
184
|
+
# 窗口内模糊找。**只在窗口内做**:全局模糊搜 34k 条对 23k 条字幕
|
|
185
|
+
# 是 8 亿次比较,而且会把无关的相似句配上,比不配还糟。
|
|
186
|
+
# 剪枝**不改结果**(同 align_unordered):两个 quick 比值都是 ratio 的上界,
|
|
187
|
+
# 够不着当前最好分的不必真算(lower5 原始轨 37 s -> 8 s,JSON 逐字节相同)。
|
|
188
|
+
# a=字幕、b=剧本的先后保持原样。
|
|
189
|
+
lo, hi = cursor, min(len(script), cursor + window)
|
|
190
|
+
best, best_s = -1, fuzzy
|
|
191
|
+
sm.set_seq1(t)
|
|
192
|
+
for i in range(lo, hi):
|
|
193
|
+
sm.set_seq2(sn[i])
|
|
194
|
+
if sm.real_quick_ratio() <= best_s or sm.quick_ratio() <= best_s:
|
|
195
|
+
continue
|
|
196
|
+
s = sm.ratio()
|
|
197
|
+
if s > best_s:
|
|
198
|
+
best, best_s = i, s
|
|
199
|
+
if best >= 0:
|
|
200
|
+
hit, n_fuzzy = best, n_fuzzy + 1
|
|
201
|
+
script[best].sim = round(best_s, 3)
|
|
202
|
+
else:
|
|
203
|
+
n_none += 1
|
|
204
|
+
sub_hit.append(hit)
|
|
205
|
+
if hit >= 0:
|
|
206
|
+
if script[hit].matched < 0:
|
|
207
|
+
script[hit].matched = si
|
|
208
|
+
cursor = hit + 1
|
|
209
|
+
return {"subs": len(subs), "exact": n_exact, "fuzzy": n_fuzzy,
|
|
210
|
+
"unmatched_subs": n_none, "jumps": n_jump, "sub_hit": sub_hit}
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def align_unordered(script: list[Entry], subs: list[tuple[float, str, str]],
|
|
214
|
+
lo: int, hi: int, fuzzy: float) -> dict:
|
|
215
|
+
"""不走游标,只问"区间内每条剧本,字幕里有没有人读到过"。
|
|
216
|
+
|
|
217
|
+
**用途 1 必须用这个。** 游标对齐假设字幕流跟着剧本顺序走——对一条字幕带成立,
|
|
218
|
+
对"画面全部文字"不成立:合并流里 UI、水印、时间戳、弹幕占了八成以上,
|
|
219
|
+
它们不在剧本里,却会把游标甩得到处跳。实测同一份数据,
|
|
220
|
+
游标口径把召回从 62%(只看字幕带)压到 39%,**加进更多正确的文字反而更低**,
|
|
221
|
+
那不是管线的性质,是量法的错。
|
|
222
|
+
|
|
223
|
+
代价:无序判据会高估——同一句话在别处出现也算命中。所以它只适合回答
|
|
224
|
+
"画面上这处文字有没有被抓到",不适合回答"时间对不对"。
|
|
225
|
+
"""
|
|
226
|
+
idx = {}
|
|
227
|
+
for si, (_, t, _) in enumerate(subs):
|
|
228
|
+
n = norm(t)
|
|
229
|
+
if n:
|
|
230
|
+
idx.setdefault(n, si)
|
|
231
|
+
by_len: dict[int, list[str]] = {}
|
|
232
|
+
for n in idx:
|
|
233
|
+
by_len.setdefault(len(n), []).append(n)
|
|
234
|
+
n_exact = n_fuzzy = 0
|
|
235
|
+
for e in script[lo:hi + 1]:
|
|
236
|
+
n = norm(e.text)
|
|
237
|
+
if not n:
|
|
238
|
+
continue
|
|
239
|
+
if n in idx:
|
|
240
|
+
e.matched, n_exact = idx[n], n_exact + 1
|
|
241
|
+
continue
|
|
242
|
+
# 长度门按**比例**放宽,不是固定 ±3:我们的 cue 可能把说话人名和台词并在一起,
|
|
243
|
+
# 或者 OCR 多吞/少读几个字,固定窗会把这些正当的匹配挡在外面。
|
|
244
|
+
# 自检口径:无序判据永远不该低于游标判据(它是更松的问法)——
|
|
245
|
+
# 固定 ±3 时实测 310 < 329,就是被这个门挡的。
|
|
246
|
+
slack = max(4, int(0.4 * len(n)))
|
|
247
|
+
best, best_s = -1, fuzzy
|
|
248
|
+
# 剪枝**不改结果**:real_quick_ratio(只看长度)和 quick_ratio(字符多重集
|
|
249
|
+
# 交集)都是 ratio 的上界,达不到当前最好分就没必要真跑 ratio。
|
|
250
|
+
# 整片上候选是万级,少了这一步一次测量要跑一小时以上。
|
|
251
|
+
# 两个序列的先后**保持原样**(a=剧本行、b=字幕行):SequenceMatcher 的
|
|
252
|
+
# ratio 不保证左右对称,换个方向可能量出别的数。
|
|
253
|
+
sm = SequenceMatcher(None)
|
|
254
|
+
sm.set_seq1(n)
|
|
255
|
+
for L in range(max(1, len(n) - slack), len(n) + slack + 1):
|
|
256
|
+
for c in by_len.get(L, ()):
|
|
257
|
+
sm.set_seq2(c)
|
|
258
|
+
if sm.real_quick_ratio() <= best_s or sm.quick_ratio() <= best_s:
|
|
259
|
+
continue
|
|
260
|
+
r = sm.ratio()
|
|
261
|
+
if r > best_s:
|
|
262
|
+
best, best_s = idx[c], r
|
|
263
|
+
if best >= 0:
|
|
264
|
+
e.matched, e.sim, n_fuzzy = best, round(best_s, 3), n_fuzzy + 1
|
|
265
|
+
return {"subs": len(subs), "exact": n_exact, "fuzzy": n_fuzzy,
|
|
266
|
+
"unmatched_subs": -1, "jumps": 0, "sub_hit": []}
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def gap_runs(flags: list[bool]) -> list[tuple[int, int]]:
|
|
270
|
+
"""连续 False 段,返回 (起点下标, 长度)。"""
|
|
271
|
+
out, i = [], 0
|
|
272
|
+
while i < len(flags):
|
|
273
|
+
if flags[i]:
|
|
274
|
+
i += 1
|
|
275
|
+
continue
|
|
276
|
+
j = i
|
|
277
|
+
while j < len(flags) and not flags[j]:
|
|
278
|
+
j += 1
|
|
279
|
+
out.append((i, j - i))
|
|
280
|
+
i = j
|
|
281
|
+
return out
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def clen(t: str) -> int:
|
|
285
|
+
"""行长(原始字符数,不做 NFKC——见 script-corpus 报告『量行长不能用 NFKC』)。"""
|
|
286
|
+
return len("".join(t.replace(chr(92) + "N", "").split()))
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
BANDS = [(1, 3), (3, 5), (5, 7), (7, 11), (11, 21), (21, 10 ** 6)]
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def gap_report(dial: list[Entry], branch_run: int, long_is_branch: bool = True) -> dict:
|
|
293
|
+
"""未命中拆成长 / 短 gap,疑似真漏按内容字数三档、按行长分档。打印并返回中间量。
|
|
294
|
+
|
|
295
|
+
`dial` 是**已经定好的分母**(该上屏的那些条目,按剧本顺序),`matched` 已填。
|
|
296
|
+
本工具的 `main` 和 `flowocr.analyze.game_align` 共用这一份,别再抄第二份。
|
|
297
|
+
|
|
298
|
+
`long_is_branch`:长 gap 能不能解释成"分支没走到"。《魔裁》的剧本是线性文件,分支只能靠
|
|
299
|
+
gap 长度猜,所以本工具是 True——长 gap 不计入疑似真漏。`game_align` 传 False:那边分支和
|
|
300
|
+
没播的一截已经按对话图 / obs 锚点排出分母,**长 gap 里剩下的没有分支解释**(点得快、别的版式、
|
|
301
|
+
圈选假阳性……),照样拆出来报,但三档表把它一起算进疑似真漏(audit-6)。
|
|
302
|
+
|
|
303
|
+
返回的 `by_class`:档 -> [该档条目数, 其中疑似真漏, 疑似真漏里落在长 gap 的],
|
|
304
|
+
口径随 `long_is_branch`(True 时前两个数都不含长 gap、第三个恒为 0)。"""
|
|
305
|
+
from collections import Counter
|
|
306
|
+
hit = [e for e in dial if e.matched >= 0]
|
|
307
|
+
flags = [e.matched >= 0 for e in dial]
|
|
308
|
+
runs = gap_runs(flags)
|
|
309
|
+
long_runs = [(i, n) for i, n in runs if n >= branch_run]
|
|
310
|
+
short = [(i, n) for i, n in runs if n < branch_run]
|
|
311
|
+
n_long = sum(n for _, n in long_runs)
|
|
312
|
+
n_short = sum(n for _, n in short)
|
|
313
|
+
print(f"未命中 {len(dial)-len(hit)} 条,拆开看:")
|
|
314
|
+
if long_is_branch:
|
|
315
|
+
print(f" 连续 ≥{branch_run} 条的段:{len(long_runs)} 段 / {n_long} 条"
|
|
316
|
+
f"({n_long/max(1,len(dial)):.1%})—— **多半是分支没走到,不算漏**")
|
|
317
|
+
print(f" 零散 <{branch_run} 条的:{len(short)} 处 / {n_short} 条"
|
|
318
|
+
f"({n_short/max(1,len(dial)):.1%})—— **这才是疑似真漏**")
|
|
319
|
+
# **这个数会"系统变差反而变好"**(methodology-audit-4 报告):
|
|
320
|
+
# 多漏几条把一段凑够 `--branch-run` 长度,整段就被挪进"多半是分支没走到",
|
|
321
|
+
# 短 gap 反而下降。反例:七条真实上屏的台词,A 命中 1/4/7 -> 短 gap 4 条;
|
|
322
|
+
# B 再漏掉第 4 条 -> 五条连续未命中被判成长 gap -> 短 gap **0 条**。
|
|
323
|
+
# 所以短 gap 只能和**命中**一起读,单独下降不等于漏得少。
|
|
324
|
+
print(f" ⚠ 短 gap **不能单独读**:再多漏几条、把一段凑到 ≥{branch_run} 长,"
|
|
325
|
+
f"它会被移进长 gap,这个数反而下降。"
|
|
326
|
+
f"\n 要看变好还是变坏,**必须连着命中一起看**(命中 {len(hit)};"
|
|
327
|
+
f"长 gap {n_long} 条里可能藏着真漏,那一侧没有逐条清单)。")
|
|
328
|
+
else:
|
|
329
|
+
print(f" 连续 ≥{branch_run} 条的段:{len(long_runs)} 段 / {n_long} 条"
|
|
330
|
+
f"({n_long/max(1,len(dial)):.1%})—— 分支 / 没播的一截已经排出分母,"
|
|
331
|
+
f"**这里没有分支解释**,一起算进疑似真漏(成段地漏,要摆帧看是什么)")
|
|
332
|
+
print(f" 零散 <{branch_run} 条的:{len(short)} 处 / {n_short} 条"
|
|
333
|
+
f"({n_short/max(1,len(dial)):.1%})")
|
|
334
|
+
print(" 疑似真漏 = 全部未命中 = 分母 − 命中(逐档都是命中的镜像,不经过 gap 分桶)。")
|
|
335
|
+
|
|
336
|
+
c = Counter(n for _, n in runs)
|
|
337
|
+
print(" 未命中段长度分布:" + " ".join(f"{k}条×{v}" for k, v in sorted(c.items())[:8]))
|
|
338
|
+
|
|
339
|
+
# 按长度分档看漏条率。**这一档不是装饰**:第一次跑就看出未命中的中位长度只有 5 字,
|
|
340
|
+
# 但"短行本来就多"和"短行更容易漏"是两回事,必须除以同档的总数才知道。
|
|
341
|
+
short_idx = {i for i, n in short for i in range(i, i + n)}
|
|
342
|
+
long_idx = {i for i, n in long_runs for i in range(i, i + n)}
|
|
343
|
+
|
|
344
|
+
# **纯标点行必须单列**(methodology-audit-2 报告):它占疑似真漏的 60-71%,
|
|
345
|
+
# 病因是 **det 在字幕带里根本没出框**(抽帧逐条确认过),
|
|
346
|
+
# 和有内容字的行(多半是匹配器没锚上)不是一回事。
|
|
347
|
+
# 混在一个百分比里报,会得出"匹配器是杠杆"这种反的结论。
|
|
348
|
+
# **按"内容字数"分三档报**(owner 2026-09-06:trivial 的那些要单独分类)。
|
|
349
|
+
# 三档的病因和取舍完全不同,混在一个漏条率里报会把结论带反:
|
|
350
|
+
# 纯标点 det 没出框;不作强求,有原文时靠上下文 momentum
|
|
351
|
+
# trivial 1-2 字 判别力接近零,锚不锚得住看上下文
|
|
352
|
+
# 有内容 >=3 字 这一档才是真的"该抓到没抓到"
|
|
353
|
+
# 口径随 long_is_branch:True 时长 gap 整段不进分母也不算漏;False 时全部未命中都算疑似真漏
|
|
354
|
+
counted = (lambda k: k not in long_idx) if long_is_branch else (lambda k: True)
|
|
355
|
+
missed_idx = short_idx if long_is_branch else short_idx | long_idx
|
|
356
|
+
n_missed = len(missed_idx)
|
|
357
|
+
print()
|
|
358
|
+
print("疑似真漏按**内容字数**分类(内容字 = 去掉标点和空白之后还剩几个字):")
|
|
359
|
+
print(f" {'类别':<16}{'剧本条目':>9}{'疑似真漏':>9}{'漏条率':>9}{'占疑似真漏':>11}"
|
|
360
|
+
+ ("" if long_is_branch else f"{'其中长 gap':>11}"))
|
|
361
|
+
by_class = {}
|
|
362
|
+
for cls in evalkit.CLASSES:
|
|
363
|
+
tot = sum(1 for k, e in enumerate(dial) if evalkit.triviality(e.text) == cls and counted(k))
|
|
364
|
+
m = sum(1 for k in missed_idx if evalkit.triviality(dial[k].text) == cls)
|
|
365
|
+
ml = 0 if long_is_branch else sum(1 for k in long_idx
|
|
366
|
+
if evalkit.triviality(dial[k].text) == cls)
|
|
367
|
+
by_class[cls] = [tot, m, ml]
|
|
368
|
+
if not tot and not m:
|
|
369
|
+
continue
|
|
370
|
+
print(f" {cls:<16}{tot:>9}{m:>9}{m/max(1,tot):>9.1%}{m/max(1,n_missed):>11.1%}"
|
|
371
|
+
+ ("" if long_is_branch else f"{ml:>11}"))
|
|
372
|
+
print(" 归因请用 `dev_tools/probe_gap_locate.py`(问『这一处 det 出框了没有』),"
|
|
373
|
+
"别用『全局模糊搜』那类判据(script_align --raw-ocr 那一刀)。")
|
|
374
|
+
print()
|
|
375
|
+
print("疑似真漏按行长分档(分母是该档全部对话条目"
|
|
376
|
+
+ (",已排除『分支没走到』的长段):" if long_is_branch else "):"))
|
|
377
|
+
print(f" {'行长':>8}{'剧本条目':>9}{'疑似真漏':>9}{'漏条率':>9}")
|
|
378
|
+
for lo, hi in BANDS:
|
|
379
|
+
tot = sum(1 for k, e in enumerate(dial) if lo <= clen(e.text) < hi and counted(k))
|
|
380
|
+
m = sum(1 for k in missed_idx if lo <= clen(dial[k].text) < hi)
|
|
381
|
+
tag = f"{lo}-{hi-1}" if hi < 10 ** 6 else f"{lo}+"
|
|
382
|
+
if tot:
|
|
383
|
+
print(f" {tag:>8}{tot:>9}{m:>9}{m/tot:>9.2%}")
|
|
384
|
+
|
|
385
|
+
return {"runs": runs, "long_runs": long_runs, "short": short, "n_long": n_long,
|
|
386
|
+
"n_short": n_short, "short_idx": short_idx, "long_idx": long_idx,
|
|
387
|
+
"by_class": by_class}
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
def main() -> int:
|
|
391
|
+
ap = argparse.ArgumentParser()
|
|
392
|
+
ap.add_argument("--script", required=True, help="script-raw 的 Scripts 目录(提供 key/类型)")
|
|
393
|
+
ap.add_argument("--text", required=True, help="script-jp 的 Scripts 目录(提供干净文本)")
|
|
394
|
+
ap.add_argument("--subs", nargs="+", required=True,
|
|
395
|
+
help="字幕文件,**按通关先后给**,不要让 shell 打乱顺序")
|
|
396
|
+
ap.add_argument("--out", default=None, help="逐条结果写到这个 JSON")
|
|
397
|
+
ap.add_argument("--window", type=int, default=400,
|
|
398
|
+
help="游标前方的搜索窗口(剧本条目数)")
|
|
399
|
+
ap.add_argument("--fuzzy", type=float, default=0.75, help="窗口内模糊匹配的相似度下限")
|
|
400
|
+
ap.add_argument("--branch-run", type=int, default=5,
|
|
401
|
+
help="未命中连续段达到这么长就算『分支没走到』,不计进疑似真漏")
|
|
402
|
+
ap.add_argument("--raw-ocr", nargs="*", default=None,
|
|
403
|
+
help="OCR 的**原始**输出(未经匹配器)。给了就把疑似真漏再拆一刀:"
|
|
404
|
+
"原始 OCR 里读到过的 = 匹配器丢的,没读到过的 = OCR 丢的。"
|
|
405
|
+
"两件事的修法完全不同,合成一个『漏条率』就没法定位")
|
|
406
|
+
ap.add_argument("--text-quality", action="store_true",
|
|
407
|
+
help="在配上的那些对上,报逐字相同率和省略号召回。"
|
|
408
|
+
"命中率只说『锚没锚上』,这两个数说『读得有多像』——"
|
|
409
|
+
"下游匹配器是相似度驱动的,短行差一个字就掉到阈值下面")
|
|
410
|
+
ap.add_argument("--unordered", action="store_true",
|
|
411
|
+
help="不走游标,只问『区间内每条剧本有没有被读到过』。"
|
|
412
|
+
"**用途 1 必须开**(合并流里八成条目不在剧本里,游标会被甩飞);"
|
|
413
|
+
"代价是会高估——同一句话在别处出现也算命中")
|
|
414
|
+
ap.add_argument("--usage1", action="store_true",
|
|
415
|
+
help="用途 1 口径:分母 = 区间内**全部**剧本条目(对话 + 辩论屏 + 选项 "
|
|
416
|
+
"+ Toast),因为它们都在屏幕上。默认是用途 2 口径:只算字幕带。"
|
|
417
|
+
"两个用途的分母不同,这是它们唯一分岔的地方")
|
|
418
|
+
ap.add_argument("--same-timeline", action="store_true",
|
|
419
|
+
help="多个字幕文件属于**同一段视频的不同轨**(用途 1:字幕带+辩论屏+UI),"
|
|
420
|
+
"按时间归并。默认是相反的假设:多个文件是一次通关的先后段落,首尾相接")
|
|
421
|
+
ap.add_argument("--raw-sim", type=float, default=0.8,
|
|
422
|
+
help="判定『原始 OCR 读到过』的相似度下限。**别用完全相等**:"
|
|
423
|
+
"`ぐっ……!?` 被读成 `ぐっ…....!?` 也算读到了,"
|
|
424
|
+
"那是匹配器的锅不是 OCR 的锅")
|
|
425
|
+
ap.add_argument("--include-cmds", default="",
|
|
426
|
+
help="把这些命令类条目放回分母(逗号分隔,如 @printDebate)。"
|
|
427
|
+
"用途 1 要统计画面全部文字时用;字幕带口径保持默认")
|
|
428
|
+
ap.add_argument("--time-range", nargs=2, type=float, default=None,
|
|
429
|
+
metavar=("S", "E"),
|
|
430
|
+
help="只取起点落在 [S, E) 秒内的字幕条目。跑窗口实验时两边都要加")
|
|
431
|
+
ap.add_argument("--span-ref", nargs="+", default=None,
|
|
432
|
+
help="用这份字幕先圈出**剧本区间**,再把统计限定到该区间。"
|
|
433
|
+
"只跑一个时间窗时必须用它:否则分母是整部剧本,"
|
|
434
|
+
"命中率会被稀释成无意义的数。两边(基线/我们)用同一个 "
|
|
435
|
+
"--span-ref,分母才一样")
|
|
436
|
+
ap.add_argument("--show", type=int, default=8, help="列几条疑似真漏")
|
|
437
|
+
args = ap.parse_args()
|
|
438
|
+
|
|
439
|
+
if args.usage1:
|
|
440
|
+
global ALL_ENTRIES
|
|
441
|
+
ALL_ENTRIES = True
|
|
442
|
+
print("(用途 1 口径:分母含辩论屏/选项/Toast,即区间内全部剧本条目)")
|
|
443
|
+
for c in (x.strip() for x in args.include_cmds.split(",") if x.strip()):
|
|
444
|
+
NON_DIALOGUE_CMDS.discard(c)
|
|
445
|
+
print(f"({c} 已放回分母)")
|
|
446
|
+
script = load_script(Path(args.script), Path(args.text))
|
|
447
|
+
|
|
448
|
+
def load_subs(paths):
|
|
449
|
+
xs = read_subs([Path(x) for x in paths], sort_by_time=args.same_timeline)
|
|
450
|
+
if args.time_range:
|
|
451
|
+
lo, hi = args.time_range
|
|
452
|
+
xs = [x for x in xs if lo <= x[0] < hi]
|
|
453
|
+
return xs
|
|
454
|
+
|
|
455
|
+
subs = load_subs(args.subs)
|
|
456
|
+
rng = (f",限定 {args.time_range[0]/60:.0f}-{args.time_range[1]/60:.0f} min"
|
|
457
|
+
if args.time_range else "")
|
|
458
|
+
print(f"剧本 {len(script)} 条,字幕 {len(subs)} 条({len(args.subs)} 个文件{rng})")
|
|
459
|
+
|
|
460
|
+
# 只跑一个时间窗时,分母必须是**该窗覆盖到的那段剧本**,不是整部剧本。
|
|
461
|
+
# 用一份参照字幕圈出区间,两边共用同一个区间,命中率才可比。
|
|
462
|
+
span = None
|
|
463
|
+
if args.span_ref:
|
|
464
|
+
ref = load_subs(args.span_ref)
|
|
465
|
+
align(script, ref, args.window, args.fuzzy)
|
|
466
|
+
got = [i for i, e in enumerate(script) if e.matched >= 0]
|
|
467
|
+
if not got:
|
|
468
|
+
raise SystemExit("--span-ref 一条都没对上剧本,圈不出区间")
|
|
469
|
+
span = (min(got), max(got))
|
|
470
|
+
print(f"区间由参照字幕({len(ref)} 条)圈出:剧本 [{span[0]}, {span[1]}]"
|
|
471
|
+
f",共 {span[1]-span[0]+1} 条")
|
|
472
|
+
for e in script: # 重置,别把参照的命中算进被测那一份
|
|
473
|
+
e.matched, e.sim = -1, 0.0
|
|
474
|
+
|
|
475
|
+
if args.unordered:
|
|
476
|
+
if span is None:
|
|
477
|
+
raise SystemExit("--unordered 必须配 --span-ref:无序判据得先有区间,否则分母是整部剧本")
|
|
478
|
+
st = align_unordered(script, subs, span[0], span[1], args.fuzzy)
|
|
479
|
+
print(f"字幕侧(无序口径):{st['subs']} 条;"
|
|
480
|
+
f"剧本被命中 精确 {st['exact']} + 模糊 {st['fuzzy']}"
|
|
481
|
+
f"(**无序判据不报『对不上剧本』**:用途 1 的合并流里本来就大量是 UI、"
|
|
482
|
+
f"水印、时间戳,不在剧本里不等于幽灵)")
|
|
483
|
+
else:
|
|
484
|
+
st = align(script, subs, args.window, args.fuzzy)
|
|
485
|
+
n_empty = st["subs"] - st["exact"] - st["fuzzy"] - st["unmatched_subs"]
|
|
486
|
+
print(f"字幕侧:精确 {st['exact']}({st['exact']/max(1,st['subs']):.1%})、"
|
|
487
|
+
f"模糊 {st['fuzzy']}、**对不上剧本 {st['unmatched_subs']}"
|
|
488
|
+
f"({st['unmatched_subs']/max(1,st['subs']):.1%},这批是幽灵条目的上限)**、"
|
|
489
|
+
f"跨窗跳转 {st['jumps']}"
|
|
490
|
+
+ (f"、归一化后为空 {n_empty}(不进上面任何一档)" if n_empty else ""))
|
|
491
|
+
# **在匹配后的 .ass 上,这个数结构上只能是 0。**
|
|
492
|
+
# match_ref 把匹配不上的 cue 整条丢弃、留下的正文一律换成剧本原文,
|
|
493
|
+
# 所以"对不上剧本"根本无从产生(methodology-audit-2 报告)。
|
|
494
|
+
# 幽灵条目只能在**匹配前**的原始轨上量。
|
|
495
|
+
if st["unmatched_subs"] == 0 and st["fuzzy"] == 0 and st["subs"] > 100:
|
|
496
|
+
print(" ⚠ **对不上剧本 = 0 且模糊 = 0**:这几乎一定是"
|
|
497
|
+
"在**匹配器的输出**上量的(正文已被换成剧本原文),"
|
|
498
|
+
"这个 0 是结构性的,**不能当成『没有幽灵条目』的证据**。"
|
|
499
|
+
"要量幽灵请拿匹配前的原始轨。")
|
|
500
|
+
|
|
501
|
+
lo_i, hi_i = span if span else (0, len(script) - 1)
|
|
502
|
+
dial = [e for i, e in enumerate(script) if is_dialogue(e) and lo_i <= i <= hi_i]
|
|
503
|
+
n_other = sum(1 for i, e in enumerate(script)
|
|
504
|
+
if not is_dialogue(e) and lo_i <= i <= hi_i)
|
|
505
|
+
hit = [e for e in dial if e.matched >= 0]
|
|
506
|
+
print()
|
|
507
|
+
tail = ("(用途 1 口径:选项/Toast/命令**已算进分母**)" if args.usage1
|
|
508
|
+
else f"(另有 {n_other} 条是选项/Toast/命令,不计)")
|
|
509
|
+
print(f"剧本侧:**该上屏的对话类 {len(dial)} 条**{tail},"
|
|
510
|
+
f"命中 {len(hit)}({len(hit)/max(1,len(dial)):.1%})")
|
|
511
|
+
|
|
512
|
+
gr = gap_report(dial, args.branch_run)
|
|
513
|
+
runs, long_runs, short = gr["runs"], gr["long_runs"], gr["short"]
|
|
514
|
+
n_long, n_short, short_idx = gr["n_long"], gr["n_short"], gr["short_idx"]
|
|
515
|
+
|
|
516
|
+
if args.text_quality:
|
|
517
|
+
# 配上之后,读得**有多像**。用途 2 里这一项直接决定下游匹配器锚不锚得住:
|
|
518
|
+
# match.py 是相似度驱动的,短行差一个字就掉到阈值下面。
|
|
519
|
+
BRK = chr(92) + "N"
|
|
520
|
+
ELL = re.compile("…+")
|
|
521
|
+
pairs = [(script[i].text.replace(BRK, ""), subs[script[i].matched][1])
|
|
522
|
+
for i in range(lo_i, hi_i + 1) if script[i].matched >= 0]
|
|
523
|
+
if pairs:
|
|
524
|
+
ex = sum(1 for x, y in pairs if norm(x) == norm(y))
|
|
525
|
+
tail = re.compile(r"[.。!!??、,))」』 ·・…]+$")
|
|
526
|
+
ex_t = sum(1 for x, y in pairs if tail.sub("", norm(x)) == tail.sub("", norm(y)))
|
|
527
|
+
ns = sum(len(ELL.findall(x)) for x, y in pairs)
|
|
528
|
+
no = sum(len(ELL.findall(y)) for x, y in pairs)
|
|
529
|
+
print()
|
|
530
|
+
warn = (" ⚠ 无序口径下多条剧本可能配到同一条字幕,这些数会被重复计入"
|
|
531
|
+
if args.unordered else "")
|
|
532
|
+
print(f"文本质量(在配上的 {len(pairs)} 对上):{warn}")
|
|
533
|
+
print(f" 逐字相同 {ex}({ex/len(pairs):.1%});"
|
|
534
|
+
f"剥掉两边尾部标点后 {ex_t}({ex_t/len(pairs):.1%})")
|
|
535
|
+
print(f" 省略号串:剧本 {ns} 处,我们读出 {no} 处"
|
|
536
|
+
f"(**召回 {no/max(1,ns):.1%}**)")
|
|
537
|
+
|
|
538
|
+
if args.usage1:
|
|
539
|
+
# 用途 1 的重点不是总召回,是**各类上屏文字分别抓到多少**:
|
|
540
|
+
# 字幕带是一回事,辩论屏、选项、Toast 走的是别的 UI、别的字号、别的位置。
|
|
541
|
+
import collections as _c
|
|
542
|
+
grp = _c.defaultdict(lambda: [0, 0])
|
|
543
|
+
for e in dial:
|
|
544
|
+
k = e.cmd or ("Choice" if e.kind == "Choice" else
|
|
545
|
+
"Toast" if e.kind == "Toast" else "对话/旁白")
|
|
546
|
+
grp[k][0] += 1
|
|
547
|
+
grp[k][1] += e.matched >= 0
|
|
548
|
+
print()
|
|
549
|
+
print("按条目类型拆开(用途 1 的重点):")
|
|
550
|
+
print(f" {'类型':<16}{'剧本条目':>9}{'命中':>7}{'召回':>9}")
|
|
551
|
+
for k, (tot, got) in sorted(grp.items(), key=lambda x: -x[1][0]):
|
|
552
|
+
print(f" {k:<16}{tot:>9}{got:>7}{got/max(1,tot):>9.1%}")
|
|
553
|
+
|
|
554
|
+
if args.raw_ocr:
|
|
555
|
+
# **不能用完全相等做归因。** 实测望言把 `ぐっ……!?` 读成 `ぐっ…....!?`
|
|
556
|
+
# (省略号读成点),完全相等会把它算成"OCR 没读出来",但那明明是读到了、
|
|
557
|
+
# 只是匹配器没锚上。两件事的修法完全不同,归错了整条结论就偏。
|
|
558
|
+
# 所以先按完全相等找,再在**同长度附近**做一次模糊找。
|
|
559
|
+
raw_list = [norm(t) for _, t, _ in read_subs([Path(x) for x in args.raw_ocr])]
|
|
560
|
+
seen = set(raw_list)
|
|
561
|
+
by_len: dict[int, list[str]] = {}
|
|
562
|
+
for t in seen:
|
|
563
|
+
by_len.setdefault(len(t), []).append(t)
|
|
564
|
+
|
|
565
|
+
def read_by_ocr(t: str) -> bool:
|
|
566
|
+
if t in seen:
|
|
567
|
+
return True
|
|
568
|
+
for L in range(max(1, len(t) - 3), len(t) + 4):
|
|
569
|
+
for c in by_len.get(L, ()):
|
|
570
|
+
if SequenceMatcher(None, t, c).ratio() >= args.raw_sim:
|
|
571
|
+
return True
|
|
572
|
+
return False
|
|
573
|
+
|
|
574
|
+
missed = [dial[k] for k in sorted(short_idx)]
|
|
575
|
+
hit_raw = {id(e): read_by_ocr(norm(e.text)) for e in missed}
|
|
576
|
+
n_in = sum(1 for e in missed if hit_raw[id(e)])
|
|
577
|
+
print()
|
|
578
|
+
print(f"再拆一刀(原始 OCR 去重 {len(seen)} 条):")
|
|
579
|
+
print(f" {'行长':>7}{'疑似真漏':>9}{'OCR 读到过':>11}{'OCR 也没读出':>13}")
|
|
580
|
+
for lo, hi in BANDS:
|
|
581
|
+
sel = [e for e in missed if lo <= clen(e.text) < hi]
|
|
582
|
+
if not sel:
|
|
583
|
+
continue
|
|
584
|
+
k = sum(1 for e in sel if hit_raw[id(e)])
|
|
585
|
+
tag = f"{lo}-{hi-1}" if hi < 10 ** 6 else f"{lo}+"
|
|
586
|
+
print(f" {tag:>7}{len(sel):>9}{k:>11}{len(sel)-k:>13}")
|
|
587
|
+
pu = [e for e in missed if evalkit.is_punct_only(e.text)]
|
|
588
|
+
pu_in = sum(1 for e in pu if hit_raw[id(e)])
|
|
589
|
+
print(" 按内容字数分档看这一刀(**判据在 trivial 那两档上已经饱和**):")
|
|
590
|
+
for cls in evalkit.CLASSES:
|
|
591
|
+
sel = [e for e in missed if evalkit.triviality(e.text) == cls]
|
|
592
|
+
if not sel:
|
|
593
|
+
continue
|
|
594
|
+
k = sum(1 for e in sel if hit_raw[id(e)])
|
|
595
|
+
print(f" {cls:<16}{evalkit.denom(k, len(sel)):>18} 判为『匹配器丢的』")
|
|
596
|
+
print(f" → **{n_in/max(1,len(missed)):.1%} 是匹配器丢的**(OCR 明明读到过),"
|
|
597
|
+
f"**{1-n_in/max(1,len(missed)):.1%} 是 OCR 就没读出来**")
|
|
598
|
+
# **这一刀是全局搜,对纯标点行几乎恒真**(methodology-audit-2 报告):
|
|
599
|
+
# 整片里 `……!` 这类行有几千条,"别处出现过"必然成立,
|
|
600
|
+
# 于是它们被一律记成"匹配器丢的"——而抽帧看到的是那一处 det 根本没出框。
|
|
601
|
+
# docs/dev-guide/verification.md「归因:问『这一处发生了什么』」 早写过这条,这里又踩了一次。
|
|
602
|
+
print(f" ⚠ **上面这刀是全局模糊搜,不问『这一处发生了什么』。**"
|
|
603
|
+
f"纯标点行 {evalkit.denom(pu_in, len(pu))} 被判成"
|
|
604
|
+
f"『匹配器丢的』,而它们占疑似真漏的 "
|
|
605
|
+
f"{evalkit.denom(len(pu), len(missed))}——"
|
|
606
|
+
f"整片里这类行有几千条,判据在这一档上已经饱和。"
|
|
607
|
+
f"\n 要归因请用 `dev_tools/probe_gap_locate.py`(位置判据:"
|
|
608
|
+
f"这一时刻字幕带里 det 出框了没有)。")
|
|
609
|
+
|
|
610
|
+
if short and args.show:
|
|
611
|
+
print(f"\n疑似真漏抽样(前 {args.show} 条):")
|
|
612
|
+
for i, n in short[:args.show]:
|
|
613
|
+
for e in dial[i:i + n]:
|
|
614
|
+
print(f" [{e.kind:<10}] {e.src:<34} {e.text[:52]}")
|
|
615
|
+
|
|
616
|
+
if args.out:
|
|
617
|
+
p = Path(args.out)
|
|
618
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
619
|
+
p.write_text(json.dumps({
|
|
620
|
+
"script_entries": len(script), "dialogue_entries": len(dial),
|
|
621
|
+
"subs": st["subs"], "exact": st["exact"], "fuzzy": st["fuzzy"],
|
|
622
|
+
"unmatched_subs": st["unmatched_subs"],
|
|
623
|
+
"dialogue_hit": len(hit),
|
|
624
|
+
"gap_long_runs": len(long_runs), "gap_long_entries": n_long,
|
|
625
|
+
"gap_short_places": len(short), "gap_short_entries": n_short,
|
|
626
|
+
"branch_run": args.branch_run,
|
|
627
|
+
# ⚠ `missed` **只有短 gap**(`--raw-ocr` 那套归因就是按它做的)。长 gap 另列一份:
|
|
628
|
+
# 2026-09-18 审计发现有工具把 `missed` 当"全部未命中"用(`rec_engine_arm.sh` 的命中集合差),
|
|
629
|
+
# 而长 gap 不为 0 时那就是静默少算。**要"全部未命中"的人自己把两份并起来**,
|
|
630
|
+
# 并且拿 `gap_short_entries + gap_long_entries` 对一次总数。
|
|
631
|
+
"missed": [{"key": e.key, "kind": e.kind, "src": e.src, "text": e.text}
|
|
632
|
+
for i, n in short for e in dial[i:i + n]],
|
|
633
|
+
"missed_long": [{"key": e.key, "kind": e.kind, "src": e.src, "text": e.text}
|
|
634
|
+
for i, n in long_runs for e in dial[i:i + n]],
|
|
635
|
+
}, ensure_ascii=False, indent=1), encoding="utf-8")
|
|
636
|
+
print(f"\n逐条结果 -> {p}")
|
|
637
|
+
return 0
|
|
638
|
+
|
|
639
|
+
|
|
640
|
+
if __name__ == "__main__":
|
|
641
|
+
raise SystemExit(main())
|