flowocr 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. flowocr/__init__.py +15 -0
  2. flowocr/analyze/__init__.py +8 -0
  3. flowocr/analyze/align.py +151 -0
  4. flowocr/analyze/build_tracks.py +2886 -0
  5. flowocr/analyze/cluster_layers.py +155 -0
  6. flowocr/analyze/game_align.py +652 -0
  7. flowocr/analyze/gamescript.py +1220 -0
  8. flowocr/analyze/gtdbundle.py +306 -0
  9. flowocr/analyze/match.py +67 -0
  10. flowocr/analyze/matchers/__init__.py +5 -0
  11. flowocr/analyze/matchers/gametext.py +70 -0
  12. flowocr/analyze/merge_nameplate.py +178 -0
  13. flowocr/analyze/models/slot_pair.json +148 -0
  14. flowocr/analyze/nameplate.py +148 -0
  15. flowocr/analyze/pair_features.py +168 -0
  16. flowocr/analyze/pair_model.py +127 -0
  17. flowocr/analyze/refine_boundaries.py +409 -0
  18. flowocr/analyze/script_align.py +641 -0
  19. flowocr/analyze/scriptmatch.py +1018 -0
  20. flowocr/analyze/slot_learned.py +306 -0
  21. flowocr/analyze/slot_lines.py +251 -0
  22. flowocr/analyze/slot_modes.py +143 -0
  23. flowocr/analyze/slot_pairs.py +562 -0
  24. flowocr/analyze/slot_veto.py +104 -0
  25. flowocr/analyze/uigate.py +351 -0
  26. flowocr/artifacts/__init__.py +5 -0
  27. flowocr/artifacts/evalkit.py +123 -0
  28. flowocr/artifacts/matchedio.py +75 -0
  29. flowocr/artifacts/srtio.py +166 -0
  30. flowocr/artifacts/tracksio.py +345 -0
  31. flowocr/extensions.py +77 -0
  32. flowocr/extract/__init__.py +8 -0
  33. flowocr/extract/childproc.py +118 -0
  34. flowocr/extract/decode_proc.py +511 -0
  35. flowocr/extract/decode_shards.py +574 -0
  36. flowocr/extract/detpost.py +215 -0
  37. flowocr/extract/edge_proc.py +314 -0
  38. flowocr/extract/edge_refine.py +598 -0
  39. flowocr/extract/fast_det.py +214 -0
  40. flowocr/extract/ffcheck.py +87 -0
  41. flowocr/extract/framegrid.py +265 -0
  42. flowocr/extract/framesource.py +1264 -0
  43. flowocr/extract/ocr_args.py +754 -0
  44. flowocr/extract/ocr_complete.py +231 -0
  45. flowocr/extract/ocr_parallel.py +208 -0
  46. flowocr/extract/ort_server.py +632 -0
  47. flowocr/extract/ortclient.py +363 -0
  48. flowocr/extract/ptsclock.py +150 -0
  49. flowocr/extract/recdecode.py +78 -0
  50. flowocr/extract/recort.py +162 -0
  51. flowocr/extract/recpack.py +94 -0
  52. flowocr/extract/recpool.py +206 -0
  53. flowocr/extract/recprep.py +71 -0
  54. flowocr/extract/refine_video.py +271 -0
  55. flowocr/extract/regions.py +387 -0
  56. flowocr/extract/reuse_v2.py +593 -0
  57. flowocr/extract/run_groups.py +247 -0
  58. flowocr/extract/run_ocr2.py +1605 -0
  59. flowocr/extract/supervisor.py +261 -0
  60. flowocr/extract/timeline.py +84 -0
  61. flowocr/extract/typewriter_fuse.py +394 -0
  62. flowocr/models.py +263 -0
  63. flowocr/output/__init__.py +3 -0
  64. flowocr/output/export.py +205 -0
  65. flowocr/output/layout.py +210 -0
  66. flowocr/output/presets/__init__.py +6 -0
  67. flowocr/output/presets/_overlay.py +40 -0
  68. flowocr/output/presets/default.py +21 -0
  69. flowocr/output/presets/default_all.py +20 -0
  70. flowocr/output/presets/dev.py +21 -0
  71. flowocr/output/presets/matched_srt.py +32 -0
  72. flowocr/output/presets/script.py +20 -0
  73. flowocr/output/presets/srt_main.py +33 -0
  74. flowocr/output/render.py +82 -0
  75. flowocr/output/run_srt.py +83 -0
  76. flowocr/output/script.py +541 -0
  77. flowocr/paths.py +168 -0
  78. flowocr/provenance.py +119 -0
  79. flowocr/typeset/__init__.py +9 -0
  80. flowocr/typeset/__main__.py +38 -0
  81. flowocr/typeset/assfile.py +216 -0
  82. flowocr/typeset/core.py +821 -0
  83. flowocr/typeset/fx/__init__.py +11 -0
  84. flowocr-0.1.0.dist-info/METADATA +109 -0
  85. flowocr-0.1.0.dist-info/RECORD +91 -0
  86. flowocr-0.1.0.dist-info/WHEEL +5 -0
  87. flowocr-0.1.0.dist-info/entry_points.txt +10 -0
  88. flowocr-0.1.0.dist-info/licenses/LICENSE +674 -0
  89. flowocr-0.1.0.dist-info/licenses/LICENSES/Apache-2.0.txt +201 -0
  90. flowocr-0.1.0.dist-info/licenses/LICENSES/PP-OCRv6-NOTICE.md +18 -0
  91. flowocr-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,394 @@
1
+ """打字机的**首字出现 / 全字出现**:多信号融合(`refine_boundaries` 调它;纯逻辑,不碰视频,守卫够得着)。
2
+
3
+ 为什么不再用"整框相关系数过门"(2026-09-12,owner 看预览说前两个时间点不准):
4
+ 最后一两个字对整框相关系数的贡献太小,**还没打完就过门**,全字出现系统性偏早 4 帧以上;
5
+ 首字出现拿"不像窗口首帧"判,背景一动、上一句淡出也触发。
6
+
7
+ 几路互相独立的信号,各有各的死角:
8
+
9
+ | 信号 | 首字 | 全字 | 死角 |
10
+ | --- | --- | --- | --- |
11
+ | OCR 中途读数(`ocr_fit`):打字中途读到的半截字数对时间做直线,外推到第 0.5 / N−0.5 个字 | ✓ | ✓ | 多数事件只有 1–2 个中途读数(只有 1 个时借同一段视频的典型速度) |
12
+ | 首字格"开始出现"(`rise_onset`):首字格字形一致率从定稿帧往回走到单调上升的起点 | ✓ | | 立绘压在首字格上 |
13
+ | 逐字格定稿时刻 + Theil–Sen 直线(`cell_fit`):截距 = 首字,末端 = 全字 | ✓ | ✓ | 半角字 / 淡入长的字体上截距偏晚 |
14
+ | 尾字格定稿(`glyph_onset`) | | ✓ | 淡入长时偏晚 |
15
+ | 整框不像窗口首帧(旧判据) | ✓ | | 背景一动就偏早 |
16
+
17
+ 融合(`fuse`):**OCR 证据给硬区间**(首字 ∈ [t_start − 一个采样间隔, t_start];全字不早于首字、
18
+ 也不早于任何一次"字还没读全"的读数),区间外的候选丢掉;剩下的按 `FUSE_TOL_US` 聚类取成员最多的一簇的中位数,
19
+ 平票取列表里靠前的(顺序就是单信号可信度)。一个都不剩就退回采样级值、夹进区间,并记 `none`。
20
+ 全字**只防晚很多**:不晚于 OCR 读全 + `LATE_CAP_US`;没有两路同意时不晚于最早候选 + `LATE_CAP_US`、记 `capped`
21
+ (owner:晚 10 帧以内都还好,晚很多是灾难,早很多无非一开始就全显示)。
22
+
23
+ 常数只有像素 / 物理量(字形偏离 40、容差 30、一致率 0.6、同意 = 3 帧、"没读全" = 不到 80% 字数),
24
+ **没有 ms/char**(product-goals 第 7 条),也没有按游戏特化。
25
+
26
+ 验证(五款游戏 + 两款 galgame / RPG,逐帧盲标,`dev_tools/typewriter_eval.py` + `data/gt/typewriter-*.json`)
27
+ 见 text-effects 报告的"多信号融合"一节。
28
+ """
29
+ from __future__ import annotations
30
+
31
+ from difflib import SequenceMatcher
32
+
33
+ import numpy as np
34
+
35
+ GLYPH_DIFF = 40
36
+ """字形像素:和格内中位数的差至少这么多(灰度级)。"""
37
+ GLYPH_TOL = 30
38
+ """一帧里的字形像素和模板差这么多以内算"已经是最终样子"。"""
39
+ GLYPH_THR = 0.6
40
+ """字形像素里"已经是最终样子"的比例过这个门,这一格就算定稿。"""
41
+ EMPTY_THR = 0.90
42
+ """旧判据:整框和窗口首帧的相关系数低于这个,算"有新东西了"。"""
43
+ RISE_EPS = 0.01
44
+ """`rise_onset` 往回走时的去抖(分数是像素比例,这个量级是噪声)。"""
45
+ FUSE_TOL_US = 50_000
46
+ """两个候选差这么多以内算"同意"(60 fps 下 3 帧)。"""
47
+ LATE_CAP_US = 166_667
48
+ """全字最多允许比 OCR 读全 / 最早候选晚这么多(10 帧 @60fps)——owner 给的"晚了还能接受"的上限。"""
49
+ FUSE_PARTIAL = 0.8
50
+ """OCR 读到的字数不到终稿这么多,算"那一刻还没打完"。"""
51
+ OCR_MIN_SIM = 0.6
52
+ """中途读数和终稿前缀的相似度门。"""
53
+ CELL_MIN_PX = 8
54
+ """一个字格里字形像素少于这么多就不给它定时(空格 / 标点缝)。"""
55
+
56
+
57
+ def chars(s: str) -> str:
58
+ return "".join(s.split())
59
+
60
+
61
+ def ncc(a: np.ndarray, b: np.ndarray) -> float:
62
+ """归一化相关系数;常量图恒为 0(audit-4 C5 那条的来源,调用方要自己判"没找到")。"""
63
+ if a.shape != b.shape or a.size == 0:
64
+ return 0.0
65
+ a = a.astype(np.float32) - a.mean()
66
+ b = b.astype(np.float32) - b.mean()
67
+ d = float(np.linalg.norm(a) * np.linalg.norm(b))
68
+ return float((a * b).sum() / d) if d > 1e-6 else 0.0
69
+
70
+
71
+ def polarity(img: np.ndarray) -> bool:
72
+ """True = 亮字。看的是更极端的那一侧(偏离量的 95 分位)——一行里字占的像素最多。"""
73
+ dev = img.astype(np.int16) - int(np.median(img))
74
+ bright, dark = dev[dev > 0], -dev[dev < 0]
75
+ return bool((np.percentile(bright, 95) if bright.size else 0) >= (np.percentile(dark, 95) if dark.size else 0))
76
+
77
+
78
+ def glyph_mask(tpl: np.ndarray, bright: bool | None = None) -> np.ndarray:
79
+ """模板里算字形的像素:只取字那一侧的极性——两侧都算会把描边、底板边缘、背景纹理收进来。"""
80
+ t = tpl.astype(np.int16)
81
+ dev = t - int(np.median(t))
82
+ if bright is None:
83
+ bright = polarity(tpl)
84
+ return dev >= GLYPH_DIFF if bright else dev <= -GLYPH_DIFF
85
+
86
+
87
+ def glyph_score(frames: np.ndarray, tpl: np.ndarray, bright: bool | None = None) -> np.ndarray:
88
+ """逐帧:模板字形像素里,和模板差 ≤ `GLYPH_TOL` 的比例。字形像素太少(< 20)就全是 NaN。"""
89
+ mask = glyph_mask(tpl, bright)
90
+ if mask.sum() < 20:
91
+ return np.full(len(frames), np.nan)
92
+ t = tpl.astype(np.int16)
93
+ return (np.abs(frames.astype(np.int16) - t)[:, mask] <= GLYPH_TOL).mean(axis=1)
94
+
95
+
96
+ def last_onset(ok) -> int | None:
97
+ """最后一段一直持续到窗口末尾的"过门"的起点下标;末帧都不过门 = 没找到。"""
98
+ ok = list(ok)
99
+ if not ok or not ok[-1]:
100
+ return None
101
+ i = len(ok) - 1
102
+ while i > 0 and ok[i - 1]:
103
+ i -= 1
104
+ return i
105
+
106
+
107
+ def rise_onset(scores) -> int | None:
108
+ """首字格"开始出现":从最后一段过门的起点往回走,分数还在单调下降就继续,停在局部最低点的下一帧。
109
+ 字一帧弹出(星铁 / 绝区零 / 像素 RPG)时和定稿同一帧;逐字淡入(魔裁)时退到淡入起点。"""
110
+ s = [float(x) for x in scores]
111
+ i1 = next((i for i in range(len(s) - 1, -1, -1) if not s[i] >= GLYPH_THR), None)
112
+ if i1 is None or i1 == len(s) - 1:
113
+ return None
114
+ i1 += 1
115
+ i = i1
116
+ while i > 0 and s[i - 1] < s[i] - RISE_EPS:
117
+ i -= 1
118
+ return min(i + 1, i1) if i < i1 else i1
119
+
120
+
121
+ def theil_sen(xs, ys) -> tuple[float, float] | None:
122
+ """稳健直线 y = a + b·x(b 夹到 ≥ 0:字不会倒着打)。被立绘挡住 / 背景污染的字格当离群值。"""
123
+ if len(xs) < 2:
124
+ return None
125
+ x = np.asarray(xs, float)
126
+ y = np.asarray(ys, float)
127
+ i, j = np.triu_indices(len(x), 1)
128
+ dx = x[j] - x[i]
129
+ ok = dx != 0
130
+ b = max(0.0, float(np.median((y[j] - y[i])[ok] / dx[ok]))) if ok.any() else 0.0
131
+ return float(np.median(y - b * x)), b
132
+
133
+
134
+ def cell_fit(frames: np.ndarray, ts: list[int], x0: int, x1: int, n_chars: int) -> tuple[float, float] | None:
135
+ """整行按字数等分成格,每格找"定稿"帧,Theil–Sen 拟合 → (首字, 全字)。"""
136
+ if n_chars < 2:
137
+ return None
138
+ k = len(frames) - 1
139
+ mask = glyph_mask(frames[k], polarity(frames[k]))
140
+ close = np.abs(frames.astype(np.int16) - frames[k].astype(np.int16)) <= GLYPH_TOL
141
+ cw = (x1 - x0) / n_chars
142
+ xs, ys = [], []
143
+ for c in range(n_chars):
144
+ a, b = int(round(x0 + c * cw)), int(round(x0 + (c + 1) * cw))
145
+ m = mask[:, a:b]
146
+ if m.sum() < CELL_MIN_PX:
147
+ continue
148
+ o = last_onset(close[:, :, a:b][:, m].mean(axis=1) >= GLYPH_THR)
149
+ if o is not None:
150
+ xs.append(c)
151
+ ys.append(ts[o])
152
+ fit = theil_sen(xs, ys)
153
+ if fit is None:
154
+ return None
155
+ a, b = fit
156
+ return a, a + b * (n_chars - 1)
157
+
158
+
159
+ def ocr_points(rows, box, final: str) -> list[tuple[int, int]]:
160
+ """窗口里的 OCR 观测 → [(t_us, 读到几个字)]:同一行位置(y 重叠过半、左沿差不到 1.5 个字高)、
161
+ 文本像终稿的前缀。`rows` = [(t_us, box, text)],调用方先按时间圈好。"""
162
+ x0, y0, _, y1 = box
163
+ h = max(1.0, y1 - y0)
164
+ N = len(final)
165
+ best: dict[int, tuple[float, int]] = {}
166
+ for t, bx, text in rows:
167
+ if min(bx[3], y1) - max(bx[1], y0) < 0.5 * h or abs(bx[0] - x0) > 1.5 * h:
168
+ continue
169
+ q = chars(text)
170
+ if not q:
171
+ continue
172
+ s = SequenceMatcher(None, q, final[:len(q)]).ratio()
173
+ if s >= OCR_MIN_SIM and (t not in best or s > best[t][0]):
174
+ best[t] = (s, min(len(q), N))
175
+ return sorted((t, n) for t, (_, n) in best.items())
176
+
177
+
178
+ OCR_SLOPES_MIN = 5
179
+ """全片 OCR 打字速度(`v_ocr`)至少要几条拟得出斜率的句子,不够就不给、退回逐字格斜率 `v_cell`(2026-09-24)。
180
+ 正常素材要么一句都拟不出(魔裁:打字快,一句在一个采样间隔内打完),要么很多句(gi-s2 31、zzz-s1 9、hsr 364——都是过了下面单调性那道门之后的数,
181
+ 和 `v_cell` 差 2%~4%);只有两三句的时候那几句多半是框抖出来的假中途读数——yuka-f1 遮罩臂 3 句把全片定成 163 ms/字(真值约 17),
182
+ 首字外推整体挪位(ocr-regions 末尾)。"""
183
+
184
+
185
+ def ocr_slope(pts, n_chars: int) -> float | None:
186
+ """这条的打字速度(µs/字),要至少两个不同时刻的中途读数,而且**字数随时间严格增加**(打字只会长):
187
+ 停在同一个字数上、或者读少了(`3, 12, 11, 11, 11` 这种,框抖 / 读半截),不是在打字,不拿来拟速度。"""
188
+ part = sorted((t, n) for t, n in pts if 0 < n < n_chars - 0.5)
189
+ if len({t for t, _ in part}) < 2 or len({n for _, n in part}) < 2: # 字数没变拟合不出速度
190
+ return None
191
+ if any(n1 <= n0 for (_, n0), (_, n1) in zip(part, part[1:])):
192
+ return None
193
+ b, _ = np.polyfit([n for _, n in part], [t for t, _ in part], 1)
194
+ return float(b) if b > 0 else None
195
+
196
+
197
+ def ocr_fit(pts, n_chars: int, t_start: int, t_full_sampled: int, span_us: int,
198
+ v: float | None) -> tuple[float | None, float | None]:
199
+ """中途读数外推 → (首字, 全字),夹进 OCR 证据给的区间。只有一个读数时借 `v`(同一段视频的典型速度)。
200
+ 多个读数但字数**不随时间严格增加**(停住 / 读少了,`ocr_slope` 同一道门)时不拿它们拟直线,按只有第一个读数处理。"""
201
+ part = sorted((t, n) for t, n in pts if 0 < n < n_chars - 0.5)
202
+ typing = all(n1 > n0 for (_, n0), (_, n1) in zip(part, part[1:]))
203
+ if len({t for t, _ in part}) >= 2 and len({n for _, n in part}) >= 2 and typing:
204
+ b, a = np.polyfit([n for _, n in part], [t for t, _ in part], 1)
205
+ first, full = a + b * 0.5, a + b * (n_chars - 0.5)
206
+ elif part and v:
207
+ t1, n1 = part[0]
208
+ first = t1 - (n1 - 0.5) * v
209
+ full = first + (n_chars - 1) * v
210
+ else:
211
+ return None, None
212
+ first = min(max(first, t_start - span_us), t_start)
213
+ lo = max([t for t, _ in part] + [first])
214
+ return first, min(max(full, lo), t_full_sampled)
215
+
216
+
217
+ def fuse(cands, lo: float, hi: float, fallback: float) -> tuple[float, str]:
218
+ """区间内的候选按 `FUSE_TOL_US` 聚类,取成员最多的一簇的中位数(平票取列表里靠前的)。
219
+ 返回 (值, "agree" 至少两路同意 / "single" 只剩一路 / "none" 一路都不剩、退回 fallback)。"""
220
+ c = [v for v in cands if v is not None and lo - FUSE_TOL_US <= v <= hi + FUSE_TOL_US]
221
+ if not c:
222
+ return min(max(fallback, lo), hi), "none"
223
+ best = None
224
+ for i, v in enumerate(c):
225
+ mem = [w for w in c if abs(w - v) <= FUSE_TOL_US]
226
+ key = (len(mem), -i)
227
+ if best is None or key > best[0]:
228
+ best = (key, mem)
229
+ mem = sorted(best[1])
230
+ mid = len(mem) // 2
231
+ val = mem[mid] if len(mem) % 2 else (mem[mid - 1] + mem[mid]) / 2
232
+ return min(max(val, lo), hi), ("agree" if len(mem) >= 2 else "single")
233
+
234
+
235
+ def snap(ts: list[int], t: float | None) -> int | None:
236
+ return None if t is None else min(ts, key=lambda x: abs(x - t))
237
+
238
+
239
+ def estimate(frames: np.ndarray, ts: list[int], inner: tuple[int, int, int, int], text: str,
240
+ t_start: int, t_full_sampled: int, span_us: int, ocr_pts, v_ocr: float | None,
241
+ v_cell: float | None = None) -> dict | None:
242
+ """`signals` + `combine` 一步到位(整段速度不需要从别的条目借时用)。"""
243
+ sig = signals(frames, ts, inner, text)
244
+ return None if sig is None else combine(sig, t_start, t_full_sampled, span_us, ocr_pts, v_ocr, v_cell)
245
+
246
+
247
+ def signals(frames: np.ndarray, ts: list[int], inner: tuple[int, int, int, int], text: str) -> dict | None:
248
+ """一条事件的像素信号(解码时算,不需要别的条目)。
249
+
250
+ `frames`:从 t_start 前一个多采样间隔到 t_full_sampled 后一个采样间隔的**连续**帧,框外扩几像素的灰度裁剪;
251
+ 最后一帧当"最终样子"的模板。`inner` = 框在裁剪里的 (x0, y0, x1, y1)。时间都是 µs。"""
252
+ final = chars(text)
253
+ N = len(final)
254
+ if N < 1 or len(frames) < 3:
255
+ return None
256
+ ts = list(ts)
257
+ k = len(frames) - 1
258
+ x0, y0, x1, y1 = inner
259
+ cell = max(1, y1 - y0) # CJK 一个字宽 ≈ 字高
260
+ # 旧判据:整框不像窗口首帧。常量图的相关系数恒为 0(C5),一律"不像"——那不是证据,不给候选。
261
+ # 整叠一次算(原来逐帧调 ncc,占 signals 的 2/3;两遍合一遍之后它跑在 OCR 旁边的线程里,2026-09-16)
262
+ first_old = None
263
+ stds = frames.std(axis=(1, 2))
264
+ if stds[0] >= 1.0:
265
+ F = frames.astype(np.float32).astype(np.float64)
266
+ A = F - F.mean(axis=(1, 2))[:, None, None]
267
+ den = np.sqrt((A * A).sum(axis=(1, 2))) * np.sqrt((A[0] * A[0]).sum())
268
+ num = (A * A[0]).sum(axis=(1, 2))
269
+ cc = np.where(den > 1e-6, num / np.where(den > 1e-6, den, 1.0), 0.0)
270
+ hit = np.nonzero((stds >= 1.0) & (cc < EMPTY_THR))[0]
271
+ first_old = ts[int(hit[0])] if hit.size else None
272
+ # 首字格"开始出现"(极性按整行判:一格里立绘占得多时,格内判会判成立绘的暗边)
273
+ head = frames[:, :, :x0 + cell]
274
+ r = rise_onset(glyph_score(head, head[k], polarity(frames[k])))
275
+ first_rise = ts[r] if r is not None else None
276
+ # 尾字格定稿(极性按这一格自己判)
277
+ tail = frames[:, :, max(0, x1 - cell):]
278
+ g = last_onset(glyph_score(tail, tail[k]) >= GLYPH_THR)
279
+ full_glyph = ts[g] if g is not None else None
280
+ return {"ts": ts, "n": N, "first_old": first_old, "first_rise": first_rise, "full_glyph": full_glyph,
281
+ "cell": cell_fit(frames, ts, x0, x1, N)}
282
+
283
+
284
+ def cell_speed(sig: dict | None) -> float | None:
285
+ """逐字格直线的斜率(µs/字);给"OCR 拟不出速度时借全片中位数"用。"""
286
+ if not sig or not sig["cell"] or sig["n"] < 2:
287
+ return None
288
+ first, full = sig["cell"]
289
+ v = (full - first) / (sig["n"] - 1)
290
+ return v if v > 0 else None
291
+
292
+
293
+ def combine(sig: dict, t_start: int, t_full_sampled: int, span_us: int, ocr_pts,
294
+ v_ocr: float | None, v_cell: float | None) -> dict:
295
+ """像素信号 + OCR 中途读数 → 融合后的首字 / 全字。
296
+
297
+ 打字速度只在**首字**外推里退到 `v_cell`:魔裁打字快,每条只有一个中途读数,OCR 拟不出速度,
298
+ 借逐字格斜率的全片中位数后首字 4–10 帧从 7/28 降到 2/28;同一个速度借给全字外推却系统性偏晚(淡入长,
299
+ 像素定稿比 OCR 读全晚),所以全字只认 OCR 自己拟出来的速度。"""
300
+ ts, N, cf = sig["ts"], sig["n"], sig["cell"]
301
+ first_old, first_rise, full_glyph = sig["first_old"], sig["first_rise"], sig["full_glyph"]
302
+ ocr_first, _ = ocr_fit(ocr_pts, N, t_start, t_full_sampled, span_us, v_ocr or v_cell)
303
+ _, ocr_full = ocr_fit(ocr_pts, N, t_start, t_full_sampled, span_us, v_ocr)
304
+
305
+ # 外推值**落到区间下界上**(速度说"比上一个采样点还早")就不投票(2026-09-24,ocr-regions):它是四个候选里唯一不看这一条像素的,
306
+ # 平票时又排第一——框挪 1~4 px、或全片中位速度被别的条目带偏,它就压在下界上赢下平票,首字早整整一个采样间隔
307
+ # (yuka-f5 裁剪臂 design 6 条早 10 帧以上、代价 230;去掉之后 31,默认产物三段真值逐项相同、gi-s2 首字 8 -> 7)。
308
+ # 不在下界上的外推值照旧排第一:它把淡入偏晚的像素测量拉回来(完全不让它投票 f1 / f5 首字代价 18 / 15 -> 36 / 36)。
309
+ # ⚠ **上界(t_start)不对称地保留**:夹在上界上的外推值按同一个道理也只是界,但让它也不投票 f5 design 首字 15 -> 18、其余不变
310
+ # (a1-mask-reuse 实验里的 fuse_tiebreak.py 的 unpin2 臂)——别按对称性顺手改
311
+ ex = snap(ts, ocr_first)
312
+ if ex is not None and ex <= t_start - span_us + FUSE_TOL_US:
313
+ ex = None
314
+ first, first_how = fuse([ex, first_rise, snap(ts, cf[0] if cf else None), first_old],
315
+ t_start - span_us, t_start, t_start)
316
+ partial = [t for t, n in ocr_pts if n < FUSE_PARTIAL * N]
317
+ full_lo = max([t_start - span_us, first] + partial)
318
+ full_cands = [full_glyph, snap(ts, cf[1] if cf else None), snap(ts, ocr_full)]
319
+ # 只防"晚很多"(owner 2026-09-12:晚 10 帧以内都还好;晚很多是灾难——译文迟迟打不完——
320
+ # 早很多无非一开始就全显示)。两道上限:OCR 读全之后 LATE_CAP_US 以内;没有两路同意时,
321
+ # 不晚于区间内最早的候选 + LATE_CAP_US。10 帧以内的判断一律不动
322
+ full, full_how = fuse(full_cands, full_lo, min(t_full_sampled + span_us, t_full_sampled + LATE_CAP_US),
323
+ t_full_sampled)
324
+ if full_how == "single":
325
+ inside = [v for v in full_cands if v is not None
326
+ and full_lo - FUSE_TOL_US <= v <= t_full_sampled + span_us + FUSE_TOL_US]
327
+ if inside and full > min(inside) + LATE_CAP_US:
328
+ full, full_how = max(full_lo, min(inside) + LATE_CAP_US), "capped"
329
+ # 吸附之后首字仍**不晚于 t_start**(那一刻已经读到字了,是实测的 pts):`ts` 是按锚点折算的估计,锚点的 pts 带容器时间基的舍入
330
+ # (yuka 是毫秒;时间网格上的采样帧不在整毫秒处时差到 ±0.5 ms),吸附会越过 t_start 零点几毫秒、产物校验拒收(2026-09-24,--fps 2.2)
331
+ first_s = min(snap(ts, first), t_start)
332
+ return {"first": first_s, "full": max(snap(ts, full), first_s),
333
+ "first_how": first_how, "full_how": full_how}
334
+
335
+
336
+ # ---------- 结尾:完全显示结束 / 完全消失(2026-09-12,淡出 / 交叉淡出) ----------
337
+ #
338
+ # 和开头的"首字 / 全字"平行:没有淡出时两个时刻重合(一帧切掉),有淡出时中间就是淡出时长。
339
+ # 整框相关系数对"整体均匀变淡"不敏感(亮度归一化掉了),原神的 3–4 帧淡出上旧终点晚 28 帧;
340
+ # 这里逐字格量**字形对比度**——(字形像素 − 本帧背景) / (模板里的字形 − 模板背景),背景亮度随帧变也不怕。
341
+
342
+ FADE_HI = 0.85
343
+ """对比度 ≥ 这个算"完全清楚"。"""
344
+ FADE_LO = 0.15
345
+ """对比度 < 这个算"看不见"。"""
346
+
347
+
348
+ def contrast_curves(frames: np.ndarray, inner: tuple[int, int, int, int], n_chars: int) -> list[np.ndarray]:
349
+ """逐字格的对比度曲线(模板 = 窗口首帧,要求那时字还完整)。字形像素或背景像素太少的格不给。"""
350
+ x0, y0, x1, y1 = inner
351
+ tpl = frames[0]
352
+ bright = polarity(tpl)
353
+ cw = (x1 - x0) / max(1, n_chars)
354
+ out = []
355
+ for c in range(n_chars):
356
+ a, b = int(round(x0 + c * cw)), int(round(x0 + (c + 1) * cw))
357
+ if b - a < 2:
358
+ continue
359
+ t = tpl[y0:y1, a:b].astype(np.float32)
360
+ dev = t - np.median(t)
361
+ m = dev >= GLYPH_DIFF if bright else dev <= -GLYPH_DIFF
362
+ if m.sum() < CELL_MIN_PX or (~m).sum() < CELL_MIN_PX:
363
+ continue
364
+ denom = t[m] - np.median(t[~m])
365
+ seg = frames[:, y0:y1, a:b].astype(np.float32)
366
+ bg = np.median(seg[:, ~m], axis=1)
367
+ out.append(np.clip(np.median((seg[:, m] - bg[:, None]) / denom[None, :], axis=1), -0.5, 1.5))
368
+ return out
369
+
370
+
371
+ def end_keyframes(frames: np.ndarray, ts: list[int], inner: tuple[int, int, int, int],
372
+ text: str) -> tuple[int | None, int | None]:
373
+ """(完全显示结束, 完全消失),µs。
374
+
375
+ 完全显示结束 = 从窗口开头起,10% 分位字格仍 ≥ FADE_HI 的最后一帧(窗口首帧就不完整 = 给不出);
376
+ 完全消失 = 之后第一帧 75% 分位字格 < FADE_LO——**不要求持续**:同位置很快会出下一句。
377
+ 分位数而不是全体:个别格被立绘 / 描边污染不牵连整行。消失侧原来取 90% 分位,星铁一条字消失后
378
+ 有一两格被背景亮纹撑在 0.15–0.20,整行晚判 17 帧(2026-09-12);75% 容得下四分之一的格被污染。"""
379
+ n = len(chars(text))
380
+ if n < 1 or len(frames) < 3:
381
+ return None, None
382
+ cs = contrast_curves(frames, inner, n)
383
+ if len(cs) < 2:
384
+ return None, None
385
+ C = np.stack(cs)
386
+ q_lo, q_hi = np.quantile(C, 0.1, axis=0), np.quantile(C, 0.75, axis=0)
387
+ a = None
388
+ if q_lo[0] >= FADE_HI:
389
+ a = 0
390
+ while a + 1 < len(ts) and q_lo[a + 1] >= FADE_HI:
391
+ a += 1
392
+ start = a + 1 if a is not None else 0
393
+ b = next((i for i in range(start, len(ts)) if q_hi[i] < FADE_LO), None)
394
+ return (ts[a] if a is not None else None), (ts[b] if b is not None else None)
flowocr/models.py ADDED
@@ -0,0 +1,263 @@
1
+ """ORT 路要的 ONNX 模型:**全用官方发布的**(owner 2026-09-22),钉版本、验 sha256、缺了就取。
2
+
3
+ PaddlePaddle 自己在 Hugging Face 上发了 PP-OCRv6 的 ONNX(Apache-2.0,`LICENSES/PP-OCRv6-NOTICE.md`)。
4
+ 以前用的是我们自己拿 paddle2onnx 转的那几份(`tools/export_*_onnx.sh`,落在 onnxrt 实验目录的 `models/` 下,
5
+ 按约定随时可删、发行包里也没有);换成官方的之后,源码安装和发行包取的是同一份、谁都能从上游核对。
6
+ owner 同日定:**旧模型产的数据和新模型的区别按噪音级算**(`ocr_args.NOISE_EQUIV` 里有 `det_onnx` / `rec_onnx`),不重量。
7
+
8
+ **我们只动一处**:rec 的图末尾加 `ArgMax` + `ReduceMax` 两个节点(CTC 贪心解码只要这两样),**权重一个字节不动**。
9
+ 不融的话 worker 每批要把 `[batch, T, 18710]` 的 fp32 logits 整个拷回主机(批 32 约 191 MB,recort 文件头),
10
+ 融了只回两个 `[batch, T]` 小张量——融合是 ORT 那条路快的主要原因(inference-runtime 第 2 条)。
11
+ **不再转半精度**:官方只发 fp32;端到端 fp32 + 融合 53.7 s 对 fp16 + 融合 54.9 s(yuka-f5 10 分钟,同一处),
12
+ 半精度没买到墙钟,还在 ORT 1.26 / 1.27 的 CPU EP 上打不开(project-structure 记录)。
13
+
14
+ 放在哪:`paths.models_root()`(`FLOWOCR_MODELS` > 数据根的 `models/`),每个仓库一个子目录。
15
+ 下载走 `HF_ENDPOINT`(默认 `https://huggingface.co`,国内可设镜像)。先写 `.part` 再改名,中途断了不留半个文件。
16
+
17
+ python -m flowocr.models fetch # 默认要的(rec + det,默认都走 ORT)
18
+ python -m flowocr.models fetch rec # 只取点名的
19
+ python -m flowocr.models list
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import argparse
24
+ import hashlib
25
+ import os
26
+ import sys
27
+ import urllib.request
28
+ import zipfile
29
+ from dataclasses import dataclass, field
30
+ from pathlib import Path
31
+
32
+ from flowocr import paths
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class Model:
37
+ repo: str
38
+ rev: str
39
+ """Hugging Face 上的**提交号**(不是分支名):`resolve/<rev>/` 下的内容不可变。"""
40
+ files: dict[str, str] = field(default_factory=dict)
41
+ """文件名 -> sha256。下载完逐个核,对不上就删掉、报错。"""
42
+ fuse_argmax: bool = False
43
+
44
+ @property
45
+ def dir(self) -> Path:
46
+ return paths.models_root() / self.repo.split("/")[-1]
47
+
48
+ @property
49
+ def onnx(self) -> Path:
50
+ """管线真正加载的那个文件。"""
51
+ return self.dir / ("inference_argmax.onnx" if self.fuse_argmax else "inference.onnx")
52
+
53
+
54
+ MODELS = {
55
+ "rec": Model("PaddlePaddle/PP-OCRv6_medium_rec_onnx", "50c7eacafc52fa7bcf4194e8cd08e46f8558504b", {
56
+ "inference.onnx": "9c09abf0957f7968c7586464b7397b84ad2387a0497a351af40e9acc71b673ba",
57
+ "inference.yml": "991b700facf5b50a7de193468207d5f4255b538dde0d312ae3b7c7a9b6873129",
58
+ }, fuse_argmax=True),
59
+ "det": Model("PaddlePaddle/PP-OCRv6_medium_det_onnx", "61323801669c338b7891481ec7bac61ce31b576a", {
60
+ "inference.onnx": "eb13b44b25bb36f89528b68720af8a61d9cf381176107f465db1757b65d086e1",
61
+ "inference.yml": "7298d5ead546584af2504d03355f881ac7a7bc0eb1e282d3e159277c1d0af871",
62
+ }),
63
+ }
64
+ """rec 的字表就用它自己仓库里的 `inference.yml`(`recort` 按模型所在目录找),模型和字表出自同一个提交。
65
+ 2026-09-22 核过:它和 PaddleX 模型目录(`official_models/PP-OCRv6_medium_rec`)里那份 `inference.yml` **逐字节相同**(18,710 类)。
66
+ rec 的 `inference.onnx` 的 sha256 和 Hugging Face 上 LFS 记的 oid 一致;其余三个是取下来时算的(提交号钉住,内容本就不可变)。"""
67
+
68
+ DEFAULT = ("rec", "det")
69
+ """默认配置要的:rec 和 det(2026-09-24 起 det 也走 ORT,D1)。原来只列 rec,翻默认时没跟着改,
70
+ 断网的机器上 `python -m flowocr.models fetch` 之后默认配置仍缺 det 的 ONNX(2026-09-24 Codex 审计)。"""
71
+
72
+
73
+ def sha256(p: Path) -> str:
74
+ h = hashlib.sha256()
75
+ with open(p, "rb") as f:
76
+ for chunk in iter(lambda: f.read(1 << 20), b""):
77
+ h.update(chunk)
78
+ return h.hexdigest()
79
+
80
+
81
+ def _download(url: str, dst: Path, want_sha: str) -> None:
82
+ tmp = dst.with_name(dst.name + ".part")
83
+ req = urllib.request.Request(url, headers={"User-Agent": "flowocr"})
84
+ print(f"[模型] 下载 {url}", flush=True)
85
+ h = hashlib.sha256()
86
+ with urllib.request.urlopen(req, timeout=60) as r, open(tmp, "wb") as f:
87
+ for chunk in iter(lambda: r.read(1 << 20), b""):
88
+ f.write(chunk)
89
+ h.update(chunk)
90
+ got = h.hexdigest()
91
+ if want_sha and got != want_sha:
92
+ tmp.unlink(missing_ok=True)
93
+ raise RuntimeError(f"{url} 的 sha256 是 {got},钉的是 {want_sha}——没装,别往下跑")
94
+ os.replace(tmp, dst)
95
+
96
+
97
+ def fuse_argmax(src: Path, dst: Path) -> None:
98
+ """rec 图末尾加 `ArgMax` / `ReduceMax`:输出从 `[batch, T, C]` 的 logits 变成 `cls`(int64)+ `prob`(float),形状 `[batch, T]`。
99
+ 只加节点、改输出声明,**不碰权重和已有节点**。和原来 一次性探针 `export_rec_onnx.sh` 里那段是同一个做法。"""
100
+ import onnx
101
+ from onnx import TensorProto, helper
102
+ m = onnx.load(str(src))
103
+ g = m.graph
104
+ y = g.output[0]
105
+ g.node.append(helper.make_node("ArgMax", [y.name], ["cls"], axis=-1, keepdims=0))
106
+ g.node.append(helper.make_node("ReduceMax", [y.name], ["prob"], axes=[-1], keepdims=0))
107
+ dims = [d.dim_param or d.dim_value for d in y.type.tensor_type.shape.dim][:2]
108
+ del g.output[:]
109
+ g.output.extend([helper.make_tensor_value_info("cls", TensorProto.INT64, dims),
110
+ helper.make_tensor_value_info("prob", TensorProto.FLOAT, dims)])
111
+ tmp = dst.with_name(dst.name + ".part")
112
+ onnx.save(m, str(tmp))
113
+ os.replace(tmp, dst)
114
+
115
+
116
+ def _src_mark(m: Model) -> Path:
117
+ """派生文件旁边记"它是从哪份源文件派生的"(源文件的 sha256)。"""
118
+ return m.onnx.with_name(m.onnx.name + ".src-sha256")
119
+
120
+
121
+ def derived_stale(m: Model) -> bool:
122
+ """派生的 argmax 图要不要重做:不在、没有来源记录、或来源不是现在钉的那份源文件。
123
+ ⚠ 只看"派生文件在不在"是错的(Codex 复审):表里换了提交号 / 源文件重新下载之后,旧的派生图照样被拿去用。"""
124
+ if not m.fuse_argmax:
125
+ return False
126
+ mark = _src_mark(m)
127
+ return not (m.onnx.is_file() and mark.is_file()
128
+ and mark.read_text(encoding="utf-8").strip() == m.files["inference.onnx"])
129
+
130
+
131
+ def ensure(m: Model) -> Path:
132
+ """一个模型就位:源文件缺了 / sha 对不上就(重新)下载,派生图过期就重做。返回管线要加载的那个文件。"""
133
+ m.dir.mkdir(parents=True, exist_ok=True)
134
+ base = os.environ.get("HF_ENDPOINT", "https://huggingface.co").rstrip("/")
135
+ for fn, want in m.files.items():
136
+ p = m.dir / fn
137
+ if p.is_file() and (not want or sha256(p) == want):
138
+ continue
139
+ _download(f"{base}/{m.repo}/resolve/{m.rev}/{fn}", p, want)
140
+ if derived_stale(m):
141
+ fuse_argmax(m.dir / "inference.onnx", m.onnx)
142
+ _src_mark(m).write_text(m.files["inference.onnx"] + "\n", encoding="utf-8")
143
+ print(f"[模型] {m.onnx.name}:融进 argmax({m.onnx.stat().st_size / 2**20:.1f} MB)", flush=True)
144
+ return m.onnx
145
+
146
+
147
+ def fetch(name: str) -> Path:
148
+ """按名字取一个模型(`ensure(MODELS[name])`)。"""
149
+ return ensure(MODELS[name])
150
+
151
+
152
+ _ready: dict[str, str] = {}
153
+ """本进程里核过的(每个进程核一次源文件 sha256,76 MB 约 0.15 s;run_ocr2 一趟会问好几次)。"""
154
+
155
+
156
+ def path(name: str) -> str:
157
+ """管线用的入口:默认模型的路径,**缺了 / 过期了就当场取**(同 PaddleX 首次运行下它自己的模型)。
158
+ 取不到(断网、sha 对不上)就抛,报错里带手动取的命令。"""
159
+ if name in _ready:
160
+ return _ready[name]
161
+ m = MODELS[name]
162
+ try:
163
+ _ready[name] = str(ensure(m))
164
+ except Exception as exc:
165
+ raise RuntimeError(f"默认 {name} 模型不在(或核不上){m.onnx},自动下载失败:{exc}。"
166
+ f"联网后跑 `flowocr-models fetch {name}`(镜像用 HF_ENDPOINT),或用 GitHub Release 上的模型包 `flowocr-models install`") from exc
167
+ return _ready[name]
168
+
169
+
170
+ def meta(p: str) -> dict:
171
+ """写进 obs `_meta.models` 的那一条:文件名 + sha256——复用判据只比路径字符串,**同名文件换了内容它看不出来**,这里记下内容。"""
172
+ q = Path(p)
173
+ return {"file": q.name, "sha256": sha256(q) if q.is_file() else None}
174
+
175
+
176
+ def pack(dst: Path) -> list[str]:
177
+ """离线包(发版时由维护者打,挂在 GitHub Release):默认模型的**源文件**(不含派生的 argmax 图,`install` 之后首次用时就地融)
178
+ + `LICENSES/` 里的许可与改动说明。只打核过 sha256 的文件;源码 checkout 里才有 `LICENSES/`。"""
179
+ lic = paths.CODE_ROOT / "LICENSES"
180
+ if not lic.is_dir():
181
+ raise SystemExit(f"找不到 {lic}:离线包要在源码 checkout 里打(要带上模型的许可说明)")
182
+ names = []
183
+ with zipfile.ZipFile(dst, "w", zipfile.ZIP_STORED) as z:
184
+ for n in DEFAULT:
185
+ m = MODELS[n]
186
+ for fn, want in m.files.items():
187
+ src = m.dir / fn
188
+ if not src.is_file() or sha256(src) != want:
189
+ raise SystemExit(f"{src} 不在或 sha256 对不上:先 `flowocr-models fetch {n}`")
190
+ z.write(src, f"{m.dir.name}/{fn}")
191
+ names.append(f"{m.dir.name}/{fn}")
192
+ for f in sorted(lic.iterdir()):
193
+ z.write(f, f"LICENSES/{f.name}")
194
+ return names
195
+
196
+
197
+ def install(src: Path) -> list[str]:
198
+ """把 `pack` 打的离线包装进模型根:只认表里的模型文件、逐个核 sha256,对不上就不写;`LICENSES/<文件名>` 一并放进模型根。
199
+ 名字之外的一律拒绝(`LICENSES/../x` 这类穿出模型根的也算);写文件先写 `.part` 再改名,中途断了不留半个文件。"""
200
+ known = {f"{m.dir.name}/{fn}": (m, fn, want) for m in MODELS.values() for fn, want in m.files.items()}
201
+ root = paths.models_root()
202
+ done = []
203
+ with zipfile.ZipFile(src) as z:
204
+ for info in z.infolist():
205
+ if info.is_dir():
206
+ continue
207
+ name = info.filename
208
+ data = z.read(info)
209
+ if name in known:
210
+ m, fn, want = known[name]
211
+ if hashlib.sha256(data).hexdigest() != want:
212
+ raise SystemExit(f"{name} 的 sha256 对不上:包坏了,重新下一份")
213
+ dst = m.dir / fn
214
+ else:
215
+ head, sep, leaf = name.partition("/")
216
+ if not (head == "LICENSES" and sep and leaf and leaf not in (".", "..")
217
+ and not any(c in leaf for c in '/\\:')):
218
+ raise SystemExit(f"包里有不认识的文件 {name}:不是 `flowocr-models pack` 打的包")
219
+ dst = root / "LICENSES" / leaf
220
+ dst.parent.mkdir(parents=True, exist_ok=True)
221
+ part = dst.with_name(dst.name + ".part")
222
+ part.write_bytes(data)
223
+ os.replace(part, dst)
224
+ done.append(str(dst))
225
+ return done
226
+
227
+
228
+ def main(argv: list[str] | None = None) -> int:
229
+ ap = argparse.ArgumentParser(prog="flowocr-models", description="取 / 列官方 ONNX 模型")
230
+ sub = ap.add_subparsers(dest="cmd", required=True)
231
+ f = sub.add_parser("fetch", help="下载并核对(已在就跳过)")
232
+ f.add_argument("names", nargs="*", help=f"{' / '.join(MODELS)},默认 {' '.join(DEFAULT)}")
233
+ sub.add_parser("list", help="列出模型根里有什么")
234
+ sub.add_parser("install", help="装离线包(GitHub Release 上的模型 zip):核过 sha256 再放进模型根").add_argument("zip")
235
+ sub.add_parser("pack", help="打离线包(维护者发版用)").add_argument("zip")
236
+ a = ap.parse_args(argv)
237
+ if a.cmd == "install":
238
+ for p in install(Path(a.zip)):
239
+ print(p)
240
+ print(f"装好了:{paths.models_root()}(第一次跑会就地融 argmax 图,不联网)")
241
+ return 0
242
+ if a.cmd == "pack":
243
+ print("\n".join(pack(Path(a.zip))))
244
+ print(f"-> {a.zip}({Path(a.zip).stat().st_size / 2**20:.1f} MB)")
245
+ return 0
246
+ if a.cmd == "fetch":
247
+ bad = [n for n in a.names if n not in MODELS]
248
+ if bad:
249
+ ap.error(f"不认识的模型 {bad}(有 {list(MODELS)})")
250
+ for n in a.names or DEFAULT:
251
+ print(f"{n}: {fetch(n)}")
252
+ return 0
253
+ print(f"模型根:{paths.models_root()}")
254
+ for n, m in MODELS.items():
255
+ src_ok = all((m.dir / fn).is_file() for fn in m.files)
256
+ state = ("在" if m.onnx.is_file() and not derived_stale(m) else
257
+ "源文件在,argmax 图第一次用时就地融(不联网)" if src_ok else "缺")
258
+ print(f" {n}: {m.repo}@{m.rev[:8]} -> {m.onnx}({state})")
259
+ return 0
260
+
261
+
262
+ if __name__ == "__main__":
263
+ sys.exit(main())
@@ -0,0 +1,3 @@
1
+ """阶段 3:输出。从阶段 2 产物投影出 SRT(`export`)和可编辑的字幕稿(`script`,特效另由阶段 4 `flowocr.typeset` 生成),
2
+ 入口是 `render` 和 `presets/` 下的内置预设;两个阶段共用的排版在 `layout`。
3
+ """