flowocr 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. flowocr/__init__.py +15 -0
  2. flowocr/analyze/__init__.py +8 -0
  3. flowocr/analyze/align.py +151 -0
  4. flowocr/analyze/build_tracks.py +2886 -0
  5. flowocr/analyze/cluster_layers.py +155 -0
  6. flowocr/analyze/game_align.py +652 -0
  7. flowocr/analyze/gamescript.py +1220 -0
  8. flowocr/analyze/gtdbundle.py +306 -0
  9. flowocr/analyze/match.py +67 -0
  10. flowocr/analyze/matchers/__init__.py +5 -0
  11. flowocr/analyze/matchers/gametext.py +70 -0
  12. flowocr/analyze/merge_nameplate.py +178 -0
  13. flowocr/analyze/models/slot_pair.json +148 -0
  14. flowocr/analyze/nameplate.py +148 -0
  15. flowocr/analyze/pair_features.py +168 -0
  16. flowocr/analyze/pair_model.py +127 -0
  17. flowocr/analyze/refine_boundaries.py +409 -0
  18. flowocr/analyze/script_align.py +641 -0
  19. flowocr/analyze/scriptmatch.py +1018 -0
  20. flowocr/analyze/slot_learned.py +306 -0
  21. flowocr/analyze/slot_lines.py +251 -0
  22. flowocr/analyze/slot_modes.py +143 -0
  23. flowocr/analyze/slot_pairs.py +562 -0
  24. flowocr/analyze/slot_veto.py +104 -0
  25. flowocr/analyze/uigate.py +351 -0
  26. flowocr/artifacts/__init__.py +5 -0
  27. flowocr/artifacts/evalkit.py +123 -0
  28. flowocr/artifacts/matchedio.py +75 -0
  29. flowocr/artifacts/srtio.py +166 -0
  30. flowocr/artifacts/tracksio.py +345 -0
  31. flowocr/extensions.py +77 -0
  32. flowocr/extract/__init__.py +8 -0
  33. flowocr/extract/childproc.py +118 -0
  34. flowocr/extract/decode_proc.py +511 -0
  35. flowocr/extract/decode_shards.py +574 -0
  36. flowocr/extract/detpost.py +215 -0
  37. flowocr/extract/edge_proc.py +314 -0
  38. flowocr/extract/edge_refine.py +598 -0
  39. flowocr/extract/fast_det.py +214 -0
  40. flowocr/extract/ffcheck.py +87 -0
  41. flowocr/extract/framegrid.py +265 -0
  42. flowocr/extract/framesource.py +1264 -0
  43. flowocr/extract/ocr_args.py +754 -0
  44. flowocr/extract/ocr_complete.py +231 -0
  45. flowocr/extract/ocr_parallel.py +208 -0
  46. flowocr/extract/ort_server.py +632 -0
  47. flowocr/extract/ortclient.py +363 -0
  48. flowocr/extract/ptsclock.py +150 -0
  49. flowocr/extract/recdecode.py +78 -0
  50. flowocr/extract/recort.py +162 -0
  51. flowocr/extract/recpack.py +94 -0
  52. flowocr/extract/recpool.py +206 -0
  53. flowocr/extract/recprep.py +71 -0
  54. flowocr/extract/refine_video.py +271 -0
  55. flowocr/extract/regions.py +387 -0
  56. flowocr/extract/reuse_v2.py +593 -0
  57. flowocr/extract/run_groups.py +247 -0
  58. flowocr/extract/run_ocr2.py +1605 -0
  59. flowocr/extract/supervisor.py +261 -0
  60. flowocr/extract/timeline.py +84 -0
  61. flowocr/extract/typewriter_fuse.py +394 -0
  62. flowocr/models.py +263 -0
  63. flowocr/output/__init__.py +3 -0
  64. flowocr/output/export.py +205 -0
  65. flowocr/output/layout.py +210 -0
  66. flowocr/output/presets/__init__.py +6 -0
  67. flowocr/output/presets/_overlay.py +40 -0
  68. flowocr/output/presets/default.py +21 -0
  69. flowocr/output/presets/default_all.py +20 -0
  70. flowocr/output/presets/dev.py +21 -0
  71. flowocr/output/presets/matched_srt.py +32 -0
  72. flowocr/output/presets/script.py +20 -0
  73. flowocr/output/presets/srt_main.py +33 -0
  74. flowocr/output/render.py +82 -0
  75. flowocr/output/run_srt.py +83 -0
  76. flowocr/output/script.py +541 -0
  77. flowocr/paths.py +168 -0
  78. flowocr/provenance.py +119 -0
  79. flowocr/typeset/__init__.py +9 -0
  80. flowocr/typeset/__main__.py +38 -0
  81. flowocr/typeset/assfile.py +216 -0
  82. flowocr/typeset/core.py +821 -0
  83. flowocr/typeset/fx/__init__.py +11 -0
  84. flowocr-0.1.0.dist-info/METADATA +109 -0
  85. flowocr-0.1.0.dist-info/RECORD +91 -0
  86. flowocr-0.1.0.dist-info/WHEEL +5 -0
  87. flowocr-0.1.0.dist-info/entry_points.txt +10 -0
  88. flowocr-0.1.0.dist-info/licenses/LICENSE +674 -0
  89. flowocr-0.1.0.dist-info/licenses/LICENSES/Apache-2.0.txt +201 -0
  90. flowocr-0.1.0.dist-info/licenses/LICENSES/PP-OCRv6-NOTICE.md +18 -0
  91. flowocr-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,541 @@
1
+ """阶段 3:从阶段 2 产物投影出**字幕稿**(一份能在 Aegisub / mpv 里直接预览、编辑的 ASS),特效另由阶段 4(`flowocr.typeset`)生成。
2
+
3
+ 方案:script-fx 计划(owner 2026-09-27)。这里**只投影**,不重新判断(artifacts.md 第 1 条):
4
+
5
+ * 对齐方式是阶段 2 判好的 `regions[].align`;matched 各层盖哪些块是 `scriptmatch.overlay_items` 判好的;
6
+ * tracks 取哪些事件、哪些是常驻 UI,按**轨**取(`tracksio.cue_lines`,UI 的权威清单 `ui_lines` 是逐轨的);
7
+ * 效果(打字机 / 淡入淡出)的判定在这里做一次,写进字幕稿(淡入淡出是 `\\fad`,打字机是 Effect 里 `fo` 的 `tw`),阶段 4 不回读阶段 2。
8
+
9
+ 条目(`Item`)= 一组同起同止的行。来源两个:
10
+
11
+ * `tracks_items`:**一个事件一条**,每个事件只画一次。不按 cue 出:默认的 `--srt-mode segment` 把一条常驻行
12
+ 切进好几条 2 秒的 cue,按 cue 出会画好几遍(底板叠成不透明、打字机从头放好几次);
13
+ * `project_items`:一个匹配条目一条,译文按原文行数重分(`overlay_lines`)。
14
+
15
+ 一条条目写成字幕稿里的一行(源行):Style 按角色、Actor 是区域号、Effect 是阶段 4 要的其余参数 `fo:…`
16
+ (原文框、打字机时长、轨迹、`was`…,格式见 artifacts.md 的字幕稿一节);正文是一个 tag 块(`\\an` `\\pos` `\\fs` `\\fad`)+ 纯文字。
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import hashlib
21
+ import json
22
+ from collections import Counter
23
+ from dataclasses import dataclass, field, replace
24
+ from pathlib import Path
25
+
26
+ from flowocr import provenance
27
+ from flowocr.artifacts import matchedio, tracksio
28
+ from flowocr.artifacts.matchedio import LAYERS
29
+ from flowocr.output import export, layout
30
+ from flowocr.typeset import assfile
31
+
32
+ FORMAT_VERSION = "1"
33
+ """字幕稿格式的版本,写在 `[Script Info]` 的 `FlowOCR Script` 键里。"""
34
+
35
+ STYLES = ("Body", "Extra", "Name", "Choice", "Offmain", "Panel", "UI")
36
+ """字幕稿的样式 = 角色(Style 表示"这是什么",不表示"在哪")。`Offmain` 是 matched 的"主轨外"那一层;
37
+ 条目在不在主轨另由 `fo` 参数 `off` 记(dev 按它改字色)。"""
38
+
39
+ PREVIEW_BOX = "&H20101010"
40
+ """预览底框的颜色(`BorderStyle=3` 用 OutlineColour 画框):近黑、α 0x20。只供预览,阶段 4 的生成行不用它。"""
41
+
42
+ TYPEWRITER_MODES = ("off", "on", "dim")
43
+ """打字机效果:`on` 检出了就逐字显出;`dim` 全字出现之前整句半透明(审计时刻用,阶段 4 的选项);`off` 不做。"""
44
+
45
+ TW_MIN_FRAMES = 2
46
+ """首字出现到全字出现不到这么多帧的不算打字机(回抠用原生帧、采样级用采样间隔):
47
+ 即时出现的字回抠也会给一个晚一两帧的全字时刻;采样级的全字判据会被"后面某帧多读两个字"骗出一个采样间隔。"""
48
+
49
+ FADE_IN_MAX_US = 300_000
50
+ """复刻的淡入最长这么多(owner 2026-09-25:拿不准的时候,字能及时看清更重要)。回抠量到的"首字 → 全字"
51
+ 不一定是淡入:整段判不清时它常常是打字过程(gi-s2 回抠产物上 0.4–1.2 s),画成那么长的淡入,字会一直发虚。
52
+ 主轨上量到的真淡出多数 ≤ 2 帧、最长约 15 帧(text-effects 报告),这个上限盖得住。
53
+ ⚠ owner 2026-09-25:有点绝对,待观察(followups 清单)。"""
54
+
55
+ INSTANT_MAX_SPREAD = 1.0
56
+ """回抠的整段判定 `instant` 有两种来源(`refine_boundaries`):每字生长的中位时长不到一帧(字确实没有生长),
57
+ 或每字耗时的离散度 `iqr_over_median` 超过 1.0(判不清)。只有前一种才整段关掉逐字显出;
58
+ 离散度超过这个数的按逐条的帧数门走——拿不准时跟着原文打字的节奏,不压成一段长淡入。"""
59
+
60
+
61
+ @dataclass
62
+ class Item:
63
+ """一组同起同止的行。`events` 是效果(打字机 / 淡入淡出)取时刻用的正文事件,只有 `line` 才带;
64
+ `boxes` 是在动的事件的轨迹;`translated`:叠的是剧本译文(matched),不是这个框里原有的那段字(tracks)——
65
+ 译文的行是重分过的,锚轴取整组的(`layout.anchor_axis`),字号有保底(`layout.OVERLAY_MIN_SCALE`);
66
+ 原文压回自己的框,锚轴取这一行自己的框,字号只按框定。"""
67
+ start_us: int
68
+ end_us: int
69
+ rows: list[list[float]]
70
+ lines: list[str]
71
+ kind: str # line / name / choice / offmain / block
72
+ events: list[dict] = field(default_factory=list)
73
+ region: int = -1
74
+ main: bool = True
75
+ ui: bool = False
76
+ align: str = "center"
77
+ translated: bool = False
78
+ boxes: list | None = None
79
+ cue: str = ""
80
+ row_h: int = 0 # block:原文行高(字号从它换算)
81
+ layer: str = "" # matched 的层(body / extra / …);tracks 的条目是空
82
+ key: str | None = None # matched:剧本行的键(跨阶段 2 重跑仍然稳定)
83
+
84
+
85
+ def overlay_lines(text: str, rows: int) -> list[str]:
86
+ """译文分成**恰好** `rows` 行(和原字幕逐行对位,底板才能逐行画):剧本自带换行数正好就照它,
87
+ 否则把整句按字数均分。`rows ≤ 1` 时剧本换行照留。"""
88
+ ls = [x for x in text.replace("\\N", "\n").split("\n") if x.strip()] or [text]
89
+ if rows <= 1 or len(ls) == rows:
90
+ return ls
91
+ s = "".join(ls)
92
+ k = -(-len(s) // rows)
93
+ return [s[i:i + k] for i in range(0, len(s), k)]
94
+
95
+
96
+ def main_regions(doc: dict) -> set[int]:
97
+ main = tracksio.main_track(doc)
98
+ if not main:
99
+ return set()
100
+ return set(main.get("regions") or ([main["region"]] if "region" in main else []))
101
+
102
+
103
+ def region_align(doc: dict) -> dict[int, str]:
104
+ return {r["index"]: r.get("align", "center") for r in doc.get("regions", [])}
105
+
106
+
107
+ def need_align(doc: dict, where: str) -> None:
108
+ """叠加要阶段 2 判好的对齐方式;旧产物没有就当场报错,不静默退回居中。"""
109
+ miss = [r["index"] for r in doc["regions"] if "align" not in r]
110
+ if miss:
111
+ raise ValueError(f"{where} 的区域没有对齐判定(regions[].align,缺 {len(miss)} 个)——"
112
+ "这份产物早于对齐判据,重跑 build_tracks(只是阶段 2,不重跑 OCR)")
113
+
114
+
115
+ def tracks_items(doc: dict, keep: str) -> list[Item]:
116
+ """tracks → 条目:**一个事件一条、只画一次**,按轨挑事件(UI 的作用域跟着轨走)。
117
+
118
+ * `keep="main"`:主轨的 cue 过一遍 `cue_lines`(和 `srt_main` 同一个行集)+ 名牌轨的事件;
119
+ * `keep="all"`:全部区域轨,UI 事件也取(`drop_filtered=False`),按**那条区域轨自己的** `ui_lines`
120
+ 和全局的 `ui_footprint` 标成 UI;振り仮名(`ruby` 标)也取、按 UI 标(`dev` 画成 UI 的样子,查注音判据靠它)。
121
+ 主轨和名牌轨都是区域轨的另一种投影,不再走。
122
+ 都按事件 id 去重(跨轨、跨 cue)。"""
123
+ picked: dict[int, bool] = {}
124
+ cue_of: dict[int, str] = {}
125
+ if keep == "main":
126
+ main = tracksio.main_track(doc)
127
+ if main is None:
128
+ raise ValueError("这份产物没有主轨(provenance.main_track 为空),`default` 没有可叠的")
129
+ for tr in [main, *(t for t in doc["tracks"] if t["kind"] == "nameplate")]:
130
+ for cue in tr["cues"]:
131
+ for e in tracksio.cue_lines(doc, tr, cue):
132
+ picked.setdefault(e["id"], False)
133
+ cue_of.setdefault(e["id"], cue["id"])
134
+ else:
135
+ for tr in doc["tracks"]:
136
+ if tr["kind"] != "region":
137
+ continue
138
+ ui = set(tr.get("ui_lines") or ())
139
+ for cue in tr["cues"]:
140
+ for e in tracksio.cue_lines(doc, tr, cue, drop_filtered=False):
141
+ flagged = (e["text"].strip() in ui
142
+ or bool({"ui_footprint", "ruby"} & set(e.get("flags") or ())))
143
+ picked[e["id"]] = picked.get(e["id"], False) or flagged
144
+ cue_of.setdefault(e["id"], cue["id"])
145
+ mains, aligns = main_regions(doc), region_align(doc)
146
+ out = []
147
+ for eid in sorted(picked):
148
+ e = doc["events"][eid]
149
+ moving = len(e.get("boxes") or []) >= 2
150
+ out.append(Item(e["t_start"], e["t_end"], [list(e["box"])], [e["text"]],
151
+ "name" if "nameplate" in (e.get("flags") or []) else "line",
152
+ events=[e], region=e["region"], main=e["region"] in mains if mains else True,
153
+ ui=picked[eid], align="center" if moving else aligns.get(e["region"], "center"),
154
+ boxes=e["boxes"] if moving else None, cue=cue_of[eid]))
155
+ return out
156
+
157
+
158
+ def project_items(items: list[dict], doc: dict, lang: str, layers, used: set[int] | None = None) -> list[Item]:
159
+ """matched → 条目:**投影**,按层和语言把 `overlay_items` 判好的条目摆成 `Item`,不判断任何东西。
160
+ 这一层没有 `lang` 的译文就不叠(画面原样)。区域取条目第一个事件的;对齐取那个区域的,多行用公共轴。
161
+ `used` 给了就把真画出来的条目用到的事件 id 记进去(`unmatched_items` 用)。"""
162
+ mains, aligns = main_regions(doc), region_align(doc)
163
+ out = []
164
+ for it in items:
165
+ item = _project_one(it, doc, lang, layers, mains, aligns)
166
+ if item is None:
167
+ continue
168
+ out.append(item)
169
+ if used is not None:
170
+ used.update(it.get("events") or ())
171
+ return out
172
+
173
+
174
+ def _project_one(it: dict, doc: dict, lang: str, layers, mains, aligns) -> Item | None:
175
+ """一个 matched 条目 → `Item`;不在要的层里、或没有 `lang` 的译文就是 None(不叠)。"""
176
+ if it["layer"] not in layers:
177
+ return None
178
+ evs = [doc["events"][i] for i in it.get("events") or []]
179
+ reg = evs[0]["region"] if evs else -1
180
+ kw = dict(region=reg, main=reg in mains if mains else True, align=aligns.get(reg, "center"), translated=True,
181
+ layer=it["layer"], key=it.get("key"))
182
+ if it["layer"] == "panel":
183
+ if not it["paras"].get(lang):
184
+ return None
185
+ return Item(it["start_us"], it["end_us"], it["rows"], it["paras"][lang], "block", row_h=it["row_h"], **kw)
186
+ text = it.get(lang)
187
+ if not text:
188
+ return None
189
+ if it["layer"] in ("body", "extra"):
190
+ return Item(it["start_us"], it["end_us"], it["rows"], overlay_lines(text, len(it["rows"])), "line",
191
+ events=evs, **kw)
192
+ return Item(it["start_us"], it["end_us"], it["rows"], [text], it["layer"], **kw) # choice / offmain / name:一块一行,不做打字机 / 淡入淡出
193
+
194
+
195
+ PANEL_COVER = 0.8
196
+ """补画 OCR 原文时,框有这么多落在一块已画的阅读面板(`block`)里的事件,**面板在屏的那段时间**不画:面板条目不记它盖住的事件。"""
197
+
198
+
199
+ def unmatched_items(doc: dict, used: set[int], blocks: list[Item] = ()) -> list[Item]:
200
+ """tracks 里没被任何已画出的 matched 条目用到的事件,照 tracks 的画法叠 OCR 原文(全部区域轨,UI 照标)。
201
+ 全部保留(`keep=all`)渲染 matched 时用:没认领上剧本的行、认领了但没这个语种译文的行、关掉的层都还在画面上,
202
+ 看得出匹配漏了什么(owner 2026-09-25)。已画的阅读面板底下的事件只扣掉面板在屏的时段(面板条目不记事件,按框和时间判),
203
+ 面板前后照画;扣剩不到一个采样间隔的碎段不画——面板的起止只有采样级精度,那一截在它的误差里。"""
204
+ out = []
205
+ for x in tracks_items(doc, "all"):
206
+ e = x.events[0]
207
+ if e["id"] in used:
208
+ continue
209
+ b = e["box"]
210
+ a = max(0.0, b[2] - b[0]) * max(0.0, b[3] - b[1])
211
+ cover = []
212
+ for blk in blocks:
213
+ r = blk.rows[0]
214
+ if not (blk.start_us < x.end_us and x.start_us < blk.end_us):
215
+ continue # 时间上不相交的面板和这条无关(同位置别的时段的面板不算)
216
+ ix = max(0.0, min(b[2], r[2]) - max(b[0], r[0])) * max(0.0, min(b[3], r[3]) - max(b[1], r[1]))
217
+ if a > 0 and ix >= PANEL_COVER * a:
218
+ cover.append((blk.start_us, blk.end_us))
219
+ if not cover:
220
+ out.append(x) # 没被面板盖到的照原样(本身短于一个采样间隔的也画)
221
+ continue
222
+ pieces, s = [], x.start_us
223
+ for cs, ce in sorted(cover):
224
+ if cs > s:
225
+ pieces.append((s, min(cs, x.end_us)))
226
+ s = max(s, ce)
227
+ pieces.append((s, x.end_us))
228
+ out += [x if (ps, pe) == (x.start_us, x.end_us) else replace(x, start_us=ps, end_us=pe)
229
+ for ps, pe in pieces if pe - ps >= doc["frame_us"]]
230
+ return out
231
+
232
+
233
+ def native_frame_us(doc: dict) -> int:
234
+ """原生帧长:从产物 `meta` 里 obs 带进来的源帧率算;没有就退到采样间隔(更保守)。"""
235
+ fps = (doc.get("meta") or {}).get("src_fps")
236
+ return int(round(1e6 / fps)) if fps else int(doc["frame_us"])
237
+
238
+
239
+ FRAGMENT_COVER = 0.9
240
+ """框有这么多落在同一条目里**更早出现**的另一个事件框内的事件,算那一行的碎片(`typing_events`)。"""
241
+
242
+ SLOW_TW_FACTOR, SLOW_TW_MIN_ITEMS, SLOW_TW_MIN_CHARS = 3.0, 5, 8
243
+ """打字机字速(毫秒 / 字)比这份字幕稿里**主轨**打字机条目的中位数慢这么多倍的,渲染时列出来报警(owner 2026-09-25:
244
+ 明显慢于主轨上其他打字机的,先当成出了问题去查)。只比主轨上至少这么多字的条目:两三个字的碎片、注音跟着正文一起显出,
245
+ 折成每字毫秒天然偏大,不说明问题;够格的条目不到这么多条就不算中位数、不报。"""
246
+
247
+
248
+ def typing_events(events: list[dict]) -> list[dict]:
249
+ """算效果时刻用的事件:去掉**碎片**——框几乎整个落在同一条目里一个更早出现的事件框内的事件。
250
+ det 偶尔在一两个采样点把一行静止的字切成两框,切出来的那块另起一个事件、被一起认领进这句台词,
251
+ 它的起点晚(一行早就打完了),拿它的"全字出现"会把打字机拖到它身上(gi-s2 129.5 s:
252
+ 两行 1 秒打完,被 135.5 s 冒出的半截行拖成 6 秒)。"""
253
+ def area(b):
254
+ return max(0.0, b[2] - b[0]) * max(0.0, b[3] - b[1])
255
+
256
+ def inside(f, o):
257
+ a = area(f["box"])
258
+ ix = [max(f["box"][0], o["box"][0]), max(f["box"][1], o["box"][1]),
259
+ min(f["box"][2], o["box"][2]), min(f["box"][3], o["box"][3])]
260
+ return a > 0 and area(ix) >= FRAGMENT_COVER * a
261
+
262
+ return [f for f in events
263
+ if not any(o is not f and o["t_start"] < f["t_start"] and inside(f, o) for o in events)]
264
+
265
+
266
+ def effect_times(it: Item, doc: dict, frame_us: int, typewriter: bool = True) -> tuple[int, int, int]:
267
+ """(打字机时长, 淡入, 淡出),单位 µs;不做的是 0。只有不动的 `line` 才做。
268
+
269
+ * 时刻只从 `typing_events` 取(一行被 det 切出来的碎片不算)。
270
+ * 打字机:条目的事件被回抠过(带 `t_full_how`)就只认回抠的 `t_full`;整段素材判成 `instant`、而且是
271
+ "字确实没有生长"那种(离散度 ≤ `INSTANT_MAX_SPREAD`)才不做,判不清的、没判定(样本 < 8)的按逐条的门走;没被回抠的(没回抠的产物、或 `--labels` / `--regions` 没挑到的区域)
272
+ 用采样级的 `t_full_sampled`。按事件判、不按整份产物判:只回抠了部分区域时,其余区域照样有采样级的效果。
273
+ 两种都要 ≥ `TW_MIN_FRAMES` 帧(回抠按原生帧、采样级和"全字没量出来"的回抠值按采样间隔)。
274
+ * 淡出:`t_full_end`(回抠)到条目结束,不足一帧的不做——回抠会把没淡出的也填上,夹在结尾前一点。
275
+ 成员里有**开放边界**(`t_end_how = open`:没看到它消失)的整条不做——不知道它什么时候、怎么消失,
276
+ 不能借别的行的 `t_full_end` 给整句(多行译文连底板)做淡出。
277
+ * 淡入:**实际没做**逐字显出时才复刻(首字出现 → 回抠的全字出现),不足 `TW_MIN_FRAMES` 个原生帧、
278
+ 或全字没量出来的不做,最长 `FADE_IN_MAX_US`。"""
279
+ if it.kind != "line" or it.boxes or not it.events:
280
+ return 0, 0, 0
281
+ s, e = it.start_us, it.end_us
282
+ refined = "text_effect" in doc and any("t_full_how" in ev for ev in it.events)
283
+ core = typing_events(it.events)
284
+
285
+ def last(key):
286
+ v = [ev[key] for ev in core if ev.get(key) is not None]
287
+ return min(max(v), e) if v else None
288
+ full = last("t_full")
289
+ # 回抠只量到首字、全字没量出来(`t_full_how.full == "none"`)时,t_full 是融合的回退值,精度是采样级的:
290
+ # 打字机按采样间隔的门量,不然"多读两个字"骗出来的假打字机会从 2 个原生帧的门漏过去;也不拿它复刻淡入
291
+ unmeasured = "none" in [(ev.get("t_full_how") or {}).get("full") for ev in core if ev.get("t_full") is not None]
292
+ tw = 0
293
+ if typewriter and refined:
294
+ eff = doc["text_effect"]
295
+ no_growth = eff.get("verdict") == "instant" and eff.get("iqr_over_median", 0.0) <= INSTANT_MAX_SPREAD
296
+ step = doc["frame_us"] if unmeasured else frame_us
297
+ if full is not None and not no_growth and full - s >= TW_MIN_FRAMES * step:
298
+ tw = full - s
299
+ elif typewriter:
300
+ fs = last("t_full_sampled")
301
+ if fs is not None and fs - s >= TW_MIN_FRAMES * doc["frame_us"]:
302
+ tw = fs - s
303
+ fend = None if any(ev.get("t_end_how") == "open" for ev in core) else last("t_full_end")
304
+ fout = e - fend if fend is not None and e - fend >= frame_us else 0
305
+ # 淡入也要 ≥ 2 个原生帧:即时出现的字回抠给的全字时刻晚一两帧,不该变成一个 1–2 帧的 \fad
306
+ fin = full - s if tw == 0 and full is not None and not unmeasured and full - s >= TW_MIN_FRAMES * frame_us else 0
307
+ return tw, min(fin, FADE_IN_MAX_US), fout
308
+
309
+
310
+ def effect_why(it: Item, doc: dict, tw: int, fin: int, fout: int) -> str | None:
311
+ """效果判定的依据,一句话(只有 `why=on` 写进字幕稿,`dev` 审计用)。"""
312
+ if it.kind != "line" or not it.events:
313
+ return None
314
+ if it.boxes:
315
+ return "在动,不做效果"
316
+ refined = "text_effect" in doc and any("t_full_how" in ev for ev in it.events)
317
+ src = "回抠" if refined else "采样级"
318
+ if refined and (doc["text_effect"] or {}).get("verdict"):
319
+ src += f",素材判 {doc['text_effect']['verdict']}"
320
+ parts = [f"打字机 {tw / 1e6:.2f}s" if tw else "", f"淡入 {fin / 1e6:.2f}s" if fin else "",
321
+ f"淡出 {fout / 1e6:.2f}s" if fout else ""]
322
+ return ";".join(p for p in parts if p) + f"({src})" if any(parts) else f"无效果({src})"
323
+
324
+
325
+ def slow_typewriters(speeds: list[tuple[float, int]], label: str = "") -> list[tuple[float, int]]:
326
+ """字速比打字机条目中位数慢 `SLOW_TW_FACTOR` 倍的(毫秒 / 字,起点 µs)。有就打一行报警、逐条列出时刻:
327
+ 这种多半不是原文打得慢,而是时刻被别的东西拖长了(碎片、错认领),要去查。"""
328
+ if len(speeds) < SLOW_TW_MIN_ITEMS:
329
+ return []
330
+ med = sorted(v for v, _ in speeds)[len(speeds) // 2]
331
+ slow = sorted(((v, t) for v, t in speeds if v > SLOW_TW_FACTOR * med), key=lambda x: x[1])
332
+ if slow:
333
+ where = "、".join(f"{t / 1e6:.1f} s({v:.0f} ms/字)" for v, t in slow[:12]) + ("…" if len(slow) > 12 else "")
334
+ print(f"⚠ {label + ':' if label else ''}打字机偏慢 {len(slow)} 条(中位数 {med:.0f} ms/字,超过 {SLOW_TW_FACTOR:g} 倍):{where}——先当成出了问题去查")
335
+ return slow
336
+
337
+
338
+ # ---- 写字幕稿 ----
339
+
340
+ def style_of(it: Item) -> str:
341
+ if it.kind == "block":
342
+ return "Panel"
343
+ if it.kind == "choice":
344
+ return "Choice"
345
+ if it.kind == "offmain":
346
+ return "Offmain"
347
+ if it.ui:
348
+ return "UI"
349
+ if it.kind == "name":
350
+ return "Name"
351
+ return "Extra" if it.layer == "extra" else "Body"
352
+
353
+
354
+ def num(v: float) -> str:
355
+ """框和轨迹的坐标写进 `fo`:整数写整数,其余照 `repr` 写全,读回来逐位相同。"""
356
+ return str(int(v)) if float(v).is_integer() else repr(float(v))
357
+
358
+
359
+ def rows_param(rows: list[list[float]]) -> str:
360
+ """数值用空格分隔:参数在 Effect 字段里,逗号会切断 ASS 行的字段。"""
361
+ return "|".join(" ".join(num(x) for x in r) for r in rows)
362
+
363
+
364
+ def path_param(boxes: list) -> str:
365
+ return "|".join(f"{int(t)}:" + " ".join(num(x) for x in b) for t, b in boxes)
366
+
367
+
368
+ def view_rows(it: Item, H: int) -> list[list[float]]:
369
+ """译文比原文框多出几行时,按原来一行的高度扩出去的行框(阶段 4 `layout.expand_rows` 同一规则;字幕稿的预览位置和字号按它算)。"""
370
+ extra = len(it.lines) - len(it.rows)
371
+ return layout.expand_rows(it.rows, extra, H) if it.translated and extra > 0 and it.kind != "block" else it.rows
372
+
373
+
374
+ def draft_line(it: Item, size: int, W: int, H: int, widths: dict, tw_us: int, fin: int, fout: int, why: str | None,
375
+ font: str = export.FONT_CN) -> str:
376
+ """一个条目 → 字幕稿里的一行源行:正文是一个主流 tag 块 + 纯文字(行间 `\\N`),阶段 4 的参数 `fo:…` 放在 Effect 字段——
377
+ 正文里没有参数,改字的时候碰不到(owner 2026-09-27 看 Aegisub 里的字幕稿后定:正文尽量是纯文字、便于编辑;参数放 Effect)。"""
378
+ t0, t1 = export.ass_time(it.start_us, True), export.ass_time(it.end_us, False)
379
+ actor = f"r{it.region}" if it.region >= 0 else ""
380
+ ev = list(dict.fromkeys(e["id"] for e in it.events if "id" in e)) # matched 条目的事件可能重复列出
381
+ fo: dict = {"ev": " ".join(map(str, ev)) or None, "key": it.key, "box": rows_param(it.rows)}
382
+ if it.kind == "block":
383
+ size, wrapped = layout.block_layout(it.rows[0], it.lines, it.row_h, font, widths)
384
+ pos = "%.0f,%.0f" % (it.rows[0][0], it.rows[0][1])
385
+ fo.update({"wrap": it.row_h, "plate": "box", "off": not it.main,
386
+ "was": f"pos:{pos.replace(',', ' ')}|fs:{size}|an:7"})
387
+ head = "{\\an7\\pos(%s)\\fs%d\\q2}" % (pos, size)
388
+ body = "\\N".join(export.ass_text(ln) for ln in wrapped)
389
+ return f"Dialogue: 0,{t0},{t1},{style_of(it)},{actor},0,0,0,{assfile.encode_fo(fo)},{head}{body}"
390
+ rows = view_rows(it, H)
391
+ tw = [layout.line_width(font, ln, widths) * size for ln in it.lines]
392
+ per_row = len(it.lines) == len(rows)
393
+ axes = layout.axes_in_screen(rows, tw if per_row else [max(tw, default=0.0)] * len(rows),
394
+ it.align, it.translated, W)
395
+ x = axes[0]
396
+ y = (rows[0][1] + rows[-1][3]) / 2
397
+ if it.boxes: # 预览:整条一个 \move 从首点到末点,精确轨迹在 path 里
398
+ p0, p1 = tuple(it.boxes[0][1][:2]), tuple(it.boxes[-1][1][:2])
399
+ base = (it.rows[0][0], it.rows[0][1])
400
+ where = layout.pos_tag(x, y, (it.start_us, it.end_us, p0, p1), base)
401
+ else:
402
+ where = "\\pos(%.0f,%.0f)" % (x, y)
403
+ owned = where[1:].replace("(", ":", 1).rstrip(")").replace(",", " ") # "pos:x y" / "move:x0 y0 x1 y1 0 t"
404
+ # 原文框本来装几行:OCR 原文带换行时一个框装着几行(tracks 事件本身带换行),比框数多;阶段 4 只把超出的行往外扩
405
+ cap = sum(len([s for s in ln.split("\n") if s.strip()]) or 1 for ln in it.lines) if not it.translated else len(it.rows)
406
+ fo.update({"fit": "tr" if it.translated else "src", "rows": True, "plate": True,
407
+ "cap": cap if cap != len(it.rows) else None,
408
+ "path": path_param(it.boxes) if it.boxes else None, "tw": tw_us // 1000 or None,
409
+ "off": not it.main, "why": why, "was": f"{owned}|fs:{size}|an:{layout.ANCHOR[it.align]}"})
410
+ fad = "\\fad(%d,%d)" % (fin // 1000, fout // 1000) if fin or fout else ""
411
+ head = "{\\an%d%s\\fs%d%s}" % (layout.ANCHOR[it.align], where, size, fad)
412
+ body = "\\N".join(export.ass_text(ln) for ln in it.lines)
413
+ return f"Dialogue: 0,{t0},{t1},{style_of(it)},{actor},0,0,0,{assfile.encode_fo(fo)},{head}{body}"
414
+
415
+
416
+ def style_lines(font: str = export.FONT_CN) -> list[str]:
417
+ """字幕稿的样式:一个角色一个,参数相同。`BorderStyle=3` 出预览底框(各渲染器都认,框用 OutlineColour、外扩 `OVERLAY_PAD`);
418
+ 次要色全透明、阴影 0:人要自己调某一行的打字机节奏时写 `\\k` 族 tag,预览就是逐字显出(`\\ko` 藏不住阴影,第 0 步探针量过)。"""
419
+ return [f"Style: {s},{font},48,&H00FFFFFF,&HFF000000,{PREVIEW_BOX},&H00000000,"
420
+ f"0,0,0,0,100,100,0,0,3,{layout.OVERLAY_PAD},0,5,0,0,0,1" for s in STYLES]
421
+
422
+
423
+ def export_script(path: Path, items: list[Item], doc: dict, typewriter: bool = True, fade: bool = True,
424
+ why: bool = False, info: dict | None = None) -> tuple[Counter, dict]:
425
+ """把条目写成字幕稿。返回(计数:各种类、打字机 / 淡入淡出条数、对齐;量好的字宽——连跑阶段 4 时交给它,不用再量一遍)。
426
+
427
+ * 字号 = `layout.fit_line`(原文只按框,译文有保底、不超画面);同栏同时在屏的选项取中位数(`layout.same_size`);
428
+ 面板按 `layout.block_layout` 折行;
429
+ * 位置:单行就是阶段 4 的位置;多行是整组的锚轴和几行的竖直中心(阶段 4 按原框逐行摆,预览里行距是字体的);
430
+ 在动的写一个首点到末点的 `\\move` 供预览,精确轨迹在 `fo` 的 `path`;
431
+ * 效果见 `effect_times`,打字机记成 `fo` 的 `tw`(正文保持纯文字)、淡入淡出写成 `\\fad`。"""
432
+ W, H = doc["size"]
433
+ frame_us = native_frame_us(doc)
434
+ font = export.FONT_CN
435
+ widths = export.ink_widths((font, "\n".join(layout.width_keys(it.lines, it.kind == "block"))) for it in items)
436
+ sizes = [layout.fit_line(view_rows(it, H), it.lines, font, widths, it.translated, W) if it.kind != "block" else 0
437
+ for it in items]
438
+ choice = [i for i, it in enumerate(items) if it.kind in ("choice", "offmain")]
439
+ for i, s in zip(choice, layout.same_size([(items[i].start_us, items[i].end_us, items[i].rows[0]) for i in choice],
440
+ [sizes[i] for i in choice])):
441
+ sizes[i] = s
442
+ lines_out, n = [], Counter()
443
+ speeds: list[tuple[float, int]] = [] # 主轨打字机条目的(毫秒 / 字,起点),见 slow_typewriters
444
+ for it, size in zip(items, sizes):
445
+ n[it.kind] += 1
446
+ tw_us, fin, fout = effect_times(it, doc, frame_us, typewriter) if it.kind != "block" else (0, 0, 0)
447
+ if not fade:
448
+ fin = fout = 0
449
+ n["typewriter"] += tw_us > 0
450
+ n["fade"] += bool(fin or fout)
451
+ if it.kind != "block":
452
+ n[f"align_{it.align}"] += 1
453
+ n_chars = sum(len(ln) for ln in it.lines)
454
+ if tw_us > 0 and it.main and n_chars >= SLOW_TW_MIN_CHARS:
455
+ speeds.append((tw_us / 1000 / (n_chars - 1), it.start_us))
456
+ lines_out.append(draft_line(it, size, W, H, widths, tw_us, fin, fout,
457
+ effect_why(it, doc, tw_us, fin, fout) if why else None, font))
458
+ head = {"FlowOCR Script": FORMAT_VERSION, **(info or {})}
459
+ Path(path).write_text(export.ass_doc(W, H, style_lines(font), lines_out, info=head), encoding="utf-8")
460
+ n["typewriter_slow"] = len(slow_typewriters(speeds, Path(path).name))
461
+ return n, widths
462
+
463
+
464
+ KEEPS = ("main", "all")
465
+
466
+
467
+ def events_fingerprint(doc: dict) -> str:
468
+ """tracks 的 `events[]` 的内容指纹(sha256 前 12 位)。字幕稿里的 `ev` 是 `events[]` 的下标,重跑阶段 2 会重新编号——
469
+ 以后要按 `ev` 回到产物(合并、回流),先核字幕稿记的这个指纹和手上的产物对不对得上。"""
470
+ blob = json.dumps(doc["events"], sort_keys=True, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
471
+ return hashlib.sha256(blob).hexdigest()[:12]
472
+
473
+
474
+ def render_script(document: dict, output_dir, options: dict | None, suffix: str = ""
475
+ ) -> tuple[Path, Counter, list[Item], dict]:
476
+ """阶段 3 的实现:吃 tracks(叠 OCR 原文)或 matched(叠剧本译文),写一份字幕稿。
477
+
478
+ options(字符串,`render --opt` 原样给的):
479
+ * `keep`:`main`(默认:tracks 主轨 + 名牌轨;matched 的 `body` / `extra` / `name`)/ `all`(全部区域轨、全部层,
480
+ matched 另把没叠译文的事件照 tracks 画 OCR 原文,见 `unmatched_items`);
481
+ * `why`:`on` 把效果判定的依据写进每行的 `fo`(dev 审计用);
482
+ * `typewriter`(off / on / dim,dim 在这里等于 on)、`fade`(off / on):off 就不判、不写;
483
+ * matched 还有 `lang`(jp,默认)/ `cn`、`layers`(逗号分隔,默认随 `keep`)、
484
+ `tracks`(产它的那份 `*-tracks.json`,默认按 `provenance.subs` 找);
485
+ * `name`:文件名(默认 `<tag>[-<lang>]-<suffix>.script.ass`,`suffix` 默认是 `keep`;叠加预设给的是预设名)。"""
486
+ o = dict(options or {})
487
+ keep = str(o.get("keep", "main"))
488
+ if keep not in KEEPS:
489
+ raise ValueError(f"keep 只能取 {list(KEEPS)}:{keep!r}")
490
+ typewriter = str(o.get("typewriter", "on"))
491
+ if typewriter not in TYPEWRITER_MODES:
492
+ raise ValueError(f"typewriter 只能取 {list(TYPEWRITER_MODES)}:{typewriter!r}")
493
+ fade = str(o.get("fade", "on"))
494
+ if fade not in ("off", "on"):
495
+ raise ValueError(f"fade 只能是 off / on:{fade!r}")
496
+ why = str(o.get("why", "off"))
497
+ if why not in ("off", "on"):
498
+ raise ValueError(f"why 只能是 off / on:{why!r}")
499
+ layers = (tuple(x for x in str(o["layers"]).split(",") if x) if o.get("layers")
500
+ else ("body", "extra", "name") if keep == "main" else LAYERS)
501
+ if not set(layers) <= set(LAYERS):
502
+ raise ValueError(f"layers 只能取 {','.join(LAYERS)} 里的:{o.get('layers')}")
503
+ out = Path(output_dir)
504
+ if document.get("schema") == matchedio.SCHEMA:
505
+ matchedio.validate(document, need_overlay=True)
506
+ lang = str(o.get("lang", "jp"))
507
+ if lang not in ("jp", "cn"):
508
+ raise ValueError(f"lang 只能是 jp / cn:{lang!r}")
509
+ tracks = str(o.get("tracks") or matchedio.tracks_path(document))
510
+ if not tracks or not Path(tracks).is_file():
511
+ raise FileNotFoundError(
512
+ f"找不到这份匹配产物对应的 tracks(provenance.subs = {matchedio.tracks_path(document)!r});"
513
+ f"叠加要它的框和画面尺寸,用 options['tracks'] 指过去")
514
+ doc = tracksio.load(Path(tracks))
515
+ need_align(doc, tracks)
516
+ used: set[int] = set()
517
+ items = project_items(document["overlay"]["items"], doc, lang, layers, used)
518
+ note = f"剧本译文 {lang},层 {','.join(layers)}"
519
+ if keep == "all":
520
+ rest = unmatched_items(doc, used, [it for it in items if it.kind == "block"])
521
+ items += rest
522
+ note += f",另 {len(rest)} 个没叠译文的事件画 OCR 原文"
523
+ tag = doc.get("tag") or Path(tracks).stem.removesuffix("-refined").removesuffix("-tracks")
524
+ base = f"{tag}-{lang}"
525
+ source = f"{tag} {matchedio.SCHEMA}"
526
+ else:
527
+ doc = document
528
+ need_align(doc, str(doc.get("tag") or "这份 tracks"))
529
+ items = tracks_items(doc, keep)
530
+ base = str(doc.get("tag") or "tracks")
531
+ note = f"OCR 原文,{'主轨 + 名牌' if keep == 'main' else '全部区域轨'}"
532
+ source = f"{base} {tracksio.SCHEMA}"
533
+ out.mkdir(parents=True, exist_ok=True)
534
+ p = out / str(o.get("name") or f"{base}-{suffix or keep}.script.ass")
535
+ info = {"FlowOCR Source": f"{source} events-sha256:{events_fingerprint(doc)}",
536
+ "FlowOCR Code": provenance.git_head(), "FlowOCR Note": note}
537
+ n, widths = export_script(p, items, doc, typewriter != "off", fade == "on", why == "on", info)
538
+ kinds = "、".join(f"{k} {n[k]}" for k in ("line", "name", "choice", "offmain", "block") if n[k])
539
+ aligns = "、".join(f"{a} {n['align_' + a]}" for a in tracksio.ALIGNS if n["align_" + a])
540
+ print(f"-> {p}(字幕稿,{len(items)} 条:{kinds};对齐 {aligns};打字机 {n['typewriter']} 条、淡入淡出 {n['fade']} 条)")
541
+ return p, n, items, widths