flowocr 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. flowocr/__init__.py +15 -0
  2. flowocr/analyze/__init__.py +8 -0
  3. flowocr/analyze/align.py +151 -0
  4. flowocr/analyze/build_tracks.py +2886 -0
  5. flowocr/analyze/cluster_layers.py +155 -0
  6. flowocr/analyze/game_align.py +652 -0
  7. flowocr/analyze/gamescript.py +1220 -0
  8. flowocr/analyze/gtdbundle.py +306 -0
  9. flowocr/analyze/match.py +67 -0
  10. flowocr/analyze/matchers/__init__.py +5 -0
  11. flowocr/analyze/matchers/gametext.py +70 -0
  12. flowocr/analyze/merge_nameplate.py +178 -0
  13. flowocr/analyze/models/slot_pair.json +148 -0
  14. flowocr/analyze/nameplate.py +148 -0
  15. flowocr/analyze/pair_features.py +168 -0
  16. flowocr/analyze/pair_model.py +127 -0
  17. flowocr/analyze/refine_boundaries.py +409 -0
  18. flowocr/analyze/script_align.py +641 -0
  19. flowocr/analyze/scriptmatch.py +1018 -0
  20. flowocr/analyze/slot_learned.py +306 -0
  21. flowocr/analyze/slot_lines.py +251 -0
  22. flowocr/analyze/slot_modes.py +143 -0
  23. flowocr/analyze/slot_pairs.py +562 -0
  24. flowocr/analyze/slot_veto.py +104 -0
  25. flowocr/analyze/uigate.py +351 -0
  26. flowocr/artifacts/__init__.py +5 -0
  27. flowocr/artifacts/evalkit.py +123 -0
  28. flowocr/artifacts/matchedio.py +75 -0
  29. flowocr/artifacts/srtio.py +166 -0
  30. flowocr/artifacts/tracksio.py +345 -0
  31. flowocr/extensions.py +77 -0
  32. flowocr/extract/__init__.py +8 -0
  33. flowocr/extract/childproc.py +118 -0
  34. flowocr/extract/decode_proc.py +511 -0
  35. flowocr/extract/decode_shards.py +574 -0
  36. flowocr/extract/detpost.py +215 -0
  37. flowocr/extract/edge_proc.py +314 -0
  38. flowocr/extract/edge_refine.py +598 -0
  39. flowocr/extract/fast_det.py +214 -0
  40. flowocr/extract/ffcheck.py +87 -0
  41. flowocr/extract/framegrid.py +265 -0
  42. flowocr/extract/framesource.py +1264 -0
  43. flowocr/extract/ocr_args.py +754 -0
  44. flowocr/extract/ocr_complete.py +231 -0
  45. flowocr/extract/ocr_parallel.py +208 -0
  46. flowocr/extract/ort_server.py +632 -0
  47. flowocr/extract/ortclient.py +363 -0
  48. flowocr/extract/ptsclock.py +150 -0
  49. flowocr/extract/recdecode.py +78 -0
  50. flowocr/extract/recort.py +162 -0
  51. flowocr/extract/recpack.py +94 -0
  52. flowocr/extract/recpool.py +206 -0
  53. flowocr/extract/recprep.py +71 -0
  54. flowocr/extract/refine_video.py +271 -0
  55. flowocr/extract/regions.py +387 -0
  56. flowocr/extract/reuse_v2.py +593 -0
  57. flowocr/extract/run_groups.py +247 -0
  58. flowocr/extract/run_ocr2.py +1605 -0
  59. flowocr/extract/supervisor.py +261 -0
  60. flowocr/extract/timeline.py +84 -0
  61. flowocr/extract/typewriter_fuse.py +394 -0
  62. flowocr/models.py +263 -0
  63. flowocr/output/__init__.py +3 -0
  64. flowocr/output/export.py +205 -0
  65. flowocr/output/layout.py +210 -0
  66. flowocr/output/presets/__init__.py +6 -0
  67. flowocr/output/presets/_overlay.py +40 -0
  68. flowocr/output/presets/default.py +21 -0
  69. flowocr/output/presets/default_all.py +20 -0
  70. flowocr/output/presets/dev.py +21 -0
  71. flowocr/output/presets/matched_srt.py +32 -0
  72. flowocr/output/presets/script.py +20 -0
  73. flowocr/output/presets/srt_main.py +33 -0
  74. flowocr/output/render.py +82 -0
  75. flowocr/output/run_srt.py +83 -0
  76. flowocr/output/script.py +541 -0
  77. flowocr/paths.py +168 -0
  78. flowocr/provenance.py +119 -0
  79. flowocr/typeset/__init__.py +9 -0
  80. flowocr/typeset/__main__.py +38 -0
  81. flowocr/typeset/assfile.py +216 -0
  82. flowocr/typeset/core.py +821 -0
  83. flowocr/typeset/fx/__init__.py +11 -0
  84. flowocr-0.1.0.dist-info/METADATA +109 -0
  85. flowocr-0.1.0.dist-info/RECORD +91 -0
  86. flowocr-0.1.0.dist-info/WHEEL +5 -0
  87. flowocr-0.1.0.dist-info/entry_points.txt +10 -0
  88. flowocr-0.1.0.dist-info/licenses/LICENSE +674 -0
  89. flowocr-0.1.0.dist-info/licenses/LICENSES/Apache-2.0.txt +201 -0
  90. flowocr-0.1.0.dist-info/licenses/LICENSES/PP-OCRv6-NOTICE.md +18 -0
  91. flowocr-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,306 @@
1
+ """游戏文本包(`gtd-bundle/1`):找包、校验、读表。
2
+
3
+ 包由 game-text-data 项目 `uv run gametext <游戏> bundle` 出(协议在那边 `docs/architecture.md` 的"产物包协议"一节),一款游戏一个:
4
+ `<game>-<version>-<指纹前 12 位>.zip`,里面 `<game>/manifest.json` + 数据文件 + `report.json` + `NOTICE.txt`。
5
+ 不公开分发,用户私下拿到后放进 `paths.gametext_root()`(默认数据根的 `gametext/`)即可,**不用解压**;解压了也认。
6
+
7
+ - **找**:`--gametext` / `FLOWOCR_GAMETEXT` / 默认位置可以是一个 `.zip`、一个带 `manifest.json` 的目录(开发机直接指
8
+ `../game-text-data/corpus/<游戏>`),或一个装着若干包的目录(按清单的 `game` 挑;解压出的目录往下找两层,见 `candidates`)。
9
+ 同一款游戏找到指纹不同的多份就报错列出来,不按版本号猜;指纹相同的几份是同一份语料,用排在前面的那份。
10
+ 没有清单的目录不认——不留旧格式的读法。
11
+ - **校验**:清单 `schema` 要认得、要有调用方要的表和语种(`ja` / `zh-Hans`)、指纹按清单重算要对上;
12
+ 调用方要的每张表还要带调用方认得的**内容契约主版本**(清单 `tables.<表>.contract`,形如 `genshin.reminders/1`,
13
+ 字段的类型、能否为 null、取值写在 game-text-data 那边的契约声明里):缺 `contract` 是加契约之前打的旧包,
14
+ 主版本不同是字段含义变了,都拒绝;每个读到的文件
15
+ 边读边算存储字节的 `sha256` 与解压后的 `content_sha256`,读完和清单比。缓存命中不读数据时用 `verify_stored`
16
+ 只核存储字节(每款几十 MB)。哈希防的是拷坏和版本对不上,不是安全机制。
17
+ - **身份**:下游只认 `fingerprint`(按内容算,换压缩、改文件名都不变)。`provenance()` 是剧本里记的那一份。
18
+ 清单的 `rev` 是 game-text-data 发布时记进指纹表的提取行为版本(整数;本地包是 `<rev>+<指纹前 8 位>`),
19
+ 只抄下来给人看——核"是不是维护者发出的那份"是拿指纹去对那边公开的指纹表,这里不联网、不校验它。
20
+ 命令行(`main`)打印指纹之前先 `verify` 全部文件的两种哈希,否则内容被换过的包也能打印出一个对得上表的指纹。
21
+ """
22
+ from __future__ import annotations
23
+
24
+ import hashlib
25
+ import io
26
+ import json
27
+ import sys
28
+ import zipfile
29
+ from dataclasses import dataclass
30
+ from pathlib import Path
31
+
32
+ from flowocr import paths
33
+
34
+ SCHEMA = "gtd-bundle/1"
35
+ MANIFEST = "manifest.json"
36
+ CHUNK = 1 << 20
37
+
38
+
39
+ class BundleError(SystemExit):
40
+ """找不到包、包坏了、和清单对不上。继承 SystemExit:CLI 直接带着这句话退出,守卫里照样能 catch。"""
41
+
42
+
43
+ def fingerprint_of(files: list[dict]) -> str:
44
+ """协议里的指纹:按 `(table, lang)` 排序的 `table\\tlang\\tcontent_sha256\\n` 串接的 sha256,只算数据表。"""
45
+ items = sorted((f["table"], f["lang"], f["content_sha256"]) for f in files if "table" in f)
46
+ return "sha256:" + hashlib.sha256("".join("%s\t%s\t%s\n" % it for it in items).encode("utf-8")).hexdigest()
47
+
48
+
49
+ class _Tee:
50
+ """读的同时喂哈希(给 zstd 的 stream_reader 当源)。"""
51
+
52
+ def __init__(self, f, h):
53
+ self.f, self.h = f, h
54
+
55
+ def read(self, n: int = -1) -> bytes:
56
+ b = self.f.read(n)
57
+ self.h.update(b)
58
+ return b
59
+
60
+
61
+ @dataclass
62
+ class Bundle:
63
+ source: Path # .zip 或目录
64
+ manifest: dict
65
+ prefix: str = "" # zip 里的成员前缀(`<game>/`);目录形态为空
66
+
67
+ @property
68
+ def game(self) -> str:
69
+ return self.manifest["game"]
70
+
71
+ @property
72
+ def fingerprint(self) -> str:
73
+ return self.manifest["fingerprint"]
74
+
75
+ def provenance(self) -> dict:
76
+ """剧本 provenance 里的 `gametext`。`source` 只记文件 / 目录名,不记绝对路径(产物会被分享)。"""
77
+ m = self.manifest
78
+ up = m.get("upstream") or {}
79
+ return {"game": m["game"], "fingerprint": m["fingerprint"], "rev": m.get("rev"), "version": up.get("version"),
80
+ "upstream": {k: up.get(k) for k in ("repo", "commit", "version")},
81
+ "extractor": m.get("extractor") or {},
82
+ "integrity_checked": bool(up.get("integrity_checked")),
83
+ "source": self.source.name}
84
+
85
+ def _files(self) -> dict[tuple[str, str], dict]:
86
+ return {(f["table"], f["lang"]): f for f in self.manifest["files"] if "table" in f}
87
+
88
+ def _open(self, rel: str):
89
+ if self.source.suffix.lower() == ".zip":
90
+ zf = zipfile.ZipFile(self.source)
91
+ try:
92
+ raw = zf.open(self.prefix + rel)
93
+ except KeyError:
94
+ zf.close()
95
+ raise BundleError(f"文本包 {self.source.name} 里缺 {self.prefix + rel}(清单里有)")
96
+ return _Closing(raw, zf)
97
+ p = self.source / rel
98
+ if not p.is_file():
99
+ raise BundleError(f"文本包 {self.source} 里缺 {rel}(清单里有)")
100
+ return open(p, "rb")
101
+
102
+ def has(self, tables, langs=("ja", "zh-Hans")) -> bool:
103
+ """可选的表:这几张表 × 语种都在清单里就 True(旧包没有的表调用方自己跳过,不报错)。"""
104
+ have = self._files()
105
+ return all((t, lang) in have for t in tables for lang in langs)
106
+
107
+ def require(self, tables: dict[str, int], langs=("ja", "zh-Hans")) -> None:
108
+ """调用方要的表({表: 认得的契约主版本})× 语种都得在清单里,每张表的契约主版本要对上——
109
+ 缺了就报缺什么(`--langs chs` 收窄跑出的包会缺日文),版本不对就报是哪张表、包里是几、这里要几。"""
110
+ have = self._files()
111
+ miss = [f"{t}×{lang}" for t in tables for lang in langs if (t, lang) not in have]
112
+ if miss:
113
+ raise BundleError(f"文本包 {self.source.name}({self.game})缺 {', '.join(miss)}:"
114
+ f"清单里的语种是 {sorted(self.manifest.get('langs') or {})}")
115
+ for t, major in tables.items():
116
+ label = ((self.manifest.get("tables") or {}).get(t) or {}).get("contract")
117
+ if label is None:
118
+ raise BundleError(f"文本包 {self.source.name}({self.game})的 {t} 没有内容契约版本:这是加契约之前打的包,"
119
+ f"game-text-data 那边用新代码重新 bundle")
120
+ if label != f"{self.game}.{t}/{major}":
121
+ raise BundleError(f"文本包 {self.source.name} 的 {t} 是契约 {label},这里只认 {self.game}.{t}/{major}:"
122
+ f"字段含义变了,flowocr 要先跟上(看 game-text-data 的 CHANGELOG)")
123
+
124
+ def _chunks(self, f: dict):
125
+ """逐块给出清单里这个文件解压后的字节,边读边算两种哈希;**读完**和清单比,对不上就报错
126
+ (读到一半就停的调用方不做这次核对)。解压 / zip 的 CRC 出错都算"包坏了",不让用户看底层栈。"""
127
+ import zstandard
128
+ raw_h, con_h = hashlib.sha256(), hashlib.sha256()
129
+ zst = f["path"].endswith(".zst")
130
+ try:
131
+ with self._open(f["path"]) as raw:
132
+ src = _Tee(raw, raw_h)
133
+ stream = zstandard.ZstdDecompressor().stream_reader(src) if zst else src
134
+ for chunk in iter(lambda: stream.read(CHUNK), b""):
135
+ if zst:
136
+ con_h.update(chunk)
137
+ yield chunk
138
+ while src.read(CHUNK): # zstd 帧后面若还有字节,也要进存储哈希
139
+ pass
140
+ except (zstandard.ZstdError, zipfile.BadZipFile) as exc:
141
+ raise self._broken(f, exc)
142
+ content = con_h if zst else raw_h
143
+ if raw_h.hexdigest() != f["sha256"] or content.hexdigest() != f.get("content_sha256", f["sha256"]):
144
+ raise BundleError(f"文本包 {self.source.name} 的 {f['path']} 和清单对不上(sha256)——包坏了或被改过,重新拿一份")
145
+
146
+ def _broken(self, f: dict, exc: Exception) -> BundleError:
147
+ return BundleError(f"文本包 {self.source.name} 的 {f['path']} 读不下去({type(exc).__name__}: {exc})"
148
+ f"——包坏了或被改过,重新拿一份")
149
+
150
+ def iter_rows(self, table: str, lang: str):
151
+ """逐行读一张表(`_chunks`:边读边核两种哈希)。JSON 解码出错也算包坏了(UnicodeDecodeError 是 ValueError)。"""
152
+ f = self._files()[(table, lang)]
153
+ pending = b""
154
+ try:
155
+ for chunk in self._chunks(f):
156
+ lines = (pending + chunk).split(b"\n")
157
+ pending = lines.pop()
158
+ rows = [json.loads(ln) for ln in lines if ln.strip()]
159
+ yield from rows
160
+ if pending.strip():
161
+ yield json.loads(pending)
162
+ except ValueError as exc:
163
+ raise self._broken(f, exc)
164
+
165
+ def verify(self) -> None:
166
+ """清单里的每个文件都核一遍:存储字节的 `sha256`,数据文件再核解压后的 `content_sha256`。
167
+ 拿指纹去对指纹表之前用——只核存储字节的话,改了内容、同时改掉清单里存储哈希的包照样过,
168
+ 而指纹(只由内容哈希算)还和表对得上。"""
169
+ for f in self.manifest["files"]:
170
+ for _ in self._chunks(f):
171
+ pass
172
+
173
+ def verify_stored(self, tables, langs=("ja", "zh-Hans")) -> None:
174
+ """只核存储字节的 sha256(缓存命中、不解压时用)。"""
175
+ have = self._files()
176
+ for t in tables:
177
+ for lang in langs:
178
+ f = have[(t, lang)]
179
+ h = hashlib.sha256()
180
+ with self._open(f["path"]) as raw:
181
+ for chunk in iter(lambda: raw.read(CHUNK), b""):
182
+ h.update(chunk)
183
+ if h.hexdigest() != f["sha256"]:
184
+ raise BundleError(f"文本包 {self.source.name} 的 {f['path']} 和清单对不上(sha256)——包坏了或被改过,重新拿一份")
185
+
186
+
187
+ class _Closing(io.RawIOBase):
188
+ """zip 成员 + 它的 ZipFile,一起关。"""
189
+
190
+ def __init__(self, raw, zf):
191
+ self.raw, self.zf = raw, zf
192
+
193
+ def read(self, n: int = -1) -> bytes:
194
+ return self.raw.read(n)
195
+
196
+ def readable(self) -> bool:
197
+ return True
198
+
199
+ def close(self) -> None:
200
+ try:
201
+ self.raw.close()
202
+ finally:
203
+ self.zf.close()
204
+ super().close()
205
+
206
+
207
+ def _load_manifest(text: str, where: str) -> dict:
208
+ try:
209
+ m = json.loads(text)
210
+ except ValueError as exc:
211
+ raise BundleError(f"{where} 的清单不是合法 JSON:{exc}")
212
+ if m.get("schema") != SCHEMA:
213
+ raise BundleError(f"{where} 的清单 schema 是 {m.get('schema')!r},这里只认 {SCHEMA}(game-text-data 那边重新 bundle)")
214
+ for k in ("game", "fingerprint", "files", "langs"):
215
+ if k not in m:
216
+ raise BundleError(f"{where} 的清单缺 {k}")
217
+ if fingerprint_of(m["files"]) != m["fingerprint"]:
218
+ raise BundleError(f"{where} 的清单自相矛盾:按文件重算的指纹不是它记的 {m['fingerprint']}")
219
+ return m
220
+
221
+
222
+ def open_bundle(p: Path) -> Bundle:
223
+ """一个包:`.zip`,或带 `manifest.json` 的目录。"""
224
+ if p.is_file() and p.suffix.lower() == ".zip":
225
+ try:
226
+ with zipfile.ZipFile(p) as zf:
227
+ names = [n for n in zf.namelist() if n.count("/") == 1 and n.endswith("/" + MANIFEST)]
228
+ if len(names) != 1:
229
+ raise BundleError(f"{p.name} 不是文本包:顶层目录下应当恰好一份 {MANIFEST},找到 {len(names)} 份")
230
+ m = _load_manifest(zf.read(names[0]).decode("utf-8"), p.name)
231
+ prefix = names[0][: -len(MANIFEST)]
232
+ except zipfile.BadZipFile as exc:
233
+ raise BundleError(f"{p} 不是有效的 zip:{exc}")
234
+ if prefix != m["game"] + "/":
235
+ raise BundleError(f"{p.name}:成员目录 {prefix!r} 和清单的 game {m['game']!r} 不一致")
236
+ return Bundle(p, m, prefix)
237
+ if (p / MANIFEST).is_file():
238
+ return Bundle(p, _load_manifest((p / MANIFEST).read_text(encoding="utf-8"), str(p)))
239
+ raise BundleError(f"{p} 既不是文本包(.zip),也不是带 {MANIFEST} 的目录")
240
+
241
+
242
+ def _has_manifest(p: Path) -> bool:
243
+ try:
244
+ with zipfile.ZipFile(p) as zf:
245
+ return any(n.count("/") == 1 and n.endswith("/" + MANIFEST) for n in zf.namelist())
246
+ except zipfile.BadZipFile:
247
+ return False
248
+
249
+
250
+ def candidates(root: Path) -> list[Path]:
251
+ """`root` 下可能是包的东西:它自己;或它下面一层的 `.zip` 与带清单的子目录,以及再下一层带清单的目录——
252
+ 包解压出来是 `<game>/manifest.json`,而 Windows 的"全部解压缩"默认再套一层以 zip 命名的文件夹。"""
253
+ if root.is_file() or (root / MANIFEST).is_file():
254
+ return [root]
255
+ if not root.is_dir():
256
+ return []
257
+ found = list(root.glob("*.zip"))
258
+ for q in root.iterdir():
259
+ if not q.is_dir():
260
+ continue
261
+ if (q / MANIFEST).is_file():
262
+ found.append(q)
263
+ else:
264
+ found += [s for s in q.iterdir() if s.is_dir() and (s / MANIFEST).is_file()]
265
+ return sorted(found)
266
+
267
+
268
+ def locate(game: str, where: str | Path | None = None) -> Bundle:
269
+ """找 `game` 的包。`where` 不给就用 `paths.gametext_root()`。"""
270
+ root = Path(where).expanduser().resolve() if where else paths.gametext_root()
271
+ bundles = []
272
+ for q in candidates(root):
273
+ if q.is_file() and q.suffix.lower() == ".zip" and not _has_manifest(q):
274
+ # 只给人看的一行(驱动里 same_ref 那段也会打出来):别拿 stdout 解析
275
+ print(f" [文本包] 跳过 {q.name}:不是文本包(顶层目录下没有 {MANIFEST})", file=sys.stderr, flush=True)
276
+ continue
277
+ bundles.append(open_bundle(q)) # 有清单但清单坏了的照样报错
278
+ found = [b for b in bundles if b.game == game]
279
+ if not found:
280
+ raise BundleError(f"没找到 {game} 的游戏文本包(在 {root} 找的)。包从 game-text-data 的 `bundle` 来,"
281
+ f"放进这个目录即可,或用 --gametext / FLOWOCR_GAMETEXT 指过去")
282
+ # 指纹相同就是同一份语料(例如 zip 和它解压出的目录并存),用排在前面的那份;内容不同才要用户挑
283
+ if len({b.fingerprint for b in found}) > 1:
284
+ listing = [f"{b.source.name}({b.fingerprint.removeprefix('sha256:')[:12]})" for b in found]
285
+ raise BundleError(f"{root} 下 {game} 的文本包有 {len(found)} 份、内容不同:{listing}——"
286
+ f"用 --gametext 指定其中一份")
287
+ return found[0]
288
+
289
+
290
+ def main(argv: list[str] | None = None) -> int:
291
+ import argparse
292
+ ap = argparse.ArgumentParser(prog="python -m flowocr.analyze.gtdbundle",
293
+ description="找游戏文本包、核对包里每个文件的哈希,再打印它的身份(指纹 / 位置 / 全部来源信息)")
294
+ ap.add_argument("game")
295
+ ap.add_argument("--gametext", default=None, help="包 / 包目录(默认 paths.gametext_root())")
296
+ ap.add_argument("--field", choices=("fingerprint", "source", "json"), default="fingerprint")
297
+ a = ap.parse_args(argv)
298
+ b = locate(a.game, a.gametext)
299
+ b.verify() # 打印出的指纹是拿去对指纹表的,先证明包里的内容就是清单说的那份
300
+ print(json.dumps(b.provenance(), ensure_ascii=False) if a.field == "json"
301
+ else b.fingerprint if a.field == "fingerprint" else str(b.source))
302
+ return 0
303
+
304
+
305
+ if __name__ == "__main__":
306
+ raise SystemExit(main())
@@ -0,0 +1,67 @@
1
+ """阶段 2 的**匹配**入口:拿一份 `*-tracks.json`,跑一个匹配器(内置的或用户自己写的),写出匹配产物。
2
+
3
+ python -m flowocr.analyze.match out/gs-hsr/hsr-tracks.json --matcher gametext \\
4
+ --ctx ref=out/gametext/hsr-ref.json --out out/gametext/hsr-matched.json --opt lang=cn
5
+ python -m flowocr.analyze.match out/x/x-tracks.json --matcher D:/mine/matcher.py --out tmp/x-matched.json
6
+
7
+ 匹配器从哪来、入口长什么样:`flowocr.extensions`(`match(document, context, options) -> dict`,
8
+ `cluster_patch` 是可选的另一半,由 `build_tracks --matcher` 在聚类那一步用)。
9
+ `--ctx k=v` 进 `context`(`tracks_path` / `tag` / `argv` 自动带上),`--opt k=v` 进 `options`,值都是字符串、匹配器自己解释。
10
+
11
+ 产物拿去阶段 3:`python -m flowocr.output.render <out.json> --preset matched_srt|default|dev`
12
+ (内置匹配器 `gametext` 写的是 `flowocr-matched/2`;用户匹配器写自己的格式,配自己的预设)。
13
+ ⚠ 换预设不重跑匹配、换匹配器不重跑提取——这就是三个阶段各写一份产物的意义。
14
+
15
+ `scriptmatch` 的 CLI 仍在(它带着评测口径的报表和一堆 ASS 旋钮);这里是**通用**那条,
16
+ 内置和用户实现走同一条路。
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import argparse
21
+ import json
22
+ import sys
23
+ from pathlib import Path
24
+
25
+ from flowocr import extensions
26
+ from flowocr.artifacts import tracksio
27
+
28
+
29
+ def kv(pairs: list[str], what: str) -> dict:
30
+ out = {}
31
+ for x in pairs:
32
+ if "=" not in x:
33
+ raise SystemExit(f"{what} 要 K=V:{x}")
34
+ k, v = x.split("=", 1)
35
+ out[k] = v
36
+ return out
37
+
38
+
39
+ def main(argv: list[str] | None = None) -> int:
40
+ ap = argparse.ArgumentParser(description="跑一个匹配器(阶段 2)",
41
+ formatter_class=argparse.RawDescriptionHelpFormatter, epilog=__doc__)
42
+ ap.add_argument("tracks", help="*-tracks.json(阶段 2 的聚类产物)")
43
+ ap.add_argument("--matcher", required=True,
44
+ help="内置名(%s)/ 模块名 / .py 路径" % "、".join(extensions.builtin_names("matcher")))
45
+ ap.add_argument("--out", required=True, help="匹配产物 JSON")
46
+ ap.add_argument("--ctx", action="append", default=[], metavar="K=V",
47
+ help="给匹配器的上下文,可多次(内置 gametext 要 `ref=<gamescript 的剧本 JSON>`)")
48
+ ap.add_argument("--opt", action="append", default=[], metavar="K=V", help="给匹配器的选项,可多次")
49
+ a = ap.parse_args(argv)
50
+
51
+ src = Path(a.tracks)
52
+ doc = tracksio.load(src)
53
+ context = {"tracks_path": str(src), "tag": src.stem.removesuffix("-tracks"),
54
+ "argv": list(sys.argv[1:] if argv is None else argv), **kv(a.ctx, "--ctx")}
55
+ matcher = extensions.load(a.matcher, "matcher")
56
+ out_doc = matcher.match(doc, context, kv(a.opt, "--opt"))
57
+ if not isinstance(out_doc, dict):
58
+ raise SystemExit(f"匹配器 `{a.matcher}` 的 match() 要返回 dict,给的是 {type(out_doc).__name__}")
59
+ p = Path(a.out)
60
+ p.parent.mkdir(parents=True, exist_ok=True)
61
+ p.write_text(json.dumps(out_doc, ensure_ascii=False, indent=1), encoding="utf-8")
62
+ print(f"-> {p}(schema {out_doc.get('schema', '<无>')},{len(out_doc.get('cues') or ())} 条)")
63
+ return 0
64
+
65
+
66
+ if __name__ == "__main__":
67
+ raise SystemExit(main())
@@ -0,0 +1,5 @@
1
+ """内置匹配器。一个模块一个匹配器,入口形状见 `flowocr.extensions`(`match` 必有、`cluster_patch` 可选)。
2
+
3
+ 现在只有 `gametext`(游戏文本库:原神 / 星铁 / 绝区零)。用户自己写的匹配器不必放进来——
4
+ `extensions.load("<路径或模块名>", "matcher")` 同样能加载。
5
+ """
@@ -0,0 +1,70 @@
1
+ """内置匹配器:**游戏文本库**(原神 / 星铁 / 绝区零的文本包,`gtd-bundle/1`,由 `gamescript` 读成剧本;game-text-corpus 报告)。
2
+
3
+ 两个入口(`flowocr.extensions`):
4
+
5
+ * `cluster_patch(context, options)`:按游戏给默认聚类器(build_tracks)的参数覆盖。表就是 `PATCHES`
6
+ (原 `tools/game_patches.py`,2026-09-22 搬进来;那个文件现在是把字典翻成 CLI 参数的薄适配)。
7
+ `context` 给 `game`(`genshin` / `starrail` / `zzz`)或 `tag`(素材标签,前缀认游戏,同 `gamescript.TAG_GAME`)。
8
+ * `match(document, context, options)`:`document` 是 `*-tracks.json` 的 dict;`context["ref"]` 是 `gamescript` 产的剧本 JSON 路径,
9
+ `context["tracks_path"]` 是那份 tracks 的路径(记进 provenance;`feed == "main"` 时从它取主轨 SRT)。
10
+ 返回 `*-matched.json` 的 dict(`scriptmatch.SCHEMA`)。判据和 CLI `scriptmatch.py` 是同一份 `run_match`。
11
+
12
+ `options` 的键和默认值同 `scriptmatch.py` 的旋钮:`feed`(nonoise)、`feed_panel`(keep)、`fuzzy`(0.75)、
13
+ `merge_gap`(`scriptmatch.MERGE_GAP`)、`overlay`(True)、`readable_textmap` / `obs`(None)。
14
+ """
15
+ from __future__ import annotations
16
+
17
+ from pathlib import Path
18
+
19
+ from flowocr.analyze import game_align as GA
20
+ from flowocr.analyze import gamescript as GS
21
+ from flowocr.analyze import scriptmatch as SM
22
+
23
+ PATCHES: dict[str, dict] = {
24
+ # 星铁:窗内并区的时间门。遗器故事的阅读面板和字幕带**从不同屏**,却因为空间上挨着被并查集
25
+ # 串成一块(game-text-corpus 报告);这道门要求"同屏过 or 时段交错 ≥2 次"才连边。
26
+ # 整场 hsr +34 条命中,cue 1,645 → 1,450、对不上剧本 344 → 269、重复认领 811 → 654,两段切片 ±0。
27
+ # **原神 / 绝区零不开**:那两款上是 −49 / −58 / −30,病在"主轨只能挑一个区域"接不住被分开的带。
28
+ "starrail": {"region_time_gate": 2},
29
+ }
30
+ """游戏 -> build_tracks 的参数覆盖。键是参数名(`region_time_gate`),不是 CLI 旗标。"""
31
+
32
+
33
+ def _flag(v) -> bool:
34
+ """`options` 的值可能是命令行来的**字符串**(`--opt overlay=0`)——`bool("0")` 是 True,
35
+ 静默把"关掉"读成"开着"。这里按字面认(2026-09-22)。"""
36
+ if isinstance(v, str):
37
+ return v.strip().lower() not in ("", "0", "false", "no", "off")
38
+ return bool(v)
39
+
40
+
41
+ def game_of(context: dict) -> str | None:
42
+ """`context["game"]` 优先;否则按素材标签前缀认(`gi-s1` → genshin),认不出就 None。"""
43
+ game = context.get("game")
44
+ if game:
45
+ return str(game)
46
+ tag = str(context.get("tag") or "")
47
+ return GS.TAG_GAME.get(tag.split("-")[0].rstrip("0123456789")) if tag else None
48
+
49
+
50
+ def cluster_patch(context: dict, options: dict | None = None) -> dict:
51
+ """这款游戏要给默认聚类器的参数覆盖;没有就是空字典(不改默认)。"""
52
+ return dict(PATCHES.get(game_of(context) or "", {}))
53
+
54
+
55
+ def match(document: dict, context: dict, options: dict | None = None) -> dict:
56
+ """见文件头。`document` 不动;返回新的 matched dict。"""
57
+ o = dict(options or {})
58
+ ref_path = context.get("ref")
59
+ if not ref_path:
60
+ raise ValueError("gametext 匹配器要 context['ref'](gamescript 产的剧本 JSON 路径)")
61
+ ref = GA.load_ref(Path(ref_path))
62
+ tracks_path = str(context.get("tracks_path") or "")
63
+ params = {"feed": o.get("feed", "nonoise"), "feed_panel": o.get("feed_panel", "keep"),
64
+ "fuzzy": float(o.get("fuzzy", 0.75)), "merge_gap": float(o.get("merge_gap", SM.MERGE_GAP))}
65
+ res = SM.run_match(ref, tracks_path, params["feed"], params["feed_panel"], params["fuzzy"], params["merge_gap"],
66
+ overlay=_flag(o.get("overlay", True)), readable_textmap=o.get("readable_textmap"),
67
+ obs=o.get("obs"), doc=document)
68
+ st = SM.match_stats(ref, res["recs"], res["suspect_raw"])
69
+ return SM.matched_doc(ref, res["recs"], res["n_raw"], st, str(ref_path), res["subs_path"], res["tp"],
70
+ list(context.get("argv") or []), params, res["items"], res["ov_stats"])
@@ -0,0 +1,178 @@
1
+ """把**名牌区域**并进主轨,产出一条给匹配器用的 SRT。
2
+
3
+ 来历(methodology-audit-2 报告):input1 上四组对照里,只有"并名牌"这一组
4
+ 能同时拉动两个指标——命中 5,814 → **5,880(+66)**、疑似真漏 133 → **94**,
5
+ 而且**纯标点行的漏条率 97.6% → 52.4%**。owner 说的"上下文 momentum"就是这个:
6
+ 名牌把"去掉标点一个字不剩"的 cue 撑成一条有边界、有判别力的条目。
7
+
8
+ 那次挑名牌用的是**角色名表**(探针能用、修法不能用)。这个工具消费的是
9
+ `nameplate.py` 的几何判据,**一个字都不看**,所以换素材也能跑。
10
+
11
+ 判据的准确性(`probe_nameplate_geom.py` 拿名表离线量的,五部整片,
12
+ 产物用 `--slot-span-ratio 2.0` 建):精确率 55–72%、召回 74–90%。
13
+ **精确率不到 100% 是这条路的成本**:挑中的 run 里有三成不是名字,
14
+ 它们会作为多余条目进匹配器——代价只能靠端到端的命中/漏条量,不能靠"看着像"。
15
+
16
+ 用法:
17
+ python -m flowocr.analyze.merge_nameplate out/yuka-f1-sr20/f1-tracks.json \
18
+ --out tmp/match/ours1-merged.srt
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import argparse
23
+ import json
24
+ import sys
25
+ from pathlib import Path
26
+
27
+ from flowocr.analyze import build_tracks as bt # noqa: E402 (只借 git_head,别再抄一份)
28
+ from flowocr.artifacts import evalkit # noqa: E402
29
+ from flowocr.analyze import nameplate # noqa: E402
30
+ from flowocr.artifacts import tracksio # noqa: E402
31
+ from flowocr.artifacts import srtio # noqa: E402
32
+
33
+
34
+ def region_srt(outdir: Path, doc: dict, r: dict) -> Path:
35
+ """区域的 SRT 文件名**从产物里读**,不用 tag+index+label 拼。
36
+
37
+ 拼出来的名字和磁盘上的名字是两回事:label 一变就是新文件名,
38
+ 2026-09-07 那次整部 input4 量错就是拼/猜文件名的同一形状。
39
+ """
40
+ for tr in doc["tracks"]:
41
+ if tr["kind"] == "region" and tr["region"] == r["i"]:
42
+ return outdir / tr["srt"]
43
+ raise SystemExit(f"tracks.json 里没有 region{r['i']:02d} 这条轨")
44
+
45
+
46
+ def main() -> int:
47
+ ap = argparse.ArgumentParser(description=__doc__,
48
+ formatter_class=argparse.RawDescriptionHelpFormatter)
49
+ ap.add_argument("tracks", help="build_tracks 产的 *-tracks.json")
50
+ ap.add_argument("--out", required=True, help="并好的 SRT 写到哪")
51
+ ap.add_argument("--no-band-repair", dest="band_repair", action="store_false",
52
+ help="**默认开**:把和主轨同一条带的碎片也并回来(cy 差 ≤--band-dy、"
53
+ "字高比在 --band-h 之间)。收紧 --slot-span-ratio 会把主字幕带"
54
+ "切下来一块——f5 的 region07 里有 328 种内容字 ≥6 的真台词,"
55
+ "命中因此掉了 261;并回来是 2,534 → 2,835。"
56
+ "五部里只有 f2/f5 真有碎片可并,另外三部开不开都一样,"
57
+ "**所以默认开是安全的**。这个开关是留给对照实验的")
58
+ ap.add_argument("--paste", action="store_true",
59
+ help="**把名牌贴进时间重叠最大的那条正文 cue 的首行**,"
60
+ # `%%`:argparse 对 help 串做 % 格式化,单个 `%` 会让
61
+ # `--help` 抛 ValueError(这一处一直是坏的,2026-09-09 修)
62
+ "而不是让它单独成条(`nameplate.attach`,贴法 99.8%% 准)。"
63
+ "存在的理由(methodology-audit-3 报告):"
64
+ "单独成条时,只有名字的 cue 在匹配器眼里就是一条短台词,"
65
+ "会去抢剧本里那些含名字的行——五部实测**超额认领**"
66
+ "(输出里比整部剧本多出来的条目——**报警,不是已确认的错**)"
67
+ "从 81/138/140/68/39 "
68
+ "涨到 546/540/360/257/126。贴进正文则不产生新条目。"
69
+ "同带碎片不受影响(那是真台词,照旧独立成条)")
70
+ ap.add_argument("--band-dy", type=float, default=0.02)
71
+ ap.add_argument("--band-h", type=float, nargs=2, default=[0.75, 1.35])
72
+ nameplate.add_args(ap)
73
+ ap.add_argument("--np-source", default="track", choices=("track", "geom"),
74
+ help="名牌从哪来:`track` = build_tracks 打标投影的那条名牌轨"
75
+ "(默认,判据在 flowocr.analyze.uigate);`geom` = nameplate.py 的七条阈值挑区域"
76
+ "(五部里四部挑中 0 个,留作对照)")
77
+ a = ap.parse_args()
78
+
79
+ tp = Path(a.tracks)
80
+ tag = tp.name.removesuffix("-tracks.json")
81
+ d = tracksio.load(tp)
82
+ d["regions"] = tracksio.regions_with_runs(d)
83
+ regs = nameplate.region_stats(d)
84
+ evalkit.require_nonzero(len(regs), "有 run 的区域")
85
+ main_r = max(regs, key=lambda x: x["score"])
86
+ # **名牌从哪来**(2026-09-20 起默认走第一条):
87
+ # `track`:`build_tracks` 打好标、投影成的那条 `kind="nameplate"` 轨——
88
+ # 判据是 `tools/uigate.pick_nameplates`(框位 + 相对位置 + 时长 ≥ 台词)。
89
+ # `geom` :`nameplate.py` 的七条阈值挑**区域**——五部里**四部挑中 0 个**(ui-gate 计划),
90
+ # 而且病根在输入:三部的名牌 run 被巨型区域吞了,区域侧根本没得挑。
91
+ np_track = next((t for t in d["tracks"] if t.get("kind") == "nameplate"), None)
92
+ if a.np_source == "track" and np_track is None:
93
+ raise SystemExit("这份 tracks.json 里没有名牌轨——用新版 build_tracks 重建,"
94
+ "或者显式 `--np-source geom` 走旧的七条阈值")
95
+ picked = [] if a.np_source == "track" else nameplate.pick(regs, a)[1]
96
+ geo_picked = list(picked) # 几何判据挑的那些;下面 band_repair 还会往里加
97
+
98
+ main_srt = region_srt(tp.parent, d, main_r)
99
+ if not main_srt.exists():
100
+ raise SystemExit(f"主轨 SRT 不在:{main_srt}(tracks.json 和产物目录对不上?)")
101
+ cues = srtio.read_srt(main_srt, sort=True)
102
+ print(f"主轨 region{main_r['i']:02d}({main_r['label']}){len(cues)} 条")
103
+
104
+ if not picked and a.np_source == "geom":
105
+ print("**没挑中任何名牌区域**——七条阈值在这部素材上不成立,原样输出主轨。"
106
+ "⚠ 五部整片里**四部**是这个结果(ui-gate 计划):"
107
+ "病根多半在输入——名牌的 run 被巨型区域吞了。默认的 `--np-source track` 不吃这一套。")
108
+ if a.band_repair:
109
+ band = [r for r in regs
110
+ if r["i"] != main_r["i"] and r not in picked and r["n"] >= a.min_runs
111
+ and abs(r["cy"] - main_r["cy"]) <= a.band_dy
112
+ and a.band_h[0] <= r["h"] / main_r["h"] <= a.band_h[1]]
113
+ if band:
114
+ print(f" 同带碎片 {len(band)} 个(收紧闸门切下来的那些)")
115
+ picked = picked + band
116
+
117
+ extra, plates = [], []
118
+ if a.np_source == "track":
119
+ np_srt = tp.parent / np_track["srt"]
120
+ if not np_srt.exists():
121
+ raise SystemExit(f"名牌轨的 SRT 不在:{np_srt}")
122
+ plates = srtio.read_srt(np_srt, sort=True)
123
+ print(f" 名牌轨 {np_track['srt']}:{len(plates)} 条"
124
+ + ("" if a.paste else " ⚠ 没给 --paste,它们会独立成条——"
125
+ "**别这么喂给匹配器**(methodology-audit-3:会去抢含名字的剧本行)"))
126
+ if not a.paste:
127
+ extra, plates = plates, []
128
+ for r in picked:
129
+ p = region_srt(tp.parent, d, r)
130
+ if not p.exists():
131
+ print(f" ⚠ region{r['i']:02d} 的 SRT 不在({p.name}),跳过")
132
+ continue
133
+ cs = srtio.read_srt(p, sort=True)
134
+ # --paste 下,**几何判据挑中的名牌**贴进正文;同带碎片是真台词,照旧独立成条
135
+ (plates if (a.paste and r in geo_picked) else extra).extend(cs)
136
+ print(f" + region{r['i']:02d}({r['label']}){len(cs)} 条"
137
+ f" cy={r['cy']:.3f} 字高 {r['h']:.0f} 时长中位 {r['dur']:.1f}s"
138
+ + (" [贴进正文]" if a.paste and r in geo_picked else ""))
139
+
140
+ allc = sorted(cues + extra, key=lambda c: (c.start, c.end))
141
+ if plates:
142
+ plates.sort(key=lambda c: c.start)
143
+ blocks, tagged = nameplate.attach(allc, plates)
144
+ print(f" 名牌 {len(plates)} 条按**时间重叠最大**贴进正文:"
145
+ f"{evalkit.denom(tagged, len(blocks))} 的 cue 贴上了名字"
146
+ f"(贴法 99.8% 准,speaker-name-line 报告)")
147
+ else:
148
+ blocks = [(int(round(c.start * 1e6)), int(round(c.end * 1e6)), list(c.lines))
149
+ for c in allc]
150
+ n = srtio.write_srt_blocks(a.out, blocks)
151
+ # **写完自己数一遍**:这个项目在"条数口径"上栽过一次 30 小时的错
152
+ got = srtio.count_cues(a.out)
153
+ if got != n:
154
+ raise SystemExit(f"写了 {n} 条但文件里数出 {got} 条——不要拿这份数据继续算")
155
+ print(f"-> {a.out}:{len(cues)} + {len(extra)} = **{n} 条**"
156
+ + (f"(另有 {len(plates)} 条名牌贴进了正文,不单独成条)" if plates else "")
157
+ + f"(`grep -c -- '-->' {a.out}` 数得到同一个数)")
158
+
159
+ # **这一级也要能追溯到是哪一版代码、哪些参数产的**(methodology-audit-3 报告):
160
+ # 它和 build_tracks 一样是"产出匹配器输入"的一级代码,可之前既不写 provenance、
161
+ # 也不落日志——挑中了哪几个区域、有没有打出"没挑中任何名牌区域",只在终端上闪过。
162
+ prov = tp.parent / f"{tag}-nameplate.json"
163
+ prov.write_text(json.dumps({
164
+ "tool": "merge_nameplate.py", "git_head": bt.git_head(),
165
+ "argv": sys.argv[1:], "tracks": str(tp), "out": str(a.out),
166
+ "main_region": main_r["i"], "main_cues": len(cues),
167
+ "picked": [{"i": r["i"], "label": r["label"], "cy": round(r["cy"], 3),
168
+ "h": r["h"], "dur": r["dur"], "n_runs": r["n"],
169
+ "band_repair": r not in geo_picked} for r in picked],
170
+ "extra_cues": len(extra), "cues": n,
171
+ "args": {k: v for k, v in sorted(vars(a).items())},
172
+ }, ensure_ascii=False, indent=1), encoding="utf-8")
173
+ print(f" 版本与选区 -> {prov.name}")
174
+ return 0
175
+
176
+
177
+ if __name__ == "__main__":
178
+ raise SystemExit(main())