pgn-postmortem 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,400 @@
1
+ """Reading PGN collections: find the files, read every game in them, keep the
2
+ player's games once each, and strip what the source attached to the moves.
3
+
4
+ Stripping keeps the headers and the mainline moves only. Every comment
5
+ (a source ``[%eval]`` included), variation and NAG is dropped: the library
6
+ re-analyzes every game itself, and old collections carry stale engine notes
7
+ (CLAUDE.md, *decided*). The headers that only describe such notes
8
+ (``DROPPED_HEADERS``) go too, and so does ``PostmortemAnalysis``, the marker
9
+ the analysis step writes: a game read back from the library's own output is
10
+ a stripped game again, to be analyzed again wherever it is written. One
11
+ header of our own is added: ``PostmortemId``, the game's content id.
12
+
13
+ The one exception is ``Collection.read(..., keep_analysis=True)``, which the
14
+ site uses (``pgn_postmortem.site``): a game carrying ``PostmortemAnalysis`` is
15
+ then kept as the analysis step wrote it, with its ``[%eval]`` comments, NAGs
16
+ and engine lines. A game without that header is still stripped, so a
17
+ source's own ``[%eval]`` is never trusted (ROADMAP.md, F-1, *out of scope*).
18
+
19
+ Games are identified by their content, not by where they were found, so the
20
+ same game in two files is kept once, and a game read back from the library's
21
+ own output has the same id as before it was analyzed.
22
+
23
+ The duplicate rule, exactly (``game_id``; the owner's decision of
24
+ 2026-09-24, for every game length): a game's identity is its **start
25
+ position, moves, result and date**. The id is the first 10 hex digits of the
26
+ SHA-1 of:
27
+
28
+ - the start position as a normalized FEN: python-chess's rendering of the
29
+ position the ``FEN`` header sets up, and empty for the standard start, so a
30
+ ``FEN`` header that spells out the standard start is the same as none;
31
+ - the ``Result`` header as written (``*`` when there is none);
32
+ - the ``Date`` header as written (``????.??.??``, python-chess's placeholder,
33
+ when there is none);
34
+ - the mainline moves in UCI.
35
+
36
+ The players' names are not part of it, so a game exported under two of the
37
+ player's names or aliases is one game. Two games with the same id are one
38
+ game, and the first one read is kept (inputs in the order given, files sorted
39
+ within a directory or a pattern, games in file order). What follows from it:
40
+
41
+ - copies whose ``Date`` headers differ in any way are kept twice: one with no
42
+ date or a partial date (``2019.??.??``) and one with the full date, or dates
43
+ written differently (``2019.03.14`` and ``2019.3.14``);
44
+ - copies whose ``Result`` headers differ (``1-0`` and ``*``), or that differ
45
+ in any move, are kept twice;
46
+ - two different games with the same start, moves, result and date are merged:
47
+ a short trap, or the same opening line agreed drawn, played twice on the
48
+ same day against different opponents keeps only the first game.
49
+ """
50
+
51
+ from __future__ import annotations
52
+
53
+ import glob
54
+ import hashlib
55
+ import io
56
+ import re
57
+ from collections.abc import Iterable, Iterator
58
+ from dataclasses import dataclass, field
59
+ from pathlib import Path
60
+
61
+ import chess
62
+ import chess.pgn
63
+
64
+ ID_HEADER = "PostmortemId"
65
+ ANALYSIS_HEADER = "PostmortemAnalysis" # written only by the analysis step (pgn_postmortem.analysis)
66
+ DROPPED_HEADERS = {"Annotator", "PlyCount", "CurrentPosition", ANALYSIS_HEADER}
67
+
68
+ GLOB_CHARS = set("*?[")
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class CollectedGame:
73
+ """One game of a collection, already stripped."""
74
+
75
+ id: str
76
+ game: chess.pgn.Game
77
+ origin: str # the file it was first found in, and its position there, e.g. "club/2019.pgn#2"
78
+
79
+ @property
80
+ def filename(self) -> str:
81
+ """``<date>-<id>.pgn``, the date zero-padded (``2019-03-14``), so a
82
+ plain directory listing is chronological (see ``file_stem``)."""
83
+ return f"{file_stem(self.game, self.id)}.pgn"
84
+
85
+
86
+ @dataclass
87
+ class ReadReport:
88
+ files: int = 0
89
+ read: int = 0
90
+ unparseable: int = 0
91
+ empty: int = 0
92
+ not_player: int = 0
93
+ duplicates: int = 0
94
+ kept: int = 0
95
+ warnings: list[str] = field(default_factory=list)
96
+
97
+ def summary(self) -> str:
98
+ return (
99
+ f"Read {self.read} game(s) from {self.files} file(s): {self.not_player} not the player's, "
100
+ f"{self.duplicates} duplicate(s), {self.empty} without moves, {self.unparseable} unparseable; "
101
+ f"kept {self.kept}."
102
+ )
103
+
104
+
105
+ def find_pgn_files(inputs: Iterable[str | Path]) -> list[Path]:
106
+ """The PGN files named by ``inputs``, in order and each once.
107
+
108
+ Each input is a file (taken whatever its extension), a directory (every
109
+ ``*.pgn`` below it, recursively) or a glob pattern (``**`` crosses
110
+ directories). An existing path is always taken as that path, even when
111
+ its name contains ``*``, ``?`` or ``[``; only an input that does not
112
+ exist is expanded as a pattern. A path that does not exist, or a pattern
113
+ that matches nothing, raises ``FileNotFoundError``: a typo should not read
114
+ as an empty collection.
115
+ """
116
+ found: list[Path] = []
117
+ seen: set[Path] = set()
118
+
119
+ def add(path: Path) -> None:
120
+ key = path.resolve()
121
+ if key not in seen:
122
+ seen.add(key)
123
+ found.append(path)
124
+
125
+ def add_dir(path: Path) -> None:
126
+ for child in sorted(p for p in path.rglob("*") if p.is_file() and p.suffix.lower() == ".pgn"):
127
+ add(child)
128
+
129
+ for item in inputs:
130
+ text = str(item)
131
+ path = Path(text).expanduser()
132
+ if GLOB_CHARS & set(text) and not path.exists():
133
+ matches = sorted(Path(p) for p in glob.glob(text, recursive=True))
134
+ if not matches:
135
+ raise FileNotFoundError(f"no file matches {text}")
136
+ for match in matches:
137
+ if match.is_dir():
138
+ add_dir(match)
139
+ else:
140
+ add(match)
141
+ continue
142
+ if path.is_dir():
143
+ add_dir(path)
144
+ elif path.is_file():
145
+ add(path)
146
+ else:
147
+ raise FileNotFoundError(f"no such file or directory: {text}")
148
+ return found
149
+
150
+
151
+ def read_text(path: Path) -> str:
152
+ """The file's text, decoded once for the whole file: UTF-8 (with or
153
+ without a BOM) if the whole file is valid UTF-8; else Windows-1252, what
154
+ old Windows-era collections are usually in (a superset of Latin-1's
155
+ letters, with curly quotes and dashes); else, for the five bytes
156
+ Windows-1252 leaves undefined, Latin-1, which decodes any byte, so a file
157
+ is never rejected for its encoding.
158
+
159
+ The decision is per file, not per game: a file that concatenates UTF-8
160
+ and Windows-1252 games is not valid UTF-8, so its UTF-8 games come out
161
+ as mojibake (``Jürgen``). Convert such a file to UTF-8 first."""
162
+ data = path.read_bytes()
163
+ for encoding in ("utf-8-sig", "cp1252"):
164
+ try:
165
+ return data.decode(encoding)
166
+ except UnicodeDecodeError:
167
+ pass
168
+ return data.decode("latin-1")
169
+
170
+
171
+ def iter_games(path: Path, report: ReadReport) -> Iterator[tuple[str, chess.pgn.Game]]:
172
+ """Every game in a (possibly multi-game) PGN file. A game python-chess
173
+ cannot parse cleanly is reported and skipped, not half-read."""
174
+ handle = io.StringIO(read_text(path))
175
+ index = 0
176
+ while (game := chess.pgn.read_game(handle)) is not None:
177
+ index += 1
178
+ report.read += 1
179
+ origin = f"{path}#{index}"
180
+ if game.errors:
181
+ report.unparseable += 1
182
+ report.warnings.append(f"skipping unparseable game {origin}: {game.errors[0]}")
183
+ continue
184
+ yield origin, game
185
+
186
+
187
+ def game_id(game: chess.pgn.Game) -> str:
188
+ """A short id computed from the game's start position, moves, result and
189
+ date (the exact rule and what follows from it are in the module
190
+ docstring). Comments, variations, NAGs, the players' names and the
191
+ headers stripping drops or adds do not change it."""
192
+ start = game.board().fen()
193
+ parts = [
194
+ "" if start == chess.STARTING_FEN else start,
195
+ game.headers.get("Result", "*"),
196
+ game.headers.get("Date", ""),
197
+ " ".join(move.uci() for move in game.mainline_moves()),
198
+ ]
199
+ return hashlib.sha1("|".join(parts).encode()).hexdigest()[:10]
200
+
201
+
202
+ def is_number(text: str) -> bool:
203
+ """Whether ``text`` is ASCII digits only. ``str.isdigit`` also accepts
204
+ digits such as ``²`` that ``int`` rejects and a file name should not carry."""
205
+ return re.fullmatch(r"[0-9]+", text) is not None
206
+
207
+
208
+ def date_fields(date: str) -> tuple[str, str, str]:
209
+ """The year, month and day of a PGN ``Date`` as written (``""`` when missing)."""
210
+ y, m, d = (date.split(".") + ["", "", ""])[:3]
211
+ return y, m, d
212
+
213
+
214
+ def file_stem(game: chess.pgn.Game, gid: str) -> str:
215
+ """``<yyyy>-<mm>-<dd>-<id>`` from the ``Date`` header: month and day
216
+ zero-padded to two digits (``2019.3.14`` gives ``2019-03-14``), an unknown
217
+ month or day as ``00``, and ``undated-<id>`` when the year is unknown. A
218
+ year, month or day counts as known only when it is ASCII digits
219
+ (``is_number``), so ``²019.01.01`` is undated. Only the file name is
220
+ normalized; the id keeps the header as written."""
221
+ y, m, d = date_fields(game.headers.get("Date", ""))
222
+ if not is_number(y):
223
+ return f"undated-{gid}"
224
+ return f"{y}-{m.zfill(2) if is_number(m) else '00'}-{d.zfill(2) if is_number(d) else '00'}-{gid}"
225
+
226
+
227
+ def strip_game(game: chess.pgn.Game, gid: str) -> chess.pgn.Game:
228
+ """A copy of ``game`` with its headers and mainline moves only."""
229
+ out = chess.pgn.Game()
230
+ for key, value in game.headers.items():
231
+ if key not in DROPPED_HEADERS:
232
+ out.headers[key] = value
233
+ if "FEN" in game.headers:
234
+ out.setup(game.board())
235
+ out.headers[ID_HEADER] = gid
236
+
237
+ node: chess.pgn.GameNode = out
238
+ for move in game.mainline_moves():
239
+ node = node.add_variation(move)
240
+ return out
241
+
242
+
243
+ def analyzed_id(path: Path) -> str | None:
244
+ """The ``PostmortemId`` of the game in ``path`` if the analysis step
245
+ wrote it (its first game carries the ``PostmortemAnalysis`` marker), else
246
+ None: a stripped game, or any other file."""
247
+ headers = chess.pgn.read_headers(io.StringIO(read_text(path)))
248
+ if headers is None or ANALYSIS_HEADER not in headers:
249
+ return None
250
+ return headers.get(ID_HEADER)
251
+
252
+
253
+ def analyzed_ids(out_dir: Path) -> set[str]:
254
+ """The ids of the games already analyzed into ``out_dir``: the
255
+ ``PostmortemId`` of each file there whose first game carries the
256
+ ``PostmortemAnalysis`` marker, whatever the file is called. A file
257
+ without it (a game that was only stripped, or anything else) does not
258
+ count."""
259
+ return {gid for path in out_dir.glob("*.pgn") if (gid := analyzed_id(path))}
260
+
261
+
262
+ def format_game(game: chess.pgn.Game) -> str:
263
+ """PGN text wrapped at 80 columns, the usual PGN convention."""
264
+ exporter = chess.pgn.StringExporter(columns=80)
265
+ return game.accept(exporter) + "\n"
266
+
267
+
268
+ def keep_analyzed(game: chess.pgn.Game, gid: str) -> chess.pgn.Game:
269
+ """``game`` as the analysis step wrote it, with its ``PostmortemId``."""
270
+ game.headers[ID_HEADER] = gid
271
+ return game
272
+
273
+
274
+ def player_names(player: str | None, aliases: Iterable[str]) -> set[str]:
275
+ return {name.strip().casefold() for name in [player or "", *aliases] if name and name.strip()}
276
+
277
+
278
+ class Collection:
279
+ """The games read from one or more PGN collections: the player's games
280
+ only, each once, stripped. ``report`` says what was left out and why.
281
+ ``player`` and ``aliases`` are the names the games were read with (none
282
+ for a collection made directly from its games); the site's quiz page
283
+ uses them (``build_site``)."""
284
+
285
+ def __init__(
286
+ self,
287
+ games: list[CollectedGame],
288
+ report: ReadReport | None = None,
289
+ *,
290
+ player: str | None = None,
291
+ aliases: Iterable[str] = (),
292
+ ):
293
+ self.games = games
294
+ self.report = report or ReadReport(kept=len(games))
295
+ self.player = player
296
+ self.aliases = tuple(aliases)
297
+
298
+ @classmethod
299
+ def read(
300
+ cls,
301
+ inputs: str | Path | Iterable[str | Path],
302
+ player: str | None = None,
303
+ aliases: Iterable[str] = (),
304
+ keep_analysis: bool = False,
305
+ ) -> Collection:
306
+ """Read ``inputs`` (files, directories, globs; see ``find_pgn_files``).
307
+
308
+ With a ``player`` or ``aliases``, only games where White or Black is
309
+ one of those names (compared without regard to case or surrounding
310
+ spaces) are kept; with neither, every game is.
311
+
312
+ With ``keep_analysis``, a game carrying the analysis step's
313
+ ``PostmortemAnalysis`` header keeps its analysis instead of being
314
+ stripped, and when a game is found both analyzed and not, the
315
+ analyzed copy is the one kept (in the place of the first copy read),
316
+ so a site can be built from the games and their analyses in any order.
317
+ """
318
+ if isinstance(inputs, str | Path):
319
+ inputs = [inputs]
320
+ aliases = tuple(aliases) # kept with the games, so read once
321
+ names = player_names(player, aliases)
322
+ report = ReadReport()
323
+ games: list[CollectedGame] = []
324
+ seen: dict[str, int] = {} # id -> index in games
325
+
326
+ for path in find_pgn_files(inputs):
327
+ report.files += 1
328
+ for origin, game in iter_games(path, report):
329
+ if names and not (
330
+ game.headers.get("White", "").strip().casefold() in names
331
+ or game.headers.get("Black", "").strip().casefold() in names
332
+ ):
333
+ report.not_player += 1
334
+ continue
335
+ if game.next() is None:
336
+ report.empty += 1
337
+ continue
338
+ gid = game_id(game)
339
+ analyzed = keep_analysis and ANALYSIS_HEADER in game.headers
340
+ if gid in seen:
341
+ report.duplicates += 1
342
+ first = seen[gid]
343
+ if analyzed and ANALYSIS_HEADER not in games[first].game.headers:
344
+ games[first] = CollectedGame(gid, keep_analyzed(game, gid), origin)
345
+ continue
346
+ seen[gid] = len(games)
347
+ kept = keep_analyzed(game, gid) if analyzed else strip_game(game, gid)
348
+ games.append(CollectedGame(gid, kept, origin))
349
+
350
+ report.kept = len(games)
351
+ return cls(games, report, player=player, aliases=aliases)
352
+
353
+ def __iter__(self) -> Iterator[CollectedGame]:
354
+ return iter(self.games)
355
+
356
+ def __len__(self) -> int:
357
+ return len(self.games)
358
+
359
+ def write(self, out_dir: str | Path) -> list[Path]:
360
+ """Write every game, stripped, to ``out_dir/<date>-<id>.pgn``, and
361
+ return the paths written.
362
+
363
+ A game already analyzed into ``out_dir`` is not written, so reading
364
+ new games into an analysis directory never throws away analysis
365
+ already done, nor adds a stripped copy next to it. "Already analyzed"
366
+ is the analysis step's own test (``analyzed_ids``): a file there whose
367
+ first game carries the ``PostmortemAnalysis`` marker and a
368
+ ``PostmortemId`` equal to this game's content id (``game_id``: start
369
+ position, moves, result and date). It goes by the id, not the file
370
+ name, so a game analyzed under an older name (before dates in names
371
+ were zero-padded, ``2019-3-14-<id>.pgn``) still counts. Any other
372
+ file at the game's name (a stripped copy, or something else) is
373
+ overwritten."""
374
+ out_dir = Path(out_dir)
375
+ out_dir.mkdir(parents=True, exist_ok=True)
376
+ done_ids = analyzed_ids(out_dir)
377
+ paths = []
378
+ for item in self.games:
379
+ if item.id in done_ids:
380
+ continue
381
+ path = out_dir / item.filename
382
+ path.write_text(format_game(item.game), encoding="utf-8")
383
+ paths.append(path)
384
+ return paths
385
+
386
+ def analyze(self, out_dir: str | Path, **options):
387
+ """Analyze the games with Stockfish into ``out_dir``; see
388
+ ``pgn_postmortem.analysis.analyze_games`` for the options."""
389
+ from pgn_postmortem.analysis import analyze_games
390
+
391
+ return analyze_games(self.games, out_dir, **options)
392
+
393
+ def build_site(self, out_dir: str | Path, **options):
394
+ """Write the static site for these games to ``out_dir``, with the quiz
395
+ page for the names they were read with (unless ``player`` or
396
+ ``aliases`` is given); see ``pgn_postmortem.site.build_site`` for the
397
+ options."""
398
+ from pgn_postmortem.site import build_site
399
+
400
+ return build_site(self, out_dir, **options)