cronos-extract 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. cronos_extract/Database.py +374 -0
  2. cronos_extract/Datafile.py +263 -0
  3. cronos_extract/Datamodel.py +301 -0
  4. cronos_extract/__init__.py +65 -0
  5. cronos_extract/_api/__init__.py +2 -0
  6. cronos_extract/_api/bank.py +439 -0
  7. cronos_extract/_api/crack.py +151 -0
  8. cronos_extract/_api/datafiles.py +120 -0
  9. cronos_extract/_api/diagnostics.py +115 -0
  10. cronos_extract/_api/errors.py +22 -0
  11. cronos_extract/_api/info.py +79 -0
  12. cronos_extract/_api/kod.py +54 -0
  13. cronos_extract/_api/values.py +209 -0
  14. cronos_extract/_cli/__init__.py +2 -0
  15. cronos_extract/_cli/crack.py +417 -0
  16. cronos_extract/_cli/csv_out.py +141 -0
  17. cronos_extract/_cli/export.py +287 -0
  18. cronos_extract/_cli/inspect.py +250 -0
  19. cronos_extract/_cli/jsonl_out.py +83 -0
  20. cronos_extract/_cli/names.py +58 -0
  21. cronos_extract/_cli/options.py +73 -0
  22. cronos_extract/_cli/report.py +201 -0
  23. cronos_extract/_cli/sql_out.py +128 -0
  24. cronos_extract/_diagnostic.py +59 -0
  25. cronos_extract/_format/__init__.py +2 -0
  26. cronos_extract/_format/files.py +35 -0
  27. cronos_extract/_format/header.py +92 -0
  28. cronos_extract/_format/record.py +192 -0
  29. cronos_extract/_format/tad.py +103 -0
  30. cronos_extract/cli.py +144 -0
  31. cronos_extract/hexdump.py +122 -0
  32. cronos_extract/koddecoder.py +456 -0
  33. cronos_extract/kodump.py +87 -0
  34. cronos_extract/py.typed +0 -0
  35. cronos_extract/readers.py +111 -0
  36. cronos_extract/survey.py +144 -0
  37. cronos_extract-1.0.0.dist-info/METADATA +394 -0
  38. cronos_extract-1.0.0.dist-info/RECORD +41 -0
  39. cronos_extract-1.0.0.dist-info/WHEEL +4 -0
  40. cronos_extract-1.0.0.dist-info/entry_points.txt +3 -0
  41. cronos_extract-1.0.0.dist-info/licenses/LICENSE +22 -0
@@ -0,0 +1,374 @@
1
+ # ABOUTME: Database: opens the Cro*.dat/.tad file pairs found in a CronosPro database directory.
2
+ # ABOUTME: Decodes the database and table definitions from CroStru.
3
+ import argparse
4
+ import os
5
+ import re
6
+ import struct
7
+ from binascii import b2a_hex
8
+ from collections.abc import Collection
9
+ from contextlib import ExitStack
10
+ from typing import Self
11
+
12
+ from . import koddecoder
13
+ from ._diagnostic import STRU_FILE, Diagnostic, DiagnosticKind, Reporter, for_table_definition
14
+ from ._format.files import open_regular_file
15
+ from .Datafile import Datafile
16
+ from .Datamodel import TableDefinition, is_table_key, undecodable_table
17
+ from .hexdump import strescape, toout
18
+ from .koddecoder import KODcoding
19
+ from .readers import ByteReader, decode_cp1251
20
+
21
+ # Printed after a database definition error: a KOD that isn't the database's own decodes the definition as garbage.
22
+ KOD_HINT = (
23
+ "If the KOD used to read this database is not its own, the definition decodes as garbage; "
24
+ "cronos-extract crack strucrack can derive the database's KOD."
25
+ )
26
+
27
+ # The files a Database opens unless told otherwise.
28
+ ALL_FILES = ("Stru", "Index", "Bank", "Sys")
29
+
30
+
31
+ class StoppingReporter:
32
+ """
33
+ A Reporter that passes each Diagnostic to `report` and remembers the first exception `report` raises.
34
+
35
+ The readers catch broad exceptions, so they can swallow a callback's request to stop. Once an exception is
36
+ remembered, every later call raises it again without calling `report`, and raise_remembered raises it for the
37
+ caller to check after a reader returns or raises.
38
+ """
39
+
40
+ def __init__(self, report: Reporter) -> None:
41
+ self._report = report
42
+ self._error: BaseException | None = None
43
+
44
+ def __call__(self, diagnostic: Diagnostic) -> None:
45
+ self.raise_remembered()
46
+ try:
47
+ self._report(diagnostic)
48
+ except BaseException as e:
49
+ self._error = e
50
+ raise
51
+
52
+ def raise_remembered(self) -> None:
53
+ """Raise the exception `report` raised, if it raised one."""
54
+ if self._error is not None:
55
+ raise self._error
56
+
57
+
58
+ class Database:
59
+ """represent the entire database, consisting of Stru, Index and Bank files"""
60
+
61
+ def __init__(
62
+ self,
63
+ dbdir: str,
64
+ compact: bool,
65
+ kod: KODcoding | None,
66
+ report: Reporter,
67
+ files: Collection[str] = ALL_FILES,
68
+ ) -> None:
69
+ """
70
+ `dbdir` is the directory containing the Cro*.dat and Cro*.tad files.
71
+ `compact` if set, the .tad file is not cached in memory, making dumps 15 % slower
72
+ `kod` is a KOD coder object, or None to read the records without KOD decoding.
73
+ `report` receives a Diagnostic for each problem that reading survives, such as a part of the database
74
+ definition that is not laid out as expected.
75
+ `files` names the components to open, from ALL_FILES; the others are None.
76
+ """
77
+ self.dbdir = dbdir
78
+ self.compact = compact
79
+ self.kod = kod
80
+ self.files = files
81
+ self.report = report
82
+
83
+ # Stru+Index+Bank for the components for most databases
84
+ self.stru = self.getfile("Stru")
85
+ self.index = self.getfile("Index")
86
+ self.bank = self.getfile("Bank")
87
+
88
+ # the Sys file resides in the "Program Files\Cronos" directory, and
89
+ # contains an index of all known databases.
90
+ self.sys = self.getfile("Sys")
91
+
92
+ @classmethod
93
+ def from_datafiles(
94
+ cls, dbdir: str, compact: bool, kod: KODcoding | None, stru: Datafile, bank: Datafile, report: Reporter
95
+ ) -> Self:
96
+ """
97
+ Make a Database of the CroStru and CroBank Datafiles `stru` and `bank`, which the caller has opened.
98
+ Closing the Database closes them.
99
+ """
100
+ db = cls(dbdir, compact, kod, report, files=())
101
+ db.stru = stru
102
+ db.bank = bank
103
+ return db
104
+
105
+ def close(self) -> None:
106
+ """
107
+ Close the files of every component of the database.
108
+ """
109
+ for datafile in (self.stru, self.index, self.bank, self.sys):
110
+ if datafile:
111
+ datafile.close()
112
+
113
+ def __enter__(self) -> Self:
114
+ return self
115
+
116
+ def __exit__(self, *exc_info: object) -> None:
117
+ self.close()
118
+
119
+ def getfile(self, name: str) -> Datafile | None:
120
+ """
121
+ Returns a Datafile object for `name`.
122
+ this function expects a `Cro<name>.dat` and a `Cro<name>.tad` file.
123
+ When no such files exist, only one of them does, or one cannot be opened or is not a regular file,
124
+ then None is returned.
125
+
126
+ A component not named in `files` is not opened, and None is returned.
127
+
128
+ `name` is matched case insensitively
129
+ """
130
+ if name not in self.files:
131
+ return None
132
+ try:
133
+ datname = self.getname(name, "dat")
134
+ tadname = self.getname(name, "tad")
135
+ if datname and tadname:
136
+ return self.opendatafile(name, datname, tadname)
137
+ except OSError:
138
+ return None
139
+ return None
140
+
141
+ def opendatafile(self, name: str, datname: str, tadname: str) -> Datafile:
142
+ """
143
+ Open a .dat/.tad pair as a Datafile, closing both files again if it can't be read.
144
+ """
145
+ with ExitStack() as stack:
146
+ dat = stack.enter_context(open_regular_file(datname))
147
+ tad = stack.enter_context(open_regular_file(tadname))
148
+ datafile = Datafile(name, dat, tad, self.compact, self.kod, self.report)
149
+ stack.pop_all()
150
+ return datafile
151
+
152
+ def getname(self, name: str, ext: str) -> str | None:
153
+ """
154
+ Get a case-insensitive filename match for 'name.ext'.
155
+ Returns None when no matching file was not found.
156
+ """
157
+ basename = f"Cro{name}.{ext}"
158
+ for fn in os.listdir(self.dbdir):
159
+ if basename.lower() == fn.lower():
160
+ return os.path.join(self.dbdir, fn)
161
+ return None
162
+
163
+ def dump(self, args: argparse.Namespace) -> None:
164
+ """
165
+ Calls the `dump` method on all database components.
166
+ """
167
+ if self.stru:
168
+ self.stru.dump(args)
169
+ if self.index:
170
+ self.index.dump(args)
171
+ if self.bank:
172
+ self.bank.dump(args)
173
+ if self.sys:
174
+ self.sys.dump(args)
175
+
176
+ def missing_stru_message(self) -> str:
177
+ """
178
+ Returns the message that explains that the database directory has no CroStru files.
179
+ """
180
+ return f"no CroStru.dat and CroStru.tad found in {self.dbdir}, which hold the table definitions"
181
+
182
+ def report_structure(self, message: str, record: int | None = None) -> None:
183
+ """
184
+ Report `message` as an unexpected_structure Diagnostic about CroStru, at `record` when it is given.
185
+ """
186
+ self.report(Diagnostic(DiagnosticKind.UNEXPECTED_STRUCTURE, message, file=STRU_FILE, record=record))
187
+
188
+ def decode_db_definition(self, data: bytes) -> dict[str, bytes]:
189
+ """
190
+ decode the 'bank' / database definition
191
+
192
+ Raises ValueError when a key stored by reference names a CroStru record that is not open, out of range,
193
+ deleted, or when the definition is cut off.
194
+ """
195
+ rd = ByteReader(data)
196
+
197
+ d: dict[str, bytes] = dict()
198
+ try:
199
+ while not rd.eof():
200
+ keyname = rd.readname()
201
+ if keyname in d:
202
+ self.report_structure(f"duplicate key: {keyname}", record=1)
203
+
204
+ index_or_length = rd.readdword()
205
+ if index_or_length >> 31:
206
+ d[keyname] = rd.readbytes(index_or_length & 0x7FFFFFFF)
207
+ else:
208
+ if self.stru is None:
209
+ raise ValueError(
210
+ f'key "{keyname}" refers to CroStru record {index_or_length}, but CroStru is not open'
211
+ )
212
+ if not 1 <= index_or_length <= self.stru.nrofrecords:
213
+ raise ValueError(
214
+ f'key "{keyname}" refers to CroStru record {index_or_length}, '
215
+ f"which CroStru does not hold ({self.stru.nrofrecords} records)"
216
+ )
217
+ refdata = self.stru.readrec(index_or_length)
218
+ if refdata is None:
219
+ raise ValueError(
220
+ f'key "{keyname}" refers to CroStru record {index_or_length}, which is deleted'
221
+ )
222
+ if refdata[:1] != b"\x04":
223
+ self.report_structure("expected refdata to start with 0x04", record=index_or_length)
224
+ d[keyname] = refdata[1:]
225
+ except EOFError as e:
226
+ raise ValueError(f"the database definition is cut off after {len(d)} keys") from e
227
+ return d
228
+
229
+ def dump_db_definition(self, args: argparse.Namespace, dbdict: dict[str, bytes]) -> None:
230
+ """
231
+ decode the 'bank' / database definition
232
+ """
233
+ for k, v in dbdict.items():
234
+ if re.search(b"[^\x0d\x0a\x09\x20-\x7e\xc0-\xff]", v):
235
+ print(f"{k:<20} - {toout(args, v)}")
236
+ else:
237
+ print(f'{k:<20} - "{strescape(v)}"')
238
+
239
+ def read_db_definition(self) -> dict[str, bytes]:
240
+ """
241
+ Read and decode the database definition from CroStru record 1.
242
+ Raises ValueError when CroStru is not open, has no record 1, when it is deleted, or when it can't be
243
+ decoded.
244
+ """
245
+ if self.stru is None:
246
+ raise ValueError("CroStru is not open, so it has no database definition")
247
+ if self.stru.nrofrecords < 1:
248
+ raise ValueError("CroStru holds no records, so it has no database definition")
249
+ dbinfo = self.stru.readrec(1)
250
+ if dbinfo is None:
251
+ raise ValueError("CroStru record 1, which holds the database definition, is deleted")
252
+ if dbinfo[:1] != b"\x03":
253
+ self.report_structure("expected dbinfo to start with 0x03", record=1)
254
+ return self.decode_db_definition(dbinfo[1:])
255
+
256
+ def dump_db_table_defs(self, args: argparse.Namespace) -> None:
257
+ """
258
+ decode the table defs from recid #1, which always has table-id #3
259
+ Note that I don't know if it is better to refer to this by recid, or by table-id.
260
+
261
+ other table-id's found in CroStru:
262
+ #4 -> large values referenced from tableid#3
263
+ """
264
+ dbdef = self.read_db_definition()
265
+ self.dump_db_definition(args, dbdef)
266
+
267
+ report = StoppingReporter(self.report)
268
+ for k, v in dbdef.items():
269
+ if is_table_key(k):
270
+ print(f"== {k} ==")
271
+ try:
272
+ tbdef = TableDefinition(
273
+ v, dbdef.get("BaseImage" + k[4:], b""), report=for_table_definition(report, k)
274
+ )
275
+ except Exception as e:
276
+ report.raise_remembered()
277
+ report(undecodable_table(k, e))
278
+ continue
279
+ report.raise_remembered()
280
+ tbdef.dump(args)
281
+ elif k == "NS1":
282
+ self.dump_ns1(v)
283
+
284
+ def dump_ns1(self, data: bytes) -> None:
285
+ if len(data) < 2:
286
+ self.report_structure("NS1 is unexpectedly short")
287
+ return
288
+ (
289
+ unk1,
290
+ sh,
291
+ ) = struct.unpack_from("<BB", data, 0)
292
+
293
+ # NS1 is encoded with the default KOD table,
294
+ # so we are not using stru.kod here.
295
+ ns1kod = koddecoder.new()
296
+ decoded_data = ns1kod.decode(sh, data[2:])
297
+
298
+ if len(decoded_data) < 12:
299
+ self.report_structure("NS1 is unexpectedly short")
300
+ return
301
+ (
302
+ serial,
303
+ unk2,
304
+ pwlen,
305
+ ) = struct.unpack_from("<LLL", decoded_data, 0)
306
+ password = decode_cp1251(decoded_data[12 : 12 + pwlen])
307
+
308
+ print(f"== NS1: ({unk1:02x},{sh:02x}) -> {serial:6d}, {unk2:d}, {pwlen:d}:'{password}'")
309
+
310
+ def recdump(self, args: argparse.Namespace) -> None:
311
+ """
312
+ Function for outputing record contents of the various .dat files.
313
+
314
+ This function is mostly useful for reverse-engineering the database format.
315
+ Raises ValueError naming the file when the chosen file is not open.
316
+ """
317
+ if args.index:
318
+ name = "Index"
319
+ elif args.sys:
320
+ name = "Sys"
321
+ elif args.stru:
322
+ name = "Stru"
323
+ else:
324
+ name = "Bank"
325
+ # Each component is kept in the attribute named after it: index, sys, stru and bank.
326
+ dbfile = getattr(self, name.lower())
327
+
328
+ if not dbfile:
329
+ raise ValueError(f"no Cro{name}.dat and Cro{name}.tad in {self.dbdir}")
330
+ nerr = 0
331
+ nr_recnone = 0
332
+ nr_recempty = 0
333
+ tabidxref = [0] * 256
334
+ bytexref = [0] * 256
335
+ for i in range(1, min(args.maxrecs, dbfile.nrofrecords) + 1):
336
+ try:
337
+ data = dbfile.readrec(i)
338
+ if args.find1d:
339
+ if data and (data.find(b"\x1d") > 0 or data.find(b"\x1b") > 0):
340
+ print(f"record with '1d': {i:d} -> {b2a_hex(data)}")
341
+ break
342
+
343
+ elif not args.stats:
344
+ if data is None:
345
+ print(f"{i:5d}: <deleted>")
346
+ else:
347
+ print(f"{i:5d}: {toout(args, data)}")
348
+ else:
349
+ if data is None:
350
+ nr_recnone += 1
351
+ elif not len(data):
352
+ nr_recempty += 1
353
+ else:
354
+ tabidxref[data[0]] += 1
355
+ for b in data[1:]:
356
+ bytexref[b] += 1
357
+ nerr = 0
358
+ except Exception as e:
359
+ print(f"{i:5d}: <{e}>")
360
+ if args.debug:
361
+ raise
362
+ nerr += 1
363
+ if nerr > 5:
364
+ break
365
+
366
+ if args.stats:
367
+ print(f"-- table-id stats --, {nr_recnone:d} * none, {nr_recempty:d} * empty")
368
+ for k, v in enumerate(tabidxref):
369
+ if v:
370
+ print(f"{v:5d} * {k:02x}")
371
+ print("-- byte stats --")
372
+ for k, v in enumerate(bytexref):
373
+ if v:
374
+ print(f"{v:5d} * {k:02x}")
@@ -0,0 +1,263 @@
1
+ # ABOUTME: Datafile: reads the records of one CronosPro .dat file through its .tad index.
2
+ # ABOUTME: Decodes each record through _format/record.py and dumps them, byte range by byte range, for inspect.
3
+ import argparse
4
+ import dataclasses
5
+ import io
6
+ from collections.abc import Iterator
7
+ from typing import BinaryIO
8
+
9
+ from ._diagnostic import Diagnostic, DiagnosticKind, Reporter
10
+ from ._format.header import read_dat_header, read_kod_check
11
+ from ._format.record import RecordParts, RecordSource, decode_record, decompress, is_compressed, read_stored
12
+ from ._format.tad import DELETED_LENGTH, TadEntry, tad_layout
13
+ from .hexdump import tohex, toout
14
+ from .koddecoder import KODcoding, select_kod
15
+
16
+
17
+ class Datafile:
18
+ """
19
+ Represent a single .dat with it's .tad index file.
20
+
21
+ `report` receives each problem that reading survives, as a Diagnostic.
22
+ """
23
+
24
+ def __init__(
25
+ self,
26
+ name: str,
27
+ dat: BinaryIO,
28
+ tad: BinaryIO,
29
+ compact: bool,
30
+ kod: KODcoding | None,
31
+ report: Reporter,
32
+ ) -> None:
33
+ self.report = report
34
+ self.name = name
35
+ self.dat = dat
36
+ self.tad = tad
37
+ self.compact = compact
38
+
39
+ self.readdathdr()
40
+ self.readtad()
41
+
42
+ self.dat.seek(0, io.SEEK_END)
43
+ self.datsize = self.dat.tell()
44
+
45
+ self.kod, problem = select_kod(self.header, kod, f"Cro{self.name}.dat")
46
+ if problem is not None:
47
+ self.report(problem)
48
+ self.source = RecordSource(self.name, self.readdata, self.datsize, self.blocksize, self.use64bit, self.kod)
49
+
50
+ def close(self) -> None:
51
+ """
52
+ Close the .dat and .tad files.
53
+ """
54
+ self.dat.close()
55
+ self.tad.close()
56
+
57
+ def readdathdr(self) -> None:
58
+ """
59
+ Read the .dat file header, and the KOD check bytes that follow it.
60
+ In a v3 file the 19 byte header is followed by 0xE9 random bytes, generated by
61
+ 'srand(time())' followed by 0xE9 times obfuscate(rand()). In a v4 file, bytes 19 to 255 are
62
+ the same in every Cro file of the database: a block KOD-encoded like record 0 with the database's
63
+ own KOD, whose first 8 plaintext bytes, the KOD check bytes, are zero.
64
+ """
65
+ header = read_dat_header(self.dat, where=f"Cro{self.name}.dat")
66
+ header = dataclasses.replace(header, kod_check=read_kod_check(self.dat))
67
+ layout = tad_layout(header.version)
68
+ if layout is None:
69
+ raise ValueError(
70
+ f"Cro{self.name}.dat is CronosPro version {header.version_text}, whose .tad index this release "
71
+ "cannot read"
72
+ )
73
+ self.header = header
74
+ self.layout = layout
75
+ self.hdrunk = header.unknown
76
+ self.version = header.version
77
+ self.encoding = header.encoding
78
+ self.blocksize = header.blocksize
79
+ self.use64bit = header.use64bit
80
+
81
+ # blocksize
82
+ # 0040 -> Bank
83
+ # 0400 -> Index or Sys
84
+ # 0200 -> Stru or Sys
85
+
86
+ # encoding
87
+ # bit0 = 'KOD encoded'
88
+ # bit1 = compressed
89
+
90
+ def readtad(self) -> None:
91
+ """
92
+ read and decode the .tad file.
93
+ """
94
+ self.tad.seek(0)
95
+ header_size = self.layout.header.size
96
+ hdrdata = self.tad.read(header_size)
97
+ if len(hdrdata) < header_size:
98
+ raise ValueError(f"Cro{self.name}.tad is shorter than its {header_size}-byte header")
99
+ self.nrdeleted, self.firstdeleted = self.layout.deleted_counts(hdrdata)
100
+
101
+ self.tadhdrlen = self.tad.tell()
102
+ self.tadentrysize = self.layout.entry.size
103
+ self.idxdata = b""
104
+ if self.compact:
105
+ self.tad.seek(0, io.SEEK_END)
106
+ else:
107
+ self.idxdata = self.tad.read()
108
+ self.tadsize = self.tad.tell() - self.tadhdrlen
109
+ self.nrofrecords = self.tadsize // self.tadentrysize
110
+ if self.tadsize % self.tadentrysize:
111
+ self.report(
112
+ Diagnostic(DiagnosticKind.UNEXPECTED_STRUCTURE, "leftover data in .tad", file=f"Cro{self.name}.dat")
113
+ )
114
+
115
+ def entry(self, index: int) -> TadEntry:
116
+ """
117
+ The .tad entry of the record at `index`, counted from 0. With `compact`, it is read from the .tad file
118
+ instead of the cached copy.
119
+
120
+ Raises ValueError naming the file and the index when it is not 0 to nrofrecords - 1.
121
+ """
122
+ if not 0 <= index < self.nrofrecords:
123
+ raise ValueError(
124
+ f"Cro{self.name}.tad has no entry {index}; its entries are numbered 0 to {self.nrofrecords - 1}"
125
+ )
126
+ if self.compact:
127
+ self.tad.seek(self.tadhdrlen + index * self.tadentrysize)
128
+ raw = self.tad.read(self.tadentrysize)
129
+ else:
130
+ start = index * self.tadentrysize
131
+ raw = self.idxdata[start : start + self.tadentrysize]
132
+ return self.layout.parse(raw)
133
+
134
+ def readdata(self, ofs: int, size: int) -> bytes:
135
+ """
136
+ Read raw data from the .dat file.
137
+
138
+ Returns b"" without seeking when `ofs` is outside 0..self.datsize: some filesystems raise OSError on a
139
+ seek far past the end of the file, where seeking within the file (or exactly to its end) does not.
140
+ """
141
+ if not 0 <= ofs <= self.datsize:
142
+ return b""
143
+ self.dat.seek(ofs)
144
+ return self.dat.read(size)
145
+
146
+ def read_record(self, recno: int) -> RecordParts | None:
147
+ """
148
+ Record `recno`, counted from 1, decoded and decompressed, or None when it is deleted.
149
+ Raises ValueError, naming the record and the file, when it is not in the file or cannot be decoded.
150
+ """
151
+ if not 1 <= recno <= self.nrofrecords:
152
+ raise ValueError(
153
+ f"Cro{self.name}.dat has no record {recno}; its records are numbered 1 to {self.nrofrecords}"
154
+ )
155
+ entry = self.entry(recno - 1)
156
+ if entry.deleted:
157
+ return None
158
+ return decode_record(self.source, recno, entry)
159
+
160
+ def readrec(self, recno: int) -> bytes | None:
161
+ """
162
+ Extract and decode a single record, or None when it is deleted.
163
+ Compressed data whose CRC-32 does not match is kept and reported through `report` as checksum_mismatch.
164
+ Raises ValueError when the record is not in the file or cannot be decoded.
165
+ """
166
+ parts = self.read_record(recno)
167
+ if parts is None:
168
+ return None
169
+ if parts.mismatched_chunks:
170
+ self.report(
171
+ Diagnostic(
172
+ DiagnosticKind.CHECKSUM_MISMATCH,
173
+ "compressed data whose checksum does not match; it is kept",
174
+ file=f"Cro{self.name}.dat",
175
+ record=recno,
176
+ )
177
+ )
178
+ return parts.data
179
+
180
+ def enumunreferenced(self, ranges: list[tuple[int, int, str]], filesize: int) -> Iterator[tuple[int, int]]:
181
+ """
182
+ From a list of used byte ranges and the filesize, enumerate the list of unused byte ranges
183
+ """
184
+ o = 0
185
+ for start, end, _desc in sorted(ranges):
186
+ if start > o:
187
+ yield o, start - o
188
+ o = end
189
+ if o < filesize:
190
+ yield o, filesize - o
191
+
192
+ def dump(self, args: argparse.Namespace) -> None:
193
+ """
194
+ Dump decodes all data referenced from the .tad file.
195
+ And optionally print out all unreferenced byte ranges in the .dat file.
196
+
197
+ This function is mostly useful for reverse-engineering the database format.
198
+
199
+ the `args` object controls how data is decoded.
200
+ """
201
+ print(
202
+ f"hdr: {self.name:<6} dat: {self.hdrunk:04x} {self.version} "
203
+ f"enc:{self.encoding:04x} bs:{self.blocksize:04x}, "
204
+ f"tad: {self.nrdeleted:08x} {self.firstdeleted:08x}"
205
+ )
206
+
207
+ ranges: list[tuple[int, int, str]] = [] # keep track of used bytes in the .dat file.
208
+
209
+ for i in range(self.nrofrecords):
210
+ entry = self.entry(i)
211
+ idx = i + 1
212
+ if args.maxrecs and i == args.maxrecs:
213
+ break
214
+ if entry.length == DELETED_LENGTH:
215
+ print(f"{idx:5d}: {entry.offset:08x} {entry.length:08x} {entry.checksum:08x}")
216
+ continue
217
+
218
+ # A deleted v4 entry keeps its data, so it is dumped like a live one and marked.
219
+ deleted = " <deleted>" if entry.deleted else ""
220
+ ofs, ln, flags, chk = entry.offset, entry.length, entry.flags, entry.checksum
221
+ ranges.append((ofs, ofs + ln, f"item #{i:d}"))
222
+ decflags = [" ", " "]
223
+ infostr = ""
224
+
225
+ try:
226
+ parts = read_stored(self.source, idx, entry, require_whole=False)
227
+ except ValueError as e:
228
+ print(f"{idx:5d}: {ofs:08x}-{ofs + ln:08x}: ({flags:02x}:{chk:08x}) <{e}>{deleted}")
229
+ continue
230
+ if parts.extended:
231
+ infostr = ";".join(f"{value:08x}" for value in [parts.chain[0], parts.length, *parts.chain[1:]])
232
+ for blockofs in parts.chain[:-1]:
233
+ ranges.append((blockofs, blockofs + self.blocksize, f"item #{i:d} ext"))
234
+ decflags[0] = "+"
235
+ elif parts.data:
236
+ decflags[0] = "*"
237
+
238
+ if not self.encoding & 1:
239
+ decflags[0] = " "
240
+
241
+ data = parts.data
242
+ mismatch = ""
243
+ if args.decompress and is_compressed(data):
244
+ try:
245
+ data, mismatched = decompress(data, f"record {idx} in Cro{self.name}.dat")
246
+ except ValueError as e:
247
+ print(f"{idx:5d}: {ofs:08x}-{ofs + ln:08x}: ({flags:02x}:{chk:08x}) <{e}>{deleted}")
248
+ continue
249
+ decflags[1] = "@"
250
+ if mismatched:
251
+ mismatch = " <checksum mismatch>"
252
+
253
+ # TODO: separate handling for v4
254
+ print(
255
+ f"{i + 1:5d}: {ofs:08x}-{ofs + ln:08x}: ({flags:02x}:{chk:08x}) "
256
+ f"{infostr} {''.join(decflags)}{toout(args, data)} {tohex(parts.tail)}{mismatch}{deleted}"
257
+ )
258
+
259
+ if args.verbose:
260
+ # output parts not referenced in the .tad file.
261
+ for o, length in self.enumunreferenced(ranges, self.datsize):
262
+ dat = self.readdata(o, length)
263
+ print(f"{o:08x}-{o + length:08x}: {toout(args, dat)}")