pyxaf 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pyxaf/__init__.py ADDED
@@ -0,0 +1,108 @@
1
+ """pyxaf — read and validate every iteration of the Dutch Auditfile Financieel.
2
+
3
+ Supported: ADF (``CLAIR1.00.00``), CLAIR2, XAF 3.0, 3.1, 3.2, 3.2.1 and 4.0. The core has no
4
+ dependencies outside the standard library.
5
+
6
+ Example:
7
+ >>> import pyxaf
8
+ >>> with pyxaf.open("2024.xaf") as af: # doctest: +SKIP
9
+ ... print(af.version, af.company.name)
10
+ ... for line in af.lines():
11
+ ... ...
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from importlib.metadata import PackageNotFoundError
17
+ from importlib.metadata import version as _version
18
+
19
+ from .detect import detect
20
+ from .errors import (
21
+ EncryptedAuditfileError,
22
+ ForbiddenConstructError,
23
+ LimitExceededError,
24
+ MissingExtraError,
25
+ NotAnAuditfileError,
26
+ PyxafError,
27
+ XmlSyntaxError,
28
+ )
29
+ from .findings import CODES, Finding, Severity
30
+ from .formats import Family, FormatInfo, NamespaceStatus, Version
31
+ from .models import (
32
+ AccountKind,
33
+ AccountType,
34
+ Address,
35
+ Company,
36
+ ForeignAmount,
37
+ Header,
38
+ Journal,
39
+ JournalKind,
40
+ JournalType,
41
+ LedgerAccount,
42
+ Line,
43
+ OpeningBalance,
44
+ Period,
45
+ Relation,
46
+ RelationKind,
47
+ RelationType,
48
+ RgsRef,
49
+ Side,
50
+ Transaction,
51
+ TransactionTotals,
52
+ VatCode,
53
+ VatLine,
54
+ )
55
+ from .raw import RawRecord
56
+ from .reader import AuditFile, open
57
+ from .validate import ValidationReport, validate
58
+
59
+ try:
60
+ __version__ = _version("pyxaf")
61
+ except PackageNotFoundError: # pragma: no cover - running from a source tree
62
+ __version__ = "0.0.0"
63
+
64
+ __all__ = [
65
+ "CODES",
66
+ "AccountKind",
67
+ "AccountType",
68
+ "Address",
69
+ "AuditFile",
70
+ "Company",
71
+ "EncryptedAuditfileError",
72
+ "Family",
73
+ "Finding",
74
+ "ForbiddenConstructError",
75
+ "ForeignAmount",
76
+ "FormatInfo",
77
+ "Header",
78
+ "Journal",
79
+ "JournalKind",
80
+ "JournalType",
81
+ "LedgerAccount",
82
+ "LimitExceededError",
83
+ "Line",
84
+ "MissingExtraError",
85
+ "NamespaceStatus",
86
+ "NotAnAuditfileError",
87
+ "OpeningBalance",
88
+ "Period",
89
+ "PyxafError",
90
+ "RawRecord",
91
+ "Relation",
92
+ "RelationKind",
93
+ "RelationType",
94
+ "RgsRef",
95
+ "Severity",
96
+ "Side",
97
+ "Transaction",
98
+ "TransactionTotals",
99
+ "ValidationReport",
100
+ "VatCode",
101
+ "VatLine",
102
+ "Version",
103
+ "XmlSyntaxError",
104
+ "__version__",
105
+ "detect",
106
+ "open",
107
+ "validate",
108
+ ]
pyxaf/__main__.py ADDED
@@ -0,0 +1,34 @@
1
+ """Command-line entry point (``python -m pyxaf`` / ``pyxaf``)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+
7
+
8
+ def main() -> None:
9
+ """Run the CLI (needs the ``cli`` extra)."""
10
+ try:
11
+ from .cli import app # noqa: PLC0415
12
+ except ImportError as exc: # pragma: no cover - depends on installed extras
13
+ if "typer" not in str(exc) and "click" not in str(exc):
14
+ raise
15
+ sys.stderr.write("The pyxaf command line needs the 'cli' extra: pip install 'pyxaf[cli]'\n")
16
+ raise SystemExit(3) from None
17
+ import typer # noqa: PLC0415 - installed with the cli extra
18
+
19
+ try:
20
+ code = app(standalone_mode=False)
21
+ except typer.Abort:
22
+ raise SystemExit(130) from None
23
+ except typer.TyperException as exc: # usage errors must not look like "errors found" (2)
24
+ show = getattr(exc, "show", None)
25
+ if show is not None:
26
+ show()
27
+ else:
28
+ sys.stderr.write(f"pyxaf: {exc}\n")
29
+ raise SystemExit(3) from None
30
+ raise SystemExit(code if isinstance(code, int) else 0)
31
+
32
+
33
+ if __name__ == "__main__":
34
+ main()
pyxaf/_adf.py ADDED
@@ -0,0 +1,414 @@
1
+ """Reader for the fixed-width ASCII auditfile (ADF, ``CLAIR1.00.00``, 1999).
2
+
3
+ Layout from the Belastingdienst *Handleiding Auditfile* (Dec 2001, erratum Feb 2003): one header
4
+ line followed by mutation lines, CR LF separated, fields left-justified and space-padded. ADF has no
5
+ master-data section: accounts, relations and journals are collected from the mutation lines.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import codecs
11
+ from collections.abc import Iterator
12
+ from decimal import Decimal
13
+ from typing import TYPE_CHECKING, NamedTuple
14
+
15
+ from ._normalize import NegativePolicy
16
+ from ._source import Source
17
+ from .findings import FindingCollector
18
+ from .models import (
19
+ AccountType,
20
+ Address,
21
+ Company,
22
+ ForeignAmount,
23
+ Header,
24
+ Journal,
25
+ JournalType,
26
+ LedgerAccount,
27
+ Line,
28
+ Relation,
29
+ RelationType,
30
+ Side,
31
+ Transaction,
32
+ TransactionTotals,
33
+ VatLine,
34
+ )
35
+ from .raw import RawRecord
36
+ from .values import parse_adf_amount, parse_adf_date, parse_int
37
+
38
+ if TYPE_CHECKING:
39
+ from .reader import _Master
40
+
41
+ __all__ = ["ADF_HEADER", "ADF_LINE", "AdfField", "AdfReader"]
42
+
43
+
44
+ class AdfField(NamedTuple):
45
+ """A fixed-width ADF field (``start`` is 1-based as in the manual)."""
46
+
47
+ name: str
48
+ start: int
49
+ length: int
50
+ required: bool = False
51
+ kind: str = "C" # C character, N numeric (decimals), D date
52
+ decimals: int = 0
53
+
54
+ def slice(self, line: str) -> str:
55
+ return line[self.start - 1 : self.start - 1 + self.length]
56
+
57
+
58
+ ADF_HEADER: tuple[AdfField, ...] = (
59
+ AdfField("versie", 1, 12, True),
60
+ AdfField("boekhoudpakket", 13, 50),
61
+ AdfField("administratiecode", 63, 20),
62
+ AdfField("jaarPeriode", 83, 15, True),
63
+ AdfField("fiscaalnummer", 98, 15),
64
+ AdfField("naamOnderneming", 113, 50, True),
65
+ AdfField("adres", 163, 30),
66
+ AdfField("plaats", 193, 30),
67
+ AdfField("aantalMutaties", 223, 10, False, "N", 0),
68
+ AdfField("datumAanmaak", 233, 10, False, "D"),
69
+ AdfField("tellingDebet", 243, 16, False, "N", 2),
70
+ AdfField("tellingCredit", 259, 16, False, "N", 2),
71
+ )
72
+
73
+ ADF_LINE: tuple[AdfField, ...] = (
74
+ AdfField("dagboekcode", 1, 20),
75
+ AdfField("dagboekomschrijving", 21, 30),
76
+ AdfField("periode", 51, 5),
77
+ AdfField("volgnummer", 56, 10),
78
+ AdfField("regelnummer", 66, 5, False, "N", 0),
79
+ AdfField("identificatieJournaalpost", 71, 20),
80
+ AdfField("verwerkingsdatum", 91, 10, False, "D"),
81
+ AdfField("grootboekrekeningcode", 101, 15, True),
82
+ AdfField("soortGrootboekrekening", 116, 5),
83
+ AdfField("cluster", 121, 15),
84
+ AdfField("grootboekrekeningnaam", 136, 30, True),
85
+ AdfField("mutatiedatum", 166, 10, True, "D"),
86
+ AdfField("boekstuknummer", 176, 15),
87
+ AdfField("soortMutatie", 191, 5),
88
+ AdfField("relatieAndereAdministraties", 196, 15),
89
+ AdfField("kostenplaats", 211, 15),
90
+ AdfField("kostensoort", 226, 15),
91
+ AdfField("kostendrager", 241, 15),
92
+ AdfField("omschrijving", 256, 30, True),
93
+ AdfField("debet", 286, 16, True, "N", 2),
94
+ AdfField("credit", 302, 16, True, "N", 2),
95
+ AdfField("btwCode", 318, 5),
96
+ AdfField("valuta", 323, 10),
97
+ AdfField("koers", 333, 13, False, "N", 6),
98
+ AdfField("debCredNummer", 346, 15),
99
+ AdfField("debCredSoort", 361, 5),
100
+ AdfField("debCredFiscaalNummer", 366, 16),
101
+ AdfField("debCredNaam", 382, 30),
102
+ AdfField("debCredAdres", 412, 30),
103
+ AdfField("debCredPostcode", 442, 10),
104
+ AdfField("debCredPlaats", 452, 30),
105
+ AdfField("debCredLand", 482, 15),
106
+ )
107
+ ADF_LINE_LENGTH = 496
108
+ _ZERO = Decimal(0)
109
+
110
+
111
+ def _s(v: str | None) -> str | None:
112
+ if v is None:
113
+ return None
114
+ v = v.strip()
115
+ return v or None
116
+
117
+
118
+ class AdfReader:
119
+ """Streams an ADF file (re-reading the source per pass)."""
120
+
121
+ def __init__(
122
+ self,
123
+ source: Source,
124
+ *,
125
+ encoding: str,
126
+ policy: NegativePolicy,
127
+ findings: FindingCollector,
128
+ value_findings: bool = True,
129
+ ) -> None:
130
+ self._src = source
131
+ self._encoding = encoding
132
+ self._policy = policy
133
+ self._findings = findings
134
+ self._vf = value_findings
135
+ self.company: Company = Company(name=None)
136
+ self.totals: TransactionTotals | None = None
137
+ self.header_raw: RawRecord | None = None
138
+ self._reported_replacements = False
139
+
140
+ # ------------------------------------------------------------------ lines
141
+ def _lines(self) -> Iterator[tuple[int, str]]:
142
+ decoder = codecs.getincrementaldecoder(self._encoding)(errors="replace")
143
+ buf = ""
144
+ n = 0
145
+ replaced = 0
146
+ for chunk in self._src.chunks():
147
+ text = decoder.decode(chunk)
148
+ replaced += text.count("\ufffd")
149
+ buf += text
150
+ *lines, buf = buf.split("\n")
151
+ for ln in lines:
152
+ n += 1
153
+ yield n, ln.rstrip("\r\x1a")
154
+ buf += decoder.decode(b"", final=True)
155
+ if buf.rstrip("\r\x1a").strip():
156
+ yield n + 1, buf.rstrip("\r\x1a")
157
+ if replaced and self._vf and not self._reported_replacements:
158
+ self._reported_replacements = True
159
+ self._findings.add(
160
+ "XAF1006",
161
+ f"{replaced} byte(s) not valid in {self._encoding} were replaced by U+FFFD "
162
+ "(pass encoding= to choose another code page)",
163
+ )
164
+
165
+ def _record(self, tag: str, spec: tuple[AdfField, ...], text: str, line_no: int) -> RawRecord:
166
+ return RawRecord(tag, {f.name: f.slice(text) for f in spec}, None, line_no)
167
+
168
+ def _num(self, rec: RawRecord, f: AdfField) -> Decimal | None:
169
+ text = rec.fields[f.name]
170
+ if not text.strip():
171
+ return None
172
+ value = parse_adf_amount(text, f.decimals)
173
+ if value is None and self._vf:
174
+ self._findings.add(
175
+ "XAF3018",
176
+ f"ADF field {f.name} is not a valid number",
177
+ line=rec.line,
178
+ path=f.name,
179
+ value=text,
180
+ )
181
+ return value
182
+
183
+ def _date(self, rec: RawRecord, name: str) -> object:
184
+ text = rec.fields[name]
185
+ if not text.strip():
186
+ return None
187
+ value = parse_adf_date(text)
188
+ if value is None and self._vf:
189
+ self._findings.add(
190
+ "XAF3017",
191
+ f"ADF field {name} is not a valid date",
192
+ line=rec.line,
193
+ path=name,
194
+ value=text,
195
+ )
196
+ return value
197
+
198
+ def _check(self, rec: RawRecord, spec: tuple[AdfField, ...], text: str) -> None:
199
+ if not self._vf:
200
+ return
201
+ for f in spec:
202
+ if f.required and not rec.fields[f.name].strip():
203
+ self._findings.add(
204
+ "XAF3010", f"required ADF field {f.name} is empty", line=rec.line, path=f.name
205
+ )
206
+ if len(text) > (ADF_LINE_LENGTH if spec is ADF_LINE else 274) + 2:
207
+ self._findings.add(
208
+ "XAF3050",
209
+ f"line is {len(text)} characters long; data beyond the last field is ignored",
210
+ line=rec.line,
211
+ )
212
+
213
+ # ---------------------------------------------------------------- master
214
+ def load_master(self, master: _Master) -> None:
215
+ """Read the header and collect accounts, relations and journals from all lines."""
216
+ hdr_fields = {f.name: f for f in ADF_HEADER}
217
+ acc_seq = rel_seq = 0
218
+ it = self._lines()
219
+ for n, text in it:
220
+ if n == 1:
221
+ rec = self._record("adfHeader", ADF_HEADER, text, n)
222
+ self.header_raw = rec
223
+ self._check(rec, ADF_HEADER, text)
224
+ f = rec.fields
225
+ created = self._date(rec, "datumAanmaak")
226
+ master.header = Header(
227
+ fiscal_year=_s(f["jaarPeriode"]),
228
+ start_date=None,
229
+ end_date=None,
230
+ currency=None,
231
+ created=created, # type: ignore[arg-type]
232
+ software_name=_s(f["boekhoudpakket"]),
233
+ software_version=None,
234
+ declared_version=_s(f["versie"]),
235
+ raw=rec,
236
+ )
237
+ addr = Address(kind="street", street=_s(f["adres"]), city=_s(f["plaats"]))
238
+ self.company = Company(
239
+ name=_s(f["naamOnderneming"]),
240
+ identifier=_s(f["administratiecode"]),
241
+ tax_registration_id=_s(f["fiscaalnummer"]),
242
+ addresses=(addr,) if addr.street or addr.city else (),
243
+ raw=rec,
244
+ )
245
+ count = f["aantalMutaties"].strip()
246
+ self.totals = TransactionTotals(
247
+ lines_count=parse_int(count) if count else None,
248
+ total_debit=self._num(rec, hdr_fields["tellingDebet"]),
249
+ total_credit=self._num(rec, hdr_fields["tellingCredit"]),
250
+ )
251
+ continue
252
+ if not text.strip():
253
+ continue
254
+ rec = self._record("adfLine", ADF_LINE, text, n)
255
+ f = rec.fields
256
+ acc_id = f["grootboekrekeningcode"].strip()
257
+ if acc_id and acc_id not in master.accounts:
258
+ master.accounts[acc_id] = LedgerAccount(
259
+ id=acc_id,
260
+ description=_s(f["grootboekrekeningnaam"]),
261
+ account_type=AccountType.from_code(f["soortGrootboekrekening"]),
262
+ lead_code=_s(f["cluster"]),
263
+ seq=acc_seq,
264
+ raw=rec,
265
+ )
266
+ acc_seq += 1
267
+ rel_id = f["debCredNummer"].strip()
268
+ if rel_id and rel_id not in master.relations:
269
+ addr = Address(
270
+ kind="street",
271
+ street=_s(f["debCredAdres"]),
272
+ postal_code=_s(f["debCredPostcode"]),
273
+ city=_s(f["debCredPlaats"]),
274
+ country=_s(f["debCredLand"]),
275
+ )
276
+ master.relations[rel_id] = Relation(
277
+ id=rel_id,
278
+ name=_s(f["debCredNaam"]),
279
+ relation_type=RelationType.from_code(f["debCredSoort"]),
280
+ tax_registration_id=_s(f["debCredFiscaalNummer"]),
281
+ addresses=(addr,) if any((addr.street, addr.city, addr.country)) else (),
282
+ seq=rel_seq,
283
+ raw=rec,
284
+ )
285
+ rel_seq += 1
286
+ jr_id = f["dagboekcode"].strip()
287
+ if jr_id not in master.journals:
288
+ master.journals[jr_id] = Journal(
289
+ id=jr_id,
290
+ description=_s(f["dagboekomschrijving"]),
291
+ journal_type=JournalType.from_code(f["dagboekomschrijving"]),
292
+ seq=len(master.journals),
293
+ raw=rec,
294
+ )
295
+ master.journals_complete = True
296
+
297
+ def load_journals(self, journals: dict[str, Journal]) -> None:
298
+ """Journals are collected by :meth:`load_master` already."""
299
+
300
+ # ------------------------------------------------------------ transactions
301
+ def raw_lines(self) -> Iterator[tuple[str | None, RawRecord]]:
302
+ """Yield ``(journal_id, line_record)`` for every mutation line."""
303
+ for n, text in self._lines():
304
+ if n == 1 or not text.strip():
305
+ continue
306
+ rec = self._record("adfLine", ADF_LINE, text, n)
307
+ yield rec.fields["dagboekcode"].strip() or None, rec
308
+
309
+ def transactions(self) -> Iterator[Transaction]:
310
+ """Group consecutive lines with the same journal, period and entry number."""
311
+ spec = {f.name: f for f in ADF_LINE}
312
+ tx_seq = 0
313
+ line_seq = 0
314
+ current_key: tuple[str, ...] | None = None
315
+ pending: list[Line] = []
316
+ first_rec: RawRecord | None = None
317
+ for n, text in self._lines():
318
+ if n == 1 or not text.strip():
319
+ continue
320
+ rec = self._record("adfLine", ADF_LINE, text, n)
321
+ self._check(rec, ADF_LINE, text)
322
+ f = rec.fields
323
+ key = (
324
+ f["dagboekcode"].strip(),
325
+ f["periode"].strip(),
326
+ f["volgnummer"].strip(),
327
+ f["identificatieJournaalpost"].strip(),
328
+ )
329
+ if key != current_key and pending:
330
+ yield self._tx(first_rec, pending, tx_seq)
331
+ tx_seq += 1
332
+ pending = []
333
+ if not pending:
334
+ current_key = key
335
+ first_rec = rec
336
+ pending.append(self._line(rec, spec, line_seq, tx_seq))
337
+ line_seq += 1
338
+ if pending:
339
+ yield self._tx(first_rec, pending, tx_seq)
340
+
341
+ def _tx(self, rec: RawRecord | None, lines: list[Line], seq: int) -> Transaction:
342
+ assert rec is not None
343
+ f = rec.fields
344
+ period_key = f["periode"].strip() or None
345
+ date = lines[0].transaction_date
346
+ return Transaction(
347
+ seq=seq,
348
+ journal_id=f["dagboekcode"].strip(),
349
+ number=_s(f["volgnummer"]) or _s(f["identificatieJournaalpost"]),
350
+ description=None,
351
+ period_key=period_key,
352
+ period_number=parse_int(period_key),
353
+ date=date,
354
+ lines=tuple(lines),
355
+ raw=rec,
356
+ )
357
+
358
+ def _line(self, rec: RawRecord, spec: dict[str, AdfField], seq: int, tx_seq: int) -> Line:
359
+ f = rec.fields
360
+ d = self._num(rec, spec["debet"]) or _ZERO
361
+ c = self._num(rec, spec["credit"]) or _ZERO
362
+ if (d < 0 or c < 0) and self._vf:
363
+ self._findings.add("XAF7001", "negative debit/credit amount", line=rec.line)
364
+ if self._policy == "abs":
365
+ d, c = abs(d), abs(c)
366
+ if d and c and self._vf:
367
+ self._findings.add(
368
+ "XAF7013", "line has both a debit and a credit amount", line=rec.line
369
+ )
370
+ signed = abs(d - c) if d == c else d - c # no -0.00
371
+ if c and not d:
372
+ amount, side = c, Side.CREDIT
373
+ elif d and not c:
374
+ amount, side = d, Side.DEBIT
375
+ else:
376
+ amount, side = abs(signed), Side.DEBIT if signed >= 0 else Side.CREDIT
377
+ debit, credit = (signed, _ZERO) if signed >= 0 else (_ZERO, -signed)
378
+ currency = _s(f["valuta"])
379
+ rate = self._num(rec, spec["koers"])
380
+ foreign = (
381
+ ForeignAmount(currency=currency, amount=None, exchange_rate=rate)
382
+ if currency or rate is not None
383
+ else None
384
+ )
385
+ vat_code = _s(f["btwCode"])
386
+ period_key = f["periode"].strip() or None
387
+ return Line(
388
+ seq=seq,
389
+ number=_s(f["regelnummer"]),
390
+ account_id=_s(f["grootboekrekeningcode"]),
391
+ amount=amount,
392
+ side=side,
393
+ signed_amount=signed,
394
+ debit=debit,
395
+ credit=credit,
396
+ description=_s(f["omschrijving"]),
397
+ document_ref=_s(f["boekstuknummer"]),
398
+ effective_date=self._date(rec, "mutatiedatum"), # type: ignore[arg-type]
399
+ relation_id=_s(f["debCredNummer"]),
400
+ cost_center=_s(f["kostenplaats"]),
401
+ cost_unit=_s(f["kostendrager"]),
402
+ vat=(
403
+ VatLine(code=vat_code, percentage=None, amount=None, side=None, signed_amount=None),
404
+ )
405
+ if vat_code
406
+ else (),
407
+ foreign=foreign,
408
+ journal_id=f["dagboekcode"].strip(),
409
+ transaction_number=_s(f["volgnummer"]) or _s(f["identificatieJournaalpost"]),
410
+ transaction_seq=tx_seq,
411
+ period_key=period_key,
412
+ transaction_date=self._date(rec, "verwerkingsdatum"), # type: ignore[arg-type]
413
+ raw=rec,
414
+ )