pycuf 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pycuf/__init__.py ADDED
@@ -0,0 +1,93 @@
1
+ """pycuf: read, check and analyse CUF-XML construction cost estimates.
2
+
3
+ CUF-XML (*Calculatie Uitwissel Formaat*) is the Dutch exchange format for construction cost
4
+ estimates (begrotingen), version 4.003. pycuf reads it into a typed model, computes costs under
5
+ an explicit :class:`~pycuf.policy.Policy`, reports every problem as a coded finding, and exports
6
+ normalized tables. The core has no dependencies outside the standard library.
7
+
8
+ Example:
9
+ >>> import pycuf
10
+ >>> cuf = pycuf.read("begroting.xml") # doctest: +SKIP
11
+ >>> totals = cuf.totals() # doctest: +SKIP
12
+ >>> totals.estimate.total # doctest: +SKIP
13
+ Decimal('8562.175')
14
+ >>> for finding in cuf.validate().findings: # doctest: +SKIP
15
+ ... print(finding)
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from importlib.metadata import PackageNotFoundError
21
+ from importlib.metadata import version as _version
22
+
23
+ from . import policy
24
+ from .calc import Totals
25
+ from .errors import (
26
+ ForbiddenConstructError,
27
+ LimitExceededError,
28
+ MissingExtraError,
29
+ NotCufError,
30
+ PycufError,
31
+ XmlSyntaxError,
32
+ )
33
+ from .findings import CODES, Finding, Severity
34
+ from .models import (
35
+ Bundle,
36
+ Costs,
37
+ CostType,
38
+ Estimate,
39
+ Line,
40
+ Project,
41
+ QuantityLine,
42
+ ResourceLine,
43
+ SortCode,
44
+ SortCodeEntry,
45
+ SortCodeScheme,
46
+ StatedTotals,
47
+ Tail,
48
+ TailItem,
49
+ )
50
+ from .policy import Policy
51
+ from .raw import RawElement
52
+ from .reader import CufFile, read
53
+ from .validate import ValidationReport, validate
54
+
55
+ try:
56
+ __version__ = _version("pycuf")
57
+ except PackageNotFoundError: # pragma: no cover - running from a source tree
58
+ __version__ = "0.0.0"
59
+
60
+ __all__ = [
61
+ "CODES",
62
+ "Bundle",
63
+ "CostType",
64
+ "Costs",
65
+ "CufFile",
66
+ "Estimate",
67
+ "Finding",
68
+ "ForbiddenConstructError",
69
+ "LimitExceededError",
70
+ "Line",
71
+ "MissingExtraError",
72
+ "NotCufError",
73
+ "Policy",
74
+ "Project",
75
+ "PycufError",
76
+ "QuantityLine",
77
+ "RawElement",
78
+ "ResourceLine",
79
+ "Severity",
80
+ "SortCode",
81
+ "SortCodeEntry",
82
+ "SortCodeScheme",
83
+ "StatedTotals",
84
+ "Tail",
85
+ "TailItem",
86
+ "Totals",
87
+ "ValidationReport",
88
+ "XmlSyntaxError",
89
+ "__version__",
90
+ "policy",
91
+ "read",
92
+ "validate",
93
+ ]
pycuf/__main__.py ADDED
@@ -0,0 +1,34 @@
1
+ """Command-line entry point (``python -m pycuf`` / ``pycuf``)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+
7
+
8
+ def main() -> None:
9
+ """Run the CLI (needs the ``cli`` extra)."""
10
+ try:
11
+ from .cli import app # noqa: PLC0415
12
+ except ImportError as exc: # pragma: no cover - depends on installed extras
13
+ if "typer" not in str(exc) and "click" not in str(exc):
14
+ raise
15
+ sys.stderr.write("The pycuf command line needs the 'cli' extra: pip install 'pycuf[cli]'\n")
16
+ raise SystemExit(3) from None
17
+ import typer # noqa: PLC0415 - installed with the cli extra
18
+
19
+ try:
20
+ code = app(standalone_mode=False)
21
+ except typer.Abort:
22
+ raise SystemExit(130) from None
23
+ except typer.TyperException as exc: # usage errors must not look like "errors found" (2)
24
+ show = getattr(exc, "show", None)
25
+ if show is not None:
26
+ show()
27
+ else:
28
+ sys.stderr.write(f"pycuf: {exc}\n")
29
+ raise SystemExit(3) from None
30
+ raise SystemExit(code if isinstance(code, int) else 0)
31
+
32
+
33
+ if __name__ == "__main__":
34
+ main()
pycuf/_arrow.py ADDED
@@ -0,0 +1,166 @@
1
+ """Arrow interoperability via nanoarrow (``pycuf[arrow]``), and polars/pandas/Parquet on top.
2
+
3
+ Each table becomes one record batch, built once from raw buffers: numbers as
4
+ ``decimal128(28, 15)`` (16-byte little-endian integers computed exactly from the ``Decimal``
5
+ digits, independent of the caller's decimal context; no float anywhere), dates as ``date32``,
6
+ timestamps as ``timestamp[us]``, with validity bitmaps. ``requested_schema`` is ignored, as the
7
+ Arrow PyCapsule interface allows; consumers cast themselves.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import datetime as dt
13
+ from collections.abc import Sequence
14
+ from decimal import MAX_EMAX, MIN_EMIN, ROUND_HALF_EVEN, Context, Decimal
15
+ from pathlib import Path
16
+ from typing import Any
17
+
18
+ from ._optional import require
19
+ from .tables import NUMBER, Column, OnInexact, Row
20
+
21
+ __all__ = ["build_batch", "schema", "stream", "to_pandas", "to_polars", "write_parquet"]
22
+
23
+ _EPOCH = dt.date(1970, 1, 1).toordinal()
24
+ _EPOCH_DT = dt.datetime(1970, 1, 1)
25
+ _ROUND = Context(prec=80, rounding=ROUND_HALF_EVEN, Emin=MIN_EMIN, Emax=MAX_EMAX)
26
+
27
+
28
+ def _na() -> Any:
29
+ return require("nanoarrow", "arrow")
30
+
31
+
32
+ def _type(na: Any, col: Column) -> Any:
33
+ if col.type == "number":
34
+ return na.decimal128(*NUMBER)
35
+ return {
36
+ "string": na.string(),
37
+ "int64": na.int64(),
38
+ "date": na.date32(),
39
+ "timestamp": na.timestamp("us"),
40
+ "bool": na.bool_(),
41
+ }[col.type]
42
+
43
+
44
+ def schema(columns: Sequence[Column]) -> Any:
45
+ """Nanoarrow struct schema for ``columns``."""
46
+ na = _na()
47
+ return na.struct({c.name: _type(na, c) for c in columns})
48
+
49
+
50
+ def _unscaled(v: Decimal, scale: int, precision: int) -> int | None:
51
+ """The exact unscaled integer of ``v`` at ``scale``, or ``None`` if it does not fit.
52
+
53
+ ``None`` means that ``v`` needs more decimals than ``scale`` or more than ``precision`` digits.
54
+ Both are checked before any integer is built, so a short value with a huge exponent
55
+ (``0E-1000000000000``) or a long coefficient costs nothing.
56
+ """
57
+ sign, digits, exp = v.as_tuple()
58
+ assert isinstance(exp, int)
59
+ end = len(digits)
60
+ while end and digits[end - 1] == 0: # 1.500 and 15E-1 are the same number
61
+ end -= 1
62
+ if not end:
63
+ return 0
64
+ shift = exp + len(digits) - end + scale
65
+ if shift < 0 or end + shift > precision:
66
+ return None
67
+ n: int = int("".join(map(str, digits[:end]))) * 10**shift
68
+ return -n if sign else n
69
+
70
+
71
+ def _validity(values: Sequence[Any]) -> tuple[bytes | None, int]:
72
+ nulls = 0
73
+ bits = bytearray((len(values) + 7) // 8)
74
+ for i, v in enumerate(values):
75
+ if v is None:
76
+ nulls += 1
77
+ else:
78
+ bits[i >> 3] |= 1 << (i & 7)
79
+ return (bytes(bits) if nulls else None), nulls
80
+
81
+
82
+ def _decimal_column(
83
+ na: Any, table: str, col: Column, values: list[Decimal | None], policy: OnInexact
84
+ ) -> Any:
85
+ precision, scale = NUMBER
86
+ quantum = Decimal(1).scaleb(-scale, context=_ROUND)
87
+ out = bytearray(16 * len(values))
88
+ for i, v in enumerate(values):
89
+ if v is None:
90
+ continue
91
+ where = f"table {table!r}, column {col.name!r}, row {i}"
92
+ if not v.is_finite():
93
+ raise ValueError(f"{where}: {v} is not a finite number")
94
+ n = _unscaled(v, scale, precision)
95
+ if n is None and policy == "round" and v.adjusted() < precision - scale:
96
+ # only values whose integer part fits are worth rounding (and fit _ROUND's precision)
97
+ n = _unscaled(v.quantize(quantum, context=_ROUND), scale, precision)
98
+ if n is None:
99
+ if policy == "null":
100
+ values[i] = None
101
+ continue
102
+ raise ValueError(
103
+ f"{where}: value {v} does not fit decimal128{NUMBER} exactly; pass "
104
+ "on_inexact='null' or 'round' to export anyway"
105
+ )
106
+ out[16 * i : 16 * i + 16] = n.to_bytes(16, "little", signed=True)
107
+ validity, nulls = _validity(values)
108
+ return na.c_array_from_buffers(
109
+ na.decimal128(precision, scale), len(values), [validity, bytes(out)], null_count=nulls
110
+ )
111
+
112
+
113
+ def _column(na: Any, table: str, col: Column, values: list[Any], policy: OnInexact) -> Any:
114
+ if col.type == "number":
115
+ return _decimal_column(na, table, col, values, policy)
116
+ if col.type == "date":
117
+ days = [None if v is None else v.toordinal() - _EPOCH for v in values]
118
+ return na.c_array(days, na.date32())
119
+ if col.type == "timestamp":
120
+ micros = [
121
+ None if v is None else (v - _EPOCH_DT) // dt.timedelta(microseconds=1) for v in values
122
+ ]
123
+ return na.c_array(micros, na.timestamp("us"))
124
+ return na.c_array(values, _type(na, col))
125
+
126
+
127
+ def build_batch(
128
+ table: str, columns: Sequence[Column], rows: Sequence[Row], on_inexact: OnInexact = "raise"
129
+ ) -> Any:
130
+ """Build one nanoarrow struct array (a record batch) from ``rows``."""
131
+ if on_inexact not in ("raise", "null", "round"):
132
+ raise ValueError("on_inexact must be 'raise', 'null' or 'round'")
133
+ na = _na()
134
+ cols = [list(c) for c in zip(*rows, strict=True)] if rows else [[] for _ in columns]
135
+ children = [_column(na, table, c, v, on_inexact) for c, v in zip(columns, cols, strict=True)]
136
+ return na.c_array_from_buffers(schema(columns), len(rows), [None], children=children)
137
+
138
+
139
+ def stream(columns: Sequence[Column], batch: Any) -> Any:
140
+ """A fresh nanoarrow ``CArrayStream`` over ``batch``."""
141
+ na = _na()
142
+ from nanoarrow.c_array_stream import CArrayStream # noqa: PLC0415
143
+
144
+ return CArrayStream.from_c_arrays([batch], na.c_schema(schema(columns)), validate=False)
145
+
146
+
147
+ def to_polars(batch: Any) -> Any:
148
+ """A polars DataFrame from ``batch`` (no pyarrow needed)."""
149
+ pl = require("polars", "polars")
150
+ return pl.DataFrame(batch)
151
+
152
+
153
+ def to_pandas(columns: Sequence[Column], batch: Any) -> Any:
154
+ """A pandas DataFrame with ``ArrowDtype`` columns (needs pandas + pyarrow)."""
155
+ pd = require("pandas", "pandas")
156
+ pa = require("pyarrow", "pandas")
157
+ table = pa.table(stream(columns, batch))
158
+ return table.to_pandas(types_mapper=pd.ArrowDtype)
159
+
160
+
161
+ def write_parquet(columns: Sequence[Column], batch: Any, path: Path) -> str:
162
+ """Write ``batch`` as a Parquet file (needs ``pycuf[parquet]``)."""
163
+ pa = require("pyarrow", "parquet")
164
+ pq = require("pyarrow.parquet", "parquet")
165
+ pq.write_table(pa.table(stream(columns, batch)), str(path))
166
+ return str(path)