pycuf 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pycuf/__init__.py +93 -0
- pycuf/__main__.py +34 -0
- pycuf/_arrow.py +166 -0
- pycuf/_build.py +596 -0
- pycuf/_encoding.py +179 -0
- pycuf/_numeric.py +52 -0
- pycuf/_optional.py +21 -0
- pycuf/_source.py +90 -0
- pycuf/_xml.py +197 -0
- pycuf/calc.py +398 -0
- pycuf/cli.py +401 -0
- pycuf/errors.py +55 -0
- pycuf/findings.py +255 -0
- pycuf/models.py +524 -0
- pycuf/policy.py +222 -0
- pycuf/py.typed +0 -0
- pycuf/raw.py +100 -0
- pycuf/reader.py +403 -0
- pycuf/spec.py +561 -0
- pycuf/tables.py +441 -0
- pycuf/validate.py +272 -0
- pycuf/values.py +199 -0
- pycuf-0.1.0.dist-info/METADATA +461 -0
- pycuf-0.1.0.dist-info/RECORD +27 -0
- pycuf-0.1.0.dist-info/WHEEL +4 -0
- pycuf-0.1.0.dist-info/entry_points.txt +3 -0
- pycuf-0.1.0.dist-info/licenses/LICENSE +21 -0
pycuf/__init__.py
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""pycuf: read, check and analyse CUF-XML construction cost estimates.
|
|
2
|
+
|
|
3
|
+
CUF-XML (*Calculatie Uitwissel Formaat*) is the Dutch exchange format for construction cost
|
|
4
|
+
estimates (begrotingen), version 4.003. pycuf reads it into a typed model, computes costs under
|
|
5
|
+
an explicit :class:`~pycuf.policy.Policy`, reports every problem as a coded finding, and exports
|
|
6
|
+
normalized tables. The core has no dependencies outside the standard library.
|
|
7
|
+
|
|
8
|
+
Example:
|
|
9
|
+
>>> import pycuf
|
|
10
|
+
>>> cuf = pycuf.read("begroting.xml") # doctest: +SKIP
|
|
11
|
+
>>> totals = cuf.totals() # doctest: +SKIP
|
|
12
|
+
>>> totals.estimate.total # doctest: +SKIP
|
|
13
|
+
Decimal('8562.175')
|
|
14
|
+
>>> for finding in cuf.validate().findings: # doctest: +SKIP
|
|
15
|
+
... print(finding)
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from importlib.metadata import PackageNotFoundError
|
|
21
|
+
from importlib.metadata import version as _version
|
|
22
|
+
|
|
23
|
+
from . import policy
|
|
24
|
+
from .calc import Totals
|
|
25
|
+
from .errors import (
|
|
26
|
+
ForbiddenConstructError,
|
|
27
|
+
LimitExceededError,
|
|
28
|
+
MissingExtraError,
|
|
29
|
+
NotCufError,
|
|
30
|
+
PycufError,
|
|
31
|
+
XmlSyntaxError,
|
|
32
|
+
)
|
|
33
|
+
from .findings import CODES, Finding, Severity
|
|
34
|
+
from .models import (
|
|
35
|
+
Bundle,
|
|
36
|
+
Costs,
|
|
37
|
+
CostType,
|
|
38
|
+
Estimate,
|
|
39
|
+
Line,
|
|
40
|
+
Project,
|
|
41
|
+
QuantityLine,
|
|
42
|
+
ResourceLine,
|
|
43
|
+
SortCode,
|
|
44
|
+
SortCodeEntry,
|
|
45
|
+
SortCodeScheme,
|
|
46
|
+
StatedTotals,
|
|
47
|
+
Tail,
|
|
48
|
+
TailItem,
|
|
49
|
+
)
|
|
50
|
+
from .policy import Policy
|
|
51
|
+
from .raw import RawElement
|
|
52
|
+
from .reader import CufFile, read
|
|
53
|
+
from .validate import ValidationReport, validate
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
__version__ = _version("pycuf")
|
|
57
|
+
except PackageNotFoundError: # pragma: no cover - running from a source tree
|
|
58
|
+
__version__ = "0.0.0"
|
|
59
|
+
|
|
60
|
+
__all__ = [
|
|
61
|
+
"CODES",
|
|
62
|
+
"Bundle",
|
|
63
|
+
"CostType",
|
|
64
|
+
"Costs",
|
|
65
|
+
"CufFile",
|
|
66
|
+
"Estimate",
|
|
67
|
+
"Finding",
|
|
68
|
+
"ForbiddenConstructError",
|
|
69
|
+
"LimitExceededError",
|
|
70
|
+
"Line",
|
|
71
|
+
"MissingExtraError",
|
|
72
|
+
"NotCufError",
|
|
73
|
+
"Policy",
|
|
74
|
+
"Project",
|
|
75
|
+
"PycufError",
|
|
76
|
+
"QuantityLine",
|
|
77
|
+
"RawElement",
|
|
78
|
+
"ResourceLine",
|
|
79
|
+
"Severity",
|
|
80
|
+
"SortCode",
|
|
81
|
+
"SortCodeEntry",
|
|
82
|
+
"SortCodeScheme",
|
|
83
|
+
"StatedTotals",
|
|
84
|
+
"Tail",
|
|
85
|
+
"TailItem",
|
|
86
|
+
"Totals",
|
|
87
|
+
"ValidationReport",
|
|
88
|
+
"XmlSyntaxError",
|
|
89
|
+
"__version__",
|
|
90
|
+
"policy",
|
|
91
|
+
"read",
|
|
92
|
+
"validate",
|
|
93
|
+
]
|
pycuf/__main__.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Command-line entry point (``python -m pycuf`` / ``pycuf``)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def main() -> None:
|
|
9
|
+
"""Run the CLI (needs the ``cli`` extra)."""
|
|
10
|
+
try:
|
|
11
|
+
from .cli import app # noqa: PLC0415
|
|
12
|
+
except ImportError as exc: # pragma: no cover - depends on installed extras
|
|
13
|
+
if "typer" not in str(exc) and "click" not in str(exc):
|
|
14
|
+
raise
|
|
15
|
+
sys.stderr.write("The pycuf command line needs the 'cli' extra: pip install 'pycuf[cli]'\n")
|
|
16
|
+
raise SystemExit(3) from None
|
|
17
|
+
import typer # noqa: PLC0415 - installed with the cli extra
|
|
18
|
+
|
|
19
|
+
try:
|
|
20
|
+
code = app(standalone_mode=False)
|
|
21
|
+
except typer.Abort:
|
|
22
|
+
raise SystemExit(130) from None
|
|
23
|
+
except typer.TyperException as exc: # usage errors must not look like "errors found" (2)
|
|
24
|
+
show = getattr(exc, "show", None)
|
|
25
|
+
if show is not None:
|
|
26
|
+
show()
|
|
27
|
+
else:
|
|
28
|
+
sys.stderr.write(f"pycuf: {exc}\n")
|
|
29
|
+
raise SystemExit(3) from None
|
|
30
|
+
raise SystemExit(code if isinstance(code, int) else 0)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
if __name__ == "__main__":
|
|
34
|
+
main()
|
pycuf/_arrow.py
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""Arrow interoperability via nanoarrow (``pycuf[arrow]``), and polars/pandas/Parquet on top.
|
|
2
|
+
|
|
3
|
+
Each table becomes one record batch, built once from raw buffers: numbers as
|
|
4
|
+
``decimal128(28, 15)`` (16-byte little-endian integers computed exactly from the ``Decimal``
|
|
5
|
+
digits, independent of the caller's decimal context; no float anywhere), dates as ``date32``,
|
|
6
|
+
timestamps as ``timestamp[us]``, with validity bitmaps. ``requested_schema`` is ignored, as the
|
|
7
|
+
Arrow PyCapsule interface allows; consumers cast themselves.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import datetime as dt
|
|
13
|
+
from collections.abc import Sequence
|
|
14
|
+
from decimal import MAX_EMAX, MIN_EMIN, ROUND_HALF_EVEN, Context, Decimal
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from ._optional import require
|
|
19
|
+
from .tables import NUMBER, Column, OnInexact, Row
|
|
20
|
+
|
|
21
|
+
__all__ = ["build_batch", "schema", "stream", "to_pandas", "to_polars", "write_parquet"]
|
|
22
|
+
|
|
23
|
+
_EPOCH = dt.date(1970, 1, 1).toordinal()
|
|
24
|
+
_EPOCH_DT = dt.datetime(1970, 1, 1)
|
|
25
|
+
_ROUND = Context(prec=80, rounding=ROUND_HALF_EVEN, Emin=MIN_EMIN, Emax=MAX_EMAX)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _na() -> Any:
|
|
29
|
+
return require("nanoarrow", "arrow")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _type(na: Any, col: Column) -> Any:
|
|
33
|
+
if col.type == "number":
|
|
34
|
+
return na.decimal128(*NUMBER)
|
|
35
|
+
return {
|
|
36
|
+
"string": na.string(),
|
|
37
|
+
"int64": na.int64(),
|
|
38
|
+
"date": na.date32(),
|
|
39
|
+
"timestamp": na.timestamp("us"),
|
|
40
|
+
"bool": na.bool_(),
|
|
41
|
+
}[col.type]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def schema(columns: Sequence[Column]) -> Any:
|
|
45
|
+
"""Nanoarrow struct schema for ``columns``."""
|
|
46
|
+
na = _na()
|
|
47
|
+
return na.struct({c.name: _type(na, c) for c in columns})
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _unscaled(v: Decimal, scale: int, precision: int) -> int | None:
|
|
51
|
+
"""The exact unscaled integer of ``v`` at ``scale``, or ``None`` if it does not fit.
|
|
52
|
+
|
|
53
|
+
``None`` means that ``v`` needs more decimals than ``scale`` or more than ``precision`` digits.
|
|
54
|
+
Both are checked before any integer is built, so a short value with a huge exponent
|
|
55
|
+
(``0E-1000000000000``) or a long coefficient costs nothing.
|
|
56
|
+
"""
|
|
57
|
+
sign, digits, exp = v.as_tuple()
|
|
58
|
+
assert isinstance(exp, int)
|
|
59
|
+
end = len(digits)
|
|
60
|
+
while end and digits[end - 1] == 0: # 1.500 and 15E-1 are the same number
|
|
61
|
+
end -= 1
|
|
62
|
+
if not end:
|
|
63
|
+
return 0
|
|
64
|
+
shift = exp + len(digits) - end + scale
|
|
65
|
+
if shift < 0 or end + shift > precision:
|
|
66
|
+
return None
|
|
67
|
+
n: int = int("".join(map(str, digits[:end]))) * 10**shift
|
|
68
|
+
return -n if sign else n
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _validity(values: Sequence[Any]) -> tuple[bytes | None, int]:
|
|
72
|
+
nulls = 0
|
|
73
|
+
bits = bytearray((len(values) + 7) // 8)
|
|
74
|
+
for i, v in enumerate(values):
|
|
75
|
+
if v is None:
|
|
76
|
+
nulls += 1
|
|
77
|
+
else:
|
|
78
|
+
bits[i >> 3] |= 1 << (i & 7)
|
|
79
|
+
return (bytes(bits) if nulls else None), nulls
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _decimal_column(
|
|
83
|
+
na: Any, table: str, col: Column, values: list[Decimal | None], policy: OnInexact
|
|
84
|
+
) -> Any:
|
|
85
|
+
precision, scale = NUMBER
|
|
86
|
+
quantum = Decimal(1).scaleb(-scale, context=_ROUND)
|
|
87
|
+
out = bytearray(16 * len(values))
|
|
88
|
+
for i, v in enumerate(values):
|
|
89
|
+
if v is None:
|
|
90
|
+
continue
|
|
91
|
+
where = f"table {table!r}, column {col.name!r}, row {i}"
|
|
92
|
+
if not v.is_finite():
|
|
93
|
+
raise ValueError(f"{where}: {v} is not a finite number")
|
|
94
|
+
n = _unscaled(v, scale, precision)
|
|
95
|
+
if n is None and policy == "round" and v.adjusted() < precision - scale:
|
|
96
|
+
# only values whose integer part fits are worth rounding (and fit _ROUND's precision)
|
|
97
|
+
n = _unscaled(v.quantize(quantum, context=_ROUND), scale, precision)
|
|
98
|
+
if n is None:
|
|
99
|
+
if policy == "null":
|
|
100
|
+
values[i] = None
|
|
101
|
+
continue
|
|
102
|
+
raise ValueError(
|
|
103
|
+
f"{where}: value {v} does not fit decimal128{NUMBER} exactly; pass "
|
|
104
|
+
"on_inexact='null' or 'round' to export anyway"
|
|
105
|
+
)
|
|
106
|
+
out[16 * i : 16 * i + 16] = n.to_bytes(16, "little", signed=True)
|
|
107
|
+
validity, nulls = _validity(values)
|
|
108
|
+
return na.c_array_from_buffers(
|
|
109
|
+
na.decimal128(precision, scale), len(values), [validity, bytes(out)], null_count=nulls
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _column(na: Any, table: str, col: Column, values: list[Any], policy: OnInexact) -> Any:
|
|
114
|
+
if col.type == "number":
|
|
115
|
+
return _decimal_column(na, table, col, values, policy)
|
|
116
|
+
if col.type == "date":
|
|
117
|
+
days = [None if v is None else v.toordinal() - _EPOCH for v in values]
|
|
118
|
+
return na.c_array(days, na.date32())
|
|
119
|
+
if col.type == "timestamp":
|
|
120
|
+
micros = [
|
|
121
|
+
None if v is None else (v - _EPOCH_DT) // dt.timedelta(microseconds=1) for v in values
|
|
122
|
+
]
|
|
123
|
+
return na.c_array(micros, na.timestamp("us"))
|
|
124
|
+
return na.c_array(values, _type(na, col))
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def build_batch(
|
|
128
|
+
table: str, columns: Sequence[Column], rows: Sequence[Row], on_inexact: OnInexact = "raise"
|
|
129
|
+
) -> Any:
|
|
130
|
+
"""Build one nanoarrow struct array (a record batch) from ``rows``."""
|
|
131
|
+
if on_inexact not in ("raise", "null", "round"):
|
|
132
|
+
raise ValueError("on_inexact must be 'raise', 'null' or 'round'")
|
|
133
|
+
na = _na()
|
|
134
|
+
cols = [list(c) for c in zip(*rows, strict=True)] if rows else [[] for _ in columns]
|
|
135
|
+
children = [_column(na, table, c, v, on_inexact) for c, v in zip(columns, cols, strict=True)]
|
|
136
|
+
return na.c_array_from_buffers(schema(columns), len(rows), [None], children=children)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def stream(columns: Sequence[Column], batch: Any) -> Any:
|
|
140
|
+
"""A fresh nanoarrow ``CArrayStream`` over ``batch``."""
|
|
141
|
+
na = _na()
|
|
142
|
+
from nanoarrow.c_array_stream import CArrayStream # noqa: PLC0415
|
|
143
|
+
|
|
144
|
+
return CArrayStream.from_c_arrays([batch], na.c_schema(schema(columns)), validate=False)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def to_polars(batch: Any) -> Any:
|
|
148
|
+
"""A polars DataFrame from ``batch`` (no pyarrow needed)."""
|
|
149
|
+
pl = require("polars", "polars")
|
|
150
|
+
return pl.DataFrame(batch)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def to_pandas(columns: Sequence[Column], batch: Any) -> Any:
|
|
154
|
+
"""A pandas DataFrame with ``ArrowDtype`` columns (needs pandas + pyarrow)."""
|
|
155
|
+
pd = require("pandas", "pandas")
|
|
156
|
+
pa = require("pyarrow", "pandas")
|
|
157
|
+
table = pa.table(stream(columns, batch))
|
|
158
|
+
return table.to_pandas(types_mapper=pd.ArrowDtype)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def write_parquet(columns: Sequence[Column], batch: Any, path: Path) -> str:
|
|
162
|
+
"""Write ``batch`` as a Parquet file (needs ``pycuf[parquet]``)."""
|
|
163
|
+
pa = require("pyarrow", "parquet")
|
|
164
|
+
pq = require("pyarrow.parquet", "parquet")
|
|
165
|
+
pq.write_table(pa.table(stream(columns, batch)), str(path))
|
|
166
|
+
return str(path)
|