liander-open-data 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- liander_open_data-0.1.0/.gitignore +22 -0
- liander_open_data-0.1.0/CHANGELOG.md +18 -0
- liander_open_data-0.1.0/LICENSE +21 -0
- liander_open_data-0.1.0/PKG-INFO +122 -0
- liander_open_data-0.1.0/README.md +84 -0
- liander_open_data-0.1.0/docs/changelog.md +1 -0
- liander_open_data-0.1.0/docs/cli.md +89 -0
- liander_open_data-0.1.0/docs/contributing.md +52 -0
- liander_open_data-0.1.0/docs/data-model.md +114 -0
- liander_open_data-0.1.0/docs/data.md +71 -0
- liander_open_data-0.1.0/docs/index.md +51 -0
- liander_open_data-0.1.0/docs/installation.md +58 -0
- liander_open_data-0.1.0/docs/quickstart.md +108 -0
- liander_open_data-0.1.0/docs/recipes.md +83 -0
- liander_open_data-0.1.0/docs/reference/database.md +11 -0
- liander_open_data-0.1.0/docs/reference/dictionary.md +15 -0
- liander_open_data-0.1.0/docs/releasing.md +59 -0
- liander_open_data-0.1.0/mkdocs.yml +78 -0
- liander_open_data-0.1.0/pyproject.toml +90 -0
- liander_open_data-0.1.0/src/liander_open_data/__init__.py +49 -0
- liander_open_data-0.1.0/src/liander_open_data/__main__.py +5 -0
- liander_open_data-0.1.0/src/liander_open_data/cli.py +178 -0
- liander_open_data-0.1.0/src/liander_open_data/data/sbi_dictionary.csv +1428 -0
- liander_open_data-0.1.0/src/liander_open_data/database.py +595 -0
- liander_open_data-0.1.0/src/liander_open_data/dictionary.py +265 -0
- liander_open_data-0.1.0/src/liander_open_data/py.typed +0 -0
- liander_open_data-0.1.0/tests/conftest.py +51 -0
- liander_open_data-0.1.0/tests/test_cli.py +43 -0
- liander_open_data-0.1.0/tests/test_database.py +153 -0
- liander_open_data-0.1.0/tests/test_dictionary.py +95 -0
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Python-generated files
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[oc]
|
|
4
|
+
build/
|
|
5
|
+
dist/
|
|
6
|
+
wheels/
|
|
7
|
+
*.egg-info
|
|
8
|
+
|
|
9
|
+
# Virtual environments
|
|
10
|
+
.venv
|
|
11
|
+
|
|
12
|
+
# Local data (too large for git / PyPI; see docs/installation.md)
|
|
13
|
+
data/Profiles/
|
|
14
|
+
*.db
|
|
15
|
+
*.db.tmp
|
|
16
|
+
|
|
17
|
+
# Tooling
|
|
18
|
+
site/
|
|
19
|
+
.pytest_cache/
|
|
20
|
+
.ruff_cache/
|
|
21
|
+
.coverage
|
|
22
|
+
coverage.xml
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and the project uses
|
|
5
|
+
[Semantic Versioning](https://semver.org/).
|
|
6
|
+
|
|
7
|
+
## [0.1.0] - 2026-10-10
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- `build_database()` / `liander-open-data build`: build a single SQLite file
|
|
12
|
+
from the Liander profile CSVs and the SBI dictionary.
|
|
13
|
+
- `ProfileDatabase` query API: `get_profile`, `get_profiles`, `lookup`,
|
|
14
|
+
`recommended_profile`, `dictionary`, `list_profiles` and `query`.
|
|
15
|
+
- Parser for the Liander *Overzicht met per SBI-code een aanbevolen profiel*
|
|
16
|
+
workbook, with a cleaned copy bundled in the package.
|
|
17
|
+
- Command line interface with `build`, `info`, `list`, `lookup`, `search`,
|
|
18
|
+
`export` and `dictionary` commands.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Lingkang
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: liander-open-data
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Lightweight SQLite database and Python API for Liander's open SBI load profiles.
|
|
5
|
+
Project-URL: Homepage, https://github.com/lingkang95/Liander_open_data
|
|
6
|
+
Project-URL: Documentation, https://github.com/lingkang95/Liander_open_data/tree/master/docs
|
|
7
|
+
Project-URL: Source, https://github.com/lingkang95/Liander_open_data
|
|
8
|
+
Project-URL: Issues, https://github.com/lingkang95/Liander_open_data/issues
|
|
9
|
+
Project-URL: Changelog, https://github.com/lingkang95/Liander_open_data/blob/master/CHANGELOG.md
|
|
10
|
+
Author-email: Lingkang <l.jin@tue.nl>
|
|
11
|
+
License-Expression: MIT
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Keywords: electricity,energy,liander,load profiles,open data,sbi,sqlite
|
|
14
|
+
Classifier: Development Status :: 3 - Alpha
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Intended Audience :: Science/Research
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Topic :: Database
|
|
25
|
+
Classifier: Topic :: Scientific/Engineering
|
|
26
|
+
Classifier: Typing :: Typed
|
|
27
|
+
Requires-Python: >=3.10
|
|
28
|
+
Requires-Dist: numpy>=1.23
|
|
29
|
+
Requires-Dist: pandas>=1.5
|
|
30
|
+
Provides-Extra: all
|
|
31
|
+
Requires-Dist: openpyxl>=3.1; extra == 'all'
|
|
32
|
+
Requires-Dist: pyarrow>=12; extra == 'all'
|
|
33
|
+
Provides-Extra: excel
|
|
34
|
+
Requires-Dist: openpyxl>=3.1; extra == 'excel'
|
|
35
|
+
Provides-Extra: parquet
|
|
36
|
+
Requires-Dist: pyarrow>=12; extra == 'parquet'
|
|
37
|
+
Description-Content-Type: text/markdown
|
|
38
|
+
|
|
39
|
+
# liander-open-data
|
|
40
|
+
|
|
41
|
+
[](https://pypi.org/project/liander-open-data/)
|
|
42
|
+
[](https://pypi.org/project/liander-open-data/)
|
|
43
|
+
[](https://github.com/lingkang95/Liander_open_data/actions/workflows/ci.yml)
|
|
44
|
+
|
|
45
|
+
A lightweight **SQLite database** and Python API for the
|
|
46
|
+
[Liander open data](https://www.liander.nl/over-ons/open-data) **SBI load profiles**:
|
|
47
|
+
15-minute electricity profiles per SBI code (Dutch Standard Industrial
|
|
48
|
+
Classification), together with Liander's dictionary of the *recommended
|
|
49
|
+
profile per SBI code*.
|
|
50
|
+
|
|
51
|
+
- One portable `.db` file holding all profiles and the SBI dictionary, which
|
|
52
|
+
any SQLite tool can open.
|
|
53
|
+
- A small Python API that returns `pandas` DataFrames.
|
|
54
|
+
- A `liander-open-data` command line tool.
|
|
55
|
+
- The SBI dictionary (1,427 codes) ships with the package and works without
|
|
56
|
+
any download.
|
|
57
|
+
|
|
58
|
+
## Installation
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install liander-open-data
|
|
62
|
+
# optional extras: read the original .xlsx dictionary / export Parquet
|
|
63
|
+
pip install "liander-open-data[excel,parquet]"
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## Quick start
|
|
67
|
+
|
|
68
|
+
The profile CSVs (`<SBI code>.csv`, ~1.2 GB) are not bundled. Download them
|
|
69
|
+
from Liander and build the database once:
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
liander-open-data build data/Profiles -o liander_profiles.db
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Then query it from Python:
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
from liander_open_data import ProfileDatabase
|
|
79
|
+
|
|
80
|
+
with ProfileDatabase("liander_profiles.db") as db:
|
|
81
|
+
db.lookup("8411") # dictionary entry for an SBI code
|
|
82
|
+
df = db.get_profile("8411", columns=["mean", "p90"])
|
|
83
|
+
rec = db.recommended_profile("84111") # Liander's recommended profile
|
|
84
|
+
wide = db.get_profiles(["01", "8411"], column="mean", start="2023-06-01")
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Or from the command line:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
liander-open-data lookup 8411
|
|
91
|
+
liander-open-data search onderwijs --with-profile
|
|
92
|
+
liander-open-data export 8411 -o 8411.csv --columns mean p90
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Or with plain SQL:
|
|
96
|
+
|
|
97
|
+
```sql
|
|
98
|
+
SELECT code, AVG(mean) AS avg_load
|
|
99
|
+
FROM profile_timeseries
|
|
100
|
+
WHERE timestamp BETWEEN '2023-07-01' AND '2023-08-01'
|
|
101
|
+
GROUP BY code ORDER BY avg_load DESC;
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## Documentation
|
|
105
|
+
|
|
106
|
+
Full documentation, covering the data model, CLI reference, API reference and
|
|
107
|
+
the release guide, is in the [`docs/`](docs/) folder. To browse it as a
|
|
108
|
+
website locally, run `uv run mkdocs serve`.
|
|
109
|
+
|
|
110
|
+
## Development
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
uv sync --all-groups
|
|
114
|
+
uv run pytest
|
|
115
|
+
uv run mkdocs serve # live documentation preview
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
## Data and license
|
|
119
|
+
|
|
120
|
+
The code is released under the MIT license. The profiles and SBI dictionary
|
|
121
|
+
are published by Liander N.V. as open data; check Liander's terms before
|
|
122
|
+
redistributing them.
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# liander-open-data
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/liander-open-data/)
|
|
4
|
+
[](https://pypi.org/project/liander-open-data/)
|
|
5
|
+
[](https://github.com/lingkang95/Liander_open_data/actions/workflows/ci.yml)
|
|
6
|
+
|
|
7
|
+
A lightweight **SQLite database** and Python API for the
|
|
8
|
+
[Liander open data](https://www.liander.nl/over-ons/open-data) **SBI load profiles**:
|
|
9
|
+
15-minute electricity profiles per SBI code (Dutch Standard Industrial
|
|
10
|
+
Classification), together with Liander's dictionary of the *recommended
|
|
11
|
+
profile per SBI code*.
|
|
12
|
+
|
|
13
|
+
- One portable `.db` file holding all profiles and the SBI dictionary, which
|
|
14
|
+
any SQLite tool can open.
|
|
15
|
+
- A small Python API that returns `pandas` DataFrames.
|
|
16
|
+
- A `liander-open-data` command line tool.
|
|
17
|
+
- The SBI dictionary (1,427 codes) ships with the package and works without
|
|
18
|
+
any download.
|
|
19
|
+
|
|
20
|
+
## Installation
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install liander-open-data
|
|
24
|
+
# optional extras: read the original .xlsx dictionary / export Parquet
|
|
25
|
+
pip install "liander-open-data[excel,parquet]"
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Quick start
|
|
29
|
+
|
|
30
|
+
The profile CSVs (`<SBI code>.csv`, ~1.2 GB) are not bundled. Download them
|
|
31
|
+
from Liander and build the database once:
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
liander-open-data build data/Profiles -o liander_profiles.db
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Then query it from Python:
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
from liander_open_data import ProfileDatabase
|
|
41
|
+
|
|
42
|
+
with ProfileDatabase("liander_profiles.db") as db:
|
|
43
|
+
db.lookup("8411") # dictionary entry for an SBI code
|
|
44
|
+
df = db.get_profile("8411", columns=["mean", "p90"])
|
|
45
|
+
rec = db.recommended_profile("84111") # Liander's recommended profile
|
|
46
|
+
wide = db.get_profiles(["01", "8411"], column="mean", start="2023-06-01")
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Or from the command line:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
liander-open-data lookup 8411
|
|
53
|
+
liander-open-data search onderwijs --with-profile
|
|
54
|
+
liander-open-data export 8411 -o 8411.csv --columns mean p90
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Or with plain SQL:
|
|
58
|
+
|
|
59
|
+
```sql
|
|
60
|
+
SELECT code, AVG(mean) AS avg_load
|
|
61
|
+
FROM profile_timeseries
|
|
62
|
+
WHERE timestamp BETWEEN '2023-07-01' AND '2023-08-01'
|
|
63
|
+
GROUP BY code ORDER BY avg_load DESC;
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## Documentation
|
|
67
|
+
|
|
68
|
+
Full documentation, covering the data model, CLI reference, API reference and
|
|
69
|
+
the release guide, is in the [`docs/`](docs/) folder. To browse it as a
|
|
70
|
+
website locally, run `uv run mkdocs serve`.
|
|
71
|
+
|
|
72
|
+
## Development
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
uv sync --all-groups
|
|
76
|
+
uv run pytest
|
|
77
|
+
uv run mkdocs serve # live documentation preview
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Data and license
|
|
81
|
+
|
|
82
|
+
The code is released under the MIT license. The profiles and SBI dictionary
|
|
83
|
+
are published by Liander N.V. as open data; check Liander's terms before
|
|
84
|
+
redistributing them.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
--8<-- "CHANGELOG.md"
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# Command line
|
|
2
|
+
|
|
3
|
+
Installing the package provides the `liander-open-data` command (also
|
|
4
|
+
available as `python -m liander_open_data`). Commands that read a database
|
|
5
|
+
accept `--db PATH` (default: `liander_profiles.db` in the current
|
|
6
|
+
directory).
|
|
7
|
+
|
|
8
|
+
```text
|
|
9
|
+
liander-open-data [--version] [-v] COMMAND ...
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
| Command | Purpose |
|
|
13
|
+
|--------------|-----------------------------------------------------------|
|
|
14
|
+
| `build` | Build the database from the profile CSVs |
|
|
15
|
+
| `info` | Show build metadata |
|
|
16
|
+
| `list` | List all stored profiles |
|
|
17
|
+
| `lookup` | Show the dictionary entry of one SBI code |
|
|
18
|
+
| `search` | Search codes by description text or code prefix |
|
|
19
|
+
| `export` | Export a profile to CSV or Parquet |
|
|
20
|
+
| `dictionary` | Convert the SBI workbook (or the bundled copy) to CSV |
|
|
21
|
+
|
|
22
|
+
## `build`
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
liander-open-data build PROFILES_DIR [-o OUTPUT] [-d DICTIONARY] [--overwrite] [-q]
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
- `PROFILES_DIR`: directory containing `<SBI code>.csv` files.
|
|
29
|
+
- `-o, --output`: database file to create (default `liander_profiles.db`).
|
|
30
|
+
- `-d, --dictionary`: `.xlsx` workbook or `.csv` export. Defaults to the
|
|
31
|
+
bundled dictionary.
|
|
32
|
+
- `--overwrite`: replace an existing database.
|
|
33
|
+
- `-q, --quiet`: hide the progress bar.
|
|
34
|
+
|
|
35
|
+
## `lookup`
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
$ liander-open-data lookup 8411
|
|
39
|
+
code: 8411
|
|
40
|
+
description: Algemeen overheidsbestuur
|
|
41
|
+
warning: None
|
|
42
|
+
n_timeseries: 638
|
|
43
|
+
best_profile: 841
|
|
44
|
+
standard_profile: KO_KANTOOR_ONDERWIJS
|
|
45
|
+
corr_best_profile: 0.3404778207107602
|
|
46
|
+
corr_standard_profile: 0.2784444209758644
|
|
47
|
+
ko_profile: KO_KANTOOR_ONDERWIJS
|
|
48
|
+
has_profile: True
|
|
49
|
+
recommended_profile: 841
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Unknown codes fall back to the nearest parent code.
|
|
53
|
+
|
|
54
|
+
## `search`
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
liander-open-data search onderwijs --with-profile
|
|
58
|
+
liander-open-data search 85
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## `export`
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
liander-open-data export CODE -o FILE [-c COL ...] [--start TS] [--end TS] [--recommended]
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
- The file extension of `FILE` sets the format: `.parquet` needs the
|
|
68
|
+
`parquet` extra, anything else is written as CSV.
|
|
69
|
+
- `--recommended` exports the profile Liander recommends for `CODE` instead
|
|
70
|
+
of `CODE`'s own profile.
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
liander-open-data export 8411 -o 8411_summer.csv -c mean p90 \
|
|
74
|
+
--start 2023-06-01 --end 2023-09-01
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
## `dictionary`
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
# convert a (newer) workbook to the clean CSV format
|
|
81
|
+
liander-open-data dictionary "Overzicht met per SBI-code een aanbevolen profiel.xlsx" -o sbi.csv
|
|
82
|
+
# or dump the bundled copy
|
|
83
|
+
liander-open-data dictionary -o sbi.csv
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## Exit status
|
|
87
|
+
|
|
88
|
+
`0` on success. `1` on an expected error such as a missing file, an unknown
|
|
89
|
+
code, or a bad column; the message is printed to stderr.
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
## Set up
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
git clone https://github.com/lingkang95/Liander_open_data.git
|
|
7
|
+
cd Liander_open_data
|
|
8
|
+
uv sync --all-groups
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Everyday commands
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
uv run pytest # tests + doctests
|
|
15
|
+
uv run pytest --cov # with coverage
|
|
16
|
+
uv run ruff check . # lint
|
|
17
|
+
uv run ruff format . # format
|
|
18
|
+
uv run mkdocs serve # docs at http://127.0.0.1:8000
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
The tests use small synthetic CSVs and need no real Liander data.
|
|
22
|
+
|
|
23
|
+
## Project layout
|
|
24
|
+
|
|
25
|
+
```text
|
|
26
|
+
src/liander_open_data/
|
|
27
|
+
├── __init__.py # public API
|
|
28
|
+
├── database.py # schema, build_database, ProfileDatabase
|
|
29
|
+
├── dictionary.py # SBI workbook parsing, SBICode
|
|
30
|
+
├── cli.py # liander-open-data command
|
|
31
|
+
└── data/
|
|
32
|
+
└── sbi_dictionary.csv # bundled, cleaned dictionary
|
|
33
|
+
tests/ # pytest suite
|
|
34
|
+
docs/ # this documentation (MkDocs Material)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Updating the bundled dictionary
|
|
38
|
+
|
|
39
|
+
When Liander publishes a new workbook:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
uv run liander-open-data dictionary path/to/new.xlsx -o src/liander_open_data/data/sbi_dictionary.csv
|
|
43
|
+
uv run pytest
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Review the diff of the CSV, add a changelog entry and release a new version.
|
|
47
|
+
|
|
48
|
+
## Changing the schema
|
|
49
|
+
|
|
50
|
+
If you change the layout of the tables in `database.py`, bump
|
|
51
|
+
`SCHEMA_VERSION`, update [Data model](data-model.md), and note the change in
|
|
52
|
+
the changelog.
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# Data model
|
|
2
|
+
|
|
3
|
+
The database is a plain SQLite 3 file. You can open it with
|
|
4
|
+
`ProfileDatabase`, with `sqlite3.connect()`, or with any SQL tool.
|
|
5
|
+
|
|
6
|
+
```mermaid
|
|
7
|
+
erDiagram
|
|
8
|
+
sbi_codes ||--o| profiles : "has"
|
|
9
|
+
profiles ||--|{ profile_values : "contains"
|
|
10
|
+
timestamps ||--|{ profile_values : "indexes"
|
|
11
|
+
|
|
12
|
+
sbi_codes {
|
|
13
|
+
TEXT code PK
|
|
14
|
+
TEXT description
|
|
15
|
+
INTEGER level
|
|
16
|
+
TEXT parent_code
|
|
17
|
+
TEXT warning
|
|
18
|
+
INTEGER n_timeseries
|
|
19
|
+
TEXT best_profile
|
|
20
|
+
TEXT standard_profile
|
|
21
|
+
REAL corr_best_profile
|
|
22
|
+
REAL corr_standard_profile
|
|
23
|
+
TEXT ko_profile
|
|
24
|
+
INTEGER has_profile
|
|
25
|
+
}
|
|
26
|
+
profiles {
|
|
27
|
+
TEXT code PK
|
|
28
|
+
INTEGER n_points
|
|
29
|
+
TEXT start
|
|
30
|
+
TEXT end
|
|
31
|
+
INTEGER interval_minutes
|
|
32
|
+
TEXT source_file
|
|
33
|
+
}
|
|
34
|
+
timestamps {
|
|
35
|
+
INTEGER slot PK
|
|
36
|
+
TEXT timestamp
|
|
37
|
+
}
|
|
38
|
+
profile_values {
|
|
39
|
+
TEXT code PK
|
|
40
|
+
INTEGER slot PK
|
|
41
|
+
REAL min
|
|
42
|
+
REAL p10
|
|
43
|
+
REAL median
|
|
44
|
+
REAL mean
|
|
45
|
+
REAL p90
|
|
46
|
+
REAL max
|
|
47
|
+
REAL median_renorm
|
|
48
|
+
REAL mean_renorm
|
|
49
|
+
REAL p90_renorm
|
|
50
|
+
}
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Tables
|
|
54
|
+
|
|
55
|
+
### `sbi_codes`
|
|
56
|
+
|
|
57
|
+
The full SBI dictionary, with one row per code. See [The data](data.md#sbi-dictionary)
|
|
58
|
+
for the meaning of each column. Two columns are derived:
|
|
59
|
+
|
|
60
|
+
- `level`: number of digits (2 to 5).
|
|
61
|
+
- `parent_code`: the code one level up (`8411` → `841`), or `NULL` for
|
|
62
|
+
2-digit codes.
|
|
63
|
+
|
|
64
|
+
### `profiles`
|
|
65
|
+
|
|
66
|
+
One row per profile loaded from a CSV: number of points, first and last
|
|
67
|
+
timestamp, and the detected interval in minutes.
|
|
68
|
+
|
|
69
|
+
### `timestamps`
|
|
70
|
+
|
|
71
|
+
The time axis shared by all profiles. Storing an integer `slot` per value
|
|
72
|
+
instead of the full timestamp string keeps the database compact.
|
|
73
|
+
Timestamps use ISO format `YYYY-MM-DD HH:MM:SS`, so string comparison
|
|
74
|
+
matches chronological order.
|
|
75
|
+
|
|
76
|
+
### `profile_values`
|
|
77
|
+
|
|
78
|
+
The measurements, one row per `(code, slot)`. This is a `WITHOUT ROWID`
|
|
79
|
+
table clustered on its primary key, so reading one profile is a single range
|
|
80
|
+
scan.
|
|
81
|
+
|
|
82
|
+
### `profile_timeseries` (view)
|
|
83
|
+
|
|
84
|
+
`profile_values` joined with `timestamps`, for convenient ad-hoc SQL:
|
|
85
|
+
|
|
86
|
+
```sql
|
|
87
|
+
SELECT timestamp, mean, p90
|
|
88
|
+
FROM profile_timeseries
|
|
89
|
+
WHERE code = '8411' AND timestamp >= '2023-12-25' AND timestamp < '2023-12-27';
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
### `metadata`
|
|
93
|
+
|
|
94
|
+
Key/value pairs describing the build: `schema_version`, `package_version`,
|
|
95
|
+
`built_at`, `n_profiles`, `n_sbi_codes`, `n_timestamps` and `value_columns`.
|
|
96
|
+
|
|
97
|
+
## Size and performance
|
|
98
|
+
|
|
99
|
+
For the 2023 dataset (203 profiles × 35,040 points × 9 statistics):
|
|
100
|
+
|
|
101
|
+
| Item | Value |
|
|
102
|
+
|-------------------------------------|-------------|
|
|
103
|
+
| Source CSVs | ≈ 1.2 GB |
|
|
104
|
+
| SQLite database | ≈ 717 MB |
|
|
105
|
+
| Build time | < 1 min |
|
|
106
|
+
| `get_profile()` (one full year) | ≈ 0.1 s |
|
|
107
|
+
| `get_profiles()` (all, one column) | ≈ 5 s |
|
|
108
|
+
|
|
109
|
+
## Schema versioning
|
|
110
|
+
|
|
111
|
+
`SCHEMA_VERSION` is stored in the `metadata` table. When a new release
|
|
112
|
+
changes the layout incompatibly, the version is bumped and
|
|
113
|
+
`ProfileDatabase` logs a warning when it opens an older file. Rebuild with
|
|
114
|
+
`--overwrite` to upgrade.
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
# The data
|
|
2
|
+
|
|
3
|
+
## Profile CSVs
|
|
4
|
+
|
|
5
|
+
Each file `<SBI code>.csv` contains one calendar year (2023) of 15-minute
|
|
6
|
+
values, 35,040 rows in total. Each row holds statistics over all metered
|
|
7
|
+
connections of businesses registered under that SBI code:
|
|
8
|
+
|
|
9
|
+
| Column | Meaning |
|
|
10
|
+
|-----------------|----------------------------------------------------------------|
|
|
11
|
+
| *(index)* | Timestamp of the start of the 15-minute interval |
|
|
12
|
+
| `min` | Minimum over all connections |
|
|
13
|
+
| `p10` | 10th percentile |
|
|
14
|
+
| `median` | Median |
|
|
15
|
+
| `mean` | Mean |
|
|
16
|
+
| `p90` | 90th percentile |
|
|
17
|
+
| `max` | Maximum |
|
|
18
|
+
| `median_renorm` | Median, renormalised |
|
|
19
|
+
| `mean_renorm` | Mean, renormalised |
|
|
20
|
+
| `p90_renorm` | 90th percentile, renormalised |
|
|
21
|
+
|
|
22
|
+
Values are normalised and dimensionless, in the range 0 to 1. Use them as a
|
|
23
|
+
*shape* and scale it to an actual consumption (see [Recipes](recipes.md)).
|
|
24
|
+
|
|
25
|
+
!!! info "Timestamps"
|
|
26
|
+
The source files contain exactly 365 × 96 = 35,040 points with no
|
|
27
|
+
daylight-saving gaps or duplicates. Timestamps are stored as given,
|
|
28
|
+
without a time zone.
|
|
29
|
+
|
|
30
|
+
SBI codes are hierarchical: `84` → `842` → `8423` → `84232`. Profiles exist
|
|
31
|
+
at several levels, and a profile at a higher level aggregates more
|
|
32
|
+
connections.
|
|
33
|
+
|
|
34
|
+
## SBI dictionary
|
|
35
|
+
|
|
36
|
+
The workbook *Overzicht met per SBI-code een aanbevolen profiel.xlsx* lists
|
|
37
|
+
all 1,427 SBI codes. The package renames its columns to English:
|
|
38
|
+
|
|
39
|
+
| Workbook column | Field | Description |
|
|
40
|
+
|-------------------------------------|-------------------------|--------------------------------------------------------------|
|
|
41
|
+
| `code` | `code` | SBI code (string, leading zeros kept) |
|
|
42
|
+
| `description` | `description` | Dutch description |
|
|
43
|
+
| `Warning` | `warning` | Set when < 10 time series were available (less reliable) |
|
|
44
|
+
| `Tijdseries in deze SBI` | `n_timeseries` | Number of connections behind the profile |
|
|
45
|
+
| `Profiel hoogste correlatie` | `best_profile` | Profile with the highest correlation (SBI code or `KO_*`) |
|
|
46
|
+
| `Standaard profiel` | `standard_profile` | Standard `KO_*` profile for the sector |
|
|
47
|
+
| `correlatie geaggregeerde profiel` | `corr_best_profile` | Correlation with `best_profile` |
|
|
48
|
+
| `correlatie standaard profiel` | `corr_standard_profile` | Correlation with `standard_profile` |
|
|
49
|
+
| `KO_profiel` | `ko_profile` | Recommended standard profile |
|
|
50
|
+
| `Profiel aanwezig` | `has_profile` | A measured profile exists for this exact code |
|
|
51
|
+
|
|
52
|
+
### Cleaning rules
|
|
53
|
+
|
|
54
|
+
The parser applies these rules to the raw workbook:
|
|
55
|
+
|
|
56
|
+
- Codes are kept as strings. Codes that lost their leading zero in Excel
|
|
57
|
+
(`2`, `6`, `8`, `9`) are restored to `02`, `06`, `08`, `09`.
|
|
58
|
+
- `"NA"`, `"nan"`, empty cells and the placeholder `0` in text columns become
|
|
59
|
+
`None`.
|
|
60
|
+
- A correlation of exactly `0` is a placeholder for *not computed* and
|
|
61
|
+
becomes `None`.
|
|
62
|
+
- Leading and trailing whitespace is stripped from descriptions.
|
|
63
|
+
- In the database, `has_profile` is recomputed from the CSV files actually
|
|
64
|
+
loaded, so it always matches the `profiles` table.
|
|
65
|
+
|
|
66
|
+
### Standard profiles (`KO_*`)
|
|
67
|
+
|
|
68
|
+
Values such as `KO_INDUSTRIE`, `KO_KANTOOR_ONDERWIJS`, `KO_AGRARIER`,
|
|
69
|
+
`KO_GLASTUINBOUW`, `KO_LOGISTIEK` and `KO_OVERIG` refer to standard
|
|
70
|
+
profiles. They are **not** included in the profile CSVs. `Onbekend` means
|
|
71
|
+
*unknown*, and `Onbekend, geen eans` means *unknown, no connections (EANs)*.
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# liander-open-data
|
|
2
|
+
|
|
3
|
+
**liander-open-data** packs the
|
|
4
|
+
[Liander open data](https://www.liander.nl/over-ons/open-data) *SBI load
|
|
5
|
+
profiles* into a single, lightweight **SQLite database** and gives you a
|
|
6
|
+
small Python API and command line tool to query it.
|
|
7
|
+
|
|
8
|
+
Liander, a Dutch distribution system operator, publishes 15-minute
|
|
9
|
+
electricity consumption profiles aggregated per **SBI code** (*Standaard
|
|
10
|
+
Bedrijfsindeling*, the Dutch Standard Industrial Classification), plus a
|
|
11
|
+
workbook that tells you **which profile to use for each SBI code**. This
|
|
12
|
+
package combines both.
|
|
13
|
+
|
|
14
|
+
## Highlights
|
|
15
|
+
|
|
16
|
+
- **One file, no server.** All 203 profiles (≈7 million rows) and 1,427 SBI
|
|
17
|
+
codes go into one portable `.db` file. Python, the `sqlite3` CLI,
|
|
18
|
+
DBeaver, DuckDB and R can all open it.
|
|
19
|
+
- **Lossless.** Values are stored as 64-bit floats and round-trip exactly to
|
|
20
|
+
the source CSVs.
|
|
21
|
+
- **Fast enough.** One full-year profile loads in about 0.1 s, and all
|
|
22
|
+
profiles for a single statistic load in about 5 s.
|
|
23
|
+
- **Batteries included.** The cleaned SBI dictionary ships with the package,
|
|
24
|
+
so `load_dictionary()` works without any download.
|
|
25
|
+
- **Typed, tested, documented.**
|
|
26
|
+
|
|
27
|
+
## At a glance
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
from liander_open_data import ProfileDatabase
|
|
31
|
+
|
|
32
|
+
with ProfileDatabase("liander_profiles.db") as db:
|
|
33
|
+
entry = db.lookup("8411")
|
|
34
|
+
print(entry.description, "->", entry.recommended_profile)
|
|
35
|
+
# Algemeen overheidsbestuur -> 841
|
|
36
|
+
|
|
37
|
+
df = db.recommended_profile("8411", columns=["mean", "p90"])
|
|
38
|
+
df.resample("D").mean().plot()
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
```mermaid
|
|
42
|
+
flowchart LR
|
|
43
|
+
A["Profiles/*.csv<br/>(1.2 GB)"] --> C[build_database]
|
|
44
|
+
B["SBI dictionary<br/>(.xlsx / bundled)"] --> C
|
|
45
|
+
C --> D[("liander_profiles.db<br/>SQLite")]
|
|
46
|
+
D --> E[ProfileDatabase API]
|
|
47
|
+
D --> F[liander-open-data CLI]
|
|
48
|
+
D --> G[Any SQL tool]
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Next: [Installation](installation.md) · [Quick start](quickstart.md)
|