ucgmsim-imdb 2026.9.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,10 @@
1
+ name: CI
2
+ on: [pull_request]
3
+ jobs:
4
+ ci:
5
+ uses: ucgmsim/meta-ci-action/.github/workflows/ci.yml@main
6
+ with:
7
+ package-dir: imdb
8
+ uv-extra-args: "--all-groups"
9
+ enable-coverage: true
10
+ cov-package: imdb
@@ -0,0 +1,19 @@
1
+ name: Claude PR Review
2
+
3
+ # Triggered by a "@claude review" comment on a pull request.
4
+ #
5
+ # Why issue_comment and not pull_request: GitHub withholds secrets from runs
6
+ # triggered by fork pull requests, so a pull_request trigger would silently
7
+ # fail on exactly the PRs that most need review. issue_comment runs in the base
8
+ # repository's context with secrets available. The action independently checks
9
+ # that the commenting user has write access before doing anything, so the gate
10
+ # is "a member of this org asked for this review".
11
+ on:
12
+ issue_comment:
13
+ types: [created]
14
+
15
+ jobs:
16
+ review:
17
+ uses: ucgmsim/meta-ci-action/.github/workflows/claude-review.yml@main
18
+ secrets:
19
+ claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
@@ -0,0 +1,63 @@
1
+ name: Publish to PyPI
2
+ run-name: Publish Python Package to PyPI for release ${{ github.event.release.tag_name || inputs.tag_name || github.ref_name }}
3
+
4
+ on:
5
+ release:
6
+ types: [published]
7
+ workflow_dispatch:
8
+ inputs:
9
+ tag_name:
10
+ description: "Git tag to checkout and publish"
11
+ required: false
12
+ type: string
13
+
14
+ jobs:
15
+ build:
16
+ name: Build distribution 📦
17
+ runs-on: ubuntu-latest
18
+
19
+ steps:
20
+ - name: Checkout code
21
+ uses: actions/checkout@v4
22
+ with:
23
+ ref: ${{ inputs.tag_name || github.ref }}
24
+ fetch-depth: 0
25
+
26
+ - name: Set up Python
27
+ uses: actions/setup-python@v5
28
+ with:
29
+ python-version: "3.12"
30
+ - name: Install pypa/build
31
+ run: >-
32
+ python3 -m
33
+ pip install
34
+ build
35
+ --user
36
+ - name: Build a binary wheel and a source tarball
37
+ run: python3 -m build
38
+ - name: Store the distribution packages
39
+ uses: actions/upload-artifact@v4
40
+ with:
41
+ name: python-package-distributions
42
+ path: dist/
43
+ publish-to-pypi:
44
+ name: Publish to PyPI
45
+ needs:
46
+ - build
47
+ runs-on: ubuntu-latest
48
+
49
+ environment:
50
+ name: pypi
51
+ url: https://pypi.org/p/ucgmsim-imdb
52
+
53
+ permissions:
54
+ id-token: write
55
+
56
+ steps:
57
+ - name: Download all the dists
58
+ uses: actions/download-artifact@v4
59
+ with:
60
+ name: python-package-distributions
61
+ path: dist/
62
+ - name: Publish distribution to PyPI
63
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,181 @@
1
+ .DS_Store
2
+ .idea
3
+ *.log
4
+ tmp/
5
+ # Created by https://www.toptal.com/developers/gitignore/api/python
6
+ # Edit at https://www.toptal.com/developers/gitignore?templates=python
7
+
8
+ ### Python ###
9
+ # Byte-compiled / optimized / DLL files
10
+ __pycache__/
11
+
12
+ *.py[cod]
13
+ *$py.class
14
+
15
+ # C extensions
16
+ *.so
17
+
18
+ # Distribution / packaging
19
+ .Python
20
+ build/
21
+ develop-eggs/
22
+ dist/
23
+ downloads/
24
+ eggs/
25
+ .eggs/
26
+ lib/
27
+ lib64/
28
+ parts/
29
+ sdist/
30
+ var/
31
+ wheels/
32
+ share/python-wheels/
33
+ *.egg-info/
34
+ .installed.cfg
35
+ *.egg
36
+ MANIFEST
37
+
38
+ # PyInstaller
39
+ # Usually these files are written by a python script from a template
40
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
41
+ *.manifest
42
+ *.spec
43
+
44
+ # Installer logs
45
+ pip-log.txt
46
+ pip-delete-this-directory.txt
47
+
48
+ # Unit test / coverage reports
49
+ htmlcov/
50
+ .tox/
51
+ .nox/
52
+ .coverage
53
+ .coverage.*
54
+ .cache
55
+ nosetests.xml
56
+ coverage.xml
57
+ *.cover
58
+ *.py,cover
59
+ .hypothesis/
60
+ .pytest_cache/
61
+ cover/
62
+
63
+ # Translations
64
+ *.mo
65
+ *.pot
66
+
67
+ # Django stuff:
68
+ *.log
69
+ local_settings.py
70
+ db.sqlite3
71
+ db.sqlite3-journal
72
+
73
+ # Flask stuff:
74
+ instance/
75
+ .webassets-cache
76
+
77
+ # Scrapy stuff:
78
+ .scrapy
79
+
80
+ # Sphinx documentation
81
+ docs/_build/
82
+
83
+ # PyBuilder
84
+ .pybuilder/
85
+ target/
86
+
87
+ # Jupyter Notebook
88
+ .ipynb_checkpoints
89
+
90
+ # IPython
91
+ profile_default/
92
+ ipython_config.py
93
+
94
+ # pyenv
95
+ # For a library or package, you might want to ignore these files since the code is
96
+ # intended to run in multiple environments; otherwise, check them in:
97
+ # .python-version
98
+
99
+ # pipenv
100
+ # According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
101
+ # However, in case of collaboration, if having platform-specific dependencies or dependencies
102
+ # having no cross-platform support, pipenv may install dependencies that don't work, or not
103
+ # install all needed dependencies.
104
+ #Pipfile.lock
105
+
106
+ # poetry
107
+ # Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
108
+ # This is especially recommended for binary packages to ensure reproducibility, and is more
109
+ # commonly ignored for libraries.
110
+ # https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
111
+ #poetry.lock
112
+
113
+ # pdm
114
+ # Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
115
+ #pdm.lock
116
+ # pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
117
+ # in version control.
118
+ # https://pdm.fming.dev/#use-with-ide
119
+ .pdm.toml
120
+
121
+ # PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
122
+ __pypackages__/
123
+
124
+ # Celery stuff
125
+ celerybeat-schedule
126
+ celerybeat.pid
127
+
128
+ # SageMath parsed files
129
+ *.sage.py
130
+
131
+ # Environments
132
+ .env
133
+ .venv
134
+ env/
135
+ venv/
136
+ ENV/
137
+ env.bak/
138
+ venv.bak/
139
+
140
+ # Spyder project settings
141
+ .spyderproject
142
+ .spyproject
143
+
144
+ # Rope project settings
145
+ .ropeproject
146
+
147
+ # mkdocs documentation
148
+ /site
149
+
150
+ # mypy
151
+ .mypy_cache/
152
+ .dmypy.json
153
+ dmypy.json
154
+
155
+ # Pyre type checker
156
+ .pyre/
157
+
158
+ # pytype static type analyzer
159
+ .pytype/
160
+
161
+ # Cython debug symbols
162
+ cython_debug/
163
+
164
+ # PyCharm
165
+ # JetBrains specific template is maintained in a separate JetBrains.gitignore that can
166
+ # be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
167
+ # and can be added to the global gitignore or merged into this file. For a more nuclear
168
+ # option (not recommended) you can uncomment the following to ignore the entire idea folder.
169
+ #.idea/
170
+
171
+ ### Python Patch ###
172
+ # Poetry local configuration file - https://python-poetry.org/docs/configuration/#local-configuration
173
+ poetry.toml
174
+
175
+ # ruff
176
+ .ruff_cache/
177
+
178
+ # LSP config files
179
+ pyrightconfig.json
180
+
181
+ # End of https://www.toptal.com/developers/gitignore/api/python
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 UC Eq Engineering
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,196 @@
1
+ Metadata-Version: 2.4
2
+ Name: ucgmsim-imdb
3
+ Version: 2026.9.1
4
+ Summary: A library for reading and writing intensity measure databases
5
+ Author: ucgmsim
6
+ Requires-Python: >=3.12
7
+ Description-Content-Type: text/markdown
8
+ License-File: LICENSE
9
+ Requires-Dist: ibis-framework[duckdb]>=9
10
+ Requires-Dist: numpy>=2
11
+ Requires-Dist: pandas>=3
12
+ Dynamic: license-file
13
+
14
+ # IMDB
15
+
16
+ A library for reading and writing intensity measure databases (IMDBs), DuckDB
17
+ databases of intensity measures (IMs) from physics-based ground-motion simulation,
18
+ empirical ground-motion model (GMM) prediction, and observed ground motion.
19
+
20
+ One database per run set. Every database uses the same schema unchanged and is
21
+ self-contained. A database may mix `kind`s of record freely, distinguished per row. Schema is documented in full in `imdb/schema.py` (the single source of truth for the DDL); this README summarises it.
22
+
23
+ ## Schema
24
+
25
+ Thirteen tables: four dimensions (`events`, `realisations`, `sites`, `site_event`),
26
+ one identity table (`records`), three IM tables (`psa_ims`, `fas_ims`,
27
+ `scalars_ims`), two IM vocabulary tables (`periods`, `frequencies`), and three
28
+ documentation tables (`db_meta`, `im_units`, `notes`).
29
+
30
+ A ground motion is identified by `(rel_id, site_id, component, kind, gmm_key)`.
31
+ Response spectra and Fourier spectra are stored as one array per record; scalar IMs
32
+ as named columns. Every IM column has a paired `<IM>_sigma` column/array (ln-space
33
+ total standard deviation), populated for `kind = "gmm"` records and NULL for
34
+ `"simulated"`/`"observed"`.
35
+
36
+ ```mermaid
37
+ erDiagram
38
+ events ||--o{ realisations : "FK, declared"
39
+ events ||--o{ site_event : "logical"
40
+ sites ||--o{ site_event : "logical"
41
+ realisations ||--o{ records : "logical"
42
+ sites ||--o{ records : "logical"
43
+ records ||--o| psa_ims : "record_int_id"
44
+ records ||--o| fas_ims : "record_int_id"
45
+ records ||--o| scalars_ims : "record_int_id"
46
+ periods ||--o{ psa_ims : "period_index indexes pSA[]"
47
+ frequencies ||--o{ fas_ims : "freq_index indexes FAS[]"
48
+
49
+ events {
50
+ INTEGER event_int_id PK
51
+ VARCHAR event_id UK "stable identity"
52
+ FLOAT magnitude
53
+ tect_type_t tect_type "ENUM, 4 values"
54
+ FLOAT dip
55
+ FLOAT dip_dir
56
+ FLOAT dtop
57
+ FLOAT dbottom
58
+ FLOAT length
59
+ VARCHAR source_wkt
60
+ VARCHAR trace_wkt
61
+ VARCHAR domain_wkt
62
+ VARCHAR metadata "JSON"
63
+ }
64
+
65
+ realisations {
66
+ INTEGER rel_int_id PK
67
+ VARCHAR rel_id UK "stable identity"
68
+ INTEGER event_int_id FK
69
+ FLOAT magnitude
70
+ FLOAT rake
71
+ FLOAT hypo_lat
72
+ FLOAT hypo_lon
73
+ FLOAT hypo_depth
74
+ VARCHAR metadata "JSON"
75
+ }
76
+
77
+ sites {
78
+ INTEGER site_int_id PK
79
+ VARCHAR site_id UK "stable identity"
80
+ FLOAT lat
81
+ FLOAT lon
82
+ FLOAT vs30 "m/s"
83
+ FLOAT z1p0 "km"
84
+ FLOAT z2p5 "km"
85
+ VARCHAR metadata "JSON"
86
+ }
87
+
88
+ site_event {
89
+ INTEGER site_int_id "logical key"
90
+ INTEGER event_int_id "logical key"
91
+ FLOAT rrup "km, event level"
92
+ FLOAT rjb "km, event level"
93
+ FLOAT rx "km, event level"
94
+ FLOAT ry "km, event level"
95
+ VARCHAR metadata "JSON"
96
+ }
97
+
98
+ records {
99
+ BIGINT record_int_id "nextval, file-local, no PK"
100
+ INTEGER event_int_id "denormalised, derived from rel_int_id"
101
+ INTEGER rel_int_id "logical key"
102
+ INTEGER site_int_id "logical key"
103
+ VARCHAR component "logical key"
104
+ record_kind_t kind "ENUM: simulated, gmm, observed. logical key"
105
+ VARCHAR gmm_key "logical key. NULL unless kind=gmm"
106
+ }
107
+
108
+ psa_ims {
109
+ BIGINT record_int_id "no row means no pSA"
110
+ FLOAT_ARRAY pSA "one array per record"
111
+ FLOAT_ARRAY pSA_sigma "ln-space total sigma, same grid as pSA"
112
+ }
113
+
114
+ fas_ims {
115
+ BIGINT record_int_id "no row means no FAS"
116
+ FLOAT_ARRAY FAS "one array per record"
117
+ FLOAT_ARRAY FAS_sigma "ln-space total sigma, same grid as FAS"
118
+ }
119
+
120
+ scalars_ims {
121
+ BIGINT record_int_id "no row means no scalars"
122
+ FLOAT PGA "g"
123
+ FLOAT PGV "cm/s"
124
+ FLOAT PGD "cm"
125
+ FLOAT CAV "m/s, NULL on rotd"
126
+ FLOAT AI "m/s, NULL on rotd"
127
+ FLOAT Ds575 "s, NULL on rotd"
128
+ FLOAT Ds595 "s, NULL on rotd"
129
+ }
130
+ ```
131
+
132
+ *(`db_meta`, `notes`, `im_units`, `periods` and `frequencies` are omitted from the
133
+ diagram above for space; see `imdb/schema.py` for the full DDL, including the
134
+ `_sigma` column on every scalar IM.)*
135
+
136
+ ### Key conventions
137
+
138
+ - **Identity**: `event_id`, `rel_id` and `site_id` are stable. The integer
139
+ surrogates (`event_int_id`, `rel_int_id`, `site_int_id`, `record_int_id`) are
140
+ assigned at ingest and change on rebuild; nothing outside the database may
141
+ reference them.
142
+ - **Array indexing is 1-based**: `periods.period_index` and `frequencies.freq_index`
143
+ match DuckDB list indexing, so `pSA[period_index]` and `FAS[freq_index]` need no
144
+ offset.
145
+ - **IM coverage is row presence**: a record has at most one row in each of
146
+ `psa_ims`, `fas_ims` and `scalars_ims`. A missing row means that IM type is not
147
+ held for that record, not NULL.
148
+ - **Components**: `000`, `090`, `ver`, `geom`, `rotd0`, `rotd50`, `rotd100`. A
149
+ database may hold any subset, listed in `db_meta.components`; the writer
150
+ validates against it. `scalars_ims.CAV`, `AI`, `Ds575` and `Ds595` are NULL for
151
+ `rotd*` components; `PGA`, `PGV` and `PGD` are populated for every component.
152
+ - **Record kind**: `simulated` (physics-based simulation), `gmm` (empirical GMM
153
+ prediction) or `observed` (recorded ground motion). `gmm_key` identifies the
154
+ model, e.g. `"Bradley_2013"`, and is NULL unless `kind = "gmm"`.
155
+ - **Units**: linear, physical units; log is a read-time transform (`g` for pSA/PGA,
156
+ `cm/s` for PGV, `cm` for PGD, `m/s` for CAV/AI, `s` for Ds575/Ds595, see
157
+ `im_units`). Every `_sigma` column/array is the exception: ln-space total
158
+ standard deviation, dimensionless.
159
+ - **Constraints**: `PRIMARY KEY`/`UNIQUE`/`FOREIGN KEY` appear only on the four
160
+ dimension tables. The large tables (`site_event`, `records`, `psa_ims`,
161
+ `fas_ims`, `scalars_ims`) have none; their logical keys are documented in
162
+ `notes` and enforced by the writer, not the schema.
163
+
164
+ ## Usage
165
+
166
+ ```python
167
+ from imdb import IMDB
168
+
169
+ # read
170
+ with IMDB("run_set.duckdb") as db:
171
+ records = db.get_records(event_ids=["event1"], component="rotd50")
172
+ psa = db.get_psa(periods=[0.1, 1.0], event_ids=["event1"])
173
+ scalars = db.get_scalars(ims=["PGA", "PGV"])
174
+
175
+ # write
176
+ db = IMDB.create("new.duckdb", periods=[0.1, 0.2, 1.0])
177
+ db.add_events(events_df)
178
+ db.add_realisations(realisations_df)
179
+ db.add_sites(sites_df)
180
+ db.add_site_event(site_event_df)
181
+ db.add_records(
182
+ records_df
183
+ ) # rel_id, site_id, component, kind, pSA, FAS, scalar IM columns
184
+ db.validate()
185
+ db.close()
186
+ ```
187
+
188
+ ## Development
189
+
190
+ ```
191
+ uv sync --all-groups
192
+ uv run pytest -q
193
+ uv run ruff check
194
+ uv run ruff format
195
+ uv run ty check
196
+ ```
@@ -0,0 +1,183 @@
1
+ # IMDB
2
+
3
+ A library for reading and writing intensity measure databases (IMDBs), DuckDB
4
+ databases of intensity measures (IMs) from physics-based ground-motion simulation,
5
+ empirical ground-motion model (GMM) prediction, and observed ground motion.
6
+
7
+ One database per run set. Every database uses the same schema unchanged and is
8
+ self-contained. A database may mix `kind`s of record freely, distinguished per row. Schema is documented in full in `imdb/schema.py` (the single source of truth for the DDL); this README summarises it.
9
+
10
+ ## Schema
11
+
12
+ Thirteen tables: four dimensions (`events`, `realisations`, `sites`, `site_event`),
13
+ one identity table (`records`), three IM tables (`psa_ims`, `fas_ims`,
14
+ `scalars_ims`), two IM vocabulary tables (`periods`, `frequencies`), and three
15
+ documentation tables (`db_meta`, `im_units`, `notes`).
16
+
17
+ A ground motion is identified by `(rel_id, site_id, component, kind, gmm_key)`.
18
+ Response spectra and Fourier spectra are stored as one array per record; scalar IMs
19
+ as named columns. Every IM column has a paired `<IM>_sigma` column/array (ln-space
20
+ total standard deviation), populated for `kind = "gmm"` records and NULL for
21
+ `"simulated"`/`"observed"`.
22
+
23
+ ```mermaid
24
+ erDiagram
25
+ events ||--o{ realisations : "FK, declared"
26
+ events ||--o{ site_event : "logical"
27
+ sites ||--o{ site_event : "logical"
28
+ realisations ||--o{ records : "logical"
29
+ sites ||--o{ records : "logical"
30
+ records ||--o| psa_ims : "record_int_id"
31
+ records ||--o| fas_ims : "record_int_id"
32
+ records ||--o| scalars_ims : "record_int_id"
33
+ periods ||--o{ psa_ims : "period_index indexes pSA[]"
34
+ frequencies ||--o{ fas_ims : "freq_index indexes FAS[]"
35
+
36
+ events {
37
+ INTEGER event_int_id PK
38
+ VARCHAR event_id UK "stable identity"
39
+ FLOAT magnitude
40
+ tect_type_t tect_type "ENUM, 4 values"
41
+ FLOAT dip
42
+ FLOAT dip_dir
43
+ FLOAT dtop
44
+ FLOAT dbottom
45
+ FLOAT length
46
+ VARCHAR source_wkt
47
+ VARCHAR trace_wkt
48
+ VARCHAR domain_wkt
49
+ VARCHAR metadata "JSON"
50
+ }
51
+
52
+ realisations {
53
+ INTEGER rel_int_id PK
54
+ VARCHAR rel_id UK "stable identity"
55
+ INTEGER event_int_id FK
56
+ FLOAT magnitude
57
+ FLOAT rake
58
+ FLOAT hypo_lat
59
+ FLOAT hypo_lon
60
+ FLOAT hypo_depth
61
+ VARCHAR metadata "JSON"
62
+ }
63
+
64
+ sites {
65
+ INTEGER site_int_id PK
66
+ VARCHAR site_id UK "stable identity"
67
+ FLOAT lat
68
+ FLOAT lon
69
+ FLOAT vs30 "m/s"
70
+ FLOAT z1p0 "km"
71
+ FLOAT z2p5 "km"
72
+ VARCHAR metadata "JSON"
73
+ }
74
+
75
+ site_event {
76
+ INTEGER site_int_id "logical key"
77
+ INTEGER event_int_id "logical key"
78
+ FLOAT rrup "km, event level"
79
+ FLOAT rjb "km, event level"
80
+ FLOAT rx "km, event level"
81
+ FLOAT ry "km, event level"
82
+ VARCHAR metadata "JSON"
83
+ }
84
+
85
+ records {
86
+ BIGINT record_int_id "nextval, file-local, no PK"
87
+ INTEGER event_int_id "denormalised, derived from rel_int_id"
88
+ INTEGER rel_int_id "logical key"
89
+ INTEGER site_int_id "logical key"
90
+ VARCHAR component "logical key"
91
+ record_kind_t kind "ENUM: simulated, gmm, observed. logical key"
92
+ VARCHAR gmm_key "logical key. NULL unless kind=gmm"
93
+ }
94
+
95
+ psa_ims {
96
+ BIGINT record_int_id "no row means no pSA"
97
+ FLOAT_ARRAY pSA "one array per record"
98
+ FLOAT_ARRAY pSA_sigma "ln-space total sigma, same grid as pSA"
99
+ }
100
+
101
+ fas_ims {
102
+ BIGINT record_int_id "no row means no FAS"
103
+ FLOAT_ARRAY FAS "one array per record"
104
+ FLOAT_ARRAY FAS_sigma "ln-space total sigma, same grid as FAS"
105
+ }
106
+
107
+ scalars_ims {
108
+ BIGINT record_int_id "no row means no scalars"
109
+ FLOAT PGA "g"
110
+ FLOAT PGV "cm/s"
111
+ FLOAT PGD "cm"
112
+ FLOAT CAV "m/s, NULL on rotd"
113
+ FLOAT AI "m/s, NULL on rotd"
114
+ FLOAT Ds575 "s, NULL on rotd"
115
+ FLOAT Ds595 "s, NULL on rotd"
116
+ }
117
+ ```
118
+
119
+ *(`db_meta`, `notes`, `im_units`, `periods` and `frequencies` are omitted from the
120
+ diagram above for space; see `imdb/schema.py` for the full DDL, including the
121
+ `_sigma` column on every scalar IM.)*
122
+
123
+ ### Key conventions
124
+
125
+ - **Identity**: `event_id`, `rel_id` and `site_id` are stable. The integer
126
+ surrogates (`event_int_id`, `rel_int_id`, `site_int_id`, `record_int_id`) are
127
+ assigned at ingest and change on rebuild; nothing outside the database may
128
+ reference them.
129
+ - **Array indexing is 1-based**: `periods.period_index` and `frequencies.freq_index`
130
+ match DuckDB list indexing, so `pSA[period_index]` and `FAS[freq_index]` need no
131
+ offset.
132
+ - **IM coverage is row presence**: a record has at most one row in each of
133
+ `psa_ims`, `fas_ims` and `scalars_ims`. A missing row means that IM type is not
134
+ held for that record, not NULL.
135
+ - **Components**: `000`, `090`, `ver`, `geom`, `rotd0`, `rotd50`, `rotd100`. A
136
+ database may hold any subset, listed in `db_meta.components`; the writer
137
+ validates against it. `scalars_ims.CAV`, `AI`, `Ds575` and `Ds595` are NULL for
138
+ `rotd*` components; `PGA`, `PGV` and `PGD` are populated for every component.
139
+ - **Record kind**: `simulated` (physics-based simulation), `gmm` (empirical GMM
140
+ prediction) or `observed` (recorded ground motion). `gmm_key` identifies the
141
+ model, e.g. `"Bradley_2013"`, and is NULL unless `kind = "gmm"`.
142
+ - **Units**: linear, physical units; log is a read-time transform (`g` for pSA/PGA,
143
+ `cm/s` for PGV, `cm` for PGD, `m/s` for CAV/AI, `s` for Ds575/Ds595, see
144
+ `im_units`). Every `_sigma` column/array is the exception: ln-space total
145
+ standard deviation, dimensionless.
146
+ - **Constraints**: `PRIMARY KEY`/`UNIQUE`/`FOREIGN KEY` appear only on the four
147
+ dimension tables. The large tables (`site_event`, `records`, `psa_ims`,
148
+ `fas_ims`, `scalars_ims`) have none; their logical keys are documented in
149
+ `notes` and enforced by the writer, not the schema.
150
+
151
+ ## Usage
152
+
153
+ ```python
154
+ from imdb import IMDB
155
+
156
+ # read
157
+ with IMDB("run_set.duckdb") as db:
158
+ records = db.get_records(event_ids=["event1"], component="rotd50")
159
+ psa = db.get_psa(periods=[0.1, 1.0], event_ids=["event1"])
160
+ scalars = db.get_scalars(ims=["PGA", "PGV"])
161
+
162
+ # write
163
+ db = IMDB.create("new.duckdb", periods=[0.1, 0.2, 1.0])
164
+ db.add_events(events_df)
165
+ db.add_realisations(realisations_df)
166
+ db.add_sites(sites_df)
167
+ db.add_site_event(site_event_df)
168
+ db.add_records(
169
+ records_df
170
+ ) # rel_id, site_id, component, kind, pSA, FAS, scalar IM columns
171
+ db.validate()
172
+ db.close()
173
+ ```
174
+
175
+ ## Development
176
+
177
+ ```
178
+ uv sync --all-groups
179
+ uv run pytest -q
180
+ uv run ruff check
181
+ uv run ruff format
182
+ uv run ty check
183
+ ```