dirigent-integration 0.23.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_integration/__init__.py +29 -0
- dirigent_integration/py.typed +0 -0
- dirigent_integration/shelves/dhis2-analytics-to-csv-report.yaml +174 -0
- dirigent_integration/shelves/dhis2-export-to-s3.yaml +66 -0
- dirigent_integration/shelves/dhis2-metadata-snapshot.yaml +143 -0
- dirigent_integration/shelves/dhis2-tracker-weekly-window.yaml +162 -0
- dirigent_integration/shelves/dhis2-values-per-org-unit-to-parquet.yaml +186 -0
- dirigent_integration/shelves/dhis2-values-to-parquet.yaml +64 -0
- dirigent_integration/shelves/fhir/README.md +79 -0
- dirigent_integration/shelves/fhir/dhis2-to-fhir-observations.yaml +212 -0
- dirigent_integration/shelves/fhir/fhir-capture-bundle-to-data-values.yaml +230 -0
- dirigent_integration/shelves/fhir/fhir-conceptmap-driven-mapping.yaml +222 -0
- dirigent_integration/shelves/fhir/fhir-encounter-to-event.yaml +248 -0
- dirigent_integration/shelves/fhir/fhir-measure-report-to-analytics-check.yaml +205 -0
- dirigent_integration/shelves/fhir/fhir-nightly-window-sync.yaml +182 -0
- dirigent_integration/shelves/fhir/fhir-patient-to-tracked-entity.yaml +225 -0
- dirigent_integration/shelves/fhir/fhir-questionnaire-response-to-data-values.yaml +187 -0
- dirigent_integration/shelves/fhir/fhir-subscription-webhook-to-dhis2.yaml +218 -0
- dirigent_integration/shelves/inbound/README.md +32 -0
- dirigent_integration/shelves/inbound/csv-drop-to-data-values.yaml +176 -0
- dirigent_integration/shelves/inbound/fan-out-per-facility-import.yaml +202 -0
- dirigent_integration/shelves/inbound/fhir-observations-to-data-values.yaml +193 -0
- dirigent_integration/shelves/inbound/http-json-to-data-values.yaml +151 -0
- dirigent_integration/shelves/inbound/outside-to-dhis2-with-checks.yaml +246 -0
- dirigent_integration/shelves/inbound/parquet-lakehouse-to-dhis2.yaml +158 -0
- dirigent_integration/shelves/inbound/webhook-payload-to-tracker-event.yaml +176 -0
- dirigent_integration/shelves/inbound/weekly-window-pull-and-import.yaml +175 -0
- dirigent_integration/shelves/parquet-to-dhis2-import.yaml +195 -0
- dirigent_integration-0.23.2.dist-info/METADATA +231 -0
- dirigent_integration-0.23.2.dist-info/RECORD +33 -0
- dirigent_integration-0.23.2.dist-info/WHEEL +4 -0
- dirigent_integration-0.23.2.dist-info/entry_points.txt +3 -0
- dirigent_integration-0.23.2.dist-info/licenses/LICENSE +15 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""The control center's own contribution: the cross-boundary example shelves.
|
|
2
|
+
|
|
3
|
+
This distribution contributes no blocks and no connection kinds. Every block its documents
|
|
4
|
+
name comes from a pack the integration assembles; what is only ever authored here is the
|
|
5
|
+
document that puts two packs in one flow, and this is how that corpus reaches an instance.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from collections.abc import Sequence
|
|
9
|
+
from importlib.resources import files
|
|
10
|
+
from importlib.resources.abc import Traversable
|
|
11
|
+
|
|
12
|
+
from dirigent_plugin import extension
|
|
13
|
+
|
|
14
|
+
#: The directory the cross-boundary shelves live in, inside this distribution.
|
|
15
|
+
SHELVES_DIRECTORY = "shelves"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class IntegrationPlugin:
|
|
19
|
+
"""The plugin object the host discovers under the dirigent.plugins.v1 entry-point group."""
|
|
20
|
+
|
|
21
|
+
@extension
|
|
22
|
+
def examples(self) -> Sequence[Traversable]:
|
|
23
|
+
"""Contribute the cross-boundary shelves this distribution carries, and nothing else."""
|
|
24
|
+
return [files(__package__ or "dirigent_integration") / SHELVES_DIRECTORY]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
plugin = IntegrationPlugin()
|
|
28
|
+
|
|
29
|
+
__all__ = ["SHELVES_DIRECTORY", "IntegrationPlugin", "plugin"]
|
|
File without changes
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
# An analytics query turned into a csv somebody can open, and a webhook saying it is there.
|
|
2
|
+
#
|
|
3
|
+
# DHIS2 answers /api/analytics with a grid: a `headers` list naming the columns, a `rows` list
|
|
4
|
+
# of positional arrays, and a `metaData` block holding the display names for every uid that
|
|
5
|
+
# appears. That is a good wire format and a poor report -- nothing outside DHIS2 knows that
|
|
6
|
+
# column 2 is the value -- so the middle of this pipeline is the reshaping, and it is the part
|
|
7
|
+
# worth reading. Three packs meet here: dhis2 for the query, builtin for the jq, the csv codec
|
|
8
|
+
# and the webhook, storage-s3 for where the report lands.
|
|
9
|
+
#
|
|
10
|
+
# What happens, hop by hop:
|
|
11
|
+
#
|
|
12
|
+
# query dhis2.analytics_query, aggregate mode, broken down by indicator and period with
|
|
13
|
+
# the org unit fixed as a filter. The grid comes back as `body`, a value.
|
|
14
|
+
# rows the grid becomes a list of flat objects, one per grid row, with the uids
|
|
15
|
+
# resolved to names out of metaData: {indicator, period, value}. That is the
|
|
16
|
+
# shape a csv has, and the grid is not.
|
|
17
|
+
# staged storage.write, that value as one json object. A converter reads one URI and
|
|
18
|
+
# writes another, so a value the run is holding is put down before it is
|
|
19
|
+
# re-encoded -- and that write is the only door a value leaves a run by.
|
|
20
|
+
# report convert.std json -> csv, reading the staged object and writing to s3://. The
|
|
21
|
+
# header line is the object keys. Output: source, target, bytes_written.
|
|
22
|
+
# summary how many rows and where they went, built before the notification so the webhook
|
|
23
|
+
# body is a value and not four references.
|
|
24
|
+
# notify webhook.post of that summary to whatever URL the caller named. Nothing is signed
|
|
25
|
+
# here; use sign_with when the receiver verifies an HMAC.
|
|
26
|
+
#
|
|
27
|
+
# To make it yours: point indicators at your own uids (semicolons separate them, which is
|
|
28
|
+
# DHIS2's own syntax for several items on one dimension), change period to a fixed period or
|
|
29
|
+
# another relative one, and give notify_url a receiver. The default is Postman Echo, which
|
|
30
|
+
# reflects what it is posted, so the run shows the body that went out without a receiver of
|
|
31
|
+
# your own. Which connection backs s3:// is an instance setting (storage_connections:
|
|
32
|
+
# {s3: <code>}).
|
|
33
|
+
#
|
|
34
|
+
# dg run --local examples/dhis2-analytics-to-csv-report.yaml
|
|
35
|
+
# dg run --local examples/dhis2-analytics-to-csv-report.yaml -p period=LAST_4_QUARTERS
|
|
36
|
+
|
|
37
|
+
format: dirigent/v1
|
|
38
|
+
kind: pipeline
|
|
39
|
+
code: dhis2-analytics-to-csv-report
|
|
40
|
+
name: An analytics grid as a csv report
|
|
41
|
+
description: Query DHIS2 analytics, flatten the grid into rows, write a csv to S3, and announce it.
|
|
42
|
+
|
|
43
|
+
tags: [dhis2, s3, csv, cross-boundary]
|
|
44
|
+
|
|
45
|
+
requires:
|
|
46
|
+
blocks:
|
|
47
|
+
- dhis2.analytics_query
|
|
48
|
+
- transform.jq
|
|
49
|
+
- storage.write
|
|
50
|
+
- convert.std
|
|
51
|
+
- webhook.post
|
|
52
|
+
|
|
53
|
+
# The public DHIS2 play server, carried inline so this document runs standalone. An instance
|
|
54
|
+
# you own names a connection it holds instead of carrying one.
|
|
55
|
+
connections:
|
|
56
|
+
dhis2-demo:
|
|
57
|
+
kind: dhis2
|
|
58
|
+
config:
|
|
59
|
+
base_url: https://play.im.dhis2.org/stable-2-43-1
|
|
60
|
+
basic_username: admin
|
|
61
|
+
basic_password: district
|
|
62
|
+
timeout: 60s
|
|
63
|
+
|
|
64
|
+
params:
|
|
65
|
+
type: object
|
|
66
|
+
properties:
|
|
67
|
+
indicators:
|
|
68
|
+
type: string
|
|
69
|
+
# ANC 1 coverage and ANC 3 coverage on the play server. Semicolons are DHIS2's own
|
|
70
|
+
# separator for several items on one dimension, so this stays a single string.
|
|
71
|
+
default: Uvn6LCg7dVU;OdiHJayrsKo
|
|
72
|
+
period:
|
|
73
|
+
type: string
|
|
74
|
+
default: LAST_12_MONTHS
|
|
75
|
+
description: A relative period keyword or an ISO period, as DHIS2's pe dimension takes.
|
|
76
|
+
org_unit:
|
|
77
|
+
type: string
|
|
78
|
+
default: ImspTQPwCqd
|
|
79
|
+
bucket:
|
|
80
|
+
type: string
|
|
81
|
+
default: dirigent-reports
|
|
82
|
+
notify_url:
|
|
83
|
+
type: string
|
|
84
|
+
default: https://postman-echo.com/post
|
|
85
|
+
description: Where the summary is posted; the default reflects it back for inspection.
|
|
86
|
+
|
|
87
|
+
steps:
|
|
88
|
+
query:
|
|
89
|
+
block: dhis2.analytics_query
|
|
90
|
+
config:
|
|
91
|
+
connection: dhis2-demo
|
|
92
|
+
mode: aggregate
|
|
93
|
+
# dimension axes are what the grid is broken down by; filter axes fix a dimension the
|
|
94
|
+
# grid is not broken down by. Putting ou in filter is what makes the report two columns
|
|
95
|
+
# wide instead of three.
|
|
96
|
+
dimension:
|
|
97
|
+
- dx:${params.indicators}
|
|
98
|
+
- pe:${params.period}
|
|
99
|
+
filter:
|
|
100
|
+
- ou:${params.org_unit}
|
|
101
|
+
# UID keeps the grid's own cells as uids and the display names in metaData, which the
|
|
102
|
+
# next step resolves. Asking DHIS2 for NAME here would flatten the report but lose the
|
|
103
|
+
# uid, and a report a machine reads again wants both.
|
|
104
|
+
output_id_scheme: UID
|
|
105
|
+
|
|
106
|
+
rows:
|
|
107
|
+
block: transform.jq
|
|
108
|
+
depends_on: [query]
|
|
109
|
+
config:
|
|
110
|
+
input: ${steps.query.output.body}
|
|
111
|
+
# The grid is positional, so the column positions are looked up by header name rather
|
|
112
|
+
# than hard-coded: DHIS2 has moved them between versions. metaData.items maps every uid
|
|
113
|
+
# in the answer to its display name, which is what turns dx:Uvn6LCg7dVU into a label.
|
|
114
|
+
program: |
|
|
115
|
+
. as $grid
|
|
116
|
+
| ([$grid.headers[].name] | index("dx")) as $dx
|
|
117
|
+
| ([$grid.headers[].name] | index("pe")) as $pe
|
|
118
|
+
| ([$grid.headers[].name] | index("value")) as $value
|
|
119
|
+
| $grid.metaData.items as $names
|
|
120
|
+
| [ $grid.rows[]
|
|
121
|
+
| {
|
|
122
|
+
indicator: ($names[.[$dx]].name // .[$dx]),
|
|
123
|
+
period: ($names[.[$pe]].name // .[$pe]),
|
|
124
|
+
value: .[$value]
|
|
125
|
+
}
|
|
126
|
+
]
|
|
127
|
+
|
|
128
|
+
staged:
|
|
129
|
+
block: storage.write
|
|
130
|
+
depends_on: [rows]
|
|
131
|
+
config:
|
|
132
|
+
# The run's own scratch space: these bytes exist only so the converter has a URI to
|
|
133
|
+
# read, and the report the caller keeps is the csv below.
|
|
134
|
+
target: ${run.scratch}/analytics/${params.period}.json
|
|
135
|
+
value: ${steps.rows.output.value}
|
|
136
|
+
|
|
137
|
+
report:
|
|
138
|
+
block: convert.std
|
|
139
|
+
depends_on: [staged]
|
|
140
|
+
config:
|
|
141
|
+
source: ${steps.staged.output.uri}
|
|
142
|
+
from: json
|
|
143
|
+
to: csv
|
|
144
|
+
# The report outlives the run, so ${run.scratch} is out: retention sweeps a run's prefix
|
|
145
|
+
# with the run. ${artifacts}/<path> is the portable way to keep something -- the instance's
|
|
146
|
+
# own storage root, which nothing sweeps -- and a named bucket is the other case, which is
|
|
147
|
+
# this one: the csv has to land in the reporting bucket people already fetch reports from.
|
|
148
|
+
target: s3://${params.bucket}/dhis2/analytics/${params.period}.csv
|
|
149
|
+
|
|
150
|
+
summary:
|
|
151
|
+
block: transform.jq
|
|
152
|
+
depends_on: [report]
|
|
153
|
+
config:
|
|
154
|
+
input:
|
|
155
|
+
rows: ${steps.rows.output.value}
|
|
156
|
+
csv: ${steps.report.output.target}
|
|
157
|
+
bytes: ${steps.report.output.bytes_written}
|
|
158
|
+
# Counting happens here rather than in the notification because a webhook body should be
|
|
159
|
+
# a value that was computed, not four references a receiver has to trust were resolved.
|
|
160
|
+
program: |
|
|
161
|
+
{report: .csv, report_bytes: .bytes, row_count: (.rows | length)}
|
|
162
|
+
|
|
163
|
+
notify:
|
|
164
|
+
block: webhook.post
|
|
165
|
+
depends_on: [summary]
|
|
166
|
+
retry:
|
|
167
|
+
max_attempts: 3
|
|
168
|
+
backoff: 30s
|
|
169
|
+
config:
|
|
170
|
+
url: ${params.notify_url}
|
|
171
|
+
body: ${steps.summary.output.value}
|
|
172
|
+
# A receiver that queues the notification answers 202, which is delivered as far as this
|
|
173
|
+
# step is concerned: posting it is what the step is responsible for.
|
|
174
|
+
success_status: [200, 201, 202]
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# Export a DHIS2 data value set and archive it in object storage.
|
|
2
|
+
#
|
|
3
|
+
# This pipeline is the integration's own: it belongs to no single pack, because it spans two.
|
|
4
|
+
# The read is a dhis2.data_value_set_export from the dhis2 pack; the write and the archive land
|
|
5
|
+
# on the s3:// scheme the dirigent-storage-s3 pack contributes. dirigent-core knows neither
|
|
6
|
+
# pack -- only an assembled environment where both are installed can run this at all.
|
|
7
|
+
#
|
|
8
|
+
# Three hops, and what each one hands on:
|
|
9
|
+
#
|
|
10
|
+
# export dhis2.data_value_set_export. The set comes back as `body`, a value in the run.
|
|
11
|
+
# stage storage.write, that value to a URI. A block does not write storage itself: a
|
|
12
|
+
# value leaves a run through this one door. Output: uri, bytes_written,
|
|
13
|
+
# content_type.
|
|
14
|
+
# archive storage.copy, the object the write named onto a second, dated key. A copy moves
|
|
15
|
+
# bytes nobody has to look at, so it is bounded by the storage and not by the
|
|
16
|
+
# worker.
|
|
17
|
+
#
|
|
18
|
+
# Which connection backs the s3:// scheme is an instance setting (storage_connections:
|
|
19
|
+
# {s3: <code>}), so the document carries only the dhis2 connection it names directly.
|
|
20
|
+
#
|
|
21
|
+
# dg run --local examples/dhis2-export-to-s3.yaml
|
|
22
|
+
|
|
23
|
+
format: dirigent/v1
|
|
24
|
+
kind: pipeline
|
|
25
|
+
code: dhis2-export-to-s3
|
|
26
|
+
name: Export a data value set to S3
|
|
27
|
+
description: Read a DHIS2 data value set, write it into object storage, then archive a dated copy.
|
|
28
|
+
|
|
29
|
+
tags: [dhis2, s3, cross-boundary]
|
|
30
|
+
|
|
31
|
+
requires:
|
|
32
|
+
blocks:
|
|
33
|
+
- dhis2.data_value_set_export
|
|
34
|
+
- storage.write
|
|
35
|
+
- storage.copy
|
|
36
|
+
|
|
37
|
+
connections:
|
|
38
|
+
dhis2-demo:
|
|
39
|
+
kind: dhis2
|
|
40
|
+
config:
|
|
41
|
+
base_url: https://play.im.dhis2.org/stable-2-43-1
|
|
42
|
+
basic_username: admin
|
|
43
|
+
basic_password: district
|
|
44
|
+
timeout: 60s
|
|
45
|
+
|
|
46
|
+
steps:
|
|
47
|
+
export:
|
|
48
|
+
block: dhis2.data_value_set_export
|
|
49
|
+
config:
|
|
50
|
+
connection: dhis2-demo
|
|
51
|
+
data_set: BfMAe6Itzgt
|
|
52
|
+
period: "202507"
|
|
53
|
+
org_unit: vSbt6cezomG
|
|
54
|
+
stage:
|
|
55
|
+
block: storage.write
|
|
56
|
+
depends_on: [export]
|
|
57
|
+
config:
|
|
58
|
+
target: s3://dirigent-exports/dhis2/BfMAe6Itzgt-202507.json
|
|
59
|
+
value: ${steps.export.output.body}
|
|
60
|
+
archive:
|
|
61
|
+
block: storage.copy
|
|
62
|
+
depends_on: [stage]
|
|
63
|
+
config:
|
|
64
|
+
# Where the write said the object landed, not the same key typed twice.
|
|
65
|
+
source: ${steps.stage.output.uri}
|
|
66
|
+
target: s3://dirigent-archive/dhis2/BfMAe6Itzgt-202507.json
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
# A dated snapshot of the metadata a downstream system needs, and then the data that uses it.
|
|
2
|
+
#
|
|
3
|
+
# Every DHIS2 export is a pile of uids. What the uids mean -- which data element, which
|
|
4
|
+
# facility, which category combination -- lives in metadata that changes underneath a
|
|
5
|
+
# warehouse without warning. So a load that will still be readable in a year snapshots the
|
|
6
|
+
# metadata beside the data, at the same moment, and this is that step. It spans dhis2 for the
|
|
7
|
+
# reads, builtin for the reshape and the composition, and storage-s3 for where it lands, which
|
|
8
|
+
# is why no single pack could own it.
|
|
9
|
+
#
|
|
10
|
+
# What happens, hop by hop:
|
|
11
|
+
#
|
|
12
|
+
# collections one dhis2.metadata read per resource named in params. The block resolves the
|
|
13
|
+
# instance's version on connect and reads through that version's accessor, so
|
|
14
|
+
# the same document reads a 2.42 or a 2.43 instance unchanged. Each answer is
|
|
15
|
+
# the collection under its own key: {"dataElements": [...]}.
|
|
16
|
+
# snapshot the fan-out read back as one list and zipped against the resource names into a
|
|
17
|
+
# single document, {"dataElements": [...], "organisationUnits": [...]}.
|
|
18
|
+
# store storage.write, that document as one object in s3. A jq step hands on a value
|
|
19
|
+
# and a value leaves a run through this one door, so the write is its own hop.
|
|
20
|
+
# One object, one fetch, everything a reader needs to resolve a uid.
|
|
21
|
+
# counts the same list, counted: what each collection held on the day of the snapshot.
|
|
22
|
+
# This is the number worth alerting on -- a data element count that halved
|
|
23
|
+
# overnight is a broken credential, not a tidy-up.
|
|
24
|
+
# then_export pipeline.run of dhis2-export-to-s3, the example beside this one. Composition
|
|
25
|
+
# is the point: the metadata is snapshotted first, the data is read second, and
|
|
26
|
+
# a reader of that data has the metadata that was true when it was read.
|
|
27
|
+
#
|
|
28
|
+
# To make it yours: change resources to the collections your consumers actually resolve, and
|
|
29
|
+
# point pipeline at your own export. `wait: true` keeps this run open until the child settles,
|
|
30
|
+
# so a failed export fails the snapshot run too; set it false to fire and forget. The child is
|
|
31
|
+
# resolved by code on the instance, so it must be applied before this document runs -- which is
|
|
32
|
+
# why it is not named in requires here, where a local validation holds no instance to check
|
|
33
|
+
# against. Which connection backs s3:// is an instance setting (storage_connections: {s3: <code>}).
|
|
34
|
+
#
|
|
35
|
+
# dg apply examples/dhis2-export-to-s3.yaml
|
|
36
|
+
# dg apply examples/dhis2-metadata-snapshot.yaml
|
|
37
|
+
# dg run dhis2-metadata-snapshot --watch
|
|
38
|
+
|
|
39
|
+
format: dirigent/v1
|
|
40
|
+
kind: pipeline
|
|
41
|
+
code: dhis2-metadata-snapshot
|
|
42
|
+
name: A metadata snapshot, then the data
|
|
43
|
+
description: Snapshot DHIS2 metadata collections into S3, then run the export that uses them.
|
|
44
|
+
|
|
45
|
+
tags: [dhis2, s3, composition, cross-boundary]
|
|
46
|
+
|
|
47
|
+
requires:
|
|
48
|
+
blocks:
|
|
49
|
+
- dhis2.metadata
|
|
50
|
+
- transform.jq
|
|
51
|
+
- storage.write
|
|
52
|
+
- pipeline.run
|
|
53
|
+
|
|
54
|
+
# The public DHIS2 play server, carried inline so this document runs standalone. An instance
|
|
55
|
+
# you own names a connection it holds instead.
|
|
56
|
+
connections:
|
|
57
|
+
dhis2-demo:
|
|
58
|
+
kind: dhis2
|
|
59
|
+
config:
|
|
60
|
+
base_url: https://play.im.dhis2.org/stable-2-43-1
|
|
61
|
+
basic_username: admin
|
|
62
|
+
basic_password: district
|
|
63
|
+
timeout: 60s
|
|
64
|
+
|
|
65
|
+
params:
|
|
66
|
+
type: object
|
|
67
|
+
properties:
|
|
68
|
+
resources:
|
|
69
|
+
type: array
|
|
70
|
+
# DHIS2 collection names, spelled as they appear in the API path. These two are what a
|
|
71
|
+
# warehouse needs to make an export readable: what was measured, and where.
|
|
72
|
+
default: [dataElements, organisationUnits]
|
|
73
|
+
items:
|
|
74
|
+
type: string
|
|
75
|
+
fields:
|
|
76
|
+
type: string
|
|
77
|
+
default: id,name,shortName
|
|
78
|
+
description: The DHIS2 fields selector applied to every collection in the snapshot.
|
|
79
|
+
bucket:
|
|
80
|
+
type: string
|
|
81
|
+
default: dirigent-exports
|
|
82
|
+
|
|
83
|
+
steps:
|
|
84
|
+
collections:
|
|
85
|
+
block: dhis2.metadata
|
|
86
|
+
for_each: ${params.resources}
|
|
87
|
+
config:
|
|
88
|
+
connection: dhis2-demo
|
|
89
|
+
resource: ${item}
|
|
90
|
+
fields: ${params.fields}
|
|
91
|
+
# Paging off: a snapshot wants the whole collection, and a half-read one is worse than
|
|
92
|
+
# no snapshot at all.
|
|
93
|
+
paging: false
|
|
94
|
+
|
|
95
|
+
snapshot:
|
|
96
|
+
block: transform.jq
|
|
97
|
+
depends_on: [collections]
|
|
98
|
+
config:
|
|
99
|
+
input:
|
|
100
|
+
resources: ${params.resources}
|
|
101
|
+
pages: ${steps.collections.output}
|
|
102
|
+
# The fan-out's output is its items' outputs in item order, so it lines up with the
|
|
103
|
+
# resource list positionally. That alignment is only safe because the step above has no
|
|
104
|
+
# `items: continue` -- a tolerated failure would drop an element and silently shift
|
|
105
|
+
# every collection one name to the left.
|
|
106
|
+
program: |
|
|
107
|
+
[.resources, [.pages[].body]]
|
|
108
|
+
| transpose
|
|
109
|
+
| map({(.[0]): (.[1][.[0]] // [])})
|
|
110
|
+
| add
|
|
111
|
+
|
|
112
|
+
store:
|
|
113
|
+
block: storage.write
|
|
114
|
+
depends_on: [snapshot]
|
|
115
|
+
config:
|
|
116
|
+
target: s3://${params.bucket}/dhis2/metadata/snapshot.json
|
|
117
|
+
value: ${steps.snapshot.output.value}
|
|
118
|
+
|
|
119
|
+
counts:
|
|
120
|
+
block: transform.jq
|
|
121
|
+
depends_on: [collections]
|
|
122
|
+
config:
|
|
123
|
+
input:
|
|
124
|
+
resources: ${params.resources}
|
|
125
|
+
pages: ${steps.collections.output}
|
|
126
|
+
program: |
|
|
127
|
+
[.resources, [.pages[].body]]
|
|
128
|
+
| transpose
|
|
129
|
+
| map({resource: .[0], rows: ((.[1][.[0]] // []) | length)})
|
|
130
|
+
|
|
131
|
+
then_export:
|
|
132
|
+
block: pipeline.run
|
|
133
|
+
# Both branches, so the run does not start the export before the snapshot it is supposed to
|
|
134
|
+
# be described by has actually landed.
|
|
135
|
+
depends_on: [store, counts]
|
|
136
|
+
config:
|
|
137
|
+
pipeline: dhis2-export-to-s3
|
|
138
|
+
# Waiting makes this run the whole story: one run id covers the metadata and the data,
|
|
139
|
+
# and a failed export is visible here rather than only in the child.
|
|
140
|
+
wait: true
|
|
141
|
+
# A child that finished completed_with_errors is a partial export, and a snapshot paired
|
|
142
|
+
# with a partial export should not report success.
|
|
143
|
+
strict: true
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
# A weekly tracker extract that reads the week that just closed, not the moment it started.
|
|
2
|
+
#
|
|
3
|
+
# A schedule answers "when does this start". A window answers "what does this run cover", and
|
|
4
|
+
# for an extract that is the only question that matters: the Monday-morning run is meant to
|
|
5
|
+
# collect last week, not everything the instance has ever recorded. Every schedule-fired run
|
|
6
|
+
# carries the interval that just closed as ${run.window.start} up to but not including
|
|
7
|
+
# ${run.window.end}, and consecutive firings tile the timeline with no overlap and no gap.
|
|
8
|
+
#
|
|
9
|
+
# Four packs meet here: dhis2 for the tracker read, builtin for the jq and the ndjson codec,
|
|
10
|
+
# parquet for the encoding, storage-s3 for both destinations.
|
|
11
|
+
#
|
|
12
|
+
# What happens, hop by hop:
|
|
13
|
+
#
|
|
14
|
+
# read dhis2.tracker, kind: events, scoped to a programme and an org unit subtree, with
|
|
15
|
+
# updated_after set to the window's start. DHIS2 answers a page: the objects under
|
|
16
|
+
# `instances` and a `page` block beside them.
|
|
17
|
+
# in_window the upper bound. DHIS2's tracker read takes updated_after and nothing that means
|
|
18
|
+
# updated_before, so the open end of the window is applied here instead. Without
|
|
19
|
+
# it, an event edited while the run was in flight would be collected twice: once by
|
|
20
|
+
# this firing and again by the next one.
|
|
21
|
+
# staged storage.write, the windowed events as one json object in s3://. A converter reads
|
|
22
|
+
# one URI and writes another, so a value the run is holding is put down first --
|
|
23
|
+
# and that write is the only door a value leaves a run by.
|
|
24
|
+
# as_ndjson convert.std json -> ndjson, beside it. One event per line is the format a reader
|
|
25
|
+
# streams rather than parses whole, and it is what the parquet encoder reads next.
|
|
26
|
+
# as_parquet convert.arrow ndjson -> parquet, beside that. Parquet is bytes and never
|
|
27
|
+
# travels as a value: source and target are URIs and nothing is carried between
|
|
28
|
+
# them. Each converter's output names the target it wrote.
|
|
29
|
+
# archive a dated second copy, keyed by the window rather than by the clock, so a backfill
|
|
30
|
+
# of an old week writes that week's key and not today's.
|
|
31
|
+
#
|
|
32
|
+
# To make it yours: change program and org_unit to your own, adjust the cron and the timezone,
|
|
33
|
+
# and point bucket at a bucket you own. Which connection backs s3:// is an instance setting
|
|
34
|
+
# (storage_connections: {s3: <code>}).
|
|
35
|
+
#
|
|
36
|
+
# The window is a property of the run, not of this document -- nothing below declares it, the
|
|
37
|
+
# cadence computes it -- so an ad hoc run has to be handed one, and a run carrying no window
|
|
38
|
+
# refuses the reference rather than quietly asking DHIS2 for everything it has:
|
|
39
|
+
#
|
|
40
|
+
# dg run --local examples/dhis2-tracker-weekly-window.yaml --window 2026-06-01..2026-06-08
|
|
41
|
+
# dg apply examples/dhis2-tracker-weekly-window.yaml
|
|
42
|
+
# dg backfill dhis2-tracker-weekly-window --schedule weekly \
|
|
43
|
+
# --from 2026-01-05T03:00:00Z --to 2026-03-02T03:00:00Z --dry-run
|
|
44
|
+
|
|
45
|
+
format: dirigent/v1
|
|
46
|
+
kind: pipeline
|
|
47
|
+
code: dhis2-tracker-weekly-window
|
|
48
|
+
name: A week of tracker events, as parquet
|
|
49
|
+
description: Read the tracker events of the week that just closed and archive them as parquet.
|
|
50
|
+
|
|
51
|
+
tags: [dhis2, parquet, s3, schedule, cross-boundary]
|
|
52
|
+
|
|
53
|
+
requires:
|
|
54
|
+
blocks:
|
|
55
|
+
- dhis2.tracker
|
|
56
|
+
- transform.jq
|
|
57
|
+
- storage.write
|
|
58
|
+
- convert.std
|
|
59
|
+
- convert.arrow
|
|
60
|
+
- storage.copy
|
|
61
|
+
|
|
62
|
+
# The public DHIS2 play server, carried inline so this document runs standalone. An instance
|
|
63
|
+
# you own names a connection it holds instead.
|
|
64
|
+
connections:
|
|
65
|
+
dhis2-demo:
|
|
66
|
+
kind: dhis2
|
|
67
|
+
config:
|
|
68
|
+
base_url: https://play.im.dhis2.org/stable-2-43-1
|
|
69
|
+
basic_username: admin
|
|
70
|
+
basic_password: district
|
|
71
|
+
timeout: 60s
|
|
72
|
+
|
|
73
|
+
params:
|
|
74
|
+
type: object
|
|
75
|
+
properties:
|
|
76
|
+
program:
|
|
77
|
+
type: string
|
|
78
|
+
default: IpHINAT79UW
|
|
79
|
+
description: The tracker programme to read; the default is the play server's child programme.
|
|
80
|
+
org_unit:
|
|
81
|
+
type: string
|
|
82
|
+
default: ImspTQPwCqd
|
|
83
|
+
bucket:
|
|
84
|
+
type: string
|
|
85
|
+
default: dirigent-exports
|
|
86
|
+
|
|
87
|
+
steps:
|
|
88
|
+
read:
|
|
89
|
+
block: dhis2.tracker
|
|
90
|
+
timeout: 30m
|
|
91
|
+
config:
|
|
92
|
+
connection: dhis2-demo
|
|
93
|
+
kind: events
|
|
94
|
+
program: ${params.program}
|
|
95
|
+
org_unit: ${params.org_unit}
|
|
96
|
+
# The org unit is a root, so the read has to descend it; SELECTED would return the
|
|
97
|
+
# events recorded at the root itself, which is almost none of them.
|
|
98
|
+
ou_mode: DESCENDANTS
|
|
99
|
+
# The lower bound of the window, handed to DHIS2 so the instance does the filtering
|
|
100
|
+
# rather than sending a year of events across for us to discard.
|
|
101
|
+
updated_after: ${run.window.start}
|
|
102
|
+
fields: event,program,programStage,orgUnit,occurredAt,updatedAt,status
|
|
103
|
+
page_size: 500
|
|
104
|
+
|
|
105
|
+
in_window:
|
|
106
|
+
block: transform.jq
|
|
107
|
+
depends_on: [read]
|
|
108
|
+
config:
|
|
109
|
+
input:
|
|
110
|
+
events: ${steps.read.output.body.instances}
|
|
111
|
+
until: ${run.window.end}
|
|
112
|
+
# Half-open: >= start came from DHIS2, < end is applied here. Both ends written out is
|
|
113
|
+
# what stops an event that was edited during the run being collected twice.
|
|
114
|
+
program: |
|
|
115
|
+
.until as $until
|
|
116
|
+
| [.events[] | select(.updatedAt < $until)]
|
|
117
|
+
|
|
118
|
+
staged:
|
|
119
|
+
block: storage.write
|
|
120
|
+
depends_on: [in_window]
|
|
121
|
+
config:
|
|
122
|
+
target: s3://${params.bucket}/dhis2/tracker/${params.program}/${run.window.start}.json
|
|
123
|
+
value: ${steps.in_window.output.value}
|
|
124
|
+
|
|
125
|
+
as_ndjson:
|
|
126
|
+
block: convert.std
|
|
127
|
+
depends_on: [staged]
|
|
128
|
+
config:
|
|
129
|
+
source: ${steps.staged.output.uri}
|
|
130
|
+
from: json
|
|
131
|
+
to: ndjson
|
|
132
|
+
target: s3://${params.bucket}/dhis2/tracker/${params.program}/${run.window.start}.ndjson
|
|
133
|
+
|
|
134
|
+
as_parquet:
|
|
135
|
+
block: convert.arrow
|
|
136
|
+
depends_on: [as_ndjson]
|
|
137
|
+
config:
|
|
138
|
+
source: ${steps.as_ndjson.output.target}
|
|
139
|
+
from: ndjson
|
|
140
|
+
to: parquet
|
|
141
|
+
target: s3://${params.bucket}/dhis2/tracker/${params.program}/${run.window.start}.parquet
|
|
142
|
+
|
|
143
|
+
archive:
|
|
144
|
+
block: storage.copy
|
|
145
|
+
depends_on: [as_parquet]
|
|
146
|
+
config:
|
|
147
|
+
source: ${steps.as_parquet.output.target}
|
|
148
|
+
# Keyed by the window's start, so re-running or backfilling a week overwrites that week's
|
|
149
|
+
# archive object and nothing else. Keying it by the wall clock would file a backfill of
|
|
150
|
+
# January under the day the backfill happened to run.
|
|
151
|
+
target: s3://dirigent-archive/dhis2/tracker/${params.program}/${run.window.start}.parquet
|
|
152
|
+
|
|
153
|
+
triggers:
|
|
154
|
+
schedules:
|
|
155
|
+
- code: weekly
|
|
156
|
+
name: Monday morning, Oslo time
|
|
157
|
+
description: Reads the week that closed at three on Monday morning.
|
|
158
|
+
cron: "0 3 * * 1"
|
|
159
|
+
# The zone belongs to the schedule, so the window it derives is a week of Oslo's clock
|
|
160
|
+
# rather than a fixed 168 hours: the daylight-saving weeks come out right without
|
|
161
|
+
# anything in the steps above knowing that daylight saving exists.
|
|
162
|
+
timezone: Europe/Oslo
|