dirigent-examples 0.15.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_examples/__init__.py +22 -0
- dirigent_examples/py.typed +0 -0
- dirigent_examples/shelves/README.md +299 -0
- dirigent_examples/shelves/composition/README.md +18 -0
- dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
- dirigent_examples/shelves/composition/composition-child.yaml +64 -0
- dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
- dirigent_examples/shelves/connections.yaml +52 -0
- dirigent_examples/shelves/demo/README.md +19 -0
- dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
- dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
- dirigent_examples/shelves/demo/requires.yaml +65 -0
- dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
- dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
- dirigent_examples/shelves/docker/README.md +29 -0
- dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
- dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
- dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
- dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
- dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
- dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
- dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
- dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
- dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
- dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
- dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
- dirigent_examples/shelves/execute/README.md +16 -0
- dirigent_examples/shelves/execute/long-log.yaml +89 -0
- dirigent_examples/shelves/failure/README.md +20 -0
- dirigent_examples/shelves/failure/error-handler.yaml +89 -0
- dirigent_examples/shelves/failure/optional-step.yaml +82 -0
- dirigent_examples/shelves/failure/retries.yaml +75 -0
- dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
- dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
- dirigent_examples/shelves/git/README.md +32 -0
- dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
- dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
- dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
- dirigent_examples/shelves/graph/README.md +22 -0
- dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
- dirigent_examples/shelves/graph/fan-in.yaml +80 -0
- dirigent_examples/shelves/graph/fan-out.yaml +66 -0
- dirigent_examples/shelves/graph/linear.yaml +66 -0
- dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
- dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
- dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
- dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
- dirigent_examples/shelves/hello-world.yaml +30 -0
- dirigent_examples/shelves/open-data/README.md +67 -0
- dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
- dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
- dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
- dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
- dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
- dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
- dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
- dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
- dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
- dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
- dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
- dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
- dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
- dirigent_examples/shelves/patterns/README.md +144 -0
- dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
- dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
- dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
- dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
- dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
- dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
- dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
- dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
- dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
- dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
- dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
- dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
- dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
- dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
- dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
- dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
- dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
- dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
- dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
- dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
- dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
- dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
- dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
- dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
- dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
- dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
- dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
- dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
- dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
- dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
- dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
- dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
- dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
- dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
- dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
- dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
- dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
- dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
- dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
- dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
- dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
- dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
- dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
- dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
- dirigent_examples/shelves/python/README.md +31 -0
- dirigent_examples/shelves/python/apply_and_run.py +52 -0
- dirigent_examples/shelves/python/ci_gate.py +76 -0
- dirigent_examples/shelves/python/connections.py +61 -0
- dirigent_examples/shelves/python/error_handling.py +84 -0
- dirigent_examples/shelves/python/follow_logs.py +39 -0
- dirigent_examples/shelves/python/list_and_filter.py +52 -0
- dirigent_examples/shelves/queues/README.md +59 -0
- dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
- dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
- dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
- dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
- dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
- dirigent_examples/shelves/recipes/README.md +130 -0
- dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
- dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
- dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
- dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
- dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
- dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
- dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
- dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
- dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
- dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
- dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
- dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
- dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
- dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
- dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
- dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
- dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
- dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
- dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
- dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
- dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
- dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
- dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
- dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
- dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
- dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
- dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
- dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
- dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
- dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
- dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
- dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
- dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
- dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
- dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
- dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
- dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
- dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
- dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
- dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
- dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
- dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
- dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
- dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
- dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
- dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
- dirigent_examples/shelves/s3/README.md +34 -0
- dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
- dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
- dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
- dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
- dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
- dirigent_examples/shelves/schemas/README.md +36 -0
- dirigent_examples/shelves/schemas/echo-reading.json +18 -0
- dirigent_examples/shelves/schemas/ou-record.json +13 -0
- dirigent_examples/shelves/schemas/station-reading.json +13 -0
- dirigent_examples/shelves/sensors/README.md +16 -0
- dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
- dirigent_examples/shelves/sensors/time-window.yaml +61 -0
- dirigent_examples/shelves/sql/README.md +52 -0
- dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
- dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
- dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
- dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
- dirigent_examples/shelves/sql/warehouse.sql +42 -0
- dirigent_examples/shelves/transform/README.md +36 -0
- dirigent_examples/shelves/transform/csv-report.yaml +55 -0
- dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
- dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
- dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
- dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
- dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
- dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
- dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
- dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
- dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
- dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
- dirigent_examples/shelves/triggers/README.md +45 -0
- dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
- dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
- dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
- dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
- dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
- dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
- dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
- dirigent_examples/shelves/validate/README.md +31 -0
- dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
- dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
- dirigent_examples-0.15.0.dist-info/METADATA +21 -0
- dirigent_examples-0.15.0.dist-info/RECORD +216 -0
- dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
- dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
- dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
# Watching one HDX dataset for a new version, with a file in storage standing in for state.
|
|
2
|
+
#
|
|
3
|
+
# The Humanitarian Data Exchange runs CKAN, whose API is keyless for public datasets:
|
|
4
|
+
# /api/3/action/package_show?id=<slug> answers {"success": true, "result": {...}}, and the
|
|
5
|
+
# result carries `metadata_modified` -- an ISO timestamp that moves whenever anything about the
|
|
6
|
+
# dataset changes -- plus a `resources` list, one entry per downloadable file with its own
|
|
7
|
+
# `download_url` and `last_modified`.
|
|
8
|
+
#
|
|
9
|
+
# THE PATTERN THIS TEACHES IS THE MARKER. There is no state block: nothing in dirigent
|
|
10
|
+
# remembers, between runs, what a previous run saw. What there is, is storage, and a marker is
|
|
11
|
+
# just a small object written at a stable path holding what the last run observed. A sensor
|
|
12
|
+
# looks for it, storage.read brings it back into the run, the comparison decides, and
|
|
13
|
+
# storage.write puts the new one down. The whole state machine is a handful of steps and one
|
|
14
|
+
# file, and it works the same against s3:// as against a local artifact root.
|
|
15
|
+
#
|
|
16
|
+
# Two consequences worth reading before copying this:
|
|
17
|
+
#
|
|
18
|
+
# - The marker below is written under ${run.scratch}, which is deleted with the run, so every
|
|
19
|
+
# `--local` run is a first sighting: the sensor skips, the comparison skips with it, and the
|
|
20
|
+
# run ends succeeded having only written a marker. That is the honest first-run behaviour,
|
|
21
|
+
# and it is what the graph shows. On an instance, write the marker at a stable location --
|
|
22
|
+
# s3://your-bucket/markers/... -- and the second run is the one that compares.
|
|
23
|
+
# - The marker path is written out in the two steps that name it, the sensor and the last
|
|
24
|
+
# write, rather than kept in a parameter, because a parameter default is literal text:
|
|
25
|
+
# ${run.scratch} in a default is those characters and not a reference to anything.
|
|
26
|
+
#
|
|
27
|
+
# What happens, hop by hop:
|
|
28
|
+
#
|
|
29
|
+
# dataset one GET of the package. CKAN answers success/result even for an error, so the
|
|
30
|
+
# step's own 2xx check is not the whole story; the transform below reads .result
|
|
31
|
+
# and would fail loudly on anything else.
|
|
32
|
+
# current what this run saw: the id, the stamp, and the resource list, reduced to what a
|
|
33
|
+
# consumer actually needs.
|
|
34
|
+
# seen is there a marker? Five seconds is not a wait for the world, it is a lookup with
|
|
35
|
+
# a deadline; an absent marker skips this step and everything reading it.
|
|
36
|
+
# recalled storage.read of that object, which is the only way the previous stamp comes back
|
|
37
|
+
# into the run.
|
|
38
|
+
# previous the stamp taken out of it.
|
|
39
|
+
# moved the comparison. Equal stamps produce an empty list, a moved stamp produces the
|
|
40
|
+
# resource urls: the decision is data, not a status.
|
|
41
|
+
# decided storage.write, the decision put down as json, since the converter below reads a
|
|
42
|
+
# URI rather than a value.
|
|
43
|
+
# written that list as ndjson, so an empty decision is an empty file.
|
|
44
|
+
# changed the gate. One byte separates "the dataset moved" from "nothing happened", and
|
|
45
|
+
# a skip here skips the copy behind it while the run stays green.
|
|
46
|
+
# publish storage.copy, promoting the url list to a path a downstream pipeline watches.
|
|
47
|
+
# Copying rather than rewriting keeps the published object byte-identical to what
|
|
48
|
+
# the decision produced.
|
|
49
|
+
# current_marker what the next run should compare against, computed under rule: all_done so
|
|
50
|
+
# it is produced whether the dataset moved, did not move, or was seen for the
|
|
51
|
+
# first time.
|
|
52
|
+
# remember that marker written over the old one. Writing it last is what makes a failed run
|
|
53
|
+
# repeatable: the marker only advances on a run that got all the way here.
|
|
54
|
+
#
|
|
55
|
+
# To make it yours: change dataset_id to any HDX slug, and give the marker a stable home.
|
|
56
|
+
# The same shape watches any CKAN instance -- data.gov, the EU portal, a national one -- since
|
|
57
|
+
# package_show and metadata_modified are CKAN's, not HDX's.
|
|
58
|
+
#
|
|
59
|
+
# dg run --local examples/open-data/hdx-dataset-watch.yaml
|
|
60
|
+
# dg run --local examples/open-data/hdx-dataset-watch.yaml -p dataset_id=malawi-healthsites
|
|
61
|
+
|
|
62
|
+
format: dirigent/v1
|
|
63
|
+
kind: pipeline
|
|
64
|
+
code: hdx-dataset-watch
|
|
65
|
+
name: HDX dataset watch
|
|
66
|
+
description: |
|
|
67
|
+
Poll one **HDX** (CKAN) dataset for a new `metadata_modified`, comparing against a marker
|
|
68
|
+
object in storage, and publish the resource url list when it moves.
|
|
69
|
+
|
|
70
|
+
There is no state block: the marker file *is* the state, and a sensor looking for it is how
|
|
71
|
+
a run asks whether there was a previous one.
|
|
72
|
+
|
|
73
|
+
tags: [open-data, http, sensor, storage, transform, starter]
|
|
74
|
+
|
|
75
|
+
requires:
|
|
76
|
+
blocks:
|
|
77
|
+
- http.request
|
|
78
|
+
- transform.jq
|
|
79
|
+
- convert.std
|
|
80
|
+
- storage.exists
|
|
81
|
+
- storage.read
|
|
82
|
+
- storage.write
|
|
83
|
+
- storage.copy
|
|
84
|
+
|
|
85
|
+
params:
|
|
86
|
+
type: object
|
|
87
|
+
additionalProperties: false
|
|
88
|
+
properties:
|
|
89
|
+
dataset_id:
|
|
90
|
+
type: string
|
|
91
|
+
default: malawi-healthsites
|
|
92
|
+
description: An HDX dataset slug, as it appears in its page URL.
|
|
93
|
+
|
|
94
|
+
steps:
|
|
95
|
+
dataset:
|
|
96
|
+
block: http.request
|
|
97
|
+
config:
|
|
98
|
+
url: https://data.humdata.org/api/3/action/package_show
|
|
99
|
+
query:
|
|
100
|
+
id: ${params.dataset_id}
|
|
101
|
+
# A package with many resources and a long description is still small, but not tiny.
|
|
102
|
+
max_response: 8mb
|
|
103
|
+
|
|
104
|
+
current:
|
|
105
|
+
block: transform.jq
|
|
106
|
+
depends_on: [dataset]
|
|
107
|
+
config:
|
|
108
|
+
input: ${steps.dataset.output.body}
|
|
109
|
+
# download_url rather than url: CKAN carries both, and for an uploaded file the plain
|
|
110
|
+
# url can be the landing page rather than the bytes.
|
|
111
|
+
program: |
|
|
112
|
+
.result
|
|
113
|
+
| {
|
|
114
|
+
id: .name,
|
|
115
|
+
title: .title,
|
|
116
|
+
metadata_modified: .metadata_modified,
|
|
117
|
+
resources: [.resources[]
|
|
118
|
+
| {name, format, last_modified, url: (.download_url // .url)}]
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
seen:
|
|
122
|
+
block: storage.exists
|
|
123
|
+
depends_on: [current]
|
|
124
|
+
poll: 1s
|
|
125
|
+
deadline: 5s
|
|
126
|
+
# No marker means no previous run to compare against. That is not a failure and not a
|
|
127
|
+
# reason to publish everything; it is a first sighting, and the branch below skips.
|
|
128
|
+
on_timeout: skip
|
|
129
|
+
config:
|
|
130
|
+
uri: ${run.scratch}/markers/${params.dataset_id}.json
|
|
131
|
+
|
|
132
|
+
recalled:
|
|
133
|
+
block: storage.read
|
|
134
|
+
depends_on: [seen]
|
|
135
|
+
config:
|
|
136
|
+
source: ${steps.seen.output.uri}
|
|
137
|
+
|
|
138
|
+
previous:
|
|
139
|
+
block: transform.jq
|
|
140
|
+
depends_on: [recalled]
|
|
141
|
+
config:
|
|
142
|
+
input: ${steps.recalled.output.value}
|
|
143
|
+
program: |
|
|
144
|
+
{metadata_modified}
|
|
145
|
+
|
|
146
|
+
moved:
|
|
147
|
+
block: transform.jq
|
|
148
|
+
depends_on: [current, previous]
|
|
149
|
+
config:
|
|
150
|
+
input:
|
|
151
|
+
current: ${steps.current.output.value}
|
|
152
|
+
previous: ${steps.previous.output.value}
|
|
153
|
+
# An empty list is a decision, not an absence: it says the stamps matched.
|
|
154
|
+
program: |
|
|
155
|
+
if .current.metadata_modified == .previous.metadata_modified
|
|
156
|
+
then []
|
|
157
|
+
else [.current.resources[].url]
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
decided:
|
|
161
|
+
block: storage.write
|
|
162
|
+
depends_on: [moved]
|
|
163
|
+
config:
|
|
164
|
+
target: ${run.scratch}/moved/${params.dataset_id}.json
|
|
165
|
+
value: ${steps.moved.output.value}
|
|
166
|
+
|
|
167
|
+
written:
|
|
168
|
+
block: convert.std
|
|
169
|
+
depends_on: [decided]
|
|
170
|
+
config:
|
|
171
|
+
source: ${steps.decided.output.uri}
|
|
172
|
+
target: ${run.scratch}/moved/${params.dataset_id}.ndjson
|
|
173
|
+
from: json
|
|
174
|
+
to: ndjson
|
|
175
|
+
|
|
176
|
+
changed:
|
|
177
|
+
block: storage.exists
|
|
178
|
+
depends_on: [written]
|
|
179
|
+
poll: 1s
|
|
180
|
+
deadline: 3s
|
|
181
|
+
on_timeout: skip
|
|
182
|
+
config:
|
|
183
|
+
uri: ${steps.written.output.target}
|
|
184
|
+
min_size: 1b
|
|
185
|
+
|
|
186
|
+
publish:
|
|
187
|
+
block: storage.copy
|
|
188
|
+
depends_on: [written, changed]
|
|
189
|
+
config:
|
|
190
|
+
source: ${steps.written.output.target}
|
|
191
|
+
target: ${run.scratch}/published/${params.dataset_id}-resources.ndjson
|
|
192
|
+
|
|
193
|
+
current_marker:
|
|
194
|
+
block: transform.jq
|
|
195
|
+
depends_on: [current, changed]
|
|
196
|
+
# all_done, because the marker has to advance on every outcome above it: a first sighting
|
|
197
|
+
# that skipped the comparison, a run where nothing moved, and a run that published.
|
|
198
|
+
rule: all_done
|
|
199
|
+
config:
|
|
200
|
+
input: ${steps.current.output.value}
|
|
201
|
+
program: |
|
|
202
|
+
{id, metadata_modified, resources: (.resources | length)}
|
|
203
|
+
|
|
204
|
+
remember:
|
|
205
|
+
block: storage.write
|
|
206
|
+
depends_on: [current_marker]
|
|
207
|
+
config:
|
|
208
|
+
# The same path the sensor at the top looks for, which is what makes this run's
|
|
209
|
+
# observation the next run's "previous".
|
|
210
|
+
target: ${run.scratch}/markers/${params.dataset_id}.json
|
|
211
|
+
value: ${steps.current_marker.output.value}
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# Form submissions out of KoboToolbox and into a csv, with the one credential on this shelf.
|
|
2
|
+
#
|
|
3
|
+
# NEEDS A TOKEN. Everything else here is public and keyless; Kobo is not, and it should not be:
|
|
4
|
+
# the submissions are somebody's household survey. The API token is per-account, printed at
|
|
5
|
+
# <server>/token/?format=json once you are logged in, and it is sent as `Authorization: Token
|
|
6
|
+
# <token>`. There is no anonymous mode to fall back to, so this document is the one example on
|
|
7
|
+
# the shelf that cannot run against a stranger's server.
|
|
8
|
+
#
|
|
9
|
+
# Where the token belongs. It is a parameter here so the document reads as one piece, and a
|
|
10
|
+
# parameter is *not* a secret store: it is recorded with the run and visible to anyone who can
|
|
11
|
+
# read it. On an instance, create an http connection holding the token and give the step
|
|
12
|
+
# `connection:` instead of `url:` and `headers:` --
|
|
13
|
+
#
|
|
14
|
+
# dg connection create http kobo \
|
|
15
|
+
# --set base_url=https://kf.kobotoolbox.org \
|
|
16
|
+
# --set 'headers.Authorization=Token <your token>'
|
|
17
|
+
#
|
|
18
|
+
# -- and the credential is encrypted at rest, redacted in every API response, and rotated in
|
|
19
|
+
# one place instead of in every run's parameters.
|
|
20
|
+
#
|
|
21
|
+
# The API: GET /api/v2/assets/{uid}/data/?format=json answers {"count": N, "next": ..., "results":
|
|
22
|
+
# [...]}, one object per submission. Kobo's own fields are underscore-prefixed -- `_id`,
|
|
23
|
+
# `_uuid`, `_submission_time`, `_geolocation` -- and everything else is a question, named by its
|
|
24
|
+
# XLSForm column name, with a group's questions namespaced as `group/question`.
|
|
25
|
+
#
|
|
26
|
+
# What happens, hop by hop:
|
|
27
|
+
#
|
|
28
|
+
# submissions one GET. `limit` is Kobo's page size; a form with more submissions than that
|
|
29
|
+
# answers a `next` url, and reading past the first page means following it -- which
|
|
30
|
+
# this document deliberately does not do, because a paging loop is not a step.
|
|
31
|
+
# rows the flattening. The Kobo metadata worth keeping becomes stable columns, and the
|
|
32
|
+
# questions named in `columns` become the rest. Naming the columns is what keeps a
|
|
33
|
+
# csv rectangular: submissions collected before a question was added simply do not
|
|
34
|
+
# have that key, and a codec cannot invent a header from a row that lacks it.
|
|
35
|
+
# staged storage.write, the rows as json. A converter reads one URI and writes another, so
|
|
36
|
+
# the rows are put down before they can be re-encoded.
|
|
37
|
+
# report json to csv in storage, named after the asset so several forms can share a
|
|
38
|
+
# prefix.
|
|
39
|
+
#
|
|
40
|
+
# To make it yours: set server if you self-host, asset_uid to the form's uid (it is in the URL
|
|
41
|
+
# of the form's page), and columns to the questions you actually report on.
|
|
42
|
+
#
|
|
43
|
+
# dg run --local examples/open-data/kobo-submissions-to-csv.yaml \
|
|
44
|
+
# -p asset_uid=aBcDeFgHiJkLmNoPqRsTuV -p token=<your token> \
|
|
45
|
+
# -p columns='["respondent_age","district"]'
|
|
46
|
+
|
|
47
|
+
format: dirigent/v1
|
|
48
|
+
kind: pipeline
|
|
49
|
+
code: kobo-submissions-to-csv
|
|
50
|
+
name: Kobo submissions to csv
|
|
51
|
+
description: |
|
|
52
|
+
Submissions for one **KoboToolbox** form, flattened to named columns and written as csv.
|
|
53
|
+
|
|
54
|
+
The only document on this shelf that needs a credential: Kobo has no anonymous read, and the
|
|
55
|
+
token belongs in a connection on any instance you keep this on.
|
|
56
|
+
|
|
57
|
+
tags: [open-data, http, storage, transform, credential, csv]
|
|
58
|
+
|
|
59
|
+
requires:
|
|
60
|
+
blocks:
|
|
61
|
+
- http.request
|
|
62
|
+
- transform.jq
|
|
63
|
+
- storage.write
|
|
64
|
+
- convert.std
|
|
65
|
+
|
|
66
|
+
params:
|
|
67
|
+
type: object
|
|
68
|
+
required: [asset_uid, token]
|
|
69
|
+
additionalProperties: false
|
|
70
|
+
properties:
|
|
71
|
+
server:
|
|
72
|
+
type: string
|
|
73
|
+
default: https://kf.kobotoolbox.org
|
|
74
|
+
description: The Kobo server; kf.kobotoolbox.org is the global one, and self-hosting is common.
|
|
75
|
+
asset_uid:
|
|
76
|
+
type: string
|
|
77
|
+
description: The form's uid, from its page URL.
|
|
78
|
+
token:
|
|
79
|
+
type: string
|
|
80
|
+
description: |
|
|
81
|
+
A Kobo API token, required: there is no public read. Given as a parameter only so this
|
|
82
|
+
document stands alone; on an instance it belongs in a connection.
|
|
83
|
+
columns:
|
|
84
|
+
type: array
|
|
85
|
+
default: []
|
|
86
|
+
items:
|
|
87
|
+
type: string
|
|
88
|
+
description: Question names to include as columns; empty means the keys of the first submission.
|
|
89
|
+
limit:
|
|
90
|
+
type: integer
|
|
91
|
+
default: 200
|
|
92
|
+
description: Kobo's page size; this document reads one page and does not follow `next`.
|
|
93
|
+
|
|
94
|
+
steps:
|
|
95
|
+
submissions:
|
|
96
|
+
block: http.request
|
|
97
|
+
config:
|
|
98
|
+
url: ${params.server}/api/v2/assets/${params.asset_uid}/data/
|
|
99
|
+
headers:
|
|
100
|
+
# Kobo's scheme is the literal word Token, not Bearer.
|
|
101
|
+
Authorization: Token ${params.token}
|
|
102
|
+
query:
|
|
103
|
+
format: json
|
|
104
|
+
limit: ${params.limit}
|
|
105
|
+
# A survey with photos and long text answers is not small.
|
|
106
|
+
max_response: 64mb
|
|
107
|
+
|
|
108
|
+
rows:
|
|
109
|
+
block: transform.jq
|
|
110
|
+
depends_on: [submissions]
|
|
111
|
+
config:
|
|
112
|
+
input:
|
|
113
|
+
payload: ${steps.submissions.output.body}
|
|
114
|
+
columns: ${params.columns}
|
|
115
|
+
# The underscore-prefixed keys are Kobo's own and are renamed here; the questions are
|
|
116
|
+
# taken by name so every row has the same shape, which is what a csv requires. A
|
|
117
|
+
# submission missing one of them gets a null rather than a shorter row.
|
|
118
|
+
program: |
|
|
119
|
+
. as {$payload, $columns}
|
|
120
|
+
| ($payload.results // []) as $rows
|
|
121
|
+
| (if ($columns | length) > 0
|
|
122
|
+
then $columns
|
|
123
|
+
else ($rows[0] // {} | keys | map(select(startswith("_") | not)))
|
|
124
|
+
end) as $questions
|
|
125
|
+
| [$rows[]
|
|
126
|
+
| . as $row
|
|
127
|
+
| {
|
|
128
|
+
submission_id: ._id,
|
|
129
|
+
uuid: ._uuid,
|
|
130
|
+
submitted_at: ._submission_time,
|
|
131
|
+
submitted_by: (._submitted_by // null)
|
|
132
|
+
}
|
|
133
|
+
+ (reduce $questions[] as $q ({}; .[$q] = ($row[$q] // null)))]
|
|
134
|
+
|
|
135
|
+
staged:
|
|
136
|
+
block: storage.write
|
|
137
|
+
depends_on: [rows]
|
|
138
|
+
config:
|
|
139
|
+
target: ${run.scratch}/kobo/${params.asset_uid}.json
|
|
140
|
+
value: ${steps.rows.output.value}
|
|
141
|
+
|
|
142
|
+
report:
|
|
143
|
+
block: convert.std
|
|
144
|
+
depends_on: [staged]
|
|
145
|
+
config:
|
|
146
|
+
source: ${steps.staged.output.uri}
|
|
147
|
+
target: ${run.scratch}/kobo/${params.asset_uid}.csv
|
|
148
|
+
from: json
|
|
149
|
+
to: csv
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
# A short list of facility names turned into coordinates, one request per second, as asked.
|
|
2
|
+
#
|
|
3
|
+
# Nominatim is OpenStreetMap's geocoder, public and keyless: GET /search?q=...&format=jsonv2
|
|
4
|
+
# answers a list of candidates, best first, each with `lat` and `lon` as strings, a
|
|
5
|
+
# `display_name`, a `category`/`type` pair saying what kind of thing it matched, and an
|
|
6
|
+
# `importance` score. An empty list is the honest answer for a name it cannot place.
|
|
7
|
+
#
|
|
8
|
+
# The usage policy is the design constraint here, not the API. The public instance allows one
|
|
9
|
+
# request per second from one client and asks for a User-Agent that identifies the application.
|
|
10
|
+
# A fan-out would send every name at once, which is the polite way to get blocked, so the
|
|
11
|
+
# lookups are written out as a chain with a `time.sleep` between them: the rate limit is in the
|
|
12
|
+
# shape of the graph, where it is visible, rather than in a comment nobody enforces. Three names
|
|
13
|
+
# is what a chain is good for; a list of three hundred belongs on your own Nominatim, and then
|
|
14
|
+
# it is a fan-out.
|
|
15
|
+
#
|
|
16
|
+
# What happens, hop by hop:
|
|
17
|
+
#
|
|
18
|
+
# facilities the input, written into the document with value.const. It is a list of objects
|
|
19
|
+
# rather than strings, so a name a person recognises and the query string sent to
|
|
20
|
+
# the geocoder can differ -- which they always end up doing.
|
|
21
|
+
# lookup_* one search each, indexed out of that list. A geocoder cannot promise a match, so
|
|
22
|
+
# each of these tolerates failure: continue_on_failure means a name that cannot be
|
|
23
|
+
# resolved leaves a row with no coordinates instead of ending the run.
|
|
24
|
+
# pause_* one second, which is the whole point of the chain.
|
|
25
|
+
# table the join. The three answers arrive as one list in this step's input, the first
|
|
26
|
+
# candidate of each is taken, and the strings Nominatim returns for lat and lon are
|
|
27
|
+
# turned into numbers -- a coordinate that has to be parsed before it can be
|
|
28
|
+
# compared is not a coordinate.
|
|
29
|
+
#
|
|
30
|
+
# To make it yours: replace the value.const list, and set user_agent to something that says who
|
|
31
|
+
# you are. Any name that resolves ambiguously is worth sending with more context in `query`:
|
|
32
|
+
# Nominatim reads "Central Hospital, Lilongwe, Malawi" far better than "Central Hospital".
|
|
33
|
+
#
|
|
34
|
+
# dg run --local examples/open-data/nominatim-geocode-facilities.yaml
|
|
35
|
+
|
|
36
|
+
format: dirigent/v1
|
|
37
|
+
kind: pipeline
|
|
38
|
+
code: nominatim-geocode-facilities
|
|
39
|
+
name: Geocode facilities with Nominatim
|
|
40
|
+
description: |
|
|
41
|
+
A handful of facility names geocoded through **Nominatim**, one request per second, joined
|
|
42
|
+
into one table of coordinates.
|
|
43
|
+
|
|
44
|
+
The rate limit is expressed as the shape of the graph: a chain with a `time.sleep` between
|
|
45
|
+
the lookups, not a fan-out.
|
|
46
|
+
|
|
47
|
+
tags: [open-data, http, sensor, transform, geocode]
|
|
48
|
+
|
|
49
|
+
requires:
|
|
50
|
+
blocks:
|
|
51
|
+
- value.const
|
|
52
|
+
- http.request
|
|
53
|
+
- time.sleep
|
|
54
|
+
- transform.jq
|
|
55
|
+
|
|
56
|
+
params:
|
|
57
|
+
type: object
|
|
58
|
+
additionalProperties: false
|
|
59
|
+
properties:
|
|
60
|
+
user_agent:
|
|
61
|
+
type: string
|
|
62
|
+
default: dirigent-example/1.0 (https://github.com/winterop-com/dirigent)
|
|
63
|
+
description: Required by the usage policy; identify the application and how to reach you.
|
|
64
|
+
country_codes:
|
|
65
|
+
type: string
|
|
66
|
+
default: mw
|
|
67
|
+
description: ISO 3166-1 alpha-2 codes, comma-separated, that the search is confined to.
|
|
68
|
+
|
|
69
|
+
steps:
|
|
70
|
+
facilities:
|
|
71
|
+
block: value.const
|
|
72
|
+
config:
|
|
73
|
+
# The list the run works from. `name` is what the output is keyed by and `query` is what
|
|
74
|
+
# is asked, because the name a register holds is rarely the string a geocoder wants.
|
|
75
|
+
value:
|
|
76
|
+
- name: Kamuzu Central Hospital
|
|
77
|
+
query: Kamuzu Central Hospital, Lilongwe, Malawi
|
|
78
|
+
- name: Queen Elizabeth Central Hospital
|
|
79
|
+
query: Queen Elizabeth Central Hospital, Blantyre, Malawi
|
|
80
|
+
- name: Mzuzu Central Hospital
|
|
81
|
+
query: Mzuzu Central Hospital, Mzuzu, Malawi
|
|
82
|
+
|
|
83
|
+
lookup_1:
|
|
84
|
+
block: http.request
|
|
85
|
+
depends_on: [facilities]
|
|
86
|
+
# A name the geocoder cannot place is a fact about the name, not a broken pipeline; the
|
|
87
|
+
# chain carries on and the table below records the gap.
|
|
88
|
+
continue_on_failure: true
|
|
89
|
+
config:
|
|
90
|
+
url: https://nominatim.openstreetmap.org/search
|
|
91
|
+
headers:
|
|
92
|
+
User-Agent: ${params.user_agent}
|
|
93
|
+
query:
|
|
94
|
+
q: ${steps.facilities.output.value.0.query}
|
|
95
|
+
# jsonv2 rather than json: it is the current shape, and it names the match's category
|
|
96
|
+
# and type, which is how you tell a hospital from a bus stop called after one.
|
|
97
|
+
format: jsonv2
|
|
98
|
+
limit: 1
|
|
99
|
+
countrycodes: ${params.country_codes}
|
|
100
|
+
|
|
101
|
+
pause_1:
|
|
102
|
+
block: time.sleep
|
|
103
|
+
depends_on: [lookup_1]
|
|
104
|
+
# The policy is one request per second from one client, so the wait sits between the
|
|
105
|
+
# requests and not beside them.
|
|
106
|
+
config:
|
|
107
|
+
for: 1s
|
|
108
|
+
|
|
109
|
+
lookup_2:
|
|
110
|
+
block: http.request
|
|
111
|
+
depends_on: [facilities, pause_1]
|
|
112
|
+
continue_on_failure: true
|
|
113
|
+
config:
|
|
114
|
+
url: https://nominatim.openstreetmap.org/search
|
|
115
|
+
headers:
|
|
116
|
+
User-Agent: ${params.user_agent}
|
|
117
|
+
query:
|
|
118
|
+
q: ${steps.facilities.output.value.1.query}
|
|
119
|
+
format: jsonv2
|
|
120
|
+
limit: 1
|
|
121
|
+
countrycodes: ${params.country_codes}
|
|
122
|
+
|
|
123
|
+
pause_2:
|
|
124
|
+
block: time.sleep
|
|
125
|
+
depends_on: [lookup_2]
|
|
126
|
+
config:
|
|
127
|
+
for: 1s
|
|
128
|
+
|
|
129
|
+
lookup_3:
|
|
130
|
+
block: http.request
|
|
131
|
+
depends_on: [facilities, pause_2]
|
|
132
|
+
continue_on_failure: true
|
|
133
|
+
config:
|
|
134
|
+
url: https://nominatim.openstreetmap.org/search
|
|
135
|
+
headers:
|
|
136
|
+
User-Agent: ${params.user_agent}
|
|
137
|
+
query:
|
|
138
|
+
q: ${steps.facilities.output.value.2.query}
|
|
139
|
+
format: jsonv2
|
|
140
|
+
limit: 1
|
|
141
|
+
countrycodes: ${params.country_codes}
|
|
142
|
+
|
|
143
|
+
table:
|
|
144
|
+
block: transform.jq
|
|
145
|
+
depends_on: [facilities, lookup_1, lookup_2, lookup_3]
|
|
146
|
+
config:
|
|
147
|
+
input:
|
|
148
|
+
facilities: ${steps.facilities.output.value}
|
|
149
|
+
answers:
|
|
150
|
+
- ${steps.lookup_1.output.body}
|
|
151
|
+
- ${steps.lookup_2.output.body}
|
|
152
|
+
- ${steps.lookup_3.output.body}
|
|
153
|
+
# lat and lon arrive as strings; tonumber here is what stops every later consumer from
|
|
154
|
+
# guessing. A name with no candidate keeps its row, with nulls where the coordinate
|
|
155
|
+
# would be, so the gap is countable.
|
|
156
|
+
program: |
|
|
157
|
+
[range(.facilities | length) as $i
|
|
158
|
+
| .facilities[$i] as $facility
|
|
159
|
+
| (.answers[$i][0] // null) as $hit
|
|
160
|
+
| {
|
|
161
|
+
name: $facility.name,
|
|
162
|
+
query: $facility.query,
|
|
163
|
+
matched: ($hit != null),
|
|
164
|
+
display_name: ($hit.display_name // null),
|
|
165
|
+
category: ($hit.category // null),
|
|
166
|
+
kind: ($hit.type // null),
|
|
167
|
+
latitude: (if $hit then ($hit.lat | tonumber) else null end),
|
|
168
|
+
longitude: (if $hit then ($hit.lon | tonumber) else null end)
|
|
169
|
+
}]
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# The same survey story against ODK Central, where the credential lives in a connection.
|
|
2
|
+
#
|
|
3
|
+
# NEEDS AN ODK CENTRAL SERVER AND AN ACCOUNT. Central serves nothing anonymously, so this
|
|
4
|
+
# document names a connection the instance holds rather than carrying anything itself:
|
|
5
|
+
#
|
|
6
|
+
# dg connection create http odk-central \
|
|
7
|
+
# --set base_url=https://central.example.org \
|
|
8
|
+
# --set basic_username=you@example.org \
|
|
9
|
+
# --set basic_password='<your password>'
|
|
10
|
+
#
|
|
11
|
+
# Central accepts HTTP Basic on the /v1 API, which is what makes a plain http connection enough
|
|
12
|
+
# and why this document has no session step: the alternative is POST /v1/sessions to trade the
|
|
13
|
+
# password for a bearer token, which is a second request and a token to keep alive for no gain
|
|
14
|
+
# in a pipeline that runs once an hour.
|
|
15
|
+
#
|
|
16
|
+
# The API: every form exposes an OData feed at
|
|
17
|
+
# /v1/projects/{id}/forms/{xmlFormId}.svc/Submissions, answering {"@odata.context": ...,
|
|
18
|
+
# "value": [...]} -- one object per submission, questions at the top level, groups nested as
|
|
19
|
+
# objects, and Central's own metadata under `__id` and `__system` (`submissionDate`,
|
|
20
|
+
# `submitterId`, `reviewState`). Central will also hand you the whole form as csv at
|
|
21
|
+
# /v1/projects/{id}/forms/{xmlFormId}/submissions.csv, and that is the better answer when a csv
|
|
22
|
+
# is all you want; the feed is what you read when the pipeline has to filter, reshape, or check
|
|
23
|
+
# the data before anything downstream sees it.
|
|
24
|
+
#
|
|
25
|
+
# What happens, hop by hop:
|
|
26
|
+
#
|
|
27
|
+
# submissions one GET through the connection. $top is OData's page size and $skip its
|
|
28
|
+
# offset; Central also honours $filter over __system/submissionDate, which is the
|
|
29
|
+
# natural place to put a run window on a scheduled load.
|
|
30
|
+
# rows the flattening: Central's metadata renamed into stable columns, the named
|
|
31
|
+
# questions read with a path so a grouped question is reachable as "group/field",
|
|
32
|
+
# and every row given the same keys.
|
|
33
|
+
# staged storage.write, the rows as json, because a converter reads a URI rather than a
|
|
34
|
+
# value the run is holding.
|
|
35
|
+
# report csv in storage, named for the form.
|
|
36
|
+
#
|
|
37
|
+
# To make it yours: set project and form_id, list the questions you report on, and add a
|
|
38
|
+
# $filter to the query once the form has more submissions than one page.
|
|
39
|
+
#
|
|
40
|
+
# dg apply examples/open-data/odk-central-submissions.yaml
|
|
41
|
+
# dg run odk-central-submissions -p project=1 -p form_id=household-survey --watch
|
|
42
|
+
|
|
43
|
+
format: dirigent/v1
|
|
44
|
+
kind: pipeline
|
|
45
|
+
code: odk-central-submissions
|
|
46
|
+
name: ODK Central submissions
|
|
47
|
+
description: |
|
|
48
|
+
Submissions for one **ODK Central** form, read from its OData feed through a connection that
|
|
49
|
+
holds the credential, flattened to named columns and written as csv.
|
|
50
|
+
|
|
51
|
+
Central serves nothing anonymously, so the credential is a connection the instance holds --
|
|
52
|
+
the same shape as the Kobo document, with the secret in the right place.
|
|
53
|
+
|
|
54
|
+
tags: [open-data, http, storage, transform, credential, csv]
|
|
55
|
+
|
|
56
|
+
requires:
|
|
57
|
+
blocks:
|
|
58
|
+
- http.request
|
|
59
|
+
- transform.jq
|
|
60
|
+
- storage.write
|
|
61
|
+
- convert.std
|
|
62
|
+
connections:
|
|
63
|
+
- odk-central
|
|
64
|
+
|
|
65
|
+
params:
|
|
66
|
+
type: object
|
|
67
|
+
required: [project, form_id]
|
|
68
|
+
additionalProperties: false
|
|
69
|
+
properties:
|
|
70
|
+
project:
|
|
71
|
+
type: integer
|
|
72
|
+
description: The numeric project id, as it appears in Central's own URLs.
|
|
73
|
+
form_id:
|
|
74
|
+
type: string
|
|
75
|
+
description: The form's xmlFormId, which is the id inside the XLSForm and not its title.
|
|
76
|
+
columns:
|
|
77
|
+
type: array
|
|
78
|
+
default: []
|
|
79
|
+
items:
|
|
80
|
+
type: string
|
|
81
|
+
description: |
|
|
82
|
+
Question paths to include, slash-separated for a grouped question ("group/field");
|
|
83
|
+
empty means the top-level keys of the first submission.
|
|
84
|
+
top:
|
|
85
|
+
type: integer
|
|
86
|
+
default: 250
|
|
87
|
+
description: OData's page size; this document reads one page.
|
|
88
|
+
|
|
89
|
+
steps:
|
|
90
|
+
submissions:
|
|
91
|
+
block: http.request
|
|
92
|
+
config:
|
|
93
|
+
# The base URL is the connection's, so moving from a staging Central to a production one
|
|
94
|
+
# edits the connection and no pipeline.
|
|
95
|
+
connection: odk-central
|
|
96
|
+
path: /v1/projects/${params.project}/forms/${params.form_id}.svc/Submissions
|
|
97
|
+
query:
|
|
98
|
+
# OData's own parameter names, dollar signs included.
|
|
99
|
+
$top: ${params.top}
|
|
100
|
+
# Without this the feed answers only the current version's fields, which silently
|
|
101
|
+
# drops answers collected under an earlier form version.
|
|
102
|
+
$expand: "*"
|
|
103
|
+
max_response: 64mb
|
|
104
|
+
|
|
105
|
+
rows:
|
|
106
|
+
block: transform.jq
|
|
107
|
+
depends_on: [submissions]
|
|
108
|
+
config:
|
|
109
|
+
input:
|
|
110
|
+
payload: ${steps.submissions.output.body}
|
|
111
|
+
columns: ${params.columns}
|
|
112
|
+
# getpath is what makes a grouped question reachable: "hh/members" is a path into a
|
|
113
|
+
# nested object, and splitting on "/" turns the column name into that path.
|
|
114
|
+
program: |
|
|
115
|
+
. as {$payload, $columns}
|
|
116
|
+
| ($payload.value // []) as $rows
|
|
117
|
+
| (if ($columns | length) > 0
|
|
118
|
+
then $columns
|
|
119
|
+
else ($rows[0] // {} | keys | map(select(startswith("__") | not)))
|
|
120
|
+
end) as $questions
|
|
121
|
+
| [$rows[]
|
|
122
|
+
| . as $row
|
|
123
|
+
| {
|
|
124
|
+
submission_id: .__id,
|
|
125
|
+
submitted_at: .__system.submissionDate,
|
|
126
|
+
submitter_id: .__system.submitterId,
|
|
127
|
+
review_state: .__system.reviewState
|
|
128
|
+
}
|
|
129
|
+
+ (reduce $questions[] as $q
|
|
130
|
+
({}; .[$q] = ($row | getpath($q | split("/")) // null)))]
|
|
131
|
+
|
|
132
|
+
staged:
|
|
133
|
+
block: storage.write
|
|
134
|
+
depends_on: [rows]
|
|
135
|
+
config:
|
|
136
|
+
target: ${run.scratch}/odk/${params.form_id}.json
|
|
137
|
+
value: ${steps.rows.output.value}
|
|
138
|
+
|
|
139
|
+
report:
|
|
140
|
+
block: convert.std
|
|
141
|
+
depends_on: [staged]
|
|
142
|
+
config:
|
|
143
|
+
source: ${steps.staged.output.uri}
|
|
144
|
+
target: ${run.scratch}/odk/${params.form_id}.csv
|
|
145
|
+
from: json
|
|
146
|
+
to: csv
|