dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,211 @@
1
+ # Watching one HDX dataset for a new version, with a file in storage standing in for state.
2
+ #
3
+ # The Humanitarian Data Exchange runs CKAN, whose API is keyless for public datasets:
4
+ # /api/3/action/package_show?id=<slug> answers {"success": true, "result": {...}}, and the
5
+ # result carries `metadata_modified` -- an ISO timestamp that moves whenever anything about the
6
+ # dataset changes -- plus a `resources` list, one entry per downloadable file with its own
7
+ # `download_url` and `last_modified`.
8
+ #
9
+ # THE PATTERN THIS TEACHES IS THE MARKER. There is no state block: nothing in dirigent
10
+ # remembers, between runs, what a previous run saw. What there is, is storage, and a marker is
11
+ # just a small object written at a stable path holding what the last run observed. A sensor
12
+ # looks for it, storage.read brings it back into the run, the comparison decides, and
13
+ # storage.write puts the new one down. The whole state machine is a handful of steps and one
14
+ # file, and it works the same against s3:// as against a local artifact root.
15
+ #
16
+ # Two consequences worth reading before copying this:
17
+ #
18
+ # - The marker below is written under ${run.scratch}, which is deleted with the run, so every
19
+ # `--local` run is a first sighting: the sensor skips, the comparison skips with it, and the
20
+ # run ends succeeded having only written a marker. That is the honest first-run behaviour,
21
+ # and it is what the graph shows. On an instance, write the marker at a stable location --
22
+ # s3://your-bucket/markers/... -- and the second run is the one that compares.
23
+ # - The marker path is written out in the two steps that name it, the sensor and the last
24
+ # write, rather than kept in a parameter, because a parameter default is literal text:
25
+ # ${run.scratch} in a default is those characters and not a reference to anything.
26
+ #
27
+ # What happens, hop by hop:
28
+ #
29
+ # dataset one GET of the package. CKAN answers success/result even for an error, so the
30
+ # step's own 2xx check is not the whole story; the transform below reads .result
31
+ # and would fail loudly on anything else.
32
+ # current what this run saw: the id, the stamp, and the resource list, reduced to what a
33
+ # consumer actually needs.
34
+ # seen is there a marker? Five seconds is not a wait for the world, it is a lookup with
35
+ # a deadline; an absent marker skips this step and everything reading it.
36
+ # recalled storage.read of that object, which is the only way the previous stamp comes back
37
+ # into the run.
38
+ # previous the stamp taken out of it.
39
+ # moved the comparison. Equal stamps produce an empty list, a moved stamp produces the
40
+ # resource urls: the decision is data, not a status.
41
+ # decided storage.write, the decision put down as json, since the converter below reads a
42
+ # URI rather than a value.
43
+ # written that list as ndjson, so an empty decision is an empty file.
44
+ # changed the gate. One byte separates "the dataset moved" from "nothing happened", and
45
+ # a skip here skips the copy behind it while the run stays green.
46
+ # publish storage.copy, promoting the url list to a path a downstream pipeline watches.
47
+ # Copying rather than rewriting keeps the published object byte-identical to what
48
+ # the decision produced.
49
+ # current_marker what the next run should compare against, computed under rule: all_done so
50
+ # it is produced whether the dataset moved, did not move, or was seen for the
51
+ # first time.
52
+ # remember that marker written over the old one. Writing it last is what makes a failed run
53
+ # repeatable: the marker only advances on a run that got all the way here.
54
+ #
55
+ # To make it yours: change dataset_id to any HDX slug, and give the marker a stable home.
56
+ # The same shape watches any CKAN instance -- data.gov, the EU portal, a national one -- since
57
+ # package_show and metadata_modified are CKAN's, not HDX's.
58
+ #
59
+ # dg run --local examples/open-data/hdx-dataset-watch.yaml
60
+ # dg run --local examples/open-data/hdx-dataset-watch.yaml -p dataset_id=malawi-healthsites
61
+
62
+ format: dirigent/v1
63
+ kind: pipeline
64
+ code: hdx-dataset-watch
65
+ name: HDX dataset watch
66
+ description: |
67
+ Poll one **HDX** (CKAN) dataset for a new `metadata_modified`, comparing against a marker
68
+ object in storage, and publish the resource url list when it moves.
69
+
70
+ There is no state block: the marker file *is* the state, and a sensor looking for it is how
71
+ a run asks whether there was a previous one.
72
+
73
+ tags: [open-data, http, sensor, storage, transform, starter]
74
+
75
+ requires:
76
+ blocks:
77
+ - http.request
78
+ - transform.jq
79
+ - convert.std
80
+ - storage.exists
81
+ - storage.read
82
+ - storage.write
83
+ - storage.copy
84
+
85
+ params:
86
+ type: object
87
+ additionalProperties: false
88
+ properties:
89
+ dataset_id:
90
+ type: string
91
+ default: malawi-healthsites
92
+ description: An HDX dataset slug, as it appears in its page URL.
93
+
94
+ steps:
95
+ dataset:
96
+ block: http.request
97
+ config:
98
+ url: https://data.humdata.org/api/3/action/package_show
99
+ query:
100
+ id: ${params.dataset_id}
101
+ # A package with many resources and a long description is still small, but not tiny.
102
+ max_response: 8mb
103
+
104
+ current:
105
+ block: transform.jq
106
+ depends_on: [dataset]
107
+ config:
108
+ input: ${steps.dataset.output.body}
109
+ # download_url rather than url: CKAN carries both, and for an uploaded file the plain
110
+ # url can be the landing page rather than the bytes.
111
+ program: |
112
+ .result
113
+ | {
114
+ id: .name,
115
+ title: .title,
116
+ metadata_modified: .metadata_modified,
117
+ resources: [.resources[]
118
+ | {name, format, last_modified, url: (.download_url // .url)}]
119
+ }
120
+
121
+ seen:
122
+ block: storage.exists
123
+ depends_on: [current]
124
+ poll: 1s
125
+ deadline: 5s
126
+ # No marker means no previous run to compare against. That is not a failure and not a
127
+ # reason to publish everything; it is a first sighting, and the branch below skips.
128
+ on_timeout: skip
129
+ config:
130
+ uri: ${run.scratch}/markers/${params.dataset_id}.json
131
+
132
+ recalled:
133
+ block: storage.read
134
+ depends_on: [seen]
135
+ config:
136
+ source: ${steps.seen.output.uri}
137
+
138
+ previous:
139
+ block: transform.jq
140
+ depends_on: [recalled]
141
+ config:
142
+ input: ${steps.recalled.output.value}
143
+ program: |
144
+ {metadata_modified}
145
+
146
+ moved:
147
+ block: transform.jq
148
+ depends_on: [current, previous]
149
+ config:
150
+ input:
151
+ current: ${steps.current.output.value}
152
+ previous: ${steps.previous.output.value}
153
+ # An empty list is a decision, not an absence: it says the stamps matched.
154
+ program: |
155
+ if .current.metadata_modified == .previous.metadata_modified
156
+ then []
157
+ else [.current.resources[].url]
158
+ end
159
+
160
+ decided:
161
+ block: storage.write
162
+ depends_on: [moved]
163
+ config:
164
+ target: ${run.scratch}/moved/${params.dataset_id}.json
165
+ value: ${steps.moved.output.value}
166
+
167
+ written:
168
+ block: convert.std
169
+ depends_on: [decided]
170
+ config:
171
+ source: ${steps.decided.output.uri}
172
+ target: ${run.scratch}/moved/${params.dataset_id}.ndjson
173
+ from: json
174
+ to: ndjson
175
+
176
+ changed:
177
+ block: storage.exists
178
+ depends_on: [written]
179
+ poll: 1s
180
+ deadline: 3s
181
+ on_timeout: skip
182
+ config:
183
+ uri: ${steps.written.output.target}
184
+ min_size: 1b
185
+
186
+ publish:
187
+ block: storage.copy
188
+ depends_on: [written, changed]
189
+ config:
190
+ source: ${steps.written.output.target}
191
+ target: ${run.scratch}/published/${params.dataset_id}-resources.ndjson
192
+
193
+ current_marker:
194
+ block: transform.jq
195
+ depends_on: [current, changed]
196
+ # all_done, because the marker has to advance on every outcome above it: a first sighting
197
+ # that skipped the comparison, a run where nothing moved, and a run that published.
198
+ rule: all_done
199
+ config:
200
+ input: ${steps.current.output.value}
201
+ program: |
202
+ {id, metadata_modified, resources: (.resources | length)}
203
+
204
+ remember:
205
+ block: storage.write
206
+ depends_on: [current_marker]
207
+ config:
208
+ # The same path the sensor at the top looks for, which is what makes this run's
209
+ # observation the next run's "previous".
210
+ target: ${run.scratch}/markers/${params.dataset_id}.json
211
+ value: ${steps.current_marker.output.value}
@@ -0,0 +1,149 @@
1
+ # Form submissions out of KoboToolbox and into a csv, with the one credential on this shelf.
2
+ #
3
+ # NEEDS A TOKEN. Everything else here is public and keyless; Kobo is not, and it should not be:
4
+ # the submissions are somebody's household survey. The API token is per-account, printed at
5
+ # <server>/token/?format=json once you are logged in, and it is sent as `Authorization: Token
6
+ # <token>`. There is no anonymous mode to fall back to, so this document is the one example on
7
+ # the shelf that cannot run against a stranger's server.
8
+ #
9
+ # Where the token belongs. It is a parameter here so the document reads as one piece, and a
10
+ # parameter is *not* a secret store: it is recorded with the run and visible to anyone who can
11
+ # read it. On an instance, create an http connection holding the token and give the step
12
+ # `connection:` instead of `url:` and `headers:` --
13
+ #
14
+ # dg connection create http kobo \
15
+ # --set base_url=https://kf.kobotoolbox.org \
16
+ # --set 'headers.Authorization=Token <your token>'
17
+ #
18
+ # -- and the credential is encrypted at rest, redacted in every API response, and rotated in
19
+ # one place instead of in every run's parameters.
20
+ #
21
+ # The API: GET /api/v2/assets/{uid}/data/?format=json answers {"count": N, "next": ..., "results":
22
+ # [...]}, one object per submission. Kobo's own fields are underscore-prefixed -- `_id`,
23
+ # `_uuid`, `_submission_time`, `_geolocation` -- and everything else is a question, named by its
24
+ # XLSForm column name, with a group's questions namespaced as `group/question`.
25
+ #
26
+ # What happens, hop by hop:
27
+ #
28
+ # submissions one GET. `limit` is Kobo's page size; a form with more submissions than that
29
+ # answers a `next` url, and reading past the first page means following it -- which
30
+ # this document deliberately does not do, because a paging loop is not a step.
31
+ # rows the flattening. The Kobo metadata worth keeping becomes stable columns, and the
32
+ # questions named in `columns` become the rest. Naming the columns is what keeps a
33
+ # csv rectangular: submissions collected before a question was added simply do not
34
+ # have that key, and a codec cannot invent a header from a row that lacks it.
35
+ # staged storage.write, the rows as json. A converter reads one URI and writes another, so
36
+ # the rows are put down before they can be re-encoded.
37
+ # report json to csv in storage, named after the asset so several forms can share a
38
+ # prefix.
39
+ #
40
+ # To make it yours: set server if you self-host, asset_uid to the form's uid (it is in the URL
41
+ # of the form's page), and columns to the questions you actually report on.
42
+ #
43
+ # dg run --local examples/open-data/kobo-submissions-to-csv.yaml \
44
+ # -p asset_uid=aBcDeFgHiJkLmNoPqRsTuV -p token=<your token> \
45
+ # -p columns='["respondent_age","district"]'
46
+
47
+ format: dirigent/v1
48
+ kind: pipeline
49
+ code: kobo-submissions-to-csv
50
+ name: Kobo submissions to csv
51
+ description: |
52
+ Submissions for one **KoboToolbox** form, flattened to named columns and written as csv.
53
+
54
+ The only document on this shelf that needs a credential: Kobo has no anonymous read, and the
55
+ token belongs in a connection on any instance you keep this on.
56
+
57
+ tags: [open-data, http, storage, transform, credential, csv]
58
+
59
+ requires:
60
+ blocks:
61
+ - http.request
62
+ - transform.jq
63
+ - storage.write
64
+ - convert.std
65
+
66
+ params:
67
+ type: object
68
+ required: [asset_uid, token]
69
+ additionalProperties: false
70
+ properties:
71
+ server:
72
+ type: string
73
+ default: https://kf.kobotoolbox.org
74
+ description: The Kobo server; kf.kobotoolbox.org is the global one, and self-hosting is common.
75
+ asset_uid:
76
+ type: string
77
+ description: The form's uid, from its page URL.
78
+ token:
79
+ type: string
80
+ description: |
81
+ A Kobo API token, required: there is no public read. Given as a parameter only so this
82
+ document stands alone; on an instance it belongs in a connection.
83
+ columns:
84
+ type: array
85
+ default: []
86
+ items:
87
+ type: string
88
+ description: Question names to include as columns; empty means the keys of the first submission.
89
+ limit:
90
+ type: integer
91
+ default: 200
92
+ description: Kobo's page size; this document reads one page and does not follow `next`.
93
+
94
+ steps:
95
+ submissions:
96
+ block: http.request
97
+ config:
98
+ url: ${params.server}/api/v2/assets/${params.asset_uid}/data/
99
+ headers:
100
+ # Kobo's scheme is the literal word Token, not Bearer.
101
+ Authorization: Token ${params.token}
102
+ query:
103
+ format: json
104
+ limit: ${params.limit}
105
+ # A survey with photos and long text answers is not small.
106
+ max_response: 64mb
107
+
108
+ rows:
109
+ block: transform.jq
110
+ depends_on: [submissions]
111
+ config:
112
+ input:
113
+ payload: ${steps.submissions.output.body}
114
+ columns: ${params.columns}
115
+ # The underscore-prefixed keys are Kobo's own and are renamed here; the questions are
116
+ # taken by name so every row has the same shape, which is what a csv requires. A
117
+ # submission missing one of them gets a null rather than a shorter row.
118
+ program: |
119
+ . as {$payload, $columns}
120
+ | ($payload.results // []) as $rows
121
+ | (if ($columns | length) > 0
122
+ then $columns
123
+ else ($rows[0] // {} | keys | map(select(startswith("_") | not)))
124
+ end) as $questions
125
+ | [$rows[]
126
+ | . as $row
127
+ | {
128
+ submission_id: ._id,
129
+ uuid: ._uuid,
130
+ submitted_at: ._submission_time,
131
+ submitted_by: (._submitted_by // null)
132
+ }
133
+ + (reduce $questions[] as $q ({}; .[$q] = ($row[$q] // null)))]
134
+
135
+ staged:
136
+ block: storage.write
137
+ depends_on: [rows]
138
+ config:
139
+ target: ${run.scratch}/kobo/${params.asset_uid}.json
140
+ value: ${steps.rows.output.value}
141
+
142
+ report:
143
+ block: convert.std
144
+ depends_on: [staged]
145
+ config:
146
+ source: ${steps.staged.output.uri}
147
+ target: ${run.scratch}/kobo/${params.asset_uid}.csv
148
+ from: json
149
+ to: csv
@@ -0,0 +1,169 @@
1
+ # A short list of facility names turned into coordinates, one request per second, as asked.
2
+ #
3
+ # Nominatim is OpenStreetMap's geocoder, public and keyless: GET /search?q=...&format=jsonv2
4
+ # answers a list of candidates, best first, each with `lat` and `lon` as strings, a
5
+ # `display_name`, a `category`/`type` pair saying what kind of thing it matched, and an
6
+ # `importance` score. An empty list is the honest answer for a name it cannot place.
7
+ #
8
+ # The usage policy is the design constraint here, not the API. The public instance allows one
9
+ # request per second from one client and asks for a User-Agent that identifies the application.
10
+ # A fan-out would send every name at once, which is the polite way to get blocked, so the
11
+ # lookups are written out as a chain with a `time.sleep` between them: the rate limit is in the
12
+ # shape of the graph, where it is visible, rather than in a comment nobody enforces. Three names
13
+ # is what a chain is good for; a list of three hundred belongs on your own Nominatim, and then
14
+ # it is a fan-out.
15
+ #
16
+ # What happens, hop by hop:
17
+ #
18
+ # facilities the input, written into the document with value.const. It is a list of objects
19
+ # rather than strings, so a name a person recognises and the query string sent to
20
+ # the geocoder can differ -- which they always end up doing.
21
+ # lookup_* one search each, indexed out of that list. A geocoder cannot promise a match, so
22
+ # each of these tolerates failure: continue_on_failure means a name that cannot be
23
+ # resolved leaves a row with no coordinates instead of ending the run.
24
+ # pause_* one second, which is the whole point of the chain.
25
+ # table the join. The three answers arrive as one list in this step's input, the first
26
+ # candidate of each is taken, and the strings Nominatim returns for lat and lon are
27
+ # turned into numbers -- a coordinate that has to be parsed before it can be
28
+ # compared is not a coordinate.
29
+ #
30
+ # To make it yours: replace the value.const list, and set user_agent to something that says who
31
+ # you are. Any name that resolves ambiguously is worth sending with more context in `query`:
32
+ # Nominatim reads "Central Hospital, Lilongwe, Malawi" far better than "Central Hospital".
33
+ #
34
+ # dg run --local examples/open-data/nominatim-geocode-facilities.yaml
35
+
36
+ format: dirigent/v1
37
+ kind: pipeline
38
+ code: nominatim-geocode-facilities
39
+ name: Geocode facilities with Nominatim
40
+ description: |
41
+ A handful of facility names geocoded through **Nominatim**, one request per second, joined
42
+ into one table of coordinates.
43
+
44
+ The rate limit is expressed as the shape of the graph: a chain with a `time.sleep` between
45
+ the lookups, not a fan-out.
46
+
47
+ tags: [open-data, http, sensor, transform, geocode]
48
+
49
+ requires:
50
+ blocks:
51
+ - value.const
52
+ - http.request
53
+ - time.sleep
54
+ - transform.jq
55
+
56
+ params:
57
+ type: object
58
+ additionalProperties: false
59
+ properties:
60
+ user_agent:
61
+ type: string
62
+ default: dirigent-example/1.0 (https://github.com/winterop-com/dirigent)
63
+ description: Required by the usage policy; identify the application and how to reach you.
64
+ country_codes:
65
+ type: string
66
+ default: mw
67
+ description: ISO 3166-1 alpha-2 codes, comma-separated, that the search is confined to.
68
+
69
+ steps:
70
+ facilities:
71
+ block: value.const
72
+ config:
73
+ # The list the run works from. `name` is what the output is keyed by and `query` is what
74
+ # is asked, because the name a register holds is rarely the string a geocoder wants.
75
+ value:
76
+ - name: Kamuzu Central Hospital
77
+ query: Kamuzu Central Hospital, Lilongwe, Malawi
78
+ - name: Queen Elizabeth Central Hospital
79
+ query: Queen Elizabeth Central Hospital, Blantyre, Malawi
80
+ - name: Mzuzu Central Hospital
81
+ query: Mzuzu Central Hospital, Mzuzu, Malawi
82
+
83
+ lookup_1:
84
+ block: http.request
85
+ depends_on: [facilities]
86
+ # A name the geocoder cannot place is a fact about the name, not a broken pipeline; the
87
+ # chain carries on and the table below records the gap.
88
+ continue_on_failure: true
89
+ config:
90
+ url: https://nominatim.openstreetmap.org/search
91
+ headers:
92
+ User-Agent: ${params.user_agent}
93
+ query:
94
+ q: ${steps.facilities.output.value.0.query}
95
+ # jsonv2 rather than json: it is the current shape, and it names the match's category
96
+ # and type, which is how you tell a hospital from a bus stop called after one.
97
+ format: jsonv2
98
+ limit: 1
99
+ countrycodes: ${params.country_codes}
100
+
101
+ pause_1:
102
+ block: time.sleep
103
+ depends_on: [lookup_1]
104
+ # The policy is one request per second from one client, so the wait sits between the
105
+ # requests and not beside them.
106
+ config:
107
+ for: 1s
108
+
109
+ lookup_2:
110
+ block: http.request
111
+ depends_on: [facilities, pause_1]
112
+ continue_on_failure: true
113
+ config:
114
+ url: https://nominatim.openstreetmap.org/search
115
+ headers:
116
+ User-Agent: ${params.user_agent}
117
+ query:
118
+ q: ${steps.facilities.output.value.1.query}
119
+ format: jsonv2
120
+ limit: 1
121
+ countrycodes: ${params.country_codes}
122
+
123
+ pause_2:
124
+ block: time.sleep
125
+ depends_on: [lookup_2]
126
+ config:
127
+ for: 1s
128
+
129
+ lookup_3:
130
+ block: http.request
131
+ depends_on: [facilities, pause_2]
132
+ continue_on_failure: true
133
+ config:
134
+ url: https://nominatim.openstreetmap.org/search
135
+ headers:
136
+ User-Agent: ${params.user_agent}
137
+ query:
138
+ q: ${steps.facilities.output.value.2.query}
139
+ format: jsonv2
140
+ limit: 1
141
+ countrycodes: ${params.country_codes}
142
+
143
+ table:
144
+ block: transform.jq
145
+ depends_on: [facilities, lookup_1, lookup_2, lookup_3]
146
+ config:
147
+ input:
148
+ facilities: ${steps.facilities.output.value}
149
+ answers:
150
+ - ${steps.lookup_1.output.body}
151
+ - ${steps.lookup_2.output.body}
152
+ - ${steps.lookup_3.output.body}
153
+ # lat and lon arrive as strings; tonumber here is what stops every later consumer from
154
+ # guessing. A name with no candidate keeps its row, with nulls where the coordinate
155
+ # would be, so the gap is countable.
156
+ program: |
157
+ [range(.facilities | length) as $i
158
+ | .facilities[$i] as $facility
159
+ | (.answers[$i][0] // null) as $hit
160
+ | {
161
+ name: $facility.name,
162
+ query: $facility.query,
163
+ matched: ($hit != null),
164
+ display_name: ($hit.display_name // null),
165
+ category: ($hit.category // null),
166
+ kind: ($hit.type // null),
167
+ latitude: (if $hit then ($hit.lat | tonumber) else null end),
168
+ longitude: (if $hit then ($hit.lon | tonumber) else null end)
169
+ }]
@@ -0,0 +1,146 @@
1
+ # The same survey story against ODK Central, where the credential lives in a connection.
2
+ #
3
+ # NEEDS AN ODK CENTRAL SERVER AND AN ACCOUNT. Central serves nothing anonymously, so this
4
+ # document names a connection the instance holds rather than carrying anything itself:
5
+ #
6
+ # dg connection create http odk-central \
7
+ # --set base_url=https://central.example.org \
8
+ # --set basic_username=you@example.org \
9
+ # --set basic_password='<your password>'
10
+ #
11
+ # Central accepts HTTP Basic on the /v1 API, which is what makes a plain http connection enough
12
+ # and why this document has no session step: the alternative is POST /v1/sessions to trade the
13
+ # password for a bearer token, which is a second request and a token to keep alive for no gain
14
+ # in a pipeline that runs once an hour.
15
+ #
16
+ # The API: every form exposes an OData feed at
17
+ # /v1/projects/{id}/forms/{xmlFormId}.svc/Submissions, answering {"@odata.context": ...,
18
+ # "value": [...]} -- one object per submission, questions at the top level, groups nested as
19
+ # objects, and Central's own metadata under `__id` and `__system` (`submissionDate`,
20
+ # `submitterId`, `reviewState`). Central will also hand you the whole form as csv at
21
+ # /v1/projects/{id}/forms/{xmlFormId}/submissions.csv, and that is the better answer when a csv
22
+ # is all you want; the feed is what you read when the pipeline has to filter, reshape, or check
23
+ # the data before anything downstream sees it.
24
+ #
25
+ # What happens, hop by hop:
26
+ #
27
+ # submissions one GET through the connection. $top is OData's page size and $skip its
28
+ # offset; Central also honours $filter over __system/submissionDate, which is the
29
+ # natural place to put a run window on a scheduled load.
30
+ # rows the flattening: Central's metadata renamed into stable columns, the named
31
+ # questions read with a path so a grouped question is reachable as "group/field",
32
+ # and every row given the same keys.
33
+ # staged storage.write, the rows as json, because a converter reads a URI rather than a
34
+ # value the run is holding.
35
+ # report csv in storage, named for the form.
36
+ #
37
+ # To make it yours: set project and form_id, list the questions you report on, and add a
38
+ # $filter to the query once the form has more submissions than one page.
39
+ #
40
+ # dg apply examples/open-data/odk-central-submissions.yaml
41
+ # dg run odk-central-submissions -p project=1 -p form_id=household-survey --watch
42
+
43
+ format: dirigent/v1
44
+ kind: pipeline
45
+ code: odk-central-submissions
46
+ name: ODK Central submissions
47
+ description: |
48
+ Submissions for one **ODK Central** form, read from its OData feed through a connection that
49
+ holds the credential, flattened to named columns and written as csv.
50
+
51
+ Central serves nothing anonymously, so the credential is a connection the instance holds --
52
+ the same shape as the Kobo document, with the secret in the right place.
53
+
54
+ tags: [open-data, http, storage, transform, credential, csv]
55
+
56
+ requires:
57
+ blocks:
58
+ - http.request
59
+ - transform.jq
60
+ - storage.write
61
+ - convert.std
62
+ connections:
63
+ - odk-central
64
+
65
+ params:
66
+ type: object
67
+ required: [project, form_id]
68
+ additionalProperties: false
69
+ properties:
70
+ project:
71
+ type: integer
72
+ description: The numeric project id, as it appears in Central's own URLs.
73
+ form_id:
74
+ type: string
75
+ description: The form's xmlFormId, which is the id inside the XLSForm and not its title.
76
+ columns:
77
+ type: array
78
+ default: []
79
+ items:
80
+ type: string
81
+ description: |
82
+ Question paths to include, slash-separated for a grouped question ("group/field");
83
+ empty means the top-level keys of the first submission.
84
+ top:
85
+ type: integer
86
+ default: 250
87
+ description: OData's page size; this document reads one page.
88
+
89
+ steps:
90
+ submissions:
91
+ block: http.request
92
+ config:
93
+ # The base URL is the connection's, so moving from a staging Central to a production one
94
+ # edits the connection and no pipeline.
95
+ connection: odk-central
96
+ path: /v1/projects/${params.project}/forms/${params.form_id}.svc/Submissions
97
+ query:
98
+ # OData's own parameter names, dollar signs included.
99
+ $top: ${params.top}
100
+ # Without this the feed answers only the current version's fields, which silently
101
+ # drops answers collected under an earlier form version.
102
+ $expand: "*"
103
+ max_response: 64mb
104
+
105
+ rows:
106
+ block: transform.jq
107
+ depends_on: [submissions]
108
+ config:
109
+ input:
110
+ payload: ${steps.submissions.output.body}
111
+ columns: ${params.columns}
112
+ # getpath is what makes a grouped question reachable: "hh/members" is a path into a
113
+ # nested object, and splitting on "/" turns the column name into that path.
114
+ program: |
115
+ . as {$payload, $columns}
116
+ | ($payload.value // []) as $rows
117
+ | (if ($columns | length) > 0
118
+ then $columns
119
+ else ($rows[0] // {} | keys | map(select(startswith("__") | not)))
120
+ end) as $questions
121
+ | [$rows[]
122
+ | . as $row
123
+ | {
124
+ submission_id: .__id,
125
+ submitted_at: .__system.submissionDate,
126
+ submitter_id: .__system.submitterId,
127
+ review_state: .__system.reviewState
128
+ }
129
+ + (reduce $questions[] as $q
130
+ ({}; .[$q] = ($row | getpath($q | split("/")) // null)))]
131
+
132
+ staged:
133
+ block: storage.write
134
+ depends_on: [rows]
135
+ config:
136
+ target: ${run.scratch}/odk/${params.form_id}.json
137
+ value: ${steps.rows.output.value}
138
+
139
+ report:
140
+ block: convert.std
141
+ depends_on: [staged]
142
+ config:
143
+ source: ${steps.staged.output.uri}
144
+ target: ${run.scratch}/odk/${params.form_id}.csv
145
+ from: json
146
+ to: csv