dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,130 @@
1
+ # Recipes
2
+
3
+ One question per file. Every recipe here is a small, complete, runnable answer to something
4
+ a person doing data work actually asks -- how do I group and sum, how do I join two lists,
5
+ what survives a parquet round trip, where do credentials live -- and its header comment is
6
+ the lesson: what it does, what goes in and what comes out, why each step is configured the
7
+ way it is, and what to change to make it yours.
8
+
9
+ Nothing here needs infrastructure. The data is written inline, generated by a jq program, or
10
+ fetched from [Postman Echo](https://postman-echo.com), a public service that answers with
11
+ the request it was given. Nothing needs the unsafe-block allowlist: no recipe on this shelf
12
+ runs code on a worker.
13
+
14
+ ```bash
15
+ dg run --local examples/recipes/jq-group-by-and-sum.yaml
16
+ ```
17
+
18
+ One recipe is handed something: [schema-referenced.yaml](schema-referenced.yaml) names a
19
+ schema the instance holds, so a local run is given it, and without it the run is refused
20
+ before anything executes -- which is the point that recipe makes.
21
+
22
+ ```bash
23
+ dg run --local --schema examples/schemas/ou-record.json examples/recipes/schema-referenced.yaml
24
+ ```
25
+
26
+ Four of them fail on purpose, and their headers say so. Failing is the lesson in two of them
27
+ and a parameter away in the others:
28
+
29
+ | Recipe | Ends as | Because |
30
+ | --- | --- | --- |
31
+ | [schema-refuses-then-rule.yaml](schema-refuses-then-rule.yaml) | `failed` | The gate refuses a level of 0, the load is skipped, and the `one_failed` handler runs |
32
+ | [reconcile-two-sources.yaml](reconcile-two-sources.yaml) | `failed` | Three rows differ beyond the tolerance and the limit is two; `-p max_differing=5` passes |
33
+ | [http-success-status-list.yaml](http-success-status-list.yaml) | `succeeded` | 418 is listed as success; `-p status=500` is not, and fails |
34
+ | [csv-header-rules.yaml](csv-header-rules.yaml) | `succeeded` | `-p header=station,station,celsius` provokes the codec's refusal |
35
+
36
+ ## jq: reshaping a value
37
+
38
+ | File | The question it answers |
39
+ | --- | --- |
40
+ | [jq-group-by-and-sum.yaml](jq-group-by-and-sum.yaml) | How do I total a column per group, with a grand total that still adds up? |
41
+ | [jq-join-two-lists.yaml](jq-join-two-lists.yaml) | How do I join two lists on a key, without losing the rows that match nothing? |
42
+ | [jq-pivot-wide-to-long.yaml](jq-pivot-wide-to-long.yaml) | How do I melt a column-per-measure table into one record per measurement? |
43
+ | [jq-long-to-wide.yaml](jq-long-to-wide.yaml) | How do I widen it back, with an explicit fill in every unreported cell? |
44
+ | [jq-dedupe-by-key.yaml](jq-dedupe-by-key.yaml) | Which row wins when a key repeats -- first, last, or newest by timestamp? |
45
+ | [jq-window-dates.yaml](jq-window-dates.yaml) | How do I build a window of days and ISO weeks, and why is `strptime` alone not a date? |
46
+ | [jq-nested-to-flat.yaml](jq-nested-to-flat.yaml) | How do I flatten nested records into the flat ones a csv or parquet writer accepts? |
47
+ | [jq-defaults-and-nulls.yaml](jq-defaults-and-nulls.yaml) | How do I default a field without `//` turning every `false` back into `true`? |
48
+ | [jq-string-cleaning.yaml](jq-string-cleaning.yaml) | How do I trim, case-fold and extract the strings a spreadsheet export brings in? |
49
+ | [jq-top-n.yaml](jq-top-n.yaml) | How do I take the top N without hiding the ties or the tail? |
50
+ | [jq-running-totals.yaml](jq-running-totals.yaml) | How do I add cumulative sums, deltas and a moving average to an ordered series? |
51
+ | [jq-validate-in-jq-vs-schema.yaml](jq-validate-in-jq-vs-schema.yaml) | When do I sort a batch in jq, and when do I gate it with a schema? |
52
+
53
+ ## map and filter: the element-wise verbs
54
+
55
+ | File | The question it answers |
56
+ | --- | --- |
57
+ | [map-enrich-with-lookup.yaml](map-enrich-with-lookup.yaml) | How do I enrich every row from a lookup table a `map.jq` program cannot see? |
58
+ | [filter-by-predicate.yaml](filter-by-predicate.yaml) | How do I write a predicate that answers true or false, since truthiness is not applied? |
59
+ | [filter-then-map-then-reduce.yaml](filter-then-map-then-reduce.yaml) | Which verb is which, in the order a pipeline usually needs them? |
60
+
61
+ ## Converting between formats
62
+
63
+ | File | The question it answers |
64
+ | --- | --- |
65
+ | [csv-to-ndjson.yaml](csv-to-ndjson.yaml) | How do I read a csv, and who gives the string cells their types? |
66
+ | [ndjson-to-parquet.yaml](ndjson-to-parquet.yaml) | How do I write records as parquet, and read the file back to see what landed? |
67
+ | [parquet-round-trip-types.yaml](parquet-round-trip-types.yaml) | Which types survive a parquet round trip, and which does a csv flatten? |
68
+ | [csv-header-rules.yaml](csv-header-rules.yaml) | What are the header rules, and which two headers are refused outright? |
69
+ | [json-to-csv-flattening.yaml](json-to-csv-flattening.yaml) | How do I turn a nested API payload into a csv somebody opens in a spreadsheet? |
70
+
71
+ ## Validating a shape
72
+
73
+ | File | The question it answers |
74
+ | --- | --- |
75
+ | [schema-carried.yaml](schema-carried.yaml) | How do I carry the shape in the document, so the pipeline runs with nothing handed to it? |
76
+ | [schema-referenced.yaml](schema-referenced.yaml) | How do I name a schema the instance holds, and declare that dependency? |
77
+ | [schema-formats.yaml](schema-formats.yaml) | Which `format` keywords actually assert here, and what does a pack contribute? |
78
+ | [schema-refuses-then-rule.yaml](schema-refuses-then-rule.yaml) | What runs after a gate refuses -- and what does not? |
79
+
80
+ ## Storage
81
+
82
+ | File | The question it answers |
83
+ | --- | --- |
84
+ | [storage-write-then-read.yaml](storage-write-then-read.yaml) | How do I park a payload in storage and pick it up again a step later? |
85
+ | [storage-copy-dated-archive.yaml](storage-copy-dated-archive.yaml) | How do I keep a dated copy, and what should the key layout be? |
86
+ | [storage-exists-gate.yaml](storage-exists-gate.yaml) | How do I wait for an object, and what stops me reading a half-written one? |
87
+ | [large-output-to-storage.yaml](large-output-to-storage.yaml) | What happens to a payload past the inline threshold, and what should I do about it? |
88
+ | [storage-manifest-of-a-fan-out.yaml](storage-manifest-of-a-fan-out.yaml) | How do I write one file per item and then say, in the run, what landed? |
89
+ | [report-to-file.yaml](report-to-file.yaml) | How do I render a page, write it to a file, and read it back to see that it landed? |
90
+
91
+ ## HTTP and webhooks
92
+
93
+ | File | The question it answers |
94
+ | --- | --- |
95
+ | [http-get-with-query.yaml](http-get-with-query.yaml) | How do I build a query without hand-writing a URL, and which field holds the answer? |
96
+ | [http-post-json-echo.yaml](http-post-json-echo.yaml) | How do I POST a body assembled upstream, as structure rather than as text? |
97
+ | [http-fetch-validate-post.yaml](http-fetch-validate-post.yaml) | How do I fetch, check what came back, reshape it, check what I made, and post it on? |
98
+ | [http-headers-and-auth-connection.yaml](http-headers-and-auth-connection.yaml) | Where do credentials live, and what belongs in a header map instead? |
99
+ | [http-save-body-to-storage.yaml](http-save-body-to-storage.yaml) | How do I keep a response body as an object somebody else can fetch? |
100
+ | [http-post-file-from-storage.yaml](http-post-file-from-storage.yaml) | How do I POST a file that lives in storage? |
101
+ | [http-success-status-list.yaml](http-success-status-list.yaml) | How do I accept a non-2xx that is a real answer, without silencing failures? |
102
+ | [http-follow-redirects.yaml](http-follow-redirects.yaml) | Why is a redirect not followed by default, and what do I lose when it is? |
103
+ | [http-timeout-override.yaml](http-timeout-override.yaml) | Which of the three deadlines around a call am I actually setting? |
104
+ | [webhook-post-hmac.yaml](webhook-post-hmac.yaml) | What exactly gets signed, with what, and what does a receiver check? |
105
+ | [webhook-post-summary.yaml](webhook-post-summary.yaml) | What belongs in a notification body somebody has to act on? |
106
+ | [report-to-webhook.yaml](report-to-webhook.yaml) | How do I POST a rendered page to a receiver, and what travels beside it? |
107
+
108
+ ## Whole small pipelines
109
+
110
+ | File | The question it answers |
111
+ | --- | --- |
112
+ | [etl-csv-clean-validate-parquet.yaml](etl-csv-clean-validate-parquet.yaml) | What does a whole small ETL look like, from a messy csv to a checked parquet file? |
113
+ | [report-daily-digest.yaml](report-daily-digest.yaml) | How do I compute a day's numbers once and let the csv, the digest and the notification agree? |
114
+ | [reconcile-two-sources.yaml](reconcile-two-sources.yaml) | How do I say exactly how two systems disagree, in four buckets and with a tolerance? |
115
+ | [pagination-by-fan-out.yaml](pagination-by-fan-out.yaml) | How do I fetch several pages at once, and what does `for_each` not let me do? |
116
+
117
+ ## The run's own report
118
+
119
+ | File | The question it answers |
120
+ | --- | --- |
121
+ | [report-built-in.yaml](report-built-in.yaml) | How do I get a page about the run itself, without writing a line of template? |
122
+ | [http-post-report.yaml](http-post-report.yaml) | How do I write that page myself, and which of the run's facts may it read? |
123
+
124
+ ## The pages behind them
125
+
126
+ [docs/jq.md](../../docs/jq.md) teaches the language, [docs/transforms.md](../../docs/transforms.md)
127
+ the four verbs and their frames, [docs/json-schema.md](../../docs/json-schema.md) the shapes,
128
+ [docs/reports.md](../../docs/reports.md) every fact a report template may read,
129
+ and [docs/blocks.md](../../docs/blocks.md) is generated from the live catalog, so it is the
130
+ honest answer to what a block's config actually takes.
@@ -0,0 +1,147 @@
1
+ # The four rules the csv codec keeps about headers, and the two headers it refuses.
2
+ #
3
+ # In: a csv whose rows are ragged, and a list of records whose keys are not all the same.
4
+ # Out: {read, written, header} -- the ragged csv read into records, and records written back
5
+ # out as a csv whose header covers every key any of them has.
6
+ #
7
+ # A convert step reads one URI and writes another, so each half below is three hops: the
8
+ # document is put in storage, converted, and read back out. The two conversions are the
9
+ # middle hop of each half, and the csv rules are what they keep.
10
+ #
11
+ # Reading, two rules:
12
+ #
13
+ # The header names the columns, and every cell is a string. A row with fewer cells than
14
+ # the header names gets an empty string in the missing ones rather than a missing key, so
15
+ # every record has the same shape.
16
+ # A row with MORE cells than the header names is refused, naming the row: cells with no
17
+ # column are data nobody can address, and dropping them silently is worse than stopping.
18
+ #
19
+ # Writing, two rules:
20
+ #
21
+ # The header is the union of the keys the rows have, in first-seen order across the
22
+ # records. A key only the third record carries is still a column, and the records without
23
+ # it get an empty cell -- a header taken from the first record alone would silently drop
24
+ # data. Within one record the order is whatever the value carries by the time it reaches
25
+ # the codec, which is not always the order it was typed in; a report whose header is part
26
+ # of its contract builds its objects field by field in a jq step, which keeps that order.
27
+ # A nested value has no csv spelling, so it is refused naming the row and the key. Flatten
28
+ # first: jq-nested-to-flat.yaml is that step.
29
+ #
30
+ # Two headers are refused outright when reading, and both refusals are worth provoking once:
31
+ #
32
+ # station,station,celsius -> the csv header repeats a column name: 'station' at columns
33
+ # 1, 2; a record key names one column, so rename them before
34
+ # converting
35
+ # station,,celsius -> the csv header has no name at column 2, and a record key
36
+ # names its column; name it before converting
37
+ #
38
+ # Both are refusals rather than repairs because a repair has to invent a name -- station_2,
39
+ # column_2 -- and every reader downstream then depends on an invention this codec made up.
40
+ #
41
+ # To see either one, run with -p header='station,station,celsius'; that run fails on purpose
42
+ # with the message above.
43
+ #
44
+ # dg run --local examples/recipes/csv-header-rules.yaml
45
+ # dg run --local examples/recipes/csv-header-rules.yaml -p header=station,station,celsius
46
+ # dg run --local examples/recipes/csv-header-rules.yaml -p header=station,,celsius
47
+
48
+ format: dirigent/v1
49
+ kind: pipeline
50
+ code: csv-header-rules
51
+ name: CSV header rules
52
+ description: How the csv codec reads a ragged file and writes a union header, and the two headers it refuses to read at all.
53
+
54
+ tags: [recipes, storage, transform]
55
+
56
+ requires:
57
+ blocks:
58
+ - value.const
59
+ - transform.jq
60
+ - storage.write
61
+ - convert.std
62
+ - storage.read
63
+
64
+ params:
65
+ type: object
66
+ properties:
67
+ header:
68
+ type: string
69
+ default: station,date,celsius
70
+ description: The header row the reading half is given; a repeated or empty name is refused.
71
+
72
+ steps:
73
+ ragged:
74
+ block: transform.jq
75
+ config:
76
+ input: ${params.header}
77
+ # The rows are short on purpose: st-2 has no celsius cell at all.
78
+ program: |
79
+ . + "\nst-1,2026-01-01,4.5\nst-2,2026-01-01\nst-3,2026-01-01,7\n"
80
+
81
+ ragged_file:
82
+ block: storage.write
83
+ depends_on: [ragged]
84
+ config:
85
+ target: ${run.scratch}/ragged.csv
86
+ text: ${steps.ragged.output.value}
87
+ content_type: text/csv
88
+
89
+ read:
90
+ block: convert.std
91
+ depends_on: [ragged_file]
92
+ config:
93
+ source: ${steps.ragged_file.output.uri}
94
+ target: ${run.scratch}/ragged.json
95
+ from: csv
96
+ to: json
97
+
98
+ records:
99
+ block: storage.read
100
+ depends_on: [read]
101
+ config:
102
+ # The converted object is application/json, so it comes back as a value.
103
+ source: ${steps.read.output.target}
104
+
105
+ uneven:
106
+ block: value.const
107
+ config:
108
+ value:
109
+ - { station: st-1, celsius: 4.5 }
110
+ # A key the first record does not have: it is still a column.
111
+ - { station: st-2, celsius: -1, battery: 96 }
112
+ - { station: st-3, note: replaced sensor }
113
+
114
+ uneven_file:
115
+ block: storage.write
116
+ depends_on: [uneven]
117
+ config:
118
+ target: ${run.scratch}/uneven.json
119
+ # Canonical JSON: the keys reach the codec sorted, whatever order they were typed in.
120
+ value: ${steps.uneven.output.value}
121
+
122
+ write:
123
+ block: convert.std
124
+ depends_on: [uneven_file]
125
+ config:
126
+ source: ${steps.uneven_file.output.uri}
127
+ target: ${run.scratch}/uneven.csv
128
+ from: json
129
+ to: csv
130
+
131
+ written:
132
+ block: storage.read
133
+ depends_on: [write]
134
+ config:
135
+ source: ${steps.write.output.target}
136
+
137
+ report:
138
+ block: transform.jq
139
+ depends_on: [records, written]
140
+ config:
141
+ input:
142
+ read: ${steps.records.output.value}
143
+ written: ${steps.written.output.text}
144
+ program: |
145
+ {read: .read,
146
+ written: .written,
147
+ header: (.written | split("\n")[0] | split(","))}
@@ -0,0 +1,107 @@
1
+ # A csv turned into ndjson, and the one thing the conversion will not do for you: types.
2
+ #
3
+ # In: an inline csv with a header row -- station, date, celsius, active.
4
+ # Out: ndjson in the run's scratch space, one JSON object per line, and beside it the same
5
+ # records with celsius as a number and active as a boolean.
6
+ #
7
+ # All three text formats convert.std knows say the same thing -- a sequence of records --
8
+ # so a conversion is reading that sequence out of one spelling and writing it in another.
9
+ # csv to ndjson is the one a line-oriented sink wants, because a consumer can read one line
10
+ # at a time instead of holding the whole array.
11
+ #
12
+ # A convert step is a storage operation, the way storage.copy is: it reads one URI and
13
+ # writes another, and no record passes through the step itself. So the csv is put at a URI
14
+ # first and the ndjson is read back from one afterwards, which is the three hops below.
15
+ #
16
+ # source storage.write, the inline csv at a URI. A real pipeline writes the body an
17
+ # http.request answered with here instead, and nothing else in the document
18
+ # moves.
19
+ # as_ndjson convert.std csv -> ndjson. Output: source, target, bytes_written.
20
+ # lines storage.read of the target. Nothing recognises the .ndjson extension, so the
21
+ # step says what the object is; application/x-ndjson is text rather than the
22
+ # JSON family, so it comes back as `text`.
23
+ #
24
+ # Every cell read out of a csv is a string. "4.5" comes back as the three characters, not as
25
+ # a number, because a csv carries no types and a codec that guessed would be wrong on the
26
+ # column where 007 is a code and 1-2 is a range. So the typing step is a jq program, in the
27
+ # document that knows what the columns mean, and it is the second half of every csv read.
28
+ #
29
+ # ndjson is one JSON value per line, which is why the typing program splits on newlines,
30
+ # drops the empty last one, and fromjson each.
31
+ #
32
+ # To change it: -p empty_is_null=true reads an empty cell as null rather than as the empty
33
+ # string, which is the right rule for a numeric column and the wrong one for a text column.
34
+ #
35
+ # dg run --local examples/recipes/csv-to-ndjson.yaml
36
+ # dg run --local examples/recipes/csv-to-ndjson.yaml -p empty_is_null=true
37
+
38
+ format: dirigent/v1
39
+ kind: pipeline
40
+ code: csv-to-ndjson
41
+ name: CSV to ndjson, with the typing that follows
42
+ description: Re-spell a csv as one JSON object per line, then give the string cells the types the columns actually have.
43
+
44
+ tags: [recipes, storage, transform]
45
+
46
+ requires:
47
+ blocks:
48
+ - storage.write
49
+ - convert.std
50
+ - storage.read
51
+ - transform.jq
52
+
53
+ params:
54
+ type: object
55
+ properties:
56
+ empty_is_null:
57
+ type: boolean
58
+ default: false
59
+ description: Read an empty cell as null rather than as an empty string.
60
+
61
+ steps:
62
+ source:
63
+ block: storage.write
64
+ config:
65
+ target: ${run.scratch}/readings.csv
66
+ text: |
67
+ station,date,celsius,active
68
+ st-1,2026-01-01,4.5,true
69
+ st-2,2026-01-01,-1,false
70
+ st-3,2026-01-01,,true
71
+ content_type: text/csv
72
+
73
+ as_ndjson:
74
+ block: convert.std
75
+ depends_on: [source]
76
+ config:
77
+ source: ${steps.source.output.uri}
78
+ target: ${run.scratch}/readings.ndjson
79
+ from: csv
80
+ to: ndjson
81
+
82
+ lines:
83
+ block: storage.read
84
+ depends_on: [as_ndjson]
85
+ config:
86
+ source: ${steps.as_ndjson.output.target}
87
+ content_type: application/x-ndjson
88
+
89
+ typed:
90
+ block: transform.jq
91
+ depends_on: [lines]
92
+ config:
93
+ input:
94
+ lines: ${steps.lines.output.text}
95
+ empty_is_null: ${params.empty_is_null}
96
+ program: |
97
+ . as {$lines, $empty_is_null}
98
+ | [$lines
99
+ | split("\n")[]
100
+ | select(length > 0)
101
+ | fromjson
102
+ | {station,
103
+ date,
104
+ celsius: (if .celsius == ""
105
+ then (if $empty_is_null then null else 0 end)
106
+ else (.celsius | tonumber) end),
107
+ active: (.active == "true")}]
@@ -0,0 +1,207 @@
1
+ # A whole small ETL: a messy csv in, a checked parquet file out.
2
+ #
3
+ # In: an inline csv with padded strings, mixed case, a typed-as-text numeric column, an
4
+ # empty cell, and a duplicate row.
5
+ # Out: a parquet file in the run's scratch space, written only from rows that passed a
6
+ # schema gate, plus a report of what was dropped and why.
7
+ #
8
+ # The order of the nine steps is the recipe, and every one of them is a different kind of
9
+ # work that wants its own step:
10
+ #
11
+ # source storage.write, the inline csv at a URI. A conversion reads one URI and writes
12
+ # another, the way storage.copy does, so the csv it reads has to be an object
13
+ # first. A real pipeline writes a downloaded body here instead.
14
+ # parse convert.std csv -> json. Nothing passes through the step: bytes at one URI,
15
+ # bytes at another.
16
+ # rows storage.read of the parsed object, which is where the records enter the run.
17
+ # Every cell arrives as a string, because a csv has no types and a codec that
18
+ # guessed would be wrong on the column where 007 is a code.
19
+ # clean the string work: trim, case-fold, and turn the text columns that are really
20
+ # numbers into numbers. This is where a csv's missing type information is
21
+ # supplied, by the document that knows what the columns mean.
22
+ # sort the rows that cannot be loaded are separated from the ones that can, with a
23
+ # reason kept per row. Dropping them silently is what makes a load report "412
24
+ # rows" when 460 arrived.
25
+ # gate validate.schema over the accepted rows. The clean step and the schema are two
26
+ # statements of the same expectation, and the gate is what makes them agree:
27
+ # anything the cleaner let through that the schema refuses fails the run here,
28
+ # at the boundary, rather than inside a parquet file somebody reads next month.
29
+ # checked storage.write of what the gate passed, because the writer below takes a URI
30
+ # and not a value. Nothing but the gate's own output is written here.
31
+ # write convert.arrow json -> parquet. Types survive here, which is the reason for
32
+ # doing all of the above before writing rather than after.
33
+ # report the run's own account of itself: counts, the rejected keys, and the file.
34
+ #
35
+ # Everything downstream of the gate reads ${steps.gate.output.value}, which is the input
36
+ # unchanged. That reference is the document's proof that no unchecked row reached the file.
37
+ #
38
+ # To change it: -p min_celsius=-50 loosens the range the schema allows; -p strict=true makes
39
+ # the sorter refuse a duplicate key instead of keeping the first, which fails the run rather
40
+ # than quietly preferring one row over another.
41
+ #
42
+ # The schema is carried in the document, so no server stores this one: it runs with
43
+ # `dg run --local`, and on an instance the same shape is created once with `dg schema create`.
44
+ #
45
+ # dg run --local examples/recipes/etl-csv-clean-validate-parquet.yaml
46
+ # dg run --local examples/recipes/etl-csv-clean-validate-parquet.yaml -p strict=true # fails, on purpose
47
+
48
+ format: dirigent/v1
49
+ kind: pipeline
50
+ code: etl-csv-clean-validate-parquet
51
+ name: CSV to checked parquet
52
+ description: Parse a messy csv, clean and type it, sort out the unloadable rows, hold the rest to a schema, and write parquet.
53
+
54
+ tags: [recipes, storage, transform, validate, parquet]
55
+
56
+ requires:
57
+ blocks:
58
+ - storage.write
59
+ - storage.read
60
+ - convert.std
61
+ - convert.arrow
62
+ - transform.jq
63
+ - validate.schema
64
+
65
+ schemas:
66
+ recipe-reading-rows:
67
+ type: array
68
+ minItems: 1
69
+ items:
70
+ type: object
71
+ required: [station, day, celsius, active]
72
+ additionalProperties: false
73
+ properties:
74
+ station: { type: string, pattern: "^st-[0-9]+$" }
75
+ day: { type: string, format: date }
76
+ celsius: { type: number, minimum: -90, maximum: 60 }
77
+ active: { type: boolean }
78
+
79
+ params:
80
+ type: object
81
+ properties:
82
+ day:
83
+ type: string
84
+ format: date
85
+ default: "2026-01-01"
86
+ description: The day the file is named for.
87
+ strict:
88
+ type: boolean
89
+ default: false
90
+ description: Fail on a duplicate station instead of keeping the first row for it.
91
+
92
+ steps:
93
+ source:
94
+ block: storage.write
95
+ config:
96
+ target: ${run.scratch}/readings-${params.day}.csv
97
+ # Inline so the recipe needs no network. In a real pipeline the body an http.request
98
+ # answered with is written here, and nothing below moves.
99
+ text: |
100
+ station,day,celsius,active
101
+ ST-1 ,2026-01-01,4.5,TRUE
102
+ st-2,2026-01-01,-1,false
103
+ st-3,2026-01-01,,true
104
+ st-1,2026-01-01,4.5,true
105
+ st-4,2026-01-01,not-a-number,true
106
+ content_type: text/csv
107
+
108
+ parse:
109
+ block: convert.std
110
+ depends_on: [source]
111
+ config:
112
+ source: ${steps.source.output.uri}
113
+ target: ${run.scratch}/readings-${params.day}.json
114
+ from: csv
115
+ to: json
116
+
117
+ rows:
118
+ block: storage.read
119
+ depends_on: [parse]
120
+ config:
121
+ source: ${steps.parse.output.target}
122
+ max_size: 8mb
123
+
124
+ clean:
125
+ block: transform.jq
126
+ depends_on: [rows]
127
+ config:
128
+ input:
129
+ rows: ${steps.rows.output.value}
130
+ day: ${params.day}
131
+ program: |
132
+ def trim: sub("^\\s+"; "") | sub("\\s+$"; "");
133
+ . as {$rows, $day}
134
+ | [$rows[]
135
+ | {station: (.station | trim | ascii_downcase),
136
+ day: (if .day == "" then $day else .day end),
137
+ # An empty cell is not a measurement of zero, so it stays null and is sorted
138
+ # out below rather than invented here.
139
+ celsius: (if .celsius == "" then null
140
+ else (try (.celsius | tonumber) catch null) end),
141
+ # A csv boolean is a word, and which word depends on who exported it.
142
+ active: ((.active | ascii_downcase) == "true")}]
143
+
144
+ sort:
145
+ block: transform.jq
146
+ depends_on: [clean]
147
+ config:
148
+ input:
149
+ rows: ${steps.clean.output.value}
150
+ strict: ${params.strict}
151
+ program: |
152
+ . as {$rows, $strict}
153
+ | [$rows | group_by(.station)[] | {station: .[0].station, count: length}]
154
+ as $by_station
155
+ | ([$by_station[] | select(.count > 1) | .station]) as $duplicated
156
+ | if $strict and ($duplicated | length) > 0
157
+ then error("duplicate stations: \($duplicated | join(", "))")
158
+ else . end
159
+ | {accepted: [$rows
160
+ | group_by(.station)[]
161
+ # First row wins for a duplicated key, which is a decision and not an accident.
162
+ | .[0]
163
+ | select(.celsius != null)],
164
+ rejected: [$rows[]
165
+ | select(.celsius == null)
166
+ | {station, reason: "celsius is absent or not a number"}],
167
+ duplicated: $duplicated}
168
+
169
+ gate:
170
+ block: validate.schema
171
+ depends_on: [sort]
172
+ config:
173
+ input: ${steps.sort.output.value.accepted}
174
+ schema: recipe-reading-rows
175
+
176
+ checked:
177
+ block: storage.write
178
+ depends_on: [gate]
179
+ config:
180
+ target: ${run.scratch}/checked-${params.day}.json
181
+ # From the gate, never from the sorter.
182
+ value: ${steps.gate.output.value}
183
+
184
+ write:
185
+ block: convert.arrow
186
+ depends_on: [checked]
187
+ config:
188
+ source: ${steps.checked.output.uri}
189
+ target: ${run.scratch}/readings-${params.day}.parquet
190
+ from: json
191
+ to: parquet
192
+
193
+ report:
194
+ block: transform.jq
195
+ depends_on: [sort, gate, write]
196
+ config:
197
+ input:
198
+ sorted: ${steps.sort.output.value}
199
+ checked: ${steps.gate.output.value}
200
+ uri: ${steps.write.output.target}
201
+ bytes: ${steps.write.output.bytes_written}
202
+ program: |
203
+ {file: .uri, file_bytes: .bytes,
204
+ written: (.checked | length),
205
+ rejected: (.sorted.rejected | length),
206
+ rejected_rows: .sorted.rejected,
207
+ duplicated_stations: .sorted.duplicated}