dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,82 @@
1
+ # Running totals, deltas, and a moving average over an ordered series.
2
+ #
3
+ # In: daily rainfall readings for one station, deliberately out of order in the input.
4
+ # Out: the same days sorted, each carrying cumulative, delta and mean_window, plus the
5
+ # final total.
6
+ #
7
+ # reduce and foreach are the same machine with different exhausts. reduce threads an
8
+ # accumulator through a stream and emits the final state once; foreach threads the same
9
+ # accumulator and emits once per element, which is exactly what a running total is. Reaching
10
+ # for reduce and rebuilding the list afterwards is the common way to write this twice.
11
+ #
12
+ # Sorting first is not decoration. A cumulative column over unsorted input is a column of
13
+ # numbers that means nothing, and the input here arrives out of order so the sort has
14
+ # something to do.
15
+ #
16
+ # The moving average is the one part foreach cannot do alone, because it looks backwards
17
+ # more than one step. jq indexes a list, so the window is read by position: for element i,
18
+ # the slice .[i-2:i+1] is the last three days, and a slice with a negative start would wrap
19
+ # from the end -- hence the max with 0. The first two days therefore average one and two
20
+ # days, which is a decision the recipe makes visible rather than emitting null.
21
+ #
22
+ # To change it: -p window=7 widens the moving average to a week.
23
+ #
24
+ # dg run --local examples/recipes/jq-running-totals.yaml
25
+ # dg run --local examples/recipes/jq-running-totals.yaml -p window=7
26
+
27
+ format: dirigent/v1
28
+ kind: pipeline
29
+ code: jq-running-totals
30
+ name: Running totals and moving averages
31
+ description: Sort a daily series, then add cumulative sums, day-on-day deltas, and a moving average over a configurable window.
32
+
33
+ tags: [recipes, transform, jq]
34
+
35
+ requires:
36
+ blocks:
37
+ - value.const
38
+ - transform.jq
39
+
40
+ params:
41
+ type: object
42
+ properties:
43
+ window:
44
+ type: integer
45
+ minimum: 1
46
+ default: 3
47
+ description: How many days the moving average covers, the current day included.
48
+
49
+ steps:
50
+ rows:
51
+ block: value.const
52
+ config:
53
+ value:
54
+ - { date: "2026-01-03", rainfall_mm: 4 }
55
+ - { date: "2026-01-01", rainfall_mm: 0 }
56
+ - { date: "2026-01-05", rainfall_mm: 1 }
57
+ - { date: "2026-01-02", rainfall_mm: 12 }
58
+ - { date: "2026-01-04", rainfall_mm: 7 }
59
+
60
+ series:
61
+ block: transform.jq
62
+ depends_on: [rows]
63
+ config:
64
+ input:
65
+ rows: ${steps.rows.output.value}
66
+ window: ${params.window}
67
+ program: |
68
+ . as {$rows, $window}
69
+ # ISO dates sort as strings, so no date parsing is needed to order the series.
70
+ | ($rows | sort_by(.date)) as $ordered
71
+ | ($ordered | map(.rainfall_mm)) as $values
72
+ | {days: [foreach range($ordered | length) as $i (
73
+ {total: 0, previous: null};
74
+ {total: (.total + $values[$i]),
75
+ previous: (if $i == 0 then null else $values[$i - 1] end)};
76
+ $ordered[$i]
77
+ + {cumulative: .total,
78
+ delta: (if .previous == null then null
79
+ else $values[$i] - .previous end),
80
+ mean_window: (($values[([$i - $window + 1, 0] | max):$i + 1])
81
+ | add / length)})],
82
+ total: ($values | add)}
@@ -0,0 +1,88 @@
1
+ # Clean the strings a spreadsheet exported: whitespace, case, punctuation, and codes.
2
+ #
3
+ # In: rows whose name, code and email arrived with padding, mixed case, doubled inner
4
+ # spaces, and a code carrying a suffix nobody wanted.
5
+ # Out: the same rows cleaned, plus a `changed` list naming the fields each row's cleaning
6
+ # actually touched.
7
+ #
8
+ # Every verb here is one jq builtin doing one thing, and the order matters: trim before
9
+ # comparing, collapse inner whitespace before splitting, and lowercase an email but never a
10
+ # name. A single gsub doing all of it at once is the version nobody can debug six months
11
+ # later.
12
+ #
13
+ # sub("^\\s+"; "") / sub("\\s+$"; "") trim, anchored at each end. jq has no trim verb,
14
+ # and ltrimstr only removes a literal prefix.
15
+ # gsub("\\s+"; " ") collapse a run of inner whitespace to one space.
16
+ # ascii_downcase case-fold. It is ASCII-only by name and by
17
+ # behaviour, so it leaves accented letters alone
18
+ # rather than mangling them.
19
+ # capture("(?<code>[A-Z]+-[0-9]+)") pull a code out of a longer string by naming the
20
+ # part wanted; the named group becomes a field.
21
+ # test(...) ask a yes-or-no question about a string without
22
+ # changing it.
23
+ #
24
+ # The `changed` list is what makes cleaning reviewable: it is computed by comparing each
25
+ # field before and after, so a run says which rows it had to touch rather than leaving that
26
+ # to a diff nobody takes.
27
+ #
28
+ # To change it: -p title_case=true also capitalises each word of a name, which is the kind
29
+ # of rule that is right for a report and wrong for a key.
30
+ #
31
+ # dg run --local examples/recipes/jq-string-cleaning.yaml
32
+ # dg run --local examples/recipes/jq-string-cleaning.yaml -p title_case=true
33
+
34
+ format: dirigent/v1
35
+ kind: pipeline
36
+ code: jq-string-cleaning
37
+ name: Clean strings with jq
38
+ description: Trim, collapse, case-fold and extract the strings a spreadsheet export brings in, and report which fields changed.
39
+
40
+ tags: [recipes, transform, jq]
41
+
42
+ requires:
43
+ blocks:
44
+ - value.const
45
+ - transform.jq
46
+
47
+ params:
48
+ type: object
49
+ properties:
50
+ title_case:
51
+ type: boolean
52
+ default: false
53
+ description: Capitalise each word of a cleaned name.
54
+
55
+ steps:
56
+ rows:
57
+ block: value.const
58
+ config:
59
+ value:
60
+ - { name: " harbour station ", code: "ST-1 (retired)", email: "Ops@Example.ORG " }
61
+ - { name: "RIDGE station", code: "site ST-2", email: "ridge@example.org" }
62
+ - { name: "delta station", code: "ST-3", email: " DELTA@EXAMPLE.ORG" }
63
+
64
+ cleaned:
65
+ block: transform.jq
66
+ depends_on: [rows]
67
+ config:
68
+ input:
69
+ rows: ${steps.rows.output.value}
70
+ title_case: ${params.title_case}
71
+ program: |
72
+ def trim: sub("^\\s+"; "") | sub("\\s+$"; "");
73
+ def squeeze: gsub("\\s+"; " ");
74
+ def titled: split(" ") | map(if length > 0
75
+ then (.[0:1] | ascii_upcase) + (.[1:] | ascii_downcase)
76
+ else . end) | join(" ");
77
+ . as {$rows, $title_case}
78
+ | [$rows[]
79
+ | . as $before
80
+ | {name: (.name | trim | squeeze | if $title_case then titled else . end),
81
+ # The code is whatever looks like a code inside the cell; a row with none
82
+ # keeps a null rather than the raw cell, so the failure is visible downstream.
83
+ code: (.code | trim | (capture("(?<code>[A-Z]+-[0-9]+)").code // null)),
84
+ email: (.email | trim | ascii_downcase),
85
+ looks_like_email: (.email | trim | test("^[^@\\s]+@[^@\\s]+\\.[a-zA-Z]{2,}$"))}
86
+ | . as $after
87
+ | $after + {changed: [$before | to_entries[]
88
+ | select($after[.key] != .value) | .key]}]
@@ -0,0 +1,85 @@
1
+ # The top N rows by a measure, with ties and the long tail both accounted for.
2
+ #
3
+ # In: a list of stations with an uptime percentage and a reading count.
4
+ # Out: {top, tail_summary, ties} -- the leaders, one row summarising everything below them,
5
+ # and the rows that tie with the last leader.
6
+ #
7
+ # jq has no "top n" builtin, and does not need one: sort_by ascends, reverse descends, and a
8
+ # slice takes the head. Writing it as sort_by(-.x) instead only works for numbers, and fails
9
+ # quietly the day the measure becomes a string, so the reverse spelling is the one to learn.
10
+ #
11
+ # Two things a naive top-n gets wrong, and both are in the output:
12
+ #
13
+ # Ties. A cut at position N splits a tie arbitrarily. The ties list re-reads the input
14
+ # for every row whose measure equals the last leader's, so a run says when the
15
+ # cut was arbitrary instead of pretending it was not.
16
+ # The tail. Dropping the rest hides how much was dropped. One summary row -- how many, and
17
+ # what they add up to -- keeps the total honest and costs one more slice.
18
+ #
19
+ # The slice is bounded by the list rather than by N: .[0:$n] on a list shorter than N is the
20
+ # whole list rather than an error, which is jq being forgiving in the one place a report
21
+ # wants it to be.
22
+ #
23
+ # To change it: -p n=1 narrows it, -p measure=readings ranks by the other column.
24
+ #
25
+ # dg run --local examples/recipes/jq-top-n.yaml
26
+ # dg run --local examples/recipes/jq-top-n.yaml -p n=1 -p measure=readings
27
+
28
+ format: dirigent/v1
29
+ kind: pipeline
30
+ code: jq-top-n
31
+ name: Top N with ties and a tail
32
+ description: Rank rows by a measure and keep the leaders, the rows tied with the last of them, and a summary of everything below.
33
+
34
+ tags: [recipes, transform, jq]
35
+
36
+ requires:
37
+ blocks:
38
+ - value.const
39
+ - transform.jq
40
+
41
+ params:
42
+ type: object
43
+ properties:
44
+ n:
45
+ type: integer
46
+ minimum: 1
47
+ default: 3
48
+ description: How many leaders to keep.
49
+ measure:
50
+ type: string
51
+ default: uptime
52
+ description: The column the ranking reads.
53
+
54
+ steps:
55
+ rows:
56
+ block: value.const
57
+ config:
58
+ value:
59
+ - { station: st-1, uptime: 99.4, readings: 1440 }
60
+ - { station: st-2, uptime: 87.0, readings: 1200 }
61
+ - { station: st-3, uptime: 99.9, readings: 1439 }
62
+ # A tie with st-2 on uptime, which is what the ties list exists to report.
63
+ - { station: st-4, uptime: 87.0, readings: 300 }
64
+ - { station: st-5, uptime: 62.5, readings: 900 }
65
+
66
+ ranked:
67
+ block: transform.jq
68
+ depends_on: [rows]
69
+ config:
70
+ input:
71
+ rows: ${steps.rows.output.value}
72
+ n: ${params.n}
73
+ measure: ${params.measure}
74
+ program: |
75
+ . as {$rows, $n, $measure}
76
+ | ($rows | sort_by(.[$measure]) | reverse) as $ordered
77
+ | ($ordered | .[0:$n]) as $top
78
+ | ($ordered | .[$n:]) as $rest
79
+ | {top: $top,
80
+ tail_summary: {rows: ($rest | length),
81
+ readings: ($rest | map(.readings) | add // 0)},
82
+ ties: (if ($top | length) == 0 then []
83
+ else ($top[-1][$measure]) as $cut
84
+ | [$ordered[] | select(.[$measure] == $cut) | .station]
85
+ end)}
@@ -0,0 +1,124 @@
1
+ # Two ways to check data, in one document, so the difference is visible rather than argued.
2
+ #
3
+ # In: a batch of rows, four good and two bad -- one missing a field, one with a level that
4
+ # is a string.
5
+ # Out: {accepted, rejected, counts} from the jq check, and the accepted rows again from a
6
+ # validate.schema gate that could not have passed the bad ones.
7
+ #
8
+ # The two checks answer different questions, and a pipeline usually wants both:
9
+ #
10
+ # jq sorts a batch. It looks at each row, keeps the ones it likes, and reports
11
+ # the rest with a reason -- so a bad row costs one line in a report rather
12
+ # than a failed run. It is a program, so nothing outside this document
13
+ # knows what it enforced.
14
+ # validate.schema is a gate. The shape is declarative, named, and reusable, and a value
15
+ # that does not fit fails the run at the boundary rather than travelling
16
+ # on. Nothing downstream has to trust that a jq program upstream was
17
+ # thorough: the gate's output is the proof, which is why the last step
18
+ # here reads ${steps.gate.output.value} and not the sorter's accepted list.
19
+ #
20
+ # So: jq decides which rows to keep, the schema decides what a kept row is allowed to be.
21
+ # Running the gate on rows a jq program already sorted is not redundant -- it is the check
22
+ # that the sorter and the schema still agree, and the day they stop the run says so.
23
+ #
24
+ # The rejected rows keep their reason and their index, because "6 rows in, 4 rows out" is
25
+ # not a report anybody can act on.
26
+ #
27
+ # To change it: -p min_level=2 tightens what the sorter accepts. The schema is unchanged,
28
+ # and the gate still passes, because a tighter sorter cannot produce a row the schema
29
+ # refuses -- which is exactly the relationship the two checks should have.
30
+ #
31
+ # The schema is carried in the document, so no server stores this one: it runs with
32
+ # `dg run --local`, and on an instance the same shape is created once with `dg schema create`.
33
+ #
34
+ # dg run --local examples/recipes/jq-validate-in-jq-vs-schema.yaml
35
+ # dg run --local examples/recipes/jq-validate-in-jq-vs-schema.yaml -p min_level=2
36
+
37
+ format: dirigent/v1
38
+ kind: pipeline
39
+ code: jq-validate-in-jq-vs-schema
40
+ name: Checking in jq versus checking with a schema
41
+ description: Sort a batch into accepted and rejected rows with jq, then hold the accepted ones to a carried schema at a gate.
42
+
43
+ tags: [recipes, transform, validate, jq]
44
+
45
+ requires:
46
+ blocks:
47
+ - value.const
48
+ - transform.jq
49
+ - validate.schema
50
+
51
+ schemas:
52
+ recipe-org-unit-list:
53
+ type: array
54
+ items:
55
+ type: object
56
+ required: [id, name, level]
57
+ additionalProperties: false
58
+ properties:
59
+ id: { type: string, minLength: 1 }
60
+ name: { type: string, minLength: 1 }
61
+ level: { type: integer, minimum: 1 }
62
+
63
+ params:
64
+ type: object
65
+ properties:
66
+ min_level:
67
+ type: integer
68
+ minimum: 1
69
+ default: 1
70
+ description: The lowest level the jq sorter accepts.
71
+
72
+ steps:
73
+ batch:
74
+ block: value.const
75
+ config:
76
+ value:
77
+ - { id: ou-1, name: Sierra Leone, level: 1 }
78
+ - { id: ou-2, name: Bo, level: 2 }
79
+ # No level at all.
80
+ - { id: ou-3, name: Kenema }
81
+ - { id: ou-4, name: Bombali, level: 2 }
82
+ # A level that arrived as a string, which is what a csv source always produces.
83
+ - { id: ou-5, name: Kailahun, level: "3" }
84
+ - { id: ou-6, name: Moyamba, level: 3 }
85
+
86
+ sorted:
87
+ block: transform.jq
88
+ depends_on: [batch]
89
+ config:
90
+ input:
91
+ rows: ${steps.batch.output.value}
92
+ min_level: ${params.min_level}
93
+ program: |
94
+ . as {$rows, $min_level}
95
+ | [range($rows | length) as $i
96
+ | $rows[$i]
97
+ | {row: ., index: $i,
98
+ problem: (if (.level | type) != "number" then "level is \(.level | type)"
99
+ elif .level < $min_level then "level below \($min_level)"
100
+ else null end)}] as $checked
101
+ | {accepted: [$checked[] | select(.problem == null) | .row],
102
+ rejected: [$checked[] | select(.problem != null) | {index, id: .row.id, problem}],
103
+ counts: {seen: ($rows | length),
104
+ accepted: [$checked[] | select(.problem == null)] | length,
105
+ rejected: [$checked[] | select(.problem != null)] | length}}
106
+
107
+ gate:
108
+ block: validate.schema
109
+ depends_on: [sorted]
110
+ config:
111
+ # The whole accepted list is checked at once, so one bad row fails the batch rather
112
+ # than half a batch reaching the sink.
113
+ input: ${steps.sorted.output.value.accepted}
114
+ schema: recipe-org-unit-list
115
+
116
+ load:
117
+ block: transform.jq
118
+ depends_on: [gate]
119
+ config:
120
+ # Read from the gate, never from the sorter: this reference is what proves in the
121
+ # document that nothing unchecked got this far.
122
+ input: ${steps.gate.output.value}
123
+ program: |
124
+ {loaded: length, ids: map(.id)}
@@ -0,0 +1,82 @@
1
+ # Date arithmetic in jq: a window of days, ISO weeks, and the strptime trap.
2
+ #
3
+ # In: a start date and a length in days, as parameters.
4
+ # Out: {window: {start, end}, days: [...], weeks: [{iso_week, days, dates}], trap: {...}}.
5
+ #
6
+ # jq has no date type. It has epoch seconds and two conversions on either side of them, and
7
+ # every date calculation here is arithmetic on those seconds:
8
+ #
9
+ # "2026-01-05T00:00:00Z" | fromdateiso8601 -> 1767571200
10
+ # 1767571200 | todate -> "2026-01-05T00:00:00Z"
11
+ #
12
+ # The trap has a name and it is strptime. strptime does not answer with a number: it answers
13
+ # with jq's broken-down time, an array of eight fields, and arithmetic on that array is an
14
+ # error rather than a wrong date. mktime turns it into seconds, and only then does strftime
15
+ # have something to format. The trap object at the end of the output shows both spellings so
16
+ # the difference is on screen: .broken_down is the array, .seconds is the number.
17
+ #
18
+ # A date-only string has no time and no zone, so "2026-01-05" is completed to midnight UTC
19
+ # before it is parsed. Doing that explicitly is the difference between a window that means
20
+ # the same thing on every worker and one that moves with the machine's timezone.
21
+ #
22
+ # ISO weeks come from %G and %V together, never from %Y and %V: the last days of December
23
+ # belong to week 01 of the next ISO year, and %Y-%V would file them under the year that just
24
+ # ended. The window below is chosen to cross a year boundary so that shows up in the output.
25
+ #
26
+ # Adding days as 86400 seconds is exact here because everything is UTC. It is not exact in a
27
+ # zone with daylight saving, where a day is sometimes 23 or 25 hours; a pipeline that needs
28
+ # local days does the arithmetic on dates rather than on seconds.
29
+ #
30
+ # To change it: -p start=2026-03-01 -p days=14 moves and widens the window.
31
+ #
32
+ # dg run --local examples/recipes/jq-window-dates.yaml
33
+ # dg run --local examples/recipes/jq-window-dates.yaml -p start=2026-03-01 -p days=14
34
+
35
+ format: dirigent/v1
36
+ kind: pipeline
37
+ code: jq-window-dates
38
+ name: Date windows and ISO weeks in jq
39
+ description: Build a window of days from a start date, bucket it into ISO weeks, and show why strptime alone is not a date.
40
+
41
+ tags: [recipes, transform, jq]
42
+
43
+ requires:
44
+ blocks:
45
+ - transform.jq
46
+
47
+ params:
48
+ type: object
49
+ properties:
50
+ start:
51
+ type: string
52
+ format: date
53
+ default: "2026-12-28"
54
+ description: The first day of the window, as YYYY-MM-DD.
55
+ days:
56
+ type: integer
57
+ minimum: 1
58
+ default: 10
59
+ description: How many days the window covers, the start day included.
60
+
61
+ steps:
62
+ window:
63
+ block: transform.jq
64
+ config:
65
+ input:
66
+ start: ${params.start}
67
+ days: ${params.days}
68
+ program: |
69
+ . as {$start, $days}
70
+ # The date is completed to midnight UTC before parsing, so the window means the same
71
+ # thing wherever the worker runs.
72
+ | ($start + "T00:00:00Z" | fromdateiso8601) as $from
73
+ | [range($days) | $from + (. * 86400)] as $seconds
74
+ | {window: {start: ($from | todate),
75
+ end: ($seconds[-1] | todate)},
76
+ days: [$seconds[] | strftime("%Y-%m-%d")],
77
+ weeks: ([$seconds[] | {iso_week: strftime("%G-W%V"), date: strftime("%Y-%m-%d")}]
78
+ | group_by(.iso_week)
79
+ | map({iso_week: .[0].iso_week, days: length, dates: map(.date)})),
80
+ trap: {broken_down: ($start | strptime("%Y-%m-%d")),
81
+ seconds: ($start | strptime("%Y-%m-%d") | mktime),
82
+ formatted: ($start | strptime("%Y-%m-%d") | mktime | strftime("%G-W%V"))}}
@@ -0,0 +1,141 @@
1
+ # Nested JSON into a csv somebody opens in a spreadsheet, with the nesting decided on.
2
+ #
3
+ # In: an API-shaped payload: an envelope around a list of records, each with a nested
4
+ # object, a list, and a field the report should not carry.
5
+ # Out: a csv in the run's scratch space, and its text read back for the run's Output tab.
6
+ #
7
+ # A csv row is flat, so this is three decisions rather than a conversion:
8
+ #
9
+ # The envelope. The codec wants the array, not the object wrapping it, so the jq step
10
+ # reaches into .results before anything else happens.
11
+ # The nesting. A nested value has no csv spelling and convert.std refuses it naming the
12
+ # row and the key. Every nested field is therefore pulled up into a column
13
+ # with a name a person reading the header will understand -- site_name, not
14
+ # site.name, because a dot in a header is a thing spreadsheets fight over.
15
+ # The lists. tags is a list, so the report joins it with a separator it chooses. That
16
+ # choice belongs here, in the step that knows the values contain no
17
+ # semicolons, and not in a codec that would have to guess.
18
+ #
19
+ # Column order is the last decision, and it is the reason the projection is written out
20
+ # field by field rather than generated: a report's header is part of its contract with
21
+ # whoever reads it, and a program that emits whatever keys it finds changes that header the
22
+ # day the source adds a field.
23
+ #
24
+ # Keeping that order is why the rows leave jq as text. storage.write serialises a `value`
25
+ # as canonical JSON, which sorts the keys, and the csv header is the key order of the first
26
+ # record; written as `text`, the bytes are the ones jq produced and the header is the one
27
+ # the projection spells.
28
+ #
29
+ # Five hops, and what each one hands on:
30
+ # payload value.const, the envelope a real pipeline would have fetched.
31
+ # rows transform.jq, the flat records, serialised in the step. Output: value, a
32
+ # string of JSON.
33
+ # stored storage.write, that text under a name, said to be JSON. Output: uri,
34
+ # bytes_written.
35
+ # csv convert.std, one URI to another. A conversion is a storage-object operation
36
+ # like storage.copy: it never holds the records as a value, which is why the
37
+ # rows are written before it and read after it.
38
+ # preview storage.read, the csv back as text, so the run's output shows the file
39
+ # without anybody opening it.
40
+ #
41
+ # To change it: -p separator=, changes what joins a list cell -- and is the fastest way to
42
+ # see why the default is a semicolon in a comma-separated file.
43
+ #
44
+ # dg run --local examples/recipes/json-to-csv-flattening.yaml
45
+ # dg run --local examples/recipes/json-to-csv-flattening.yaml -p separator=,
46
+
47
+ format: dirigent/v1
48
+ kind: pipeline
49
+ code: json-to-csv-flattening
50
+ name: Nested JSON to a flat csv
51
+ description: Reach into an envelope, flatten the nested fields into named columns, join the list cells, and convert the written rows into a csv artifact.
52
+
53
+ tags: [recipes, storage, transform, starter]
54
+
55
+ requires:
56
+ blocks:
57
+ - value.const
58
+ - transform.jq
59
+ - storage.write
60
+ - convert.std
61
+ - storage.read
62
+
63
+ params:
64
+ type: object
65
+ properties:
66
+ separator:
67
+ type: string
68
+ default: ";"
69
+ description: What joins the values of a list-valued field inside one cell.
70
+ day:
71
+ type: string
72
+ format: date
73
+ default: "2026-01-01"
74
+ description: The day the csv is named for.
75
+
76
+ steps:
77
+ payload:
78
+ block: value.const
79
+ config:
80
+ value:
81
+ generated_at: "2026-01-01T06:00:00Z"
82
+ page: 1
83
+ results:
84
+ - id: st-1
85
+ site: { name: Harbour, region: east, coords: { lat: 59.91, lon: 10.75 } }
86
+ tags: [coastal, tidal]
87
+ readings: { count: 1440, mean_celsius: 4.5 }
88
+ internal_note: do not publish
89
+ - id: st-2
90
+ site: { name: Ridge, region: west, coords: { lat: 60.39, lon: 5.32 } }
91
+ tags: [alpine]
92
+ readings: { count: 1200, mean_celsius: -1.0 }
93
+ internal_note: recalibrated
94
+
95
+ rows:
96
+ block: transform.jq
97
+ depends_on: [payload]
98
+ config:
99
+ input:
100
+ payload: ${steps.payload.output.value}
101
+ separator: ${params.separator}
102
+ program: |
103
+ . as {$payload, $separator}
104
+ | [$payload.results[]
105
+ | {station: .id,
106
+ site_name: .site.name,
107
+ region: .site.region,
108
+ lat: .site.coords.lat,
109
+ lon: .site.coords.lon,
110
+ tags: (.tags | join($separator)),
111
+ readings: .readings.count,
112
+ mean_celsius: .readings.mean_celsius}]
113
+ | tojson
114
+
115
+ stored:
116
+ block: storage.write
117
+ depends_on: [rows]
118
+ config:
119
+ target: ${run.scratch}/stations-${params.day}.json
120
+ # text rather than value: a value is written as canonical JSON, and canonical JSON
121
+ # sorts the keys the csv header is built from.
122
+ text: ${steps.rows.output.value}
123
+ content_type: application/json
124
+
125
+ csv:
126
+ block: convert.std
127
+ depends_on: [stored]
128
+ config:
129
+ # Where the write said the rows landed, not the same path typed twice.
130
+ source: ${steps.stored.output.uri}
131
+ # The file is the point of a report run, so it goes to storage under a dated name.
132
+ target: ${run.scratch}/stations-${params.day}.csv
133
+ from: json
134
+ to: csv
135
+
136
+ preview:
137
+ block: storage.read
138
+ depends_on: [csv]
139
+ config:
140
+ source: ${steps.csv.output.target}
141
+ max_size: 1mb