dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,104 @@
1
+ # Three deadlines sit around one HTTP call. Know which one you are setting.
2
+ #
3
+ # In: a delay in seconds, as a parameter, defaulting to 2.
4
+ # Out: a slow call that completes inside its own timeout, and its measured duration.
5
+ #
6
+ # The three, from the inside out:
7
+ #
8
+ # timeout on the step's config how long this one call may take before the
9
+ # client gives up. It overrides the connection's
10
+ # timeout for this call and nothing else.
11
+ # timeout on the connection the default for every call made through it,
12
+ # which is where a slow API's whole profile
13
+ # belongs rather than in each step.
14
+ # timeout on the step the engine's deadline on the attempt, which
15
+ # covers the block's entire execution, retries of
16
+ # an inner operation included. It is step-level
17
+ # engine semantics, uniform across every block,
18
+ # and it is what kills a step that hangs somewhere
19
+ # the HTTP client is not.
20
+ #
21
+ # The inner one is the one to reach for when a single endpoint is slow: a report that takes
22
+ # forty seconds to build does not justify raising the deadline on every call to that host,
23
+ # and a health check that should answer in one second is a step-level override in the other
24
+ # direction.
25
+ #
26
+ # https://postman-echo.com/delay/<seconds> answers after the seconds it is given, so the
27
+ # relationship is visible: with the default the call takes about 2s against a 10s budget and
28
+ # the run is green. Setting -p delay=6 against -p timeout=3 fails the step instead, as a
29
+ # timeout rather than as an error the server returned, which is a different failure class
30
+ # and retried differently.
31
+ #
32
+ # The engine's own step timeout is set here too, comfortably above the call's, so that a
33
+ # hang outside the HTTP client -- a name that never resolves, a socket that never closes --
34
+ # still ends the attempt.
35
+ #
36
+ # To change it: -p delay=6 -p timeout=3 to see the failure, or -p delay=1 for a fast run.
37
+ #
38
+ # dg run --local examples/recipes/http-timeout-override.yaml
39
+ # dg run --local examples/recipes/http-timeout-override.yaml -p delay=6 -p timeout=3 # fails, on purpose
40
+
41
+ format: dirigent/v1
42
+ kind: pipeline
43
+ code: http-timeout-override
44
+ name: A per-call timeout override
45
+ description: Override a connection's timeout for one slow call, under an engine step deadline that covers the whole attempt.
46
+
47
+ tags: [recipes, http, transform]
48
+
49
+ requires:
50
+ blocks:
51
+ - http.request
52
+ - transform.jq
53
+
54
+ connections:
55
+ echo-slow:
56
+ kind: http
57
+ config:
58
+ base_url: https://postman-echo.com
59
+ # The default for every call through this connection. Deliberately short, so the
60
+ # override below is doing real work.
61
+ timeout: 3s
62
+ health_path: /get
63
+
64
+ params:
65
+ type: object
66
+ properties:
67
+ delay:
68
+ type: integer
69
+ minimum: 0
70
+ maximum: 10
71
+ default: 2
72
+ description: How many seconds the endpoint waits before answering.
73
+ timeout:
74
+ type: number
75
+ default: 10
76
+ description: This call's own deadline, overriding the connection's 3 seconds.
77
+
78
+ steps:
79
+ slow:
80
+ block: http.request
81
+ # The engine's deadline on the whole attempt, well above the call's own, so a hang
82
+ # outside the HTTP client still ends the step.
83
+ timeout: 60s
84
+ config:
85
+ connection: echo-slow
86
+ path: /delay/${params.delay}
87
+ method: GET
88
+ # This call alone; every other call through echo-slow keeps the 3 second default.
89
+ timeout: ${params.timeout}
90
+
91
+ timing:
92
+ block: transform.jq
93
+ depends_on: [slow]
94
+ config:
95
+ input:
96
+ status: ${steps.slow.output.status}
97
+ duration_ms: ${steps.slow.output.duration_ms}
98
+ budget: ${params.timeout}
99
+ program: |
100
+ {status, duration_ms,
101
+ budget_ms: (.budget * 1000),
102
+ # A call that lands close to its budget is a call about to start failing
103
+ # intermittently, which is worth reporting before it does.
104
+ headroom_ms: ((.budget * 1000) - .duration_ms)}
@@ -0,0 +1,78 @@
1
+ # Reduce a list with repeated keys to one row per key, and say which row won.
2
+ #
3
+ # In: readings where the same station reports several times, out of order, some of them
4
+ # corrections of an earlier one.
5
+ # Out: {first_seen, last_seen, latest, duplicates} -- three different answers to "dedupe",
6
+ # side by side, plus the keys that had more than one row.
7
+ #
8
+ # Dedupe is never one operation, because "the duplicate" is not a property of the data; it
9
+ # is a decision about which row to keep:
10
+ #
11
+ # unique_by(.station) keeps the smallest row by the whole key order, which for a
12
+ # sorted-by-nothing input is arbitrary and only safe when the
13
+ # duplicates are genuinely identical.
14
+ # group_by then .[0] / .[-1] keeps the first or last row of each bucket in input order,
15
+ # which is the answer when arrival order carries meaning.
16
+ # group_by then max_by(.at) keeps the newest row by a timestamp the data carries, which
17
+ # is the only one of the three that survives a reorder.
18
+ #
19
+ # The last is almost always the right one for corrections, and this recipe reports the other
20
+ # two beside it so the difference is visible in the output rather than argued about.
21
+ #
22
+ # duplicates is computed before anything is thrown away: a dedupe that reports nothing is a
23
+ # dedupe nobody can audit.
24
+ #
25
+ # To change it: -p key=region dedupes on a different column; the timestamp column stays .at.
26
+ #
27
+ # dg run --local examples/recipes/jq-dedupe-by-key.yaml
28
+ # dg run --local examples/recipes/jq-dedupe-by-key.yaml -p key=region
29
+
30
+ format: dirigent/v1
31
+ kind: pipeline
32
+ code: jq-dedupe-by-key
33
+ name: Dedupe by key, three ways
34
+ description: Collapse repeated rows to one per key by first seen, last seen, and newest timestamp, and list the keys that repeated.
35
+
36
+ tags: [recipes, transform, jq]
37
+
38
+ requires:
39
+ blocks:
40
+ - value.const
41
+ - transform.jq
42
+
43
+ params:
44
+ type: object
45
+ properties:
46
+ key:
47
+ type: string
48
+ default: station
49
+ description: The column a row is deduplicated on.
50
+
51
+ steps:
52
+ rows:
53
+ block: value.const
54
+ config:
55
+ value:
56
+ - { station: st-1, region: east, at: "2026-01-01T06:00:00Z", celsius: 4 }
57
+ - { station: st-2, region: west, at: "2026-01-01T06:00:00Z", celsius: -1 }
58
+ # A correction: same station, later timestamp, arriving before the row it corrects.
59
+ - { station: st-1, region: east, at: "2026-01-01T09:00:00Z", celsius: 5 }
60
+ - { station: st-1, region: east, at: "2026-01-01T07:00:00Z", celsius: 40 }
61
+ - { station: st-3, region: east, at: "2026-01-01T06:00:00Z", celsius: 7 }
62
+
63
+ deduped:
64
+ block: transform.jq
65
+ depends_on: [rows]
66
+ config:
67
+ input:
68
+ key: ${params.key}
69
+ rows: ${steps.rows.output.value}
70
+ program: |
71
+ . as {$key, $rows}
72
+ | ($rows | group_by(.[$key])) as $buckets
73
+ | {first_seen: [$buckets[] | .[0]],
74
+ last_seen: [$buckets[] | .[-1]],
75
+ # max_by reads the timestamp the row carries, so a late arrival cannot win by
76
+ # arriving late and an early correction cannot lose by arriving early.
77
+ latest: [$buckets[] | max_by(.at)],
78
+ duplicates: [$buckets[] | select(length > 1) | {key: .[0][$key], rows: length}]}
@@ -0,0 +1,91 @@
1
+ # Absent, null, false, and empty string are four different things. Handle each on purpose.
2
+ #
3
+ # In: rows where a field is missing entirely, present and null, present and false, and
4
+ # present and empty -- one row for each case.
5
+ # Out: {naive, careful, report} -- the same defaulting written the wrong way and the right
6
+ # way, with a per-row report of which case each field actually was.
7
+ #
8
+ # // is the alternative operator, and it answers with its right side when the left is null
9
+ # OR false. That second half is the trap: .enabled // true can never answer false, so a row
10
+ # that switched a feature off comes back on. The naive object in the output is that bug,
11
+ # left in on purpose so the careful one below has something to be different from.
12
+ #
13
+ # The three questions that are actually being asked are different questions, and jq spells
14
+ # each of them:
15
+ #
16
+ # has("enabled") is the key present at all -- absent and null are distinguished.
17
+ # .enabled == null is the value the null value, whether or not the key is there.
18
+ # (.count // 0) is it null-or-false, which for a number is a fine default.
19
+ #
20
+ # For a boolean, the safe defaulting is `if has("x") and .x != null then .x else DEFAULT end`.
21
+ # For a string, "" is a value and // will not replace it, so an empty-means-absent rule is
22
+ # written out rather than assumed.
23
+ #
24
+ # tonumber on a string that is not a number is an error rather than a null, so it goes
25
+ # through try/catch. That is the one place a reshape may decide, because the alternative --
26
+ # a failed step -- reports a bad cell as an outage.
27
+ #
28
+ # To change it: -p strict_numbers=true makes an unparsable count null instead of zero, which
29
+ # is what a load that must not invent a measurement wants.
30
+ #
31
+ # dg run --local examples/recipes/jq-defaults-and-nulls.yaml
32
+ # dg run --local examples/recipes/jq-defaults-and-nulls.yaml -p strict_numbers=true
33
+
34
+ format: dirigent/v1
35
+ kind: pipeline
36
+ code: jq-defaults-and-nulls
37
+ name: Defaults, nulls, and the // trap
38
+ description: Distinguish absent, null, false, and empty when defaulting fields, and show what the alternative operator gets wrong.
39
+
40
+ tags: [recipes, transform, jq]
41
+
42
+ requires:
43
+ blocks:
44
+ - value.const
45
+ - transform.jq
46
+
47
+ params:
48
+ type: object
49
+ properties:
50
+ strict_numbers:
51
+ type: boolean
52
+ default: false
53
+ description: Leave an unparsable count null rather than defaulting it to zero.
54
+
55
+ steps:
56
+ rows:
57
+ block: value.const
58
+ config:
59
+ value:
60
+ - { id: a, enabled: true, count: "3", label: Harbour }
61
+ # enabled is absent: nothing was said about it.
62
+ - { id: b, count: "0", label: Ridge }
63
+ # enabled is null: something was said, and it said nothing.
64
+ - { id: c, enabled: null, count: null, label: "" }
65
+ # enabled is false: something was said, and it said no. This is the row // ruins.
66
+ - { id: d, enabled: false, count: "not a number", label: Delta }
67
+
68
+ defaults:
69
+ block: transform.jq
70
+ depends_on: [rows]
71
+ config:
72
+ input:
73
+ rows: ${steps.rows.output.value}
74
+ strict: ${params.strict_numbers}
75
+ program: |
76
+ . as {$rows, $strict}
77
+ | {naive: [$rows[] | {id, enabled: (.enabled // true)}],
78
+ careful: [$rows[]
79
+ | {id,
80
+ enabled: (if has("enabled") and .enabled != null then .enabled else true end),
81
+ count: (if .count == null then (if $strict then null else 0 end)
82
+ else (try (.count | tonumber)
83
+ catch (if $strict then null else 0 end)) end),
84
+ # An empty string is a value, so // never sees it; the rule is written out.
85
+ label: (if (.label // "") == "" then "unnamed" else .label end)}],
86
+ report: [$rows[]
87
+ | {id,
88
+ enabled_case: (if has("enabled") | not then "absent"
89
+ elif .enabled == null then "null"
90
+ elif .enabled == false then "false"
91
+ else "true" end)}]}
@@ -0,0 +1,74 @@
1
+ # Total a numeric column per group, and keep the grand total beside the groups.
2
+ #
3
+ # In: a flat list of sales rows -- {region, product, units, revenue}.
4
+ # Out: {as_of, groups: [{region, rows, units, revenue, mean_revenue}], total: {...}}.
5
+ #
6
+ # group_by sorts by the key and buckets, so every bucket is a list and the group's key is
7
+ # read back off any element of it -- .[0].region. Everything after that is ordinary list
8
+ # arithmetic: add over a mapped column, length for the count, and a division that is safe
9
+ # because group_by never produces an empty bucket.
10
+ #
11
+ # The grand total is computed from the rows rather than from the groups, because summing
12
+ # already-rounded group totals is how a report stops adding up.
13
+ #
14
+ # To change it: -p group_by=product buckets the same rows the other way. The program reads
15
+ # the key with .[$key], jq's dynamic field access, which is what lets the column be a
16
+ # parameter instead of a second program.
17
+ #
18
+ # dg run --local examples/recipes/jq-group-by-and-sum.yaml
19
+ # dg run --local examples/recipes/jq-group-by-and-sum.yaml -p group_by=product
20
+
21
+ format: dirigent/v1
22
+ kind: pipeline
23
+ code: jq-group-by-and-sum
24
+ name: Group and sum with jq
25
+ description: Bucket flat rows by one column, total two numeric columns per bucket, and keep a grand total beside them.
26
+
27
+ tags: [recipes, transform, jq]
28
+
29
+ requires:
30
+ blocks:
31
+ - value.const
32
+ - transform.jq
33
+
34
+ params:
35
+ type: object
36
+ properties:
37
+ group_by:
38
+ type: string
39
+ default: region
40
+ description: The column the rows are bucketed by.
41
+
42
+ steps:
43
+ rows:
44
+ block: value.const
45
+ config:
46
+ value:
47
+ - { region: east, product: pump, units: 3, revenue: 300 }
48
+ - { region: east, product: valve, units: 10, revenue: 250 }
49
+ - { region: west, product: pump, units: 1, revenue: 100 }
50
+ - { region: west, product: valve, units: 4, revenue: 100 }
51
+ - { region: north, product: pump, units: 2, revenue: 200 }
52
+
53
+ totals:
54
+ block: transform.jq
55
+ depends_on: [rows]
56
+ config:
57
+ # The key travels as data beside the rows rather than spliced into the program text:
58
+ # a program carrying a ${...} cannot be compiled when the document is applied.
59
+ input:
60
+ key: ${params.group_by}
61
+ rows: ${steps.rows.output.value}
62
+ program: |
63
+ . as {$key, $rows}
64
+ | {as_of: "2026-01-01",
65
+ groups: ($rows
66
+ | group_by(.[$key])
67
+ | map({group: .[0][$key],
68
+ rows: length,
69
+ units: (map(.units) | add),
70
+ revenue: (map(.revenue) | add),
71
+ mean_revenue: ((map(.revenue) | add) / length)})),
72
+ total: {rows: ($rows | length),
73
+ units: ($rows | map(.units) | add),
74
+ revenue: ($rows | map(.revenue) | add)}}
@@ -0,0 +1,77 @@
1
+ # Join two lists on a key, keeping the rows that match nothing.
2
+ #
3
+ # In: a list of stations (the dimension) and a list of readings (the facts).
4
+ # Out: {joined: [...], unmatched: [...]} -- every reading enriched, and the ones whose
5
+ # station is in no dimension row listed separately rather than silently dropped.
6
+ #
7
+ # A jq program has exactly one input, so a join composes its two sides into one object in
8
+ # the step's config and destructures them apart again in the program. INDEX builds the
9
+ # lookup table in one pass -- an object keyed by station id -- which turns the join from a
10
+ # scan per reading into a lookup per reading.
11
+ #
12
+ # Both halves of the result come out of that one index: $by_id[.station] is null for a
13
+ # reading nothing matched, so // supplies the default on the joined side and select collects
14
+ # exactly those rows on the unmatched side. An inner join alone would report four readings
15
+ # where five arrived, and never say which one went missing.
16
+ #
17
+ # To change it: index the other side to change the join's direction -- indexing the readings
18
+ # and iterating the stations answers "which stations reported at all".
19
+ #
20
+ # dg run --local examples/recipes/jq-join-two-lists.yaml
21
+ # dg run --local examples/recipes/jq-join-two-lists.yaml -p default_name=none
22
+
23
+ format: dirigent/v1
24
+ kind: pipeline
25
+ code: jq-join-two-lists
26
+ name: Join two lists with INDEX
27
+ description: Join readings to their station dimension with INDEX, keeping the unmatched rows in a list of their own.
28
+
29
+ tags: [recipes, transform, jq]
30
+
31
+ requires:
32
+ blocks:
33
+ - value.const
34
+ - transform.jq
35
+
36
+ params:
37
+ type: object
38
+ properties:
39
+ default_name:
40
+ type: string
41
+ default: unknown station
42
+ description: The name a reading gets when its station is in no dimension row.
43
+
44
+ steps:
45
+ stations:
46
+ block: value.const
47
+ config:
48
+ value:
49
+ - { id: st-1, name: Harbour, region: east }
50
+ - { id: st-2, name: Ridge, region: west }
51
+ - { id: st-3, name: Delta, region: east }
52
+
53
+ readings:
54
+ block: value.const
55
+ config:
56
+ value:
57
+ - { station: st-1, celsius: 4 }
58
+ - { station: st-2, celsius: -1 }
59
+ - { station: st-3, celsius: 7 }
60
+ # No st-9 in the dimension: this is the row the recipe exists for.
61
+ - { station: st-9, celsius: 2 }
62
+
63
+ join:
64
+ block: transform.jq
65
+ depends_on: [stations, readings]
66
+ config:
67
+ input:
68
+ stations: ${steps.stations.output.value}
69
+ readings: ${steps.readings.output.value}
70
+ fallback: ${params.default_name}
71
+ program: |
72
+ . as {$stations, $readings, $fallback}
73
+ | ($stations | INDEX(.id)) as $by_id
74
+ | {joined: [$readings[]
75
+ | . + {name: ($by_id[.station].name // $fallback),
76
+ region: ($by_id[.station].region // null)}],
77
+ unmatched: [$readings[] | select($by_id[.station] == null) | .station]}
@@ -0,0 +1,76 @@
1
+ # Turn long rows -- one record per measurement -- back into one wide row per subject.
2
+ #
3
+ # In: [{station, date, measure, value}, ...].
4
+ # Out: [{station, date, celsius, humidity, rainfall_mm}, ...], one row per station and date,
5
+ # with an explicit null in every cell the long form had no record for.
6
+ #
7
+ # This is jq-pivot-wide-to-long.yaml run backwards, and it is the harder direction, because
8
+ # widening has to decide three things the melt never had to:
9
+ #
10
+ # - What identifies a row. The key is composite here, so group_by takes a list and jq
11
+ # compares those lists element by element.
12
+ # - What the full set of columns is. It is collected from the whole input before the
13
+ # grouping, not from each group, or the rows come out ragged and something downstream
14
+ # has to reconcile them.
15
+ # - What a missing cell means. A blank object of nulls is merged under each group, so
16
+ # every output row carries every column whether that subject reported it or not.
17
+ #
18
+ # from_entries is the inverse of the melt's to_entries. Two records for the same measure in
19
+ # one group would quietly leave the last one standing; catching that is a validate.schema
20
+ # gate's job, never a reshape's.
21
+ #
22
+ # To change it: -p absent=n/a fills unreported cells with a marker instead of null, which is
23
+ # what a csv reader that cannot tell an empty cell from a missing one wants.
24
+ #
25
+ # dg run --local examples/recipes/jq-long-to-wide.yaml
26
+ # dg run --local examples/recipes/jq-long-to-wide.yaml -p absent=n/a
27
+
28
+ format: dirigent/v1
29
+ kind: pipeline
30
+ code: jq-long-to-wide
31
+ name: Pivot long rows to wide
32
+ description: Widen one-record-per-measurement rows into one row per subject, with an explicit fill in every unreported cell.
33
+
34
+ tags: [recipes, transform, jq]
35
+
36
+ requires:
37
+ blocks:
38
+ - value.const
39
+ - transform.jq
40
+
41
+ params:
42
+ type: object
43
+ properties:
44
+ absent:
45
+ type: [string, "null"]
46
+ default: null
47
+ description: What fills a cell the long form carried no record for.
48
+
49
+ steps:
50
+ long:
51
+ block: value.const
52
+ config:
53
+ value:
54
+ - { station: st-1, date: "2026-01-01", measure: celsius, value: 4 }
55
+ - { station: st-1, date: "2026-01-01", measure: humidity, value: 81 }
56
+ - { station: st-1, date: "2026-01-01", measure: rainfall_mm, value: 0 }
57
+ - { station: st-2, date: "2026-01-01", measure: celsius, value: -1 }
58
+ # st-2 reported no humidity that day: the wide row still gets the column.
59
+ - { station: st-2, date: "2026-01-01", measure: rainfall_mm, value: 3 }
60
+ - { station: st-1, date: "2026-01-02", measure: celsius, value: 6 }
61
+
62
+ wide:
63
+ block: transform.jq
64
+ depends_on: [long]
65
+ config:
66
+ input:
67
+ rows: ${steps.long.output.value}
68
+ absent: ${params.absent}
69
+ program: |
70
+ . as {$rows, $absent}
71
+ | ([$rows[].measure] | unique) as $columns
72
+ | ($columns | map({key: ., value: $absent}) | from_entries) as $blank
73
+ | [$rows
74
+ | group_by([.station, .date])[]
75
+ | {station: .[0].station, date: .[0].date}
76
+ + ($blank + (map({key: .measure, value: .value}) | from_entries))]
@@ -0,0 +1,89 @@
1
+ # Flatten deeply nested records into flat ones a csv or a parquet writer will accept.
2
+ #
3
+ # In: records with nested objects and a list -- {id, site: {name, coords: {lat, lon}},
4
+ # tags: [...]}.
5
+ # Out: flat records whose keys are the joined paths -- site.name, site.coords.lat, tags.
6
+ #
7
+ # Every row-oriented sink here refuses a nested value rather than writing it as text:
8
+ # convert.std says so naming the row and the key, and convert.arrow says the same about a
9
+ # parquet column. Flattening is therefore not a nicety before a csv, it is the step that
10
+ # makes the csv possible at all.
11
+ #
12
+ # paths is the whole recipe. It emits the path to every leaf of a value, as an array of
13
+ # keys, and getpath reads the value at one -- so join(".") is the single place that decides
14
+ # what a flattened key looks like, and changing the separator is changing that one string.
15
+ #
16
+ # The predicate handed to paths is where a boolean column goes missing. paths(f) keeps a
17
+ # path when f answers a truthy value, and for a cell holding false the tidy-looking
18
+ # paths(scalars) answers false -- so the row silently loses the column. Writing the
19
+ # predicate as a type comparison keeps it, and st-2's active: false in the output is the
20
+ # proof.
21
+ #
22
+ # Lists are the case worth deciding rather than defaulting. paths walks into them too, which
23
+ # would give tags.0 and tags.1 -- columns that move whenever an element is inserted. So the
24
+ # list-valued fields are pulled out first and joined into one cell, by the step that knows
25
+ # what the separator should be, and everything left flattens by path.
26
+ #
27
+ # To change it: -p separator=__ changes the joined key, which some warehouse loaders prefer
28
+ # to a dot they would otherwise have to quote.
29
+ #
30
+ # dg run --local examples/recipes/jq-nested-to-flat.yaml
31
+ # dg run --local examples/recipes/jq-nested-to-flat.yaml -p separator=__
32
+
33
+ format: dirigent/v1
34
+ kind: pipeline
35
+ code: jq-nested-to-flat
36
+ name: Flatten nested records
37
+ description: Flatten nested records into joined-path keys, with list fields collapsed into one cell rather than indexed columns.
38
+
39
+ tags: [recipes, transform, jq]
40
+
41
+ requires:
42
+ blocks:
43
+ - value.const
44
+ - transform.jq
45
+
46
+ params:
47
+ type: object
48
+ properties:
49
+ separator:
50
+ type: string
51
+ default: "."
52
+ description: What joins the path segments of a flattened key.
53
+
54
+ steps:
55
+ nested:
56
+ block: value.const
57
+ config:
58
+ value:
59
+ - id: st-1
60
+ site: { name: Harbour, coords: { lat: 59.91, lon: 10.75 } }
61
+ tags: [coastal, tidal]
62
+ active: true
63
+ - id: st-2
64
+ site: { name: Ridge, coords: { lat: 60.39, lon: 5.32 } }
65
+ tags: [alpine]
66
+ active: false
67
+
68
+ flat:
69
+ block: transform.jq
70
+ depends_on: [nested]
71
+ config:
72
+ input:
73
+ rows: ${steps.nested.output.value}
74
+ separator: ${params.separator}
75
+ program: |
76
+ . as {$rows, $separator}
77
+ | [$rows[]
78
+ # The list-valued fields are collapsed before the walk, so a new element never
79
+ # adds a column.
80
+ | (to_entries | map(select(.value | type == "array"))
81
+ | map({key: .key, value: (.value | join(","))}) | from_entries) as $lists
82
+ | (to_entries | map(select(.value | type != "array")) | from_entries) as $rest
83
+ # paths(f) keeps a path when f answers a truthy value, so paths(scalars) drops
84
+ # every leaf that is itself false. The predicate is written as a comparison so a
85
+ # false cell survives the flattening.
86
+ | ([$rest | paths(type | . != "object" and . != "array") as $p
87
+ | {key: ($p | join($separator)), value: getpath($p)}]
88
+ | from_entries)
89
+ + $lists]
@@ -0,0 +1,65 @@
1
+ # Turn wide rows -- one column per measure -- into long rows, one record per measurement.
2
+ #
3
+ # In: [{station, date, celsius, humidity, rainfall_mm}, ...], a column per measure.
4
+ # Out: [{station, date, measure, value}, ...], one record per measure per row.
5
+ #
6
+ # Long is the shape almost every sink wants: a data value set, a tidy dataframe, a csv whose
7
+ # header does not change when a new instrument is installed. The verb that gets there is
8
+ # to_entries, which turns whatever is left of the object into {key, value} pairs once the
9
+ # identifying columns have been removed with del. Naming the identifiers once, rather than
10
+ # the measures, is what makes the program survive a new column.
11
+ #
12
+ # Dropping nulls is the decision worth seeing. A wide row carries a cell for a measure that
13
+ # was never taken, and a long row for it would be a claim that a measurement of null
14
+ # happened. select is where that claim is refused.
15
+ #
16
+ # To change it: -p keep_nulls=true emits a long row for every cell, which is what a sink
17
+ # that distinguishes "not reported" from "not measured" wants.
18
+ #
19
+ # dg run --local examples/recipes/jq-pivot-wide-to-long.yaml
20
+ # dg run --local examples/recipes/jq-pivot-wide-to-long.yaml -p keep_nulls=true
21
+
22
+ format: dirigent/v1
23
+ kind: pipeline
24
+ code: jq-pivot-wide-to-long
25
+ name: Pivot wide rows to long
26
+ description: Melt a column-per-measure table into one record per measurement, dropping the cells that were never measured.
27
+
28
+ tags: [recipes, transform, jq]
29
+
30
+ requires:
31
+ blocks:
32
+ - value.const
33
+ - transform.jq
34
+
35
+ params:
36
+ type: object
37
+ properties:
38
+ keep_nulls:
39
+ type: boolean
40
+ default: false
41
+ description: Emit a long row for a measure that has no value, rather than dropping it.
42
+
43
+ steps:
44
+ wide:
45
+ block: value.const
46
+ config:
47
+ value:
48
+ - { station: st-1, date: "2026-01-01", celsius: 4, humidity: 81, rainfall_mm: 0 }
49
+ - { station: st-2, date: "2026-01-01", celsius: -1, humidity: 74, rainfall_mm: null }
50
+ - { station: st-1, date: "2026-01-02", celsius: 6, humidity: null, rainfall_mm: 12 }
51
+
52
+ long:
53
+ block: transform.jq
54
+ depends_on: [wide]
55
+ config:
56
+ input:
57
+ rows: ${steps.wide.output.value}
58
+ keep_nulls: ${params.keep_nulls}
59
+ program: |
60
+ . as {$rows, $keep_nulls}
61
+ | [$rows[]
62
+ | . as $row
63
+ | (del(.station, .date) | to_entries[])
64
+ | select($keep_nulls or .value != null)
65
+ | {station: $row.station, date: $row.date, measure: .key, value: .value}]