dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,86 @@
1
+ # one_failed: the handler branch, drawn in the DAG instead of coded inside a block.
2
+ #
3
+ # one_failed fires when AT LEAST ONE prerequisite failed, and skips when none did. That is
4
+ # what makes a step an error handler: it sits on the page next to the step it handles, so the
5
+ # failure path is reviewable in a pull request rather than buried in an except clause.
6
+ #
7
+ # Hop by hop:
8
+ #
9
+ # load /status/${params.status}, 500 by default, so it fails.
10
+ # alert one_failed on the load. It runs, and it is the only place the failure is
11
+ # described in words a human reads.
12
+ # carry_on all_success on the load, the default. It is skipped, because the two branches
13
+ # are exclusive by construction: whichever way the load settles, exactly one of
14
+ # the pair runs and the other skips.
15
+ #
16
+ # EXPECT THIS RUN TO FAIL. The handler running is not a repair: the load is still red and
17
+ # untolerated, so the run reports failed with the alert delivered. Pass -p status=200 and the
18
+ # mirror image happens -- carry_on runs, alert skips, and the run succeeds.
19
+ #
20
+ # One sharp edge worth knowing before reaching for this: a step marked
21
+ # continue_on_failure: true shows its dependents a success, so a one_failed handler hung below
22
+ # a tolerated step never fires. Tolerating a failure and handling one are opposite
23
+ # instructions.
24
+ #
25
+ # dg run --local examples/patterns/rule-one-failed.yaml # fails; the alert branch runs
26
+ # dg run --local examples/patterns/rule-one-failed.yaml -p status=200 # succeeds; the alert skips
27
+
28
+ format: dirigent/v1
29
+ kind: pipeline
30
+ code: rule-one-failed
31
+ name: The one_failed edge
32
+ description: |
33
+ `rule: one_failed` runs a step only when a prerequisite **failed**, and skips it when none
34
+ did. It is how an error-handling branch is drawn in the DAG.
35
+
36
+ As written the load fails, the alert branch runs, the success branch is skipped, and the
37
+ run reports `failed`.
38
+
39
+ tags: [patterns, http, graph, rules]
40
+
41
+ requires:
42
+ blocks:
43
+ - http.request
44
+
45
+ params:
46
+ type: object
47
+ properties:
48
+ status:
49
+ type: integer
50
+ description: The status the load endpoint is asked to answer with; 500 fails, 200 does not.
51
+ default: 500
52
+ enum: [200, 500]
53
+
54
+ steps:
55
+ load:
56
+ block: http.request
57
+ config:
58
+ url: "https://postman-echo.com/status/${params.status}"
59
+ method: GET
60
+
61
+ alert:
62
+ block: http.request
63
+ depends_on: [load]
64
+ # The whole file in one word. This step is unreachable on the happy path, and the graph
65
+ # says so without anything having to read a status code.
66
+ rule: one_failed
67
+ config:
68
+ url: https://postman-echo.com/post
69
+ method: POST
70
+ body:
71
+ text: the nightly load failed
72
+ # ${run.id} is what an operator pastes into dg runs show, so an alert carrying it is
73
+ # an alert somebody can act on.
74
+ run: "${run.id}"
75
+
76
+ carry_on:
77
+ block: http.request
78
+ depends_on: [load]
79
+ # The default rule, written out this once so the pair reads as a fork rather than as one
80
+ # step with an option on it.
81
+ rule: all_success
82
+ config:
83
+ url: https://postman-echo.com/post
84
+ method: POST
85
+ body:
86
+ text: the nightly load landed
@@ -0,0 +1,110 @@
1
+ # The one-time clock: a single instant, agreed in advance and reviewable in a pull request.
2
+ #
3
+ # `at:` is the third clock, and it is the cutover and the launch: the migration that happens
4
+ # once, at a moment somebody agreed to, that nobody should have to be awake for. Writing it in
5
+ # the document rather than typing it into a terminal at two in the morning is the entire
6
+ # argument for the field.
7
+ #
8
+ # HOW THE MOMENT IS READ, which is the part that goes wrong silently:
9
+ #
10
+ # at: 2027-03-28 02:30:00 with timezone: Europe/Oslo is half past two in Oslo. A naive
11
+ # moment is anchored in the zone the schedule declares -- not in
12
+ # UTC, and not in whatever the host's local time happens to be.
13
+ # at: 2027-03-28T02:30:00+01:00 is taken as given, and the timezone field is not consulted
14
+ # at all. Write the offset yourself when you mean an instant.
15
+ #
16
+ # Exactly one clock per schedule. cron, interval and at are mutually exclusive, and a schedule
17
+ # that declares none or two is refused when the document is parsed rather than at the moment
18
+ # it should have fired.
19
+ #
20
+ # AFTERWARDS. Once the instant has gone by there is no next firing, so the schedule pauses
21
+ # itself rather than disappearing: its parameters and its firing history stay readable with
22
+ # dg schedule firings, which is what you want on the morning after a cutover.
23
+ #
24
+ # Hop by hop:
25
+ #
26
+ # plan turns the agreed bounds into the batches a migration would run.
27
+ # migrate reads the plan and reports what it covered.
28
+ #
29
+ # EXPECT THIS RUN TO SUCCEED in about a second. The schedule fires only on an instance this
30
+ # document has been applied to; locally the pipeline runs on its defaults.
31
+ #
32
+ # To change it: the pinned params below are the cutover's real bounds, and the pipeline's
33
+ # defaults are a smaller rehearsal -- which is the shape to copy. Rehearse with dg run, and
34
+ # let the schedule carry the real thing.
35
+ #
36
+ # dg run --local examples/patterns/schedule-at-once.yaml
37
+ # dg run --local examples/patterns/schedule-at-once.yaml -p from=2025-01-01 -p to=2025-12-31
38
+ # dg apply examples/patterns/schedule-at-once.yaml
39
+ # dg schedule firings schedule-at-once cutover
40
+
41
+ format: dirigent/v1
42
+ kind: pipeline
43
+ code: schedule-at-once
44
+ name: One agreed instant
45
+ description: |
46
+ `at:` fires once, at a single instant, and then pauses itself so its parameters and firing
47
+ history stay readable.
48
+
49
+ A moment written without an offset is anchored in the zone the schedule declares. Written
50
+ with one, it is taken as given and the zone is not consulted.
51
+
52
+ tags: [patterns, transform, schedule]
53
+
54
+ requires:
55
+ blocks:
56
+ - transform.jq
57
+
58
+ params:
59
+ type: object
60
+ properties:
61
+ from:
62
+ type: string
63
+ format: date
64
+ description: First day of the migration window, inclusive.
65
+ default: "2026-01-01"
66
+ to:
67
+ type: string
68
+ format: date
69
+ description: Last day of the migration window, inclusive.
70
+ default: "2026-01-07"
71
+ environment:
72
+ type: string
73
+ enum: [staging, production]
74
+ default: staging
75
+
76
+ steps:
77
+ plan:
78
+ block: transform.jq
79
+ config:
80
+ input:
81
+ from: "${params.from}"
82
+ to: "${params.to}"
83
+ environment: "${params.environment}"
84
+ program: |
85
+ {environment, from, to, batches: [range(1; 13)]}
86
+
87
+ migrate:
88
+ block: transform.jq
89
+ depends_on: [plan]
90
+ config:
91
+ input: "${steps.plan.output.value}"
92
+ program: |
93
+ {environment, migrated: "\(.from)..\(.to)", batches: (.batches | length)}
94
+
95
+ triggers:
96
+ schedules:
97
+ - code: cutover
98
+ name: The agreed cutover
99
+ description: Runs the production migration once, at the moment operations agreed to.
100
+ # No offset, so this is half past four in the morning in Oslo, read through the zone
101
+ # below. It is also the night the clocks go forward, which is exactly the kind of date
102
+ # a naive moment plus a named zone gets right and a hand-computed UTC instant does not.
103
+ at: 2027-03-28 04:30:00
104
+ timezone: Europe/Oslo
105
+ # The real bounds live here, not in the pipeline's defaults: an ad hoc rehearsal should
106
+ # not accidentally be the cutover.
107
+ params:
108
+ from: "2025-01-01"
109
+ to: "2025-12-31"
110
+ environment: production
@@ -0,0 +1,121 @@
1
+ # A cron expression, and the zone that decides what it means.
2
+ #
3
+ # A cron field is five columns -- minute, hour, day of month, month, day of week -- and it
4
+ # supports the whole ordinary vocabulary: * for every, a list (1,15), a range (1-5), a step
5
+ # (*/15), and the two combined (0-23/2). The three schedules below use enough of it to read as
6
+ # a reference.
7
+ #
8
+ # The zone is the load-bearing part, and it belongs to the SCHEDULE, not to the host the
9
+ # scheduler happens to run on and not to the pipeline. That has one consequence worth stating
10
+ # plainly: "0 5 * * *" in Europe/Oslo is five in the morning in Oslo all year, so it stays
11
+ # five across a daylight-saving boundary rather than sliding to four or six. The same
12
+ # expression in UTC never moves at all, and the difference between those two sentences is
13
+ # exactly why the field exists.
14
+ #
15
+ # ONE PIPELINE, THREE CLOCKS. A schedule is not a second pipeline. The same document below is
16
+ # fired at 05:00 Oslo for the nordics, at 05:00 Kathmandu for Nepal (a zone whose offset is not
17
+ # a whole number of hours, which is a good reason never to hand-roll one), and on the first of
18
+ # the month in UTC for the archive. Each pins its own parameters, which is what makes them
19
+ # different runs of one thing rather than three copies of a document.
20
+ #
21
+ # The schedules exist only on an instance this document has been applied to. The pipeline
22
+ # itself runs anywhere, and locally it takes the pipeline's defaults rather than any
23
+ # schedule's pinned parameters.
24
+ #
25
+ # Hop by hop:
26
+ #
27
+ # plan turns the region and the day into the shape a loader would be handed.
28
+ # receipt reads the plan back and reports what was covered.
29
+ #
30
+ # EXPECT THIS RUN TO SUCCEED in about a second, with no network and no allowlist.
31
+ #
32
+ # To change it: edit an expression and apply again -- an apply brings a managed schedule up to
33
+ # date, and it never touches whether the schedule is paused, when it last fired, or the run
34
+ # history. Those are facts about one instance, so they stay in it.
35
+ #
36
+ # dg run --local examples/patterns/schedule-cron-timezone.yaml
37
+ # dg apply examples/patterns/schedule-cron-timezone.yaml
38
+ # dg schedule list schedule-cron-timezone
39
+
40
+ format: dirigent/v1
41
+ kind: pipeline
42
+ code: schedule-cron-timezone
43
+ name: Cron, in a named zone
44
+ description: |
45
+ Five columns and the zone that decides what they mean. The timezone belongs to the
46
+ schedule, so `0 5 * * *` in `Europe/Oslo` is five in the morning in Oslo across a
47
+ daylight-saving boundary.
48
+
49
+ Three schedules on one pipeline, each pinning its own parameters: a schedule is not a
50
+ second pipeline.
51
+
52
+ tags: [patterns, transform, schedule]
53
+
54
+ requires:
55
+ blocks:
56
+ - transform.jq
57
+
58
+ params:
59
+ type: object
60
+ properties:
61
+ region:
62
+ type: string
63
+ description: Which region the load covers; each schedule pins its own.
64
+ default: nordics
65
+ day:
66
+ type: string
67
+ format: date
68
+ description: The day being loaded.
69
+ default: "2026-01-01"
70
+
71
+ steps:
72
+ plan:
73
+ block: transform.jq
74
+ config:
75
+ input:
76
+ region: "${params.region}"
77
+ day: "${params.day}"
78
+ program: |
79
+ {region, day, datasets: ["cases", "climate"]}
80
+
81
+ receipt:
82
+ block: transform.jq
83
+ depends_on: [plan]
84
+ config:
85
+ input: "${steps.plan.output.value}"
86
+ program: |
87
+ {covered: .region, day: .day, datasets: (.datasets | length)}
88
+
89
+ triggers:
90
+ schedules:
91
+ - code: nordics-nightly
92
+ # A trigger carries the same quartet as anything else addressable: code addresses it,
93
+ # name and description only ever read.
94
+ name: Nightly, Oslo time
95
+ description: Fires once the upstream extract has settled for the day.
96
+ cron: "0 5 * * *"
97
+ timezone: Europe/Oslo
98
+ # Pinned parameters override the pipeline's defaults for this schedule's firings only.
99
+ params:
100
+ region: nordics
101
+ day: "2026-01-01"
102
+
103
+ - code: nepal-nightly
104
+ # Asia/Kathmandu is UTC+05:45. A named zone is the only honest way to write that, and
105
+ # the reason a schedule never takes a fixed offset.
106
+ cron: "0 5 * * *"
107
+ timezone: Asia/Kathmandu
108
+ params:
109
+ region: nepal
110
+ day: "2026-01-01"
111
+
112
+ - code: monthly-archive
113
+ # Day-of-month and month columns, and a minute that is not zero so two schedules that
114
+ # would otherwise collide on the hour do not. Every fifteen minutes would be */15 in
115
+ # the first column; every second hour, 0-23/2 in the second.
116
+ cron: "20 3 1 * *"
117
+ # UTC for a job with no human waiting on it: nothing about an archive is local.
118
+ timezone: UTC
119
+ params:
120
+ region: all
121
+ day: "2026-01-01"
@@ -0,0 +1,109 @@
1
+ # An interval: a cadence, not a calendar, and what happens when the scheduler was away.
2
+ #
3
+ # Write "every 1h" as an interval and "at 05:00 Oslo" as a cron, and neither has to fake the
4
+ # other. An interval is right when what you want is freshness -- a poll, a rolling refresh, a
5
+ # cache warm -- and nobody downstream cares which minute of the hour it lands on. A cron is
6
+ # right when the firing is a calendar moment somebody expects.
7
+ #
8
+ # A duration is the same length in every zone, so an interval never consults a calendar: it
9
+ # does not skip a day, it does not double one, and daylight saving is not a thing it can
10
+ # notice. The timezone field is still there because a schedule carries one, and an interval
11
+ # schedule simply does not read it.
12
+ #
13
+ # THE MISFIRE POLICY IS WHAT KEEPS AN HOURLY SCHEDULE FROM BECOMING A STORM. A firing late by
14
+ # less than the grace (scheduler_misfire_grace, five minutes by default) is ordinary: the
15
+ # clock advances from the slot it owed, so the cadence stays anchored where it was rather than
16
+ # drifting by the length of a tick or by how long a run took. A firing late by MORE than the
17
+ # grace is a misfire: it fires exactly once, and the next firing is computed from now,
18
+ # abandoning every slot that was missed. So a scheduler down over a weekend wakes up and runs
19
+ # this once, not sixty times. Backfilling the gap is a separate, deliberate act -- dg backfill
20
+ # -- and not something a restart does to you.
21
+ #
22
+ # Hop by hop:
23
+ #
24
+ # refresh builds the rolling window the cadence implies, from a parameter rather than from
25
+ # the clock, so the same document runs identically ad hoc.
26
+ # report reads it back and reports the coverage.
27
+ #
28
+ # EXPECT THIS RUN TO SUCCEED in about a second. Locally there is no scheduler, so the two
29
+ # schedules below are declarations that travel with the document and nothing more.
30
+ #
31
+ # To change it: concurrency: skip is on this pipeline on purpose and belongs with the cadence.
32
+ # Two firings of a rolling refresh are not worth having at once, and a refresh that takes
33
+ # longer than its interval would otherwise stack runs until something gives.
34
+ #
35
+ # dg run --local examples/patterns/schedule-interval.yaml
36
+ # dg apply examples/patterns/schedule-interval.yaml
37
+ # dg schedule list schedule-interval
38
+
39
+ format: dirigent/v1
40
+ kind: pipeline
41
+ code: schedule-interval
42
+ name: A rolling cadence
43
+ description: |
44
+ An interval is a cadence rather than a calendar moment: a duration is the same length in
45
+ every zone, so nothing here consults a clock face.
46
+
47
+ A firing missed by more than the misfire grace fires **once** and re-anchors from now, so
48
+ a scheduler that was away for a weekend does not wake up and run sixty times.
49
+
50
+ tags: [patterns, transform, schedule]
51
+
52
+ # The cadence and the concurrency policy belong together: a refresh that runs longer than its
53
+ # interval must not stack a second run on the first. concurrency-skip.yaml is what that
54
+ # policy does in detail.
55
+ concurrency: skip
56
+
57
+ requires:
58
+ blocks:
59
+ - transform.jq
60
+
61
+ params:
62
+ type: object
63
+ properties:
64
+ window_hours:
65
+ type: integer
66
+ description: How much history each refresh re-reads; each schedule pins its own.
67
+ default: 24
68
+ minimum: 1
69
+ maximum: 168
70
+ environment:
71
+ type: string
72
+ enum: [staging, production]
73
+ default: staging
74
+
75
+ steps:
76
+ refresh:
77
+ block: transform.jq
78
+ config:
79
+ input:
80
+ hours: "${params.window_hours}"
81
+ environment: "${params.environment}"
82
+ program: |
83
+ {environment, hours, buckets: [range(0; .hours; 6)]}
84
+
85
+ report:
86
+ block: transform.jq
87
+ depends_on: [refresh]
88
+ config:
89
+ input: "${steps.refresh.output.value}"
90
+ program: |
91
+ {environment, refreshed_hours: .hours, batches: (.buckets | length)}
92
+
93
+ triggers:
94
+ schedules:
95
+ - code: hourly
96
+ name: Hourly refresh
97
+ # A humane duration, not a number of seconds. 90m and 1h30m are the same schedule.
98
+ interval: 1h
99
+ params:
100
+ window_hours: 24
101
+ environment: staging
102
+
103
+ - code: six-hourly-deep
104
+ # The same pipeline at a different cadence over a wider window is a second schedule,
105
+ # not a second pipeline.
106
+ interval: 6h
107
+ params:
108
+ window_hours: 168
109
+ environment: production
@@ -0,0 +1,105 @@
1
+ # The interval a firing COVERS, and why both ends of it are written down.
2
+ #
3
+ # A schedule answers "when does this start". A window answers "what does this run cover", and
4
+ # those are different questions: the nightly load at 05:00 is meant to read yesterday, not
5
+ # this instant. Every schedule-fired run carries the interval that just closed, and a step
6
+ # reads it exactly the way it reads a parameter: ${run.window.start} and ${run.window.end}.
7
+ #
8
+ # HALF-OPEN IS THE WHOLE POINT. start is included, end is excluded, so consecutive firings
9
+ # tile the timeline with no overlap and no gap. A row observed exactly on a boundary belongs
10
+ # to precisely one window. Writing the predicate out -- observed_at >= start and
11
+ # observed_at < end -- is what makes that reviewable, and it is why both ends are in the query
12
+ # rather than a single "day" the step has to interpret.
13
+ #
14
+ # THE WINDOW IS A PROPERTY OF THE RUN, NOT OF THIS DOCUMENT. Nothing below declares it; the
15
+ # cadence computes it. So an ad hoc run has to be given one, and a run that carries no window
16
+ # REFUSES the reference rather than resolving it to an empty string:
17
+ #
18
+ # ${run.window.start} cannot be resolved: this run carries no window; a schedule-fired or
19
+ # backfilled run has one, and an ad hoc run only if it was started with one
20
+ #
21
+ # That refusal is a feature. Resolving to empty would silently widen the query to everything
22
+ # the upstream system has, which is the failure nobody notices until the bill arrives.
23
+ #
24
+ # A window derived from a cron cadence is WALL-CLOCK, not a fixed number of hours. In
25
+ # Europe/Oslo the spring-forward night is 23 hours long and the fall-back night is 25, and the
26
+ # pair of instants says so honestly rather than rounding it away -- which is why the step
27
+ # below computes its own duration from the two ends instead of assuming 86400 seconds.
28
+ #
29
+ # Hop by hop:
30
+ #
31
+ # query builds the extraction predicate from the two ends and the dataset.
32
+ # receipt reads the predicate back and reports the coverage as one string.
33
+ #
34
+ # EXPECT A RUN WITH --window TO SUCCEED, and a run without one to FAIL at the first step with
35
+ # the refusal quoted above. Both are worth doing once:
36
+ #
37
+ # dg run --local examples/patterns/schedule-window-half-open.yaml --window 2026-06-01..2026-06-02
38
+ # dg run --local examples/patterns/schedule-window-half-open.yaml # refused: no window
39
+ # dg apply examples/patterns/schedule-window-half-open.yaml
40
+ # dg backfill schedule-window-half-open --schedule nightly \
41
+ # --from 2026-06-01T05:00:00Z --to 2026-06-08T05:00:00Z --dry-run
42
+
43
+ format: dirigent/v1
44
+ kind: pipeline
45
+ code: schedule-window-half-open
46
+ name: The window a firing covers
47
+ description: |
48
+ `${run.window.start}` and `${run.window.end}` are the interval a run covers: half-open, so
49
+ consecutive firings tile the timeline without overlapping or leaving a gap.
50
+
51
+ The window belongs to the run, not to this document. A run without one **refuses** the
52
+ reference rather than quietly widening the query to everything.
53
+
54
+ tags: [patterns, transform, references, schedule]
55
+
56
+ requires:
57
+ blocks:
58
+ - transform.jq
59
+
60
+ params:
61
+ type: object
62
+ properties:
63
+ dataset:
64
+ type: string
65
+ description: Which dataset the extraction reads; the schedule pins it for every firing.
66
+ default: cases
67
+
68
+ steps:
69
+ query:
70
+ block: transform.jq
71
+ config:
72
+ input:
73
+ dataset: "${params.dataset}"
74
+ # Both ends, as ISO 8601 instants. A reference that is the whole value stays typed,
75
+ # and these are strings, so they arrive as strings.
76
+ from: "${run.window.start}"
77
+ to: "${run.window.end}"
78
+ program: |
79
+ {
80
+ dataset,
81
+ from,
82
+ to,
83
+ predicate: "observed_at >= \(.from) and observed_at < \(.to)"
84
+ }
85
+
86
+ receipt:
87
+ block: transform.jq
88
+ depends_on: [query]
89
+ config:
90
+ input: "${steps.query.output.value}"
91
+ program: |
92
+ {loaded: .dataset, covering: "\(.from)..\(.to)"}
93
+
94
+ triggers:
95
+ schedules:
96
+ - code: nightly
97
+ name: Nightly, Oslo time
98
+ description: Reads the day that closed at five this morning.
99
+ cron: "0 5 * * *"
100
+ # The zone is what makes the derived window a day of Oslo's clock rather than a fixed
101
+ # number of hours, so the daylight-saving nights come out right without anything in the
102
+ # steps knowing that daylight saving exists.
103
+ timezone: Europe/Oslo
104
+ params:
105
+ dataset: cases
@@ -0,0 +1,121 @@
1
+ # http.ready: waiting for a service to say it is up, without holding anything while you wait.
2
+ #
3
+ # A sensor observes and changes nothing. Each poke here is one short GET, "not yet" is the
4
+ # EXPECTED answer rather than a failure, and between pokes the attempt is a parked row with a
5
+ # wake-up time -- so waiting an hour for a service to come up costs a handful of rows and no
6
+ # worker at all. That is the difference between this and a step that loops.
7
+ #
8
+ # The block's own config answers "what counts as ready":
9
+ #
10
+ # expect_status the statuses that mean ready. Empty means any 2xx, which is the sensible
11
+ # default; a service that answers 204 when warm and 503 when not needs
12
+ # nothing here, and one that answers 418 on purpose needs it named.
13
+ # contains an optional body matcher. Readiness ALSO requires this text in the body, so
14
+ # a load balancer answering 200 from a page that says "starting" is not ready.
15
+ # max_response how much of the body is read while looking for it.
16
+ #
17
+ # The step's fields answer "how long, how often, and what if not":
18
+ #
19
+ # poll the cadence. Its default is a minute, which is right for a real service.
20
+ # deadline when to give up. Its default is a day.
21
+ # on_timeout fail (the default) or skip.
22
+ #
23
+ # WHY A NON-2xx IS NOT AN ERROR HERE, which is the thing to internalise: an http.request
24
+ # against a 503 fails the step, because a request that was answered 503 did not do its job. An
25
+ # http.ready against the same 503 has done its job perfectly -- it observed that the service is
26
+ # not ready yet. Same status, same URL, opposite meanings, because one is an operator and the
27
+ # other is a sensor.
28
+ #
29
+ # Hop by hop:
30
+ #
31
+ # api_up pokes an endpoint that answers 200 immediately, so it is ready on the first
32
+ # poke and the run does not wait at all. It also matches on the body, which is the
33
+ # guard that separates "the port is open" from "the service is up".
34
+ # smoke_test the ordinary call that was waiting for it, and the reason the sensor is here.
35
+ # report reads the sensor's own output: which status made it ready, how long the poke
36
+ # took, and whether the body matcher was in play.
37
+ #
38
+ # EXPECT THIS RUN TO SUCCEED in about two seconds, with matched true.
39
+ #
40
+ # To change it: -p path=/status/503 never becomes ready, and after the deadline the step fails
41
+ # -- with on_timeout: skip it would skip instead, which is timeout-skips-the-step.yaml.
42
+ #
43
+ # dg run --local examples/patterns/sensor-http-ready.yaml
44
+ # dg run --local examples/patterns/sensor-http-ready.yaml -p path=/status/503 # fails at the deadline
45
+
46
+ format: dirigent/v1
47
+ kind: pipeline
48
+ code: sensor-http-ready
49
+ name: Waiting for a service
50
+ description: |
51
+ `http.ready` pokes an endpoint until it answers ready. A non-2xx is "not yet", not an
52
+ error -- the same status that would fail an `http.request`.
53
+
54
+ `expect_status` and `contains` say what ready means; `poll`, `deadline` and `on_timeout`
55
+ are step fields and say how long you are willing to wait for it.
56
+
57
+ tags: [patterns, http, sensor, transform]
58
+
59
+ requires:
60
+ blocks:
61
+ - http.ready
62
+ - http.request
63
+ - transform.jq
64
+
65
+ params:
66
+ type: object
67
+ properties:
68
+ path:
69
+ type: string
70
+ description: The readiness path; /get answers 200 at once, /status/503 never answers ready.
71
+ default: /get
72
+ dataset:
73
+ type: string
74
+ description: Echoed by the endpoint, so the body matcher below has something to match on.
75
+ default: cases
76
+
77
+ steps:
78
+ api_up:
79
+ block: http.ready
80
+ # Two seconds so the example is watchable. A real readiness check on a service that takes
81
+ # minutes to warm up polls in minutes: each poke is a scheduled row, and the cost of
82
+ # waiting is rows, not time.
83
+ poll: 2s
84
+ # Short for the same reason. The block's own default is a day, which is the right order of
85
+ # magnitude for "wait until somebody starts it".
86
+ deadline: 20s
87
+ config:
88
+ # A sensor names a target and nothing else: there is no method and no query map here,
89
+ # because looking is a GET and anything a look needs goes in the URL.
90
+ url: "https://postman-echo.com${params.path}?dataset=${params.dataset}"
91
+ # 200 only, rather than any 2xx. Naming it is what stops a 204 with an empty body from
92
+ # counting as a service that is serving.
93
+ expect_status: [200]
94
+ # The body matcher, and the reason it is worth having: a proxy in front of a starting
95
+ # service answers 200 from its own error page, and only the body tells them apart.
96
+ contains: "${params.dataset}"
97
+ max_response: 64kb
98
+
99
+ smoke_test:
100
+ block: http.request
101
+ depends_on: [api_up]
102
+ # The work that was waiting. Ordinary all_success, so it never runs against a service the
103
+ # sensor could not confirm.
104
+ config:
105
+ url: https://postman-echo.com/post
106
+ method: POST
107
+ body:
108
+ dataset: "${params.dataset}"
109
+
110
+ report:
111
+ block: transform.jq
112
+ depends_on: [smoke_test]
113
+ config:
114
+ input:
115
+ ready_status: "${steps.api_up.output.status}"
116
+ poke_ms: "${steps.api_up.output.duration_ms}"
117
+ # True when a contains matcher was configured, so a reader can tell a body-checked
118
+ # readiness from a status-only one.
119
+ matched: "${steps.api_up.output.matched}"
120
+ program: |
121
+ {ready_status, poke_ms, matched}