dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,97 @@
1
+ # concurrency: skip -- a second trigger while one run is going creates no run at all.
2
+ #
3
+ # The policy is on the PIPELINE, not on a step and not on a schedule, because the question it
4
+ # answers is about the pipeline: may two runs of this be in flight at once? Four answers:
5
+ #
6
+ # allow (the default) start it; two runs of this can safely overlap.
7
+ # skip do not start it; there is nothing to catch up on. THIS FILE.
8
+ # queue hold it until the slot is free, then release the oldest held one.
9
+ # replace cancel what is running and start this instead.
10
+ #
11
+ # WHAT A SECOND TRIGGER DOES HERE: no run is created. Not queued, not deferred, not started
12
+ # and cancelled -- nothing exists. The trigger is recorded as skipped by the concurrency
13
+ # policy and the clock moves on to its next slot. Nothing is retried later, because skip's
14
+ # claim is that the missed firing had no work of its own: the run already going will refresh
15
+ # whatever the skipped one would have.
16
+ #
17
+ # That claim is what makes skip the right policy for a REFRESH and the wrong one for an
18
+ # INCREMENT. Rebuilding a rolling view, warming a cache, re-reading the last 24 hours: any of
19
+ # those can lose a firing without losing data. Consuming a queue, appending yesterday's rows,
20
+ # advancing a cursor: each of those loses work permanently, and wants queue instead.
21
+ #
22
+ # The decision is serialised. Two triggers arriving in the same instant on different processes
23
+ # take a lock first, so exactly one of them wins the slot -- otherwise both would read
24
+ # "nothing is running" and skip would skip nothing.
25
+ #
26
+ # Hop by hop:
27
+ #
28
+ # slow_scan a sleep long enough that a second trigger while it runs is easy to arrange by
29
+ # hand. This is what stands in for the work.
30
+ # refresh rebuilds the view. Idempotent by construction, which is the property skip
31
+ # depends on.
32
+ #
33
+ # EXPECT THIS RUN TO SUCCEED in about six seconds. A single local run has nothing to overlap
34
+ # with, so the policy is exercised against a server -- start one run, then start another while
35
+ # it is going and watch the second not appear:
36
+ #
37
+ # dg run --local examples/patterns/concurrency-skip.yaml
38
+ # dg apply examples/patterns/concurrency-skip.yaml
39
+ # dg run concurrency-skip &
40
+ # dg run concurrency-skip # no run created; the policy skipped it
41
+ # dg runs list concurrency-skip # one run, not two
42
+
43
+ format: dirigent/v1
44
+ kind: pipeline
45
+ code: concurrency-skip
46
+ name: A second run that never starts
47
+ description: |
48
+ `concurrency: skip` creates no run at all when one is already in flight: not queued, not
49
+ deferred, nothing.
50
+
51
+ Right for a refresh, whose next firing subsumes the one that was skipped. Wrong for an
52
+ increment, which loses that work permanently -- use `queue` there.
53
+
54
+ tags: [patterns, sensor, transform, concurrency]
55
+
56
+ # The one line this file is about. It is a property of the pipeline, because "may two of
57
+ # these overlap" is a question about the pipeline and not about any one trigger.
58
+ concurrency: skip
59
+
60
+ requires:
61
+ blocks:
62
+ - time.sleep
63
+ - transform.jq
64
+
65
+ params:
66
+ type: object
67
+ properties:
68
+ seconds:
69
+ type: integer
70
+ description: How long the scan takes; long enough to trigger a second run by hand.
71
+ default: 5
72
+ minimum: 1
73
+ maximum: 120
74
+ view:
75
+ type: string
76
+ description: Which materialised view the refresh rebuilds.
77
+ default: cases-rolling-24h
78
+
79
+ steps:
80
+ slow_scan:
81
+ block: time.sleep
82
+ # A sensor, so the wait is a parked row rather than a held worker -- the slot is occupied
83
+ # by the RUN, not by a worker, which is why a long run can block a short one.
84
+ poll: 1s
85
+ config:
86
+ for: "${params.seconds}s"
87
+
88
+ refresh:
89
+ block: transform.jq
90
+ depends_on: [slow_scan]
91
+ # A full rebuild, not an append. That is the property skip rests on: a firing that never
92
+ # happened costs nothing, because the next one does all of the work again.
93
+ config:
94
+ input:
95
+ view: "${params.view}"
96
+ program: |
97
+ {rebuilt: .view, incremental: false}
@@ -0,0 +1,140 @@
1
+ # Naming a connection, and carrying one, in the same document -- and when each is right.
2
+ #
3
+ # A document is portable precisely because it names credentials rather than holding them. A
4
+ # step says connection: postman-echo, and the base URL, the auth, the TLS settings and the
5
+ # timeout live on the instance, encrypted at rest and redacted in every API response. Moving
6
+ # from staging to production edits the connection and not one pipeline.
7
+ #
8
+ # THE TWO FORMS:
9
+ #
10
+ # connection: <code> the ordinary one. The instance holds it, requires.connections says so,
11
+ # and an apply against an instance without it is refused up front with
12
+ # the code to create.
13
+ # connections: {...} carried in the document. For a document that has to run with no
14
+ # instance to create a connection on first: a published example, a
15
+ # reproduction, a --local run. A SERVER REFUSES a document that carries
16
+ # one, because applying it would store the credential in every version
17
+ # of the pipeline.
18
+ #
19
+ # Both appear below, which is why this document runs locally and would not apply to a server
20
+ # as it stands. That is not a trick: it is the honest way to show the two side by side.
21
+ #
22
+ # WHAT A CONNECTION BUYS BEYOND SECRECY: the base URL. A step that names one writes a path and
23
+ # not a URL, so the same document points at a test instance and a production instance by
24
+ # naming a different connection -- there is no URL to edit. It also carries the timeout and
25
+ # the health path, which is what dg connection check calls, so a connection can be tested
26
+ # without running a pipeline.
27
+ #
28
+ # A LOCAL RUN HAS NO SERVER to hold the named one, so it is handed over:
29
+ #
30
+ # dg run --local examples/patterns/connections-referenced-vs-carried.yaml \
31
+ # --connections examples/connections.yaml
32
+ #
33
+ # Without --connections the run is refused at the step that names it, because a connection
34
+ # that is not there is not something to guess at.
35
+ #
36
+ # Hop by hop:
37
+ #
38
+ # named http.request through the named connection. It writes a PATH; the base URL comes
39
+ # from the connection, and so do the credentials.
40
+ # carried the same call through the connection this document carries, so both forms are
41
+ # exercised in one run.
42
+ # absolute the third form, for completeness: an absolute url with no connection at all.
43
+ # Fine for a public endpoint, wrong the moment anything needs a credential --
44
+ # because there is nowhere in a document to put one.
45
+ # compare reads all three back. They answer identically, which is the point: a connection
46
+ # changes where a credential lives, not what a step does.
47
+ #
48
+ # EXPECT THIS RUN TO SUCCEED in about three seconds, given --connections.
49
+
50
+ format: dirigent/v1
51
+ kind: pipeline
52
+ code: connections-referenced-vs-carried
53
+ name: A connection named, and one carried
54
+ description: |
55
+ A step names a connection by code and the instance holds the credential, encrypted at rest
56
+ and redacted in every response. That is what makes a document portable.
57
+
58
+ A document may instead **carry** one, for a run with no instance to create it on -- and a
59
+ server refuses such a document, because applying it would store the secret in every
60
+ version.
61
+
62
+ tags: [patterns, http, transform, credential]
63
+
64
+ requires:
65
+ blocks:
66
+ - http.request
67
+ - transform.jq
68
+ # Named here, so an apply against an instance that does not hold it is refused with the
69
+ # code to create rather than failing at the step.
70
+ connections:
71
+ - postman-echo
72
+
73
+ # Carried, for the case with no instance to name one on. The credentials below are Postman's
74
+ # own public demo pair. basic_password is a SecretStr: sealed on the way in, and an applied
75
+ # connection reports which fields are set and never what they are set to.
76
+ connections:
77
+ echo-carried:
78
+ kind: http
79
+ config:
80
+ base_url: https://postman-echo.com
81
+ basic_username: postman
82
+ basic_password: password
83
+ timeout: 30s
84
+ # What dg connection check calls. A connection with no health path can only be tested
85
+ # by running a pipeline through it.
86
+ health_path: /get
87
+
88
+ params:
89
+ type: object
90
+ properties:
91
+ dataset:
92
+ type: string
93
+ description: Echoed by all three calls, so the three answers are comparable.
94
+ default: cases
95
+
96
+ steps:
97
+ named:
98
+ block: http.request
99
+ config:
100
+ # The code, not a URL. examples/connections.yaml is where a --local run gets it from.
101
+ connection: postman-echo
102
+ # A path, resolved against the connection's base URL. This is the line that does not
103
+ # change when the instance moves from staging to production.
104
+ path: /get
105
+ query:
106
+ dataset: "${params.dataset}"
107
+ via: named
108
+
109
+ carried:
110
+ block: http.request
111
+ config:
112
+ connection: echo-carried
113
+ path: /get
114
+ query:
115
+ dataset: "${params.dataset}"
116
+ via: carried
117
+
118
+ absolute:
119
+ block: http.request
120
+ config:
121
+ # No connection at all. Legitimate for a public endpoint with no credential and no
122
+ # environment to move between, and nothing more than that.
123
+ url: https://postman-echo.com/get
124
+ query:
125
+ dataset: "${params.dataset}"
126
+ via: absolute
127
+ # A per-call override of whatever timeout applies. On the two steps above the
128
+ # connection's own timeout would have applied instead.
129
+ timeout: 20s
130
+
131
+ compare:
132
+ block: transform.jq
133
+ depends_on: [named, carried, absolute]
134
+ config:
135
+ input:
136
+ named: "${steps.named.output.body.args.via}"
137
+ carried: "${steps.carried.output.body.args.via}"
138
+ absolute: "${steps.absolute.output.body.args.via}"
139
+ program: |
140
+ {named, carried, absolute, note: "three ways to say where; one thing done"}
@@ -0,0 +1,106 @@
1
+ # deadline: how long a step may keep waiting, and what it costs to wait that long.
2
+ #
3
+ # A sensor does not hold a worker. Each poke is one short, read-only observation -- here, one
4
+ # head request against a URI -- and between pokes the attempt is a parked row with a wake-up
5
+ # time on it. So waiting twelve hours costs a few hundred rows, not twelve hours of a worker,
6
+ # and the deadline can honestly be as long as the business answer requires.
7
+ #
8
+ # The three fields that shape a wait, and none of them lives in the block's config:
9
+ #
10
+ # poll how often to look. Cost per hour of waiting.
11
+ # deadline how long to keep looking. When to give up.
12
+ # on_timeout what giving up MEANS: fail (the default, this file) or skip
13
+ # (timeout-skips-the-step.yaml).
14
+ #
15
+ # This one takes the default, because a drop that has not landed by its deadline is a real
16
+ # problem here: somebody promised a file and did not send it, and the run being red is how
17
+ # that is noticed.
18
+ #
19
+ # Hop by hop:
20
+ #
21
+ # wait_for_drop storage.exists pokes a path under this run's own scratch space every two
22
+ # seconds. Nothing ever writes there, so it never appears, and after eight
23
+ # seconds the deadline expires and the step fails.
24
+ # load the default all_success edge, so it is skipped.
25
+ # page_someone one_failed, so it runs: the absence is reported rather than merely logged.
26
+ #
27
+ # EXPECT THIS RUN TO FAIL, in about eight seconds, with wait_for_drop failed, load skipped and
28
+ # page_someone green. The attempt's error class is rejected, not transient: an expired
29
+ # deadline is not something another attempt could fix, so a retry budget on this step would
30
+ # buy nothing. An expired `timeout` is classified the other way, and
31
+ # timeout-fails-the-step.yaml says why that matters.
32
+ #
33
+ # Note where the drop is looked for. ${run.scratch} is this run's own directory under the
34
+ # instance's artifact root, and file:// URIs are refused outside that root -- so a stored
35
+ # pipeline can never be turned into an arbitrary-file read. A real drop is an s3:// URI or a
36
+ # path under that root, never an absolute path typed into the document.
37
+ #
38
+ # To change it: on_timeout: skip turns this exact run green, which is the same document
39
+ # saying "no drop today is fine". min_size is the other guard worth knowing -- a producer
40
+ # uploading a large object makes the key visible before the bytes are all there, and a floor
41
+ # stops the load reading a half-written file.
42
+ #
43
+ # dg run --local examples/patterns/deadline-on-a-sensor.yaml # fails in ~8s, by design
44
+
45
+ format: dirigent/v1
46
+ kind: pipeline
47
+ code: deadline-on-a-sensor
48
+ name: A deadline on a wait
49
+ description: |
50
+ `poll`, `deadline` and `on_timeout` are step-level engine semantics, uniform across every
51
+ sensor and never buried in a block's config.
52
+
53
+ A parked sensor costs rows, not workers. As written the drop never lands, the deadline
54
+ expires with the default `on_timeout: fail`, and the run reports `failed`.
55
+
56
+ tags: [patterns, http, sensor, storage, timeout]
57
+
58
+ requires:
59
+ blocks:
60
+ - storage.exists
61
+ - storage.copy
62
+ - http.request
63
+
64
+ params:
65
+ type: object
66
+ properties:
67
+ day:
68
+ type: string
69
+ format: date
70
+ description: The day whose drop is expected; it is part of the object's name.
71
+ default: "2026-01-01"
72
+
73
+ steps:
74
+ wait_for_drop:
75
+ block: storage.exists
76
+ # Chosen for how soon the drop matters, not for what it costs to look: each poke is one
77
+ # head request and one row. A daily drop is polled in minutes, not seconds.
78
+ poll: 2s
79
+ # Short enough to watch. A real one is the hour by which somebody promised the file.
80
+ deadline: 8s
81
+ # on_timeout is left unwritten, so it is fail: a promised file that never arrived is a
82
+ # problem somebody has to hear about.
83
+ config:
84
+ uri: "${run.scratch}/drops/${params.day}.parquet"
85
+ # A floor, so a key that exists but is still being written is not treated as a landing.
86
+ min_size: 1kb
87
+
88
+ load:
89
+ block: storage.copy
90
+ depends_on: [wait_for_drop]
91
+ config:
92
+ # Reading the sensor's own output rather than rebuilding the URI: the sensor may have
93
+ # matched a glob, and its answer is which object actually landed.
94
+ source: "${steps.wait_for_drop.output.uri}"
95
+ target: "${run.scratch}/loaded/${params.day}.parquet"
96
+
97
+ page_someone:
98
+ block: http.request
99
+ depends_on: [wait_for_drop]
100
+ rule: one_failed
101
+ config:
102
+ url: https://postman-echo.com/post
103
+ method: POST
104
+ body:
105
+ text: the drop did not land inside its deadline
106
+ day: "${params.day}"
@@ -0,0 +1,88 @@
1
+ # items: continue -- one bad element, and the other nine carrying on without it.
2
+ #
3
+ # This is fan-out-fail-fast.yaml with one word replaced, and the pair is worth running back to
4
+ # back. continue says the elements are independent: a region that refused is a region that
5
+ # refused, not a reason to abandon the other three.
6
+ #
7
+ # Four consequences, and they are easy to blur into one:
8
+ #
9
+ # 1. The failed item settles as failed. Its attempt is red and its error is stored.
10
+ # 2. The other items run to completion regardless of when the failure happened.
11
+ # 3. The STEP counts as succeeded, so an ordinary all_success edge below it fires.
12
+ # 4. The run reports completed_with_errors -- a third status, not a shade of failed.
13
+ #
14
+ # And the sharp edge, which is rule 3's real cost: the step's output is the list of its items'
15
+ # outputs, and a failed item is ABSENT from it. Not null, absent. So the list is shorter than
16
+ # the input, index 0 is the first item that worked rather than the first one asked for, and a
17
+ # downstream step that counts rows is counting the ones that survived. Read the whole list;
18
+ # be careful with indices.
19
+ #
20
+ # If EVERY item fails the step itself failed, and a downstream step forced to run anyway with
21
+ # rule: all_done is refused rather than handed an empty list.
22
+ #
23
+ # Hop by hop:
24
+ #
25
+ # push one item per status. Item two answers 500 and fails; items one and three
26
+ # answer 200 and do not.
27
+ # surviving runs once over the whole batch, and counts what is actually there. Two, not
28
+ # three, and the document says so out loud rather than assuming three.
29
+ #
30
+ # EXPECT THIS RUN TO END completed_with_errors, with two succeeded items, one failed item, a
31
+ # succeeded step, and a surviving.count of 2.
32
+ #
33
+ # To change it: -p statuses='[500,500,500]' makes every item fail, which fails the step, which
34
+ # skips the join. That is the boundary between this file and fan-out-fail-fast.yaml.
35
+ #
36
+ # dg run --local examples/patterns/fan-out-continue.yaml # completed_with_errors
37
+ # dg run --local examples/patterns/fan-out-continue.yaml -p statuses='[500,500,500]' # failed
38
+
39
+ format: dirigent/v1
40
+ kind: pipeline
41
+ code: fan-out-continue
42
+ name: A batch that tolerates one bad item
43
+ description: |
44
+ `items: continue` lets the surviving elements finish. The failed item stays red, the step
45
+ counts as succeeded, and the run reports `completed_with_errors`.
46
+
47
+ A failed item is **absent** from the step's output list rather than null, so the batch a
48
+ downstream step reads can be shorter than the one that was asked for.
49
+
50
+ tags: [patterns, http, transform, failure, fan-out]
51
+
52
+ requires:
53
+ blocks:
54
+ - http.request
55
+ - transform.jq
56
+
57
+ params:
58
+ type: object
59
+ properties:
60
+ statuses:
61
+ type: array
62
+ description: One item per element; each item asks the endpoint for that status.
63
+ default: [200, 500, 200]
64
+ minItems: 1
65
+ maxItems: 16
66
+ items:
67
+ type: integer
68
+ enum: [200, 404, 500]
69
+
70
+ steps:
71
+ push:
72
+ block: http.request
73
+ for_each: "${params.statuses}"
74
+ # The one word this file is about.
75
+ items: continue
76
+ config:
77
+ url: "https://postman-echo.com/status/${item}"
78
+ method: GET
79
+
80
+ surviving:
81
+ block: transform.jq
82
+ depends_on: [push]
83
+ config:
84
+ input: "${steps.push.output}"
85
+ # length rather than a hard-coded three: under continue the batch is whatever survived,
86
+ # and a program that assumes otherwise reports a number nobody can trust.
87
+ program: |
88
+ {statuses: [.[].status], count: length}
@@ -0,0 +1,80 @@
1
+ # items: fail_fast -- the default item policy, and what one bad element costs the batch.
2
+ #
3
+ # A fan-out has its own failure question, separate from every other one in the engine: when
4
+ # one of N items fails, is that a bad element or a bad batch? fail_fast is the default answer
5
+ # and it says bad batch. The step fails, and it fails as a whole -- the successful items do
6
+ # not leave a partial result behind for anything downstream to read.
7
+ #
8
+ # That is the right default. A fan-out is usually N slices of one job, and half a job silently
9
+ # succeeding is worse than a red run: it is a load somebody believes happened.
10
+ #
11
+ # The list below is statuses rather than regions, because the element is what decides whether
12
+ # the item fails, and making that the visible content of the item is the shortest way to show
13
+ # it. The second element is a 500.
14
+ #
15
+ # Hop by hop:
16
+ #
17
+ # push one item per status, each calling /status/<item>. Items one and three answer 200;
18
+ # item two answers 500 and fails. Under fail_fast that is the end of the step.
19
+ # receipt the default all_success edge on a failed step, so it is skipped.
20
+ #
21
+ # EXPECT THIS RUN TO FAIL. Look at the item grid rather than the step: some items are
22
+ # succeeded, one is failed, and depending on timing a sibling may not have run at all. The
23
+ # step is failed regardless, which is the policy.
24
+ #
25
+ # To change it: fan-out-continue.yaml is this file with one word replaced, and it is worth
26
+ # running both back to back. -p statuses='[200,200,200]' makes this one green.
27
+ #
28
+ # dg run --local examples/patterns/fan-out-fail-fast.yaml # fails, by design
29
+ # dg run --local examples/patterns/fan-out-fail-fast.yaml -p statuses='[200,200]' # succeeds
30
+
31
+ format: dirigent/v1
32
+ kind: pipeline
33
+ code: fan-out-fail-fast
34
+ name: One bad item sinks the batch
35
+ description: |
36
+ `items: fail_fast` is the default: one failed element fails the whole fan-out step, and
37
+ nothing downstream reads a partial batch.
38
+
39
+ The right default, because a fan-out is usually N slices of one job, and half a job
40
+ silently succeeding is worse than a red run.
41
+
42
+ tags: [patterns, http, failure, fan-out]
43
+
44
+ requires:
45
+ blocks:
46
+ - http.request
47
+
48
+ params:
49
+ type: object
50
+ properties:
51
+ statuses:
52
+ type: array
53
+ description: One item per element; each item asks the endpoint for that status.
54
+ default: [200, 500, 200]
55
+ minItems: 1
56
+ maxItems: 16
57
+ items:
58
+ type: integer
59
+ enum: [200, 404, 500]
60
+
61
+ steps:
62
+ push:
63
+ block: http.request
64
+ for_each: "${params.statuses}"
65
+ # items is deliberately not written. Omitting it is what makes this file about the
66
+ # default rather than about a setting somebody chose.
67
+ config:
68
+ url: "https://postman-echo.com/status/${item}"
69
+ method: GET
70
+
71
+ receipt:
72
+ block: http.request
73
+ depends_on: [push]
74
+ # Skipped. Under fail_fast there is no half-batch to hand on, so the edge behaves exactly
75
+ # as it would below any other failed step.
76
+ config:
77
+ url: https://postman-echo.com/post
78
+ method: POST
79
+ body:
80
+ pushed: "${steps.push.output}"
@@ -0,0 +1,84 @@
1
+ # A fan-out whose width comes from the run's parameters.
2
+ #
3
+ # for_each: "${params.regions}" is the form to reach for when the caller decides how much work
4
+ # there is: a backfill over the months somebody names, a load over the regions this tenant
5
+ # has, a re-index of the datasets that moved. The pipeline is one document; the run is three
6
+ # items today and eleven tomorrow.
7
+ #
8
+ # The parameter schema is what keeps that honest. regions below is declared as an array of
9
+ # strings with a minimum length, so a run started with a bare string, an empty list, or a
10
+ # number is refused at the door with the reason -- before a run exists, before an item grid
11
+ # is built, before anything is called. A fan-out over an unvalidated parameter is how a
12
+ # pipeline ends up making one request per character of a string.
13
+ #
14
+ # Expansion happens when the run is CREATED, so the item grid is complete and visible from the
15
+ # start, and the reference may read params, run and item but never an upstream step's output.
16
+ #
17
+ # Hop by hop:
18
+ #
19
+ # fetch http.request once per region, with the region in the query string so each item's
20
+ # call is distinguishable in the log. Postman Echo answers with what it was sent,
21
+ # so each item's output carries its own region back.
22
+ # tally one attempt over the whole list, pulling each item's echoed region out again.
23
+ #
24
+ # EXPECT THIS RUN TO SUCCEED with four items, in a couple of seconds.
25
+ #
26
+ # To change it: pass your own list. A JSON array on the command line is the way a list-typed
27
+ # parameter is written, and dotted keys address nested leaves rather than array indices --
28
+ # writing into a list by index is refused, because a dotted path cannot say how long the array
29
+ # is meant to be.
30
+ #
31
+ # dg run --local examples/patterns/fan-out-from-params.yaml
32
+ # dg run --local examples/patterns/fan-out-from-params.yaml -p regions='["east","west"]'
33
+
34
+ format: dirigent/v1
35
+ kind: pipeline
36
+ code: fan-out-from-params
37
+ name: A fan-out sized by its parameters
38
+ description: |
39
+ `for_each: "${params.regions}"` lets the caller decide how wide the run is, while the
40
+ parameter schema keeps that from meaning anything at all.
41
+
42
+ Expansion happens when the run is created, so the item grid is complete before the first
43
+ attempt runs.
44
+
45
+ tags: [patterns, http, transform, fan-out, params]
46
+
47
+ requires:
48
+ blocks:
49
+ - http.request
50
+ - transform.jq
51
+
52
+ params:
53
+ type: object
54
+ properties:
55
+ regions:
56
+ type: array
57
+ description: The regions this run loads; one run item per element.
58
+ default: [east, west, north, south]
59
+ # The guard that matters. Without items and minItems, -p regions=east would be a
60
+ # string, and a fan-out over a string is a fan-out over nothing good.
61
+ minItems: 1
62
+ maxItems: 32
63
+ items:
64
+ type: string
65
+ minLength: 1
66
+
67
+ steps:
68
+ fetch:
69
+ block: http.request
70
+ for_each: "${params.regions}"
71
+ config:
72
+ url: https://postman-echo.com/get
73
+ query:
74
+ # Each item sends its own region, so the log line for one item names the work that
75
+ # item did rather than the work the step did.
76
+ region: "${item}"
77
+
78
+ tally:
79
+ block: transform.jq
80
+ depends_on: [fetch]
81
+ config:
82
+ input: "${steps.fetch.output}"
83
+ program: |
84
+ {loaded: [.[].body.args.region], count: length}