dirigent-examples 0.15.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_examples/__init__.py +22 -0
- dirigent_examples/py.typed +0 -0
- dirigent_examples/shelves/README.md +299 -0
- dirigent_examples/shelves/composition/README.md +18 -0
- dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
- dirigent_examples/shelves/composition/composition-child.yaml +64 -0
- dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
- dirigent_examples/shelves/connections.yaml +52 -0
- dirigent_examples/shelves/demo/README.md +19 -0
- dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
- dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
- dirigent_examples/shelves/demo/requires.yaml +65 -0
- dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
- dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
- dirigent_examples/shelves/docker/README.md +29 -0
- dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
- dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
- dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
- dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
- dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
- dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
- dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
- dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
- dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
- dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
- dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
- dirigent_examples/shelves/execute/README.md +16 -0
- dirigent_examples/shelves/execute/long-log.yaml +89 -0
- dirigent_examples/shelves/failure/README.md +20 -0
- dirigent_examples/shelves/failure/error-handler.yaml +89 -0
- dirigent_examples/shelves/failure/optional-step.yaml +82 -0
- dirigent_examples/shelves/failure/retries.yaml +75 -0
- dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
- dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
- dirigent_examples/shelves/git/README.md +32 -0
- dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
- dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
- dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
- dirigent_examples/shelves/graph/README.md +22 -0
- dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
- dirigent_examples/shelves/graph/fan-in.yaml +80 -0
- dirigent_examples/shelves/graph/fan-out.yaml +66 -0
- dirigent_examples/shelves/graph/linear.yaml +66 -0
- dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
- dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
- dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
- dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
- dirigent_examples/shelves/hello-world.yaml +30 -0
- dirigent_examples/shelves/open-data/README.md +67 -0
- dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
- dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
- dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
- dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
- dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
- dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
- dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
- dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
- dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
- dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
- dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
- dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
- dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
- dirigent_examples/shelves/patterns/README.md +144 -0
- dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
- dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
- dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
- dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
- dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
- dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
- dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
- dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
- dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
- dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
- dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
- dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
- dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
- dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
- dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
- dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
- dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
- dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
- dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
- dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
- dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
- dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
- dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
- dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
- dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
- dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
- dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
- dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
- dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
- dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
- dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
- dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
- dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
- dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
- dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
- dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
- dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
- dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
- dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
- dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
- dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
- dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
- dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
- dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
- dirigent_examples/shelves/python/README.md +31 -0
- dirigent_examples/shelves/python/apply_and_run.py +52 -0
- dirigent_examples/shelves/python/ci_gate.py +76 -0
- dirigent_examples/shelves/python/connections.py +61 -0
- dirigent_examples/shelves/python/error_handling.py +84 -0
- dirigent_examples/shelves/python/follow_logs.py +39 -0
- dirigent_examples/shelves/python/list_and_filter.py +52 -0
- dirigent_examples/shelves/queues/README.md +59 -0
- dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
- dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
- dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
- dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
- dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
- dirigent_examples/shelves/recipes/README.md +130 -0
- dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
- dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
- dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
- dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
- dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
- dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
- dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
- dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
- dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
- dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
- dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
- dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
- dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
- dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
- dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
- dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
- dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
- dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
- dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
- dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
- dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
- dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
- dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
- dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
- dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
- dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
- dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
- dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
- dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
- dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
- dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
- dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
- dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
- dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
- dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
- dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
- dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
- dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
- dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
- dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
- dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
- dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
- dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
- dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
- dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
- dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
- dirigent_examples/shelves/s3/README.md +34 -0
- dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
- dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
- dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
- dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
- dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
- dirigent_examples/shelves/schemas/README.md +36 -0
- dirigent_examples/shelves/schemas/echo-reading.json +18 -0
- dirigent_examples/shelves/schemas/ou-record.json +13 -0
- dirigent_examples/shelves/schemas/station-reading.json +13 -0
- dirigent_examples/shelves/sensors/README.md +16 -0
- dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
- dirigent_examples/shelves/sensors/time-window.yaml +61 -0
- dirigent_examples/shelves/sql/README.md +52 -0
- dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
- dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
- dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
- dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
- dirigent_examples/shelves/sql/warehouse.sql +42 -0
- dirigent_examples/shelves/transform/README.md +36 -0
- dirigent_examples/shelves/transform/csv-report.yaml +55 -0
- dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
- dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
- dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
- dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
- dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
- dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
- dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
- dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
- dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
- dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
- dirigent_examples/shelves/triggers/README.md +45 -0
- dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
- dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
- dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
- dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
- dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
- dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
- dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
- dirigent_examples/shelves/validate/README.md +31 -0
- dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
- dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
- dirigent_examples-0.15.0.dist-info/METADATA +21 -0
- dirigent_examples-0.15.0.dist-info/RECORD +216 -0
- dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
- dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
- dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# A drive step that fails, and the teardown that runs anyway.
|
|
2
|
+
#
|
|
3
|
+
# THIS PIPELINE FAILS ON PURPOSE. It is here to show what the run looks like afterwards, so
|
|
4
|
+
# the expected outcome is a failed run with a succeeded teardown in it and an empty daemon.
|
|
5
|
+
#
|
|
6
|
+
# REQUIRES A DOCKER DAEMON the worker can reach, named by DOCKER_HOST (a dind sidecar over
|
|
7
|
+
# tcp+TLS, or the local socket in dev). docs/docker.md is the family's home.
|
|
8
|
+
#
|
|
9
|
+
# Every block here declares local_execution, so the engine refuses them unless the instance
|
|
10
|
+
# allowlists their ids:
|
|
11
|
+
# dg run --local examples/docker/docker-run-failing-teardown.yaml \
|
|
12
|
+
# --enable-unsafe docker.compose.up,docker.compose.down,docker.run
|
|
13
|
+
# export DIRIGENT_ENABLED_UNSAFE_BLOCKS='["docker.compose.up","docker.compose.down","docker.run"]'
|
|
14
|
+
#
|
|
15
|
+
# Three hops:
|
|
16
|
+
# stack one nginx service, up and healthy, persisting for the rest of the run.
|
|
17
|
+
# drive a container that reaches the service, prints what it got, and then exits 1.
|
|
18
|
+
# The container's non-zero exit is the step's failure, and its stderr is what the
|
|
19
|
+
# failure message quotes.
|
|
20
|
+
# teardown the same project down again, under rule: all_done.
|
|
21
|
+
#
|
|
22
|
+
# WHAT THE RUN'S STEP LIST READS LIKE AFTERWARDS: stack succeeded, drive failed, teardown
|
|
23
|
+
# succeeded, run failed. A teardown is not an apology for a failure and does not hide one --
|
|
24
|
+
# the run's status is failed because a step of it failed, whatever ran after. That is the
|
|
25
|
+
# whole reason down is its own step rather than something up does at the end of itself:
|
|
26
|
+
# all_done is an ordinary engine rule, evaluated the same way every other rule is, so no
|
|
27
|
+
# "finally" primitive is needed and no block has to guess when a stack is finished with.
|
|
28
|
+
#
|
|
29
|
+
# WHAT THE DAEMON HOLDS AFTERWARDS: nothing of this run.
|
|
30
|
+
# docker ps -a --filter label=com.docker.compose.project
|
|
31
|
+
# lists nothing from it, because the teardown ran. Take rule: all_done off the down step and
|
|
32
|
+
# the same run leaves nginx running until somebody notices.
|
|
33
|
+
#
|
|
34
|
+
# TO MAKE IT YOURS: the drive step is standing in for whatever real work a stack is brought up
|
|
35
|
+
# for -- a migration, a test suite, a load. Give it your command; keep the shape.
|
|
36
|
+
|
|
37
|
+
format: dirigent/v1
|
|
38
|
+
kind: pipeline
|
|
39
|
+
code: docker-run-failing-teardown
|
|
40
|
+
name: A failing drive step over a stack
|
|
41
|
+
description: A container that reaches the stack and then exits 1, with the teardown running on any outcome.
|
|
42
|
+
|
|
43
|
+
tags: [docker, execute]
|
|
44
|
+
|
|
45
|
+
requires:
|
|
46
|
+
blocks:
|
|
47
|
+
- docker.compose.up
|
|
48
|
+
- docker.compose.down
|
|
49
|
+
- docker.run
|
|
50
|
+
|
|
51
|
+
steps:
|
|
52
|
+
stack:
|
|
53
|
+
block: docker.compose.up
|
|
54
|
+
deadline: 10m
|
|
55
|
+
config:
|
|
56
|
+
wait: true
|
|
57
|
+
wait_timeout: 2m
|
|
58
|
+
pull: always
|
|
59
|
+
content: |
|
|
60
|
+
services:
|
|
61
|
+
api:
|
|
62
|
+
image: nginx:alpine
|
|
63
|
+
healthcheck:
|
|
64
|
+
test: ["CMD-SHELL", "wget -qO- http://localhost/ >/dev/null 2>&1 || exit 1"]
|
|
65
|
+
interval: 2s
|
|
66
|
+
timeout: 2s
|
|
67
|
+
retries: 5
|
|
68
|
+
|
|
69
|
+
drive:
|
|
70
|
+
block: docker.run
|
|
71
|
+
depends_on: [stack]
|
|
72
|
+
deadline: 5m
|
|
73
|
+
config:
|
|
74
|
+
image: alpine:3
|
|
75
|
+
pull: true
|
|
76
|
+
network: "${steps.stack.output.default_network}"
|
|
77
|
+
# The work succeeds and the step still fails: the exit code is the whole verdict, and
|
|
78
|
+
# the last line on stderr is what the step's failure message quotes back.
|
|
79
|
+
argv:
|
|
80
|
+
- sh
|
|
81
|
+
- -c
|
|
82
|
+
- 'wget -qO- http://api/ >/dev/null && echo "the service answered"; echo "and now failing on purpose" >&2; exit 1'
|
|
83
|
+
memory: 64mb
|
|
84
|
+
pids_limit: 64
|
|
85
|
+
|
|
86
|
+
teardown:
|
|
87
|
+
block: docker.compose.down
|
|
88
|
+
depends_on: [drive]
|
|
89
|
+
# Without this the teardown would be skipped the moment drive failed, and the stack would
|
|
90
|
+
# outlive the run.
|
|
91
|
+
rule: all_done
|
|
92
|
+
deadline: 5m
|
|
93
|
+
config:
|
|
94
|
+
down_volumes: true
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# A container that talks while it works, for watching the log stream live.
|
|
2
|
+
#
|
|
3
|
+
# The block is asynchronous: execute starts the container detached and the engine probes it
|
|
4
|
+
# on a clock. Every probe reads what the container printed since the last one and appends it
|
|
5
|
+
# to the run's log, so the run's screen and `dg logs` both move while the container runs --
|
|
6
|
+
# a hundred ticks over three and a half minutes, arriving a couple at a time.
|
|
7
|
+
#
|
|
8
|
+
# REQUIRES A DOCKER SOCKET, like every docker.run step: without a reachable daemon the
|
|
9
|
+
# document applies and the step fails saying so. The id is unsafe and must be allowlisted:
|
|
10
|
+
# dg run --local examples/docker/docker-ticker.yaml --enable-unsafe docker.run
|
|
11
|
+
# export DIRIGENT_ENABLED_UNSAFE_BLOCKS='["docker.run"]' # for the instance
|
|
12
|
+
|
|
13
|
+
format: dirigent/v1
|
|
14
|
+
kind: pipeline
|
|
15
|
+
code: docker-ticker
|
|
16
|
+
name: Docker ticker
|
|
17
|
+
description: A container printing a line every two seconds, streamed into the run as it goes.
|
|
18
|
+
|
|
19
|
+
tags: [docker, execute]
|
|
20
|
+
|
|
21
|
+
requires:
|
|
22
|
+
blocks:
|
|
23
|
+
- docker.run
|
|
24
|
+
|
|
25
|
+
params:
|
|
26
|
+
type: object
|
|
27
|
+
additionalProperties: false
|
|
28
|
+
properties:
|
|
29
|
+
ticks:
|
|
30
|
+
type: integer
|
|
31
|
+
default: 100
|
|
32
|
+
minimum: 1
|
|
33
|
+
maximum: 1000
|
|
34
|
+
description: How many lines the container prints, two seconds apart.
|
|
35
|
+
|
|
36
|
+
steps:
|
|
37
|
+
tick:
|
|
38
|
+
block: docker.run
|
|
39
|
+
deadline: 40m
|
|
40
|
+
config:
|
|
41
|
+
image: alpine:3
|
|
42
|
+
pull: true
|
|
43
|
+
network: none
|
|
44
|
+
argv:
|
|
45
|
+
- sh
|
|
46
|
+
- -c
|
|
47
|
+
- 'i=1; while [ "$i" -le "${params.ticks}" ]; do echo "tick $i of ${params.ticks}"; i=$((i+1)); sleep 2; done'
|
|
48
|
+
memory: 64mb
|
|
49
|
+
pids_limit: 64
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# Execute examples
|
|
2
|
+
|
|
3
|
+
A shell step that runs code on the worker, which is why it needs its entry in
|
|
4
|
+
`DIRIGENT_ENABLED_UNSAFE_BLOCKS` before an instance will run it. The container family --
|
|
5
|
+
`docker.run`, `docker.compose.*` and `docker.build` -- now lives on its own shelf,
|
|
6
|
+
[docker/](../docker).
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
dg run --local examples/execute/long-log.yaml --enable-unsafe shell.run
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
## Pipelines
|
|
13
|
+
|
|
14
|
+
| File | What it teaches |
|
|
15
|
+
| --- | --- |
|
|
16
|
+
| [long-log.yaml](long-log.yaml) | A step that logs enough to scroll, for exercising the terminal pane rather than reading about it. |
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# A step that prints far more than fits anywhere, to see what the terminal and the UI do.
|
|
2
|
+
#
|
|
3
|
+
# Most examples print one line. Real steps print thousands, and a log viewer that is only
|
|
4
|
+
# ever tested against "hello from dirigent" is not tested. This one emits a few hundred lines
|
|
5
|
+
# of plausible batch output on purpose, so there is something to scroll, search, and wrap.
|
|
6
|
+
#
|
|
7
|
+
# It is also the example that shows what shell.run does with a stream it cannot carry inline.
|
|
8
|
+
# The block returns nine output fields for the two streams, and they answer different
|
|
9
|
+
# questions:
|
|
10
|
+
#
|
|
11
|
+
# stdout the HEAD of the stream, cut at the instance's inline capture size
|
|
12
|
+
# stdout_bytes how much the command printed altogether, cut or not
|
|
13
|
+
# stdout_truncated whether the field above is short of the stream
|
|
14
|
+
# stdout_uri where the WHOLE of it was written, never truncated
|
|
15
|
+
#
|
|
16
|
+
# So a downstream step that needs the full output reads stdout_uri and never stdout: the
|
|
17
|
+
# inline copy is a convenience for looking at, and it is the one that lies about length. The
|
|
18
|
+
# same four exist for stderr, which is captured separately -- a run that prints progress to
|
|
19
|
+
# stderr and results to stdout keeps them apart, and this one does exactly that.
|
|
20
|
+
#
|
|
21
|
+
# Lower the instance's inline capture size and the same run starts reporting truncated: true
|
|
22
|
+
# with an unchanged stdout_bytes, which is the pair worth watching.
|
|
23
|
+
#
|
|
24
|
+
# This one uses shell.run, which executes code on the worker and is refused unless the
|
|
25
|
+
# instance allowlists it: export DIRIGENT_ENABLED_UNSAFE_BLOCKS='["shell.run"]', or pass
|
|
26
|
+
# --enable-unsafe shell.run to a local run.
|
|
27
|
+
#
|
|
28
|
+
# dg run --local examples/execute/long-log.yaml --enable-unsafe shell.run
|
|
29
|
+
# dg run --local examples/execute/long-log.yaml --enable-unsafe shell.run -p lines=2000
|
|
30
|
+
|
|
31
|
+
format: dirigent/v1
|
|
32
|
+
kind: pipeline
|
|
33
|
+
code: long-log
|
|
34
|
+
name: Long log
|
|
35
|
+
description: |
|
|
36
|
+
A step that prints a few hundred lines, so a log viewer has something to do.
|
|
37
|
+
|
|
38
|
+
The point is the difference between the four `stdout_*` fields: `stdout` is the **head**
|
|
39
|
+
of the stream, `stdout_bytes` is its true length, `stdout_truncated` says whether those
|
|
40
|
+
two disagree, and `stdout_uri` is where the whole of it was written.
|
|
41
|
+
|
|
42
|
+
A downstream step that needs all of it reads the URI. The inline copy is for looking at.
|
|
43
|
+
|
|
44
|
+
tags: [execute]
|
|
45
|
+
|
|
46
|
+
requires:
|
|
47
|
+
blocks:
|
|
48
|
+
- shell.run
|
|
49
|
+
|
|
50
|
+
params:
|
|
51
|
+
type: object
|
|
52
|
+
properties:
|
|
53
|
+
lines:
|
|
54
|
+
type: integer
|
|
55
|
+
description: How many rows the batch pretends to process.
|
|
56
|
+
default: 400
|
|
57
|
+
minimum: 1
|
|
58
|
+
maximum: 100000
|
|
59
|
+
|
|
60
|
+
steps:
|
|
61
|
+
process_batch:
|
|
62
|
+
block: shell.run
|
|
63
|
+
config:
|
|
64
|
+
# Progress goes to stderr and results go to stdout, because the two streams are
|
|
65
|
+
# captured separately and this is what makes that visible rather than merely true.
|
|
66
|
+
command: |
|
|
67
|
+
echo "batch starting: ${params.lines} rows" >&2
|
|
68
|
+
i=1
|
|
69
|
+
while [ "$i" -le "${params.lines}" ]; do
|
|
70
|
+
printf 'row %05d station=st-%03d celsius=%d.%d status=accepted\n' \
|
|
71
|
+
"$i" "$((i % 250))" "$((i % 40 - 10))" "$((i % 10))"
|
|
72
|
+
if [ "$((i % 100))" -eq 0 ]; then
|
|
73
|
+
echo "progress: $i rows" >&2
|
|
74
|
+
fi
|
|
75
|
+
i=$((i + 1))
|
|
76
|
+
done
|
|
77
|
+
echo "batch finished" >&2
|
|
78
|
+
|
|
79
|
+
# Reads the measurements rather than the text, which is the honest way to depend on a
|
|
80
|
+
# stream that may have been cut. stdout_bytes is the whole length even when stdout is not.
|
|
81
|
+
report_volume:
|
|
82
|
+
block: shell.run
|
|
83
|
+
depends_on: [process_batch]
|
|
84
|
+
config:
|
|
85
|
+
argv:
|
|
86
|
+
- echo
|
|
87
|
+
- "wrote ${steps.process_batch.output.stdout_bytes} bytes to stdout
|
|
88
|
+
(truncated inline: ${steps.process_batch.output.stdout_truncated}),
|
|
89
|
+
whole stream at ${steps.process_batch.output.stdout_uri}"
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# Failure examples
|
|
2
|
+
|
|
3
|
+
How a run goes wrong on purpose: retries and the budget they spend, a timeout that ends an
|
|
4
|
+
attempt, a step allowed to fail without failing the run, and the always-runs cleanup edge.
|
|
5
|
+
Seeing these settle red is the point; a corpus where everything is green teaches nothing
|
|
6
|
+
about reading a failure.
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
dg run --local examples/failure/retries.yaml
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
## Pipelines
|
|
13
|
+
|
|
14
|
+
| File | What it teaches |
|
|
15
|
+
| --- | --- |
|
|
16
|
+
| [retries.yaml](retries.yaml) | The retry policy: attempts, backoff, and an error class that says whether trying again could ever help. |
|
|
17
|
+
| [retry-budget.yaml](retry-budget.yaml) | The budget behind the policy: an unknown failure (a non-zero exit) is retried too, and a hopeless step spends every attempt before it fails. |
|
|
18
|
+
| [step-timeout.yaml](step-timeout.yaml) | A timeout ends the attempt, and the policy decides whether another one starts. |
|
|
19
|
+
| [optional-step.yaml](optional-step.yaml) | `completed_with_errors`: a step that may fail without failing the run, and the third status that says so. |
|
|
20
|
+
| [error-handler.yaml](error-handler.yaml) | The cleanup edge: alert on failure, and always release the lock, however the load went. |
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# Error handling drawn in the DAG rather than coded inside a block.
|
|
2
|
+
#
|
|
3
|
+
# rule is the edge condition on a step's depends_on. Four values, and they are the whole
|
|
4
|
+
# vocabulary for branching on an outcome:
|
|
5
|
+
#
|
|
6
|
+
# all_success (the default) every prerequisite succeeded
|
|
7
|
+
# one_failed at least one prerequisite failed; skipped when none did
|
|
8
|
+
# all_done every prerequisite settled, whatever it settled as
|
|
9
|
+
# always runs once every prerequisite has settled, whatever the outcome -- the
|
|
10
|
+
# engine still waits for upstream to finish, it does not fire mid-flight
|
|
11
|
+
#
|
|
12
|
+
# one_failed is what makes a step an error-handler branch. all_done is where cleanup that
|
|
13
|
+
# must happen either way belongs. Both are visible in the graph, which is the point: the
|
|
14
|
+
# failure path is drawn in the DAG rather than coded inside a block.
|
|
15
|
+
#
|
|
16
|
+
# There is one other failure knob, and it is a different question. rule decides whether a
|
|
17
|
+
# step runs; continue_on_failure decides whether a step that failed sinks the run. A step
|
|
18
|
+
# marked continue_on_failure: true settles as failed, but its dependents are shown a success,
|
|
19
|
+
# so their all_success edges fire and the branch carries on, while the run ends
|
|
20
|
+
# completed_with_errors instead of failed. That is also why a one_failed handler below a
|
|
21
|
+
# tolerated step never runs: the rules are shown no failure to handle. optional-step.yaml is
|
|
22
|
+
# that flag on its own. Use it for the branch nobody should be paged about; use rule for the
|
|
23
|
+
# branch that exists to handle the failure.
|
|
24
|
+
# For a fan-out the equivalent is items: continue, which tolerates one bad element rather
|
|
25
|
+
# than one bad step.
|
|
26
|
+
#
|
|
27
|
+
# As written the load fails, so the notify branch runs and the success branch is skipped.
|
|
28
|
+
#
|
|
29
|
+
# This one uses shell.run, which executes code on the worker and is refused unless the
|
|
30
|
+
# instance allowlists it: export DIRIGENT_ENABLED_UNSAFE_BLOCKS='["shell.run"]', or pass
|
|
31
|
+
# --enable-unsafe shell.run to a local run.
|
|
32
|
+
#
|
|
33
|
+
# dg run --local examples/failure/error-handler.yaml
|
|
34
|
+
# dg run --local examples/failure/error-handler.yaml -p status=200 # the other branch
|
|
35
|
+
|
|
36
|
+
format: dirigent/v1
|
|
37
|
+
kind: pipeline
|
|
38
|
+
code: error-handler
|
|
39
|
+
name: Handle a failing step
|
|
40
|
+
description: Load a batch, alert on failure, and always release the lock afterwards.
|
|
41
|
+
|
|
42
|
+
tags: [failure, execute, http, graph]
|
|
43
|
+
|
|
44
|
+
params:
|
|
45
|
+
type: object
|
|
46
|
+
properties:
|
|
47
|
+
status:
|
|
48
|
+
type: integer
|
|
49
|
+
description: The status the load endpoint is asked to return.
|
|
50
|
+
default: 500
|
|
51
|
+
|
|
52
|
+
steps:
|
|
53
|
+
load_batch:
|
|
54
|
+
block: http.request
|
|
55
|
+
retry:
|
|
56
|
+
max_attempts: 2
|
|
57
|
+
backoff: 1s
|
|
58
|
+
config:
|
|
59
|
+
url: "https://postman-echo.com/status/${params.status}"
|
|
60
|
+
method: GET
|
|
61
|
+
|
|
62
|
+
notify_failure:
|
|
63
|
+
block: http.request
|
|
64
|
+
depends_on: [load_batch]
|
|
65
|
+
# one_failed is what makes this a handler: it runs only on the path where the load
|
|
66
|
+
# failed, and skips silently on the path where it did not.
|
|
67
|
+
rule: one_failed
|
|
68
|
+
config:
|
|
69
|
+
url: https://postman-echo.com/post
|
|
70
|
+
method: POST
|
|
71
|
+
body:
|
|
72
|
+
text: the nightly load failed
|
|
73
|
+
|
|
74
|
+
publish_success:
|
|
75
|
+
block: shell.run
|
|
76
|
+
depends_on: [load_batch]
|
|
77
|
+
rule: all_success
|
|
78
|
+
config:
|
|
79
|
+
argv: [echo, "load succeeded"]
|
|
80
|
+
|
|
81
|
+
release_lock:
|
|
82
|
+
block: http.request
|
|
83
|
+
depends_on: [notify_failure, publish_success]
|
|
84
|
+
# Exactly one of the two branches above runs and the other skips, so all_done is what
|
|
85
|
+
# releases the lock either way. all_success would wait for a step that never runs.
|
|
86
|
+
rule: all_done
|
|
87
|
+
config:
|
|
88
|
+
url: https://postman-echo.com/delete
|
|
89
|
+
method: DELETE
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# A step that is allowed to fail, and what that does to the run it is part of.
|
|
2
|
+
#
|
|
3
|
+
# continue_on_failure is one narrow claim: this step failing is not a reason to stop. The
|
|
4
|
+
# engine holds it as "tolerated", and tolerated has three consequences worth separating,
|
|
5
|
+
# because they are easy to blur into one:
|
|
6
|
+
#
|
|
7
|
+
# 1. The step itself still settles as FAILED. Its attempt is red, its error is stored, and
|
|
8
|
+
# its logs are there to read. Tolerating a failure never hides one.
|
|
9
|
+
# 2. Its dependents see SUCCEEDED. An ordinary all_success edge below a tolerated failure
|
|
10
|
+
# fires, so the branch carries on -- which is the entire point of the flag.
|
|
11
|
+
# 3. The run ends completed_with_errors, not succeeded. A run with a tolerated failure in
|
|
12
|
+
# it is not a clean run, and it does not get to claim it was.
|
|
13
|
+
#
|
|
14
|
+
# Consequence 2 has a sharp edge. Because dependents see SUCCEEDED, a rule: one_failed handler
|
|
15
|
+
# hung below a tolerated step never fires: there is, as far as the rules are concerned, no
|
|
16
|
+
# failure to handle. Tolerating a failure and handling one are opposite instructions, so a
|
|
17
|
+
# step should not be asked for both.
|
|
18
|
+
#
|
|
19
|
+
# The failure here is local and instant -- a command that exits non-zero -- so the run settles
|
|
20
|
+
# in about a second and needs no network. What it costs to fail is the point, not what fails.
|
|
21
|
+
#
|
|
22
|
+
# This one uses shell.run, which executes code on the worker and is refused unless the
|
|
23
|
+
# instance allowlists it: export DIRIGENT_ENABLED_UNSAFE_BLOCKS='["shell.run"]', or pass
|
|
24
|
+
# --enable-unsafe shell.run to a local run.
|
|
25
|
+
#
|
|
26
|
+
# dg run --local examples/failure/optional-step.yaml --enable-unsafe shell.run # completed_with_errors
|
|
27
|
+
# dg run --local examples/failure/optional-step.yaml --enable-unsafe shell.run -p exit_code=0 # succeeded
|
|
28
|
+
|
|
29
|
+
format: dirigent/v1
|
|
30
|
+
kind: pipeline
|
|
31
|
+
code: optional-step
|
|
32
|
+
name: Optional step
|
|
33
|
+
description: |
|
|
34
|
+
Import a batch, push metrics on a best-effort basis, and publish either way.
|
|
35
|
+
|
|
36
|
+
`continue_on_failure: true` makes the metrics push **optional**: it settles as failed,
|
|
37
|
+
its dependents see it as succeeded and carry on, and the run reports
|
|
38
|
+
`completed_with_errors` rather than `succeeded`.
|
|
39
|
+
|
|
40
|
+
Nothing is hidden. The red attempt is still there to open.
|
|
41
|
+
|
|
42
|
+
tags: [failure, execute]
|
|
43
|
+
|
|
44
|
+
requires:
|
|
45
|
+
blocks:
|
|
46
|
+
- shell.run
|
|
47
|
+
|
|
48
|
+
params:
|
|
49
|
+
type: object
|
|
50
|
+
properties:
|
|
51
|
+
exit_code:
|
|
52
|
+
type: integer
|
|
53
|
+
description: What the metrics push exits with; 0 makes the whole run green.
|
|
54
|
+
default: 3
|
|
55
|
+
minimum: 0
|
|
56
|
+
maximum: 125
|
|
57
|
+
|
|
58
|
+
steps:
|
|
59
|
+
import_batch:
|
|
60
|
+
block: shell.run
|
|
61
|
+
config:
|
|
62
|
+
argv: [echo, "imported 412 rows"]
|
|
63
|
+
|
|
64
|
+
push_metrics:
|
|
65
|
+
block: shell.run
|
|
66
|
+
depends_on: [import_batch]
|
|
67
|
+
# The one flag this example is about. Nobody should be paged because a metrics endpoint
|
|
68
|
+
# was unhappy, and nobody should be told the run was clean either.
|
|
69
|
+
continue_on_failure: true
|
|
70
|
+
# No retry policy: one attempt, and the exit code is deterministic. retry-budget.yaml is
|
|
71
|
+
# where a failure is spent against a budget instead.
|
|
72
|
+
config:
|
|
73
|
+
command: "echo 'metrics endpoint refused' >&2; exit ${params.exit_code}"
|
|
74
|
+
|
|
75
|
+
publish:
|
|
76
|
+
block: shell.run
|
|
77
|
+
depends_on: [push_metrics]
|
|
78
|
+
# The default rule, left unwritten, is all_success -- and it fires, because a tolerated
|
|
79
|
+
# failure reads as succeeded to whatever depends on it. Without the flag above this step
|
|
80
|
+
# would be skipped and the run would be failed.
|
|
81
|
+
config:
|
|
82
|
+
argv: [echo, "published, with or without metrics"]
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# Retry policy per step, and the difference between a retry and a tolerated failure.
|
|
2
|
+
#
|
|
3
|
+
# A retry is the attempt-level question: should we try this again? The delay is data, so no
|
|
4
|
+
# worker sleeps through a backoff -- a new attempt row is written with available_at in the
|
|
5
|
+
# future and picked up when it comes due. Whether a retry happens at all also depends on
|
|
6
|
+
# the block's own error classification: a 5xx is transient and retried, a 4xx is rejected
|
|
7
|
+
# and never is, however much budget is left.
|
|
8
|
+
#
|
|
9
|
+
# continue_on_failure is the step-level question: what does this failure mean downstream?
|
|
10
|
+
# A tolerated failure reads as succeeded to its dependents, so the branch carries on, while
|
|
11
|
+
# the run itself reports completed_with_errors rather than succeeded.
|
|
12
|
+
#
|
|
13
|
+
# As written, publish hits a deliberate 503 three times over six seconds and then gives up,
|
|
14
|
+
# which is what makes the retry visible. Pass a different status to watch the publish
|
|
15
|
+
# succeed -- the optional cache still times out, so the run still ends completed_with_errors:
|
|
16
|
+
#
|
|
17
|
+
# This one uses shell.run, which executes code on the worker and is refused unless the
|
|
18
|
+
# instance allowlists it: export DIRIGENT_ENABLED_UNSAFE_BLOCKS='["shell.run"]', or pass
|
|
19
|
+
# --enable-unsafe shell.run to a local run.
|
|
20
|
+
#
|
|
21
|
+
# dg run --local examples/failure/retries.yaml # completed_with_errors
|
|
22
|
+
# dg run --local examples/failure/retries.yaml -p status=200 # publish succeeds; completed_with_errors
|
|
23
|
+
|
|
24
|
+
format: dirigent/v1
|
|
25
|
+
kind: pipeline
|
|
26
|
+
code: retries
|
|
27
|
+
name: Retry behavior
|
|
28
|
+
description: Publish to a service that refuses, then warm an optional cache that is slow.
|
|
29
|
+
|
|
30
|
+
tags: [failure, execute, http, retry]
|
|
31
|
+
|
|
32
|
+
params:
|
|
33
|
+
type: object
|
|
34
|
+
properties:
|
|
35
|
+
status:
|
|
36
|
+
type: integer
|
|
37
|
+
description: The status the publish endpoint is asked to return.
|
|
38
|
+
default: 503
|
|
39
|
+
|
|
40
|
+
steps:
|
|
41
|
+
publish:
|
|
42
|
+
block: http.request
|
|
43
|
+
continue_on_failure: true
|
|
44
|
+
retry:
|
|
45
|
+
# 2s, then 4s, then give up: multiplier doubles the wait and max_backoff caps it, so a
|
|
46
|
+
# long budget cannot turn into a long sleep. jitter spreads the retries of a fan-out
|
|
47
|
+
# that all failed at once, instead of firing them again in the same instant.
|
|
48
|
+
max_attempts: 3
|
|
49
|
+
backoff: 2s
|
|
50
|
+
max_backoff: 10s
|
|
51
|
+
multiplier: 2.0
|
|
52
|
+
jitter: 0.2
|
|
53
|
+
config:
|
|
54
|
+
url: "https://postman-echo.com/status/${params.status}"
|
|
55
|
+
method: GET
|
|
56
|
+
|
|
57
|
+
warm_cache:
|
|
58
|
+
block: http.request
|
|
59
|
+
depends_on: [publish]
|
|
60
|
+
continue_on_failure: true
|
|
61
|
+
# The endpoint sleeps five seconds and the timeout is one, so this fails on time rather
|
|
62
|
+
# than on the service: an optional step is not allowed to hold the run open.
|
|
63
|
+
timeout: 1s
|
|
64
|
+
retry:
|
|
65
|
+
max_attempts: 2
|
|
66
|
+
backoff: 1s
|
|
67
|
+
config:
|
|
68
|
+
url: https://postman-echo.com/delay/5
|
|
69
|
+
method: GET
|
|
70
|
+
|
|
71
|
+
finish:
|
|
72
|
+
block: shell.run
|
|
73
|
+
depends_on: [warm_cache]
|
|
74
|
+
config:
|
|
75
|
+
argv: [echo, "published if it was accepted, cache warmed if it was willing"]
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# A budget of three attempts, spent in full on a failure nobody can explain.
|
|
2
|
+
#
|
|
3
|
+
# retries.yaml shows a retry against a service that answers 503, where the block itself calls
|
|
4
|
+
# the failure transient. This is the other half of the rule, and the one that surprises
|
|
5
|
+
# people: a failure the engine cannot classify is retried too.
|
|
6
|
+
#
|
|
7
|
+
# There are exactly three classes, and only one of them ends a step immediately:
|
|
8
|
+
#
|
|
9
|
+
# transient network, a 5xx, a timeout retried while the budget lasts
|
|
10
|
+
# rejected validation, auth, a 4xx never retried, however much budget is left
|
|
11
|
+
# unknown anything else retried while the budget lasts
|
|
12
|
+
#
|
|
13
|
+
# A command that exits non-zero is unknown. shell.run cannot know whether exit 1 meant a
|
|
14
|
+
# corrupt input file or a database that was briefly away, so it declines to guess, and the
|
|
15
|
+
# engine spends the budget rather than assuming the worst. That is what makes max_attempts
|
|
16
|
+
# worth writing on a step whose failures are ordinary program failures rather than HTTP.
|
|
17
|
+
#
|
|
18
|
+
# The consequence is visible here and worth sitting with: this step is hopeless, and it is
|
|
19
|
+
# retried anyway, three times, over roughly three seconds. Nothing knows it is hopeless.
|
|
20
|
+
# A retry budget is a bet that the failure was luck, and the bet is paid whether or not it
|
|
21
|
+
# was -- which is why a step that fails deterministically wants max_attempts: 1, and why
|
|
22
|
+
# step-timeout.yaml has no retry at all.
|
|
23
|
+
#
|
|
24
|
+
# The delay is data rather than a sleep. Each failure writes the next attempt's available_at
|
|
25
|
+
# into the future and the worker moves on, so nothing holds a worker slot for the backoff.
|
|
26
|
+
# Watch the attempts on the step: there are three, a second or so apart, each with its own
|
|
27
|
+
# stderr, and the third one settles the step as failed and the run with it.
|
|
28
|
+
#
|
|
29
|
+
# This one uses shell.run, which executes code on the worker and is refused unless the
|
|
30
|
+
# instance allowlists it: export DIRIGENT_ENABLED_UNSAFE_BLOCKS='["shell.run"]', or pass
|
|
31
|
+
# --enable-unsafe shell.run to a local run.
|
|
32
|
+
#
|
|
33
|
+
# Only config is interpolated, so max_attempts below is a literal: a retry budget is part of
|
|
34
|
+
# the pipeline's shape, not something a caller talks the engine into at run time.
|
|
35
|
+
#
|
|
36
|
+
# dg run --local examples/failure/retry-budget.yaml --enable-unsafe shell.run # fails after 3 attempts
|
|
37
|
+
|
|
38
|
+
format: dirigent/v1
|
|
39
|
+
kind: pipeline
|
|
40
|
+
code: retry-budget
|
|
41
|
+
name: Retry budget
|
|
42
|
+
description: |
|
|
43
|
+
A local command that always exits 1, given three attempts to prove it.
|
|
44
|
+
|
|
45
|
+
A non-zero exit is classified `unknown`, and **unknown is retried**: only `rejected`
|
|
46
|
+
ends a step on its first failure. The budget is spent in full, about a second apart,
|
|
47
|
+
and then the step fails.
|
|
48
|
+
|
|
49
|
+
A retry budget is a bet that the failure was luck. This one loses that bet three times,
|
|
50
|
+
which is what makes it worth reading.
|
|
51
|
+
|
|
52
|
+
tags: [failure, execute, retry]
|
|
53
|
+
|
|
54
|
+
requires:
|
|
55
|
+
blocks:
|
|
56
|
+
- shell.run
|
|
57
|
+
|
|
58
|
+
steps:
|
|
59
|
+
reconcile:
|
|
60
|
+
block: shell.run
|
|
61
|
+
retry:
|
|
62
|
+
# The budget counts the first try, so 3 here means one attempt and two retries.
|
|
63
|
+
max_attempts: 3
|
|
64
|
+
# A second between attempts, doubling, so the whole thing settles in about three
|
|
65
|
+
# seconds instead of the thirty the default backoff would take.
|
|
66
|
+
backoff: 1s
|
|
67
|
+
max_backoff: 4s
|
|
68
|
+
multiplier: 2.0
|
|
69
|
+
jitter: 0.1
|
|
70
|
+
config:
|
|
71
|
+
# Deterministic: there is no exit code where this succeeds, and the engine has no way
|
|
72
|
+
# to know that. It prints which attempt it is on, so the three are told apart in logs.
|
|
73
|
+
command: "echo 'reconcile failed: ledger is short 3 rows' >&2; exit 1"
|
|
74
|
+
|
|
75
|
+
# Never reached. The default all_success edge is not satisfied by a step that spent its
|
|
76
|
+
# budget and failed, and there is no continue_on_failure here to tolerate it -- that is
|
|
77
|
+
# optional-step.yaml, where the failure is allowed and the branch carries on.
|
|
78
|
+
settle:
|
|
79
|
+
block: shell.run
|
|
80
|
+
depends_on: [reconcile]
|
|
81
|
+
config:
|
|
82
|
+
argv: [echo, "only reached if a retry ever worked"]
|