dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,134 @@
1
+ # A payload well past the inline threshold, and the two places size actually matters.
2
+ #
3
+ # In: a row count, defaulting to 5000 rows -- a few hundred kilobytes of JSON.
4
+ # Out: one large file in the run's scratch space, a csv beside it, and a small summary read
5
+ # back out of the first.
6
+ #
7
+ # Two different caps get confused with each other, and only one of them is a decision this
8
+ # document makes:
9
+ #
10
+ # inline_artifact_max an instance setting, 16KB by default. Every successful attempt
11
+ # keeps its whole structured output, whatever its size, so
12
+ # ${steps.x.output.*} never pays a storage round trip. What the cap
13
+ # governs is where the artifact copy of that output lives: at or
14
+ # below it, inlined in the attempt's reference row; above it,
15
+ # streamed to the run's scratch prefix with the row holding the URI
16
+ # instead. Nothing in a document says any of this, and the file it
17
+ # leaves behind is named after the attempt rather than after
18
+ # anything a person was looking for.
19
+ # max_size what this document says, on the storage.read below. A value comes
20
+ # back into the run only through a read, and a read holds what it
21
+ # reads, so the step names the ceiling it will hold and an object
22
+ # above it is refused rather than truncated.
23
+ #
24
+ # So the write is not an optimisation of the threshold, it is the alternative to relying on
25
+ # it: a result worth keeping is a result named, written and read back deliberately, and a
26
+ # result nobody names a file for is one the engine spills for you and nobody finds.
27
+ #
28
+ # Five hops, and what each one hands on:
29
+ # generate transform.jq, the big list. Its output is far past the threshold, so this is
30
+ # the attempt whose artifact copy the engine spills for you.
31
+ # store storage.write, the same value under a name a person can look for.
32
+ # as_csv convert.std, URI to URI. The codec never holds the list in either direction.
33
+ # reread storage.read, the file back as a value, bounded by max_size.
34
+ # summary transform.jq, a handful of numbers -- the shape that belongs in a step output,
35
+ # which is read by a person on a run page and by a reference in another step's
36
+ # config, and neither of those wants a megabyte.
37
+ #
38
+ # To change it: -p rows=200000 makes the file tens of megabytes. Past 32mb the reread is
39
+ # refused naming the object and the cap, and raising max_size is the deliberate act that
40
+ # allows it.
41
+ #
42
+ # dg run --local examples/recipes/large-output-to-storage.yaml
43
+ # dg run --local examples/recipes/large-output-to-storage.yaml -p rows=50000
44
+
45
+ format: dirigent/v1
46
+ kind: pipeline
47
+ code: large-output-to-storage
48
+ name: A large result through storage
49
+ description: Generate a payload far past the inline artifact threshold, write it to storage under a name of its own, and keep only a bounded summary in a step output.
50
+
51
+ tags: [recipes, storage, transform]
52
+
53
+ requires:
54
+ blocks:
55
+ - transform.jq
56
+ - storage.write
57
+ - convert.std
58
+ - storage.read
59
+
60
+ params:
61
+ type: object
62
+ properties:
63
+ rows:
64
+ type: integer
65
+ minimum: 1
66
+ default: 5000
67
+ description: How many rows are generated; 5000 is a few hundred kilobytes of JSON.
68
+ day:
69
+ type: string
70
+ format: date
71
+ default: "2026-01-01"
72
+ description: The day the file is named for.
73
+
74
+ steps:
75
+ generate:
76
+ block: transform.jq
77
+ config:
78
+ input:
79
+ rows: ${params.rows}
80
+ day: ${params.day}
81
+ program: |
82
+ . as {$rows, $day}
83
+ | [range($rows)
84
+ | {id: "row-\(.)",
85
+ day: $day,
86
+ station: "st-\(. % 250)",
87
+ celsius: ((. % 37) - 12),
88
+ note: "generated for the storage recipe"}]
89
+
90
+ store:
91
+ block: storage.write
92
+ depends_on: [generate]
93
+ config:
94
+ target: ${run.scratch}/large-${params.day}.json
95
+ value: ${steps.generate.output.value}
96
+
97
+ as_csv:
98
+ block: convert.std
99
+ depends_on: [store]
100
+ config:
101
+ source: ${steps.store.output.uri}
102
+ target: ${run.scratch}/large-${params.day}.csv
103
+ from: json
104
+ to: csv
105
+
106
+ reread:
107
+ block: storage.read
108
+ depends_on: [store, as_csv]
109
+ config:
110
+ source: ${steps.store.output.uri}
111
+ # Written out here because this is the step where a bigger -p rows would meet it.
112
+ max_size: 32mb
113
+
114
+ summary:
115
+ block: transform.jq
116
+ depends_on: [reread]
117
+ config:
118
+ input: ${steps.reread.output.value}
119
+ program: |
120
+ {rows: length,
121
+ stations: ([.[].station] | unique | length),
122
+ mean_celsius: ([.[].celsius] | add / length)}
123
+
124
+ sizes:
125
+ block: transform.jq
126
+ depends_on: [store, as_csv, summary]
127
+ config:
128
+ input:
129
+ json_bytes: ${steps.store.output.bytes_written}
130
+ csv_bytes: ${steps.as_csv.output.bytes_written}
131
+ summary: ${steps.summary.output.value}
132
+ program: |
133
+ . + {inline_threshold_bytes: 16384,
134
+ json_over_threshold: (.json_bytes > 16384)}
@@ -0,0 +1,96 @@
1
+ # Enrich every row from a lookup table, and see why the table cannot live in the map.
2
+ #
3
+ # In: a list of readings, and a small dimension table of stations.
4
+ # Out: one enriched row per reading, same length as the input, each carrying the station's
5
+ # name and region and a derived fahrenheit column.
6
+ #
7
+ # A map.jq program is handed one element and nothing else. There is no second input, no
8
+ # --arg, and no way for the program to reach another step's output, so the lookup table has
9
+ # to arrive attached to the elements. That attaching is a whole-value job, and it is the
10
+ # transform.jq step above the map: it indexes the dimension once and merges the matched
11
+ # fields into each row.
12
+ #
13
+ # Splicing the table into the program text with a ${...} would work exactly once, and would
14
+ # cost the thing that makes a jq step cheap: a program with a reference in it cannot be
15
+ # compiled when the document is applied, so a syntax error would wait for the run.
16
+ #
17
+ # What is left for map.jq is the per-element arithmetic, and there the verb is worth having.
18
+ # The frame calls the program once per element and asserts the length afterwards, so this
19
+ # step cannot drop a row or invent one, and a program that fails does it naming the element
20
+ # by its 0-based index instead of failing the whole batch anonymously.
21
+ #
22
+ # To change it: -p require_match=true makes the enrichment refuse a reading whose station is
23
+ # not in the dimension. That run fails on purpose, with "element 2: no station row for
24
+ # st-9" -- the loud alternative to the null-filled row the default produces.
25
+ #
26
+ # dg run --local examples/recipes/map-enrich-with-lookup.yaml
27
+ # dg run --local examples/recipes/map-enrich-with-lookup.yaml -p require_match=true
28
+
29
+ format: dirigent/v1
30
+ kind: pipeline
31
+ code: map-enrich-with-lookup
32
+ name: Enrich rows from a lookup table
33
+ description: Merge a dimension table into every row with a whole-value step, then compute the per-row columns with map.jq.
34
+
35
+ tags: [recipes, transform, map]
36
+
37
+ requires:
38
+ blocks:
39
+ - value.const
40
+ - transform.jq
41
+ - map.jq
42
+
43
+ params:
44
+ type: object
45
+ properties:
46
+ require_match:
47
+ type: boolean
48
+ default: false
49
+ description: Fail the element whose station is in no dimension row, rather than filling nulls.
50
+
51
+ steps:
52
+ stations:
53
+ block: value.const
54
+ config:
55
+ value:
56
+ - { id: st-1, name: Harbour, region: east }
57
+ - { id: st-2, name: Ridge, region: west }
58
+
59
+ readings:
60
+ block: value.const
61
+ config:
62
+ value:
63
+ - { station: st-1, celsius: 4 }
64
+ - { station: st-2, celsius: -1 }
65
+ - { station: st-9, celsius: 21 }
66
+
67
+ attach:
68
+ block: transform.jq
69
+ depends_on: [stations, readings]
70
+ config:
71
+ input:
72
+ stations: ${steps.stations.output.value}
73
+ readings: ${steps.readings.output.value}
74
+ require_match: ${params.require_match}
75
+ program: |
76
+ . as {$stations, $readings, $require_match}
77
+ | ($stations | INDEX(.id)) as $by_id
78
+ | [$readings[]
79
+ | . + {name: ($by_id[.station].name // null),
80
+ region: ($by_id[.station].region // null),
81
+ require_match: $require_match}]
82
+
83
+ enrich:
84
+ block: map.jq
85
+ depends_on: [attach]
86
+ config:
87
+ input: ${steps.attach.output.value}
88
+ # error() fails the element, and the frame reports which index it was, so a run that
89
+ # refuses an unmatched reading still says which one.
90
+ program: |
91
+ if .require_match and .name == null
92
+ then error("no station row for \(.station)")
93
+ else {station, name, region,
94
+ celsius,
95
+ fahrenheit: (.celsius * 9 / 5 + 32 | round)}
96
+ end
@@ -0,0 +1,130 @@
1
+ # Records written as parquet, and read back to prove what landed.
2
+ #
3
+ # In: a list of typed records, built inline.
4
+ # Out: a parquet file in the run's scratch space, and the same records read back out of it.
5
+ #
6
+ # A conversion is a storage-object operation, like storage.copy: it reads one URI, writes
7
+ # another, and never holds the records as a value. So a value in the run is written before
8
+ # it is converted, and a result the run wants to look at is read back after -- which is what
9
+ # the first and last hops here are. Parquet makes that concrete, being bytes rather than
10
+ # text: there has never been an inline spelling for it.
11
+ #
12
+ # Six hops, and what each one hands on:
13
+ # rows value.const, the records a real pipeline would have fetched.
14
+ # stored storage.write, that value as a JSON object. Output: uri, bytes_written.
15
+ # as_ndjson convert.std, json to ndjson. One record per line, no array to hold, which
16
+ # is the shape a stream of records already has.
17
+ # write convert.arrow, ndjson to parquet. Output: source, target, bytes_written.
18
+ # read_back convert.arrow, parquet to json, into a file of its own.
19
+ # landed storage.read, that file as a value, which is what the summary counts.
20
+ #
21
+ # convert.arrow infers one type per column from the records: boolean, int64, float64, or
22
+ # string. A column whose rows disagree is refused naming the column rather than coerced, and
23
+ # integers and floats unify to a float column because JSON calls both a number.
24
+ #
25
+ # The read back is not decoration. It is how a run says what actually landed rather than
26
+ # what was intended: the file's own bytes are decoded, and the row count and the columns
27
+ # come from the file, so a truncated or half-written object shows up here as a failure
28
+ # instead of downstream tomorrow.
29
+ #
30
+ # convert.arrow ships in dirigent-parquet, on pyarrow, where convert.std deliberately stands
31
+ # on the standard library alone. Installing that pack is what puts the block in the catalog,
32
+ # and naming it in requires.blocks is what refuses this document on an instance without it.
33
+ #
34
+ # To change it: point the targets at s3://<bucket>/... and the same steps write and read an
35
+ # object store, because the scheme is the only thing that changes.
36
+ #
37
+ # dg run --local examples/recipes/ndjson-to-parquet.yaml
38
+ # dg run --local examples/recipes/ndjson-to-parquet.yaml -p day=2026-02-01
39
+
40
+ format: dirigent/v1
41
+ kind: pipeline
42
+ code: ndjson-to-parquet
43
+ name: Records to parquet and back
44
+ description: Write typed records into the run's scratch space, convert them through ndjson into parquet, and read the file back to see what landed.
45
+
46
+ tags: [recipes, storage, transform, parquet, starter]
47
+
48
+ requires:
49
+ blocks:
50
+ - value.const
51
+ - storage.write
52
+ - convert.std
53
+ - convert.arrow
54
+ - storage.read
55
+ - transform.jq
56
+
57
+ params:
58
+ type: object
59
+ properties:
60
+ day:
61
+ type: string
62
+ format: date
63
+ default: "2026-01-01"
64
+ description: The day the written file is named for.
65
+
66
+ steps:
67
+ rows:
68
+ block: value.const
69
+ config:
70
+ value:
71
+ - { station: st-1, celsius: 4.5, readings: 1440, active: true }
72
+ - { station: st-2, celsius: -1.0, readings: 1200, active: false }
73
+ - { station: st-3, celsius: 7.25, readings: 1439, active: true }
74
+
75
+ stored:
76
+ block: storage.write
77
+ depends_on: [rows]
78
+ config:
79
+ target: ${run.scratch}/readings-${params.day}.json
80
+ value: ${steps.rows.output.value}
81
+
82
+ as_ndjson:
83
+ block: convert.std
84
+ depends_on: [stored]
85
+ config:
86
+ source: ${steps.stored.output.uri}
87
+ target: ${run.scratch}/readings-${params.day}.ndjson
88
+ from: json
89
+ to: ndjson
90
+
91
+ write:
92
+ block: convert.arrow
93
+ depends_on: [as_ndjson]
94
+ config:
95
+ source: ${steps.as_ndjson.output.target}
96
+ target: ${run.scratch}/readings-${params.day}.parquet
97
+ from: ndjson
98
+ to: parquet
99
+
100
+ read_back:
101
+ block: convert.arrow
102
+ depends_on: [write]
103
+ config:
104
+ # The URI the write reported, rather than the same expression written twice: one place
105
+ # decides where the file went.
106
+ source: ${steps.write.output.target}
107
+ target: ${run.scratch}/readings-${params.day}-back.json
108
+ from: parquet
109
+ to: json
110
+
111
+ landed:
112
+ block: storage.read
113
+ depends_on: [read_back]
114
+ config:
115
+ source: ${steps.read_back.output.target}
116
+ max_size: 1mb
117
+
118
+ summary:
119
+ block: transform.jq
120
+ depends_on: [write, landed]
121
+ config:
122
+ input:
123
+ uri: ${steps.write.output.target}
124
+ file_bytes: ${steps.write.output.bytes_written}
125
+ records: ${steps.landed.output.value}
126
+ program: |
127
+ {uri, file_bytes,
128
+ rows: (.records | length),
129
+ columns: (.records[0] | keys),
130
+ records}
@@ -0,0 +1,115 @@
1
+ # Fetching several pages at once, with the page numbers as a literal list.
2
+ #
3
+ # In: a list of page numbers, as a parameter.
4
+ # Out: one HTTP call per page, each with its own status and retry budget, and a fan-in step
5
+ # that stitches the pages back into one list.
6
+ #
7
+ # for_each is expanded when the run is created, so it reads params.*, item and run.* -- and
8
+ # never another step's output. That is the constraint this recipe is built around, and it
9
+ # cuts both ways:
10
+ #
11
+ # What you get. The item grid exists from the moment the run is visible, so a person
12
+ # watching sees six pages and their statuses rather than a step that might turn into six
13
+ # later. Items run concurrently, each with its own status, error and retry budget, so one
14
+ # slow page does not serialise the rest and one failed page is recorded against itself.
15
+ # What you give up. A fan-out whose cardinality depends on what an earlier call returned
16
+ # is not expressible: this is why the pages are a literal list. When the count really is
17
+ # unknown, either the first call reports it and a second pipeline is run with it as a
18
+ # parameter, or the pipeline walks the cursor one call at a time and gives up the
19
+ # concurrency.
20
+ #
21
+ # items decides what one bad page means. Under the default, fail_fast, any failed item fails
22
+ # the step. Under continue, the step succeeds as long as one item did, the failures are
23
+ # recorded against their own items, and the run reports completed_with_errors -- which is
24
+ # the setting for a report that is worth producing from five pages out of six, and the wrong
25
+ # setting for a load that must be whole.
26
+ #
27
+ # The fan-in reads ${steps.pages.output} -- the list of the items' outputs, in item order --
28
+ # rather than .output.value, which is a field of one item's output and not of the list. Only
29
+ # items that succeeded contribute.
30
+ #
31
+ # https://postman-echo.com/get echoes the query it was given, so each item's response
32
+ # carries its own page number and the stitching is visibly per-page.
33
+ #
34
+ # To change it: -p pages='[1,2]' fetches fewer; -p page_size=10 changes what each call asks
35
+ # for. In a real pipeline the pages usually come from an earlier run's count.
36
+ #
37
+ # dg run --local examples/recipes/pagination-by-fan-out.yaml
38
+ # dg run --local examples/recipes/pagination-by-fan-out.yaml -p pages='[1,2]'
39
+
40
+ format: dirigent/v1
41
+ kind: pipeline
42
+ code: pagination-by-fan-out
43
+ name: Pagination as a fan-out
44
+ description: Fetch a literal list of pages concurrently with for_each, each item its own attempt, then stitch the pages into one list.
45
+
46
+ tags: [recipes, http, transform, graph]
47
+
48
+ requires:
49
+ blocks:
50
+ - http.request
51
+ - transform.jq
52
+
53
+ params:
54
+ type: object
55
+ properties:
56
+ pages:
57
+ type: array
58
+ default: [1, 2, 3, 4]
59
+ items:
60
+ type: integer
61
+ minimum: 1
62
+ description: The page numbers to fetch; one item, and one call, per element.
63
+ page_size:
64
+ type: integer
65
+ minimum: 1
66
+ default: 25
67
+ description: What each call asks for.
68
+ day:
69
+ type: string
70
+ format: date
71
+ default: "2026-01-01"
72
+ description: The day every page is fetched for.
73
+
74
+ steps:
75
+ pages:
76
+ block: http.request
77
+ for_each: ${params.pages}
78
+ # fail_fast: a partial read is not a page of results, it is a gap. A report that is
79
+ # worth producing from most of its pages says items: continue instead.
80
+ items: fail_fast
81
+ # Each item gets this budget of its own, so one flaky page is retried and the rest are
82
+ # untouched.
83
+ retry:
84
+ max_attempts: 3
85
+ backoff: 1s
86
+ config:
87
+ url: https://postman-echo.com/get
88
+ method: GET
89
+ query:
90
+ page: ${item}
91
+ page_size: ${params.page_size}
92
+ day: ${params.day}
93
+ timeout: 20s
94
+
95
+ stitch:
96
+ block: transform.jq
97
+ depends_on: [pages]
98
+ config:
99
+ # The fan-out's output is the list of its items' outputs, in item order.
100
+ input: ${steps.pages.output}
101
+ program: |
102
+ . as $items
103
+ | [$items[]
104
+ | {page: (.body.args.page | tonumber),
105
+ status: .status,
106
+ duration_ms: .duration_ms,
107
+ url: .body.url}] as $fetched
108
+ | {pages: ($fetched | length),
109
+ # In item order, which is the order the pages were asked for and not the order
110
+ # they came back in.
111
+ page_numbers: [$fetched[].page],
112
+ ordered: ([$fetched[].page] == ([$fetched[].page] | sort)),
113
+ slowest_ms: ([$fetched[].duration_ms] | max),
114
+ total_ms: ([$fetched[].duration_ms] | add),
115
+ fetched: $fetched}
@@ -0,0 +1,163 @@
1
+ # What survives a round trip through parquet, and what a csv would have flattened instead.
2
+ #
3
+ # In: one record carrying every JSON scalar -- a string, an integer, a float, a boolean, a
4
+ # null, a numeric string, and a column whose values are integers and floats together.
5
+ # Out: {before, after_parquet, after_csv, kinds, changed}, so the two formats can be
6
+ # compared column by column in one output.
7
+ #
8
+ # Parquet has a schema and csv does not, and that single difference is the whole recipe:
9
+ #
10
+ # integer survives as an int64.
11
+ # float survives as a float64.
12
+ # boolean survives as a boolean, not as "true".
13
+ # null survives as a null, in a column that stays typed.
14
+ # numeric string survives as a string. "007" comes back "007", where a csv reader is
15
+ # left guessing and a spreadsheet would have eaten the leading zeros.
16
+ # mixed number integers and floats unify to one float column, because JSON calls both
17
+ # a number. The values still come back equal, since JSON has one number
18
+ # spelling, so `changed.parquet` is empty -- the loss is in the column's
19
+ # type, which a later reader sees and a jq comparison cannot.
20
+ #
21
+ # The csv leg is the control. Every cell in a csv is text on the way out and text on the way
22
+ # back, so after it every value is a string and the booleans, the numbers and the nulls are
23
+ # indistinguishable from the words that spell them.
24
+ #
25
+ # Both legs are four hops, and they are four rather than two because a conversion is a
26
+ # storage-object operation: it reads one URI and writes another, and never carries a value.
27
+ # The record is written once, converted out, converted back, and read into the run again --
28
+ # so `before` is the value this document wrote and `after` is what came off a disk.
29
+ #
30
+ # The comparison is computed rather than described: `changed` lists the fields whose value
31
+ # is not identical after the round trip, so this recipe stays honest if a codec ever changes
32
+ # under it.
33
+ #
34
+ # A column whose rows disagree in a way that cannot unify is refused by name at the write
35
+ # rather than coerced. Adding {"code": 7} to one of the rows below is the fastest way to see
36
+ # it: "column 'code' holds both integer and string values, and a parquet column carries one
37
+ # type; reshape it before converting".
38
+ #
39
+ # dg run --local examples/recipes/parquet-round-trip-types.yaml
40
+
41
+ format: dirigent/v1
42
+ kind: pipeline
43
+ code: parquet-round-trip-types
44
+ name: What survives a parquet round trip
45
+ description: Write one record of every JSON scalar, convert it out to parquet and back, and compare it against the same record round-tripped through csv.
46
+
47
+ tags: [recipes, storage, transform, parquet]
48
+
49
+ requires:
50
+ blocks:
51
+ - value.const
52
+ - storage.write
53
+ - convert.std
54
+ - convert.arrow
55
+ - storage.read
56
+ - transform.jq
57
+
58
+ params:
59
+ type: object
60
+ properties:
61
+ day:
62
+ type: string
63
+ format: date
64
+ default: "2026-01-01"
65
+ description: The day the written file is named for.
66
+
67
+ steps:
68
+ rows:
69
+ block: value.const
70
+ config:
71
+ value:
72
+ - text: Harbour
73
+ whole: 1440
74
+ fractional: 4.5
75
+ flag: true
76
+ absent: null
77
+ code: "007"
78
+ mixed: 1
79
+ - text: Ridge
80
+ whole: 1200
81
+ fractional: -1.25
82
+ flag: false
83
+ absent: null
84
+ code: "042"
85
+ # The other value in this column is an integer, so the column unifies to float.
86
+ mixed: 2.5
87
+
88
+ stored:
89
+ block: storage.write
90
+ depends_on: [rows]
91
+ config:
92
+ target: ${run.scratch}/types-${params.day}.json
93
+ value: ${steps.rows.output.value}
94
+
95
+ write_parquet:
96
+ block: convert.arrow
97
+ depends_on: [stored]
98
+ config:
99
+ source: ${steps.stored.output.uri}
100
+ target: ${run.scratch}/types-${params.day}.parquet
101
+ from: json
102
+ to: parquet
103
+
104
+ back_from_parquet:
105
+ block: convert.arrow
106
+ depends_on: [write_parquet]
107
+ config:
108
+ source: ${steps.write_parquet.output.target}
109
+ target: ${run.scratch}/types-${params.day}-from-parquet.json
110
+ from: parquet
111
+ to: json
112
+
113
+ read_parquet:
114
+ block: storage.read
115
+ depends_on: [back_from_parquet]
116
+ config:
117
+ source: ${steps.back_from_parquet.output.target}
118
+ max_size: 1mb
119
+
120
+ write_csv:
121
+ block: convert.std
122
+ depends_on: [stored]
123
+ config:
124
+ source: ${steps.stored.output.uri}
125
+ target: ${run.scratch}/types-${params.day}.csv
126
+ from: json
127
+ to: csv
128
+
129
+ back_from_csv:
130
+ block: convert.std
131
+ depends_on: [write_csv]
132
+ config:
133
+ source: ${steps.write_csv.output.target}
134
+ target: ${run.scratch}/types-${params.day}-from-csv.json
135
+ from: csv
136
+ to: json
137
+
138
+ read_csv:
139
+ block: storage.read
140
+ depends_on: [back_from_csv]
141
+ config:
142
+ source: ${steps.back_from_csv.output.target}
143
+ max_size: 1mb
144
+
145
+ compare:
146
+ block: transform.jq
147
+ depends_on: [rows, read_parquet, read_csv]
148
+ config:
149
+ input:
150
+ before: ${steps.rows.output.value}
151
+ after_parquet: ${steps.read_parquet.output.value}
152
+ after_csv: ${steps.read_csv.output.value}
153
+ program: |
154
+ def kinds: map_values(type);
155
+ . as {$before, $after_parquet, $after_csv}
156
+ | {before: $before[0], after_parquet: $after_parquet[0], after_csv: $after_csv[0],
157
+ kinds: {before: ($before[0] | kinds),
158
+ after_parquet: ($after_parquet[0] | kinds),
159
+ after_csv: ($after_csv[0] | kinds)},
160
+ changed: {parquet: [$before[0] | to_entries[]
161
+ | select($after_parquet[0][.key] != .value) | .key],
162
+ csv: [$before[0] | to_entries[]
163
+ | select($after_csv[0][.key] != .value) | .key]}}