dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,70 @@
1
+ # The analytics reshape: a flat list in, one row per group out, and a total beside them.
2
+ #
3
+ # group_by is the whole of it. jq sorts the list by the key and hands back a list of lists,
4
+ # one per distinct key, so an aggregate is `map(.rainfall_mm) | add` over each of those
5
+ # lists and the group's key is `.[0].region`. The groups come out in sorted key order --
6
+ # east, north, west -- because group_by sorts before it groups.
7
+ #
8
+ # The total is computed from the original list rather than by adding the group totals up.
9
+ # Both are correct here; only one stays correct when the groups are filtered.
10
+ #
11
+ # The program contains no ${...} reference, so jq compiles it when the document is applied
12
+ # and a typo is an issue at steps.by_region.config rather than a run that fails tonight. A
13
+ # config that carries a reference cannot be compiled early, which is why the data is the
14
+ # input and never spliced into the program.
15
+ #
16
+ # The readings are written inline so the example needs no network. In a real pipeline this is
17
+ # a reference to an upstream step's output: an http.request's body, or a storage.read's value.
18
+ #
19
+ # What it prints, and the arithmetic to check it against:
20
+ #
21
+ # east 2 stations, 12 + 8 = 20mm, mean 10
22
+ # north 3 stations, 30 + 10 + 5 = 45mm, mean 15
23
+ # west 1 station, 7 = 7mm, mean 7
24
+ # total 6 stations, 20 + 45 + 7 = 72mm, mean 12
25
+ #
26
+ # dg run --local examples/transform/jq-group-and-aggregate.yaml
27
+
28
+ format: dirigent/v1
29
+ kind: pipeline
30
+ code: jq-group-and-aggregate
31
+ name: Group and aggregate with jq
32
+ description: Group a flat list of readings by region, aggregate each group, and total them.
33
+
34
+ tags: [transform]
35
+
36
+ requires:
37
+ blocks:
38
+ - transform.jq
39
+
40
+ steps:
41
+ by_region:
42
+ block: transform.jq
43
+ config:
44
+ input:
45
+ - { station: st-1, region: east, rainfall_mm: 12 }
46
+ - { station: st-2, region: east, rainfall_mm: 8 }
47
+ - { station: st-3, region: north, rainfall_mm: 30 }
48
+ - { station: st-4, region: north, rainfall_mm: 10 }
49
+ - { station: st-5, region: north, rainfall_mm: 5 }
50
+ - { station: st-6, region: west, rainfall_mm: 7 }
51
+ # One output, so the step's value is that one object.
52
+ program: |
53
+ . as $readings
54
+ | {
55
+ regions: [
56
+ $readings
57
+ | group_by(.region)[]
58
+ | {
59
+ region: .[0].region,
60
+ stations: length,
61
+ total_mm: (map(.rainfall_mm) | add),
62
+ mean_mm: (map(.rainfall_mm) | add / length)
63
+ }
64
+ ],
65
+ total: {
66
+ stations: ($readings | length),
67
+ total_mm: ($readings | map(.rainfall_mm) | add),
68
+ mean_mm: ($readings | map(.rainfall_mm) | add / length)
69
+ }
70
+ }
@@ -0,0 +1,98 @@
1
+ # Enriching one dataset from another: two upstream steps, one join, and no reference in
2
+ # any of the three programs.
3
+ #
4
+ # The pattern is the third step's config. Two outputs are composed into a single inline
5
+ # input, keyed by what they are:
6
+ #
7
+ # input:
8
+ # stations: ${steps.stations.output.value}
9
+ # readings: ${steps.readings.output.value}
10
+ #
11
+ # so the program destructures .stations and .readings and never mentions a step. That keeps
12
+ # the references in typed config, where the engine resolves them, and keeps the program a
13
+ # constant string that jq compiles when the document is applied. A program with a ${...}
14
+ # spliced into it could only be compiled on the worker, on the first run.
15
+ #
16
+ # INDEX(.id) turns the stations list into an object keyed by id, which is the lookup table:
17
+ # one pass to build it, then constant-time reads, instead of scanning the list per reading.
18
+ #
19
+ # st-9 has no station record, and the join keeps its row with a null name rather than
20
+ # dropping it -- an outer join, not an inner one. `// null` is what says so. jq already
21
+ # yields null for a missing key, so the operator changes no output; it is written because a
22
+ # reader should see the decision in the program rather than infer it from jq's semantics.
23
+ # A pipeline that wants unmatched readings gone writes `select(...)` and means it.
24
+ #
25
+ # The two upstream programs are projections over inline parameter defaults, so the example
26
+ # needs no network. In a real pipeline they are two http.request steps against two systems,
27
+ # and this is the step that puts the two answers together.
28
+ #
29
+ # What it prints: three readings, st-1 and st-3 carrying a station_name and a region, st-9
30
+ # carrying null for both.
31
+ #
32
+ # dg run --local examples/transform/jq-join-two-sources.yaml
33
+
34
+ format: dirigent/v1
35
+ kind: pipeline
36
+ code: jq-join-two-sources
37
+ name: Join two sources with jq
38
+ description: Enrich a list of readings with the station records they refer to, keeping unmatched rows.
39
+
40
+ tags: [transform, graph]
41
+
42
+ requires:
43
+ blocks:
44
+ - transform.jq
45
+
46
+ params:
47
+ type: object
48
+ additionalProperties: false
49
+ properties:
50
+ stations:
51
+ type: array
52
+ description: The station records to enrich from.
53
+ items:
54
+ type: object
55
+ default:
56
+ - { id: st-1, name: Bo Central, region: east }
57
+ - { id: st-2, name: Kenema North, region: east }
58
+ - { id: st-3, name: Makeni West, region: west }
59
+ readings:
60
+ type: array
61
+ description: The readings to enrich, one of which names an unknown station.
62
+ items:
63
+ type: object
64
+ default:
65
+ - { station: st-1, celsius: 4.5 }
66
+ - { station: st-3, celsius: 1.2 }
67
+ - { station: st-9, celsius: -2.0 }
68
+
69
+ steps:
70
+ stations:
71
+ block: transform.jq
72
+ config:
73
+ input: ${params.stations}
74
+ # A projection, standing in for whatever the station system actually returns.
75
+ program: |
76
+ [.[] | {id, name, region}]
77
+
78
+ readings:
79
+ block: transform.jq
80
+ config:
81
+ input: ${params.readings}
82
+ program: |
83
+ [.[] | {station, celsius}]
84
+
85
+ join:
86
+ block: transform.jq
87
+ depends_on: [stations, readings]
88
+ config:
89
+ # The two outputs, composed into one value under names the program knows.
90
+ input:
91
+ stations: ${steps.stations.output.value}
92
+ readings: ${steps.readings.output.value}
93
+ program: |
94
+ . as {$stations, $readings}
95
+ | ($stations | INDEX(.id)) as $by_id
96
+ | [$readings[]
97
+ | . + {station_name: ($by_id[.station].name // null),
98
+ region: ($by_id[.station].region // null)}]
@@ -0,0 +1,91 @@
1
+ # Reshaping data between two steps, with no subprocess and no allowlist entry.
2
+ #
3
+ # transform.jq runs a jq program over one whole value. The program is compiled when the
4
+ # document is applied, so a syntax error is an issue at steps.<name>.config beside every
5
+ # other one the document has, rather than a run that fails at three in the morning. jq opens
6
+ # no file and no socket, so the engine executes nothing on the worker: unlike every example
7
+ # that reaches for shell.run to do this, this one needs no --enable-unsafe and no network.
8
+ #
9
+ # The shape is the one a reshape usually has. reshape pulls the active readings out of a
10
+ # payload and flattens them; per_region is mapped over the regions and reads that list once
11
+ # per element; summary reads the whole fan-out back as one value. Note where the per-item
12
+ # value goes: the item is data in config.input, not text spliced into the program, so the
13
+ # program stays a program and the data stays data.
14
+ #
15
+ # A jq program is a stream of outputs, and the step's output follows it: one output is the
16
+ # value, several are the list of them, and none at all is refused, because a step has to
17
+ # produce something for the next one to read.
18
+ #
19
+ # The input is written inline so the example needs no network. In a real pipeline it is a
20
+ # reference to an upstream step's output: an http.request's body, or the value a storage.read
21
+ # brought in. The result travels the same way, and reaches a file through a storage.write.
22
+ #
23
+ # dg run --local examples/transform/jq-reshape.yaml
24
+ # dg run --local examples/transform/jq-reshape.yaml -p regions='["east"]'
25
+
26
+ format: dirigent/v1
27
+ kind: pipeline
28
+ code: jq-reshape
29
+ name: Reshape with jq
30
+ description: Reshape a reading payload with jq, then map the result over one region each.
31
+
32
+ tags: [transform]
33
+
34
+ requires:
35
+ blocks:
36
+ - transform.jq
37
+
38
+ params:
39
+ type: object
40
+ properties:
41
+ regions:
42
+ type: array
43
+ default: [east, west, north]
44
+ items:
45
+ type: string
46
+
47
+ steps:
48
+ reshape:
49
+ block: transform.jq
50
+ config:
51
+ input:
52
+ generated_at: "2026-01-01T06:00:00Z"
53
+ readings:
54
+ - { station: st-1, region: east, status: active, celsius: 4 }
55
+ - { station: st-2, region: east, status: retired, celsius: null }
56
+ - { station: st-3, region: west, status: active, celsius: 1.2 }
57
+ - { station: st-4, region: north, status: active, celsius: -3.4 }
58
+ # A second active east reading, so mean_celsius is (4 + 8) / 2 = 6, an average
59
+ # of two rather than a passthrough of one.
60
+ - { station: st-5, region: east, status: active, celsius: 8 }
61
+ # One output, so the step's output value is that one array.
62
+ program: |
63
+ [.readings[] | select(.status == "active") | {station, region, celsius}]
64
+
65
+ per_region:
66
+ block: transform.jq
67
+ depends_on: [reshape]
68
+ # for_each is expanded when the run is created, so it reads params and not a step's
69
+ # output. The reshaped list is read in config, where an output reference belongs.
70
+ for_each: "${params.regions}"
71
+ config:
72
+ input:
73
+ region: "${item}"
74
+ rows: "${steps.reshape.output.value}"
75
+ program: |
76
+ . as {$region, $rows}
77
+ | {region: $region,
78
+ stations: [$rows[] | select(.region == $region) | .station],
79
+ mean_celsius: ([$rows[] | select(.region == $region) | .celsius]
80
+ | if length > 0 then add / length else null end)}
81
+
82
+ summary:
83
+ block: transform.jq
84
+ depends_on: [per_region]
85
+ config:
86
+ # A fan-out step's output is the list of its items' outputs, in item order, and a
87
+ # transform's output is an object carrying value, so the list is unwrapped once here.
88
+ input: "${steps.per_region.output}"
89
+ program: |
90
+ [.[] | .value | select(.stations | length > 0)]
91
+ | {regions_with_data: length, stations: (map(.stations) | flatten | sort)}
@@ -0,0 +1,112 @@
1
+ # A value leaving the run and coming back: storage.write out, storage.read in.
2
+ #
3
+ # A transform works on a value and answers with one, so a list that should survive as a
4
+ # file leaves through storage.write and returns through storage.read. Those two blocks are
5
+ # the only doors, and this document walks through both of them: generate builds a list,
6
+ # store puts it in the run's scratch space, reload reads it back as a value, cold keeps the
7
+ # readings below zero, keep writes those out as their own file, and summary folds what cold
8
+ # produced down to the handful of fields the run answers with.
9
+ #
10
+ # What each hop hands the next:
11
+ #
12
+ # - a transform's output is value, which is what the next transform's input reads
13
+ # - a write's output is uri, bytes_written and content_type, so the read that follows
14
+ # names ${steps.store.output.uri} rather than repeating the path
15
+ # - a read's output is value for the json family and text otherwise, decided by what the
16
+ # object says it is; readings.json says application/json by its extension
17
+ #
18
+ # A read holds the object whole, so max_size bounds it and an object past that is refused
19
+ # rather than truncated. Bytes nobody has to look at move with storage.copy instead.
20
+ #
21
+ # The list is built in jq from range, so the example needs no network and no fixture file.
22
+ # In a real pipeline generate is an http.request and its body goes straight to store.
23
+ #
24
+ # What it prints, and the arithmetic to check it against: celsius is (i % 40) - 10, so ten
25
+ # of every forty stations are below zero, and 2000 stations give 500 cold ones. The coldest
26
+ # is st-0 at -10, and the three regions all appear among them.
27
+ #
28
+ # dg run --local examples/transform/jq-stream-through-storage.yaml
29
+ # dg run --local examples/transform/jq-stream-through-storage.yaml -p stations=200
30
+
31
+ format: dirigent/v1
32
+ kind: pipeline
33
+ code: jq-stream-through-storage
34
+ name: A value through storage
35
+ description: Write a generated list to storage, read it back as a value, and write the smaller result out the same way.
36
+
37
+ tags: [transform, storage]
38
+
39
+ requires:
40
+ blocks:
41
+ - transform.jq
42
+ - storage.read
43
+ - storage.write
44
+
45
+ params:
46
+ type: object
47
+ additionalProperties: false
48
+ properties:
49
+ stations:
50
+ type: integer
51
+ minimum: 1
52
+ maximum: 100000
53
+ default: 2000
54
+ description: How many station readings to generate.
55
+
56
+ steps:
57
+ generate:
58
+ block: transform.jq
59
+ config:
60
+ input:
61
+ count: ${params.stations}
62
+ # About 100KB at the default count.
63
+ program: |
64
+ [range(.count) as $i
65
+ | {station: "st-\($i)",
66
+ region: ["east", "west", "north"][$i % 3],
67
+ celsius: ($i % 40) - 10}]
68
+
69
+ store:
70
+ block: storage.write
71
+ depends_on: [generate]
72
+ config:
73
+ target: ${run.scratch}/readings.json
74
+ value: ${steps.generate.output.value}
75
+
76
+ reload:
77
+ block: storage.read
78
+ depends_on: [store]
79
+ config:
80
+ source: ${steps.store.output.uri}
81
+ # The default is 1mb, and the generated list is well under it at every station count
82
+ # this document allows.
83
+ max_size: 8mb
84
+
85
+ cold:
86
+ block: transform.jq
87
+ depends_on: [reload]
88
+ config:
89
+ # readings.json is application/json, so the read decoded it and the value is the list
90
+ # itself rather than its text.
91
+ input: ${steps.reload.output.value}
92
+ program: |
93
+ [.[] | select(.celsius < 0) | {station, region, celsius}]
94
+
95
+ keep:
96
+ block: storage.write
97
+ depends_on: [cold]
98
+ config:
99
+ target: ${run.scratch}/cold.json
100
+ value: ${steps.cold.output.value}
101
+
102
+ summary:
103
+ block: transform.jq
104
+ depends_on: [cold]
105
+ config:
106
+ # No second read: the cold list is already a value in this run, and a write is for
107
+ # what leaves it. This output is small, so it inlines, and it is the run's answer.
108
+ input: ${steps.cold.output.value}
109
+ program: |
110
+ {cold_stations: length,
111
+ coldest: (min_by(.celsius) | .station),
112
+ regions: (map(.region) | unique)}
@@ -0,0 +1,56 @@
1
+ # One record per line, which is what a consumer that streams wants.
2
+ #
3
+ # ndjson is the spelling for a sequence that is appended to or read without parsing the
4
+ # whole: each line is one complete JSON record. This writes an array to storage, converts
5
+ # that object to ndjson and back, and reads the result in again -- and the round trip is
6
+ # the point, because the three formats say the same thing, so converting is re-spelling,
7
+ # never reshaping.
8
+ #
9
+ # A conversion is a storage-object operation: source uri in, target uri out, no value in
10
+ # between. storage.write is what turns the records this run holds into an object to convert,
11
+ # and storage.read is what brings the last one back as a value the run can show.
12
+ #
13
+ # dg run --local examples/transform/ndjson-round-trip.yaml --keep
14
+
15
+ format: dirigent/v1
16
+ kind: pipeline
17
+ code: ndjson-round-trip
18
+ name: One record per line
19
+ description: Re-spell a stored JSON array as ndjson, convert it back, and read the result in as a value.
20
+
21
+ tags: [transform, storage]
22
+
23
+ steps:
24
+ records:
25
+ block: value.const
26
+ config:
27
+ value:
28
+ - { station: st-1, celsius: 4.5 }
29
+ - { station: st-3, celsius: 1.2 }
30
+ store:
31
+ block: storage.write
32
+ depends_on: [records]
33
+ config:
34
+ target: ${run.scratch}/records.json
35
+ value: ${steps.records.output.value}
36
+ as_lines:
37
+ block: convert.std
38
+ depends_on: [store]
39
+ config:
40
+ source: ${steps.store.output.uri}
41
+ target: ${run.scratch}/records.ndjson
42
+ from: json
43
+ to: ndjson
44
+ back:
45
+ block: convert.std
46
+ depends_on: [as_lines]
47
+ config:
48
+ source: ${steps.as_lines.output.target}
49
+ target: ${run.scratch}/round-trip.json
50
+ from: ndjson
51
+ to: json
52
+ read_back:
53
+ block: storage.read
54
+ depends_on: [back]
55
+ config:
56
+ source: ${steps.back.output.target}
@@ -0,0 +1,68 @@
1
+ # Records into parquet and back, where the types survive the trip.
2
+ #
3
+ # Parquet is the typed spelling at rest: where csv turns every value into a string, a
4
+ # parquet column knows it holds an integer or a float, and convert.arrow infers that schema
5
+ # from the records as it writes. This stores a batch as json, converts it to parquet, reads
6
+ # that file back into json, and loads it as a value -- and the numbers come back numbers
7
+ # rather than strings.
8
+ #
9
+ # Every convert step here names two uris and carries nothing: parquet is bytes, and a
10
+ # conversion is a storage-object operation either way. storage.write is the hop that puts
11
+ # the records in storage to begin with, and storage.read the hop that brings the result
12
+ # back where a later step, or a person reading the run, can see it.
13
+ #
14
+ # convert.arrow ships in the dirigent-parquet pack; installing it is what puts the block in
15
+ # the catalog.
16
+ #
17
+ # dg run --local examples/transform/parquet-round-trip.yaml --keep
18
+
19
+ format: dirigent/v1
20
+ kind: pipeline
21
+ code: parquet-round-trip
22
+ name: A parquet round trip
23
+ description: Write records to a parquet file, then read them back with their types intact.
24
+
25
+ tags: [transform, storage]
26
+
27
+ requires:
28
+ blocks:
29
+ - value.const
30
+ - storage.read
31
+ - storage.write
32
+ - convert.arrow
33
+
34
+ steps:
35
+ readings:
36
+ block: value.const
37
+ config:
38
+ value:
39
+ - { station: st-1, region: east, celsius: 4.5, active: true }
40
+ - { station: st-3, region: west, celsius: 1.2, active: true }
41
+ - { station: st-4, region: north, celsius: -3.4, active: false }
42
+ store:
43
+ block: storage.write
44
+ depends_on: [readings]
45
+ config:
46
+ target: ${run.scratch}/readings.json
47
+ value: ${steps.readings.output.value}
48
+ write:
49
+ block: convert.arrow
50
+ depends_on: [store]
51
+ config:
52
+ source: ${steps.store.output.uri}
53
+ target: ${run.scratch}/readings.parquet
54
+ from: json
55
+ to: parquet
56
+ read_back:
57
+ block: convert.arrow
58
+ depends_on: [write]
59
+ config:
60
+ source: ${steps.write.output.target}
61
+ target: ${run.scratch}/round-trip.json
62
+ from: parquet
63
+ to: json
64
+ load:
65
+ block: storage.read
66
+ depends_on: [read_back]
67
+ config:
68
+ source: ${steps.read_back.output.target}
@@ -0,0 +1,142 @@
1
+ # A csv re-encoded once and then reshaped, with no subprocess and no allowlist entry.
2
+ #
3
+ # convert.std is a codec, not a language: it has no program, only a source, a target, a
4
+ # from and a to, and the pair is checked when the document is applied. It converts between
5
+ # json, ndjson and csv, and everything a csv carries is text, so every cell it reads out is
6
+ # a string. That is the first thing this document has to deal with: celsius arrives as
7
+ # "4.5", and the jq program says tonumber where it wants a number. A codec that guessed
8
+ # would not be a codec.
9
+ #
10
+ # The shape is csv in, csv out, and storage is what the two blocks meet over. source writes
11
+ # the inline csv as an object; parse re-encodes it as json; rows reads that json in as a
12
+ # value; active reshapes it with jq; per_region is mapped over the regions; report gathers
13
+ # the fan-out; store puts that array back in storage, and as_csv converts it to csv again.
14
+ #
15
+ # Two things about crossing between the two blocks. A convert step names uris and carries no
16
+ # value at all, so a value reaching one goes through storage.write and a value coming out of
17
+ # one through storage.read. And a csv cell holds one value, so the stations of a region are
18
+ # joined into one before they are converted -- a nested value is refused rather than
19
+ # silently stringified.
20
+ #
21
+ # dg run --local examples/transform/std-convert-fan-out.yaml --keep
22
+ # dg run --local examples/transform/std-convert-fan-out.yaml -p regions='["east"]'
23
+
24
+ format: dirigent/v1
25
+ kind: pipeline
26
+ code: std-convert-fan-out
27
+ name: Convert and fan out
28
+ description: |
29
+ Store an inline csv, re-encode it as json, reshape it with jq, and fan out over the
30
+ regions.
31
+
32
+ Two blocks meet here, and storage is what they meet over:
33
+
34
+ 1. `convert.std` is a **codec** -- a `from` and a `to`, over a source and a target uri
35
+ 2. `transform.jq` is a **language** -- a program, over a value
36
+
37
+ A value reaches a codec through `storage.write` and leaves it through `storage.read`.
38
+
39
+ tags: [transform, graph, storage]
40
+
41
+ requires:
42
+ blocks:
43
+ - convert.std
44
+ - transform.jq
45
+ - storage.read
46
+ - storage.write
47
+
48
+ params:
49
+ type: object
50
+ properties:
51
+ regions:
52
+ type: array
53
+ default: [east, west, north]
54
+ items:
55
+ type: string
56
+
57
+ steps:
58
+ source:
59
+ block: storage.write
60
+ config:
61
+ target: ${run.scratch}/stations.csv
62
+ # Written inline so the example needs no network. In a real pipeline this object is
63
+ # the csv an upstream http.request already put in storage, and parse starts there.
64
+ text: |
65
+ station,region,status,celsius
66
+ st-1,east,active,4.5
67
+ st-2,east,retired,
68
+ st-3,west,active,1.2
69
+ st-4,north,active,-3.4
70
+ st-5,north,active,-1.6
71
+ content_type: text/csv
72
+
73
+ parse:
74
+ block: convert.std
75
+ depends_on: [source]
76
+ config:
77
+ source: ${steps.source.output.uri}
78
+ target: ${run.scratch}/stations.json
79
+ from: csv
80
+ to: json
81
+
82
+ rows:
83
+ block: storage.read
84
+ depends_on: [parse]
85
+ config:
86
+ source: ${steps.parse.output.target}
87
+
88
+ active:
89
+ block: transform.jq
90
+ depends_on: [rows]
91
+ config:
92
+ # Every cell came out of the csv as a string, so the program says tonumber for the one
93
+ # column it wants as a number.
94
+ input: ${steps.rows.output.value}
95
+ program: |
96
+ [.[] | select(.status == "active") | {station, region, celsius: (.celsius | tonumber)}]
97
+
98
+ per_region:
99
+ block: transform.jq
100
+ depends_on: [active]
101
+ # for_each is expanded when the run is created, so it reads params and not a step's
102
+ # output: the regions cannot come out of the csv the run is about to parse.
103
+ for_each: "${params.regions}"
104
+ config:
105
+ input:
106
+ region: "${item}"
107
+ rows: "${steps.active.output.value}"
108
+ program: |
109
+ . as {$region, $rows}
110
+ | [$rows[] | select(.region == $region)] as $here
111
+ | {region: $region,
112
+ stations: ([$here[] | .station] | join(" ")),
113
+ mean_celsius: (if $here == [] then null else ([$here[] | .celsius] | add / length) end)}
114
+
115
+ report:
116
+ block: transform.jq
117
+ depends_on: [per_region]
118
+ config:
119
+ # A fan-out step's output is the list of its items' outputs, in item order, so the
120
+ # object carrying value is unwrapped once here.
121
+ input: "${steps.per_region.output}"
122
+ program: |
123
+ [.[] | .value]
124
+
125
+ store:
126
+ block: storage.write
127
+ depends_on: [report]
128
+ config:
129
+ target: ${run.scratch}/regions.json
130
+ value: ${steps.report.output.value}
131
+
132
+ as_csv:
133
+ block: convert.std
134
+ depends_on: [store]
135
+ config:
136
+ source: ${steps.store.output.uri}
137
+ target: ${run.scratch}/regions.csv
138
+ # The header is the union of the keys the rows have, in first-seen order, and
139
+ # storage.write put them down as canonical JSON, so that order is alphabetical. A null
140
+ # mean -- a region the csv had no active reading for -- is an empty cell.
141
+ from: json
142
+ to: csv