dirigent-examples 0.15.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. dirigent_examples/__init__.py +22 -0
  2. dirigent_examples/py.typed +0 -0
  3. dirigent_examples/shelves/README.md +299 -0
  4. dirigent_examples/shelves/composition/README.md +18 -0
  5. dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
  6. dirigent_examples/shelves/composition/composition-child.yaml +64 -0
  7. dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
  8. dirigent_examples/shelves/connections.yaml +52 -0
  9. dirigent_examples/shelves/demo/README.md +19 -0
  10. dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
  11. dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
  12. dirigent_examples/shelves/demo/requires.yaml +65 -0
  13. dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
  14. dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
  15. dirigent_examples/shelves/docker/README.md +29 -0
  16. dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
  17. dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
  18. dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
  19. dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
  20. dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
  21. dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
  22. dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
  23. dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
  24. dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
  25. dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
  26. dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
  27. dirigent_examples/shelves/execute/README.md +16 -0
  28. dirigent_examples/shelves/execute/long-log.yaml +89 -0
  29. dirigent_examples/shelves/failure/README.md +20 -0
  30. dirigent_examples/shelves/failure/error-handler.yaml +89 -0
  31. dirigent_examples/shelves/failure/optional-step.yaml +82 -0
  32. dirigent_examples/shelves/failure/retries.yaml +75 -0
  33. dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
  34. dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
  35. dirigent_examples/shelves/git/README.md +32 -0
  36. dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
  37. dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
  38. dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
  39. dirigent_examples/shelves/graph/README.md +22 -0
  40. dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
  41. dirigent_examples/shelves/graph/fan-in.yaml +80 -0
  42. dirigent_examples/shelves/graph/fan-out.yaml +66 -0
  43. dirigent_examples/shelves/graph/linear.yaml +66 -0
  44. dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
  45. dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
  46. dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
  47. dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
  48. dirigent_examples/shelves/hello-world.yaml +30 -0
  49. dirigent_examples/shelves/open-data/README.md +67 -0
  50. dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
  51. dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
  52. dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
  53. dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
  54. dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
  55. dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
  56. dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
  57. dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
  58. dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
  59. dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
  60. dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
  61. dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
  62. dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
  63. dirigent_examples/shelves/patterns/README.md +144 -0
  64. dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
  65. dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
  66. dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
  67. dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
  68. dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
  69. dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
  70. dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
  71. dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
  72. dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
  73. dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
  74. dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
  75. dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
  76. dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
  77. dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
  78. dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
  79. dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
  80. dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
  81. dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
  82. dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
  83. dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
  84. dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
  85. dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
  86. dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
  87. dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
  88. dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
  89. dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
  90. dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
  91. dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
  92. dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
  93. dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
  94. dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
  95. dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
  96. dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
  97. dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
  98. dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
  99. dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
  100. dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
  101. dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
  102. dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
  103. dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
  104. dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
  105. dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
  106. dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
  107. dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
  108. dirigent_examples/shelves/python/README.md +31 -0
  109. dirigent_examples/shelves/python/apply_and_run.py +52 -0
  110. dirigent_examples/shelves/python/ci_gate.py +76 -0
  111. dirigent_examples/shelves/python/connections.py +61 -0
  112. dirigent_examples/shelves/python/error_handling.py +84 -0
  113. dirigent_examples/shelves/python/follow_logs.py +39 -0
  114. dirigent_examples/shelves/python/list_and_filter.py +52 -0
  115. dirigent_examples/shelves/queues/README.md +59 -0
  116. dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
  117. dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
  118. dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
  119. dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
  120. dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
  121. dirigent_examples/shelves/recipes/README.md +130 -0
  122. dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
  123. dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
  124. dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
  125. dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
  126. dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
  127. dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
  128. dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
  129. dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
  130. dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
  131. dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
  132. dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
  133. dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
  134. dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
  135. dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
  136. dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
  137. dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
  138. dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
  139. dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
  140. dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
  141. dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
  142. dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
  143. dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
  144. dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
  145. dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
  146. dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
  147. dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
  148. dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
  149. dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
  150. dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
  151. dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
  152. dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
  153. dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
  154. dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
  155. dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
  156. dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
  157. dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
  158. dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
  159. dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
  160. dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
  161. dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
  162. dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
  163. dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
  164. dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
  165. dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
  166. dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
  167. dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
  168. dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
  169. dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
  170. dirigent_examples/shelves/s3/README.md +34 -0
  171. dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
  172. dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
  173. dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
  174. dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
  175. dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
  176. dirigent_examples/shelves/schemas/README.md +36 -0
  177. dirigent_examples/shelves/schemas/echo-reading.json +18 -0
  178. dirigent_examples/shelves/schemas/ou-record.json +13 -0
  179. dirigent_examples/shelves/schemas/station-reading.json +13 -0
  180. dirigent_examples/shelves/sensors/README.md +16 -0
  181. dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
  182. dirigent_examples/shelves/sensors/time-window.yaml +61 -0
  183. dirigent_examples/shelves/sql/README.md +52 -0
  184. dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
  185. dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
  186. dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
  187. dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
  188. dirigent_examples/shelves/sql/warehouse.sql +42 -0
  189. dirigent_examples/shelves/transform/README.md +36 -0
  190. dirigent_examples/shelves/transform/csv-report.yaml +55 -0
  191. dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
  192. dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
  193. dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
  194. dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
  195. dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
  196. dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
  197. dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
  198. dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
  199. dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
  200. dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
  201. dirigent_examples/shelves/triggers/README.md +45 -0
  202. dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
  203. dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
  204. dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
  205. dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
  206. dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
  207. dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
  208. dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
  209. dirigent_examples/shelves/validate/README.md +31 -0
  210. dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
  211. dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
  212. dirigent_examples-0.15.0.dist-info/METADATA +21 -0
  213. dirigent_examples-0.15.0.dist-info/RECORD +216 -0
  214. dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
  215. dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
  216. dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,146 @@
1
+ # SQL over files: DuckDB reads a parquet artifact and writes a csv one.
2
+ #
3
+ # NEEDS NOTHING BUT THE ENGINE: no network, no daemon, no server, no allowlist. It does need
4
+ # duckdb on the worker, which is the dirigent-blocks[duckdb] extra, and convert.arrow, which
5
+ # is the dirigent-parquet pack.
6
+ # dg run --local examples/sql/duckdb-parquet-to-report.yaml --keep
7
+ # docs/sql.md is the family's home.
8
+ #
9
+ # RUNS WHEREVER THE ARTIFACTS LIVE. ${run.scratch} is a local directory under `dg dev` and a
10
+ # bucket on the compose stack, and this document does not care: a file:// artifact becomes a
11
+ # path duckdb opens, and an s3:// one is opened by duckdb's httpfs extension on the credentials
12
+ # the s3 scheme is configured from. The image ships that extension; a bare install does
13
+ # `python -c "import duckdb; duckdb.connect().execute('INSTALL httpfs')"` once.
14
+ #
15
+ # Five hops, and what each one hands on:
16
+ # readings three records as a constant, standing in for whatever produced them.
17
+ # as_json storage.write puts them in the run's scratch. Output: the uri they landed at.
18
+ # store convert.arrow re-encodes that object as parquet. Output: the target uri.
19
+ # summarise sql.query reads that parquet FILE and groups it. Output: the rows, inline.
20
+ # report sql.execute writes the answer out again as a csv artifact.
21
+ #
22
+ # WHY THE WRITE IS ITS OWN HOP. convert.arrow works on storage objects the way storage.copy
23
+ # does -- a source uri, a target uri, no value -- so records a step is holding become an
24
+ # object through storage.write before a codec can touch them.
25
+ #
26
+ # THE ENGINE IS THE CONNECTION'S URL, AND NOTHING ELSE CHANGES. These are the same two blocks
27
+ # the sqlite and postgres examples use, with the same fields and the same rules. A url naming
28
+ # duckdb picks the engine that reads files; a url naming postgres picks the one that reads a
29
+ # server. That is the whole of what "an engine of the family" means.
30
+ #
31
+ # WHY :memory: IS THE RIGHT DATABASE HERE. Nothing is stored in duckdb: the data lives in the
32
+ # parquet file, and duckdb is the thing that reads it. An in-memory database is then a query
33
+ # engine with no state of its own, opened and dropped inside each step. Name a duckdb file
34
+ # instead when a pipeline wants tables that outlive one step.
35
+ #
36
+ # A FILE IS NAMED BY A BOUND PARAMETER, NEVER BY TEXT IN THE STATEMENT. read_parquet(:source)
37
+ # takes the file the same way a WHERE clause takes a value, so the family's one rule holds
38
+ # here too: ${...} resolves into params and never into the sql. The value is a storage URI,
39
+ # and on a duckdb connection a file:// URI inside the run's own directories arrives as the path
40
+ # duckdb opens, while an s3:// one is handed over whole for httpfs to open. A file:// URI
41
+ # outside the run is refused; gs:// and azure:// still ask to be copied in first.
42
+ #
43
+ # BOTH DIRECTIONS. The read is read_parquet in summarise; the write is COPY ... TO :target in
44
+ # report, which is how a query's answer becomes an artifact a later step, or a person,
45
+ # collects. csv here because the report is meant to be read; write (FORMAT parquet) instead
46
+ # and the output is a typed file the next pipeline reads back with read_parquet.
47
+ #
48
+ # WHY THE WRITE IS sql.execute AND NOT sql.query. COPY returns no rows -- it returns a count
49
+ # -- and sql.query exists to hand rows on. sql.execute is the block that writes, which is also
50
+ # why the connection it names is not read_only.
51
+ #
52
+ # TO MAKE IT YOURS: point store's target at a bucket, replace the constant with the step that
53
+ # actually produces your rows, and grow the statement. Aggregation, joins across two parquet
54
+ # files, a window function -- it is one query engine over files, so the answer to "how do I
55
+ # join these" is a JOIN.
56
+
57
+ format: dirigent/v1
58
+ kind: pipeline
59
+ code: duckdb-parquet-to-report
60
+ name: Query a parquet file with DuckDB
61
+ description: Write a parquet artifact, group it with DuckDB over the file itself, and copy the answer out as csv.
62
+
63
+ tags: [sql, storage, transform, parquet]
64
+
65
+ requires:
66
+ blocks:
67
+ - value.const
68
+ - storage.write
69
+ - convert.arrow
70
+ - sql.query
71
+ - sql.execute
72
+
73
+ connections:
74
+ analysis:
75
+ kind: sql
76
+ config:
77
+ # No file: the tables this pipeline reads are the parquet files it names, so the
78
+ # database itself is empty and lives only as long as the step that opens it.
79
+ url: "duckdb:///:memory:"
80
+
81
+ steps:
82
+ readings:
83
+ block: value.const
84
+ config:
85
+ value:
86
+ - { station: st-1, region: east, celsius: 4.5 }
87
+ - { station: st-2, region: east, celsius: 6.1 }
88
+ - { station: st-3, region: west, celsius: 1.2 }
89
+ - { station: st-4, region: north, celsius: -3.4 }
90
+
91
+ as_json:
92
+ block: storage.write
93
+ depends_on: [readings]
94
+ config:
95
+ target: ${run.scratch}/readings.json
96
+ value: ${steps.readings.output.value}
97
+
98
+ store:
99
+ block: convert.arrow
100
+ depends_on: [as_json]
101
+ config:
102
+ source: ${steps.as_json.output.uri}
103
+ # The parquet file the two SQL steps below read.
104
+ target: ${run.scratch}/readings.parquet
105
+ from: json
106
+ to: parquet
107
+
108
+ summarise:
109
+ block: sql.query
110
+ depends_on: [store]
111
+ config:
112
+ connection: analysis
113
+ sql: >
114
+ SELECT region, COUNT(*) AS stations, ROUND(AVG(celsius), 2) AS mean_celsius
115
+ FROM read_parquet(:source)
116
+ WHERE celsius >= :floor
117
+ GROUP BY region
118
+ ORDER BY region
119
+ params:
120
+ # The artifact the previous step wrote, as a uri. The block resolves it to the path
121
+ # duckdb opens, because this is a duckdb connection.
122
+ source: "${steps.store.output.target}"
123
+ # An ordinary bound value beside the file, to show they are the same mechanism.
124
+ floor: -10
125
+ max_rows: 100
126
+
127
+ report:
128
+ block: sql.execute
129
+ depends_on: [store]
130
+ config:
131
+ connection: analysis
132
+ statements:
133
+ # One statement, reading the parquet and writing the csv, because nothing is kept
134
+ # between two steps of an in-memory database.
135
+ - >
136
+ COPY (
137
+ SELECT region, COUNT(*) AS stations, ROUND(AVG(celsius), 2) AS mean_celsius
138
+ FROM read_parquet(:source)
139
+ GROUP BY region
140
+ ORDER BY region
141
+ ) TO :target (FORMAT csv, HEADER)
142
+ params:
143
+ source: "${steps.store.output.target}"
144
+ # The write side of the same rule: the target is a uri too, and the csv lands in the
145
+ # run's scratch space where storage.copy or a person can pick it up.
146
+ target: "${run.scratch}/regions.csv"
@@ -0,0 +1,111 @@
1
+ # Reading a PostgreSQL warehouse through a connection that cannot write, whatever a step asks.
2
+ #
3
+ # NEEDS AN INSTANCE holding the connection, and a PostgreSQL it can reach. Unlike the other two
4
+ # documents on this shelf, this one names its connection instead of carrying it, so there is
5
+ # nothing here to run without a server:
6
+ # dg connection create sql warehouse-read \
7
+ # --set url=postgresql+asyncpg://reader@db.example:5432/warehouse \
8
+ # --set password=... \
9
+ # --set read_only=true
10
+ # dg connection check warehouse-read
11
+ # dg apply examples/sql/sql-postgres-readonly.yaml && dg run sql-postgres-readonly
12
+ # On the compose stack that PostgreSQL is the infra/compose.sql.yaml overlay, seeded with this
13
+ # table and a reader role, and the connection names it as `reader@warehouse:5432/warehouse`.
14
+ # It is still a valid document without any of that: an instance that does not hold the
15
+ # connection refuses it at apply, naming what is missing, rather than failing on the first run.
16
+ # `dg apply --dry-run` is how to see that answer without writing anything.
17
+ # docs/sql.md is the family's home.
18
+ #
19
+ # Two hops, and what each one hands on:
20
+ # recent the rows for one site since the run's window opened, bound as parameters.
21
+ # Output: the rows, their count and the columns.
22
+ # report a constant step reading them, standing in for whatever actually consumes them.
23
+ #
24
+ # THE PASSWORD IS SET SEPARATELY, AND THAT IS ENFORCED. A url written
25
+ # postgresql+asyncpg://reader:secret@db.example/warehouse is REFUSED when the connection is
26
+ # created: url is a plain field, so a password in it would sit unencrypted in the database and
27
+ # come back out of the API. password is a sealed field -- encrypted at rest, redacted in every
28
+ # response -- and the two are put together in memory at connect time and nowhere else.
29
+ #
30
+ # READ-ONLY IS THE DATABASE'S ANSWER, NOT A CHECK HERE. With read_only: true the connection
31
+ # opens every session as SET TRANSACTION READ ONLY, so a statement that writes is refused by
32
+ # PostgreSQL itself. sql.execute refuses this connection before it opens anything at all. The
33
+ # pattern is one connection per role: this one for every reporting pipeline, a separate
34
+ # warehouse-write for the few that load data.
35
+ #
36
+ # max_rows IS A PROMISE ABOUT THE OUTPUT. A step's output is stored with the run and read back
37
+ # whole, so it is not the place for an unbounded result. The query is narrowed by both
38
+ # parameters and then bounded again here; a result past the bound fails the step rather than
39
+ # arriving truncated. sql-query-to-storage.yaml is what to do when the answer really is large.
40
+ #
41
+ # THE WINDOW IS WHY THIS IS SCHEDULABLE. ${run.window.start} is the start of the interval the
42
+ # run covers, which a scheduled or backfilled run carries. Bound as a parameter it makes one
43
+ # document work for today, for yesterday, and for a backfill of last March, with no edit.
44
+ #
45
+ # TO MAKE IT YOURS: point the connection at your warehouse, replace the query with your own,
46
+ # and give the report step a real consumer -- a transform, a validate.schema, or an
47
+ # http.request that posts the rows on.
48
+
49
+ format: dirigent/v1
50
+ kind: pipeline
51
+ code: sql-postgres-readonly
52
+ name: Read a warehouse through a read-only connection
53
+ description: Query a PostgreSQL warehouse over a connection that refuses writes, with both values bound as parameters.
54
+
55
+ tags: [sql, starter]
56
+
57
+ requires:
58
+ blocks:
59
+ - sql.query
60
+ - value.const
61
+ connections:
62
+ # Named, not carried: the instance holds it, and this document refuses to apply where it
63
+ # is absent rather than failing the first time it runs.
64
+ - warehouse-read
65
+
66
+ params:
67
+ type: object
68
+ additionalProperties: false
69
+ properties:
70
+ site:
71
+ type: string
72
+ default: north
73
+ description: The site to report on, bound as a parameter rather than spliced into the SQL.
74
+
75
+ steps:
76
+ recent:
77
+ block: sql.query
78
+ deadline: 5m
79
+ config:
80
+ connection: warehouse-read
81
+ # Both filters are placeholders. Nothing a parameter contains can change what this
82
+ # statement does: the text is fixed when the document is applied.
83
+ sql: >-
84
+ SELECT id, site, seen, value
85
+ FROM reading
86
+ WHERE site = :site AND seen >= CAST(CAST(:since AS text) AS timestamptz)
87
+ ORDER BY seen
88
+ params:
89
+ site: "${params.site}"
90
+ # The interval this run covers. A scheduled run carries one; an ad hoc run carries one
91
+ # only if it was started with --window, and a step reading it otherwise fails rather
92
+ # than quietly widening to everything.
93
+ since: "${run.window.start}"
94
+ # The cast in the statement is what PostgreSQL needs: asyncpg types a parameter by where
95
+ # it sits, so a JSON string bound straight into a timestamp column is refused by the
96
+ # driver. Casting through text lets the server do the conversion it knows.
97
+ max_rows: 500
98
+ # Longer than a fast query needs and shorter than a runaway one takes. On PostgreSQL it
99
+ # is also set as the session's statement_timeout, so the server cancels the query rather
100
+ # than leaving it running after the worker stopped waiting.
101
+ timeout: 2m
102
+
103
+ report:
104
+ block: value.const
105
+ depends_on: [recent]
106
+ config:
107
+ value:
108
+ rows: "${steps.recent.output.rows}"
109
+ row_count: "${steps.recent.output.row_count}"
110
+ # How long the database took, which is the number worth watching as a table grows.
111
+ duration_ms: "${steps.recent.output.duration_ms}"
@@ -0,0 +1,85 @@
1
+ # A query's rows, and the hop that puts them in a file.
2
+ #
3
+ # NEEDS NOTHING: no network, no daemon, no server, no allowlist. The database is a SQLite file
4
+ # in the run's own work directory, and the rows land in the run's scratch space.
5
+ # dg run --local examples/sql/sql-query-to-storage.yaml --keep
6
+ # docs/sql.md is the family's home.
7
+ #
8
+ # Three hops, and what each one hands on:
9
+ # build creates the table and inserts three rows. Output: the row counts.
10
+ # extract selects them. Output: rows, row_count and columns.
11
+ # save storage.write puts those rows in the run's scratch space as json. Output: the
12
+ # uri they landed at, how many bytes went there, and what the object is.
13
+ #
14
+ # A QUERY HANDS ROWS ON, AND NOTHING ELSE. sql.query carries its rows in the step's output,
15
+ # where a transform maps them, a validate.schema checks them and a reference names them. A
16
+ # file is one more hop, not a mode: storage.write is the only way a value leaves a run, so
17
+ # the rows that belong in a file go to it, and the rows that belong to the next step stay in
18
+ # the output.
19
+ #
20
+ # max_rows IS THE LINE, AND IT IS ABOUT MEMORY. An output is stored with the run and read back
21
+ # whole, so a hundred thousand rows in one is a hundred thousand rows in the database and in
22
+ # every read of that run. The block fails the step at that bound rather than truncating,
23
+ # because half an answer is not a smaller answer -- and since the rows are carried either way,
24
+ # writing them to a file does not raise it. A result too large to carry is one a query narrows,
25
+ # with a GROUP BY, a LIMIT, or a WHERE the database evaluates instead of the worker.
26
+ #
27
+ # WHAT THE FILE IS. storage.write puts a value down as canonical JSON -- sorted keys, no
28
+ # spaces -- so reading.json is one array of objects keyed by column name. A convert.std step
29
+ # after this one re-spells it as ndjson or csv, and convert.arrow as parquet; each of those
30
+ # reads the uri this step wrote.
31
+ #
32
+ # TO MAKE IT YOURS: point save's target at a bucket rather than scratch if the file should
33
+ # outlive the run, and follow it with the step that consumes it -- a conversion, a
34
+ # storage.copy to a landing bucket, or the load that reads it back somewhere else.
35
+
36
+ format: dirigent/v1
37
+ kind: pipeline
38
+ code: sql-query-to-storage
39
+ name: Write a query result to storage
40
+ description: Read a table with sql.query and hand its rows to storage.write, which is how a result becomes a file.
41
+
42
+ tags: [sql, storage]
43
+
44
+ requires:
45
+ blocks:
46
+ - sql.execute
47
+ - sql.query
48
+ - storage.write
49
+
50
+ connections:
51
+ work-db:
52
+ kind: sql
53
+ config:
54
+ # Relative, so the database lands in this run's work directory and needs nothing on the
55
+ # host. The same connection as the roundtrip example, for the same reason.
56
+ url: sqlite+aiosqlite:///demo.db
57
+
58
+ steps:
59
+ build:
60
+ block: sql.execute
61
+ config:
62
+ connection: work-db
63
+ statements:
64
+ - CREATE TABLE reading (id INTEGER PRIMARY KEY, site TEXT NOT NULL, value REAL NOT NULL)
65
+ - INSERT INTO reading (id, site, value) VALUES (1, :north, 2.5), (2, :south, 4.0), (3, :north, 3.5)
66
+ params:
67
+ north: north
68
+ south: south
69
+
70
+ extract:
71
+ block: sql.query
72
+ depends_on: [build]
73
+ config:
74
+ connection: work-db
75
+ sql: SELECT id, site, value FROM reading ORDER BY id
76
+ max_rows: 100
77
+
78
+ save:
79
+ block: storage.write
80
+ depends_on: [extract]
81
+ config:
82
+ # A file inside the run's scratch space, so it is cleaned up with the run. A bucket URI
83
+ # here is what makes the result outlive it.
84
+ target: "${run.scratch}/reading.json"
85
+ value: "${steps.extract.output.rows}"
@@ -0,0 +1,114 @@
1
+ # A database built, written and read back inside one run, with every value bound.
2
+ #
3
+ # NEEDS NOTHING: no network, no daemon, no server, no allowlist. The database is a SQLite file
4
+ # in the run's own work directory, and both sql blocks are ordinary.
5
+ # dg run --local examples/sql/sql-sqlite-roundtrip.yaml
6
+ # docs/sql.md is the family's home.
7
+ #
8
+ # Three hops, and what each one hands on:
9
+ # build creates the table and inserts two rows, all in one transaction. Output: how many
10
+ # rows each statement touched.
11
+ # read selects those rows back with a bound parameter. Output: the rows, their count and
12
+ # the column names.
13
+ # report a constant step that reads the rows, proving they are an ordinary reference any
14
+ # downstream step can name.
15
+ #
16
+ # THE CONNECTION HOLDS THE DATABASE, THE DOCUMENT HOLDS THE STATEMENTS. A step never writes a
17
+ # URL: it names a sql connection, and the connection is what carries the database and the
18
+ # sealed password that opens it. Moving a pipeline from staging to production is then an edit
19
+ # to one connection and to no pipeline. This one is carried in the document because a --local
20
+ # run has no instance to hold it; on a server it would be created once with
21
+ # `dg connection create sql ...` and the document would name it and stop there.
22
+ #
23
+ # WHY THE URL IS RELATIVE. A sqlite database written with a relative path is resolved against
24
+ # the run's work directory, so this run builds its database, uses it, and leaves it with
25
+ # everything else the run wrote. That is what makes this example runnable anywhere with no
26
+ # setup at all; a real pipeline names a database that already exists.
27
+ #
28
+ # BINDING IS THE POINT. Not one value is written into the SQL text. Every one is a named
29
+ # parameter -- :site, :value -- and the parameters travel to the database beside the statement,
30
+ # so a value that reads as SQL is compared as a string and matches nothing. ${...} resolves
31
+ # into params, never into sql, and that is the whole rule.
32
+ #
33
+ # ONE TRANSACTION. sql.execute takes a list, and the list is one transaction: the CREATE and
34
+ # both INSERTs commit together or not at all. A DDL statement reports -1 because SQLite does
35
+ # not say how many rows a CREATE touched, which is what -1 means everywhere in row_counts.
36
+ #
37
+ # TO MAKE IT YOURS: point the connection at a real database, drop the build step, and keep the
38
+ # read step -- reading a table with bound parameters is what most pipelines actually want.
39
+
40
+ format: dirigent/v1
41
+ kind: pipeline
42
+ code: sql-sqlite-roundtrip
43
+ name: Write a table and read it back
44
+ description: Create a SQLite database in the run's work directory, insert rows with bound parameters, and read them back.
45
+
46
+ tags: [sql]
47
+
48
+ requires:
49
+ blocks:
50
+ - sql.execute
51
+ - sql.query
52
+ - value.const
53
+
54
+ connections:
55
+ work-db:
56
+ kind: sql
57
+ config:
58
+ # Relative, so it lands in this run's work directory and needs nothing on the host. The
59
+ # driver is written out because these blocks speak to a database over an async one, and
60
+ # a url without one is refused when the connection is written.
61
+ url: sqlite+aiosqlite:///demo.db
62
+
63
+ params:
64
+ type: object
65
+ additionalProperties: false
66
+ properties:
67
+ site:
68
+ type: string
69
+ default: north
70
+ description: The site the read step asks for, bound as a parameter rather than spliced into the SQL.
71
+
72
+ steps:
73
+ build:
74
+ block: sql.execute
75
+ config:
76
+ connection: work-db
77
+ statements:
78
+ - CREATE TABLE reading (id INTEGER PRIMARY KEY, site TEXT NOT NULL, value REAL NOT NULL)
79
+ # Both inserts share one params map, which is why they name different columns of it
80
+ # rather than repeating a value.
81
+ - INSERT INTO reading (id, site, value) VALUES (1, :north, :north_value)
82
+ - INSERT INTO reading (id, site, value) VALUES (2, :south, :south_value)
83
+ params:
84
+ north: north
85
+ north_value: 2.5
86
+ south: south
87
+ south_value: 4.0
88
+
89
+ read:
90
+ block: sql.query
91
+ depends_on: [build]
92
+ config:
93
+ connection: work-db
94
+ # One statement. A document that needs two writes two steps, or uses sql.execute.
95
+ sql: SELECT id, site, value FROM reading WHERE site = :site ORDER BY id
96
+ params:
97
+ # The pipeline parameter reaches the database as a bound value and never as text in
98
+ # the statement above, which is what makes it safe to let somebody else set it.
99
+ site: "${params.site}"
100
+ # Two rows fit comfortably; a query that could return more than this fails the step
101
+ # rather than truncating, so a wider result is one to narrow or to raise this for.
102
+ max_rows: 10
103
+
104
+ report:
105
+ block: value.const
106
+ depends_on: [read]
107
+ config:
108
+ value:
109
+ # A list of objects keyed by column name, which is the shape every downstream step
110
+ # sees: a transform maps it, a validate.schema checks it, an http.request posts it.
111
+ rows: "${steps.read.output.rows}"
112
+ # The count is there whether the rows were carried inline or saved to storage.
113
+ row_count: "${steps.read.output.row_count}"
114
+ columns: "${steps.read.output.columns}"
@@ -0,0 +1,42 @@
1
+ -- What the warehouse in infra/compose.sql.yaml holds: two roles and the table the sql
2
+ -- examples on this shelf query. The stack mounts this file; nothing else reads it.
3
+ --
4
+ -- Run by the postgres image's entrypoint on first start, connected to the warehouse database
5
+ -- as its superuser. \getenv reads each role's password from the environment, so no password
6
+ -- is written here.
7
+
8
+ \getenv reader_password WAREHOUSE_READER_PASSWORD
9
+ \getenv writer_password WAREHOUSE_WRITER_PASSWORD
10
+
11
+ -- The shape examples/sql/sql-postgres-readonly.yaml selects: one row per reading, narrowed by
12
+ -- site and by the run's window.
13
+ CREATE TABLE reading (
14
+ id bigint GENERATED ALWAYS AS IDENTITY PRIMARY KEY,
15
+ site text NOT NULL,
16
+ seen timestamptz NOT NULL,
17
+ value double precision NOT NULL
18
+ );
19
+
20
+ -- Seeded relative to now, so a run whose window is the last hour or the last day finds rows
21
+ -- however long ago the volume was created.
22
+ INSERT INTO reading (site, seen, value) VALUES
23
+ ('north', now() - interval '3 hours', 11.5),
24
+ ('north', now() - interval '2 hours', 12.25),
25
+ ('north', now() - interval '1 hour', 10.75),
26
+ ('north', now() - interval '20 minutes', 13.0),
27
+ ('south', now() - interval '90 minutes', 8.5),
28
+ ('south', now() - interval '30 minutes', 9.25);
29
+
30
+ -- SELECT and nothing else. A statement that writes through this role is refused by the
31
+ -- database itself, whatever the connection or the document asks for.
32
+ CREATE ROLE reader LOGIN PASSWORD :'reader_password';
33
+ GRANT SELECT ON ALL TABLES IN SCHEMA public TO reader;
34
+ ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT ON TABLES TO reader;
35
+
36
+ -- The role a loading pipeline uses: the four statements, and the sequence an identity column
37
+ -- draws from.
38
+ CREATE ROLE writer LOGIN PASSWORD :'writer_password';
39
+ GRANT SELECT, INSERT, UPDATE, DELETE ON ALL TABLES IN SCHEMA public TO writer;
40
+ GRANT USAGE ON ALL SEQUENCES IN SCHEMA public TO writer;
41
+ ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT, INSERT, UPDATE, DELETE ON TABLES TO writer;
42
+ ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT USAGE ON SEQUENCES TO writer;
@@ -0,0 +1,36 @@
1
+ # Transform examples
2
+
3
+ These pipelines reshape data with the transform verbs -- `transform`, `map`, `filter` and
4
+ `convert` -- which run a program or a codec against a value and execute nothing on the
5
+ worker. None of them needs an allowlist entry, a network, or anything installed first.
6
+ [docs/transforms.md](../../docs/transforms.md) is the page behind them: a verb is a contract
7
+ with a promise its frame enforces, and a kind is the engine that keeps it.
8
+
9
+ A file that stars an engine is prefixed with its kind, the way a block id's
10
+ `<verb>.<kind>` names it: `jq-` for the jq engines, `std-` for the `convert.std` codec. The
11
+ rest are named for the format they round-trip.
12
+
13
+ ```bash
14
+ dg run --local examples/transform/jq-reshape.yaml
15
+ ```
16
+
17
+ ## Pipelines
18
+
19
+ | File | What it teaches |
20
+ | --- | --- |
21
+ | [jq-reshape.yaml](jq-reshape.yaml) | The whole-value reshape: one jq program between steps, then a fan-out that reads its list once per element. |
22
+ | [jq-filter-and-map.yaml](jq-filter-and-map.yaml) | The two element-wise verbs against the whole-value one: a subset, a list of the same length, and a reshape. |
23
+ | [jq-group-and-aggregate.yaml](jq-group-and-aggregate.yaml) | `group_by` and arithmetic: per-group sums and means with a total beside them. |
24
+ | [jq-join-two-sources.yaml](jq-join-two-sources.yaml) | Two upstream outputs composed into one inline input, joined with `INDEX`, unmatched rows kept with a null name. |
25
+ | [jq-stream-through-storage.yaml](jq-stream-through-storage.yaml) | The two doors on storage: `storage.write` puts a value in an object and `storage.read` brings one back, with the reshape between them. |
26
+ | [std-convert-fan-out.yaml](std-convert-fan-out.yaml) | The codec: csv to json, reshaped with jq, fanned out over the regions, and written back as csv. |
27
+ | [csv-report.yaml](csv-report.yaml) | Records shaped into flat rows and written as a csv artifact, with nothing on the allowlist. |
28
+ | [ndjson-round-trip.yaml](ndjson-round-trip.yaml) | ndjson: a JSON array re-spelled one record per line, and read back. |
29
+ | [yaml-config-to-json.yaml](yaml-config-to-json.yaml) | yaml: one document is one value, so a config becomes the object it describes, and comes back a document. |
30
+ | [xml-feed-to-ndjson.yaml](xml-feed-to-ndjson.yaml) | xml: a feed's elements as one record per line, the mapping that makes attributes and children keys, and the whole document as one object. |
31
+ | [parquet-round-trip.yaml](parquet-round-trip.yaml) | Records to parquet and back, the types surviving where csv would flatten them to strings (needs `dirigent-parquet`). |
32
+
33
+ Every program on this shelf is reference-free, so jq compiles them when the document is
34
+ applied and a syntax error is an issue beside every other one the document has. That is why
35
+ the data is always the step's `input` and never spliced into the program: a config carrying
36
+ a `${...}` cannot be checked until the run.
@@ -0,0 +1,55 @@
1
+ # A csv report, written where a person can fetch it.
2
+ #
3
+ # Four hops turn records into a file: a fixed value stands in for whatever produced the
4
+ # records, a jq program shapes them into flat rows, storage.write puts those rows in the
5
+ # run's scratch as json, and convert.std re-encodes that object as csv -- so the run's
6
+ # Output tab lists a report.csv somebody can read back. Nothing here is on the allowlist.
7
+ #
8
+ # convert.std works on storage objects, the way storage.copy does: it names a source uri
9
+ # and a target uri and never carries a value. The write is how the rows a step is holding
10
+ # become an object it can read.
11
+ #
12
+ # The rows must be flat: a nested value has no csv spelling and is refused naming the row
13
+ # and the key, so the jq step is where a list becomes one joined cell.
14
+ #
15
+ # dg run --local examples/transform/csv-report.yaml --keep
16
+
17
+ format: dirigent/v1
18
+ kind: pipeline
19
+ code: csv-report
20
+ name: A report as csv
21
+ description: Shape records into flat rows, store them as json, and re-encode that as a csv artifact.
22
+
23
+ tags: [transform, storage]
24
+
25
+ steps:
26
+ readings:
27
+ block: value.const
28
+ config:
29
+ value:
30
+ - { station: st-1, region: east, celsius: 4.5, tags: [ok, new] }
31
+ - { station: st-3, region: west, celsius: 1.2, tags: [ok] }
32
+ - { station: st-4, region: north, celsius: -3.4, tags: [] }
33
+ rows:
34
+ block: transform.jq
35
+ depends_on: [readings]
36
+ config:
37
+ input: ${steps.readings.output.value}
38
+ # Flatten for the codec: the tag list becomes one joined cell, because only this
39
+ # program knows what the separator should mean.
40
+ program: |
41
+ [.[] | {station, region, celsius, tags: (.tags | join(" "))}]
42
+ store:
43
+ block: storage.write
44
+ depends_on: [rows]
45
+ config:
46
+ target: ${run.scratch}/rows.json
47
+ value: ${steps.rows.output.value}
48
+ report:
49
+ block: convert.std
50
+ depends_on: [store]
51
+ config:
52
+ source: ${steps.store.output.uri}
53
+ target: ${run.scratch}/report.csv
54
+ from: json
55
+ to: csv
@@ -0,0 +1,70 @@
1
+ # The two constrained verbs, and what they buy over doing everything in one program.
2
+ #
3
+ # filter.jq and map.jq run a jq program once per element of a list. Neither can do what the
4
+ # other does: a filter's program answers true or false and the frame keeps the element it
5
+ # was given, so a filtered list is a subset with nothing modified, and a map's program
6
+ # returns one replacement per element, so a mapped list is exactly as long as its input.
7
+ # Written as one transform.jq program the same work is one line -- and nothing but reading
8
+ # it tells you whether it dropped rows, added rows, or edited them in place.
9
+ #
10
+ # The steps read: keep the active readings, convert each one to fahrenheit, then summarise.
11
+ # The third step is a transform because it genuinely is one: it changes the shape of the
12
+ # whole value rather than working element by element, and that is the verb that says so.
13
+ #
14
+ # The per-element contracts are strict on purpose. A map program emitting nothing, or two
15
+ # things, for one element is refused rather than quietly changing the length, and a filter
16
+ # program answering 0 or "" is refused rather than read as false: jq's truthiness is not
17
+ # applied, so a program meaning "has readings" writes .count > 0.
18
+ #
19
+ # The input is written inline so the example needs no network. In a real pipeline it is a
20
+ # reference to an upstream step's output, which is how a value that lives in storage arrives:
21
+ # through a storage.read of its own.
22
+ #
23
+ # dg run --local examples/transform/jq-filter-and-map.yaml
24
+
25
+ format: dirigent/v1
26
+ kind: pipeline
27
+ code: jq-filter-and-map
28
+ name: Filter and map with jq
29
+ description: Keep the active readings with filter.jq, convert each with map.jq, and summarise them.
30
+
31
+ tags: [transform]
32
+
33
+ requires:
34
+ blocks:
35
+ - filter.jq
36
+ - map.jq
37
+ - transform.jq
38
+
39
+ steps:
40
+ active:
41
+ block: filter.jq
42
+ config:
43
+ input:
44
+ - { station: st-1, region: east, status: active, celsius: 4.5 }
45
+ - { station: st-2, region: east, status: retired, celsius: 11.0 }
46
+ - { station: st-3, region: west, status: active, celsius: 1.2 }
47
+ - { station: st-4, region: north, status: active, celsius: -3.4 }
48
+ # The answer is the verdict and nothing else: the elements that come out are the ones
49
+ # that went in, unmodified, because the frame keeps them rather than the program.
50
+ program: |
51
+ .status == "active"
52
+
53
+ fahrenheit:
54
+ block: map.jq
55
+ depends_on: [active]
56
+ config:
57
+ input: "${steps.active.output.value}"
58
+ # One output per element, so this list is as long as the one above it.
59
+ program: |
60
+ {station, region, fahrenheit: (.celsius * 9 / 5 + 32 | round)}
61
+
62
+ summary:
63
+ block: transform.jq
64
+ depends_on: [fahrenheit]
65
+ config:
66
+ # A whole-value reshape rather than an element-wise one, which is why it is the third
67
+ # verb and not one of the other two.
68
+ input: "${steps.fahrenheit.output.value}"
69
+ program: |
70
+ {stations: length, regions: (map(.region) | unique), warmest: (max_by(.fahrenheit) | .station)}