dirigent-examples 0.15.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_examples/__init__.py +22 -0
- dirigent_examples/py.typed +0 -0
- dirigent_examples/shelves/README.md +299 -0
- dirigent_examples/shelves/composition/README.md +18 -0
- dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
- dirigent_examples/shelves/composition/composition-child.yaml +64 -0
- dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
- dirigent_examples/shelves/connections.yaml +52 -0
- dirigent_examples/shelves/demo/README.md +19 -0
- dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
- dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
- dirigent_examples/shelves/demo/requires.yaml +65 -0
- dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
- dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
- dirigent_examples/shelves/docker/README.md +29 -0
- dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
- dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
- dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
- dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
- dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
- dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
- dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
- dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
- dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
- dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
- dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
- dirigent_examples/shelves/execute/README.md +16 -0
- dirigent_examples/shelves/execute/long-log.yaml +89 -0
- dirigent_examples/shelves/failure/README.md +20 -0
- dirigent_examples/shelves/failure/error-handler.yaml +89 -0
- dirigent_examples/shelves/failure/optional-step.yaml +82 -0
- dirigent_examples/shelves/failure/retries.yaml +75 -0
- dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
- dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
- dirigent_examples/shelves/git/README.md +32 -0
- dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
- dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
- dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
- dirigent_examples/shelves/graph/README.md +22 -0
- dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
- dirigent_examples/shelves/graph/fan-in.yaml +80 -0
- dirigent_examples/shelves/graph/fan-out.yaml +66 -0
- dirigent_examples/shelves/graph/linear.yaml +66 -0
- dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
- dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
- dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
- dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
- dirigent_examples/shelves/hello-world.yaml +30 -0
- dirigent_examples/shelves/open-data/README.md +67 -0
- dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
- dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
- dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
- dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
- dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
- dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
- dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
- dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
- dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
- dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
- dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
- dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
- dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
- dirigent_examples/shelves/patterns/README.md +144 -0
- dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
- dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
- dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
- dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
- dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
- dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
- dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
- dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
- dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
- dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
- dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
- dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
- dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
- dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
- dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
- dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
- dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
- dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
- dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
- dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
- dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
- dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
- dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
- dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
- dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
- dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
- dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
- dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
- dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
- dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
- dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
- dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
- dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
- dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
- dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
- dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
- dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
- dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
- dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
- dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
- dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
- dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
- dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
- dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
- dirigent_examples/shelves/python/README.md +31 -0
- dirigent_examples/shelves/python/apply_and_run.py +52 -0
- dirigent_examples/shelves/python/ci_gate.py +76 -0
- dirigent_examples/shelves/python/connections.py +61 -0
- dirigent_examples/shelves/python/error_handling.py +84 -0
- dirigent_examples/shelves/python/follow_logs.py +39 -0
- dirigent_examples/shelves/python/list_and_filter.py +52 -0
- dirigent_examples/shelves/queues/README.md +59 -0
- dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
- dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
- dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
- dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
- dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
- dirigent_examples/shelves/recipes/README.md +130 -0
- dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
- dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
- dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
- dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
- dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
- dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
- dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
- dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
- dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
- dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
- dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
- dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
- dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
- dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
- dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
- dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
- dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
- dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
- dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
- dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
- dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
- dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
- dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
- dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
- dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
- dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
- dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
- dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
- dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
- dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
- dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
- dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
- dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
- dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
- dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
- dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
- dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
- dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
- dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
- dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
- dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
- dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
- dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
- dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
- dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
- dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
- dirigent_examples/shelves/s3/README.md +34 -0
- dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
- dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
- dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
- dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
- dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
- dirigent_examples/shelves/schemas/README.md +36 -0
- dirigent_examples/shelves/schemas/echo-reading.json +18 -0
- dirigent_examples/shelves/schemas/ou-record.json +13 -0
- dirigent_examples/shelves/schemas/station-reading.json +13 -0
- dirigent_examples/shelves/sensors/README.md +16 -0
- dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
- dirigent_examples/shelves/sensors/time-window.yaml +61 -0
- dirigent_examples/shelves/sql/README.md +52 -0
- dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
- dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
- dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
- dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
- dirigent_examples/shelves/sql/warehouse.sql +42 -0
- dirigent_examples/shelves/transform/README.md +36 -0
- dirigent_examples/shelves/transform/csv-report.yaml +55 -0
- dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
- dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
- dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
- dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
- dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
- dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
- dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
- dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
- dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
- dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
- dirigent_examples/shelves/triggers/README.md +45 -0
- dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
- dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
- dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
- dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
- dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
- dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
- dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
- dirigent_examples/shelves/validate/README.md +31 -0
- dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
- dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
- dirigent_examples-0.15.0.dist-info/METADATA +21 -0
- dirigent_examples-0.15.0.dist-info/RECORD +216 -0
- dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
- dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
- dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# SQL over files: DuckDB reads a parquet artifact and writes a csv one.
|
|
2
|
+
#
|
|
3
|
+
# NEEDS NOTHING BUT THE ENGINE: no network, no daemon, no server, no allowlist. It does need
|
|
4
|
+
# duckdb on the worker, which is the dirigent-blocks[duckdb] extra, and convert.arrow, which
|
|
5
|
+
# is the dirigent-parquet pack.
|
|
6
|
+
# dg run --local examples/sql/duckdb-parquet-to-report.yaml --keep
|
|
7
|
+
# docs/sql.md is the family's home.
|
|
8
|
+
#
|
|
9
|
+
# RUNS WHEREVER THE ARTIFACTS LIVE. ${run.scratch} is a local directory under `dg dev` and a
|
|
10
|
+
# bucket on the compose stack, and this document does not care: a file:// artifact becomes a
|
|
11
|
+
# path duckdb opens, and an s3:// one is opened by duckdb's httpfs extension on the credentials
|
|
12
|
+
# the s3 scheme is configured from. The image ships that extension; a bare install does
|
|
13
|
+
# `python -c "import duckdb; duckdb.connect().execute('INSTALL httpfs')"` once.
|
|
14
|
+
#
|
|
15
|
+
# Five hops, and what each one hands on:
|
|
16
|
+
# readings three records as a constant, standing in for whatever produced them.
|
|
17
|
+
# as_json storage.write puts them in the run's scratch. Output: the uri they landed at.
|
|
18
|
+
# store convert.arrow re-encodes that object as parquet. Output: the target uri.
|
|
19
|
+
# summarise sql.query reads that parquet FILE and groups it. Output: the rows, inline.
|
|
20
|
+
# report sql.execute writes the answer out again as a csv artifact.
|
|
21
|
+
#
|
|
22
|
+
# WHY THE WRITE IS ITS OWN HOP. convert.arrow works on storage objects the way storage.copy
|
|
23
|
+
# does -- a source uri, a target uri, no value -- so records a step is holding become an
|
|
24
|
+
# object through storage.write before a codec can touch them.
|
|
25
|
+
#
|
|
26
|
+
# THE ENGINE IS THE CONNECTION'S URL, AND NOTHING ELSE CHANGES. These are the same two blocks
|
|
27
|
+
# the sqlite and postgres examples use, with the same fields and the same rules. A url naming
|
|
28
|
+
# duckdb picks the engine that reads files; a url naming postgres picks the one that reads a
|
|
29
|
+
# server. That is the whole of what "an engine of the family" means.
|
|
30
|
+
#
|
|
31
|
+
# WHY :memory: IS THE RIGHT DATABASE HERE. Nothing is stored in duckdb: the data lives in the
|
|
32
|
+
# parquet file, and duckdb is the thing that reads it. An in-memory database is then a query
|
|
33
|
+
# engine with no state of its own, opened and dropped inside each step. Name a duckdb file
|
|
34
|
+
# instead when a pipeline wants tables that outlive one step.
|
|
35
|
+
#
|
|
36
|
+
# A FILE IS NAMED BY A BOUND PARAMETER, NEVER BY TEXT IN THE STATEMENT. read_parquet(:source)
|
|
37
|
+
# takes the file the same way a WHERE clause takes a value, so the family's one rule holds
|
|
38
|
+
# here too: ${...} resolves into params and never into the sql. The value is a storage URI,
|
|
39
|
+
# and on a duckdb connection a file:// URI inside the run's own directories arrives as the path
|
|
40
|
+
# duckdb opens, while an s3:// one is handed over whole for httpfs to open. A file:// URI
|
|
41
|
+
# outside the run is refused; gs:// and azure:// still ask to be copied in first.
|
|
42
|
+
#
|
|
43
|
+
# BOTH DIRECTIONS. The read is read_parquet in summarise; the write is COPY ... TO :target in
|
|
44
|
+
# report, which is how a query's answer becomes an artifact a later step, or a person,
|
|
45
|
+
# collects. csv here because the report is meant to be read; write (FORMAT parquet) instead
|
|
46
|
+
# and the output is a typed file the next pipeline reads back with read_parquet.
|
|
47
|
+
#
|
|
48
|
+
# WHY THE WRITE IS sql.execute AND NOT sql.query. COPY returns no rows -- it returns a count
|
|
49
|
+
# -- and sql.query exists to hand rows on. sql.execute is the block that writes, which is also
|
|
50
|
+
# why the connection it names is not read_only.
|
|
51
|
+
#
|
|
52
|
+
# TO MAKE IT YOURS: point store's target at a bucket, replace the constant with the step that
|
|
53
|
+
# actually produces your rows, and grow the statement. Aggregation, joins across two parquet
|
|
54
|
+
# files, a window function -- it is one query engine over files, so the answer to "how do I
|
|
55
|
+
# join these" is a JOIN.
|
|
56
|
+
|
|
57
|
+
format: dirigent/v1
|
|
58
|
+
kind: pipeline
|
|
59
|
+
code: duckdb-parquet-to-report
|
|
60
|
+
name: Query a parquet file with DuckDB
|
|
61
|
+
description: Write a parquet artifact, group it with DuckDB over the file itself, and copy the answer out as csv.
|
|
62
|
+
|
|
63
|
+
tags: [sql, storage, transform, parquet]
|
|
64
|
+
|
|
65
|
+
requires:
|
|
66
|
+
blocks:
|
|
67
|
+
- value.const
|
|
68
|
+
- storage.write
|
|
69
|
+
- convert.arrow
|
|
70
|
+
- sql.query
|
|
71
|
+
- sql.execute
|
|
72
|
+
|
|
73
|
+
connections:
|
|
74
|
+
analysis:
|
|
75
|
+
kind: sql
|
|
76
|
+
config:
|
|
77
|
+
# No file: the tables this pipeline reads are the parquet files it names, so the
|
|
78
|
+
# database itself is empty and lives only as long as the step that opens it.
|
|
79
|
+
url: "duckdb:///:memory:"
|
|
80
|
+
|
|
81
|
+
steps:
|
|
82
|
+
readings:
|
|
83
|
+
block: value.const
|
|
84
|
+
config:
|
|
85
|
+
value:
|
|
86
|
+
- { station: st-1, region: east, celsius: 4.5 }
|
|
87
|
+
- { station: st-2, region: east, celsius: 6.1 }
|
|
88
|
+
- { station: st-3, region: west, celsius: 1.2 }
|
|
89
|
+
- { station: st-4, region: north, celsius: -3.4 }
|
|
90
|
+
|
|
91
|
+
as_json:
|
|
92
|
+
block: storage.write
|
|
93
|
+
depends_on: [readings]
|
|
94
|
+
config:
|
|
95
|
+
target: ${run.scratch}/readings.json
|
|
96
|
+
value: ${steps.readings.output.value}
|
|
97
|
+
|
|
98
|
+
store:
|
|
99
|
+
block: convert.arrow
|
|
100
|
+
depends_on: [as_json]
|
|
101
|
+
config:
|
|
102
|
+
source: ${steps.as_json.output.uri}
|
|
103
|
+
# The parquet file the two SQL steps below read.
|
|
104
|
+
target: ${run.scratch}/readings.parquet
|
|
105
|
+
from: json
|
|
106
|
+
to: parquet
|
|
107
|
+
|
|
108
|
+
summarise:
|
|
109
|
+
block: sql.query
|
|
110
|
+
depends_on: [store]
|
|
111
|
+
config:
|
|
112
|
+
connection: analysis
|
|
113
|
+
sql: >
|
|
114
|
+
SELECT region, COUNT(*) AS stations, ROUND(AVG(celsius), 2) AS mean_celsius
|
|
115
|
+
FROM read_parquet(:source)
|
|
116
|
+
WHERE celsius >= :floor
|
|
117
|
+
GROUP BY region
|
|
118
|
+
ORDER BY region
|
|
119
|
+
params:
|
|
120
|
+
# The artifact the previous step wrote, as a uri. The block resolves it to the path
|
|
121
|
+
# duckdb opens, because this is a duckdb connection.
|
|
122
|
+
source: "${steps.store.output.target}"
|
|
123
|
+
# An ordinary bound value beside the file, to show they are the same mechanism.
|
|
124
|
+
floor: -10
|
|
125
|
+
max_rows: 100
|
|
126
|
+
|
|
127
|
+
report:
|
|
128
|
+
block: sql.execute
|
|
129
|
+
depends_on: [store]
|
|
130
|
+
config:
|
|
131
|
+
connection: analysis
|
|
132
|
+
statements:
|
|
133
|
+
# One statement, reading the parquet and writing the csv, because nothing is kept
|
|
134
|
+
# between two steps of an in-memory database.
|
|
135
|
+
- >
|
|
136
|
+
COPY (
|
|
137
|
+
SELECT region, COUNT(*) AS stations, ROUND(AVG(celsius), 2) AS mean_celsius
|
|
138
|
+
FROM read_parquet(:source)
|
|
139
|
+
GROUP BY region
|
|
140
|
+
ORDER BY region
|
|
141
|
+
) TO :target (FORMAT csv, HEADER)
|
|
142
|
+
params:
|
|
143
|
+
source: "${steps.store.output.target}"
|
|
144
|
+
# The write side of the same rule: the target is a uri too, and the csv lands in the
|
|
145
|
+
# run's scratch space where storage.copy or a person can pick it up.
|
|
146
|
+
target: "${run.scratch}/regions.csv"
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
# Reading a PostgreSQL warehouse through a connection that cannot write, whatever a step asks.
|
|
2
|
+
#
|
|
3
|
+
# NEEDS AN INSTANCE holding the connection, and a PostgreSQL it can reach. Unlike the other two
|
|
4
|
+
# documents on this shelf, this one names its connection instead of carrying it, so there is
|
|
5
|
+
# nothing here to run without a server:
|
|
6
|
+
# dg connection create sql warehouse-read \
|
|
7
|
+
# --set url=postgresql+asyncpg://reader@db.example:5432/warehouse \
|
|
8
|
+
# --set password=... \
|
|
9
|
+
# --set read_only=true
|
|
10
|
+
# dg connection check warehouse-read
|
|
11
|
+
# dg apply examples/sql/sql-postgres-readonly.yaml && dg run sql-postgres-readonly
|
|
12
|
+
# On the compose stack that PostgreSQL is the infra/compose.sql.yaml overlay, seeded with this
|
|
13
|
+
# table and a reader role, and the connection names it as `reader@warehouse:5432/warehouse`.
|
|
14
|
+
# It is still a valid document without any of that: an instance that does not hold the
|
|
15
|
+
# connection refuses it at apply, naming what is missing, rather than failing on the first run.
|
|
16
|
+
# `dg apply --dry-run` is how to see that answer without writing anything.
|
|
17
|
+
# docs/sql.md is the family's home.
|
|
18
|
+
#
|
|
19
|
+
# Two hops, and what each one hands on:
|
|
20
|
+
# recent the rows for one site since the run's window opened, bound as parameters.
|
|
21
|
+
# Output: the rows, their count and the columns.
|
|
22
|
+
# report a constant step reading them, standing in for whatever actually consumes them.
|
|
23
|
+
#
|
|
24
|
+
# THE PASSWORD IS SET SEPARATELY, AND THAT IS ENFORCED. A url written
|
|
25
|
+
# postgresql+asyncpg://reader:secret@db.example/warehouse is REFUSED when the connection is
|
|
26
|
+
# created: url is a plain field, so a password in it would sit unencrypted in the database and
|
|
27
|
+
# come back out of the API. password is a sealed field -- encrypted at rest, redacted in every
|
|
28
|
+
# response -- and the two are put together in memory at connect time and nowhere else.
|
|
29
|
+
#
|
|
30
|
+
# READ-ONLY IS THE DATABASE'S ANSWER, NOT A CHECK HERE. With read_only: true the connection
|
|
31
|
+
# opens every session as SET TRANSACTION READ ONLY, so a statement that writes is refused by
|
|
32
|
+
# PostgreSQL itself. sql.execute refuses this connection before it opens anything at all. The
|
|
33
|
+
# pattern is one connection per role: this one for every reporting pipeline, a separate
|
|
34
|
+
# warehouse-write for the few that load data.
|
|
35
|
+
#
|
|
36
|
+
# max_rows IS A PROMISE ABOUT THE OUTPUT. A step's output is stored with the run and read back
|
|
37
|
+
# whole, so it is not the place for an unbounded result. The query is narrowed by both
|
|
38
|
+
# parameters and then bounded again here; a result past the bound fails the step rather than
|
|
39
|
+
# arriving truncated. sql-query-to-storage.yaml is what to do when the answer really is large.
|
|
40
|
+
#
|
|
41
|
+
# THE WINDOW IS WHY THIS IS SCHEDULABLE. ${run.window.start} is the start of the interval the
|
|
42
|
+
# run covers, which a scheduled or backfilled run carries. Bound as a parameter it makes one
|
|
43
|
+
# document work for today, for yesterday, and for a backfill of last March, with no edit.
|
|
44
|
+
#
|
|
45
|
+
# TO MAKE IT YOURS: point the connection at your warehouse, replace the query with your own,
|
|
46
|
+
# and give the report step a real consumer -- a transform, a validate.schema, or an
|
|
47
|
+
# http.request that posts the rows on.
|
|
48
|
+
|
|
49
|
+
format: dirigent/v1
|
|
50
|
+
kind: pipeline
|
|
51
|
+
code: sql-postgres-readonly
|
|
52
|
+
name: Read a warehouse through a read-only connection
|
|
53
|
+
description: Query a PostgreSQL warehouse over a connection that refuses writes, with both values bound as parameters.
|
|
54
|
+
|
|
55
|
+
tags: [sql, starter]
|
|
56
|
+
|
|
57
|
+
requires:
|
|
58
|
+
blocks:
|
|
59
|
+
- sql.query
|
|
60
|
+
- value.const
|
|
61
|
+
connections:
|
|
62
|
+
# Named, not carried: the instance holds it, and this document refuses to apply where it
|
|
63
|
+
# is absent rather than failing the first time it runs.
|
|
64
|
+
- warehouse-read
|
|
65
|
+
|
|
66
|
+
params:
|
|
67
|
+
type: object
|
|
68
|
+
additionalProperties: false
|
|
69
|
+
properties:
|
|
70
|
+
site:
|
|
71
|
+
type: string
|
|
72
|
+
default: north
|
|
73
|
+
description: The site to report on, bound as a parameter rather than spliced into the SQL.
|
|
74
|
+
|
|
75
|
+
steps:
|
|
76
|
+
recent:
|
|
77
|
+
block: sql.query
|
|
78
|
+
deadline: 5m
|
|
79
|
+
config:
|
|
80
|
+
connection: warehouse-read
|
|
81
|
+
# Both filters are placeholders. Nothing a parameter contains can change what this
|
|
82
|
+
# statement does: the text is fixed when the document is applied.
|
|
83
|
+
sql: >-
|
|
84
|
+
SELECT id, site, seen, value
|
|
85
|
+
FROM reading
|
|
86
|
+
WHERE site = :site AND seen >= CAST(CAST(:since AS text) AS timestamptz)
|
|
87
|
+
ORDER BY seen
|
|
88
|
+
params:
|
|
89
|
+
site: "${params.site}"
|
|
90
|
+
# The interval this run covers. A scheduled run carries one; an ad hoc run carries one
|
|
91
|
+
# only if it was started with --window, and a step reading it otherwise fails rather
|
|
92
|
+
# than quietly widening to everything.
|
|
93
|
+
since: "${run.window.start}"
|
|
94
|
+
# The cast in the statement is what PostgreSQL needs: asyncpg types a parameter by where
|
|
95
|
+
# it sits, so a JSON string bound straight into a timestamp column is refused by the
|
|
96
|
+
# driver. Casting through text lets the server do the conversion it knows.
|
|
97
|
+
max_rows: 500
|
|
98
|
+
# Longer than a fast query needs and shorter than a runaway one takes. On PostgreSQL it
|
|
99
|
+
# is also set as the session's statement_timeout, so the server cancels the query rather
|
|
100
|
+
# than leaving it running after the worker stopped waiting.
|
|
101
|
+
timeout: 2m
|
|
102
|
+
|
|
103
|
+
report:
|
|
104
|
+
block: value.const
|
|
105
|
+
depends_on: [recent]
|
|
106
|
+
config:
|
|
107
|
+
value:
|
|
108
|
+
rows: "${steps.recent.output.rows}"
|
|
109
|
+
row_count: "${steps.recent.output.row_count}"
|
|
110
|
+
# How long the database took, which is the number worth watching as a table grows.
|
|
111
|
+
duration_ms: "${steps.recent.output.duration_ms}"
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
# A query's rows, and the hop that puts them in a file.
|
|
2
|
+
#
|
|
3
|
+
# NEEDS NOTHING: no network, no daemon, no server, no allowlist. The database is a SQLite file
|
|
4
|
+
# in the run's own work directory, and the rows land in the run's scratch space.
|
|
5
|
+
# dg run --local examples/sql/sql-query-to-storage.yaml --keep
|
|
6
|
+
# docs/sql.md is the family's home.
|
|
7
|
+
#
|
|
8
|
+
# Three hops, and what each one hands on:
|
|
9
|
+
# build creates the table and inserts three rows. Output: the row counts.
|
|
10
|
+
# extract selects them. Output: rows, row_count and columns.
|
|
11
|
+
# save storage.write puts those rows in the run's scratch space as json. Output: the
|
|
12
|
+
# uri they landed at, how many bytes went there, and what the object is.
|
|
13
|
+
#
|
|
14
|
+
# A QUERY HANDS ROWS ON, AND NOTHING ELSE. sql.query carries its rows in the step's output,
|
|
15
|
+
# where a transform maps them, a validate.schema checks them and a reference names them. A
|
|
16
|
+
# file is one more hop, not a mode: storage.write is the only way a value leaves a run, so
|
|
17
|
+
# the rows that belong in a file go to it, and the rows that belong to the next step stay in
|
|
18
|
+
# the output.
|
|
19
|
+
#
|
|
20
|
+
# max_rows IS THE LINE, AND IT IS ABOUT MEMORY. An output is stored with the run and read back
|
|
21
|
+
# whole, so a hundred thousand rows in one is a hundred thousand rows in the database and in
|
|
22
|
+
# every read of that run. The block fails the step at that bound rather than truncating,
|
|
23
|
+
# because half an answer is not a smaller answer -- and since the rows are carried either way,
|
|
24
|
+
# writing them to a file does not raise it. A result too large to carry is one a query narrows,
|
|
25
|
+
# with a GROUP BY, a LIMIT, or a WHERE the database evaluates instead of the worker.
|
|
26
|
+
#
|
|
27
|
+
# WHAT THE FILE IS. storage.write puts a value down as canonical JSON -- sorted keys, no
|
|
28
|
+
# spaces -- so reading.json is one array of objects keyed by column name. A convert.std step
|
|
29
|
+
# after this one re-spells it as ndjson or csv, and convert.arrow as parquet; each of those
|
|
30
|
+
# reads the uri this step wrote.
|
|
31
|
+
#
|
|
32
|
+
# TO MAKE IT YOURS: point save's target at a bucket rather than scratch if the file should
|
|
33
|
+
# outlive the run, and follow it with the step that consumes it -- a conversion, a
|
|
34
|
+
# storage.copy to a landing bucket, or the load that reads it back somewhere else.
|
|
35
|
+
|
|
36
|
+
format: dirigent/v1
|
|
37
|
+
kind: pipeline
|
|
38
|
+
code: sql-query-to-storage
|
|
39
|
+
name: Write a query result to storage
|
|
40
|
+
description: Read a table with sql.query and hand its rows to storage.write, which is how a result becomes a file.
|
|
41
|
+
|
|
42
|
+
tags: [sql, storage]
|
|
43
|
+
|
|
44
|
+
requires:
|
|
45
|
+
blocks:
|
|
46
|
+
- sql.execute
|
|
47
|
+
- sql.query
|
|
48
|
+
- storage.write
|
|
49
|
+
|
|
50
|
+
connections:
|
|
51
|
+
work-db:
|
|
52
|
+
kind: sql
|
|
53
|
+
config:
|
|
54
|
+
# Relative, so the database lands in this run's work directory and needs nothing on the
|
|
55
|
+
# host. The same connection as the roundtrip example, for the same reason.
|
|
56
|
+
url: sqlite+aiosqlite:///demo.db
|
|
57
|
+
|
|
58
|
+
steps:
|
|
59
|
+
build:
|
|
60
|
+
block: sql.execute
|
|
61
|
+
config:
|
|
62
|
+
connection: work-db
|
|
63
|
+
statements:
|
|
64
|
+
- CREATE TABLE reading (id INTEGER PRIMARY KEY, site TEXT NOT NULL, value REAL NOT NULL)
|
|
65
|
+
- INSERT INTO reading (id, site, value) VALUES (1, :north, 2.5), (2, :south, 4.0), (3, :north, 3.5)
|
|
66
|
+
params:
|
|
67
|
+
north: north
|
|
68
|
+
south: south
|
|
69
|
+
|
|
70
|
+
extract:
|
|
71
|
+
block: sql.query
|
|
72
|
+
depends_on: [build]
|
|
73
|
+
config:
|
|
74
|
+
connection: work-db
|
|
75
|
+
sql: SELECT id, site, value FROM reading ORDER BY id
|
|
76
|
+
max_rows: 100
|
|
77
|
+
|
|
78
|
+
save:
|
|
79
|
+
block: storage.write
|
|
80
|
+
depends_on: [extract]
|
|
81
|
+
config:
|
|
82
|
+
# A file inside the run's scratch space, so it is cleaned up with the run. A bucket URI
|
|
83
|
+
# here is what makes the result outlive it.
|
|
84
|
+
target: "${run.scratch}/reading.json"
|
|
85
|
+
value: "${steps.extract.output.rows}"
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# A database built, written and read back inside one run, with every value bound.
|
|
2
|
+
#
|
|
3
|
+
# NEEDS NOTHING: no network, no daemon, no server, no allowlist. The database is a SQLite file
|
|
4
|
+
# in the run's own work directory, and both sql blocks are ordinary.
|
|
5
|
+
# dg run --local examples/sql/sql-sqlite-roundtrip.yaml
|
|
6
|
+
# docs/sql.md is the family's home.
|
|
7
|
+
#
|
|
8
|
+
# Three hops, and what each one hands on:
|
|
9
|
+
# build creates the table and inserts two rows, all in one transaction. Output: how many
|
|
10
|
+
# rows each statement touched.
|
|
11
|
+
# read selects those rows back with a bound parameter. Output: the rows, their count and
|
|
12
|
+
# the column names.
|
|
13
|
+
# report a constant step that reads the rows, proving they are an ordinary reference any
|
|
14
|
+
# downstream step can name.
|
|
15
|
+
#
|
|
16
|
+
# THE CONNECTION HOLDS THE DATABASE, THE DOCUMENT HOLDS THE STATEMENTS. A step never writes a
|
|
17
|
+
# URL: it names a sql connection, and the connection is what carries the database and the
|
|
18
|
+
# sealed password that opens it. Moving a pipeline from staging to production is then an edit
|
|
19
|
+
# to one connection and to no pipeline. This one is carried in the document because a --local
|
|
20
|
+
# run has no instance to hold it; on a server it would be created once with
|
|
21
|
+
# `dg connection create sql ...` and the document would name it and stop there.
|
|
22
|
+
#
|
|
23
|
+
# WHY THE URL IS RELATIVE. A sqlite database written with a relative path is resolved against
|
|
24
|
+
# the run's work directory, so this run builds its database, uses it, and leaves it with
|
|
25
|
+
# everything else the run wrote. That is what makes this example runnable anywhere with no
|
|
26
|
+
# setup at all; a real pipeline names a database that already exists.
|
|
27
|
+
#
|
|
28
|
+
# BINDING IS THE POINT. Not one value is written into the SQL text. Every one is a named
|
|
29
|
+
# parameter -- :site, :value -- and the parameters travel to the database beside the statement,
|
|
30
|
+
# so a value that reads as SQL is compared as a string and matches nothing. ${...} resolves
|
|
31
|
+
# into params, never into sql, and that is the whole rule.
|
|
32
|
+
#
|
|
33
|
+
# ONE TRANSACTION. sql.execute takes a list, and the list is one transaction: the CREATE and
|
|
34
|
+
# both INSERTs commit together or not at all. A DDL statement reports -1 because SQLite does
|
|
35
|
+
# not say how many rows a CREATE touched, which is what -1 means everywhere in row_counts.
|
|
36
|
+
#
|
|
37
|
+
# TO MAKE IT YOURS: point the connection at a real database, drop the build step, and keep the
|
|
38
|
+
# read step -- reading a table with bound parameters is what most pipelines actually want.
|
|
39
|
+
|
|
40
|
+
format: dirigent/v1
|
|
41
|
+
kind: pipeline
|
|
42
|
+
code: sql-sqlite-roundtrip
|
|
43
|
+
name: Write a table and read it back
|
|
44
|
+
description: Create a SQLite database in the run's work directory, insert rows with bound parameters, and read them back.
|
|
45
|
+
|
|
46
|
+
tags: [sql]
|
|
47
|
+
|
|
48
|
+
requires:
|
|
49
|
+
blocks:
|
|
50
|
+
- sql.execute
|
|
51
|
+
- sql.query
|
|
52
|
+
- value.const
|
|
53
|
+
|
|
54
|
+
connections:
|
|
55
|
+
work-db:
|
|
56
|
+
kind: sql
|
|
57
|
+
config:
|
|
58
|
+
# Relative, so it lands in this run's work directory and needs nothing on the host. The
|
|
59
|
+
# driver is written out because these blocks speak to a database over an async one, and
|
|
60
|
+
# a url without one is refused when the connection is written.
|
|
61
|
+
url: sqlite+aiosqlite:///demo.db
|
|
62
|
+
|
|
63
|
+
params:
|
|
64
|
+
type: object
|
|
65
|
+
additionalProperties: false
|
|
66
|
+
properties:
|
|
67
|
+
site:
|
|
68
|
+
type: string
|
|
69
|
+
default: north
|
|
70
|
+
description: The site the read step asks for, bound as a parameter rather than spliced into the SQL.
|
|
71
|
+
|
|
72
|
+
steps:
|
|
73
|
+
build:
|
|
74
|
+
block: sql.execute
|
|
75
|
+
config:
|
|
76
|
+
connection: work-db
|
|
77
|
+
statements:
|
|
78
|
+
- CREATE TABLE reading (id INTEGER PRIMARY KEY, site TEXT NOT NULL, value REAL NOT NULL)
|
|
79
|
+
# Both inserts share one params map, which is why they name different columns of it
|
|
80
|
+
# rather than repeating a value.
|
|
81
|
+
- INSERT INTO reading (id, site, value) VALUES (1, :north, :north_value)
|
|
82
|
+
- INSERT INTO reading (id, site, value) VALUES (2, :south, :south_value)
|
|
83
|
+
params:
|
|
84
|
+
north: north
|
|
85
|
+
north_value: 2.5
|
|
86
|
+
south: south
|
|
87
|
+
south_value: 4.0
|
|
88
|
+
|
|
89
|
+
read:
|
|
90
|
+
block: sql.query
|
|
91
|
+
depends_on: [build]
|
|
92
|
+
config:
|
|
93
|
+
connection: work-db
|
|
94
|
+
# One statement. A document that needs two writes two steps, or uses sql.execute.
|
|
95
|
+
sql: SELECT id, site, value FROM reading WHERE site = :site ORDER BY id
|
|
96
|
+
params:
|
|
97
|
+
# The pipeline parameter reaches the database as a bound value and never as text in
|
|
98
|
+
# the statement above, which is what makes it safe to let somebody else set it.
|
|
99
|
+
site: "${params.site}"
|
|
100
|
+
# Two rows fit comfortably; a query that could return more than this fails the step
|
|
101
|
+
# rather than truncating, so a wider result is one to narrow or to raise this for.
|
|
102
|
+
max_rows: 10
|
|
103
|
+
|
|
104
|
+
report:
|
|
105
|
+
block: value.const
|
|
106
|
+
depends_on: [read]
|
|
107
|
+
config:
|
|
108
|
+
value:
|
|
109
|
+
# A list of objects keyed by column name, which is the shape every downstream step
|
|
110
|
+
# sees: a transform maps it, a validate.schema checks it, an http.request posts it.
|
|
111
|
+
rows: "${steps.read.output.rows}"
|
|
112
|
+
# The count is there whether the rows were carried inline or saved to storage.
|
|
113
|
+
row_count: "${steps.read.output.row_count}"
|
|
114
|
+
columns: "${steps.read.output.columns}"
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
-- What the warehouse in infra/compose.sql.yaml holds: two roles and the table the sql
|
|
2
|
+
-- examples on this shelf query. The stack mounts this file; nothing else reads it.
|
|
3
|
+
--
|
|
4
|
+
-- Run by the postgres image's entrypoint on first start, connected to the warehouse database
|
|
5
|
+
-- as its superuser. \getenv reads each role's password from the environment, so no password
|
|
6
|
+
-- is written here.
|
|
7
|
+
|
|
8
|
+
\getenv reader_password WAREHOUSE_READER_PASSWORD
|
|
9
|
+
\getenv writer_password WAREHOUSE_WRITER_PASSWORD
|
|
10
|
+
|
|
11
|
+
-- The shape examples/sql/sql-postgres-readonly.yaml selects: one row per reading, narrowed by
|
|
12
|
+
-- site and by the run's window.
|
|
13
|
+
CREATE TABLE reading (
|
|
14
|
+
id bigint GENERATED ALWAYS AS IDENTITY PRIMARY KEY,
|
|
15
|
+
site text NOT NULL,
|
|
16
|
+
seen timestamptz NOT NULL,
|
|
17
|
+
value double precision NOT NULL
|
|
18
|
+
);
|
|
19
|
+
|
|
20
|
+
-- Seeded relative to now, so a run whose window is the last hour or the last day finds rows
|
|
21
|
+
-- however long ago the volume was created.
|
|
22
|
+
INSERT INTO reading (site, seen, value) VALUES
|
|
23
|
+
('north', now() - interval '3 hours', 11.5),
|
|
24
|
+
('north', now() - interval '2 hours', 12.25),
|
|
25
|
+
('north', now() - interval '1 hour', 10.75),
|
|
26
|
+
('north', now() - interval '20 minutes', 13.0),
|
|
27
|
+
('south', now() - interval '90 minutes', 8.5),
|
|
28
|
+
('south', now() - interval '30 minutes', 9.25);
|
|
29
|
+
|
|
30
|
+
-- SELECT and nothing else. A statement that writes through this role is refused by the
|
|
31
|
+
-- database itself, whatever the connection or the document asks for.
|
|
32
|
+
CREATE ROLE reader LOGIN PASSWORD :'reader_password';
|
|
33
|
+
GRANT SELECT ON ALL TABLES IN SCHEMA public TO reader;
|
|
34
|
+
ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT ON TABLES TO reader;
|
|
35
|
+
|
|
36
|
+
-- The role a loading pipeline uses: the four statements, and the sequence an identity column
|
|
37
|
+
-- draws from.
|
|
38
|
+
CREATE ROLE writer LOGIN PASSWORD :'writer_password';
|
|
39
|
+
GRANT SELECT, INSERT, UPDATE, DELETE ON ALL TABLES IN SCHEMA public TO writer;
|
|
40
|
+
GRANT USAGE ON ALL SEQUENCES IN SCHEMA public TO writer;
|
|
41
|
+
ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT, INSERT, UPDATE, DELETE ON TABLES TO writer;
|
|
42
|
+
ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT USAGE ON SEQUENCES TO writer;
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Transform examples
|
|
2
|
+
|
|
3
|
+
These pipelines reshape data with the transform verbs -- `transform`, `map`, `filter` and
|
|
4
|
+
`convert` -- which run a program or a codec against a value and execute nothing on the
|
|
5
|
+
worker. None of them needs an allowlist entry, a network, or anything installed first.
|
|
6
|
+
[docs/transforms.md](../../docs/transforms.md) is the page behind them: a verb is a contract
|
|
7
|
+
with a promise its frame enforces, and a kind is the engine that keeps it.
|
|
8
|
+
|
|
9
|
+
A file that stars an engine is prefixed with its kind, the way a block id's
|
|
10
|
+
`<verb>.<kind>` names it: `jq-` for the jq engines, `std-` for the `convert.std` codec. The
|
|
11
|
+
rest are named for the format they round-trip.
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
dg run --local examples/transform/jq-reshape.yaml
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
## Pipelines
|
|
18
|
+
|
|
19
|
+
| File | What it teaches |
|
|
20
|
+
| --- | --- |
|
|
21
|
+
| [jq-reshape.yaml](jq-reshape.yaml) | The whole-value reshape: one jq program between steps, then a fan-out that reads its list once per element. |
|
|
22
|
+
| [jq-filter-and-map.yaml](jq-filter-and-map.yaml) | The two element-wise verbs against the whole-value one: a subset, a list of the same length, and a reshape. |
|
|
23
|
+
| [jq-group-and-aggregate.yaml](jq-group-and-aggregate.yaml) | `group_by` and arithmetic: per-group sums and means with a total beside them. |
|
|
24
|
+
| [jq-join-two-sources.yaml](jq-join-two-sources.yaml) | Two upstream outputs composed into one inline input, joined with `INDEX`, unmatched rows kept with a null name. |
|
|
25
|
+
| [jq-stream-through-storage.yaml](jq-stream-through-storage.yaml) | The two doors on storage: `storage.write` puts a value in an object and `storage.read` brings one back, with the reshape between them. |
|
|
26
|
+
| [std-convert-fan-out.yaml](std-convert-fan-out.yaml) | The codec: csv to json, reshaped with jq, fanned out over the regions, and written back as csv. |
|
|
27
|
+
| [csv-report.yaml](csv-report.yaml) | Records shaped into flat rows and written as a csv artifact, with nothing on the allowlist. |
|
|
28
|
+
| [ndjson-round-trip.yaml](ndjson-round-trip.yaml) | ndjson: a JSON array re-spelled one record per line, and read back. |
|
|
29
|
+
| [yaml-config-to-json.yaml](yaml-config-to-json.yaml) | yaml: one document is one value, so a config becomes the object it describes, and comes back a document. |
|
|
30
|
+
| [xml-feed-to-ndjson.yaml](xml-feed-to-ndjson.yaml) | xml: a feed's elements as one record per line, the mapping that makes attributes and children keys, and the whole document as one object. |
|
|
31
|
+
| [parquet-round-trip.yaml](parquet-round-trip.yaml) | Records to parquet and back, the types surviving where csv would flatten them to strings (needs `dirigent-parquet`). |
|
|
32
|
+
|
|
33
|
+
Every program on this shelf is reference-free, so jq compiles them when the document is
|
|
34
|
+
applied and a syntax error is an issue beside every other one the document has. That is why
|
|
35
|
+
the data is always the step's `input` and never spliced into the program: a config carrying
|
|
36
|
+
a `${...}` cannot be checked until the run.
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# A csv report, written where a person can fetch it.
|
|
2
|
+
#
|
|
3
|
+
# Four hops turn records into a file: a fixed value stands in for whatever produced the
|
|
4
|
+
# records, a jq program shapes them into flat rows, storage.write puts those rows in the
|
|
5
|
+
# run's scratch as json, and convert.std re-encodes that object as csv -- so the run's
|
|
6
|
+
# Output tab lists a report.csv somebody can read back. Nothing here is on the allowlist.
|
|
7
|
+
#
|
|
8
|
+
# convert.std works on storage objects, the way storage.copy does: it names a source uri
|
|
9
|
+
# and a target uri and never carries a value. The write is how the rows a step is holding
|
|
10
|
+
# become an object it can read.
|
|
11
|
+
#
|
|
12
|
+
# The rows must be flat: a nested value has no csv spelling and is refused naming the row
|
|
13
|
+
# and the key, so the jq step is where a list becomes one joined cell.
|
|
14
|
+
#
|
|
15
|
+
# dg run --local examples/transform/csv-report.yaml --keep
|
|
16
|
+
|
|
17
|
+
format: dirigent/v1
|
|
18
|
+
kind: pipeline
|
|
19
|
+
code: csv-report
|
|
20
|
+
name: A report as csv
|
|
21
|
+
description: Shape records into flat rows, store them as json, and re-encode that as a csv artifact.
|
|
22
|
+
|
|
23
|
+
tags: [transform, storage]
|
|
24
|
+
|
|
25
|
+
steps:
|
|
26
|
+
readings:
|
|
27
|
+
block: value.const
|
|
28
|
+
config:
|
|
29
|
+
value:
|
|
30
|
+
- { station: st-1, region: east, celsius: 4.5, tags: [ok, new] }
|
|
31
|
+
- { station: st-3, region: west, celsius: 1.2, tags: [ok] }
|
|
32
|
+
- { station: st-4, region: north, celsius: -3.4, tags: [] }
|
|
33
|
+
rows:
|
|
34
|
+
block: transform.jq
|
|
35
|
+
depends_on: [readings]
|
|
36
|
+
config:
|
|
37
|
+
input: ${steps.readings.output.value}
|
|
38
|
+
# Flatten for the codec: the tag list becomes one joined cell, because only this
|
|
39
|
+
# program knows what the separator should mean.
|
|
40
|
+
program: |
|
|
41
|
+
[.[] | {station, region, celsius, tags: (.tags | join(" "))}]
|
|
42
|
+
store:
|
|
43
|
+
block: storage.write
|
|
44
|
+
depends_on: [rows]
|
|
45
|
+
config:
|
|
46
|
+
target: ${run.scratch}/rows.json
|
|
47
|
+
value: ${steps.rows.output.value}
|
|
48
|
+
report:
|
|
49
|
+
block: convert.std
|
|
50
|
+
depends_on: [store]
|
|
51
|
+
config:
|
|
52
|
+
source: ${steps.store.output.uri}
|
|
53
|
+
target: ${run.scratch}/report.csv
|
|
54
|
+
from: json
|
|
55
|
+
to: csv
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# The two constrained verbs, and what they buy over doing everything in one program.
|
|
2
|
+
#
|
|
3
|
+
# filter.jq and map.jq run a jq program once per element of a list. Neither can do what the
|
|
4
|
+
# other does: a filter's program answers true or false and the frame keeps the element it
|
|
5
|
+
# was given, so a filtered list is a subset with nothing modified, and a map's program
|
|
6
|
+
# returns one replacement per element, so a mapped list is exactly as long as its input.
|
|
7
|
+
# Written as one transform.jq program the same work is one line -- and nothing but reading
|
|
8
|
+
# it tells you whether it dropped rows, added rows, or edited them in place.
|
|
9
|
+
#
|
|
10
|
+
# The steps read: keep the active readings, convert each one to fahrenheit, then summarise.
|
|
11
|
+
# The third step is a transform because it genuinely is one: it changes the shape of the
|
|
12
|
+
# whole value rather than working element by element, and that is the verb that says so.
|
|
13
|
+
#
|
|
14
|
+
# The per-element contracts are strict on purpose. A map program emitting nothing, or two
|
|
15
|
+
# things, for one element is refused rather than quietly changing the length, and a filter
|
|
16
|
+
# program answering 0 or "" is refused rather than read as false: jq's truthiness is not
|
|
17
|
+
# applied, so a program meaning "has readings" writes .count > 0.
|
|
18
|
+
#
|
|
19
|
+
# The input is written inline so the example needs no network. In a real pipeline it is a
|
|
20
|
+
# reference to an upstream step's output, which is how a value that lives in storage arrives:
|
|
21
|
+
# through a storage.read of its own.
|
|
22
|
+
#
|
|
23
|
+
# dg run --local examples/transform/jq-filter-and-map.yaml
|
|
24
|
+
|
|
25
|
+
format: dirigent/v1
|
|
26
|
+
kind: pipeline
|
|
27
|
+
code: jq-filter-and-map
|
|
28
|
+
name: Filter and map with jq
|
|
29
|
+
description: Keep the active readings with filter.jq, convert each with map.jq, and summarise them.
|
|
30
|
+
|
|
31
|
+
tags: [transform]
|
|
32
|
+
|
|
33
|
+
requires:
|
|
34
|
+
blocks:
|
|
35
|
+
- filter.jq
|
|
36
|
+
- map.jq
|
|
37
|
+
- transform.jq
|
|
38
|
+
|
|
39
|
+
steps:
|
|
40
|
+
active:
|
|
41
|
+
block: filter.jq
|
|
42
|
+
config:
|
|
43
|
+
input:
|
|
44
|
+
- { station: st-1, region: east, status: active, celsius: 4.5 }
|
|
45
|
+
- { station: st-2, region: east, status: retired, celsius: 11.0 }
|
|
46
|
+
- { station: st-3, region: west, status: active, celsius: 1.2 }
|
|
47
|
+
- { station: st-4, region: north, status: active, celsius: -3.4 }
|
|
48
|
+
# The answer is the verdict and nothing else: the elements that come out are the ones
|
|
49
|
+
# that went in, unmodified, because the frame keeps them rather than the program.
|
|
50
|
+
program: |
|
|
51
|
+
.status == "active"
|
|
52
|
+
|
|
53
|
+
fahrenheit:
|
|
54
|
+
block: map.jq
|
|
55
|
+
depends_on: [active]
|
|
56
|
+
config:
|
|
57
|
+
input: "${steps.active.output.value}"
|
|
58
|
+
# One output per element, so this list is as long as the one above it.
|
|
59
|
+
program: |
|
|
60
|
+
{station, region, fahrenheit: (.celsius * 9 / 5 + 32 | round)}
|
|
61
|
+
|
|
62
|
+
summary:
|
|
63
|
+
block: transform.jq
|
|
64
|
+
depends_on: [fahrenheit]
|
|
65
|
+
config:
|
|
66
|
+
# A whole-value reshape rather than an element-wise one, which is why it is the third
|
|
67
|
+
# verb and not one of the other two.
|
|
68
|
+
input: "${steps.fahrenheit.output.value}"
|
|
69
|
+
program: |
|
|
70
|
+
{stations: length, regions: (map(.region) | unique), warmest: (max_by(.fahrenheit) | .station)}
|