dirigent-examples 0.15.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_examples/__init__.py +22 -0
- dirigent_examples/py.typed +0 -0
- dirigent_examples/shelves/README.md +299 -0
- dirigent_examples/shelves/composition/README.md +18 -0
- dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
- dirigent_examples/shelves/composition/composition-child.yaml +64 -0
- dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
- dirigent_examples/shelves/connections.yaml +52 -0
- dirigent_examples/shelves/demo/README.md +19 -0
- dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
- dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
- dirigent_examples/shelves/demo/requires.yaml +65 -0
- dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
- dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
- dirigent_examples/shelves/docker/README.md +29 -0
- dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
- dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
- dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
- dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
- dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
- dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
- dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
- dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
- dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
- dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
- dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
- dirigent_examples/shelves/execute/README.md +16 -0
- dirigent_examples/shelves/execute/long-log.yaml +89 -0
- dirigent_examples/shelves/failure/README.md +20 -0
- dirigent_examples/shelves/failure/error-handler.yaml +89 -0
- dirigent_examples/shelves/failure/optional-step.yaml +82 -0
- dirigent_examples/shelves/failure/retries.yaml +75 -0
- dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
- dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
- dirigent_examples/shelves/git/README.md +32 -0
- dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
- dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
- dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
- dirigent_examples/shelves/graph/README.md +22 -0
- dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
- dirigent_examples/shelves/graph/fan-in.yaml +80 -0
- dirigent_examples/shelves/graph/fan-out.yaml +66 -0
- dirigent_examples/shelves/graph/linear.yaml +66 -0
- dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
- dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
- dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
- dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
- dirigent_examples/shelves/hello-world.yaml +30 -0
- dirigent_examples/shelves/open-data/README.md +67 -0
- dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
- dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
- dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
- dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
- dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
- dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
- dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
- dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
- dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
- dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
- dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
- dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
- dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
- dirigent_examples/shelves/patterns/README.md +144 -0
- dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
- dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
- dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
- dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
- dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
- dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
- dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
- dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
- dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
- dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
- dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
- dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
- dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
- dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
- dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
- dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
- dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
- dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
- dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
- dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
- dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
- dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
- dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
- dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
- dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
- dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
- dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
- dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
- dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
- dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
- dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
- dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
- dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
- dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
- dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
- dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
- dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
- dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
- dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
- dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
- dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
- dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
- dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
- dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
- dirigent_examples/shelves/python/README.md +31 -0
- dirigent_examples/shelves/python/apply_and_run.py +52 -0
- dirigent_examples/shelves/python/ci_gate.py +76 -0
- dirigent_examples/shelves/python/connections.py +61 -0
- dirigent_examples/shelves/python/error_handling.py +84 -0
- dirigent_examples/shelves/python/follow_logs.py +39 -0
- dirigent_examples/shelves/python/list_and_filter.py +52 -0
- dirigent_examples/shelves/queues/README.md +59 -0
- dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
- dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
- dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
- dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
- dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
- dirigent_examples/shelves/recipes/README.md +130 -0
- dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
- dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
- dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
- dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
- dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
- dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
- dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
- dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
- dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
- dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
- dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
- dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
- dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
- dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
- dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
- dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
- dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
- dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
- dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
- dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
- dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
- dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
- dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
- dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
- dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
- dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
- dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
- dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
- dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
- dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
- dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
- dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
- dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
- dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
- dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
- dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
- dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
- dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
- dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
- dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
- dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
- dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
- dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
- dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
- dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
- dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
- dirigent_examples/shelves/s3/README.md +34 -0
- dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
- dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
- dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
- dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
- dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
- dirigent_examples/shelves/schemas/README.md +36 -0
- dirigent_examples/shelves/schemas/echo-reading.json +18 -0
- dirigent_examples/shelves/schemas/ou-record.json +13 -0
- dirigent_examples/shelves/schemas/station-reading.json +13 -0
- dirigent_examples/shelves/sensors/README.md +16 -0
- dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
- dirigent_examples/shelves/sensors/time-window.yaml +61 -0
- dirigent_examples/shelves/sql/README.md +52 -0
- dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
- dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
- dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
- dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
- dirigent_examples/shelves/sql/warehouse.sql +42 -0
- dirigent_examples/shelves/transform/README.md +36 -0
- dirigent_examples/shelves/transform/csv-report.yaml +55 -0
- dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
- dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
- dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
- dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
- dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
- dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
- dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
- dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
- dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
- dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
- dirigent_examples/shelves/triggers/README.md +45 -0
- dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
- dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
- dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
- dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
- dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
- dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
- dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
- dirigent_examples/shelves/validate/README.md +31 -0
- dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
- dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
- dirigent_examples-0.15.0.dist-info/METADATA +21 -0
- dirigent_examples-0.15.0.dist-info/RECORD +216 -0
- dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
- dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
- dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# Recipes
|
|
2
|
+
|
|
3
|
+
One question per file. Every recipe here is a small, complete, runnable answer to something
|
|
4
|
+
a person doing data work actually asks -- how do I group and sum, how do I join two lists,
|
|
5
|
+
what survives a parquet round trip, where do credentials live -- and its header comment is
|
|
6
|
+
the lesson: what it does, what goes in and what comes out, why each step is configured the
|
|
7
|
+
way it is, and what to change to make it yours.
|
|
8
|
+
|
|
9
|
+
Nothing here needs infrastructure. The data is written inline, generated by a jq program, or
|
|
10
|
+
fetched from [Postman Echo](https://postman-echo.com), a public service that answers with
|
|
11
|
+
the request it was given. Nothing needs the unsafe-block allowlist: no recipe on this shelf
|
|
12
|
+
runs code on a worker.
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
dg run --local examples/recipes/jq-group-by-and-sum.yaml
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
One recipe is handed something: [schema-referenced.yaml](schema-referenced.yaml) names a
|
|
19
|
+
schema the instance holds, so a local run is given it, and without it the run is refused
|
|
20
|
+
before anything executes -- which is the point that recipe makes.
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
dg run --local --schema examples/schemas/ou-record.json examples/recipes/schema-referenced.yaml
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Four of them fail on purpose, and their headers say so. Failing is the lesson in two of them
|
|
27
|
+
and a parameter away in the others:
|
|
28
|
+
|
|
29
|
+
| Recipe | Ends as | Because |
|
|
30
|
+
| --- | --- | --- |
|
|
31
|
+
| [schema-refuses-then-rule.yaml](schema-refuses-then-rule.yaml) | `failed` | The gate refuses a level of 0, the load is skipped, and the `one_failed` handler runs |
|
|
32
|
+
| [reconcile-two-sources.yaml](reconcile-two-sources.yaml) | `failed` | Three rows differ beyond the tolerance and the limit is two; `-p max_differing=5` passes |
|
|
33
|
+
| [http-success-status-list.yaml](http-success-status-list.yaml) | `succeeded` | 418 is listed as success; `-p status=500` is not, and fails |
|
|
34
|
+
| [csv-header-rules.yaml](csv-header-rules.yaml) | `succeeded` | `-p header=station,station,celsius` provokes the codec's refusal |
|
|
35
|
+
|
|
36
|
+
## jq: reshaping a value
|
|
37
|
+
|
|
38
|
+
| File | The question it answers |
|
|
39
|
+
| --- | --- |
|
|
40
|
+
| [jq-group-by-and-sum.yaml](jq-group-by-and-sum.yaml) | How do I total a column per group, with a grand total that still adds up? |
|
|
41
|
+
| [jq-join-two-lists.yaml](jq-join-two-lists.yaml) | How do I join two lists on a key, without losing the rows that match nothing? |
|
|
42
|
+
| [jq-pivot-wide-to-long.yaml](jq-pivot-wide-to-long.yaml) | How do I melt a column-per-measure table into one record per measurement? |
|
|
43
|
+
| [jq-long-to-wide.yaml](jq-long-to-wide.yaml) | How do I widen it back, with an explicit fill in every unreported cell? |
|
|
44
|
+
| [jq-dedupe-by-key.yaml](jq-dedupe-by-key.yaml) | Which row wins when a key repeats -- first, last, or newest by timestamp? |
|
|
45
|
+
| [jq-window-dates.yaml](jq-window-dates.yaml) | How do I build a window of days and ISO weeks, and why is `strptime` alone not a date? |
|
|
46
|
+
| [jq-nested-to-flat.yaml](jq-nested-to-flat.yaml) | How do I flatten nested records into the flat ones a csv or parquet writer accepts? |
|
|
47
|
+
| [jq-defaults-and-nulls.yaml](jq-defaults-and-nulls.yaml) | How do I default a field without `//` turning every `false` back into `true`? |
|
|
48
|
+
| [jq-string-cleaning.yaml](jq-string-cleaning.yaml) | How do I trim, case-fold and extract the strings a spreadsheet export brings in? |
|
|
49
|
+
| [jq-top-n.yaml](jq-top-n.yaml) | How do I take the top N without hiding the ties or the tail? |
|
|
50
|
+
| [jq-running-totals.yaml](jq-running-totals.yaml) | How do I add cumulative sums, deltas and a moving average to an ordered series? |
|
|
51
|
+
| [jq-validate-in-jq-vs-schema.yaml](jq-validate-in-jq-vs-schema.yaml) | When do I sort a batch in jq, and when do I gate it with a schema? |
|
|
52
|
+
|
|
53
|
+
## map and filter: the element-wise verbs
|
|
54
|
+
|
|
55
|
+
| File | The question it answers |
|
|
56
|
+
| --- | --- |
|
|
57
|
+
| [map-enrich-with-lookup.yaml](map-enrich-with-lookup.yaml) | How do I enrich every row from a lookup table a `map.jq` program cannot see? |
|
|
58
|
+
| [filter-by-predicate.yaml](filter-by-predicate.yaml) | How do I write a predicate that answers true or false, since truthiness is not applied? |
|
|
59
|
+
| [filter-then-map-then-reduce.yaml](filter-then-map-then-reduce.yaml) | Which verb is which, in the order a pipeline usually needs them? |
|
|
60
|
+
|
|
61
|
+
## Converting between formats
|
|
62
|
+
|
|
63
|
+
| File | The question it answers |
|
|
64
|
+
| --- | --- |
|
|
65
|
+
| [csv-to-ndjson.yaml](csv-to-ndjson.yaml) | How do I read a csv, and who gives the string cells their types? |
|
|
66
|
+
| [ndjson-to-parquet.yaml](ndjson-to-parquet.yaml) | How do I write records as parquet, and read the file back to see what landed? |
|
|
67
|
+
| [parquet-round-trip-types.yaml](parquet-round-trip-types.yaml) | Which types survive a parquet round trip, and which does a csv flatten? |
|
|
68
|
+
| [csv-header-rules.yaml](csv-header-rules.yaml) | What are the header rules, and which two headers are refused outright? |
|
|
69
|
+
| [json-to-csv-flattening.yaml](json-to-csv-flattening.yaml) | How do I turn a nested API payload into a csv somebody opens in a spreadsheet? |
|
|
70
|
+
|
|
71
|
+
## Validating a shape
|
|
72
|
+
|
|
73
|
+
| File | The question it answers |
|
|
74
|
+
| --- | --- |
|
|
75
|
+
| [schema-carried.yaml](schema-carried.yaml) | How do I carry the shape in the document, so the pipeline runs with nothing handed to it? |
|
|
76
|
+
| [schema-referenced.yaml](schema-referenced.yaml) | How do I name a schema the instance holds, and declare that dependency? |
|
|
77
|
+
| [schema-formats.yaml](schema-formats.yaml) | Which `format` keywords actually assert here, and what does a pack contribute? |
|
|
78
|
+
| [schema-refuses-then-rule.yaml](schema-refuses-then-rule.yaml) | What runs after a gate refuses -- and what does not? |
|
|
79
|
+
|
|
80
|
+
## Storage
|
|
81
|
+
|
|
82
|
+
| File | The question it answers |
|
|
83
|
+
| --- | --- |
|
|
84
|
+
| [storage-write-then-read.yaml](storage-write-then-read.yaml) | How do I park a payload in storage and pick it up again a step later? |
|
|
85
|
+
| [storage-copy-dated-archive.yaml](storage-copy-dated-archive.yaml) | How do I keep a dated copy, and what should the key layout be? |
|
|
86
|
+
| [storage-exists-gate.yaml](storage-exists-gate.yaml) | How do I wait for an object, and what stops me reading a half-written one? |
|
|
87
|
+
| [large-output-to-storage.yaml](large-output-to-storage.yaml) | What happens to a payload past the inline threshold, and what should I do about it? |
|
|
88
|
+
| [storage-manifest-of-a-fan-out.yaml](storage-manifest-of-a-fan-out.yaml) | How do I write one file per item and then say, in the run, what landed? |
|
|
89
|
+
| [report-to-file.yaml](report-to-file.yaml) | How do I render a page, write it to a file, and read it back to see that it landed? |
|
|
90
|
+
|
|
91
|
+
## HTTP and webhooks
|
|
92
|
+
|
|
93
|
+
| File | The question it answers |
|
|
94
|
+
| --- | --- |
|
|
95
|
+
| [http-get-with-query.yaml](http-get-with-query.yaml) | How do I build a query without hand-writing a URL, and which field holds the answer? |
|
|
96
|
+
| [http-post-json-echo.yaml](http-post-json-echo.yaml) | How do I POST a body assembled upstream, as structure rather than as text? |
|
|
97
|
+
| [http-fetch-validate-post.yaml](http-fetch-validate-post.yaml) | How do I fetch, check what came back, reshape it, check what I made, and post it on? |
|
|
98
|
+
| [http-headers-and-auth-connection.yaml](http-headers-and-auth-connection.yaml) | Where do credentials live, and what belongs in a header map instead? |
|
|
99
|
+
| [http-save-body-to-storage.yaml](http-save-body-to-storage.yaml) | How do I keep a response body as an object somebody else can fetch? |
|
|
100
|
+
| [http-post-file-from-storage.yaml](http-post-file-from-storage.yaml) | How do I POST a file that lives in storage? |
|
|
101
|
+
| [http-success-status-list.yaml](http-success-status-list.yaml) | How do I accept a non-2xx that is a real answer, without silencing failures? |
|
|
102
|
+
| [http-follow-redirects.yaml](http-follow-redirects.yaml) | Why is a redirect not followed by default, and what do I lose when it is? |
|
|
103
|
+
| [http-timeout-override.yaml](http-timeout-override.yaml) | Which of the three deadlines around a call am I actually setting? |
|
|
104
|
+
| [webhook-post-hmac.yaml](webhook-post-hmac.yaml) | What exactly gets signed, with what, and what does a receiver check? |
|
|
105
|
+
| [webhook-post-summary.yaml](webhook-post-summary.yaml) | What belongs in a notification body somebody has to act on? |
|
|
106
|
+
| [report-to-webhook.yaml](report-to-webhook.yaml) | How do I POST a rendered page to a receiver, and what travels beside it? |
|
|
107
|
+
|
|
108
|
+
## Whole small pipelines
|
|
109
|
+
|
|
110
|
+
| File | The question it answers |
|
|
111
|
+
| --- | --- |
|
|
112
|
+
| [etl-csv-clean-validate-parquet.yaml](etl-csv-clean-validate-parquet.yaml) | What does a whole small ETL look like, from a messy csv to a checked parquet file? |
|
|
113
|
+
| [report-daily-digest.yaml](report-daily-digest.yaml) | How do I compute a day's numbers once and let the csv, the digest and the notification agree? |
|
|
114
|
+
| [reconcile-two-sources.yaml](reconcile-two-sources.yaml) | How do I say exactly how two systems disagree, in four buckets and with a tolerance? |
|
|
115
|
+
| [pagination-by-fan-out.yaml](pagination-by-fan-out.yaml) | How do I fetch several pages at once, and what does `for_each` not let me do? |
|
|
116
|
+
|
|
117
|
+
## The run's own report
|
|
118
|
+
|
|
119
|
+
| File | The question it answers |
|
|
120
|
+
| --- | --- |
|
|
121
|
+
| [report-built-in.yaml](report-built-in.yaml) | How do I get a page about the run itself, without writing a line of template? |
|
|
122
|
+
| [http-post-report.yaml](http-post-report.yaml) | How do I write that page myself, and which of the run's facts may it read? |
|
|
123
|
+
|
|
124
|
+
## The pages behind them
|
|
125
|
+
|
|
126
|
+
[docs/jq.md](../../docs/jq.md) teaches the language, [docs/transforms.md](../../docs/transforms.md)
|
|
127
|
+
the four verbs and their frames, [docs/json-schema.md](../../docs/json-schema.md) the shapes,
|
|
128
|
+
[docs/reports.md](../../docs/reports.md) every fact a report template may read,
|
|
129
|
+
and [docs/blocks.md](../../docs/blocks.md) is generated from the live catalog, so it is the
|
|
130
|
+
honest answer to what a block's config actually takes.
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
# The four rules the csv codec keeps about headers, and the two headers it refuses.
|
|
2
|
+
#
|
|
3
|
+
# In: a csv whose rows are ragged, and a list of records whose keys are not all the same.
|
|
4
|
+
# Out: {read, written, header} -- the ragged csv read into records, and records written back
|
|
5
|
+
# out as a csv whose header covers every key any of them has.
|
|
6
|
+
#
|
|
7
|
+
# A convert step reads one URI and writes another, so each half below is three hops: the
|
|
8
|
+
# document is put in storage, converted, and read back out. The two conversions are the
|
|
9
|
+
# middle hop of each half, and the csv rules are what they keep.
|
|
10
|
+
#
|
|
11
|
+
# Reading, two rules:
|
|
12
|
+
#
|
|
13
|
+
# The header names the columns, and every cell is a string. A row with fewer cells than
|
|
14
|
+
# the header names gets an empty string in the missing ones rather than a missing key, so
|
|
15
|
+
# every record has the same shape.
|
|
16
|
+
# A row with MORE cells than the header names is refused, naming the row: cells with no
|
|
17
|
+
# column are data nobody can address, and dropping them silently is worse than stopping.
|
|
18
|
+
#
|
|
19
|
+
# Writing, two rules:
|
|
20
|
+
#
|
|
21
|
+
# The header is the union of the keys the rows have, in first-seen order across the
|
|
22
|
+
# records. A key only the third record carries is still a column, and the records without
|
|
23
|
+
# it get an empty cell -- a header taken from the first record alone would silently drop
|
|
24
|
+
# data. Within one record the order is whatever the value carries by the time it reaches
|
|
25
|
+
# the codec, which is not always the order it was typed in; a report whose header is part
|
|
26
|
+
# of its contract builds its objects field by field in a jq step, which keeps that order.
|
|
27
|
+
# A nested value has no csv spelling, so it is refused naming the row and the key. Flatten
|
|
28
|
+
# first: jq-nested-to-flat.yaml is that step.
|
|
29
|
+
#
|
|
30
|
+
# Two headers are refused outright when reading, and both refusals are worth provoking once:
|
|
31
|
+
#
|
|
32
|
+
# station,station,celsius -> the csv header repeats a column name: 'station' at columns
|
|
33
|
+
# 1, 2; a record key names one column, so rename them before
|
|
34
|
+
# converting
|
|
35
|
+
# station,,celsius -> the csv header has no name at column 2, and a record key
|
|
36
|
+
# names its column; name it before converting
|
|
37
|
+
#
|
|
38
|
+
# Both are refusals rather than repairs because a repair has to invent a name -- station_2,
|
|
39
|
+
# column_2 -- and every reader downstream then depends on an invention this codec made up.
|
|
40
|
+
#
|
|
41
|
+
# To see either one, run with -p header='station,station,celsius'; that run fails on purpose
|
|
42
|
+
# with the message above.
|
|
43
|
+
#
|
|
44
|
+
# dg run --local examples/recipes/csv-header-rules.yaml
|
|
45
|
+
# dg run --local examples/recipes/csv-header-rules.yaml -p header=station,station,celsius
|
|
46
|
+
# dg run --local examples/recipes/csv-header-rules.yaml -p header=station,,celsius
|
|
47
|
+
|
|
48
|
+
format: dirigent/v1
|
|
49
|
+
kind: pipeline
|
|
50
|
+
code: csv-header-rules
|
|
51
|
+
name: CSV header rules
|
|
52
|
+
description: How the csv codec reads a ragged file and writes a union header, and the two headers it refuses to read at all.
|
|
53
|
+
|
|
54
|
+
tags: [recipes, storage, transform]
|
|
55
|
+
|
|
56
|
+
requires:
|
|
57
|
+
blocks:
|
|
58
|
+
- value.const
|
|
59
|
+
- transform.jq
|
|
60
|
+
- storage.write
|
|
61
|
+
- convert.std
|
|
62
|
+
- storage.read
|
|
63
|
+
|
|
64
|
+
params:
|
|
65
|
+
type: object
|
|
66
|
+
properties:
|
|
67
|
+
header:
|
|
68
|
+
type: string
|
|
69
|
+
default: station,date,celsius
|
|
70
|
+
description: The header row the reading half is given; a repeated or empty name is refused.
|
|
71
|
+
|
|
72
|
+
steps:
|
|
73
|
+
ragged:
|
|
74
|
+
block: transform.jq
|
|
75
|
+
config:
|
|
76
|
+
input: ${params.header}
|
|
77
|
+
# The rows are short on purpose: st-2 has no celsius cell at all.
|
|
78
|
+
program: |
|
|
79
|
+
. + "\nst-1,2026-01-01,4.5\nst-2,2026-01-01\nst-3,2026-01-01,7\n"
|
|
80
|
+
|
|
81
|
+
ragged_file:
|
|
82
|
+
block: storage.write
|
|
83
|
+
depends_on: [ragged]
|
|
84
|
+
config:
|
|
85
|
+
target: ${run.scratch}/ragged.csv
|
|
86
|
+
text: ${steps.ragged.output.value}
|
|
87
|
+
content_type: text/csv
|
|
88
|
+
|
|
89
|
+
read:
|
|
90
|
+
block: convert.std
|
|
91
|
+
depends_on: [ragged_file]
|
|
92
|
+
config:
|
|
93
|
+
source: ${steps.ragged_file.output.uri}
|
|
94
|
+
target: ${run.scratch}/ragged.json
|
|
95
|
+
from: csv
|
|
96
|
+
to: json
|
|
97
|
+
|
|
98
|
+
records:
|
|
99
|
+
block: storage.read
|
|
100
|
+
depends_on: [read]
|
|
101
|
+
config:
|
|
102
|
+
# The converted object is application/json, so it comes back as a value.
|
|
103
|
+
source: ${steps.read.output.target}
|
|
104
|
+
|
|
105
|
+
uneven:
|
|
106
|
+
block: value.const
|
|
107
|
+
config:
|
|
108
|
+
value:
|
|
109
|
+
- { station: st-1, celsius: 4.5 }
|
|
110
|
+
# A key the first record does not have: it is still a column.
|
|
111
|
+
- { station: st-2, celsius: -1, battery: 96 }
|
|
112
|
+
- { station: st-3, note: replaced sensor }
|
|
113
|
+
|
|
114
|
+
uneven_file:
|
|
115
|
+
block: storage.write
|
|
116
|
+
depends_on: [uneven]
|
|
117
|
+
config:
|
|
118
|
+
target: ${run.scratch}/uneven.json
|
|
119
|
+
# Canonical JSON: the keys reach the codec sorted, whatever order they were typed in.
|
|
120
|
+
value: ${steps.uneven.output.value}
|
|
121
|
+
|
|
122
|
+
write:
|
|
123
|
+
block: convert.std
|
|
124
|
+
depends_on: [uneven_file]
|
|
125
|
+
config:
|
|
126
|
+
source: ${steps.uneven_file.output.uri}
|
|
127
|
+
target: ${run.scratch}/uneven.csv
|
|
128
|
+
from: json
|
|
129
|
+
to: csv
|
|
130
|
+
|
|
131
|
+
written:
|
|
132
|
+
block: storage.read
|
|
133
|
+
depends_on: [write]
|
|
134
|
+
config:
|
|
135
|
+
source: ${steps.write.output.target}
|
|
136
|
+
|
|
137
|
+
report:
|
|
138
|
+
block: transform.jq
|
|
139
|
+
depends_on: [records, written]
|
|
140
|
+
config:
|
|
141
|
+
input:
|
|
142
|
+
read: ${steps.records.output.value}
|
|
143
|
+
written: ${steps.written.output.text}
|
|
144
|
+
program: |
|
|
145
|
+
{read: .read,
|
|
146
|
+
written: .written,
|
|
147
|
+
header: (.written | split("\n")[0] | split(","))}
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
# A csv turned into ndjson, and the one thing the conversion will not do for you: types.
|
|
2
|
+
#
|
|
3
|
+
# In: an inline csv with a header row -- station, date, celsius, active.
|
|
4
|
+
# Out: ndjson in the run's scratch space, one JSON object per line, and beside it the same
|
|
5
|
+
# records with celsius as a number and active as a boolean.
|
|
6
|
+
#
|
|
7
|
+
# All three text formats convert.std knows say the same thing -- a sequence of records --
|
|
8
|
+
# so a conversion is reading that sequence out of one spelling and writing it in another.
|
|
9
|
+
# csv to ndjson is the one a line-oriented sink wants, because a consumer can read one line
|
|
10
|
+
# at a time instead of holding the whole array.
|
|
11
|
+
#
|
|
12
|
+
# A convert step is a storage operation, the way storage.copy is: it reads one URI and
|
|
13
|
+
# writes another, and no record passes through the step itself. So the csv is put at a URI
|
|
14
|
+
# first and the ndjson is read back from one afterwards, which is the three hops below.
|
|
15
|
+
#
|
|
16
|
+
# source storage.write, the inline csv at a URI. A real pipeline writes the body an
|
|
17
|
+
# http.request answered with here instead, and nothing else in the document
|
|
18
|
+
# moves.
|
|
19
|
+
# as_ndjson convert.std csv -> ndjson. Output: source, target, bytes_written.
|
|
20
|
+
# lines storage.read of the target. Nothing recognises the .ndjson extension, so the
|
|
21
|
+
# step says what the object is; application/x-ndjson is text rather than the
|
|
22
|
+
# JSON family, so it comes back as `text`.
|
|
23
|
+
#
|
|
24
|
+
# Every cell read out of a csv is a string. "4.5" comes back as the three characters, not as
|
|
25
|
+
# a number, because a csv carries no types and a codec that guessed would be wrong on the
|
|
26
|
+
# column where 007 is a code and 1-2 is a range. So the typing step is a jq program, in the
|
|
27
|
+
# document that knows what the columns mean, and it is the second half of every csv read.
|
|
28
|
+
#
|
|
29
|
+
# ndjson is one JSON value per line, which is why the typing program splits on newlines,
|
|
30
|
+
# drops the empty last one, and fromjson each.
|
|
31
|
+
#
|
|
32
|
+
# To change it: -p empty_is_null=true reads an empty cell as null rather than as the empty
|
|
33
|
+
# string, which is the right rule for a numeric column and the wrong one for a text column.
|
|
34
|
+
#
|
|
35
|
+
# dg run --local examples/recipes/csv-to-ndjson.yaml
|
|
36
|
+
# dg run --local examples/recipes/csv-to-ndjson.yaml -p empty_is_null=true
|
|
37
|
+
|
|
38
|
+
format: dirigent/v1
|
|
39
|
+
kind: pipeline
|
|
40
|
+
code: csv-to-ndjson
|
|
41
|
+
name: CSV to ndjson, with the typing that follows
|
|
42
|
+
description: Re-spell a csv as one JSON object per line, then give the string cells the types the columns actually have.
|
|
43
|
+
|
|
44
|
+
tags: [recipes, storage, transform]
|
|
45
|
+
|
|
46
|
+
requires:
|
|
47
|
+
blocks:
|
|
48
|
+
- storage.write
|
|
49
|
+
- convert.std
|
|
50
|
+
- storage.read
|
|
51
|
+
- transform.jq
|
|
52
|
+
|
|
53
|
+
params:
|
|
54
|
+
type: object
|
|
55
|
+
properties:
|
|
56
|
+
empty_is_null:
|
|
57
|
+
type: boolean
|
|
58
|
+
default: false
|
|
59
|
+
description: Read an empty cell as null rather than as an empty string.
|
|
60
|
+
|
|
61
|
+
steps:
|
|
62
|
+
source:
|
|
63
|
+
block: storage.write
|
|
64
|
+
config:
|
|
65
|
+
target: ${run.scratch}/readings.csv
|
|
66
|
+
text: |
|
|
67
|
+
station,date,celsius,active
|
|
68
|
+
st-1,2026-01-01,4.5,true
|
|
69
|
+
st-2,2026-01-01,-1,false
|
|
70
|
+
st-3,2026-01-01,,true
|
|
71
|
+
content_type: text/csv
|
|
72
|
+
|
|
73
|
+
as_ndjson:
|
|
74
|
+
block: convert.std
|
|
75
|
+
depends_on: [source]
|
|
76
|
+
config:
|
|
77
|
+
source: ${steps.source.output.uri}
|
|
78
|
+
target: ${run.scratch}/readings.ndjson
|
|
79
|
+
from: csv
|
|
80
|
+
to: ndjson
|
|
81
|
+
|
|
82
|
+
lines:
|
|
83
|
+
block: storage.read
|
|
84
|
+
depends_on: [as_ndjson]
|
|
85
|
+
config:
|
|
86
|
+
source: ${steps.as_ndjson.output.target}
|
|
87
|
+
content_type: application/x-ndjson
|
|
88
|
+
|
|
89
|
+
typed:
|
|
90
|
+
block: transform.jq
|
|
91
|
+
depends_on: [lines]
|
|
92
|
+
config:
|
|
93
|
+
input:
|
|
94
|
+
lines: ${steps.lines.output.text}
|
|
95
|
+
empty_is_null: ${params.empty_is_null}
|
|
96
|
+
program: |
|
|
97
|
+
. as {$lines, $empty_is_null}
|
|
98
|
+
| [$lines
|
|
99
|
+
| split("\n")[]
|
|
100
|
+
| select(length > 0)
|
|
101
|
+
| fromjson
|
|
102
|
+
| {station,
|
|
103
|
+
date,
|
|
104
|
+
celsius: (if .celsius == ""
|
|
105
|
+
then (if $empty_is_null then null else 0 end)
|
|
106
|
+
else (.celsius | tonumber) end),
|
|
107
|
+
active: (.active == "true")}]
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
# A whole small ETL: a messy csv in, a checked parquet file out.
|
|
2
|
+
#
|
|
3
|
+
# In: an inline csv with padded strings, mixed case, a typed-as-text numeric column, an
|
|
4
|
+
# empty cell, and a duplicate row.
|
|
5
|
+
# Out: a parquet file in the run's scratch space, written only from rows that passed a
|
|
6
|
+
# schema gate, plus a report of what was dropped and why.
|
|
7
|
+
#
|
|
8
|
+
# The order of the nine steps is the recipe, and every one of them is a different kind of
|
|
9
|
+
# work that wants its own step:
|
|
10
|
+
#
|
|
11
|
+
# source storage.write, the inline csv at a URI. A conversion reads one URI and writes
|
|
12
|
+
# another, the way storage.copy does, so the csv it reads has to be an object
|
|
13
|
+
# first. A real pipeline writes a downloaded body here instead.
|
|
14
|
+
# parse convert.std csv -> json. Nothing passes through the step: bytes at one URI,
|
|
15
|
+
# bytes at another.
|
|
16
|
+
# rows storage.read of the parsed object, which is where the records enter the run.
|
|
17
|
+
# Every cell arrives as a string, because a csv has no types and a codec that
|
|
18
|
+
# guessed would be wrong on the column where 007 is a code.
|
|
19
|
+
# clean the string work: trim, case-fold, and turn the text columns that are really
|
|
20
|
+
# numbers into numbers. This is where a csv's missing type information is
|
|
21
|
+
# supplied, by the document that knows what the columns mean.
|
|
22
|
+
# sort the rows that cannot be loaded are separated from the ones that can, with a
|
|
23
|
+
# reason kept per row. Dropping them silently is what makes a load report "412
|
|
24
|
+
# rows" when 460 arrived.
|
|
25
|
+
# gate validate.schema over the accepted rows. The clean step and the schema are two
|
|
26
|
+
# statements of the same expectation, and the gate is what makes them agree:
|
|
27
|
+
# anything the cleaner let through that the schema refuses fails the run here,
|
|
28
|
+
# at the boundary, rather than inside a parquet file somebody reads next month.
|
|
29
|
+
# checked storage.write of what the gate passed, because the writer below takes a URI
|
|
30
|
+
# and not a value. Nothing but the gate's own output is written here.
|
|
31
|
+
# write convert.arrow json -> parquet. Types survive here, which is the reason for
|
|
32
|
+
# doing all of the above before writing rather than after.
|
|
33
|
+
# report the run's own account of itself: counts, the rejected keys, and the file.
|
|
34
|
+
#
|
|
35
|
+
# Everything downstream of the gate reads ${steps.gate.output.value}, which is the input
|
|
36
|
+
# unchanged. That reference is the document's proof that no unchecked row reached the file.
|
|
37
|
+
#
|
|
38
|
+
# To change it: -p min_celsius=-50 loosens the range the schema allows; -p strict=true makes
|
|
39
|
+
# the sorter refuse a duplicate key instead of keeping the first, which fails the run rather
|
|
40
|
+
# than quietly preferring one row over another.
|
|
41
|
+
#
|
|
42
|
+
# The schema is carried in the document, so no server stores this one: it runs with
|
|
43
|
+
# `dg run --local`, and on an instance the same shape is created once with `dg schema create`.
|
|
44
|
+
#
|
|
45
|
+
# dg run --local examples/recipes/etl-csv-clean-validate-parquet.yaml
|
|
46
|
+
# dg run --local examples/recipes/etl-csv-clean-validate-parquet.yaml -p strict=true # fails, on purpose
|
|
47
|
+
|
|
48
|
+
format: dirigent/v1
|
|
49
|
+
kind: pipeline
|
|
50
|
+
code: etl-csv-clean-validate-parquet
|
|
51
|
+
name: CSV to checked parquet
|
|
52
|
+
description: Parse a messy csv, clean and type it, sort out the unloadable rows, hold the rest to a schema, and write parquet.
|
|
53
|
+
|
|
54
|
+
tags: [recipes, storage, transform, validate, parquet]
|
|
55
|
+
|
|
56
|
+
requires:
|
|
57
|
+
blocks:
|
|
58
|
+
- storage.write
|
|
59
|
+
- storage.read
|
|
60
|
+
- convert.std
|
|
61
|
+
- convert.arrow
|
|
62
|
+
- transform.jq
|
|
63
|
+
- validate.schema
|
|
64
|
+
|
|
65
|
+
schemas:
|
|
66
|
+
recipe-reading-rows:
|
|
67
|
+
type: array
|
|
68
|
+
minItems: 1
|
|
69
|
+
items:
|
|
70
|
+
type: object
|
|
71
|
+
required: [station, day, celsius, active]
|
|
72
|
+
additionalProperties: false
|
|
73
|
+
properties:
|
|
74
|
+
station: { type: string, pattern: "^st-[0-9]+$" }
|
|
75
|
+
day: { type: string, format: date }
|
|
76
|
+
celsius: { type: number, minimum: -90, maximum: 60 }
|
|
77
|
+
active: { type: boolean }
|
|
78
|
+
|
|
79
|
+
params:
|
|
80
|
+
type: object
|
|
81
|
+
properties:
|
|
82
|
+
day:
|
|
83
|
+
type: string
|
|
84
|
+
format: date
|
|
85
|
+
default: "2026-01-01"
|
|
86
|
+
description: The day the file is named for.
|
|
87
|
+
strict:
|
|
88
|
+
type: boolean
|
|
89
|
+
default: false
|
|
90
|
+
description: Fail on a duplicate station instead of keeping the first row for it.
|
|
91
|
+
|
|
92
|
+
steps:
|
|
93
|
+
source:
|
|
94
|
+
block: storage.write
|
|
95
|
+
config:
|
|
96
|
+
target: ${run.scratch}/readings-${params.day}.csv
|
|
97
|
+
# Inline so the recipe needs no network. In a real pipeline the body an http.request
|
|
98
|
+
# answered with is written here, and nothing below moves.
|
|
99
|
+
text: |
|
|
100
|
+
station,day,celsius,active
|
|
101
|
+
ST-1 ,2026-01-01,4.5,TRUE
|
|
102
|
+
st-2,2026-01-01,-1,false
|
|
103
|
+
st-3,2026-01-01,,true
|
|
104
|
+
st-1,2026-01-01,4.5,true
|
|
105
|
+
st-4,2026-01-01,not-a-number,true
|
|
106
|
+
content_type: text/csv
|
|
107
|
+
|
|
108
|
+
parse:
|
|
109
|
+
block: convert.std
|
|
110
|
+
depends_on: [source]
|
|
111
|
+
config:
|
|
112
|
+
source: ${steps.source.output.uri}
|
|
113
|
+
target: ${run.scratch}/readings-${params.day}.json
|
|
114
|
+
from: csv
|
|
115
|
+
to: json
|
|
116
|
+
|
|
117
|
+
rows:
|
|
118
|
+
block: storage.read
|
|
119
|
+
depends_on: [parse]
|
|
120
|
+
config:
|
|
121
|
+
source: ${steps.parse.output.target}
|
|
122
|
+
max_size: 8mb
|
|
123
|
+
|
|
124
|
+
clean:
|
|
125
|
+
block: transform.jq
|
|
126
|
+
depends_on: [rows]
|
|
127
|
+
config:
|
|
128
|
+
input:
|
|
129
|
+
rows: ${steps.rows.output.value}
|
|
130
|
+
day: ${params.day}
|
|
131
|
+
program: |
|
|
132
|
+
def trim: sub("^\\s+"; "") | sub("\\s+$"; "");
|
|
133
|
+
. as {$rows, $day}
|
|
134
|
+
| [$rows[]
|
|
135
|
+
| {station: (.station | trim | ascii_downcase),
|
|
136
|
+
day: (if .day == "" then $day else .day end),
|
|
137
|
+
# An empty cell is not a measurement of zero, so it stays null and is sorted
|
|
138
|
+
# out below rather than invented here.
|
|
139
|
+
celsius: (if .celsius == "" then null
|
|
140
|
+
else (try (.celsius | tonumber) catch null) end),
|
|
141
|
+
# A csv boolean is a word, and which word depends on who exported it.
|
|
142
|
+
active: ((.active | ascii_downcase) == "true")}]
|
|
143
|
+
|
|
144
|
+
sort:
|
|
145
|
+
block: transform.jq
|
|
146
|
+
depends_on: [clean]
|
|
147
|
+
config:
|
|
148
|
+
input:
|
|
149
|
+
rows: ${steps.clean.output.value}
|
|
150
|
+
strict: ${params.strict}
|
|
151
|
+
program: |
|
|
152
|
+
. as {$rows, $strict}
|
|
153
|
+
| [$rows | group_by(.station)[] | {station: .[0].station, count: length}]
|
|
154
|
+
as $by_station
|
|
155
|
+
| ([$by_station[] | select(.count > 1) | .station]) as $duplicated
|
|
156
|
+
| if $strict and ($duplicated | length) > 0
|
|
157
|
+
then error("duplicate stations: \($duplicated | join(", "))")
|
|
158
|
+
else . end
|
|
159
|
+
| {accepted: [$rows
|
|
160
|
+
| group_by(.station)[]
|
|
161
|
+
# First row wins for a duplicated key, which is a decision and not an accident.
|
|
162
|
+
| .[0]
|
|
163
|
+
| select(.celsius != null)],
|
|
164
|
+
rejected: [$rows[]
|
|
165
|
+
| select(.celsius == null)
|
|
166
|
+
| {station, reason: "celsius is absent or not a number"}],
|
|
167
|
+
duplicated: $duplicated}
|
|
168
|
+
|
|
169
|
+
gate:
|
|
170
|
+
block: validate.schema
|
|
171
|
+
depends_on: [sort]
|
|
172
|
+
config:
|
|
173
|
+
input: ${steps.sort.output.value.accepted}
|
|
174
|
+
schema: recipe-reading-rows
|
|
175
|
+
|
|
176
|
+
checked:
|
|
177
|
+
block: storage.write
|
|
178
|
+
depends_on: [gate]
|
|
179
|
+
config:
|
|
180
|
+
target: ${run.scratch}/checked-${params.day}.json
|
|
181
|
+
# From the gate, never from the sorter.
|
|
182
|
+
value: ${steps.gate.output.value}
|
|
183
|
+
|
|
184
|
+
write:
|
|
185
|
+
block: convert.arrow
|
|
186
|
+
depends_on: [checked]
|
|
187
|
+
config:
|
|
188
|
+
source: ${steps.checked.output.uri}
|
|
189
|
+
target: ${run.scratch}/readings-${params.day}.parquet
|
|
190
|
+
from: json
|
|
191
|
+
to: parquet
|
|
192
|
+
|
|
193
|
+
report:
|
|
194
|
+
block: transform.jq
|
|
195
|
+
depends_on: [sort, gate, write]
|
|
196
|
+
config:
|
|
197
|
+
input:
|
|
198
|
+
sorted: ${steps.sort.output.value}
|
|
199
|
+
checked: ${steps.gate.output.value}
|
|
200
|
+
uri: ${steps.write.output.target}
|
|
201
|
+
bytes: ${steps.write.output.bytes_written}
|
|
202
|
+
program: |
|
|
203
|
+
{file: .uri, file_bytes: .bytes,
|
|
204
|
+
written: (.checked | length),
|
|
205
|
+
rejected: (.sorted.rejected | length),
|
|
206
|
+
rejected_rows: .sorted.rejected,
|
|
207
|
+
duplicated_stations: .sorted.duplicated}
|