dirigent-examples 0.15.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_examples/__init__.py +22 -0
- dirigent_examples/py.typed +0 -0
- dirigent_examples/shelves/README.md +299 -0
- dirigent_examples/shelves/composition/README.md +18 -0
- dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
- dirigent_examples/shelves/composition/composition-child.yaml +64 -0
- dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
- dirigent_examples/shelves/connections.yaml +52 -0
- dirigent_examples/shelves/demo/README.md +19 -0
- dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
- dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
- dirigent_examples/shelves/demo/requires.yaml +65 -0
- dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
- dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
- dirigent_examples/shelves/docker/README.md +29 -0
- dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
- dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
- dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
- dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
- dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
- dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
- dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
- dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
- dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
- dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
- dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
- dirigent_examples/shelves/execute/README.md +16 -0
- dirigent_examples/shelves/execute/long-log.yaml +89 -0
- dirigent_examples/shelves/failure/README.md +20 -0
- dirigent_examples/shelves/failure/error-handler.yaml +89 -0
- dirigent_examples/shelves/failure/optional-step.yaml +82 -0
- dirigent_examples/shelves/failure/retries.yaml +75 -0
- dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
- dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
- dirigent_examples/shelves/git/README.md +32 -0
- dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
- dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
- dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
- dirigent_examples/shelves/graph/README.md +22 -0
- dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
- dirigent_examples/shelves/graph/fan-in.yaml +80 -0
- dirigent_examples/shelves/graph/fan-out.yaml +66 -0
- dirigent_examples/shelves/graph/linear.yaml +66 -0
- dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
- dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
- dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
- dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
- dirigent_examples/shelves/hello-world.yaml +30 -0
- dirigent_examples/shelves/open-data/README.md +67 -0
- dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
- dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
- dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
- dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
- dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
- dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
- dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
- dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
- dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
- dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
- dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
- dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
- dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
- dirigent_examples/shelves/patterns/README.md +144 -0
- dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
- dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
- dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
- dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
- dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
- dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
- dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
- dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
- dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
- dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
- dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
- dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
- dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
- dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
- dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
- dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
- dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
- dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
- dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
- dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
- dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
- dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
- dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
- dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
- dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
- dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
- dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
- dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
- dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
- dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
- dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
- dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
- dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
- dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
- dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
- dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
- dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
- dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
- dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
- dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
- dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
- dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
- dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
- dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
- dirigent_examples/shelves/python/README.md +31 -0
- dirigent_examples/shelves/python/apply_and_run.py +52 -0
- dirigent_examples/shelves/python/ci_gate.py +76 -0
- dirigent_examples/shelves/python/connections.py +61 -0
- dirigent_examples/shelves/python/error_handling.py +84 -0
- dirigent_examples/shelves/python/follow_logs.py +39 -0
- dirigent_examples/shelves/python/list_and_filter.py +52 -0
- dirigent_examples/shelves/queues/README.md +59 -0
- dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
- dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
- dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
- dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
- dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
- dirigent_examples/shelves/recipes/README.md +130 -0
- dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
- dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
- dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
- dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
- dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
- dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
- dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
- dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
- dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
- dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
- dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
- dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
- dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
- dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
- dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
- dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
- dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
- dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
- dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
- dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
- dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
- dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
- dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
- dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
- dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
- dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
- dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
- dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
- dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
- dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
- dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
- dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
- dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
- dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
- dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
- dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
- dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
- dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
- dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
- dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
- dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
- dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
- dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
- dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
- dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
- dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
- dirigent_examples/shelves/s3/README.md +34 -0
- dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
- dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
- dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
- dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
- dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
- dirigent_examples/shelves/schemas/README.md +36 -0
- dirigent_examples/shelves/schemas/echo-reading.json +18 -0
- dirigent_examples/shelves/schemas/ou-record.json +13 -0
- dirigent_examples/shelves/schemas/station-reading.json +13 -0
- dirigent_examples/shelves/sensors/README.md +16 -0
- dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
- dirigent_examples/shelves/sensors/time-window.yaml +61 -0
- dirigent_examples/shelves/sql/README.md +52 -0
- dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
- dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
- dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
- dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
- dirigent_examples/shelves/sql/warehouse.sql +42 -0
- dirigent_examples/shelves/transform/README.md +36 -0
- dirigent_examples/shelves/transform/csv-report.yaml +55 -0
- dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
- dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
- dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
- dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
- dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
- dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
- dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
- dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
- dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
- dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
- dirigent_examples/shelves/triggers/README.md +45 -0
- dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
- dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
- dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
- dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
- dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
- dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
- dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
- dirigent_examples/shelves/validate/README.md +31 -0
- dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
- dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
- dirigent_examples-0.15.0.dist-info/METADATA +21 -0
- dirigent_examples-0.15.0.dist-info/RECORD +216 -0
- dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
- dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
- dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# Three deadlines sit around one HTTP call. Know which one you are setting.
|
|
2
|
+
#
|
|
3
|
+
# In: a delay in seconds, as a parameter, defaulting to 2.
|
|
4
|
+
# Out: a slow call that completes inside its own timeout, and its measured duration.
|
|
5
|
+
#
|
|
6
|
+
# The three, from the inside out:
|
|
7
|
+
#
|
|
8
|
+
# timeout on the step's config how long this one call may take before the
|
|
9
|
+
# client gives up. It overrides the connection's
|
|
10
|
+
# timeout for this call and nothing else.
|
|
11
|
+
# timeout on the connection the default for every call made through it,
|
|
12
|
+
# which is where a slow API's whole profile
|
|
13
|
+
# belongs rather than in each step.
|
|
14
|
+
# timeout on the step the engine's deadline on the attempt, which
|
|
15
|
+
# covers the block's entire execution, retries of
|
|
16
|
+
# an inner operation included. It is step-level
|
|
17
|
+
# engine semantics, uniform across every block,
|
|
18
|
+
# and it is what kills a step that hangs somewhere
|
|
19
|
+
# the HTTP client is not.
|
|
20
|
+
#
|
|
21
|
+
# The inner one is the one to reach for when a single endpoint is slow: a report that takes
|
|
22
|
+
# forty seconds to build does not justify raising the deadline on every call to that host,
|
|
23
|
+
# and a health check that should answer in one second is a step-level override in the other
|
|
24
|
+
# direction.
|
|
25
|
+
#
|
|
26
|
+
# https://postman-echo.com/delay/<seconds> answers after the seconds it is given, so the
|
|
27
|
+
# relationship is visible: with the default the call takes about 2s against a 10s budget and
|
|
28
|
+
# the run is green. Setting -p delay=6 against -p timeout=3 fails the step instead, as a
|
|
29
|
+
# timeout rather than as an error the server returned, which is a different failure class
|
|
30
|
+
# and retried differently.
|
|
31
|
+
#
|
|
32
|
+
# The engine's own step timeout is set here too, comfortably above the call's, so that a
|
|
33
|
+
# hang outside the HTTP client -- a name that never resolves, a socket that never closes --
|
|
34
|
+
# still ends the attempt.
|
|
35
|
+
#
|
|
36
|
+
# To change it: -p delay=6 -p timeout=3 to see the failure, or -p delay=1 for a fast run.
|
|
37
|
+
#
|
|
38
|
+
# dg run --local examples/recipes/http-timeout-override.yaml
|
|
39
|
+
# dg run --local examples/recipes/http-timeout-override.yaml -p delay=6 -p timeout=3 # fails, on purpose
|
|
40
|
+
|
|
41
|
+
format: dirigent/v1
|
|
42
|
+
kind: pipeline
|
|
43
|
+
code: http-timeout-override
|
|
44
|
+
name: A per-call timeout override
|
|
45
|
+
description: Override a connection's timeout for one slow call, under an engine step deadline that covers the whole attempt.
|
|
46
|
+
|
|
47
|
+
tags: [recipes, http, transform]
|
|
48
|
+
|
|
49
|
+
requires:
|
|
50
|
+
blocks:
|
|
51
|
+
- http.request
|
|
52
|
+
- transform.jq
|
|
53
|
+
|
|
54
|
+
connections:
|
|
55
|
+
echo-slow:
|
|
56
|
+
kind: http
|
|
57
|
+
config:
|
|
58
|
+
base_url: https://postman-echo.com
|
|
59
|
+
# The default for every call through this connection. Deliberately short, so the
|
|
60
|
+
# override below is doing real work.
|
|
61
|
+
timeout: 3s
|
|
62
|
+
health_path: /get
|
|
63
|
+
|
|
64
|
+
params:
|
|
65
|
+
type: object
|
|
66
|
+
properties:
|
|
67
|
+
delay:
|
|
68
|
+
type: integer
|
|
69
|
+
minimum: 0
|
|
70
|
+
maximum: 10
|
|
71
|
+
default: 2
|
|
72
|
+
description: How many seconds the endpoint waits before answering.
|
|
73
|
+
timeout:
|
|
74
|
+
type: number
|
|
75
|
+
default: 10
|
|
76
|
+
description: This call's own deadline, overriding the connection's 3 seconds.
|
|
77
|
+
|
|
78
|
+
steps:
|
|
79
|
+
slow:
|
|
80
|
+
block: http.request
|
|
81
|
+
# The engine's deadline on the whole attempt, well above the call's own, so a hang
|
|
82
|
+
# outside the HTTP client still ends the step.
|
|
83
|
+
timeout: 60s
|
|
84
|
+
config:
|
|
85
|
+
connection: echo-slow
|
|
86
|
+
path: /delay/${params.delay}
|
|
87
|
+
method: GET
|
|
88
|
+
# This call alone; every other call through echo-slow keeps the 3 second default.
|
|
89
|
+
timeout: ${params.timeout}
|
|
90
|
+
|
|
91
|
+
timing:
|
|
92
|
+
block: transform.jq
|
|
93
|
+
depends_on: [slow]
|
|
94
|
+
config:
|
|
95
|
+
input:
|
|
96
|
+
status: ${steps.slow.output.status}
|
|
97
|
+
duration_ms: ${steps.slow.output.duration_ms}
|
|
98
|
+
budget: ${params.timeout}
|
|
99
|
+
program: |
|
|
100
|
+
{status, duration_ms,
|
|
101
|
+
budget_ms: (.budget * 1000),
|
|
102
|
+
# A call that lands close to its budget is a call about to start failing
|
|
103
|
+
# intermittently, which is worth reporting before it does.
|
|
104
|
+
headroom_ms: ((.budget * 1000) - .duration_ms)}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# Reduce a list with repeated keys to one row per key, and say which row won.
|
|
2
|
+
#
|
|
3
|
+
# In: readings where the same station reports several times, out of order, some of them
|
|
4
|
+
# corrections of an earlier one.
|
|
5
|
+
# Out: {first_seen, last_seen, latest, duplicates} -- three different answers to "dedupe",
|
|
6
|
+
# side by side, plus the keys that had more than one row.
|
|
7
|
+
#
|
|
8
|
+
# Dedupe is never one operation, because "the duplicate" is not a property of the data; it
|
|
9
|
+
# is a decision about which row to keep:
|
|
10
|
+
#
|
|
11
|
+
# unique_by(.station) keeps the smallest row by the whole key order, which for a
|
|
12
|
+
# sorted-by-nothing input is arbitrary and only safe when the
|
|
13
|
+
# duplicates are genuinely identical.
|
|
14
|
+
# group_by then .[0] / .[-1] keeps the first or last row of each bucket in input order,
|
|
15
|
+
# which is the answer when arrival order carries meaning.
|
|
16
|
+
# group_by then max_by(.at) keeps the newest row by a timestamp the data carries, which
|
|
17
|
+
# is the only one of the three that survives a reorder.
|
|
18
|
+
#
|
|
19
|
+
# The last is almost always the right one for corrections, and this recipe reports the other
|
|
20
|
+
# two beside it so the difference is visible in the output rather than argued about.
|
|
21
|
+
#
|
|
22
|
+
# duplicates is computed before anything is thrown away: a dedupe that reports nothing is a
|
|
23
|
+
# dedupe nobody can audit.
|
|
24
|
+
#
|
|
25
|
+
# To change it: -p key=region dedupes on a different column; the timestamp column stays .at.
|
|
26
|
+
#
|
|
27
|
+
# dg run --local examples/recipes/jq-dedupe-by-key.yaml
|
|
28
|
+
# dg run --local examples/recipes/jq-dedupe-by-key.yaml -p key=region
|
|
29
|
+
|
|
30
|
+
format: dirigent/v1
|
|
31
|
+
kind: pipeline
|
|
32
|
+
code: jq-dedupe-by-key
|
|
33
|
+
name: Dedupe by key, three ways
|
|
34
|
+
description: Collapse repeated rows to one per key by first seen, last seen, and newest timestamp, and list the keys that repeated.
|
|
35
|
+
|
|
36
|
+
tags: [recipes, transform, jq]
|
|
37
|
+
|
|
38
|
+
requires:
|
|
39
|
+
blocks:
|
|
40
|
+
- value.const
|
|
41
|
+
- transform.jq
|
|
42
|
+
|
|
43
|
+
params:
|
|
44
|
+
type: object
|
|
45
|
+
properties:
|
|
46
|
+
key:
|
|
47
|
+
type: string
|
|
48
|
+
default: station
|
|
49
|
+
description: The column a row is deduplicated on.
|
|
50
|
+
|
|
51
|
+
steps:
|
|
52
|
+
rows:
|
|
53
|
+
block: value.const
|
|
54
|
+
config:
|
|
55
|
+
value:
|
|
56
|
+
- { station: st-1, region: east, at: "2026-01-01T06:00:00Z", celsius: 4 }
|
|
57
|
+
- { station: st-2, region: west, at: "2026-01-01T06:00:00Z", celsius: -1 }
|
|
58
|
+
# A correction: same station, later timestamp, arriving before the row it corrects.
|
|
59
|
+
- { station: st-1, region: east, at: "2026-01-01T09:00:00Z", celsius: 5 }
|
|
60
|
+
- { station: st-1, region: east, at: "2026-01-01T07:00:00Z", celsius: 40 }
|
|
61
|
+
- { station: st-3, region: east, at: "2026-01-01T06:00:00Z", celsius: 7 }
|
|
62
|
+
|
|
63
|
+
deduped:
|
|
64
|
+
block: transform.jq
|
|
65
|
+
depends_on: [rows]
|
|
66
|
+
config:
|
|
67
|
+
input:
|
|
68
|
+
key: ${params.key}
|
|
69
|
+
rows: ${steps.rows.output.value}
|
|
70
|
+
program: |
|
|
71
|
+
. as {$key, $rows}
|
|
72
|
+
| ($rows | group_by(.[$key])) as $buckets
|
|
73
|
+
| {first_seen: [$buckets[] | .[0]],
|
|
74
|
+
last_seen: [$buckets[] | .[-1]],
|
|
75
|
+
# max_by reads the timestamp the row carries, so a late arrival cannot win by
|
|
76
|
+
# arriving late and an early correction cannot lose by arriving early.
|
|
77
|
+
latest: [$buckets[] | max_by(.at)],
|
|
78
|
+
duplicates: [$buckets[] | select(length > 1) | {key: .[0][$key], rows: length}]}
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# Absent, null, false, and empty string are four different things. Handle each on purpose.
|
|
2
|
+
#
|
|
3
|
+
# In: rows where a field is missing entirely, present and null, present and false, and
|
|
4
|
+
# present and empty -- one row for each case.
|
|
5
|
+
# Out: {naive, careful, report} -- the same defaulting written the wrong way and the right
|
|
6
|
+
# way, with a per-row report of which case each field actually was.
|
|
7
|
+
#
|
|
8
|
+
# // is the alternative operator, and it answers with its right side when the left is null
|
|
9
|
+
# OR false. That second half is the trap: .enabled // true can never answer false, so a row
|
|
10
|
+
# that switched a feature off comes back on. The naive object in the output is that bug,
|
|
11
|
+
# left in on purpose so the careful one below has something to be different from.
|
|
12
|
+
#
|
|
13
|
+
# The three questions that are actually being asked are different questions, and jq spells
|
|
14
|
+
# each of them:
|
|
15
|
+
#
|
|
16
|
+
# has("enabled") is the key present at all -- absent and null are distinguished.
|
|
17
|
+
# .enabled == null is the value the null value, whether or not the key is there.
|
|
18
|
+
# (.count // 0) is it null-or-false, which for a number is a fine default.
|
|
19
|
+
#
|
|
20
|
+
# For a boolean, the safe defaulting is `if has("x") and .x != null then .x else DEFAULT end`.
|
|
21
|
+
# For a string, "" is a value and // will not replace it, so an empty-means-absent rule is
|
|
22
|
+
# written out rather than assumed.
|
|
23
|
+
#
|
|
24
|
+
# tonumber on a string that is not a number is an error rather than a null, so it goes
|
|
25
|
+
# through try/catch. That is the one place a reshape may decide, because the alternative --
|
|
26
|
+
# a failed step -- reports a bad cell as an outage.
|
|
27
|
+
#
|
|
28
|
+
# To change it: -p strict_numbers=true makes an unparsable count null instead of zero, which
|
|
29
|
+
# is what a load that must not invent a measurement wants.
|
|
30
|
+
#
|
|
31
|
+
# dg run --local examples/recipes/jq-defaults-and-nulls.yaml
|
|
32
|
+
# dg run --local examples/recipes/jq-defaults-and-nulls.yaml -p strict_numbers=true
|
|
33
|
+
|
|
34
|
+
format: dirigent/v1
|
|
35
|
+
kind: pipeline
|
|
36
|
+
code: jq-defaults-and-nulls
|
|
37
|
+
name: Defaults, nulls, and the // trap
|
|
38
|
+
description: Distinguish absent, null, false, and empty when defaulting fields, and show what the alternative operator gets wrong.
|
|
39
|
+
|
|
40
|
+
tags: [recipes, transform, jq]
|
|
41
|
+
|
|
42
|
+
requires:
|
|
43
|
+
blocks:
|
|
44
|
+
- value.const
|
|
45
|
+
- transform.jq
|
|
46
|
+
|
|
47
|
+
params:
|
|
48
|
+
type: object
|
|
49
|
+
properties:
|
|
50
|
+
strict_numbers:
|
|
51
|
+
type: boolean
|
|
52
|
+
default: false
|
|
53
|
+
description: Leave an unparsable count null rather than defaulting it to zero.
|
|
54
|
+
|
|
55
|
+
steps:
|
|
56
|
+
rows:
|
|
57
|
+
block: value.const
|
|
58
|
+
config:
|
|
59
|
+
value:
|
|
60
|
+
- { id: a, enabled: true, count: "3", label: Harbour }
|
|
61
|
+
# enabled is absent: nothing was said about it.
|
|
62
|
+
- { id: b, count: "0", label: Ridge }
|
|
63
|
+
# enabled is null: something was said, and it said nothing.
|
|
64
|
+
- { id: c, enabled: null, count: null, label: "" }
|
|
65
|
+
# enabled is false: something was said, and it said no. This is the row // ruins.
|
|
66
|
+
- { id: d, enabled: false, count: "not a number", label: Delta }
|
|
67
|
+
|
|
68
|
+
defaults:
|
|
69
|
+
block: transform.jq
|
|
70
|
+
depends_on: [rows]
|
|
71
|
+
config:
|
|
72
|
+
input:
|
|
73
|
+
rows: ${steps.rows.output.value}
|
|
74
|
+
strict: ${params.strict_numbers}
|
|
75
|
+
program: |
|
|
76
|
+
. as {$rows, $strict}
|
|
77
|
+
| {naive: [$rows[] | {id, enabled: (.enabled // true)}],
|
|
78
|
+
careful: [$rows[]
|
|
79
|
+
| {id,
|
|
80
|
+
enabled: (if has("enabled") and .enabled != null then .enabled else true end),
|
|
81
|
+
count: (if .count == null then (if $strict then null else 0 end)
|
|
82
|
+
else (try (.count | tonumber)
|
|
83
|
+
catch (if $strict then null else 0 end)) end),
|
|
84
|
+
# An empty string is a value, so // never sees it; the rule is written out.
|
|
85
|
+
label: (if (.label // "") == "" then "unnamed" else .label end)}],
|
|
86
|
+
report: [$rows[]
|
|
87
|
+
| {id,
|
|
88
|
+
enabled_case: (if has("enabled") | not then "absent"
|
|
89
|
+
elif .enabled == null then "null"
|
|
90
|
+
elif .enabled == false then "false"
|
|
91
|
+
else "true" end)}]}
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
# Total a numeric column per group, and keep the grand total beside the groups.
|
|
2
|
+
#
|
|
3
|
+
# In: a flat list of sales rows -- {region, product, units, revenue}.
|
|
4
|
+
# Out: {as_of, groups: [{region, rows, units, revenue, mean_revenue}], total: {...}}.
|
|
5
|
+
#
|
|
6
|
+
# group_by sorts by the key and buckets, so every bucket is a list and the group's key is
|
|
7
|
+
# read back off any element of it -- .[0].region. Everything after that is ordinary list
|
|
8
|
+
# arithmetic: add over a mapped column, length for the count, and a division that is safe
|
|
9
|
+
# because group_by never produces an empty bucket.
|
|
10
|
+
#
|
|
11
|
+
# The grand total is computed from the rows rather than from the groups, because summing
|
|
12
|
+
# already-rounded group totals is how a report stops adding up.
|
|
13
|
+
#
|
|
14
|
+
# To change it: -p group_by=product buckets the same rows the other way. The program reads
|
|
15
|
+
# the key with .[$key], jq's dynamic field access, which is what lets the column be a
|
|
16
|
+
# parameter instead of a second program.
|
|
17
|
+
#
|
|
18
|
+
# dg run --local examples/recipes/jq-group-by-and-sum.yaml
|
|
19
|
+
# dg run --local examples/recipes/jq-group-by-and-sum.yaml -p group_by=product
|
|
20
|
+
|
|
21
|
+
format: dirigent/v1
|
|
22
|
+
kind: pipeline
|
|
23
|
+
code: jq-group-by-and-sum
|
|
24
|
+
name: Group and sum with jq
|
|
25
|
+
description: Bucket flat rows by one column, total two numeric columns per bucket, and keep a grand total beside them.
|
|
26
|
+
|
|
27
|
+
tags: [recipes, transform, jq]
|
|
28
|
+
|
|
29
|
+
requires:
|
|
30
|
+
blocks:
|
|
31
|
+
- value.const
|
|
32
|
+
- transform.jq
|
|
33
|
+
|
|
34
|
+
params:
|
|
35
|
+
type: object
|
|
36
|
+
properties:
|
|
37
|
+
group_by:
|
|
38
|
+
type: string
|
|
39
|
+
default: region
|
|
40
|
+
description: The column the rows are bucketed by.
|
|
41
|
+
|
|
42
|
+
steps:
|
|
43
|
+
rows:
|
|
44
|
+
block: value.const
|
|
45
|
+
config:
|
|
46
|
+
value:
|
|
47
|
+
- { region: east, product: pump, units: 3, revenue: 300 }
|
|
48
|
+
- { region: east, product: valve, units: 10, revenue: 250 }
|
|
49
|
+
- { region: west, product: pump, units: 1, revenue: 100 }
|
|
50
|
+
- { region: west, product: valve, units: 4, revenue: 100 }
|
|
51
|
+
- { region: north, product: pump, units: 2, revenue: 200 }
|
|
52
|
+
|
|
53
|
+
totals:
|
|
54
|
+
block: transform.jq
|
|
55
|
+
depends_on: [rows]
|
|
56
|
+
config:
|
|
57
|
+
# The key travels as data beside the rows rather than spliced into the program text:
|
|
58
|
+
# a program carrying a ${...} cannot be compiled when the document is applied.
|
|
59
|
+
input:
|
|
60
|
+
key: ${params.group_by}
|
|
61
|
+
rows: ${steps.rows.output.value}
|
|
62
|
+
program: |
|
|
63
|
+
. as {$key, $rows}
|
|
64
|
+
| {as_of: "2026-01-01",
|
|
65
|
+
groups: ($rows
|
|
66
|
+
| group_by(.[$key])
|
|
67
|
+
| map({group: .[0][$key],
|
|
68
|
+
rows: length,
|
|
69
|
+
units: (map(.units) | add),
|
|
70
|
+
revenue: (map(.revenue) | add),
|
|
71
|
+
mean_revenue: ((map(.revenue) | add) / length)})),
|
|
72
|
+
total: {rows: ($rows | length),
|
|
73
|
+
units: ($rows | map(.units) | add),
|
|
74
|
+
revenue: ($rows | map(.revenue) | add)}}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
# Join two lists on a key, keeping the rows that match nothing.
|
|
2
|
+
#
|
|
3
|
+
# In: a list of stations (the dimension) and a list of readings (the facts).
|
|
4
|
+
# Out: {joined: [...], unmatched: [...]} -- every reading enriched, and the ones whose
|
|
5
|
+
# station is in no dimension row listed separately rather than silently dropped.
|
|
6
|
+
#
|
|
7
|
+
# A jq program has exactly one input, so a join composes its two sides into one object in
|
|
8
|
+
# the step's config and destructures them apart again in the program. INDEX builds the
|
|
9
|
+
# lookup table in one pass -- an object keyed by station id -- which turns the join from a
|
|
10
|
+
# scan per reading into a lookup per reading.
|
|
11
|
+
#
|
|
12
|
+
# Both halves of the result come out of that one index: $by_id[.station] is null for a
|
|
13
|
+
# reading nothing matched, so // supplies the default on the joined side and select collects
|
|
14
|
+
# exactly those rows on the unmatched side. An inner join alone would report four readings
|
|
15
|
+
# where five arrived, and never say which one went missing.
|
|
16
|
+
#
|
|
17
|
+
# To change it: index the other side to change the join's direction -- indexing the readings
|
|
18
|
+
# and iterating the stations answers "which stations reported at all".
|
|
19
|
+
#
|
|
20
|
+
# dg run --local examples/recipes/jq-join-two-lists.yaml
|
|
21
|
+
# dg run --local examples/recipes/jq-join-two-lists.yaml -p default_name=none
|
|
22
|
+
|
|
23
|
+
format: dirigent/v1
|
|
24
|
+
kind: pipeline
|
|
25
|
+
code: jq-join-two-lists
|
|
26
|
+
name: Join two lists with INDEX
|
|
27
|
+
description: Join readings to their station dimension with INDEX, keeping the unmatched rows in a list of their own.
|
|
28
|
+
|
|
29
|
+
tags: [recipes, transform, jq]
|
|
30
|
+
|
|
31
|
+
requires:
|
|
32
|
+
blocks:
|
|
33
|
+
- value.const
|
|
34
|
+
- transform.jq
|
|
35
|
+
|
|
36
|
+
params:
|
|
37
|
+
type: object
|
|
38
|
+
properties:
|
|
39
|
+
default_name:
|
|
40
|
+
type: string
|
|
41
|
+
default: unknown station
|
|
42
|
+
description: The name a reading gets when its station is in no dimension row.
|
|
43
|
+
|
|
44
|
+
steps:
|
|
45
|
+
stations:
|
|
46
|
+
block: value.const
|
|
47
|
+
config:
|
|
48
|
+
value:
|
|
49
|
+
- { id: st-1, name: Harbour, region: east }
|
|
50
|
+
- { id: st-2, name: Ridge, region: west }
|
|
51
|
+
- { id: st-3, name: Delta, region: east }
|
|
52
|
+
|
|
53
|
+
readings:
|
|
54
|
+
block: value.const
|
|
55
|
+
config:
|
|
56
|
+
value:
|
|
57
|
+
- { station: st-1, celsius: 4 }
|
|
58
|
+
- { station: st-2, celsius: -1 }
|
|
59
|
+
- { station: st-3, celsius: 7 }
|
|
60
|
+
# No st-9 in the dimension: this is the row the recipe exists for.
|
|
61
|
+
- { station: st-9, celsius: 2 }
|
|
62
|
+
|
|
63
|
+
join:
|
|
64
|
+
block: transform.jq
|
|
65
|
+
depends_on: [stations, readings]
|
|
66
|
+
config:
|
|
67
|
+
input:
|
|
68
|
+
stations: ${steps.stations.output.value}
|
|
69
|
+
readings: ${steps.readings.output.value}
|
|
70
|
+
fallback: ${params.default_name}
|
|
71
|
+
program: |
|
|
72
|
+
. as {$stations, $readings, $fallback}
|
|
73
|
+
| ($stations | INDEX(.id)) as $by_id
|
|
74
|
+
| {joined: [$readings[]
|
|
75
|
+
| . + {name: ($by_id[.station].name // $fallback),
|
|
76
|
+
region: ($by_id[.station].region // null)}],
|
|
77
|
+
unmatched: [$readings[] | select($by_id[.station] == null) | .station]}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# Turn long rows -- one record per measurement -- back into one wide row per subject.
|
|
2
|
+
#
|
|
3
|
+
# In: [{station, date, measure, value}, ...].
|
|
4
|
+
# Out: [{station, date, celsius, humidity, rainfall_mm}, ...], one row per station and date,
|
|
5
|
+
# with an explicit null in every cell the long form had no record for.
|
|
6
|
+
#
|
|
7
|
+
# This is jq-pivot-wide-to-long.yaml run backwards, and it is the harder direction, because
|
|
8
|
+
# widening has to decide three things the melt never had to:
|
|
9
|
+
#
|
|
10
|
+
# - What identifies a row. The key is composite here, so group_by takes a list and jq
|
|
11
|
+
# compares those lists element by element.
|
|
12
|
+
# - What the full set of columns is. It is collected from the whole input before the
|
|
13
|
+
# grouping, not from each group, or the rows come out ragged and something downstream
|
|
14
|
+
# has to reconcile them.
|
|
15
|
+
# - What a missing cell means. A blank object of nulls is merged under each group, so
|
|
16
|
+
# every output row carries every column whether that subject reported it or not.
|
|
17
|
+
#
|
|
18
|
+
# from_entries is the inverse of the melt's to_entries. Two records for the same measure in
|
|
19
|
+
# one group would quietly leave the last one standing; catching that is a validate.schema
|
|
20
|
+
# gate's job, never a reshape's.
|
|
21
|
+
#
|
|
22
|
+
# To change it: -p absent=n/a fills unreported cells with a marker instead of null, which is
|
|
23
|
+
# what a csv reader that cannot tell an empty cell from a missing one wants.
|
|
24
|
+
#
|
|
25
|
+
# dg run --local examples/recipes/jq-long-to-wide.yaml
|
|
26
|
+
# dg run --local examples/recipes/jq-long-to-wide.yaml -p absent=n/a
|
|
27
|
+
|
|
28
|
+
format: dirigent/v1
|
|
29
|
+
kind: pipeline
|
|
30
|
+
code: jq-long-to-wide
|
|
31
|
+
name: Pivot long rows to wide
|
|
32
|
+
description: Widen one-record-per-measurement rows into one row per subject, with an explicit fill in every unreported cell.
|
|
33
|
+
|
|
34
|
+
tags: [recipes, transform, jq]
|
|
35
|
+
|
|
36
|
+
requires:
|
|
37
|
+
blocks:
|
|
38
|
+
- value.const
|
|
39
|
+
- transform.jq
|
|
40
|
+
|
|
41
|
+
params:
|
|
42
|
+
type: object
|
|
43
|
+
properties:
|
|
44
|
+
absent:
|
|
45
|
+
type: [string, "null"]
|
|
46
|
+
default: null
|
|
47
|
+
description: What fills a cell the long form carried no record for.
|
|
48
|
+
|
|
49
|
+
steps:
|
|
50
|
+
long:
|
|
51
|
+
block: value.const
|
|
52
|
+
config:
|
|
53
|
+
value:
|
|
54
|
+
- { station: st-1, date: "2026-01-01", measure: celsius, value: 4 }
|
|
55
|
+
- { station: st-1, date: "2026-01-01", measure: humidity, value: 81 }
|
|
56
|
+
- { station: st-1, date: "2026-01-01", measure: rainfall_mm, value: 0 }
|
|
57
|
+
- { station: st-2, date: "2026-01-01", measure: celsius, value: -1 }
|
|
58
|
+
# st-2 reported no humidity that day: the wide row still gets the column.
|
|
59
|
+
- { station: st-2, date: "2026-01-01", measure: rainfall_mm, value: 3 }
|
|
60
|
+
- { station: st-1, date: "2026-01-02", measure: celsius, value: 6 }
|
|
61
|
+
|
|
62
|
+
wide:
|
|
63
|
+
block: transform.jq
|
|
64
|
+
depends_on: [long]
|
|
65
|
+
config:
|
|
66
|
+
input:
|
|
67
|
+
rows: ${steps.long.output.value}
|
|
68
|
+
absent: ${params.absent}
|
|
69
|
+
program: |
|
|
70
|
+
. as {$rows, $absent}
|
|
71
|
+
| ([$rows[].measure] | unique) as $columns
|
|
72
|
+
| ($columns | map({key: ., value: $absent}) | from_entries) as $blank
|
|
73
|
+
| [$rows
|
|
74
|
+
| group_by([.station, .date])[]
|
|
75
|
+
| {station: .[0].station, date: .[0].date}
|
|
76
|
+
+ ($blank + (map({key: .measure, value: .value}) | from_entries))]
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# Flatten deeply nested records into flat ones a csv or a parquet writer will accept.
|
|
2
|
+
#
|
|
3
|
+
# In: records with nested objects and a list -- {id, site: {name, coords: {lat, lon}},
|
|
4
|
+
# tags: [...]}.
|
|
5
|
+
# Out: flat records whose keys are the joined paths -- site.name, site.coords.lat, tags.
|
|
6
|
+
#
|
|
7
|
+
# Every row-oriented sink here refuses a nested value rather than writing it as text:
|
|
8
|
+
# convert.std says so naming the row and the key, and convert.arrow says the same about a
|
|
9
|
+
# parquet column. Flattening is therefore not a nicety before a csv, it is the step that
|
|
10
|
+
# makes the csv possible at all.
|
|
11
|
+
#
|
|
12
|
+
# paths is the whole recipe. It emits the path to every leaf of a value, as an array of
|
|
13
|
+
# keys, and getpath reads the value at one -- so join(".") is the single place that decides
|
|
14
|
+
# what a flattened key looks like, and changing the separator is changing that one string.
|
|
15
|
+
#
|
|
16
|
+
# The predicate handed to paths is where a boolean column goes missing. paths(f) keeps a
|
|
17
|
+
# path when f answers a truthy value, and for a cell holding false the tidy-looking
|
|
18
|
+
# paths(scalars) answers false -- so the row silently loses the column. Writing the
|
|
19
|
+
# predicate as a type comparison keeps it, and st-2's active: false in the output is the
|
|
20
|
+
# proof.
|
|
21
|
+
#
|
|
22
|
+
# Lists are the case worth deciding rather than defaulting. paths walks into them too, which
|
|
23
|
+
# would give tags.0 and tags.1 -- columns that move whenever an element is inserted. So the
|
|
24
|
+
# list-valued fields are pulled out first and joined into one cell, by the step that knows
|
|
25
|
+
# what the separator should be, and everything left flattens by path.
|
|
26
|
+
#
|
|
27
|
+
# To change it: -p separator=__ changes the joined key, which some warehouse loaders prefer
|
|
28
|
+
# to a dot they would otherwise have to quote.
|
|
29
|
+
#
|
|
30
|
+
# dg run --local examples/recipes/jq-nested-to-flat.yaml
|
|
31
|
+
# dg run --local examples/recipes/jq-nested-to-flat.yaml -p separator=__
|
|
32
|
+
|
|
33
|
+
format: dirigent/v1
|
|
34
|
+
kind: pipeline
|
|
35
|
+
code: jq-nested-to-flat
|
|
36
|
+
name: Flatten nested records
|
|
37
|
+
description: Flatten nested records into joined-path keys, with list fields collapsed into one cell rather than indexed columns.
|
|
38
|
+
|
|
39
|
+
tags: [recipes, transform, jq]
|
|
40
|
+
|
|
41
|
+
requires:
|
|
42
|
+
blocks:
|
|
43
|
+
- value.const
|
|
44
|
+
- transform.jq
|
|
45
|
+
|
|
46
|
+
params:
|
|
47
|
+
type: object
|
|
48
|
+
properties:
|
|
49
|
+
separator:
|
|
50
|
+
type: string
|
|
51
|
+
default: "."
|
|
52
|
+
description: What joins the path segments of a flattened key.
|
|
53
|
+
|
|
54
|
+
steps:
|
|
55
|
+
nested:
|
|
56
|
+
block: value.const
|
|
57
|
+
config:
|
|
58
|
+
value:
|
|
59
|
+
- id: st-1
|
|
60
|
+
site: { name: Harbour, coords: { lat: 59.91, lon: 10.75 } }
|
|
61
|
+
tags: [coastal, tidal]
|
|
62
|
+
active: true
|
|
63
|
+
- id: st-2
|
|
64
|
+
site: { name: Ridge, coords: { lat: 60.39, lon: 5.32 } }
|
|
65
|
+
tags: [alpine]
|
|
66
|
+
active: false
|
|
67
|
+
|
|
68
|
+
flat:
|
|
69
|
+
block: transform.jq
|
|
70
|
+
depends_on: [nested]
|
|
71
|
+
config:
|
|
72
|
+
input:
|
|
73
|
+
rows: ${steps.nested.output.value}
|
|
74
|
+
separator: ${params.separator}
|
|
75
|
+
program: |
|
|
76
|
+
. as {$rows, $separator}
|
|
77
|
+
| [$rows[]
|
|
78
|
+
# The list-valued fields are collapsed before the walk, so a new element never
|
|
79
|
+
# adds a column.
|
|
80
|
+
| (to_entries | map(select(.value | type == "array"))
|
|
81
|
+
| map({key: .key, value: (.value | join(","))}) | from_entries) as $lists
|
|
82
|
+
| (to_entries | map(select(.value | type != "array")) | from_entries) as $rest
|
|
83
|
+
# paths(f) keeps a path when f answers a truthy value, so paths(scalars) drops
|
|
84
|
+
# every leaf that is itself false. The predicate is written as a comparison so a
|
|
85
|
+
# false cell survives the flattening.
|
|
86
|
+
| ([$rest | paths(type | . != "object" and . != "array") as $p
|
|
87
|
+
| {key: ($p | join($separator)), value: getpath($p)}]
|
|
88
|
+
| from_entries)
|
|
89
|
+
+ $lists]
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# Turn wide rows -- one column per measure -- into long rows, one record per measurement.
|
|
2
|
+
#
|
|
3
|
+
# In: [{station, date, celsius, humidity, rainfall_mm}, ...], a column per measure.
|
|
4
|
+
# Out: [{station, date, measure, value}, ...], one record per measure per row.
|
|
5
|
+
#
|
|
6
|
+
# Long is the shape almost every sink wants: a data value set, a tidy dataframe, a csv whose
|
|
7
|
+
# header does not change when a new instrument is installed. The verb that gets there is
|
|
8
|
+
# to_entries, which turns whatever is left of the object into {key, value} pairs once the
|
|
9
|
+
# identifying columns have been removed with del. Naming the identifiers once, rather than
|
|
10
|
+
# the measures, is what makes the program survive a new column.
|
|
11
|
+
#
|
|
12
|
+
# Dropping nulls is the decision worth seeing. A wide row carries a cell for a measure that
|
|
13
|
+
# was never taken, and a long row for it would be a claim that a measurement of null
|
|
14
|
+
# happened. select is where that claim is refused.
|
|
15
|
+
#
|
|
16
|
+
# To change it: -p keep_nulls=true emits a long row for every cell, which is what a sink
|
|
17
|
+
# that distinguishes "not reported" from "not measured" wants.
|
|
18
|
+
#
|
|
19
|
+
# dg run --local examples/recipes/jq-pivot-wide-to-long.yaml
|
|
20
|
+
# dg run --local examples/recipes/jq-pivot-wide-to-long.yaml -p keep_nulls=true
|
|
21
|
+
|
|
22
|
+
format: dirigent/v1
|
|
23
|
+
kind: pipeline
|
|
24
|
+
code: jq-pivot-wide-to-long
|
|
25
|
+
name: Pivot wide rows to long
|
|
26
|
+
description: Melt a column-per-measure table into one record per measurement, dropping the cells that were never measured.
|
|
27
|
+
|
|
28
|
+
tags: [recipes, transform, jq]
|
|
29
|
+
|
|
30
|
+
requires:
|
|
31
|
+
blocks:
|
|
32
|
+
- value.const
|
|
33
|
+
- transform.jq
|
|
34
|
+
|
|
35
|
+
params:
|
|
36
|
+
type: object
|
|
37
|
+
properties:
|
|
38
|
+
keep_nulls:
|
|
39
|
+
type: boolean
|
|
40
|
+
default: false
|
|
41
|
+
description: Emit a long row for a measure that has no value, rather than dropping it.
|
|
42
|
+
|
|
43
|
+
steps:
|
|
44
|
+
wide:
|
|
45
|
+
block: value.const
|
|
46
|
+
config:
|
|
47
|
+
value:
|
|
48
|
+
- { station: st-1, date: "2026-01-01", celsius: 4, humidity: 81, rainfall_mm: 0 }
|
|
49
|
+
- { station: st-2, date: "2026-01-01", celsius: -1, humidity: 74, rainfall_mm: null }
|
|
50
|
+
- { station: st-1, date: "2026-01-02", celsius: 6, humidity: null, rainfall_mm: 12 }
|
|
51
|
+
|
|
52
|
+
long:
|
|
53
|
+
block: transform.jq
|
|
54
|
+
depends_on: [wide]
|
|
55
|
+
config:
|
|
56
|
+
input:
|
|
57
|
+
rows: ${steps.wide.output.value}
|
|
58
|
+
keep_nulls: ${params.keep_nulls}
|
|
59
|
+
program: |
|
|
60
|
+
. as {$rows, $keep_nulls}
|
|
61
|
+
| [$rows[]
|
|
62
|
+
| . as $row
|
|
63
|
+
| (del(.station, .date) | to_entries[])
|
|
64
|
+
| select($keep_nulls or .value != null)
|
|
65
|
+
| {station: $row.station, date: $row.date, measure: .key, value: .value}]
|