dirigent-examples 0.15.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_examples/__init__.py +22 -0
- dirigent_examples/py.typed +0 -0
- dirigent_examples/shelves/README.md +299 -0
- dirigent_examples/shelves/composition/README.md +18 -0
- dirigent_examples/shelves/composition/chained-instances.yaml +97 -0
- dirigent_examples/shelves/composition/composition-child.yaml +64 -0
- dirigent_examples/shelves/composition/composition-parent.yaml +89 -0
- dirigent_examples/shelves/connections.yaml +52 -0
- dirigent_examples/shelves/demo/README.md +19 -0
- dirigent_examples/shelves/demo/markdown-showcase.yaml +117 -0
- dirigent_examples/shelves/demo/params-showcase.yaml +92 -0
- dirigent_examples/shelves/demo/requires.yaml +65 -0
- dirigent_examples/shelves/demo/weekly-import-malawi.yaml +54 -0
- dirigent_examples/shelves/demo/weekly-import-nepal.yaml +75 -0
- dirigent_examples/shelves/docker/README.md +29 -0
- dirigent_examples/shelves/docker/docker-build-push.yaml +110 -0
- dirigent_examples/shelves/docker/docker-build-run.yaml +106 -0
- dirigent_examples/shelves/docker/docker-compose-database.yaml +124 -0
- dirigent_examples/shelves/docker/docker-compose-failing-up.yaml +83 -0
- dirigent_examples/shelves/docker/docker-compose-file.yaml +117 -0
- dirigent_examples/shelves/docker/docker-compose-profiles-env.yaml +133 -0
- dirigent_examples/shelves/docker/docker-compose-stack.yaml +70 -0
- dirigent_examples/shelves/docker/docker-hello.yaml +53 -0
- dirigent_examples/shelves/docker/docker-remote-daemon.yaml +92 -0
- dirigent_examples/shelves/docker/docker-run-failing-teardown.yaml +94 -0
- dirigent_examples/shelves/docker/docker-ticker.yaml +49 -0
- dirigent_examples/shelves/execute/README.md +16 -0
- dirigent_examples/shelves/execute/long-log.yaml +89 -0
- dirigent_examples/shelves/failure/README.md +20 -0
- dirigent_examples/shelves/failure/error-handler.yaml +89 -0
- dirigent_examples/shelves/failure/optional-step.yaml +82 -0
- dirigent_examples/shelves/failure/retries.yaml +75 -0
- dirigent_examples/shelves/failure/retry-budget.yaml +82 -0
- dirigent_examples/shelves/failure/step-timeout.yaml +96 -0
- dirigent_examples/shelves/git/README.md +32 -0
- dirigent_examples/shelves/git/git-checkout-build.yaml +125 -0
- dirigent_examples/shelves/git/git-checkout-compose.yaml +138 -0
- dirigent_examples/shelves/git/git-checkout-public.yaml +84 -0
- dirigent_examples/shelves/graph/README.md +22 -0
- dirigent_examples/shelves/graph/deep-chain.yaml +119 -0
- dirigent_examples/shelves/graph/fan-in.yaml +80 -0
- dirigent_examples/shelves/graph/fan-out.yaml +66 -0
- dirigent_examples/shelves/graph/linear.yaml +66 -0
- dirigent_examples/shelves/graph/parallel-branches.yaml +57 -0
- dirigent_examples/shelves/graph/parallel-sleep.yaml +55 -0
- dirigent_examples/shelves/graph/skip-diamond.yaml +105 -0
- dirigent_examples/shelves/graph/wide-fan.yaml +147 -0
- dirigent_examples/shelves/hello-world.yaml +30 -0
- dirigent_examples/shelves/open-data/README.md +67 -0
- dirigent_examples/shelves/open-data/feeds-composition.yaml +120 -0
- dirigent_examples/shelves/open-data/gdacs-disaster-updates.yaml +230 -0
- dirigent_examples/shelves/open-data/github-releases-relay.yaml +225 -0
- dirigent_examples/shelves/open-data/hdx-dataset-watch.yaml +211 -0
- dirigent_examples/shelves/open-data/kobo-submissions-to-csv.yaml +149 -0
- dirigent_examples/shelves/open-data/nominatim-geocode-facilities.yaml +169 -0
- dirigent_examples/shelves/open-data/odk-central-submissions.yaml +146 -0
- dirigent_examples/shelves/open-data/open-meteo-weekly-report.yaml +142 -0
- dirigent_examples/shelves/open-data/overpass-health-facilities.yaml +154 -0
- dirigent_examples/shelves/open-data/usgs-earthquakes-alert.yaml +208 -0
- dirigent_examples/shelves/open-data/who-gho-indicators-to-parquet.yaml +163 -0
- dirigent_examples/shelves/open-data/wikidata-country-reference.yaml +146 -0
- dirigent_examples/shelves/open-data/world-bank-population-trend.yaml +166 -0
- dirigent_examples/shelves/patterns/README.md +144 -0
- dirigent_examples/shelves/patterns/concurrency-queue.yaml +89 -0
- dirigent_examples/shelves/patterns/concurrency-replace.yaml +92 -0
- dirigent_examples/shelves/patterns/concurrency-skip.yaml +97 -0
- dirigent_examples/shelves/patterns/connections-referenced-vs-carried.yaml +140 -0
- dirigent_examples/shelves/patterns/deadline-on-a-sensor.yaml +106 -0
- dirigent_examples/shelves/patterns/fan-out-continue.yaml +88 -0
- dirigent_examples/shelves/patterns/fan-out-fail-fast.yaml +80 -0
- dirigent_examples/shelves/patterns/fan-out-from-params.yaml +84 -0
- dirigent_examples/shelves/patterns/fan-out-item-wise.yaml +111 -0
- dirigent_examples/shelves/patterns/fan-out-literal-list.yaml +81 -0
- dirigent_examples/shelves/patterns/fan-out-nested-objects.yaml +107 -0
- dirigent_examples/shelves/patterns/fan-out-then-join.yaml +86 -0
- dirigent_examples/shelves/patterns/log-levels.yaml +119 -0
- dirigent_examples/shelves/patterns/outputs-inline-vs-storage.yaml +140 -0
- dirigent_examples/shelves/patterns/params-every-type.yaml +259 -0
- dirigent_examples/shelves/patterns/params-validation-refuses.yaml +131 -0
- dirigent_examples/shelves/patterns/pipeline-run-child.yaml +96 -0
- dirigent_examples/shelves/patterns/pipeline-run-fire-and-forget.yaml +99 -0
- dirigent_examples/shelves/patterns/pipeline-run-strict.yaml +103 -0
- dirigent_examples/shelves/patterns/pipeline-run-wait.yaml +107 -0
- dirigent_examples/shelves/patterns/pipeline-run-with-params.yaml +124 -0
- dirigent_examples/shelves/patterns/poll-cadence.yaml +102 -0
- dirigent_examples/shelves/patterns/priority-layered.yaml +120 -0
- dirigent_examples/shelves/patterns/references-cheat-sheet.yaml +186 -0
- dirigent_examples/shelves/patterns/retry-budget-exhausted.yaml +92 -0
- dirigent_examples/shelves/patterns/retry-exponential-backoff.yaml +88 -0
- dirigent_examples/shelves/patterns/retry-only-transient.yaml +108 -0
- dirigent_examples/shelves/patterns/retry-with-jitter.yaml +102 -0
- dirigent_examples/shelves/patterns/rule-all-done.yaml +83 -0
- dirigent_examples/shelves/patterns/rule-all-success.yaml +79 -0
- dirigent_examples/shelves/patterns/rule-always.yaml +92 -0
- dirigent_examples/shelves/patterns/rule-one-failed.yaml +86 -0
- dirigent_examples/shelves/patterns/schedule-at-once.yaml +110 -0
- dirigent_examples/shelves/patterns/schedule-cron-timezone.yaml +121 -0
- dirigent_examples/shelves/patterns/schedule-interval.yaml +109 -0
- dirigent_examples/shelves/patterns/schedule-window-half-open.yaml +105 -0
- dirigent_examples/shelves/patterns/sensor-http-ready.yaml +121 -0
- dirigent_examples/shelves/patterns/sensor-storage-exists.yaml +134 -0
- dirigent_examples/shelves/patterns/step-names-and-keys.yaml +99 -0
- dirigent_examples/shelves/patterns/timeout-fails-the-step.yaml +94 -0
- dirigent_examples/shelves/patterns/timeout-skips-the-step.yaml +102 -0
- dirigent_examples/shelves/patterns/webhook-mapping-nested-payload.yaml +125 -0
- dirigent_examples/shelves/patterns/webhook-signed.yaml +144 -0
- dirigent_examples/shelves/preview/s3-parquet-to-ingestion.yaml +92 -0
- dirigent_examples/shelves/python/README.md +31 -0
- dirigent_examples/shelves/python/apply_and_run.py +52 -0
- dirigent_examples/shelves/python/ci_gate.py +76 -0
- dirigent_examples/shelves/python/connections.py +61 -0
- dirigent_examples/shelves/python/error_handling.py +84 -0
- dirigent_examples/shelves/python/follow_logs.py +39 -0
- dirigent_examples/shelves/python/list_and_filter.py +52 -0
- dirigent_examples/shelves/queues/README.md +59 -0
- dirigent_examples/shelves/queues/kafka-consume-then-transform.yaml +105 -0
- dirigent_examples/shelves/queues/kafka-produce-then-consume.yaml +124 -0
- dirigent_examples/shelves/queues/rabbitmq-consume-ack-on-success.yaml +102 -0
- dirigent_examples/shelves/queues/report-to-kafka.yaml +105 -0
- dirigent_examples/shelves/queues/report-to-rabbitmq.yaml +105 -0
- dirigent_examples/shelves/recipes/README.md +130 -0
- dirigent_examples/shelves/recipes/csv-header-rules.yaml +147 -0
- dirigent_examples/shelves/recipes/csv-to-ndjson.yaml +107 -0
- dirigent_examples/shelves/recipes/etl-csv-clean-validate-parquet.yaml +207 -0
- dirigent_examples/shelves/recipes/filter-by-predicate.yaml +116 -0
- dirigent_examples/shelves/recipes/filter-then-map-then-reduce.yaml +109 -0
- dirigent_examples/shelves/recipes/http-fetch-validate-post.yaml +142 -0
- dirigent_examples/shelves/recipes/http-follow-redirects.yaml +100 -0
- dirigent_examples/shelves/recipes/http-get-with-query.yaml +96 -0
- dirigent_examples/shelves/recipes/http-headers-and-auth-connection.yaml +110 -0
- dirigent_examples/shelves/recipes/http-post-file-from-storage.yaml +117 -0
- dirigent_examples/shelves/recipes/http-post-json-echo.yaml +105 -0
- dirigent_examples/shelves/recipes/http-post-report.yaml +183 -0
- dirigent_examples/shelves/recipes/http-save-body-to-storage.yaml +106 -0
- dirigent_examples/shelves/recipes/http-success-status-list.yaml +80 -0
- dirigent_examples/shelves/recipes/http-timeout-override.yaml +104 -0
- dirigent_examples/shelves/recipes/jq-dedupe-by-key.yaml +78 -0
- dirigent_examples/shelves/recipes/jq-defaults-and-nulls.yaml +91 -0
- dirigent_examples/shelves/recipes/jq-group-by-and-sum.yaml +74 -0
- dirigent_examples/shelves/recipes/jq-join-two-lists.yaml +77 -0
- dirigent_examples/shelves/recipes/jq-long-to-wide.yaml +76 -0
- dirigent_examples/shelves/recipes/jq-nested-to-flat.yaml +89 -0
- dirigent_examples/shelves/recipes/jq-pivot-wide-to-long.yaml +65 -0
- dirigent_examples/shelves/recipes/jq-running-totals.yaml +82 -0
- dirigent_examples/shelves/recipes/jq-string-cleaning.yaml +88 -0
- dirigent_examples/shelves/recipes/jq-top-n.yaml +85 -0
- dirigent_examples/shelves/recipes/jq-validate-in-jq-vs-schema.yaml +124 -0
- dirigent_examples/shelves/recipes/jq-window-dates.yaml +82 -0
- dirigent_examples/shelves/recipes/json-to-csv-flattening.yaml +141 -0
- dirigent_examples/shelves/recipes/large-output-to-storage.yaml +134 -0
- dirigent_examples/shelves/recipes/map-enrich-with-lookup.yaml +96 -0
- dirigent_examples/shelves/recipes/ndjson-to-parquet.yaml +130 -0
- dirigent_examples/shelves/recipes/pagination-by-fan-out.yaml +115 -0
- dirigent_examples/shelves/recipes/parquet-round-trip-types.yaml +163 -0
- dirigent_examples/shelves/recipes/reconcile-two-sources.yaml +159 -0
- dirigent_examples/shelves/recipes/report-built-in.yaml +72 -0
- dirigent_examples/shelves/recipes/report-daily-digest.yaml +186 -0
- dirigent_examples/shelves/recipes/report-to-file.yaml +131 -0
- dirigent_examples/shelves/recipes/report-to-webhook.yaml +136 -0
- dirigent_examples/shelves/recipes/schema-carried.yaml +112 -0
- dirigent_examples/shelves/recipes/schema-formats.yaml +107 -0
- dirigent_examples/shelves/recipes/schema-referenced.yaml +86 -0
- dirigent_examples/shelves/recipes/schema-refuses-then-rule.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-copy-dated-archive.yaml +114 -0
- dirigent_examples/shelves/recipes/storage-exists-gate.yaml +127 -0
- dirigent_examples/shelves/recipes/storage-manifest-of-a-fan-out.yaml +104 -0
- dirigent_examples/shelves/recipes/storage-write-then-read.yaml +119 -0
- dirigent_examples/shelves/recipes/webhook-post-hmac.yaml +132 -0
- dirigent_examples/shelves/recipes/webhook-post-summary.yaml +142 -0
- dirigent_examples/shelves/s3/README.md +34 -0
- dirigent_examples/shelves/s3/report-to-s3.yaml +93 -0
- dirigent_examples/shelves/s3/s3-copy-and-verify.yaml +105 -0
- dirigent_examples/shelves/s3/s3-csv-report.yaml +87 -0
- dirigent_examples/shelves/s3/s3-parquet-report.yaml +77 -0
- dirigent_examples/shelves/s3/s3-round-trip.yaml +125 -0
- dirigent_examples/shelves/schemas/README.md +36 -0
- dirigent_examples/shelves/schemas/echo-reading.json +18 -0
- dirigent_examples/shelves/schemas/ou-record.json +13 -0
- dirigent_examples/shelves/schemas/station-reading.json +13 -0
- dirigent_examples/shelves/sensors/README.md +16 -0
- dirigent_examples/shelves/sensors/sensor-gate.yaml +65 -0
- dirigent_examples/shelves/sensors/time-window.yaml +61 -0
- dirigent_examples/shelves/sql/README.md +52 -0
- dirigent_examples/shelves/sql/duckdb-parquet-to-report.yaml +146 -0
- dirigent_examples/shelves/sql/sql-postgres-readonly.yaml +111 -0
- dirigent_examples/shelves/sql/sql-query-to-storage.yaml +85 -0
- dirigent_examples/shelves/sql/sql-sqlite-roundtrip.yaml +114 -0
- dirigent_examples/shelves/sql/warehouse.sql +42 -0
- dirigent_examples/shelves/transform/README.md +36 -0
- dirigent_examples/shelves/transform/csv-report.yaml +55 -0
- dirigent_examples/shelves/transform/jq-filter-and-map.yaml +70 -0
- dirigent_examples/shelves/transform/jq-group-and-aggregate.yaml +70 -0
- dirigent_examples/shelves/transform/jq-join-two-sources.yaml +98 -0
- dirigent_examples/shelves/transform/jq-reshape.yaml +91 -0
- dirigent_examples/shelves/transform/jq-stream-through-storage.yaml +112 -0
- dirigent_examples/shelves/transform/ndjson-round-trip.yaml +56 -0
- dirigent_examples/shelves/transform/parquet-round-trip.yaml +68 -0
- dirigent_examples/shelves/transform/std-convert-fan-out.yaml +142 -0
- dirigent_examples/shelves/transform/xml-feed-to-ndjson.yaml +116 -0
- dirigent_examples/shelves/transform/yaml-config-to-json.yaml +104 -0
- dirigent_examples/shelves/triggers/README.md +45 -0
- dirigent_examples/shelves/triggers/at-one-time.yaml +78 -0
- dirigent_examples/shelves/triggers/cron-nightly.yaml +79 -0
- dirigent_examples/shelves/triggers/cron-windowed.yaml +86 -0
- dirigent_examples/shelves/triggers/document-nightly.yaml +80 -0
- dirigent_examples/shelves/triggers/interval-rolling.yaml +88 -0
- dirigent_examples/shelves/triggers/managed-and-manual.yaml +109 -0
- dirigent_examples/shelves/triggers/webhook-trigger.yaml +75 -0
- dirigent_examples/shelves/validate/README.md +31 -0
- dirigent_examples/shelves/validate/expects-a-shape.yaml +56 -0
- dirigent_examples/shelves/validate/the-shape-is-wrong.yaml +46 -0
- dirigent_examples-0.15.0.dist-info/METADATA +21 -0
- dirigent_examples-0.15.0.dist-info/RECORD +216 -0
- dirigent_examples-0.15.0.dist-info/WHEEL +4 -0
- dirigent_examples-0.15.0.dist-info/entry_points.txt +3 -0
- dirigent_examples-0.15.0.dist-info/licenses/LICENSE +18 -0
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
# A payload well past the inline threshold, and the two places size actually matters.
|
|
2
|
+
#
|
|
3
|
+
# In: a row count, defaulting to 5000 rows -- a few hundred kilobytes of JSON.
|
|
4
|
+
# Out: one large file in the run's scratch space, a csv beside it, and a small summary read
|
|
5
|
+
# back out of the first.
|
|
6
|
+
#
|
|
7
|
+
# Two different caps get confused with each other, and only one of them is a decision this
|
|
8
|
+
# document makes:
|
|
9
|
+
#
|
|
10
|
+
# inline_artifact_max an instance setting, 16KB by default. Every successful attempt
|
|
11
|
+
# keeps its whole structured output, whatever its size, so
|
|
12
|
+
# ${steps.x.output.*} never pays a storage round trip. What the cap
|
|
13
|
+
# governs is where the artifact copy of that output lives: at or
|
|
14
|
+
# below it, inlined in the attempt's reference row; above it,
|
|
15
|
+
# streamed to the run's scratch prefix with the row holding the URI
|
|
16
|
+
# instead. Nothing in a document says any of this, and the file it
|
|
17
|
+
# leaves behind is named after the attempt rather than after
|
|
18
|
+
# anything a person was looking for.
|
|
19
|
+
# max_size what this document says, on the storage.read below. A value comes
|
|
20
|
+
# back into the run only through a read, and a read holds what it
|
|
21
|
+
# reads, so the step names the ceiling it will hold and an object
|
|
22
|
+
# above it is refused rather than truncated.
|
|
23
|
+
#
|
|
24
|
+
# So the write is not an optimisation of the threshold, it is the alternative to relying on
|
|
25
|
+
# it: a result worth keeping is a result named, written and read back deliberately, and a
|
|
26
|
+
# result nobody names a file for is one the engine spills for you and nobody finds.
|
|
27
|
+
#
|
|
28
|
+
# Five hops, and what each one hands on:
|
|
29
|
+
# generate transform.jq, the big list. Its output is far past the threshold, so this is
|
|
30
|
+
# the attempt whose artifact copy the engine spills for you.
|
|
31
|
+
# store storage.write, the same value under a name a person can look for.
|
|
32
|
+
# as_csv convert.std, URI to URI. The codec never holds the list in either direction.
|
|
33
|
+
# reread storage.read, the file back as a value, bounded by max_size.
|
|
34
|
+
# summary transform.jq, a handful of numbers -- the shape that belongs in a step output,
|
|
35
|
+
# which is read by a person on a run page and by a reference in another step's
|
|
36
|
+
# config, and neither of those wants a megabyte.
|
|
37
|
+
#
|
|
38
|
+
# To change it: -p rows=200000 makes the file tens of megabytes. Past 32mb the reread is
|
|
39
|
+
# refused naming the object and the cap, and raising max_size is the deliberate act that
|
|
40
|
+
# allows it.
|
|
41
|
+
#
|
|
42
|
+
# dg run --local examples/recipes/large-output-to-storage.yaml
|
|
43
|
+
# dg run --local examples/recipes/large-output-to-storage.yaml -p rows=50000
|
|
44
|
+
|
|
45
|
+
format: dirigent/v1
|
|
46
|
+
kind: pipeline
|
|
47
|
+
code: large-output-to-storage
|
|
48
|
+
name: A large result through storage
|
|
49
|
+
description: Generate a payload far past the inline artifact threshold, write it to storage under a name of its own, and keep only a bounded summary in a step output.
|
|
50
|
+
|
|
51
|
+
tags: [recipes, storage, transform]
|
|
52
|
+
|
|
53
|
+
requires:
|
|
54
|
+
blocks:
|
|
55
|
+
- transform.jq
|
|
56
|
+
- storage.write
|
|
57
|
+
- convert.std
|
|
58
|
+
- storage.read
|
|
59
|
+
|
|
60
|
+
params:
|
|
61
|
+
type: object
|
|
62
|
+
properties:
|
|
63
|
+
rows:
|
|
64
|
+
type: integer
|
|
65
|
+
minimum: 1
|
|
66
|
+
default: 5000
|
|
67
|
+
description: How many rows are generated; 5000 is a few hundred kilobytes of JSON.
|
|
68
|
+
day:
|
|
69
|
+
type: string
|
|
70
|
+
format: date
|
|
71
|
+
default: "2026-01-01"
|
|
72
|
+
description: The day the file is named for.
|
|
73
|
+
|
|
74
|
+
steps:
|
|
75
|
+
generate:
|
|
76
|
+
block: transform.jq
|
|
77
|
+
config:
|
|
78
|
+
input:
|
|
79
|
+
rows: ${params.rows}
|
|
80
|
+
day: ${params.day}
|
|
81
|
+
program: |
|
|
82
|
+
. as {$rows, $day}
|
|
83
|
+
| [range($rows)
|
|
84
|
+
| {id: "row-\(.)",
|
|
85
|
+
day: $day,
|
|
86
|
+
station: "st-\(. % 250)",
|
|
87
|
+
celsius: ((. % 37) - 12),
|
|
88
|
+
note: "generated for the storage recipe"}]
|
|
89
|
+
|
|
90
|
+
store:
|
|
91
|
+
block: storage.write
|
|
92
|
+
depends_on: [generate]
|
|
93
|
+
config:
|
|
94
|
+
target: ${run.scratch}/large-${params.day}.json
|
|
95
|
+
value: ${steps.generate.output.value}
|
|
96
|
+
|
|
97
|
+
as_csv:
|
|
98
|
+
block: convert.std
|
|
99
|
+
depends_on: [store]
|
|
100
|
+
config:
|
|
101
|
+
source: ${steps.store.output.uri}
|
|
102
|
+
target: ${run.scratch}/large-${params.day}.csv
|
|
103
|
+
from: json
|
|
104
|
+
to: csv
|
|
105
|
+
|
|
106
|
+
reread:
|
|
107
|
+
block: storage.read
|
|
108
|
+
depends_on: [store, as_csv]
|
|
109
|
+
config:
|
|
110
|
+
source: ${steps.store.output.uri}
|
|
111
|
+
# Written out here because this is the step where a bigger -p rows would meet it.
|
|
112
|
+
max_size: 32mb
|
|
113
|
+
|
|
114
|
+
summary:
|
|
115
|
+
block: transform.jq
|
|
116
|
+
depends_on: [reread]
|
|
117
|
+
config:
|
|
118
|
+
input: ${steps.reread.output.value}
|
|
119
|
+
program: |
|
|
120
|
+
{rows: length,
|
|
121
|
+
stations: ([.[].station] | unique | length),
|
|
122
|
+
mean_celsius: ([.[].celsius] | add / length)}
|
|
123
|
+
|
|
124
|
+
sizes:
|
|
125
|
+
block: transform.jq
|
|
126
|
+
depends_on: [store, as_csv, summary]
|
|
127
|
+
config:
|
|
128
|
+
input:
|
|
129
|
+
json_bytes: ${steps.store.output.bytes_written}
|
|
130
|
+
csv_bytes: ${steps.as_csv.output.bytes_written}
|
|
131
|
+
summary: ${steps.summary.output.value}
|
|
132
|
+
program: |
|
|
133
|
+
. + {inline_threshold_bytes: 16384,
|
|
134
|
+
json_over_threshold: (.json_bytes > 16384)}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# Enrich every row from a lookup table, and see why the table cannot live in the map.
|
|
2
|
+
#
|
|
3
|
+
# In: a list of readings, and a small dimension table of stations.
|
|
4
|
+
# Out: one enriched row per reading, same length as the input, each carrying the station's
|
|
5
|
+
# name and region and a derived fahrenheit column.
|
|
6
|
+
#
|
|
7
|
+
# A map.jq program is handed one element and nothing else. There is no second input, no
|
|
8
|
+
# --arg, and no way for the program to reach another step's output, so the lookup table has
|
|
9
|
+
# to arrive attached to the elements. That attaching is a whole-value job, and it is the
|
|
10
|
+
# transform.jq step above the map: it indexes the dimension once and merges the matched
|
|
11
|
+
# fields into each row.
|
|
12
|
+
#
|
|
13
|
+
# Splicing the table into the program text with a ${...} would work exactly once, and would
|
|
14
|
+
# cost the thing that makes a jq step cheap: a program with a reference in it cannot be
|
|
15
|
+
# compiled when the document is applied, so a syntax error would wait for the run.
|
|
16
|
+
#
|
|
17
|
+
# What is left for map.jq is the per-element arithmetic, and there the verb is worth having.
|
|
18
|
+
# The frame calls the program once per element and asserts the length afterwards, so this
|
|
19
|
+
# step cannot drop a row or invent one, and a program that fails does it naming the element
|
|
20
|
+
# by its 0-based index instead of failing the whole batch anonymously.
|
|
21
|
+
#
|
|
22
|
+
# To change it: -p require_match=true makes the enrichment refuse a reading whose station is
|
|
23
|
+
# not in the dimension. That run fails on purpose, with "element 2: no station row for
|
|
24
|
+
# st-9" -- the loud alternative to the null-filled row the default produces.
|
|
25
|
+
#
|
|
26
|
+
# dg run --local examples/recipes/map-enrich-with-lookup.yaml
|
|
27
|
+
# dg run --local examples/recipes/map-enrich-with-lookup.yaml -p require_match=true
|
|
28
|
+
|
|
29
|
+
format: dirigent/v1
|
|
30
|
+
kind: pipeline
|
|
31
|
+
code: map-enrich-with-lookup
|
|
32
|
+
name: Enrich rows from a lookup table
|
|
33
|
+
description: Merge a dimension table into every row with a whole-value step, then compute the per-row columns with map.jq.
|
|
34
|
+
|
|
35
|
+
tags: [recipes, transform, map]
|
|
36
|
+
|
|
37
|
+
requires:
|
|
38
|
+
blocks:
|
|
39
|
+
- value.const
|
|
40
|
+
- transform.jq
|
|
41
|
+
- map.jq
|
|
42
|
+
|
|
43
|
+
params:
|
|
44
|
+
type: object
|
|
45
|
+
properties:
|
|
46
|
+
require_match:
|
|
47
|
+
type: boolean
|
|
48
|
+
default: false
|
|
49
|
+
description: Fail the element whose station is in no dimension row, rather than filling nulls.
|
|
50
|
+
|
|
51
|
+
steps:
|
|
52
|
+
stations:
|
|
53
|
+
block: value.const
|
|
54
|
+
config:
|
|
55
|
+
value:
|
|
56
|
+
- { id: st-1, name: Harbour, region: east }
|
|
57
|
+
- { id: st-2, name: Ridge, region: west }
|
|
58
|
+
|
|
59
|
+
readings:
|
|
60
|
+
block: value.const
|
|
61
|
+
config:
|
|
62
|
+
value:
|
|
63
|
+
- { station: st-1, celsius: 4 }
|
|
64
|
+
- { station: st-2, celsius: -1 }
|
|
65
|
+
- { station: st-9, celsius: 21 }
|
|
66
|
+
|
|
67
|
+
attach:
|
|
68
|
+
block: transform.jq
|
|
69
|
+
depends_on: [stations, readings]
|
|
70
|
+
config:
|
|
71
|
+
input:
|
|
72
|
+
stations: ${steps.stations.output.value}
|
|
73
|
+
readings: ${steps.readings.output.value}
|
|
74
|
+
require_match: ${params.require_match}
|
|
75
|
+
program: |
|
|
76
|
+
. as {$stations, $readings, $require_match}
|
|
77
|
+
| ($stations | INDEX(.id)) as $by_id
|
|
78
|
+
| [$readings[]
|
|
79
|
+
| . + {name: ($by_id[.station].name // null),
|
|
80
|
+
region: ($by_id[.station].region // null),
|
|
81
|
+
require_match: $require_match}]
|
|
82
|
+
|
|
83
|
+
enrich:
|
|
84
|
+
block: map.jq
|
|
85
|
+
depends_on: [attach]
|
|
86
|
+
config:
|
|
87
|
+
input: ${steps.attach.output.value}
|
|
88
|
+
# error() fails the element, and the frame reports which index it was, so a run that
|
|
89
|
+
# refuses an unmatched reading still says which one.
|
|
90
|
+
program: |
|
|
91
|
+
if .require_match and .name == null
|
|
92
|
+
then error("no station row for \(.station)")
|
|
93
|
+
else {station, name, region,
|
|
94
|
+
celsius,
|
|
95
|
+
fahrenheit: (.celsius * 9 / 5 + 32 | round)}
|
|
96
|
+
end
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# Records written as parquet, and read back to prove what landed.
|
|
2
|
+
#
|
|
3
|
+
# In: a list of typed records, built inline.
|
|
4
|
+
# Out: a parquet file in the run's scratch space, and the same records read back out of it.
|
|
5
|
+
#
|
|
6
|
+
# A conversion is a storage-object operation, like storage.copy: it reads one URI, writes
|
|
7
|
+
# another, and never holds the records as a value. So a value in the run is written before
|
|
8
|
+
# it is converted, and a result the run wants to look at is read back after -- which is what
|
|
9
|
+
# the first and last hops here are. Parquet makes that concrete, being bytes rather than
|
|
10
|
+
# text: there has never been an inline spelling for it.
|
|
11
|
+
#
|
|
12
|
+
# Six hops, and what each one hands on:
|
|
13
|
+
# rows value.const, the records a real pipeline would have fetched.
|
|
14
|
+
# stored storage.write, that value as a JSON object. Output: uri, bytes_written.
|
|
15
|
+
# as_ndjson convert.std, json to ndjson. One record per line, no array to hold, which
|
|
16
|
+
# is the shape a stream of records already has.
|
|
17
|
+
# write convert.arrow, ndjson to parquet. Output: source, target, bytes_written.
|
|
18
|
+
# read_back convert.arrow, parquet to json, into a file of its own.
|
|
19
|
+
# landed storage.read, that file as a value, which is what the summary counts.
|
|
20
|
+
#
|
|
21
|
+
# convert.arrow infers one type per column from the records: boolean, int64, float64, or
|
|
22
|
+
# string. A column whose rows disagree is refused naming the column rather than coerced, and
|
|
23
|
+
# integers and floats unify to a float column because JSON calls both a number.
|
|
24
|
+
#
|
|
25
|
+
# The read back is not decoration. It is how a run says what actually landed rather than
|
|
26
|
+
# what was intended: the file's own bytes are decoded, and the row count and the columns
|
|
27
|
+
# come from the file, so a truncated or half-written object shows up here as a failure
|
|
28
|
+
# instead of downstream tomorrow.
|
|
29
|
+
#
|
|
30
|
+
# convert.arrow ships in dirigent-parquet, on pyarrow, where convert.std deliberately stands
|
|
31
|
+
# on the standard library alone. Installing that pack is what puts the block in the catalog,
|
|
32
|
+
# and naming it in requires.blocks is what refuses this document on an instance without it.
|
|
33
|
+
#
|
|
34
|
+
# To change it: point the targets at s3://<bucket>/... and the same steps write and read an
|
|
35
|
+
# object store, because the scheme is the only thing that changes.
|
|
36
|
+
#
|
|
37
|
+
# dg run --local examples/recipes/ndjson-to-parquet.yaml
|
|
38
|
+
# dg run --local examples/recipes/ndjson-to-parquet.yaml -p day=2026-02-01
|
|
39
|
+
|
|
40
|
+
format: dirigent/v1
|
|
41
|
+
kind: pipeline
|
|
42
|
+
code: ndjson-to-parquet
|
|
43
|
+
name: Records to parquet and back
|
|
44
|
+
description: Write typed records into the run's scratch space, convert them through ndjson into parquet, and read the file back to see what landed.
|
|
45
|
+
|
|
46
|
+
tags: [recipes, storage, transform, parquet, starter]
|
|
47
|
+
|
|
48
|
+
requires:
|
|
49
|
+
blocks:
|
|
50
|
+
- value.const
|
|
51
|
+
- storage.write
|
|
52
|
+
- convert.std
|
|
53
|
+
- convert.arrow
|
|
54
|
+
- storage.read
|
|
55
|
+
- transform.jq
|
|
56
|
+
|
|
57
|
+
params:
|
|
58
|
+
type: object
|
|
59
|
+
properties:
|
|
60
|
+
day:
|
|
61
|
+
type: string
|
|
62
|
+
format: date
|
|
63
|
+
default: "2026-01-01"
|
|
64
|
+
description: The day the written file is named for.
|
|
65
|
+
|
|
66
|
+
steps:
|
|
67
|
+
rows:
|
|
68
|
+
block: value.const
|
|
69
|
+
config:
|
|
70
|
+
value:
|
|
71
|
+
- { station: st-1, celsius: 4.5, readings: 1440, active: true }
|
|
72
|
+
- { station: st-2, celsius: -1.0, readings: 1200, active: false }
|
|
73
|
+
- { station: st-3, celsius: 7.25, readings: 1439, active: true }
|
|
74
|
+
|
|
75
|
+
stored:
|
|
76
|
+
block: storage.write
|
|
77
|
+
depends_on: [rows]
|
|
78
|
+
config:
|
|
79
|
+
target: ${run.scratch}/readings-${params.day}.json
|
|
80
|
+
value: ${steps.rows.output.value}
|
|
81
|
+
|
|
82
|
+
as_ndjson:
|
|
83
|
+
block: convert.std
|
|
84
|
+
depends_on: [stored]
|
|
85
|
+
config:
|
|
86
|
+
source: ${steps.stored.output.uri}
|
|
87
|
+
target: ${run.scratch}/readings-${params.day}.ndjson
|
|
88
|
+
from: json
|
|
89
|
+
to: ndjson
|
|
90
|
+
|
|
91
|
+
write:
|
|
92
|
+
block: convert.arrow
|
|
93
|
+
depends_on: [as_ndjson]
|
|
94
|
+
config:
|
|
95
|
+
source: ${steps.as_ndjson.output.target}
|
|
96
|
+
target: ${run.scratch}/readings-${params.day}.parquet
|
|
97
|
+
from: ndjson
|
|
98
|
+
to: parquet
|
|
99
|
+
|
|
100
|
+
read_back:
|
|
101
|
+
block: convert.arrow
|
|
102
|
+
depends_on: [write]
|
|
103
|
+
config:
|
|
104
|
+
# The URI the write reported, rather than the same expression written twice: one place
|
|
105
|
+
# decides where the file went.
|
|
106
|
+
source: ${steps.write.output.target}
|
|
107
|
+
target: ${run.scratch}/readings-${params.day}-back.json
|
|
108
|
+
from: parquet
|
|
109
|
+
to: json
|
|
110
|
+
|
|
111
|
+
landed:
|
|
112
|
+
block: storage.read
|
|
113
|
+
depends_on: [read_back]
|
|
114
|
+
config:
|
|
115
|
+
source: ${steps.read_back.output.target}
|
|
116
|
+
max_size: 1mb
|
|
117
|
+
|
|
118
|
+
summary:
|
|
119
|
+
block: transform.jq
|
|
120
|
+
depends_on: [write, landed]
|
|
121
|
+
config:
|
|
122
|
+
input:
|
|
123
|
+
uri: ${steps.write.output.target}
|
|
124
|
+
file_bytes: ${steps.write.output.bytes_written}
|
|
125
|
+
records: ${steps.landed.output.value}
|
|
126
|
+
program: |
|
|
127
|
+
{uri, file_bytes,
|
|
128
|
+
rows: (.records | length),
|
|
129
|
+
columns: (.records[0] | keys),
|
|
130
|
+
records}
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
# Fetching several pages at once, with the page numbers as a literal list.
|
|
2
|
+
#
|
|
3
|
+
# In: a list of page numbers, as a parameter.
|
|
4
|
+
# Out: one HTTP call per page, each with its own status and retry budget, and a fan-in step
|
|
5
|
+
# that stitches the pages back into one list.
|
|
6
|
+
#
|
|
7
|
+
# for_each is expanded when the run is created, so it reads params.*, item and run.* -- and
|
|
8
|
+
# never another step's output. That is the constraint this recipe is built around, and it
|
|
9
|
+
# cuts both ways:
|
|
10
|
+
#
|
|
11
|
+
# What you get. The item grid exists from the moment the run is visible, so a person
|
|
12
|
+
# watching sees six pages and their statuses rather than a step that might turn into six
|
|
13
|
+
# later. Items run concurrently, each with its own status, error and retry budget, so one
|
|
14
|
+
# slow page does not serialise the rest and one failed page is recorded against itself.
|
|
15
|
+
# What you give up. A fan-out whose cardinality depends on what an earlier call returned
|
|
16
|
+
# is not expressible: this is why the pages are a literal list. When the count really is
|
|
17
|
+
# unknown, either the first call reports it and a second pipeline is run with it as a
|
|
18
|
+
# parameter, or the pipeline walks the cursor one call at a time and gives up the
|
|
19
|
+
# concurrency.
|
|
20
|
+
#
|
|
21
|
+
# items decides what one bad page means. Under the default, fail_fast, any failed item fails
|
|
22
|
+
# the step. Under continue, the step succeeds as long as one item did, the failures are
|
|
23
|
+
# recorded against their own items, and the run reports completed_with_errors -- which is
|
|
24
|
+
# the setting for a report that is worth producing from five pages out of six, and the wrong
|
|
25
|
+
# setting for a load that must be whole.
|
|
26
|
+
#
|
|
27
|
+
# The fan-in reads ${steps.pages.output} -- the list of the items' outputs, in item order --
|
|
28
|
+
# rather than .output.value, which is a field of one item's output and not of the list. Only
|
|
29
|
+
# items that succeeded contribute.
|
|
30
|
+
#
|
|
31
|
+
# https://postman-echo.com/get echoes the query it was given, so each item's response
|
|
32
|
+
# carries its own page number and the stitching is visibly per-page.
|
|
33
|
+
#
|
|
34
|
+
# To change it: -p pages='[1,2]' fetches fewer; -p page_size=10 changes what each call asks
|
|
35
|
+
# for. In a real pipeline the pages usually come from an earlier run's count.
|
|
36
|
+
#
|
|
37
|
+
# dg run --local examples/recipes/pagination-by-fan-out.yaml
|
|
38
|
+
# dg run --local examples/recipes/pagination-by-fan-out.yaml -p pages='[1,2]'
|
|
39
|
+
|
|
40
|
+
format: dirigent/v1
|
|
41
|
+
kind: pipeline
|
|
42
|
+
code: pagination-by-fan-out
|
|
43
|
+
name: Pagination as a fan-out
|
|
44
|
+
description: Fetch a literal list of pages concurrently with for_each, each item its own attempt, then stitch the pages into one list.
|
|
45
|
+
|
|
46
|
+
tags: [recipes, http, transform, graph]
|
|
47
|
+
|
|
48
|
+
requires:
|
|
49
|
+
blocks:
|
|
50
|
+
- http.request
|
|
51
|
+
- transform.jq
|
|
52
|
+
|
|
53
|
+
params:
|
|
54
|
+
type: object
|
|
55
|
+
properties:
|
|
56
|
+
pages:
|
|
57
|
+
type: array
|
|
58
|
+
default: [1, 2, 3, 4]
|
|
59
|
+
items:
|
|
60
|
+
type: integer
|
|
61
|
+
minimum: 1
|
|
62
|
+
description: The page numbers to fetch; one item, and one call, per element.
|
|
63
|
+
page_size:
|
|
64
|
+
type: integer
|
|
65
|
+
minimum: 1
|
|
66
|
+
default: 25
|
|
67
|
+
description: What each call asks for.
|
|
68
|
+
day:
|
|
69
|
+
type: string
|
|
70
|
+
format: date
|
|
71
|
+
default: "2026-01-01"
|
|
72
|
+
description: The day every page is fetched for.
|
|
73
|
+
|
|
74
|
+
steps:
|
|
75
|
+
pages:
|
|
76
|
+
block: http.request
|
|
77
|
+
for_each: ${params.pages}
|
|
78
|
+
# fail_fast: a partial read is not a page of results, it is a gap. A report that is
|
|
79
|
+
# worth producing from most of its pages says items: continue instead.
|
|
80
|
+
items: fail_fast
|
|
81
|
+
# Each item gets this budget of its own, so one flaky page is retried and the rest are
|
|
82
|
+
# untouched.
|
|
83
|
+
retry:
|
|
84
|
+
max_attempts: 3
|
|
85
|
+
backoff: 1s
|
|
86
|
+
config:
|
|
87
|
+
url: https://postman-echo.com/get
|
|
88
|
+
method: GET
|
|
89
|
+
query:
|
|
90
|
+
page: ${item}
|
|
91
|
+
page_size: ${params.page_size}
|
|
92
|
+
day: ${params.day}
|
|
93
|
+
timeout: 20s
|
|
94
|
+
|
|
95
|
+
stitch:
|
|
96
|
+
block: transform.jq
|
|
97
|
+
depends_on: [pages]
|
|
98
|
+
config:
|
|
99
|
+
# The fan-out's output is the list of its items' outputs, in item order.
|
|
100
|
+
input: ${steps.pages.output}
|
|
101
|
+
program: |
|
|
102
|
+
. as $items
|
|
103
|
+
| [$items[]
|
|
104
|
+
| {page: (.body.args.page | tonumber),
|
|
105
|
+
status: .status,
|
|
106
|
+
duration_ms: .duration_ms,
|
|
107
|
+
url: .body.url}] as $fetched
|
|
108
|
+
| {pages: ($fetched | length),
|
|
109
|
+
# In item order, which is the order the pages were asked for and not the order
|
|
110
|
+
# they came back in.
|
|
111
|
+
page_numbers: [$fetched[].page],
|
|
112
|
+
ordered: ([$fetched[].page] == ([$fetched[].page] | sort)),
|
|
113
|
+
slowest_ms: ([$fetched[].duration_ms] | max),
|
|
114
|
+
total_ms: ([$fetched[].duration_ms] | add),
|
|
115
|
+
fetched: $fetched}
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# What survives a round trip through parquet, and what a csv would have flattened instead.
|
|
2
|
+
#
|
|
3
|
+
# In: one record carrying every JSON scalar -- a string, an integer, a float, a boolean, a
|
|
4
|
+
# null, a numeric string, and a column whose values are integers and floats together.
|
|
5
|
+
# Out: {before, after_parquet, after_csv, kinds, changed}, so the two formats can be
|
|
6
|
+
# compared column by column in one output.
|
|
7
|
+
#
|
|
8
|
+
# Parquet has a schema and csv does not, and that single difference is the whole recipe:
|
|
9
|
+
#
|
|
10
|
+
# integer survives as an int64.
|
|
11
|
+
# float survives as a float64.
|
|
12
|
+
# boolean survives as a boolean, not as "true".
|
|
13
|
+
# null survives as a null, in a column that stays typed.
|
|
14
|
+
# numeric string survives as a string. "007" comes back "007", where a csv reader is
|
|
15
|
+
# left guessing and a spreadsheet would have eaten the leading zeros.
|
|
16
|
+
# mixed number integers and floats unify to one float column, because JSON calls both
|
|
17
|
+
# a number. The values still come back equal, since JSON has one number
|
|
18
|
+
# spelling, so `changed.parquet` is empty -- the loss is in the column's
|
|
19
|
+
# type, which a later reader sees and a jq comparison cannot.
|
|
20
|
+
#
|
|
21
|
+
# The csv leg is the control. Every cell in a csv is text on the way out and text on the way
|
|
22
|
+
# back, so after it every value is a string and the booleans, the numbers and the nulls are
|
|
23
|
+
# indistinguishable from the words that spell them.
|
|
24
|
+
#
|
|
25
|
+
# Both legs are four hops, and they are four rather than two because a conversion is a
|
|
26
|
+
# storage-object operation: it reads one URI and writes another, and never carries a value.
|
|
27
|
+
# The record is written once, converted out, converted back, and read into the run again --
|
|
28
|
+
# so `before` is the value this document wrote and `after` is what came off a disk.
|
|
29
|
+
#
|
|
30
|
+
# The comparison is computed rather than described: `changed` lists the fields whose value
|
|
31
|
+
# is not identical after the round trip, so this recipe stays honest if a codec ever changes
|
|
32
|
+
# under it.
|
|
33
|
+
#
|
|
34
|
+
# A column whose rows disagree in a way that cannot unify is refused by name at the write
|
|
35
|
+
# rather than coerced. Adding {"code": 7} to one of the rows below is the fastest way to see
|
|
36
|
+
# it: "column 'code' holds both integer and string values, and a parquet column carries one
|
|
37
|
+
# type; reshape it before converting".
|
|
38
|
+
#
|
|
39
|
+
# dg run --local examples/recipes/parquet-round-trip-types.yaml
|
|
40
|
+
|
|
41
|
+
format: dirigent/v1
|
|
42
|
+
kind: pipeline
|
|
43
|
+
code: parquet-round-trip-types
|
|
44
|
+
name: What survives a parquet round trip
|
|
45
|
+
description: Write one record of every JSON scalar, convert it out to parquet and back, and compare it against the same record round-tripped through csv.
|
|
46
|
+
|
|
47
|
+
tags: [recipes, storage, transform, parquet]
|
|
48
|
+
|
|
49
|
+
requires:
|
|
50
|
+
blocks:
|
|
51
|
+
- value.const
|
|
52
|
+
- storage.write
|
|
53
|
+
- convert.std
|
|
54
|
+
- convert.arrow
|
|
55
|
+
- storage.read
|
|
56
|
+
- transform.jq
|
|
57
|
+
|
|
58
|
+
params:
|
|
59
|
+
type: object
|
|
60
|
+
properties:
|
|
61
|
+
day:
|
|
62
|
+
type: string
|
|
63
|
+
format: date
|
|
64
|
+
default: "2026-01-01"
|
|
65
|
+
description: The day the written file is named for.
|
|
66
|
+
|
|
67
|
+
steps:
|
|
68
|
+
rows:
|
|
69
|
+
block: value.const
|
|
70
|
+
config:
|
|
71
|
+
value:
|
|
72
|
+
- text: Harbour
|
|
73
|
+
whole: 1440
|
|
74
|
+
fractional: 4.5
|
|
75
|
+
flag: true
|
|
76
|
+
absent: null
|
|
77
|
+
code: "007"
|
|
78
|
+
mixed: 1
|
|
79
|
+
- text: Ridge
|
|
80
|
+
whole: 1200
|
|
81
|
+
fractional: -1.25
|
|
82
|
+
flag: false
|
|
83
|
+
absent: null
|
|
84
|
+
code: "042"
|
|
85
|
+
# The other value in this column is an integer, so the column unifies to float.
|
|
86
|
+
mixed: 2.5
|
|
87
|
+
|
|
88
|
+
stored:
|
|
89
|
+
block: storage.write
|
|
90
|
+
depends_on: [rows]
|
|
91
|
+
config:
|
|
92
|
+
target: ${run.scratch}/types-${params.day}.json
|
|
93
|
+
value: ${steps.rows.output.value}
|
|
94
|
+
|
|
95
|
+
write_parquet:
|
|
96
|
+
block: convert.arrow
|
|
97
|
+
depends_on: [stored]
|
|
98
|
+
config:
|
|
99
|
+
source: ${steps.stored.output.uri}
|
|
100
|
+
target: ${run.scratch}/types-${params.day}.parquet
|
|
101
|
+
from: json
|
|
102
|
+
to: parquet
|
|
103
|
+
|
|
104
|
+
back_from_parquet:
|
|
105
|
+
block: convert.arrow
|
|
106
|
+
depends_on: [write_parquet]
|
|
107
|
+
config:
|
|
108
|
+
source: ${steps.write_parquet.output.target}
|
|
109
|
+
target: ${run.scratch}/types-${params.day}-from-parquet.json
|
|
110
|
+
from: parquet
|
|
111
|
+
to: json
|
|
112
|
+
|
|
113
|
+
read_parquet:
|
|
114
|
+
block: storage.read
|
|
115
|
+
depends_on: [back_from_parquet]
|
|
116
|
+
config:
|
|
117
|
+
source: ${steps.back_from_parquet.output.target}
|
|
118
|
+
max_size: 1mb
|
|
119
|
+
|
|
120
|
+
write_csv:
|
|
121
|
+
block: convert.std
|
|
122
|
+
depends_on: [stored]
|
|
123
|
+
config:
|
|
124
|
+
source: ${steps.stored.output.uri}
|
|
125
|
+
target: ${run.scratch}/types-${params.day}.csv
|
|
126
|
+
from: json
|
|
127
|
+
to: csv
|
|
128
|
+
|
|
129
|
+
back_from_csv:
|
|
130
|
+
block: convert.std
|
|
131
|
+
depends_on: [write_csv]
|
|
132
|
+
config:
|
|
133
|
+
source: ${steps.write_csv.output.target}
|
|
134
|
+
target: ${run.scratch}/types-${params.day}-from-csv.json
|
|
135
|
+
from: csv
|
|
136
|
+
to: json
|
|
137
|
+
|
|
138
|
+
read_csv:
|
|
139
|
+
block: storage.read
|
|
140
|
+
depends_on: [back_from_csv]
|
|
141
|
+
config:
|
|
142
|
+
source: ${steps.back_from_csv.output.target}
|
|
143
|
+
max_size: 1mb
|
|
144
|
+
|
|
145
|
+
compare:
|
|
146
|
+
block: transform.jq
|
|
147
|
+
depends_on: [rows, read_parquet, read_csv]
|
|
148
|
+
config:
|
|
149
|
+
input:
|
|
150
|
+
before: ${steps.rows.output.value}
|
|
151
|
+
after_parquet: ${steps.read_parquet.output.value}
|
|
152
|
+
after_csv: ${steps.read_csv.output.value}
|
|
153
|
+
program: |
|
|
154
|
+
def kinds: map_values(type);
|
|
155
|
+
. as {$before, $after_parquet, $after_csv}
|
|
156
|
+
| {before: $before[0], after_parquet: $after_parquet[0], after_csv: $after_csv[0],
|
|
157
|
+
kinds: {before: ($before[0] | kinds),
|
|
158
|
+
after_parquet: ($after_parquet[0] | kinds),
|
|
159
|
+
after_csv: ($after_csv[0] | kinds)},
|
|
160
|
+
changed: {parquet: [$before[0] | to_entries[]
|
|
161
|
+
| select($after_parquet[0][.key] != .value) | .key],
|
|
162
|
+
csv: [$before[0] | to_entries[]
|
|
163
|
+
| select($after_csv[0][.key] != .value) | .key]}}
|