didntrun-sdk 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. didntrun_sdk-1.0.0/.gitignore +7 -0
  2. didntrun_sdk-1.0.0/LICENSE +21 -0
  3. didntrun_sdk-1.0.0/PKG-INFO +276 -0
  4. didntrun_sdk-1.0.0/README.md +248 -0
  5. didntrun_sdk-1.0.0/pyproject.toml +104 -0
  6. didntrun_sdk-1.0.0/src/didntrun/__init__.py +25 -0
  7. didntrun_sdk-1.0.0/src/didntrun/cli.py +351 -0
  8. didntrun_sdk-1.0.0/src/didntrun/external_id.py +33 -0
  9. didntrun_sdk-1.0.0/src/didntrun/log_text.py +15 -0
  10. didntrun_sdk-1.0.0/src/didntrun/ping_kind.py +12 -0
  11. didntrun_sdk-1.0.0/src/didntrun/py.typed +0 -0
  12. didntrun_sdk-1.0.0/src/didntrun/transport/__init__.py +4 -0
  13. didntrun_sdk-1.0.0/src/didntrun/transport/base.py +84 -0
  14. didntrun_sdk-1.0.0/src/didntrun/transport/http_client.py +163 -0
  15. didntrun_sdk-1.0.0/src/didntrun/watchdog.py +508 -0
  16. didntrun_sdk-1.0.0/tests/__init__.py +0 -0
  17. didntrun_sdk-1.0.0/tests/contract/__init__.py +0 -0
  18. didntrun_sdk-1.0.0/tests/contract/conftest.py +75 -0
  19. didntrun_sdk-1.0.0/tests/contract/test_ping_contract.py +107 -0
  20. didntrun_sdk-1.0.0/tests/unit/__init__.py +0 -0
  21. didntrun_sdk-1.0.0/tests/unit/conftest.py +224 -0
  22. didntrun_sdk-1.0.0/tests/unit/test_cli_ping.py +104 -0
  23. didntrun_sdk-1.0.0/tests/unit/test_cli_process.py +165 -0
  24. didntrun_sdk-1.0.0/tests/unit/test_cli_run.py +197 -0
  25. didntrun_sdk-1.0.0/tests/unit/test_conformance_coverage.py +56 -0
  26. didntrun_sdk-1.0.0/tests/unit/test_external_id.py +34 -0
  27. didntrun_sdk-1.0.0/tests/unit/test_http_client_transport.py +199 -0
  28. didntrun_sdk-1.0.0/tests/unit/test_log_text.py +17 -0
  29. didntrun_sdk-1.0.0/tests/unit/test_package.py +12 -0
  30. didntrun_sdk-1.0.0/tests/unit/test_ping_kind.py +27 -0
  31. didntrun_sdk-1.0.0/tests/unit/test_public_api.py +28 -0
  32. didntrun_sdk-1.0.0/tests/unit/test_transport_base.py +42 -0
  33. didntrun_sdk-1.0.0/tests/unit/test_watchdog_construction.py +90 -0
  34. didntrun_sdk-1.0.0/tests/unit/test_watchdog_dsn.py +117 -0
  35. didntrun_sdk-1.0.0/tests/unit/test_watchdog_logging.py +158 -0
  36. didntrun_sdk-1.0.0/tests/unit/test_watchdog_retry.py +196 -0
  37. didntrun_sdk-1.0.0/tests/unit/test_watchdog_run.py +249 -0
  38. didntrun_sdk-1.0.0/tests/unit/test_watchdog_send.py +200 -0
@@ -0,0 +1,7 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.egg-info/
4
+ .mypy_cache/
5
+ .pytest_cache/
6
+ .ruff_cache/
7
+ dist/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Robert Štětka
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,276 @@
1
+ Metadata-Version: 2.5
2
+ Name: didntrun-sdk
3
+ Version: 1.0.0
4
+ Summary: First-party Python client for the didnt.run ping API: correct, non-blocking instrumentation by default.
5
+ Project-URL: Homepage, https://didnt.run
6
+ Project-URL: Source, https://gitlab.com/robert.stetka/didnt-run-sdk-python
7
+ Author: Robert Štětka
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: cron,didnt.run,healthcheck,heartbeat,monitoring,watchdog
11
+ Classifier: Development Status :: 5 - Production/Stable
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: System Administrators
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Topic :: System :: Monitoring
19
+ Classifier: Typing :: Typed
20
+ Requires-Python: >=3.11
21
+ Provides-Extra: contract
22
+ Requires-Dist: psycopg[binary]>=3.2; extra == 'contract'
23
+ Provides-Extra: dev
24
+ Requires-Dist: mypy>=1.13; extra == 'dev'
25
+ Requires-Dist: pytest>=8.3; extra == 'dev'
26
+ Requires-Dist: ruff>=0.8; extra == 'dev'
27
+ Description-Content-Type: text/markdown
28
+
29
+ # didntrun-sdk — Python client for didnt.run
30
+
31
+ Instrument a scheduled job against didnt.run with one line. The SDK's contract: **your job is never
32
+ slowed beyond a hard time budget and never failed by the watchdog** — every transport problem is
33
+ swallowed and logged, by default through Python's `logging` module, or through a logger you supply.
34
+
35
+ ## Install
36
+
37
+ ```bash
38
+ pip install didntrun-sdk
39
+ ```
40
+
41
+ ## Use
42
+
43
+ ```python
44
+ import os
45
+ from didntrun import Watchdog
46
+
47
+ # One env var wires everything: https://{ping-token}@it.didnt.run[?budget_ms=2000&connect_timeout_ms=1000]
48
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
49
+ ```
50
+
51
+ > Your DSN is shown once, ready to copy, when you issue a ping token in the didnt.run admin UI.
52
+
53
+ ```python
54
+ # Recommended: bracket the work. ONE completion ping — measured duration on success,
55
+ # exception context on failure (the exception is re-raised; your job's semantics don't change).
56
+ with wd.run("app.daily-report"):
57
+ generate()
58
+
59
+ # Or the decorator form:
60
+ @wd.watch("app.daily-report")
61
+ def daily_report():
62
+ ...
63
+
64
+ # Or fire the bare "I ran" ping yourself:
65
+ wd.ping("app.daily-report")
66
+ wd.ping("app.daily-report", duration_ms=1234)
67
+
68
+ # Or send explicit lifecycle pings — manual bracketing when start and finish
69
+ # live in different processes or code paths (see "Lifecycle bracketing"):
70
+ wd.start("app.daily-report")
71
+ wd.success("app.daily-report", duration_ms=1234)
72
+ wd.fail("app.daily-report")
73
+ ```
74
+
75
+ `ping()` returns `bool` (delivered vs. swallowed) if you want to observe delivery; you never have to.
76
+
77
+ ### Job identity
78
+
79
+ ```python
80
+ @wd.watch() # id defaults to f"{fn.__module__}.{fn.__qualname__}"
81
+ def daily_report():
82
+ ...
83
+ ```
84
+
85
+ **This is identity-defining.** Moving `daily_report` to another module registers a NEW job — the old
86
+ one goes stale and will alert. Prefer the explicit-id form (`@wd.watch("app.daily-report")`); every
87
+ example above uses it for that reason. Ids are normalized the same way the PHP SDK
88
+ normalizes them — leading `\` stripped, `\` → `.`, empty rejected — and capped at 255 **bytes**, which
89
+ is the length the server measures, dropping a trailing partial UTF-8 character rather than splitting
90
+ one. The mapping is kept for parity rather than necessity: Python identifiers are already dotted, so
91
+ it is a no-op for a native id.
92
+
93
+ ### Failure semantics
94
+
95
+ - Misconfiguration (a malformed DSN, an empty token, a token outside the accept set, a base URL with a
96
+ query, a fragment or an unparseable port, a DSN option that is not a plain positive integer) raises
97
+ `ValueError` **at construction**, and `@wd.watch()` raises `ValueError` **at decoration** when the
98
+ function it wraps is an `async def` or a generator (see "No `async`/`await`"). Both are wiring time,
99
+ before any job runs. After that, nothing raises except the wrapped work's own exception inside
100
+ `run()` / `watch()`, re-raised unchanged.
101
+ - Each ping gets a wall-clock budget (default 2 s, connect 1 s) with one retry inside it — see
102
+ "Delivery and retries" below for exactly which failures qualify. `401` is logged at ERROR (your token
103
+ is wrong); every other rejection at WARNING.
104
+ - Oversized `meta` (> 64 KiB body) is dropped; the ping still goes out.
105
+
106
+ ### Wiring
107
+
108
+ ```python
109
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
110
+ ```
111
+
112
+ Construct once at startup, wherever your app already wires singletons, and hand `wd` around — there's
113
+ no framework glue to install; `Watchdog` is a plain object.
114
+
115
+ Bring your own HTTP client (optional): implement the `Transport` protocol — it's `typing.Protocol`, not
116
+ an ABC, so nothing has to import from this package. The whole surface is one method:
117
+
118
+ ```python
119
+ from collections.abc import Mapping
120
+
121
+ import requests
122
+ from didntrun.transport import Transport, TransportResult, TimeBudget
123
+
124
+ class RequestsTransport(Transport):
125
+ def post(self, url: str, headers: Mapping[str, str], body: bytes, budget: TimeBudget) -> TransportResult:
126
+ try:
127
+ resp = requests.post(
128
+ url, headers=dict(headers), data=body, allow_redirects=False,
129
+ timeout=(budget.connect_ms / 1000, max(budget.total_ms, 1) / 1000),
130
+ )
131
+ return TransportResult.response(resp.status_code)
132
+ except requests.RequestException as e:
133
+ return TransportResult.failure(str(e))
134
+
135
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"], transport=RequestsTransport())
136
+ ```
137
+
138
+ Note the time budget is then only as good as your client's own timeout config — the default
139
+ `HttpClientTransport` (stdlib `http.client`) is deliberately the one thing this package doesn't
140
+ outsource: it never follows redirects and enforces the budget by tracking a deadline on
141
+ `time.monotonic()` rather than relying on a per-socket-operation `timeout=`.
142
+
143
+ ### Lifecycle bracketing
144
+
145
+ `run()` / `watch()` send one completion ping by default. Opt into **bracketing** and they also emit a
146
+ `start` ping *before* the work runs: the `start` anchor is duration-immune, so it tightens the inferred
147
+ cadence for jobs whose duration is non-trivial or variable (a terminal-only anchor drifts with run
148
+ length).
149
+
150
+ Enable it globally (constructor arg, or the `lifecycle` DSN param) with a per-call override:
151
+
152
+ ```python
153
+ # Global default via DSN: https://{token}@it.didnt.run?lifecycle=true
154
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
155
+ with wd.run("app.daily-report"): # start + completion
156
+ generate()
157
+
158
+ # Per-call override wins (None = inherit the instance default):
159
+ with wd.run("app.daily-report", lifecycle=True):
160
+ generate()
161
+ ```
162
+
163
+ A bracketed run has **two** budget windows — one ping before the work, one after — so its worst-case
164
+ added wall-clock is ~2× the per-ping budget, straddling the work. For work a single `with` block can't
165
+ wrap (it spans processes, or start and finish live in different code), call `start()` / `success()` /
166
+ `fail()` directly instead.
167
+
168
+ **Server requirement:** bracketing needs a ping-api that understands lifecycle kinds. Against an older,
169
+ kind-blind self-hosted server, leave `lifecycle` **off** — a separate `start` ping would double-count
170
+ runs in schedule inference.
171
+
172
+ **No `async`/`await`.** There's no async HTTP client in the stdlib, and shipping one would break the
173
+ zero-dependency promise. `@wd.watch()` **refuses** an `async def` or a generator function outright,
174
+ with a `ValueError` at decoration: a synchronous wrapper around a coroutine function gets the
175
+ coroutine object back immediately, so the completion ping would go out before the work had executed a
176
+ line — reporting success for a job that has not run yet, and feeding a 0 ms duration into the
177
+ hung-detection baseline. Await the work inside a sync wrapper, or keep the ping off the loop thread:
178
+
179
+ ```python
180
+ await asyncio.to_thread(wd.ping, "app.daily-report")
181
+ ```
182
+
183
+ ## Logging
184
+
185
+ Every failure is swallowed and logged. By default the SDK logs through the standard `logging` module —
186
+ `logging.getLogger("didntrun")` — with **no handler attached**, so with nothing else configured in your
187
+ process, Python's own `logging.lastResort` emits WARNING and above to **stderr**. Under a plain cron
188
+ invocation that's exactly what you want: your cron runner captures or mails it.
189
+
190
+ Attach your own handler to redirect it, the same as any other library logger:
191
+
192
+ ```python
193
+ import logging
194
+
195
+ logging.getLogger("didntrun").addHandler(logging.FileHandler("/var/log/didntrun.log"))
196
+ ```
197
+
198
+ To silence it completely:
199
+
200
+ ```python
201
+ logging.getLogger("didntrun").addHandler(logging.NullHandler())
202
+ ```
203
+
204
+ A ping that is never delivered is the one failure you cannot see from the server side — it looks
205
+ identical to a job that died before pinging. Leaving logging on is what tells the two apart.
206
+
207
+ Each failure is written as exactly one line, so a log is safe to grep and count — that's also how the
208
+ PHP SDK's deferred time-budget question gets settled for this one: count `ping rejected` against
209
+ `ping not delivered`. Control characters in a value you passed — a newline or a tab inside an
210
+ `external_id`, say — appear as `?` rather than breaking the line in two, and the same context is also
211
+ passed as `extra=` for handlers that render structured fields. Your ping token never appears in a log
212
+ line either, including inside an error string authored by a transport.
213
+
214
+ **stderr records carry no timestamp** — `logging.lastResort` uses a bare handler with no `Formatter`, so
215
+ a message on stderr says only what happened, not when. Counting failures over a week works either way,
216
+ but correlating a specific lost ping with a specific alert needs a timestamp — either attach a real
217
+ handler (above) or have your cron wrapper stamp the stream.
218
+
219
+ ## Delivery and retries
220
+
221
+ A ping is retried **once** when the failure could plausibly succeed on a second attempt:
222
+
223
+ | Outcome | Retried? |
224
+ | --- | --- |
225
+ | Connection refused / DNS failure / timeout | yes |
226
+ | `408 Request Timeout`, any `5xx` | yes |
227
+ | `429 Too Many Requests` | no |
228
+ | Any other `4xx` (`400`, `401`, `404`, …) | no |
229
+
230
+ One retry in total, never one per category. The retry has to fit inside the same time budget as the
231
+ first attempt (2 s by default), so against a slow server there may be no room for it — the SDK will
232
+ never exceed the budget to get a ping through.
233
+
234
+ ## CLI
235
+
236
+ Installing the package also installs `didnt-run` — a console script, no separate install step:
237
+
238
+ ```
239
+ didnt-run ping <id> [--duration-ms N] [--meta k=v]...
240
+ didnt-run run <id> [--lifecycle] [--capture-stderr] -- <cmd> [args...]
241
+ ```
242
+
243
+ Reads `DIDNT_RUN_DSN` from the environment, or `--dsn`. `run` wraps a shell command in your crontab:
244
+
245
+ ```cron
246
+ */5 * * * * didnt-run run app.nightly-backup -- pg_dump -Fc mydb > /backups/mydb.dump
247
+ ```
248
+
249
+ - **It exits with the wrapped command's exit code.** The crontab's semantics, and any `&&` chained
250
+ after it, are unchanged.
251
+ - **stdout and stderr pass straight through**, so cron mail still works. `--capture-stderr` *tees*
252
+ rather than swallows, keeping the last 512 bytes for `meta` — it defaults to **off**, because a
253
+ command's stderr is a population you didn't write and can't predict (`curl`, `pg_dump`, `rsync` all
254
+ print connection strings on failure on their own); a denylist of secret-looking patterns goes stale
255
+ the day some tool prints a new one, so the safe default is off and the opt-in is explicit.
256
+ - **A watchdog failure never changes the exit code.** A missing or malformed `DIDNT_RUN_DSN` logs to
257
+ stderr and the command still runs, uninstrumented — an already-registered job that stops pinging is
258
+ exactly what the watchdog alerts on. A command that can't be executed at all follows the
259
+ shell's own convention: **127** when it is not found, **126** when it is found but not executable.
260
+ The same code goes into the `fail` ping's `exit_code` meta and out as the wrapper's exit status.
261
+
262
+ A malformed `--meta` on `run` is reported on stderr and dropped — the command still runs. It was the
263
+ only watchdog-side input on that path that could stop a backup, and a metadata typo is the least
264
+ important of them. `didnt-run ping` has nothing to protect, so misconfiguration there exits `2`.
265
+ `--meta k=v` values are always strings; no type coercion is attempted.
266
+
267
+ ## Requirements
268
+
269
+ Python ≥ 3.11. Zero runtime dependencies — the SDK is stdlib only.
270
+
271
+ ## License
272
+
273
+ MIT — see [LICENSE](LICENSE).
274
+
275
+ Development happens in the didnt.run monorepo; this repository is a read-only subtree split. Please
276
+ report issues against the SDK there rather than opening merge requests here.
@@ -0,0 +1,248 @@
1
+ # didntrun-sdk — Python client for didnt.run
2
+
3
+ Instrument a scheduled job against didnt.run with one line. The SDK's contract: **your job is never
4
+ slowed beyond a hard time budget and never failed by the watchdog** — every transport problem is
5
+ swallowed and logged, by default through Python's `logging` module, or through a logger you supply.
6
+
7
+ ## Install
8
+
9
+ ```bash
10
+ pip install didntrun-sdk
11
+ ```
12
+
13
+ ## Use
14
+
15
+ ```python
16
+ import os
17
+ from didntrun import Watchdog
18
+
19
+ # One env var wires everything: https://{ping-token}@it.didnt.run[?budget_ms=2000&connect_timeout_ms=1000]
20
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
21
+ ```
22
+
23
+ > Your DSN is shown once, ready to copy, when you issue a ping token in the didnt.run admin UI.
24
+
25
+ ```python
26
+ # Recommended: bracket the work. ONE completion ping — measured duration on success,
27
+ # exception context on failure (the exception is re-raised; your job's semantics don't change).
28
+ with wd.run("app.daily-report"):
29
+ generate()
30
+
31
+ # Or the decorator form:
32
+ @wd.watch("app.daily-report")
33
+ def daily_report():
34
+ ...
35
+
36
+ # Or fire the bare "I ran" ping yourself:
37
+ wd.ping("app.daily-report")
38
+ wd.ping("app.daily-report", duration_ms=1234)
39
+
40
+ # Or send explicit lifecycle pings — manual bracketing when start and finish
41
+ # live in different processes or code paths (see "Lifecycle bracketing"):
42
+ wd.start("app.daily-report")
43
+ wd.success("app.daily-report", duration_ms=1234)
44
+ wd.fail("app.daily-report")
45
+ ```
46
+
47
+ `ping()` returns `bool` (delivered vs. swallowed) if you want to observe delivery; you never have to.
48
+
49
+ ### Job identity
50
+
51
+ ```python
52
+ @wd.watch() # id defaults to f"{fn.__module__}.{fn.__qualname__}"
53
+ def daily_report():
54
+ ...
55
+ ```
56
+
57
+ **This is identity-defining.** Moving `daily_report` to another module registers a NEW job — the old
58
+ one goes stale and will alert. Prefer the explicit-id form (`@wd.watch("app.daily-report")`); every
59
+ example above uses it for that reason. Ids are normalized the same way the PHP SDK
60
+ normalizes them — leading `\` stripped, `\` → `.`, empty rejected — and capped at 255 **bytes**, which
61
+ is the length the server measures, dropping a trailing partial UTF-8 character rather than splitting
62
+ one. The mapping is kept for parity rather than necessity: Python identifiers are already dotted, so
63
+ it is a no-op for a native id.
64
+
65
+ ### Failure semantics
66
+
67
+ - Misconfiguration (a malformed DSN, an empty token, a token outside the accept set, a base URL with a
68
+ query, a fragment or an unparseable port, a DSN option that is not a plain positive integer) raises
69
+ `ValueError` **at construction**, and `@wd.watch()` raises `ValueError` **at decoration** when the
70
+ function it wraps is an `async def` or a generator (see "No `async`/`await`"). Both are wiring time,
71
+ before any job runs. After that, nothing raises except the wrapped work's own exception inside
72
+ `run()` / `watch()`, re-raised unchanged.
73
+ - Each ping gets a wall-clock budget (default 2 s, connect 1 s) with one retry inside it — see
74
+ "Delivery and retries" below for exactly which failures qualify. `401` is logged at ERROR (your token
75
+ is wrong); every other rejection at WARNING.
76
+ - Oversized `meta` (> 64 KiB body) is dropped; the ping still goes out.
77
+
78
+ ### Wiring
79
+
80
+ ```python
81
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
82
+ ```
83
+
84
+ Construct once at startup, wherever your app already wires singletons, and hand `wd` around — there's
85
+ no framework glue to install; `Watchdog` is a plain object.
86
+
87
+ Bring your own HTTP client (optional): implement the `Transport` protocol — it's `typing.Protocol`, not
88
+ an ABC, so nothing has to import from this package. The whole surface is one method:
89
+
90
+ ```python
91
+ from collections.abc import Mapping
92
+
93
+ import requests
94
+ from didntrun.transport import Transport, TransportResult, TimeBudget
95
+
96
+ class RequestsTransport(Transport):
97
+ def post(self, url: str, headers: Mapping[str, str], body: bytes, budget: TimeBudget) -> TransportResult:
98
+ try:
99
+ resp = requests.post(
100
+ url, headers=dict(headers), data=body, allow_redirects=False,
101
+ timeout=(budget.connect_ms / 1000, max(budget.total_ms, 1) / 1000),
102
+ )
103
+ return TransportResult.response(resp.status_code)
104
+ except requests.RequestException as e:
105
+ return TransportResult.failure(str(e))
106
+
107
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"], transport=RequestsTransport())
108
+ ```
109
+
110
+ Note the time budget is then only as good as your client's own timeout config — the default
111
+ `HttpClientTransport` (stdlib `http.client`) is deliberately the one thing this package doesn't
112
+ outsource: it never follows redirects and enforces the budget by tracking a deadline on
113
+ `time.monotonic()` rather than relying on a per-socket-operation `timeout=`.
114
+
115
+ ### Lifecycle bracketing
116
+
117
+ `run()` / `watch()` send one completion ping by default. Opt into **bracketing** and they also emit a
118
+ `start` ping *before* the work runs: the `start` anchor is duration-immune, so it tightens the inferred
119
+ cadence for jobs whose duration is non-trivial or variable (a terminal-only anchor drifts with run
120
+ length).
121
+
122
+ Enable it globally (constructor arg, or the `lifecycle` DSN param) with a per-call override:
123
+
124
+ ```python
125
+ # Global default via DSN: https://{token}@it.didnt.run?lifecycle=true
126
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
127
+ with wd.run("app.daily-report"): # start + completion
128
+ generate()
129
+
130
+ # Per-call override wins (None = inherit the instance default):
131
+ with wd.run("app.daily-report", lifecycle=True):
132
+ generate()
133
+ ```
134
+
135
+ A bracketed run has **two** budget windows — one ping before the work, one after — so its worst-case
136
+ added wall-clock is ~2× the per-ping budget, straddling the work. For work a single `with` block can't
137
+ wrap (it spans processes, or start and finish live in different code), call `start()` / `success()` /
138
+ `fail()` directly instead.
139
+
140
+ **Server requirement:** bracketing needs a ping-api that understands lifecycle kinds. Against an older,
141
+ kind-blind self-hosted server, leave `lifecycle` **off** — a separate `start` ping would double-count
142
+ runs in schedule inference.
143
+
144
+ **No `async`/`await`.** There's no async HTTP client in the stdlib, and shipping one would break the
145
+ zero-dependency promise. `@wd.watch()` **refuses** an `async def` or a generator function outright,
146
+ with a `ValueError` at decoration: a synchronous wrapper around a coroutine function gets the
147
+ coroutine object back immediately, so the completion ping would go out before the work had executed a
148
+ line — reporting success for a job that has not run yet, and feeding a 0 ms duration into the
149
+ hung-detection baseline. Await the work inside a sync wrapper, or keep the ping off the loop thread:
150
+
151
+ ```python
152
+ await asyncio.to_thread(wd.ping, "app.daily-report")
153
+ ```
154
+
155
+ ## Logging
156
+
157
+ Every failure is swallowed and logged. By default the SDK logs through the standard `logging` module —
158
+ `logging.getLogger("didntrun")` — with **no handler attached**, so with nothing else configured in your
159
+ process, Python's own `logging.lastResort` emits WARNING and above to **stderr**. Under a plain cron
160
+ invocation that's exactly what you want: your cron runner captures or mails it.
161
+
162
+ Attach your own handler to redirect it, the same as any other library logger:
163
+
164
+ ```python
165
+ import logging
166
+
167
+ logging.getLogger("didntrun").addHandler(logging.FileHandler("/var/log/didntrun.log"))
168
+ ```
169
+
170
+ To silence it completely:
171
+
172
+ ```python
173
+ logging.getLogger("didntrun").addHandler(logging.NullHandler())
174
+ ```
175
+
176
+ A ping that is never delivered is the one failure you cannot see from the server side — it looks
177
+ identical to a job that died before pinging. Leaving logging on is what tells the two apart.
178
+
179
+ Each failure is written as exactly one line, so a log is safe to grep and count — that's also how the
180
+ PHP SDK's deferred time-budget question gets settled for this one: count `ping rejected` against
181
+ `ping not delivered`. Control characters in a value you passed — a newline or a tab inside an
182
+ `external_id`, say — appear as `?` rather than breaking the line in two, and the same context is also
183
+ passed as `extra=` for handlers that render structured fields. Your ping token never appears in a log
184
+ line either, including inside an error string authored by a transport.
185
+
186
+ **stderr records carry no timestamp** — `logging.lastResort` uses a bare handler with no `Formatter`, so
187
+ a message on stderr says only what happened, not when. Counting failures over a week works either way,
188
+ but correlating a specific lost ping with a specific alert needs a timestamp — either attach a real
189
+ handler (above) or have your cron wrapper stamp the stream.
190
+
191
+ ## Delivery and retries
192
+
193
+ A ping is retried **once** when the failure could plausibly succeed on a second attempt:
194
+
195
+ | Outcome | Retried? |
196
+ | --- | --- |
197
+ | Connection refused / DNS failure / timeout | yes |
198
+ | `408 Request Timeout`, any `5xx` | yes |
199
+ | `429 Too Many Requests` | no |
200
+ | Any other `4xx` (`400`, `401`, `404`, …) | no |
201
+
202
+ One retry in total, never one per category. The retry has to fit inside the same time budget as the
203
+ first attempt (2 s by default), so against a slow server there may be no room for it — the SDK will
204
+ never exceed the budget to get a ping through.
205
+
206
+ ## CLI
207
+
208
+ Installing the package also installs `didnt-run` — a console script, no separate install step:
209
+
210
+ ```
211
+ didnt-run ping <id> [--duration-ms N] [--meta k=v]...
212
+ didnt-run run <id> [--lifecycle] [--capture-stderr] -- <cmd> [args...]
213
+ ```
214
+
215
+ Reads `DIDNT_RUN_DSN` from the environment, or `--dsn`. `run` wraps a shell command in your crontab:
216
+
217
+ ```cron
218
+ */5 * * * * didnt-run run app.nightly-backup -- pg_dump -Fc mydb > /backups/mydb.dump
219
+ ```
220
+
221
+ - **It exits with the wrapped command's exit code.** The crontab's semantics, and any `&&` chained
222
+ after it, are unchanged.
223
+ - **stdout and stderr pass straight through**, so cron mail still works. `--capture-stderr` *tees*
224
+ rather than swallows, keeping the last 512 bytes for `meta` — it defaults to **off**, because a
225
+ command's stderr is a population you didn't write and can't predict (`curl`, `pg_dump`, `rsync` all
226
+ print connection strings on failure on their own); a denylist of secret-looking patterns goes stale
227
+ the day some tool prints a new one, so the safe default is off and the opt-in is explicit.
228
+ - **A watchdog failure never changes the exit code.** A missing or malformed `DIDNT_RUN_DSN` logs to
229
+ stderr and the command still runs, uninstrumented — an already-registered job that stops pinging is
230
+ exactly what the watchdog alerts on. A command that can't be executed at all follows the
231
+ shell's own convention: **127** when it is not found, **126** when it is found but not executable.
232
+ The same code goes into the `fail` ping's `exit_code` meta and out as the wrapper's exit status.
233
+
234
+ A malformed `--meta` on `run` is reported on stderr and dropped — the command still runs. It was the
235
+ only watchdog-side input on that path that could stop a backup, and a metadata typo is the least
236
+ important of them. `didnt-run ping` has nothing to protect, so misconfiguration there exits `2`.
237
+ `--meta k=v` values are always strings; no type coercion is attempted.
238
+
239
+ ## Requirements
240
+
241
+ Python ≥ 3.11. Zero runtime dependencies — the SDK is stdlib only.
242
+
243
+ ## License
244
+
245
+ MIT — see [LICENSE](LICENSE).
246
+
247
+ Development happens in the didnt.run monorepo; this repository is a read-only subtree split. Please
248
+ report issues against the SDK there rather than opening merge requests here.
@@ -0,0 +1,104 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "didntrun-sdk"
7
+ version = "1.0.0"
8
+ description = "First-party Python client for the didnt.run ping API: correct, non-blocking instrumentation by default."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Robert Štětka" }]
14
+ keywords = ["cron", "monitoring", "watchdog", "heartbeat", "healthcheck", "didnt.run"]
15
+ classifiers = [
16
+ "Development Status :: 5 - Production/Stable",
17
+ "Intended Audience :: Developers",
18
+ "Intended Audience :: System Administrators",
19
+ "License :: OSI Approved :: MIT License",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Programming Language :: Python :: 3.13",
23
+ "Topic :: System :: Monitoring",
24
+ "Typing :: Typed",
25
+ ]
26
+ dependencies = []
27
+
28
+ [project.urls]
29
+ Homepage = "https://didnt.run"
30
+ Source = "https://gitlab.com/robert.stetka/didnt-run-sdk-python"
31
+
32
+ [project.scripts]
33
+ didnt-run = "didntrun.cli:main"
34
+
35
+ [project.optional-dependencies]
36
+ dev = ["pytest>=8.3", "mypy>=1.13", "ruff>=0.8"]
37
+ contract = ["psycopg[binary]>=3.2"]
38
+
39
+ [tool.hatch.build.targets.wheel]
40
+ packages = ["src/didntrun"]
41
+
42
+ [tool.hatch.build.targets.sdist]
43
+ include = ["src", "tests", "README.md", "LICENSE", "pyproject.toml"]
44
+
45
+ [tool.ruff]
46
+ # Ruff 0.16 formats Python code blocks inside Markdown. A README's snippets are illustrative and
47
+ # hand-laid-out for reading — collapsing `def f():\n ...` onto one line and exploding a call
48
+ # across five is worse documentation, and ECS does not format the PHP SDK's README either.
49
+ extend-exclude = ["*.md"]
50
+ line-length = 110
51
+ target-version = "py311"
52
+ src = ["src", "tests"]
53
+
54
+ [tool.ruff.lint]
55
+ select = ["E", "F", "W", "I", "N", "UP", "B", "C4", "SIM", "RUF", "ANN", "PT", "S", "PTH"]
56
+ ignore = [
57
+ # `meta` is caller-supplied JSON; Mapping[str, Any] is the honest type, not a weakness.
58
+ "ANN401",
59
+ ]
60
+
61
+ [tool.ruff.lint.per-file-ignores]
62
+ # S603 flags every subprocess call. Running the wrapped command IS this module's job, and it
63
+ # deliberately runs it WITHOUT a shell — which is the thing S603 exists to encourage.
64
+ "src/didntrun/cli.py" = ["S603"]
65
+ # Tests assert, and pytest does not want return annotations on every test function. S105/S106 flag
66
+ # every literal named like a credential, and a ping token fixture IS one — deliberately fake, and
67
+ # the thing under test in the construction suite.
68
+ "tests/**" = ["S101", "S105", "S106", "ANN"]
69
+ # The four process-level CLI tests exist to observe what a REAL wrapper process does to its own
70
+ # stderr, so spawning one is the method, not an oversight. S607's `sh` is deliberate too: `2>&-` is
71
+ # how fd 2 gets closed, and a shell found on PATH is what a crontab line would use.
72
+ "tests/unit/test_cli_process.py" = ["S101", "S105", "S106", "ANN", "S603", "S607"]
73
+ # The last two contract tests call Watchdog._send on purpose: they assert on the SERVER's limits,
74
+ # and going through the public ping() would hit the SDK's own guard first and never reach the wire.
75
+ "tests/contract/test_ping_contract.py" = ["S101", "ANN", "SLF001"]
76
+
77
+ [tool.ruff.format]
78
+ docstring-code-format = true
79
+
80
+ [tool.mypy]
81
+ python_version = "3.11"
82
+ strict = true
83
+ files = ["src", "tests"]
84
+ warn_unreachable = true
85
+
86
+ # Tests are type-checked, but not required to annotate every test function: pytest calls them, not
87
+ # us, so a missing `-> None` buys nothing. Everything else strict still applies — a test that
88
+ # misuses a public type is still an error, which is the point of checking them at all.
89
+ [[tool.mypy.overrides]]
90
+ module = "tests.*"
91
+ disallow_untyped_defs = false
92
+ disallow_incomplete_defs = false
93
+ # Follows from the two above rather than adding anything: unannotated helpers are allowed here, so
94
+ # flagging every call to one would make the exemption self-cancelling. mypy 2.x reports those calls
95
+ # from inside an unannotated test body as well, where 1.x did not.
96
+ disallow_untyped_calls = false
97
+
98
+ [tool.pytest.ini_options]
99
+ testpaths = ["tests/unit"]
100
+ addopts = "-q"
101
+ # A DeprecationWarning from the stdlib is a bug report against this package's 3.11 floor.
102
+ # ResourceWarning is exempt: it fires at GC time, so with sockets in play it would make the suite
103
+ # fail on collection timing rather than on anything a reader could act on.
104
+ filterwarnings = ["error", "default::ResourceWarning"]