didntrun-sdk 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- didntrun_sdk-1.0.0/.gitignore +7 -0
- didntrun_sdk-1.0.0/LICENSE +21 -0
- didntrun_sdk-1.0.0/PKG-INFO +276 -0
- didntrun_sdk-1.0.0/README.md +248 -0
- didntrun_sdk-1.0.0/pyproject.toml +104 -0
- didntrun_sdk-1.0.0/src/didntrun/__init__.py +25 -0
- didntrun_sdk-1.0.0/src/didntrun/cli.py +351 -0
- didntrun_sdk-1.0.0/src/didntrun/external_id.py +33 -0
- didntrun_sdk-1.0.0/src/didntrun/log_text.py +15 -0
- didntrun_sdk-1.0.0/src/didntrun/ping_kind.py +12 -0
- didntrun_sdk-1.0.0/src/didntrun/py.typed +0 -0
- didntrun_sdk-1.0.0/src/didntrun/transport/__init__.py +4 -0
- didntrun_sdk-1.0.0/src/didntrun/transport/base.py +84 -0
- didntrun_sdk-1.0.0/src/didntrun/transport/http_client.py +163 -0
- didntrun_sdk-1.0.0/src/didntrun/watchdog.py +508 -0
- didntrun_sdk-1.0.0/tests/__init__.py +0 -0
- didntrun_sdk-1.0.0/tests/contract/__init__.py +0 -0
- didntrun_sdk-1.0.0/tests/contract/conftest.py +75 -0
- didntrun_sdk-1.0.0/tests/contract/test_ping_contract.py +107 -0
- didntrun_sdk-1.0.0/tests/unit/__init__.py +0 -0
- didntrun_sdk-1.0.0/tests/unit/conftest.py +224 -0
- didntrun_sdk-1.0.0/tests/unit/test_cli_ping.py +104 -0
- didntrun_sdk-1.0.0/tests/unit/test_cli_process.py +165 -0
- didntrun_sdk-1.0.0/tests/unit/test_cli_run.py +197 -0
- didntrun_sdk-1.0.0/tests/unit/test_conformance_coverage.py +56 -0
- didntrun_sdk-1.0.0/tests/unit/test_external_id.py +34 -0
- didntrun_sdk-1.0.0/tests/unit/test_http_client_transport.py +199 -0
- didntrun_sdk-1.0.0/tests/unit/test_log_text.py +17 -0
- didntrun_sdk-1.0.0/tests/unit/test_package.py +12 -0
- didntrun_sdk-1.0.0/tests/unit/test_ping_kind.py +27 -0
- didntrun_sdk-1.0.0/tests/unit/test_public_api.py +28 -0
- didntrun_sdk-1.0.0/tests/unit/test_transport_base.py +42 -0
- didntrun_sdk-1.0.0/tests/unit/test_watchdog_construction.py +90 -0
- didntrun_sdk-1.0.0/tests/unit/test_watchdog_dsn.py +117 -0
- didntrun_sdk-1.0.0/tests/unit/test_watchdog_logging.py +158 -0
- didntrun_sdk-1.0.0/tests/unit/test_watchdog_retry.py +196 -0
- didntrun_sdk-1.0.0/tests/unit/test_watchdog_run.py +249 -0
- didntrun_sdk-1.0.0/tests/unit/test_watchdog_send.py +200 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Robert Štětka
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: didntrun-sdk
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: First-party Python client for the didnt.run ping API: correct, non-blocking instrumentation by default.
|
|
5
|
+
Project-URL: Homepage, https://didnt.run
|
|
6
|
+
Project-URL: Source, https://gitlab.com/robert.stetka/didnt-run-sdk-python
|
|
7
|
+
Author: Robert Štětka
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: cron,didnt.run,healthcheck,heartbeat,monitoring,watchdog
|
|
11
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: System Administrators
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Topic :: System :: Monitoring
|
|
19
|
+
Classifier: Typing :: Typed
|
|
20
|
+
Requires-Python: >=3.11
|
|
21
|
+
Provides-Extra: contract
|
|
22
|
+
Requires-Dist: psycopg[binary]>=3.2; extra == 'contract'
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: mypy>=1.13; extra == 'dev'
|
|
25
|
+
Requires-Dist: pytest>=8.3; extra == 'dev'
|
|
26
|
+
Requires-Dist: ruff>=0.8; extra == 'dev'
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# didntrun-sdk — Python client for didnt.run
|
|
30
|
+
|
|
31
|
+
Instrument a scheduled job against didnt.run with one line. The SDK's contract: **your job is never
|
|
32
|
+
slowed beyond a hard time budget and never failed by the watchdog** — every transport problem is
|
|
33
|
+
swallowed and logged, by default through Python's `logging` module, or through a logger you supply.
|
|
34
|
+
|
|
35
|
+
## Install
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install didntrun-sdk
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Use
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
import os
|
|
45
|
+
from didntrun import Watchdog
|
|
46
|
+
|
|
47
|
+
# One env var wires everything: https://{ping-token}@it.didnt.run[?budget_ms=2000&connect_timeout_ms=1000]
|
|
48
|
+
wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
> Your DSN is shown once, ready to copy, when you issue a ping token in the didnt.run admin UI.
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
# Recommended: bracket the work. ONE completion ping — measured duration on success,
|
|
55
|
+
# exception context on failure (the exception is re-raised; your job's semantics don't change).
|
|
56
|
+
with wd.run("app.daily-report"):
|
|
57
|
+
generate()
|
|
58
|
+
|
|
59
|
+
# Or the decorator form:
|
|
60
|
+
@wd.watch("app.daily-report")
|
|
61
|
+
def daily_report():
|
|
62
|
+
...
|
|
63
|
+
|
|
64
|
+
# Or fire the bare "I ran" ping yourself:
|
|
65
|
+
wd.ping("app.daily-report")
|
|
66
|
+
wd.ping("app.daily-report", duration_ms=1234)
|
|
67
|
+
|
|
68
|
+
# Or send explicit lifecycle pings — manual bracketing when start and finish
|
|
69
|
+
# live in different processes or code paths (see "Lifecycle bracketing"):
|
|
70
|
+
wd.start("app.daily-report")
|
|
71
|
+
wd.success("app.daily-report", duration_ms=1234)
|
|
72
|
+
wd.fail("app.daily-report")
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
`ping()` returns `bool` (delivered vs. swallowed) if you want to observe delivery; you never have to.
|
|
76
|
+
|
|
77
|
+
### Job identity
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
@wd.watch() # id defaults to f"{fn.__module__}.{fn.__qualname__}"
|
|
81
|
+
def daily_report():
|
|
82
|
+
...
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
**This is identity-defining.** Moving `daily_report` to another module registers a NEW job — the old
|
|
86
|
+
one goes stale and will alert. Prefer the explicit-id form (`@wd.watch("app.daily-report")`); every
|
|
87
|
+
example above uses it for that reason. Ids are normalized the same way the PHP SDK
|
|
88
|
+
normalizes them — leading `\` stripped, `\` → `.`, empty rejected — and capped at 255 **bytes**, which
|
|
89
|
+
is the length the server measures, dropping a trailing partial UTF-8 character rather than splitting
|
|
90
|
+
one. The mapping is kept for parity rather than necessity: Python identifiers are already dotted, so
|
|
91
|
+
it is a no-op for a native id.
|
|
92
|
+
|
|
93
|
+
### Failure semantics
|
|
94
|
+
|
|
95
|
+
- Misconfiguration (a malformed DSN, an empty token, a token outside the accept set, a base URL with a
|
|
96
|
+
query, a fragment or an unparseable port, a DSN option that is not a plain positive integer) raises
|
|
97
|
+
`ValueError` **at construction**, and `@wd.watch()` raises `ValueError` **at decoration** when the
|
|
98
|
+
function it wraps is an `async def` or a generator (see "No `async`/`await`"). Both are wiring time,
|
|
99
|
+
before any job runs. After that, nothing raises except the wrapped work's own exception inside
|
|
100
|
+
`run()` / `watch()`, re-raised unchanged.
|
|
101
|
+
- Each ping gets a wall-clock budget (default 2 s, connect 1 s) with one retry inside it — see
|
|
102
|
+
"Delivery and retries" below for exactly which failures qualify. `401` is logged at ERROR (your token
|
|
103
|
+
is wrong); every other rejection at WARNING.
|
|
104
|
+
- Oversized `meta` (> 64 KiB body) is dropped; the ping still goes out.
|
|
105
|
+
|
|
106
|
+
### Wiring
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Construct once at startup, wherever your app already wires singletons, and hand `wd` around — there's
|
|
113
|
+
no framework glue to install; `Watchdog` is a plain object.
|
|
114
|
+
|
|
115
|
+
Bring your own HTTP client (optional): implement the `Transport` protocol — it's `typing.Protocol`, not
|
|
116
|
+
an ABC, so nothing has to import from this package. The whole surface is one method:
|
|
117
|
+
|
|
118
|
+
```python
|
|
119
|
+
from collections.abc import Mapping
|
|
120
|
+
|
|
121
|
+
import requests
|
|
122
|
+
from didntrun.transport import Transport, TransportResult, TimeBudget
|
|
123
|
+
|
|
124
|
+
class RequestsTransport(Transport):
|
|
125
|
+
def post(self, url: str, headers: Mapping[str, str], body: bytes, budget: TimeBudget) -> TransportResult:
|
|
126
|
+
try:
|
|
127
|
+
resp = requests.post(
|
|
128
|
+
url, headers=dict(headers), data=body, allow_redirects=False,
|
|
129
|
+
timeout=(budget.connect_ms / 1000, max(budget.total_ms, 1) / 1000),
|
|
130
|
+
)
|
|
131
|
+
return TransportResult.response(resp.status_code)
|
|
132
|
+
except requests.RequestException as e:
|
|
133
|
+
return TransportResult.failure(str(e))
|
|
134
|
+
|
|
135
|
+
wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"], transport=RequestsTransport())
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Note the time budget is then only as good as your client's own timeout config — the default
|
|
139
|
+
`HttpClientTransport` (stdlib `http.client`) is deliberately the one thing this package doesn't
|
|
140
|
+
outsource: it never follows redirects and enforces the budget by tracking a deadline on
|
|
141
|
+
`time.monotonic()` rather than relying on a per-socket-operation `timeout=`.
|
|
142
|
+
|
|
143
|
+
### Lifecycle bracketing
|
|
144
|
+
|
|
145
|
+
`run()` / `watch()` send one completion ping by default. Opt into **bracketing** and they also emit a
|
|
146
|
+
`start` ping *before* the work runs: the `start` anchor is duration-immune, so it tightens the inferred
|
|
147
|
+
cadence for jobs whose duration is non-trivial or variable (a terminal-only anchor drifts with run
|
|
148
|
+
length).
|
|
149
|
+
|
|
150
|
+
Enable it globally (constructor arg, or the `lifecycle` DSN param) with a per-call override:
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
# Global default via DSN: https://{token}@it.didnt.run?lifecycle=true
|
|
154
|
+
wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
|
|
155
|
+
with wd.run("app.daily-report"): # start + completion
|
|
156
|
+
generate()
|
|
157
|
+
|
|
158
|
+
# Per-call override wins (None = inherit the instance default):
|
|
159
|
+
with wd.run("app.daily-report", lifecycle=True):
|
|
160
|
+
generate()
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
A bracketed run has **two** budget windows — one ping before the work, one after — so its worst-case
|
|
164
|
+
added wall-clock is ~2× the per-ping budget, straddling the work. For work a single `with` block can't
|
|
165
|
+
wrap (it spans processes, or start and finish live in different code), call `start()` / `success()` /
|
|
166
|
+
`fail()` directly instead.
|
|
167
|
+
|
|
168
|
+
**Server requirement:** bracketing needs a ping-api that understands lifecycle kinds. Against an older,
|
|
169
|
+
kind-blind self-hosted server, leave `lifecycle` **off** — a separate `start` ping would double-count
|
|
170
|
+
runs in schedule inference.
|
|
171
|
+
|
|
172
|
+
**No `async`/`await`.** There's no async HTTP client in the stdlib, and shipping one would break the
|
|
173
|
+
zero-dependency promise. `@wd.watch()` **refuses** an `async def` or a generator function outright,
|
|
174
|
+
with a `ValueError` at decoration: a synchronous wrapper around a coroutine function gets the
|
|
175
|
+
coroutine object back immediately, so the completion ping would go out before the work had executed a
|
|
176
|
+
line — reporting success for a job that has not run yet, and feeding a 0 ms duration into the
|
|
177
|
+
hung-detection baseline. Await the work inside a sync wrapper, or keep the ping off the loop thread:
|
|
178
|
+
|
|
179
|
+
```python
|
|
180
|
+
await asyncio.to_thread(wd.ping, "app.daily-report")
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
## Logging
|
|
184
|
+
|
|
185
|
+
Every failure is swallowed and logged. By default the SDK logs through the standard `logging` module —
|
|
186
|
+
`logging.getLogger("didntrun")` — with **no handler attached**, so with nothing else configured in your
|
|
187
|
+
process, Python's own `logging.lastResort` emits WARNING and above to **stderr**. Under a plain cron
|
|
188
|
+
invocation that's exactly what you want: your cron runner captures or mails it.
|
|
189
|
+
|
|
190
|
+
Attach your own handler to redirect it, the same as any other library logger:
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
import logging
|
|
194
|
+
|
|
195
|
+
logging.getLogger("didntrun").addHandler(logging.FileHandler("/var/log/didntrun.log"))
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
To silence it completely:
|
|
199
|
+
|
|
200
|
+
```python
|
|
201
|
+
logging.getLogger("didntrun").addHandler(logging.NullHandler())
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
A ping that is never delivered is the one failure you cannot see from the server side — it looks
|
|
205
|
+
identical to a job that died before pinging. Leaving logging on is what tells the two apart.
|
|
206
|
+
|
|
207
|
+
Each failure is written as exactly one line, so a log is safe to grep and count — that's also how the
|
|
208
|
+
PHP SDK's deferred time-budget question gets settled for this one: count `ping rejected` against
|
|
209
|
+
`ping not delivered`. Control characters in a value you passed — a newline or a tab inside an
|
|
210
|
+
`external_id`, say — appear as `?` rather than breaking the line in two, and the same context is also
|
|
211
|
+
passed as `extra=` for handlers that render structured fields. Your ping token never appears in a log
|
|
212
|
+
line either, including inside an error string authored by a transport.
|
|
213
|
+
|
|
214
|
+
**stderr records carry no timestamp** — `logging.lastResort` uses a bare handler with no `Formatter`, so
|
|
215
|
+
a message on stderr says only what happened, not when. Counting failures over a week works either way,
|
|
216
|
+
but correlating a specific lost ping with a specific alert needs a timestamp — either attach a real
|
|
217
|
+
handler (above) or have your cron wrapper stamp the stream.
|
|
218
|
+
|
|
219
|
+
## Delivery and retries
|
|
220
|
+
|
|
221
|
+
A ping is retried **once** when the failure could plausibly succeed on a second attempt:
|
|
222
|
+
|
|
223
|
+
| Outcome | Retried? |
|
|
224
|
+
| --- | --- |
|
|
225
|
+
| Connection refused / DNS failure / timeout | yes |
|
|
226
|
+
| `408 Request Timeout`, any `5xx` | yes |
|
|
227
|
+
| `429 Too Many Requests` | no |
|
|
228
|
+
| Any other `4xx` (`400`, `401`, `404`, …) | no |
|
|
229
|
+
|
|
230
|
+
One retry in total, never one per category. The retry has to fit inside the same time budget as the
|
|
231
|
+
first attempt (2 s by default), so against a slow server there may be no room for it — the SDK will
|
|
232
|
+
never exceed the budget to get a ping through.
|
|
233
|
+
|
|
234
|
+
## CLI
|
|
235
|
+
|
|
236
|
+
Installing the package also installs `didnt-run` — a console script, no separate install step:
|
|
237
|
+
|
|
238
|
+
```
|
|
239
|
+
didnt-run ping <id> [--duration-ms N] [--meta k=v]...
|
|
240
|
+
didnt-run run <id> [--lifecycle] [--capture-stderr] -- <cmd> [args...]
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
Reads `DIDNT_RUN_DSN` from the environment, or `--dsn`. `run` wraps a shell command in your crontab:
|
|
244
|
+
|
|
245
|
+
```cron
|
|
246
|
+
*/5 * * * * didnt-run run app.nightly-backup -- pg_dump -Fc mydb > /backups/mydb.dump
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
- **It exits with the wrapped command's exit code.** The crontab's semantics, and any `&&` chained
|
|
250
|
+
after it, are unchanged.
|
|
251
|
+
- **stdout and stderr pass straight through**, so cron mail still works. `--capture-stderr` *tees*
|
|
252
|
+
rather than swallows, keeping the last 512 bytes for `meta` — it defaults to **off**, because a
|
|
253
|
+
command's stderr is a population you didn't write and can't predict (`curl`, `pg_dump`, `rsync` all
|
|
254
|
+
print connection strings on failure on their own); a denylist of secret-looking patterns goes stale
|
|
255
|
+
the day some tool prints a new one, so the safe default is off and the opt-in is explicit.
|
|
256
|
+
- **A watchdog failure never changes the exit code.** A missing or malformed `DIDNT_RUN_DSN` logs to
|
|
257
|
+
stderr and the command still runs, uninstrumented — an already-registered job that stops pinging is
|
|
258
|
+
exactly what the watchdog alerts on. A command that can't be executed at all follows the
|
|
259
|
+
shell's own convention: **127** when it is not found, **126** when it is found but not executable.
|
|
260
|
+
The same code goes into the `fail` ping's `exit_code` meta and out as the wrapper's exit status.
|
|
261
|
+
|
|
262
|
+
A malformed `--meta` on `run` is reported on stderr and dropped — the command still runs. It was the
|
|
263
|
+
only watchdog-side input on that path that could stop a backup, and a metadata typo is the least
|
|
264
|
+
important of them. `didnt-run ping` has nothing to protect, so misconfiguration there exits `2`.
|
|
265
|
+
`--meta k=v` values are always strings; no type coercion is attempted.
|
|
266
|
+
|
|
267
|
+
## Requirements
|
|
268
|
+
|
|
269
|
+
Python ≥ 3.11. Zero runtime dependencies — the SDK is stdlib only.
|
|
270
|
+
|
|
271
|
+
## License
|
|
272
|
+
|
|
273
|
+
MIT — see [LICENSE](LICENSE).
|
|
274
|
+
|
|
275
|
+
Development happens in the didnt.run monorepo; this repository is a read-only subtree split. Please
|
|
276
|
+
report issues against the SDK there rather than opening merge requests here.
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
# didntrun-sdk — Python client for didnt.run
|
|
2
|
+
|
|
3
|
+
Instrument a scheduled job against didnt.run with one line. The SDK's contract: **your job is never
|
|
4
|
+
slowed beyond a hard time budget and never failed by the watchdog** — every transport problem is
|
|
5
|
+
swallowed and logged, by default through Python's `logging` module, or through a logger you supply.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install didntrun-sdk
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Use
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
import os
|
|
17
|
+
from didntrun import Watchdog
|
|
18
|
+
|
|
19
|
+
# One env var wires everything: https://{ping-token}@it.didnt.run[?budget_ms=2000&connect_timeout_ms=1000]
|
|
20
|
+
wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
> Your DSN is shown once, ready to copy, when you issue a ping token in the didnt.run admin UI.
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
# Recommended: bracket the work. ONE completion ping — measured duration on success,
|
|
27
|
+
# exception context on failure (the exception is re-raised; your job's semantics don't change).
|
|
28
|
+
with wd.run("app.daily-report"):
|
|
29
|
+
generate()
|
|
30
|
+
|
|
31
|
+
# Or the decorator form:
|
|
32
|
+
@wd.watch("app.daily-report")
|
|
33
|
+
def daily_report():
|
|
34
|
+
...
|
|
35
|
+
|
|
36
|
+
# Or fire the bare "I ran" ping yourself:
|
|
37
|
+
wd.ping("app.daily-report")
|
|
38
|
+
wd.ping("app.daily-report", duration_ms=1234)
|
|
39
|
+
|
|
40
|
+
# Or send explicit lifecycle pings — manual bracketing when start and finish
|
|
41
|
+
# live in different processes or code paths (see "Lifecycle bracketing"):
|
|
42
|
+
wd.start("app.daily-report")
|
|
43
|
+
wd.success("app.daily-report", duration_ms=1234)
|
|
44
|
+
wd.fail("app.daily-report")
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
`ping()` returns `bool` (delivered vs. swallowed) if you want to observe delivery; you never have to.
|
|
48
|
+
|
|
49
|
+
### Job identity
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
@wd.watch() # id defaults to f"{fn.__module__}.{fn.__qualname__}"
|
|
53
|
+
def daily_report():
|
|
54
|
+
...
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
**This is identity-defining.** Moving `daily_report` to another module registers a NEW job — the old
|
|
58
|
+
one goes stale and will alert. Prefer the explicit-id form (`@wd.watch("app.daily-report")`); every
|
|
59
|
+
example above uses it for that reason. Ids are normalized the same way the PHP SDK
|
|
60
|
+
normalizes them — leading `\` stripped, `\` → `.`, empty rejected — and capped at 255 **bytes**, which
|
|
61
|
+
is the length the server measures, dropping a trailing partial UTF-8 character rather than splitting
|
|
62
|
+
one. The mapping is kept for parity rather than necessity: Python identifiers are already dotted, so
|
|
63
|
+
it is a no-op for a native id.
|
|
64
|
+
|
|
65
|
+
### Failure semantics
|
|
66
|
+
|
|
67
|
+
- Misconfiguration (a malformed DSN, an empty token, a token outside the accept set, a base URL with a
|
|
68
|
+
query, a fragment or an unparseable port, a DSN option that is not a plain positive integer) raises
|
|
69
|
+
`ValueError` **at construction**, and `@wd.watch()` raises `ValueError` **at decoration** when the
|
|
70
|
+
function it wraps is an `async def` or a generator (see "No `async`/`await`"). Both are wiring time,
|
|
71
|
+
before any job runs. After that, nothing raises except the wrapped work's own exception inside
|
|
72
|
+
`run()` / `watch()`, re-raised unchanged.
|
|
73
|
+
- Each ping gets a wall-clock budget (default 2 s, connect 1 s) with one retry inside it — see
|
|
74
|
+
"Delivery and retries" below for exactly which failures qualify. `401` is logged at ERROR (your token
|
|
75
|
+
is wrong); every other rejection at WARNING.
|
|
76
|
+
- Oversized `meta` (> 64 KiB body) is dropped; the ping still goes out.
|
|
77
|
+
|
|
78
|
+
### Wiring
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Construct once at startup, wherever your app already wires singletons, and hand `wd` around — there's
|
|
85
|
+
no framework glue to install; `Watchdog` is a plain object.
|
|
86
|
+
|
|
87
|
+
Bring your own HTTP client (optional): implement the `Transport` protocol — it's `typing.Protocol`, not
|
|
88
|
+
an ABC, so nothing has to import from this package. The whole surface is one method:
|
|
89
|
+
|
|
90
|
+
```python
|
|
91
|
+
from collections.abc import Mapping
|
|
92
|
+
|
|
93
|
+
import requests
|
|
94
|
+
from didntrun.transport import Transport, TransportResult, TimeBudget
|
|
95
|
+
|
|
96
|
+
class RequestsTransport(Transport):
|
|
97
|
+
def post(self, url: str, headers: Mapping[str, str], body: bytes, budget: TimeBudget) -> TransportResult:
|
|
98
|
+
try:
|
|
99
|
+
resp = requests.post(
|
|
100
|
+
url, headers=dict(headers), data=body, allow_redirects=False,
|
|
101
|
+
timeout=(budget.connect_ms / 1000, max(budget.total_ms, 1) / 1000),
|
|
102
|
+
)
|
|
103
|
+
return TransportResult.response(resp.status_code)
|
|
104
|
+
except requests.RequestException as e:
|
|
105
|
+
return TransportResult.failure(str(e))
|
|
106
|
+
|
|
107
|
+
wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"], transport=RequestsTransport())
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Note the time budget is then only as good as your client's own timeout config — the default
|
|
111
|
+
`HttpClientTransport` (stdlib `http.client`) is deliberately the one thing this package doesn't
|
|
112
|
+
outsource: it never follows redirects and enforces the budget by tracking a deadline on
|
|
113
|
+
`time.monotonic()` rather than relying on a per-socket-operation `timeout=`.
|
|
114
|
+
|
|
115
|
+
### Lifecycle bracketing
|
|
116
|
+
|
|
117
|
+
`run()` / `watch()` send one completion ping by default. Opt into **bracketing** and they also emit a
|
|
118
|
+
`start` ping *before* the work runs: the `start` anchor is duration-immune, so it tightens the inferred
|
|
119
|
+
cadence for jobs whose duration is non-trivial or variable (a terminal-only anchor drifts with run
|
|
120
|
+
length).
|
|
121
|
+
|
|
122
|
+
Enable it globally (constructor arg, or the `lifecycle` DSN param) with a per-call override:
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
# Global default via DSN: https://{token}@it.didnt.run?lifecycle=true
|
|
126
|
+
wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
|
|
127
|
+
with wd.run("app.daily-report"): # start + completion
|
|
128
|
+
generate()
|
|
129
|
+
|
|
130
|
+
# Per-call override wins (None = inherit the instance default):
|
|
131
|
+
with wd.run("app.daily-report", lifecycle=True):
|
|
132
|
+
generate()
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
A bracketed run has **two** budget windows — one ping before the work, one after — so its worst-case
|
|
136
|
+
added wall-clock is ~2× the per-ping budget, straddling the work. For work a single `with` block can't
|
|
137
|
+
wrap (it spans processes, or start and finish live in different code), call `start()` / `success()` /
|
|
138
|
+
`fail()` directly instead.
|
|
139
|
+
|
|
140
|
+
**Server requirement:** bracketing needs a ping-api that understands lifecycle kinds. Against an older,
|
|
141
|
+
kind-blind self-hosted server, leave `lifecycle` **off** — a separate `start` ping would double-count
|
|
142
|
+
runs in schedule inference.
|
|
143
|
+
|
|
144
|
+
**No `async`/`await`.** There's no async HTTP client in the stdlib, and shipping one would break the
|
|
145
|
+
zero-dependency promise. `@wd.watch()` **refuses** an `async def` or a generator function outright,
|
|
146
|
+
with a `ValueError` at decoration: a synchronous wrapper around a coroutine function gets the
|
|
147
|
+
coroutine object back immediately, so the completion ping would go out before the work had executed a
|
|
148
|
+
line — reporting success for a job that has not run yet, and feeding a 0 ms duration into the
|
|
149
|
+
hung-detection baseline. Await the work inside a sync wrapper, or keep the ping off the loop thread:
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
await asyncio.to_thread(wd.ping, "app.daily-report")
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
## Logging
|
|
156
|
+
|
|
157
|
+
Every failure is swallowed and logged. By default the SDK logs through the standard `logging` module —
|
|
158
|
+
`logging.getLogger("didntrun")` — with **no handler attached**, so with nothing else configured in your
|
|
159
|
+
process, Python's own `logging.lastResort` emits WARNING and above to **stderr**. Under a plain cron
|
|
160
|
+
invocation that's exactly what you want: your cron runner captures or mails it.
|
|
161
|
+
|
|
162
|
+
Attach your own handler to redirect it, the same as any other library logger:
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
import logging
|
|
166
|
+
|
|
167
|
+
logging.getLogger("didntrun").addHandler(logging.FileHandler("/var/log/didntrun.log"))
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
To silence it completely:
|
|
171
|
+
|
|
172
|
+
```python
|
|
173
|
+
logging.getLogger("didntrun").addHandler(logging.NullHandler())
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
A ping that is never delivered is the one failure you cannot see from the server side — it looks
|
|
177
|
+
identical to a job that died before pinging. Leaving logging on is what tells the two apart.
|
|
178
|
+
|
|
179
|
+
Each failure is written as exactly one line, so a log is safe to grep and count — that's also how the
|
|
180
|
+
PHP SDK's deferred time-budget question gets settled for this one: count `ping rejected` against
|
|
181
|
+
`ping not delivered`. Control characters in a value you passed — a newline or a tab inside an
|
|
182
|
+
`external_id`, say — appear as `?` rather than breaking the line in two, and the same context is also
|
|
183
|
+
passed as `extra=` for handlers that render structured fields. Your ping token never appears in a log
|
|
184
|
+
line either, including inside an error string authored by a transport.
|
|
185
|
+
|
|
186
|
+
**stderr records carry no timestamp** — `logging.lastResort` uses a bare handler with no `Formatter`, so
|
|
187
|
+
a message on stderr says only what happened, not when. Counting failures over a week works either way,
|
|
188
|
+
but correlating a specific lost ping with a specific alert needs a timestamp — either attach a real
|
|
189
|
+
handler (above) or have your cron wrapper stamp the stream.
|
|
190
|
+
|
|
191
|
+
## Delivery and retries
|
|
192
|
+
|
|
193
|
+
A ping is retried **once** when the failure could plausibly succeed on a second attempt:
|
|
194
|
+
|
|
195
|
+
| Outcome | Retried? |
|
|
196
|
+
| --- | --- |
|
|
197
|
+
| Connection refused / DNS failure / timeout | yes |
|
|
198
|
+
| `408 Request Timeout`, any `5xx` | yes |
|
|
199
|
+
| `429 Too Many Requests` | no |
|
|
200
|
+
| Any other `4xx` (`400`, `401`, `404`, …) | no |
|
|
201
|
+
|
|
202
|
+
One retry in total, never one per category. The retry has to fit inside the same time budget as the
|
|
203
|
+
first attempt (2 s by default), so against a slow server there may be no room for it — the SDK will
|
|
204
|
+
never exceed the budget to get a ping through.
|
|
205
|
+
|
|
206
|
+
## CLI
|
|
207
|
+
|
|
208
|
+
Installing the package also installs `didnt-run` — a console script, no separate install step:
|
|
209
|
+
|
|
210
|
+
```
|
|
211
|
+
didnt-run ping <id> [--duration-ms N] [--meta k=v]...
|
|
212
|
+
didnt-run run <id> [--lifecycle] [--capture-stderr] -- <cmd> [args...]
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
Reads `DIDNT_RUN_DSN` from the environment, or `--dsn`. `run` wraps a shell command in your crontab:
|
|
216
|
+
|
|
217
|
+
```cron
|
|
218
|
+
*/5 * * * * didnt-run run app.nightly-backup -- pg_dump -Fc mydb > /backups/mydb.dump
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
- **It exits with the wrapped command's exit code.** The crontab's semantics, and any `&&` chained
|
|
222
|
+
after it, are unchanged.
|
|
223
|
+
- **stdout and stderr pass straight through**, so cron mail still works. `--capture-stderr` *tees*
|
|
224
|
+
rather than swallows, keeping the last 512 bytes for `meta` — it defaults to **off**, because a
|
|
225
|
+
command's stderr is a population you didn't write and can't predict (`curl`, `pg_dump`, `rsync` all
|
|
226
|
+
print connection strings on failure on their own); a denylist of secret-looking patterns goes stale
|
|
227
|
+
the day some tool prints a new one, so the safe default is off and the opt-in is explicit.
|
|
228
|
+
- **A watchdog failure never changes the exit code.** A missing or malformed `DIDNT_RUN_DSN` logs to
|
|
229
|
+
stderr and the command still runs, uninstrumented — an already-registered job that stops pinging is
|
|
230
|
+
exactly what the watchdog alerts on. A command that can't be executed at all follows the
|
|
231
|
+
shell's own convention: **127** when it is not found, **126** when it is found but not executable.
|
|
232
|
+
The same code goes into the `fail` ping's `exit_code` meta and out as the wrapper's exit status.
|
|
233
|
+
|
|
234
|
+
A malformed `--meta` on `run` is reported on stderr and dropped — the command still runs. It was the
|
|
235
|
+
only watchdog-side input on that path that could stop a backup, and a metadata typo is the least
|
|
236
|
+
important of them. `didnt-run ping` has nothing to protect, so misconfiguration there exits `2`.
|
|
237
|
+
`--meta k=v` values are always strings; no type coercion is attempted.
|
|
238
|
+
|
|
239
|
+
## Requirements
|
|
240
|
+
|
|
241
|
+
Python ≥ 3.11. Zero runtime dependencies — the SDK is stdlib only.
|
|
242
|
+
|
|
243
|
+
## License
|
|
244
|
+
|
|
245
|
+
MIT — see [LICENSE](LICENSE).
|
|
246
|
+
|
|
247
|
+
Development happens in the didnt.run monorepo; this repository is a read-only subtree split. Please
|
|
248
|
+
report issues against the SDK there rather than opening merge requests here.
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "didntrun-sdk"
|
|
7
|
+
version = "1.0.0"
|
|
8
|
+
description = "First-party Python client for the didnt.run ping API: correct, non-blocking instrumentation by default."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Robert Štětka" }]
|
|
14
|
+
keywords = ["cron", "monitoring", "watchdog", "heartbeat", "healthcheck", "didnt.run"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 5 - Production/Stable",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Intended Audience :: System Administrators",
|
|
19
|
+
"License :: OSI Approved :: MIT License",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Programming Language :: Python :: 3.13",
|
|
23
|
+
"Topic :: System :: Monitoring",
|
|
24
|
+
"Typing :: Typed",
|
|
25
|
+
]
|
|
26
|
+
dependencies = []
|
|
27
|
+
|
|
28
|
+
[project.urls]
|
|
29
|
+
Homepage = "https://didnt.run"
|
|
30
|
+
Source = "https://gitlab.com/robert.stetka/didnt-run-sdk-python"
|
|
31
|
+
|
|
32
|
+
[project.scripts]
|
|
33
|
+
didnt-run = "didntrun.cli:main"
|
|
34
|
+
|
|
35
|
+
[project.optional-dependencies]
|
|
36
|
+
dev = ["pytest>=8.3", "mypy>=1.13", "ruff>=0.8"]
|
|
37
|
+
contract = ["psycopg[binary]>=3.2"]
|
|
38
|
+
|
|
39
|
+
[tool.hatch.build.targets.wheel]
|
|
40
|
+
packages = ["src/didntrun"]
|
|
41
|
+
|
|
42
|
+
[tool.hatch.build.targets.sdist]
|
|
43
|
+
include = ["src", "tests", "README.md", "LICENSE", "pyproject.toml"]
|
|
44
|
+
|
|
45
|
+
[tool.ruff]
|
|
46
|
+
# Ruff 0.16 formats Python code blocks inside Markdown. A README's snippets are illustrative and
|
|
47
|
+
# hand-laid-out for reading — collapsing `def f():\n ...` onto one line and exploding a call
|
|
48
|
+
# across five is worse documentation, and ECS does not format the PHP SDK's README either.
|
|
49
|
+
extend-exclude = ["*.md"]
|
|
50
|
+
line-length = 110
|
|
51
|
+
target-version = "py311"
|
|
52
|
+
src = ["src", "tests"]
|
|
53
|
+
|
|
54
|
+
[tool.ruff.lint]
|
|
55
|
+
select = ["E", "F", "W", "I", "N", "UP", "B", "C4", "SIM", "RUF", "ANN", "PT", "S", "PTH"]
|
|
56
|
+
ignore = [
|
|
57
|
+
# `meta` is caller-supplied JSON; Mapping[str, Any] is the honest type, not a weakness.
|
|
58
|
+
"ANN401",
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
[tool.ruff.lint.per-file-ignores]
|
|
62
|
+
# S603 flags every subprocess call. Running the wrapped command IS this module's job, and it
|
|
63
|
+
# deliberately runs it WITHOUT a shell — which is the thing S603 exists to encourage.
|
|
64
|
+
"src/didntrun/cli.py" = ["S603"]
|
|
65
|
+
# Tests assert, and pytest does not want return annotations on every test function. S105/S106 flag
|
|
66
|
+
# every literal named like a credential, and a ping token fixture IS one — deliberately fake, and
|
|
67
|
+
# the thing under test in the construction suite.
|
|
68
|
+
"tests/**" = ["S101", "S105", "S106", "ANN"]
|
|
69
|
+
# The four process-level CLI tests exist to observe what a REAL wrapper process does to its own
|
|
70
|
+
# stderr, so spawning one is the method, not an oversight. S607's `sh` is deliberate too: `2>&-` is
|
|
71
|
+
# how fd 2 gets closed, and a shell found on PATH is what a crontab line would use.
|
|
72
|
+
"tests/unit/test_cli_process.py" = ["S101", "S105", "S106", "ANN", "S603", "S607"]
|
|
73
|
+
# The last two contract tests call Watchdog._send on purpose: they assert on the SERVER's limits,
|
|
74
|
+
# and going through the public ping() would hit the SDK's own guard first and never reach the wire.
|
|
75
|
+
"tests/contract/test_ping_contract.py" = ["S101", "ANN", "SLF001"]
|
|
76
|
+
|
|
77
|
+
[tool.ruff.format]
|
|
78
|
+
docstring-code-format = true
|
|
79
|
+
|
|
80
|
+
[tool.mypy]
|
|
81
|
+
python_version = "3.11"
|
|
82
|
+
strict = true
|
|
83
|
+
files = ["src", "tests"]
|
|
84
|
+
warn_unreachable = true
|
|
85
|
+
|
|
86
|
+
# Tests are type-checked, but not required to annotate every test function: pytest calls them, not
|
|
87
|
+
# us, so a missing `-> None` buys nothing. Everything else strict still applies — a test that
|
|
88
|
+
# misuses a public type is still an error, which is the point of checking them at all.
|
|
89
|
+
[[tool.mypy.overrides]]
|
|
90
|
+
module = "tests.*"
|
|
91
|
+
disallow_untyped_defs = false
|
|
92
|
+
disallow_incomplete_defs = false
|
|
93
|
+
# Follows from the two above rather than adding anything: unannotated helpers are allowed here, so
|
|
94
|
+
# flagging every call to one would make the exemption self-cancelling. mypy 2.x reports those calls
|
|
95
|
+
# from inside an unannotated test body as well, where 1.x did not.
|
|
96
|
+
disallow_untyped_calls = false
|
|
97
|
+
|
|
98
|
+
[tool.pytest.ini_options]
|
|
99
|
+
testpaths = ["tests/unit"]
|
|
100
|
+
addopts = "-q"
|
|
101
|
+
# A DeprecationWarning from the stdlib is a bug report against this package's 3.11 floor.
|
|
102
|
+
# ResourceWarning is exempt: it fires at GC time, so with sockets in play it would make the suite
|
|
103
|
+
# fail on collection timing rather than on anything a reader could act on.
|
|
104
|
+
filterwarnings = ["error", "default::ResourceWarning"]
|