didntrun-sdk 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
didntrun/__init__.py ADDED
@@ -0,0 +1,25 @@
1
+ """First-party Python client for the didnt.run ping API.
2
+
3
+ from didntrun import Watchdog
4
+
5
+ wd = Watchdog.from_dsn(os.environ["DIDNT_RUN_DSN"])
6
+ wd.ping("app.daily-report")
7
+
8
+ The contract: after construction, nothing in this package may slow the instrumented job beyond its
9
+ time budget or fail it. Every transport problem is swallowed and logged.
10
+ """
11
+
12
+ from didntrun.ping_kind import PingKind
13
+ from didntrun.transport.base import TimeBudget, Transport, TransportResult
14
+ from didntrun.transport.http_client import HttpClientTransport
15
+ from didntrun.watchdog import Watchdog
16
+
17
+ __all__ = [
18
+ "HttpClientTransport",
19
+ "PingKind",
20
+ "TimeBudget",
21
+ "Transport",
22
+ "TransportResult",
23
+ "Watchdog",
24
+ ]
25
+ __version__ = "1.0.0"
didntrun/cli.py ADDED
@@ -0,0 +1,351 @@
1
+ """`didnt-run` — instrument a shell crontab without writing any Python.
2
+
3
+ No logging is configured here on purpose. The library attaches no handler, so logging.lastResort
4
+ writes WARNING and above to stderr — which is where a cron runner already looks, and which triggers
5
+ cron mail regardless of the exit code.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ import contextlib
12
+ import os
13
+ import subprocess
14
+ import sys
15
+ import threading
16
+ import time
17
+ from collections.abc import Callable, Sequence
18
+ from typing import IO
19
+
20
+ from didntrun.log_text import one_line
21
+ from didntrun.watchdog import Watchdog, _elapsed_ms
22
+
23
+ _EXIT_USAGE = 2
24
+ _REPORT_PREFIX = "didnt-run: "
25
+
26
+ #: Which subcommands take a `-- <command>` tail. Declared once and checked once, because the rule
27
+ #: "argv after `--` belongs to a subcommand that takes a command" was being spelled twice with
28
+ #: opposite polarity — `_ping` rejecting a non-empty tail, `_run` rejecting an empty one — so a
29
+ #: third subcommand would have started life with the hole open again.
30
+ _TAKES_COMMAND = frozenset({"run"})
31
+
32
+
33
+ def _check_arity(subcommand: str, external_id: str, command: list[str]) -> None:
34
+ if command and subcommand not in _TAKES_COMMAND:
35
+ # The worst failure a watchdog has. `ping` and `run` are adjacent subcommands whose argument
36
+ # lists look identical, so typing one for the other used to send a cheerful "the job ran"
37
+ # for a job that never executed — on schedule, forever, with nothing contradicting it. A
38
+ # missed alert is recoverable; a false all-clear is not.
39
+ raise _UsageError(
40
+ f"{subcommand} takes no command — did you mean "
41
+ f"`didnt-run run {external_id} -- {' '.join(command)}`?"
42
+ )
43
+ if not command and subcommand in _TAKES_COMMAND:
44
+ raise _UsageError(f"{subcommand} needs a command: didnt-run {subcommand} <id> -- <cmd> [args...]")
45
+
46
+
47
+ def _stderr_writer() -> Callable[[bytes], None]:
48
+ """Resolve where the child's stderr goes ONCE, and never let it hurt the caller.
49
+
50
+ Three shapes of `sys.stderr` reach this in the wild. A real file object (cron) has `.buffer`.
51
+ A programmatic caller — which the `argv` parameter invites — may have replaced it with a
52
+ text-only object that has none. And when fd 2 is closed (`2>&-` in a crontab line is ordinary)
53
+ CPython sets `sys.stderr` to **None** outright, which the previous version of this function
54
+ did not survive: it crashed AFTER the wrapped command had already run.
55
+
56
+ Every writer swallows its own failures. A broken pipe on OUR stderr — `didnt-run run … 2>&1 |
57
+ head` is a normal idiom, and `head` closing early is the normal case — must not become the
58
+ wrapped command's problem.
59
+ """
60
+ stream = sys.stderr
61
+ if stream is None:
62
+ return lambda _chunk: None
63
+
64
+ sink = getattr(stream, "buffer", None)
65
+ if sink is not None:
66
+
67
+ def write_bytes(chunk: bytes) -> None:
68
+ with contextlib.suppress(Exception):
69
+ sink.write(chunk)
70
+ sink.flush()
71
+
72
+ return write_bytes
73
+
74
+ def write_text(chunk: bytes) -> None:
75
+ with contextlib.suppress(Exception):
76
+ stream.write(chunk.decode("utf-8", errors="replace"))
77
+ stream.flush()
78
+
79
+ return write_text
80
+
81
+
82
+ def _quiet_broken_stderr() -> None:
83
+ """Stop a broken stderr pipe from becoming exit 120 at interpreter shutdown.
84
+
85
+ CPython exits **120** when it cannot flush sys.stderr on the way out, and that happens after
86
+ main() has returned the wrapped command's exit code — so `didnt-run run job -- cmd 2>&1 | head`
87
+ silently replaced a successful 0 with 120. Suppressing the write is not enough: the data is
88
+ still sitting in the buffer at shutdown. Pointing fd 2 at /dev/null once the pipe is known
89
+ broken gives the interpreter something it can flush.
90
+
91
+ `head`, `grep -m1` and `logger` all close early as a matter of course, so this is the ordinary
92
+ case for a crontab line, not an edge case.
93
+ """
94
+ if sys.stderr is None:
95
+ return
96
+ try:
97
+ sys.stderr.flush()
98
+ except Exception:
99
+ with contextlib.suppress(Exception):
100
+ os.dup2(os.open(os.devnull, os.O_WRONLY), sys.stderr.fileno())
101
+
102
+
103
+ def _report(message: str) -> None:
104
+ """The one place a CLI diagnostic reaches stderr.
105
+
106
+ docs/sdk-conformance.md C-9 — one call, one line — is a property of the package, not of
107
+ Watchdog. These messages interpolate raw argv (an external_id, a command line), so a CR/LF in
108
+ one would forge a second line indistinguishable from a genuine one.
109
+
110
+ It writes through _stderr_writer rather than `print(file=sys.stderr)`, because when fd 2 is
111
+ closed `sys.stderr` is None and `print(file=None)` falls back to **stdout** — which injects
112
+ `didnt-run:` chatter into the wrapped command's own output, breaking `didnt-run run j -- emit
113
+ -json | jq` in the one direction nothing tests for: not swallowing output, adding it.
114
+ """
115
+ _stderr_writer()(f"{one_line(_REPORT_PREFIX + message)}\n".encode())
116
+
117
+
118
+ #: Shell convention: 126 = found but not executable, 127 = not found. The wrapper claims the
119
+ #: crontab's semantics are unchanged, so it has to make the same distinction the shell does.
120
+ _EXIT_NOT_FOUND = 127
121
+ _EXIT_NOT_EXECUTABLE = 126
122
+ _STDERR_TAIL_BYTES = 512
123
+ _STDERR_TAIL_CHARS = 512
124
+ #: How long to let the drain finish after the child exits. Bounded so a backgrounded grandchild
125
+ #: holding the pipe open cannot keep the wrapper alive.
126
+ _DRAIN_GRACE_S = 0.25
127
+
128
+
129
+ class _UsageError(Exception):
130
+ """A usage problem: exit 2, and never run anything."""
131
+
132
+
133
+ def _split_command(argv: list[str]) -> tuple[list[str], list[str]]:
134
+ """Split on the first `--`, before argparse sees anything.
135
+
136
+ argparse.REMAINDER is quirky and semi-deprecated, and the wrapped command's own flags (`-v`,
137
+ `--force`) must never be parsed as ours.
138
+ """
139
+ if "--" in argv:
140
+ index = argv.index("--")
141
+ return argv[:index], argv[index + 1 :]
142
+ return argv, []
143
+
144
+
145
+ def _parser() -> argparse.ArgumentParser:
146
+ common = argparse.ArgumentParser(add_help=False)
147
+ common.add_argument("--dsn", default=None, help="Overrides $DIDNT_RUN_DSN.")
148
+ common.add_argument(
149
+ "--meta",
150
+ action="append",
151
+ default=[],
152
+ metavar="KEY=VALUE",
153
+ help="Extra context for the ping; repeatable. Values are always strings.",
154
+ )
155
+
156
+ parser = argparse.ArgumentParser(prog="didnt-run", description="didnt.run watchdog client.")
157
+ subparsers = parser.add_subparsers(dest="subcommand", required=True)
158
+
159
+ ping = subparsers.add_parser("ping", parents=[common], help="Send one 'the job ran' ping.")
160
+ ping.add_argument("external_id")
161
+ ping.add_argument("--duration-ms", type=int, default=None)
162
+
163
+ run = subparsers.add_parser(
164
+ "run",
165
+ parents=[common],
166
+ help="Run a command, then report how it went.",
167
+ epilog="Example: didnt-run run backups.nightly -- /usr/bin/backup.sh --full",
168
+ )
169
+ run.add_argument("external_id")
170
+ run.add_argument("--lifecycle", action="store_true", help="Also send a start ping first.")
171
+ run.add_argument(
172
+ "--capture-stderr",
173
+ action="store_true",
174
+ help="Tee stderr and put its last 512 bytes in meta. OFF by default: a failing command's "
175
+ "stderr routinely contains connection strings, and unlike an exception message you did "
176
+ "not write it and cannot predict it.",
177
+ )
178
+ return parser
179
+
180
+
181
+ def _meta(pairs: Sequence[str]) -> dict[str, str]:
182
+ meta: dict[str, str] = {}
183
+ for pair in pairs:
184
+ key, separator, value = pair.partition("=")
185
+ if not separator or not key:
186
+ raise _UsageError(f"--meta expects KEY=VALUE, got {pair!r}")
187
+ meta[key] = value
188
+ return meta
189
+
190
+
191
+ def _run_meta(pairs: Sequence[str]) -> dict[str, str]:
192
+ """`--meta` for `run`, where a typo must not stop the backup.
193
+
194
+ It was the ONLY watchdog-side input on this path that could: a missing DSN, a malformed DSN, an
195
+ empty external_id and an oversize one all run the command anyway, on the stated reasoning that
196
+ this executes at cron time rather than wiring time. A metadata typo is strictly less important
197
+ than any of those, and it was the one that silently stopped the job — forever, since a crontab
198
+ line does not get retyped.
199
+ """
200
+ try:
201
+ return _meta(pairs)
202
+ except _UsageError as error:
203
+ _report(f"{error} — continuing without it")
204
+ return {}
205
+
206
+
207
+ def _watchdog(args: argparse.Namespace) -> Watchdog:
208
+ dsn = args.dsn or os.environ.get("DIDNT_RUN_DSN", "")
209
+ if not dsn:
210
+ raise ValueError("no DSN: set $DIDNT_RUN_DSN or pass --dsn")
211
+ return Watchdog.from_dsn(dsn)
212
+
213
+
214
+ def _ping(args: argparse.Namespace, meta: dict[str, str]) -> int:
215
+
216
+ try:
217
+ watchdog = _watchdog(args)
218
+ except ValueError as error:
219
+ _report(f"{error}")
220
+ return _EXIT_USAGE
221
+
222
+ # The return value is deliberately discarded. An undeliverable ping is already a log line on
223
+ # stderr, which is what cron mails on; failing the command as well would make a brief watchdog
224
+ # outage look like a failed backup.
225
+ watchdog.ping(args.external_id, duration_ms=args.duration_ms, meta=meta or None)
226
+ return 0
227
+
228
+
229
+ def _drain(stream: IO[bytes], write: Callable[[bytes], None], tail: bytearray) -> None:
230
+ """Tee the child's stderr as it arrives, keeping the last _STDERR_TAIL_BYTES.
231
+
232
+ `os.read` rather than `stream.read(4096)`: BufferedReader.read(n) blocks until it has n bytes or
233
+ EOF, so a job that logs a line then works for three minutes had its line withheld for three
234
+ minutes — and if the wrapper died in between, that output was lost where it would have reached
235
+ cron mail unwrapped. os.read returns whatever is available.
236
+
237
+ Swallows everything. This runs on a thread alongside a command that is already executing, and
238
+ the one job it must never do is turn a problem with OUR stderr into a problem for the child.
239
+ """
240
+ with contextlib.suppress(Exception):
241
+ while True:
242
+ chunk = os.read(stream.fileno(), 4096)
243
+ if not chunk:
244
+ return
245
+ write(chunk)
246
+ tail.extend(chunk)
247
+ del tail[:-_STDERR_TAIL_BYTES]
248
+
249
+
250
+ def _execute(command: list[str], *, capture_stderr: bool) -> tuple[int, str | None]:
251
+ """Run the command. Returns (returncode, stderr tail or None).
252
+
253
+ Without capture this is a plain passthrough — the child inherits our stdio, so cron sees exactly
254
+ what it would have seen without the wrapper.
255
+
256
+ With capture the drain runs on a THREAD while this one waits on the child, which is not a
257
+ refinement but a correctness fix. Reading to EOF in the foreground waits for the last writer to
258
+ close the pipe, and a child that backgrounds anything (`sleep 5 & exit 0`) keeps it open — so
259
+ the wrapper returned five seconds after the command did, reported an inflated duration into the
260
+ p99 that feeds hung detection, and with a real daemon never returned at all. Turning on a
261
+ metadata flag must not change when the wrapper returns.
262
+ """
263
+ if not capture_stderr:
264
+ return subprocess.run(command, check=False).returncode, None
265
+
266
+ tail = bytearray()
267
+ process = subprocess.Popen(command, stderr=subprocess.PIPE)
268
+ stream = process.stderr
269
+ if stream is None: # pragma: no cover - PIPE always provides one
270
+ return process.wait(), None
271
+
272
+ drain = threading.Thread(target=_drain, args=(stream, _stderr_writer(), tail), daemon=True)
273
+ drain.start()
274
+ returncode = process.wait()
275
+ drain.join(timeout=_DRAIN_GRACE_S)
276
+ with contextlib.suppress(Exception):
277
+ stream.close()
278
+
279
+ # Truncate AFTER decoding: errors="replace" expands each orphaned byte into a 3-byte U+FFFD, so
280
+ # capping the bytearray alone let the tail exceed the 512 bytes --help promises.
281
+ return returncode, tail.decode("utf-8", errors="replace")[-_STDERR_TAIL_CHARS:]
282
+
283
+
284
+ def _run(args: argparse.Namespace, meta: dict[str, str], command: list[str]) -> int:
285
+ watchdog: Watchdog | None
286
+ try:
287
+ watchdog = _watchdog(args)
288
+ except ValueError as error:
289
+ # Deliberately NOT fatal, unlike `ping`: there is a command to protect, and this runs at
290
+ # cron time rather than at wiring time.
291
+ _report(f"{error} — running the command uninstrumented")
292
+ watchdog = None
293
+
294
+ if watchdog is not None and (args.lifecycle or watchdog.lifecycle):
295
+ watchdog.start(args.external_id, meta=meta or None)
296
+
297
+ started = time.monotonic()
298
+ try:
299
+ returncode, stderr_tail = _execute(command, capture_stderr=args.capture_stderr)
300
+ except OSError as error:
301
+ not_runnable = _EXIT_NOT_EXECUTABLE if isinstance(error, PermissionError) else _EXIT_NOT_FOUND
302
+ elapsed_ms = _elapsed_ms(started)
303
+ if watchdog is not None:
304
+ watchdog.fail(
305
+ args.external_id,
306
+ duration_ms=elapsed_ms,
307
+ meta={**meta, "exit_code": not_runnable, "error": str(error)},
308
+ )
309
+ _report(f"{error}")
310
+ return not_runnable
311
+
312
+ elapsed_ms = _elapsed_ms(started)
313
+ # subprocess reports a signal death as a NEGATIVE returncode; shells report 128 + signal.
314
+ exit_code = returncode if returncode >= 0 else 128 - returncode
315
+ outcome: dict[str, object] = {**meta, "exit_code": exit_code}
316
+ if returncode < 0:
317
+ outcome["signal"] = -returncode
318
+ if stderr_tail is not None:
319
+ outcome["stderr_tail"] = stderr_tail
320
+
321
+ if watchdog is not None:
322
+ report = watchdog.success if exit_code == 0 else watchdog.fail
323
+ report(args.external_id, duration_ms=elapsed_ms, meta=outcome)
324
+
325
+ return exit_code
326
+
327
+
328
+ def main(argv: Sequence[str] | None = None) -> int:
329
+ try:
330
+ return _main(argv)
331
+ finally:
332
+ _quiet_broken_stderr()
333
+
334
+
335
+ def _main(argv: Sequence[str] | None = None) -> int:
336
+ raw = list(sys.argv[1:] if argv is None else argv)
337
+ before, command = _split_command(raw)
338
+ try:
339
+ args = _parser().parse_args(before)
340
+ _check_arity(args.subcommand, args.external_id, command)
341
+ if args.subcommand == "ping":
342
+ # `ping` has nothing to protect, so a bad --meta is fatal here.
343
+ return _ping(args, _meta(args.meta))
344
+ return _run(args, _run_meta(args.meta), command)
345
+ except _UsageError as error:
346
+ _report(f"{error}")
347
+ return _EXIT_USAGE
348
+
349
+
350
+ if __name__ == "__main__":
351
+ sys.exit(main())
@@ -0,0 +1,33 @@
1
+ """The FQCN-friendly external_id normalizer.
2
+
3
+ IDENTITY-DEFINING: renaming a class, or moving a function between modules, renames the job. Dots
4
+ instead of backslashes so ids survive proxies (%5C mangling) and read cleanly in the SPA.
5
+ """
6
+
7
+ #: Mirrors ping-api's RecordPing::MAX_EXTERNAL_ID_LENGTH. Counted in BYTES, because the server
8
+ #: checks strlen() — see truncate().
9
+ MAX_LENGTH = 255
10
+
11
+
12
+ def normalize(raw: str) -> str | None:
13
+ """Return the canonical id, or None when it normalizes to nothing."""
14
+ identifier = raw.lstrip("\\").replace("\\", ".")
15
+ return identifier or None
16
+
17
+
18
+ def byte_length(identifier: str) -> int:
19
+ """The length the server will measure."""
20
+ return len(identifier.encode("utf-8"))
21
+
22
+
23
+ def truncate(identifier: str) -> str:
24
+ """Cut to MAX_LENGTH *bytes*, dropping a trailing partial character rather than splitting one.
25
+
26
+ `len()` counts characters and the server counts bytes, so a non-ASCII id can be under the cap by
27
+ one measure and over it by the other. Slicing bytes can land mid-sequence, which is what
28
+ errors="ignore" cleans up.
29
+ """
30
+ encoded = identifier.encode("utf-8")
31
+ if len(encoded) <= MAX_LENGTH:
32
+ return identifier
33
+ return encoded[:MAX_LENGTH].decode("utf-8", errors="ignore")
didntrun/log_text.py ADDED
@@ -0,0 +1,15 @@
1
+ """One log() call must produce exactly ONE log line, whatever the sink.
2
+
3
+ Context values arrive verbatim from the caller: `external_id` is whatever the consumer passed, and
4
+ normalize() maps backslashes to dots without stripping control characters. A CR/LF in one forges a
5
+ second line indistinguishable from a genuine SDK line; an ESC writes raw ANSI into whatever terminal
6
+ an operator is tailing.
7
+ """
8
+
9
+ import re
10
+
11
+ _CONTROL = re.compile(r"[\x00-\x1f\x7f]")
12
+
13
+
14
+ def one_line(value: str) -> str:
15
+ return _CONTROL.sub("?", value)
didntrun/ping_kind.py ADDED
@@ -0,0 +1,12 @@
1
+ """The lifecycle vocabulary. Values mirror the domain's PingKind — pinned by the contract suite."""
2
+
3
+ from enum import StrEnum
4
+
5
+
6
+ class PingKind(StrEnum):
7
+ """`RUN` is the one-ping "the job ran" signal; the other three bracket a run."""
8
+
9
+ RUN = "run"
10
+ START = "start"
11
+ SUCCESS = "success"
12
+ FAIL = "fail"
didntrun/py.typed ADDED
File without changes
@@ -0,0 +1,4 @@
1
+ from didntrun.transport.base import TimeBudget, Transport, TransportResult
2
+ from didntrun.transport.http_client import HttpClientTransport
3
+
4
+ __all__ = ["HttpClientTransport", "TimeBudget", "Transport", "TransportResult"]
@@ -0,0 +1,84 @@
1
+ """The transport seam: the interface that lets the SDK promise it never hurts the job."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ from collections.abc import Mapping
7
+ from dataclasses import dataclass
8
+ from typing import Protocol
9
+
10
+
11
+ @dataclass(frozen=True, slots=True)
12
+ class TimeBudget:
13
+ """A wall-clock ceiling for one ping, plus the share of it the connect phase may use."""
14
+
15
+ total_ms: int
16
+ connect_ms: int
17
+
18
+ def __post_init__(self) -> None:
19
+ if self.total_ms < 1 or self.connect_ms < 1:
20
+ raise ValueError("Timeout budget must be positive milliseconds.")
21
+
22
+ def remaining_from(self, started: float, *, floor_ms: int) -> TimeBudget | None:
23
+ """What is left of this budget since `started`, or None if less than `floor_ms` remains.
24
+
25
+ The guard lives with the invariant rather than in a docstring telling callers to check
26
+ first. In a package whose top-line contract is that nothing raises after construction, a
27
+ value object whose safe use is guaranteed by a comment is one rung too low — and it held
28
+ only by accident, because the retry floor that happened to call it is also >= 1.
29
+ """
30
+ remaining_ms = self.total_ms - int((time.monotonic() - started) * 1000)
31
+ if remaining_ms < floor_ms or remaining_ms < 1:
32
+ return None
33
+ return self.with_total(remaining_ms)
34
+
35
+ def with_total(self, total_ms: int) -> TimeBudget:
36
+ """The retry's shrunken budget: whatever wall-clock remains, connect capped to fit.
37
+
38
+ The caller must ensure total_ms >= 1 — __post_init__ raises otherwise, so guard the
39
+ remainder before shrinking.
40
+ """
41
+ return TimeBudget(total_ms, min(self.connect_ms, total_ms))
42
+
43
+
44
+ @dataclass(frozen=True, slots=True)
45
+ class TransportResult:
46
+ """What a transport reports back. Either a status was received, or it was not."""
47
+
48
+ status: int | None
49
+ error: str | None
50
+
51
+ @classmethod
52
+ def response(cls, status: int) -> TransportResult:
53
+ return cls(status, None)
54
+
55
+ @classmethod
56
+ def failure(cls, error: str) -> TransportResult:
57
+ return cls(None, error)
58
+
59
+ @property
60
+ def is_connection_failure(self) -> bool:
61
+ """No HTTP response at all — refused, DNS, or a timeout before the status line.
62
+
63
+ NOT the retry predicate: retry policy is the client's concern and lives in
64
+ Watchdog._is_retryable(), which also retries 408 and 5xx. This stays a dumb value object.
65
+ """
66
+ return self.status is None
67
+
68
+
69
+ class Transport(Protocol):
70
+ """A Protocol, not a base class: a requests/httpx adapter satisfies it without importing us.
71
+
72
+ Implementations MUST NOT raise, MUST NOT exceed the budget, and MUST NOT follow redirects — a
73
+ redirect-following transport silently changes ping semantics. Report every problem as
74
+ TransportResult.failure(). Return TransportResult.response() only when an actual HTTP status was
75
+ received; anything earlier is a failure.
76
+ """
77
+
78
+ def post(
79
+ self,
80
+ url: str,
81
+ headers: Mapping[str, str],
82
+ body: bytes,
83
+ budget: TimeBudget,
84
+ ) -> TransportResult: ...