jobgpumonitor-server 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,19 @@
1
+ name: CI
2
+ on:
3
+ push:
4
+ branches: [main]
5
+ pull_request:
6
+ jobs:
7
+ test:
8
+ runs-on: ubuntu-latest
9
+ strategy:
10
+ matrix:
11
+ python: ["3.9", "3.12"]
12
+ steps:
13
+ - uses: actions/checkout@v4
14
+ - uses: actions/setup-python@v5
15
+ with:
16
+ python-version: ${{ matrix.python }}
17
+ - run: pip install ruff && ruff check src tests
18
+ - run: pip install -e ".[dev]"
19
+ - run: pytest -q -p no:warnings
@@ -0,0 +1,47 @@
1
+ name: Release
2
+
3
+ # Tag a commit `vX.Y.Z` (matching the version in pyproject.toml) to build and publish to PyPI.
4
+ # Uses PyPI trusted publishing: register this repo/workflow once at
5
+ # https://pypi.org/manage/account/publishing/ (no API token needed).
6
+
7
+ on:
8
+ push:
9
+ tags: ["v*"]
10
+
11
+ jobs:
12
+ build:
13
+ runs-on: ubuntu-latest
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: actions/setup-python@v5
17
+ with:
18
+ python-version: "3.12"
19
+ - run: pip install build
20
+ - name: check tag matches package version
21
+ run: |
22
+ v=$(python -c "import tomllib;print(tomllib.load(open('pyproject.toml','rb'))['project']['version'])")
23
+ test "v$v" = "${GITHUB_REF_NAME}" || { echo "tag ${GITHUB_REF_NAME} != v$v"; exit 1; }
24
+ - run: python -m build
25
+ - uses: actions/upload-artifact@v4
26
+ with:
27
+ name: dist
28
+ path: dist/
29
+
30
+ publish:
31
+ needs: build
32
+ runs-on: ubuntu-latest
33
+ environment: pypi
34
+ permissions:
35
+ id-token: write
36
+ contents: write
37
+ steps:
38
+ - uses: actions/download-artifact@v4
39
+ with:
40
+ name: dist
41
+ path: dist/
42
+ - uses: pypa/gh-action-pypi-publish@release/v1
43
+ - name: GitHub release
44
+ uses: softprops/action-gh-release@v2
45
+ with:
46
+ files: dist/*
47
+ generate_release_notes: true
@@ -0,0 +1,14 @@
1
+ __pycache__/
2
+ *.pyc
3
+ *.egg-info/
4
+ .venv/
5
+ dist/
6
+ build/
7
+ .pytest_cache/
8
+ .ruff_cache/
9
+ .DS_Store
10
+ *.service
11
+ *.db
12
+ *.sqlite*
13
+ update.bat
14
+ !deploy/*.service
@@ -0,0 +1,104 @@
1
+ Metadata-Version: 2.5
2
+ Name: jobgpumonitor-server
3
+ Version: 0.4.0
4
+ Summary: Consumer for jobgpumonitor events: run state, alert rules, notifications (ntfy, Telegram, webhook) and a small API.
5
+ Project-URL: Homepage, https://github.com/marcpinet/jobgpumonitor-server
6
+ Author: Marc Pinet
7
+ License-Expression: MIT
8
+ Keywords: gpu,hpc,monitoring,notifications,ntfy,oar,slurm
9
+ Requires-Python: >=3.9
10
+ Provides-Extra: api
11
+ Requires-Dist: fastapi>=0.110; extra == 'api'
12
+ Requires-Dist: uvicorn>=0.27; extra == 'api'
13
+ Provides-Extra: dev
14
+ Requires-Dist: fastapi>=0.110; extra == 'dev'
15
+ Requires-Dist: httpx>=0.27; extra == 'dev'
16
+ Requires-Dist: jobgpumonitor>=0.1.0; extra == 'dev'
17
+ Requires-Dist: pytest>=8; extra == 'dev'
18
+ Requires-Dist: ruff>=0.6; extra == 'dev'
19
+ Requires-Dist: tomli>=2; (python_version < '3.11') and extra == 'dev'
20
+ Provides-Extra: toml
21
+ Requires-Dist: tomli>=2; (python_version < '3.11') and extra == 'toml'
22
+ Description-Content-Type: text/markdown
23
+
24
+ # jobgpumonitor-server
25
+
26
+ The consumer side of [jobgpumonitor](https://github.com/marcpinet/jobgpumonitor): reads the
27
+ JSONL events written by the emitter and the scheduler probe, keeps one document per run,
28
+ sends notifications (ntfy, Telegram, webhook) and serves a small read-only API.
29
+
30
+ ```
31
+ jobgpumonitor (in the job) ──┐
32
+ ├──> $JGM_DIR/runs/**.jsonl ──> jgmd serve ──> ntfy / Telegram / webhook
33
+ jgm scheduler (login node) ──┘ └──> SQLite ──> HTTP API
34
+ ```
35
+
36
+ ## Quick start (login node)
37
+
38
+ ```bash
39
+ pip install "jobgpumonitor-server[api]"
40
+ jgmd init # writes ~/.config/jgm-server/config.toml
41
+ $EDITOR ~/.config/jgm-server/config.toml # put an ntfy topic or a Telegram bot in [notify.*]
42
+ jgmd notify-test # phone should buzz
43
+ jgmd serve --api # keep it in tmux / systemd --user
44
+ ```
45
+
46
+ Then, still on the login node, `jgm scheduler` from the emitter package so that OOM,
47
+ time-outs, preemptions and the queue are reported too.
48
+
49
+ ## What you get notified about
50
+
51
+ | Alert | When |
52
+ |---|---|
53
+ | 🚀 started | the job leaves the queue (with node, GPUs, time queued, time limit) |
54
+ | ✅ finished | clean end: duration, last metrics, GPU utilisation and idle share, peak memory |
55
+ | ❌ failed / OOM / timed out / cancelled / preempted | with the traceback or the scheduler's reason, exit code, last stderr lines, path of the `.out` file |
56
+ | ⚠️ correction | the scheduler's verdict contradicts what the process reported |
57
+ | ⏱ will not finish in time | tqdm ETA overshoots the job deadline by more than 2 minutes |
58
+ | 🧠 memory | above 90 % of the cgroup / requested memory |
59
+ | 🥱 GPU idle | every visible GPU under 5 % for 15 minutes |
60
+ | 💀 stopped reporting | no heartbeat for 3 intervals while the scheduler still says RUNNING (15 min when nothing at all arrives from the cluster: the agent or the proxy is then the likelier culprit) |
61
+ | 🔄 reporting again | heartbeats are back after a 💀; the next silence is reported again |
62
+ | ⚠️ correction | the scheduler's real verdict arrives after an `UNKNOWN_ENDED` (accounting was down) |
63
+
64
+ Every alert fires once per run and is recorded, so restarting the server never re-sends.
65
+ A channel that fails (ntfy down, no network) gets the alert again later, with backoff, for
66
+ up to 6 hours; the other channels are not sent it twice. Raw events are kept
67
+ `keep_events_days` after they are received (not after they were emitted).
68
+
69
+ ## API
70
+
71
+ `jgmd serve --api` (needs the `[api]` extra) exposes on `127.0.0.1:21834`:
72
+
73
+ ```
74
+ GET /health
75
+ GET /runs?phase=running
76
+ GET /runs/<cluster>/<job>/<restart>
77
+ GET /runs/<cluster>/<job>/<restart>/events?after=0&types=metric.log,progress.update
78
+ GET /runs/<cluster>/<job>/<restart>/stream # server-sent events
79
+ GET /runs/<cluster>/<job>/<restart>/logs?stream=stdout # the job's .out/.err, live
80
+ GET /alerts
81
+ ```
82
+
83
+ Interactive docs at `/docs`. From your laptop: `ssh -L 21834:localhost:21834 cluster`.
84
+
85
+ ## CLI
86
+
87
+ ```
88
+ jgmd serve [--once] [--api] ingest, rules, notifications
89
+ jgmd runs [--phase running] table of runs
90
+ jgmd show <run_id | job id> full run document (--events N for raw events)
91
+ jgmd alerts what was sent, and through which channel
92
+ jgmd notify-test send a test message
93
+ jgmd init write the example config
94
+ ```
95
+
96
+ Configuration lives in `~/.config/jgm-server/config.toml` (see `jgmd init`); `JGMD_DIRS`,
97
+ `JGMD_NTFY_TOPIC`, `JGMD_TELEGRAM_TOKEN` / `JGMD_TELEGRAM_CHAT_ID`, `JGMD_WEBHOOK_URL` work
98
+ without a file.
99
+
100
+ ## Development
101
+
102
+ ```bash
103
+ uv venv && uv pip install -e ".[dev]" && uv run pytest
104
+ ```
@@ -0,0 +1,81 @@
1
+ # jobgpumonitor-server
2
+
3
+ The consumer side of [jobgpumonitor](https://github.com/marcpinet/jobgpumonitor): reads the
4
+ JSONL events written by the emitter and the scheduler probe, keeps one document per run,
5
+ sends notifications (ntfy, Telegram, webhook) and serves a small read-only API.
6
+
7
+ ```
8
+ jobgpumonitor (in the job) ──┐
9
+ ├──> $JGM_DIR/runs/**.jsonl ──> jgmd serve ──> ntfy / Telegram / webhook
10
+ jgm scheduler (login node) ──┘ └──> SQLite ──> HTTP API
11
+ ```
12
+
13
+ ## Quick start (login node)
14
+
15
+ ```bash
16
+ pip install "jobgpumonitor-server[api]"
17
+ jgmd init # writes ~/.config/jgm-server/config.toml
18
+ $EDITOR ~/.config/jgm-server/config.toml # put an ntfy topic or a Telegram bot in [notify.*]
19
+ jgmd notify-test # phone should buzz
20
+ jgmd serve --api # keep it in tmux / systemd --user
21
+ ```
22
+
23
+ Then, still on the login node, `jgm scheduler` from the emitter package so that OOM,
24
+ time-outs, preemptions and the queue are reported too.
25
+
26
+ ## What you get notified about
27
+
28
+ | Alert | When |
29
+ |---|---|
30
+ | 🚀 started | the job leaves the queue (with node, GPUs, time queued, time limit) |
31
+ | ✅ finished | clean end: duration, last metrics, GPU utilisation and idle share, peak memory |
32
+ | ❌ failed / OOM / timed out / cancelled / preempted | with the traceback or the scheduler's reason, exit code, last stderr lines, path of the `.out` file |
33
+ | ⚠️ correction | the scheduler's verdict contradicts what the process reported |
34
+ | ⏱ will not finish in time | tqdm ETA overshoots the job deadline by more than 2 minutes |
35
+ | 🧠 memory | above 90 % of the cgroup / requested memory |
36
+ | 🥱 GPU idle | every visible GPU under 5 % for 15 minutes |
37
+ | 💀 stopped reporting | no heartbeat for 3 intervals while the scheduler still says RUNNING (15 min when nothing at all arrives from the cluster: the agent or the proxy is then the likelier culprit) |
38
+ | 🔄 reporting again | heartbeats are back after a 💀; the next silence is reported again |
39
+ | ⚠️ correction | the scheduler's real verdict arrives after an `UNKNOWN_ENDED` (accounting was down) |
40
+
41
+ Every alert fires once per run and is recorded, so restarting the server never re-sends.
42
+ A channel that fails (ntfy down, no network) gets the alert again later, with backoff, for
43
+ up to 6 hours; the other channels are not sent it twice. Raw events are kept
44
+ `keep_events_days` after they are received (not after they were emitted).
45
+
46
+ ## API
47
+
48
+ `jgmd serve --api` (needs the `[api]` extra) exposes on `127.0.0.1:21834`:
49
+
50
+ ```
51
+ GET /health
52
+ GET /runs?phase=running
53
+ GET /runs/<cluster>/<job>/<restart>
54
+ GET /runs/<cluster>/<job>/<restart>/events?after=0&types=metric.log,progress.update
55
+ GET /runs/<cluster>/<job>/<restart>/stream # server-sent events
56
+ GET /runs/<cluster>/<job>/<restart>/logs?stream=stdout # the job's .out/.err, live
57
+ GET /alerts
58
+ ```
59
+
60
+ Interactive docs at `/docs`. From your laptop: `ssh -L 21834:localhost:21834 cluster`.
61
+
62
+ ## CLI
63
+
64
+ ```
65
+ jgmd serve [--once] [--api] ingest, rules, notifications
66
+ jgmd runs [--phase running] table of runs
67
+ jgmd show <run_id | job id> full run document (--events N for raw events)
68
+ jgmd alerts what was sent, and through which channel
69
+ jgmd notify-test send a test message
70
+ jgmd init write the example config
71
+ ```
72
+
73
+ Configuration lives in `~/.config/jgm-server/config.toml` (see `jgmd init`); `JGMD_DIRS`,
74
+ `JGMD_NTFY_TOPIC`, `JGMD_TELEGRAM_TOKEN` / `JGMD_TELEGRAM_CHAT_ID`, `JGMD_WEBHOOK_URL` work
75
+ without a file.
76
+
77
+ ## Development
78
+
79
+ ```bash
80
+ uv venv && uv pip install -e ".[dev]" && uv run pytest
81
+ ```
@@ -0,0 +1,8 @@
1
+ # Deploying on a Linux box (systemd)
2
+
3
+ 1. Copy `jobgpumonitor-server.example.service` to `../jobgpumonitor-server.service`, replace `USER`.
4
+ 2. Copy `update.example.bat` to `../update.bat`, set `HOST`. Both copies are gitignored.
5
+ 3. Run `update.bat` from Windows (or the same ssh lines from any shell). It pulls, installs into `.venv`,
6
+ installs the unit and (re)starts the service. The API listens on `127.0.0.1:21834`; put your
7
+ reverse proxy in front of it.
8
+ 4. Configuration lives in `~/.config/jgm-server/config.toml` on the host (`jgmd init` writes an example).
@@ -0,0 +1,15 @@
1
+ [Unit]
2
+ Description=jobgpumonitor server (run state, alerts, API)
3
+ After=network.target
4
+
5
+ [Service]
6
+ User=USER
7
+ Type=simple
8
+ WorkingDirectory=/home/USER/jobgpumonitor-server/
9
+ ExecStart=/home/USER/jobgpumonitor-server/.venv/bin/jgmd serve --api
10
+ Restart=always
11
+ RestartSec=5
12
+ Environment=PYTHONUNBUFFERED=1
13
+
14
+ [Install]
15
+ WantedBy=multi-user.target
@@ -0,0 +1,11 @@
1
+ @echo off
2
+ :: Deploy to a Linux host over SSH (edit HOST). update.bat itself is gitignored: copy this file.
3
+ set HOST=USER@your-host.example
4
+ ssh %HOST% "cd jobgpumonitor-server && git pull"
5
+ ssh %HOST% "sudo systemctl stop jobgpumonitor-server"
6
+ ssh %HOST% "cd jobgpumonitor-server && [ ! -d .venv ] && python3 -m venv .venv"
7
+ ssh %HOST% "cd jobgpumonitor-server && source .venv/bin/activate && pip install --upgrade pip && pip install -e '.[api,toml]'"
8
+ scp jobgpumonitor-server.service %HOST%:jobgpumonitor-server/
9
+ ssh %HOST% "sudo cp jobgpumonitor-server/jobgpumonitor-server.service /etc/systemd/system/"
10
+ ssh %HOST% "sudo systemctl enable jobgpumonitor-server && sudo systemctl daemon-reload && sudo systemctl start jobgpumonitor-server"
11
+ ssh %HOST% "sleep 2 && systemctl is-active jobgpumonitor-server && curl -s http://127.0.0.1:21834/health"
@@ -0,0 +1,42 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.25"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "jobgpumonitor-server"
7
+ version = "0.4.0"
8
+ description = "Consumer for jobgpumonitor events: run state, alert rules, notifications (ntfy, Telegram, webhook) and a small API."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ requires-python = ">=3.9"
12
+ authors = [{ name = "Marc Pinet" }]
13
+ keywords = ["slurm", "oar", "hpc", "gpu", "monitoring", "notifications", "ntfy"]
14
+ dependencies = []
15
+
16
+ [project.optional-dependencies]
17
+ toml = ["tomli>=2; python_version < '3.11'"]
18
+ api = ["fastapi>=0.110", "uvicorn>=0.27"]
19
+ dev = ["pytest>=8", "ruff>=0.6", "jobgpumonitor>=0.1.0", "httpx>=0.27", "fastapi>=0.110", "tomli>=2; python_version < '3.11'"]
20
+
21
+ [project.scripts]
22
+ jgmd = "jobgpumonitor_server.cli:main"
23
+
24
+ [project.urls]
25
+ Homepage = "https://github.com/marcpinet/jobgpumonitor-server"
26
+
27
+ [tool.hatch.build.targets.wheel]
28
+ packages = ["src/jobgpumonitor_server"]
29
+
30
+ [tool.pytest.ini_options]
31
+ testpaths = ["tests"]
32
+ addopts = "-q"
33
+
34
+ [tool.ruff]
35
+ line-length = 110
36
+ target-version = "py39"
37
+ src = ["src", "tests"]
38
+
39
+ [tool.ruff.lint]
40
+ select = ["E", "F", "I", "UP", "B"]
41
+ ignore = ["E501", "UP006", "UP035", "UP007", "UP045", "B008"]
42
+
@@ -0,0 +1,3 @@
1
+ """jobgpumonitor-server: consume jobgpumonitor events, keep run state, alert, expose an API."""
2
+
3
+ __version__ = "0.4.0"
@@ -0,0 +1,187 @@
1
+ """Read-only HTTP API over the store (optional: ``pip install "jobgpumonitor-server[api]"``).
2
+
3
+ GET /health
4
+ GET /runs?phase=running&cluster=…&limit=100
5
+ GET /runs/{run_id} (run_id contains slashes: /runs/marcel-c3/8224458/0)
6
+ GET /runs/{run_id}/events?after=0&limit=1000&types=run.heartbeat,metric.log
7
+ GET /runs/{run_id}/stream server-sent events, new events as they are ingested
8
+ GET /alerts?limit=100
9
+ GET /runs/{run_id}/logs[?stream=stdout&tail=200000] live stdout/stderr rebuilt from log.chunk events
10
+ POST /ingest JSON list of events (gzip ok), bearer = ingest_token
11
+ """
12
+
13
+ import asyncio
14
+ import gzip
15
+ import hmac
16
+ import io
17
+ import json
18
+ import os
19
+ import re
20
+ import time
21
+ import zlib
22
+ from typing import Any, Dict, List, Optional
23
+
24
+ from .store import Store
25
+
26
+
27
+ def append_events(base_dir: str, events: List[Any]) -> "tuple[int, int]":
28
+ """Append well-formed events to ``<base_dir>/runs/<run_id>/<emitter>.jsonl``. Returns (accepted, rejected)."""
29
+ accepted = rejected = 0
30
+ handles: Dict[str, Any] = {}
31
+ try:
32
+ for ev in events:
33
+ if not (isinstance(ev, dict) and isinstance(ev.get("run_id"), str) and isinstance(ev.get("emitter"), str)
34
+ and isinstance(ev.get("type"), str) and isinstance(ev.get("data"), dict)
35
+ and _RUN_ID_RE.match(ev["run_id"]) and _EMITTER_RE.match(ev["emitter"])
36
+ and not any(p in (".", "..") for p in [*ev["run_id"].split("/"), ev["emitter"]])):
37
+ rejected += 1
38
+ continue
39
+ path = os.path.join(base_dir, "runs", *ev["run_id"].split("/"), ev["emitter"] + ".jsonl")
40
+ fh = handles.get(path)
41
+ if fh is None:
42
+ os.makedirs(os.path.dirname(path), exist_ok=True)
43
+ fh = handles[path] = open(path, "a", encoding="utf-8")
44
+ fh.write(json.dumps(ev, separators=(",", ":"), ensure_ascii=False) + "\n")
45
+ accepted += 1
46
+ finally:
47
+ for fh in handles.values():
48
+ fh.flush()
49
+ fh.close()
50
+ return accepted, rejected
51
+
52
+
53
+ def create_app(store: Store, token: str = "", prefix: str = "", ingest_token: str = "", ingest_dir: str = "") -> Any:
54
+ """Build the app. With ``prefix`` (e.g. ``/jgm``) every route is served under that path,
55
+ which lets a reverse proxy expose the API as a sub-path of an existing host."""
56
+ app = _build_app(store, token, ingest_token, ingest_dir)
57
+ prefix = "/" + prefix.strip("/") if prefix and prefix.strip("/") else ""
58
+ if not prefix:
59
+ return app
60
+ from fastapi import FastAPI
61
+
62
+ outer = FastAPI(docs_url=None, redoc_url=None, openapi_url=None)
63
+ outer.mount(prefix, app)
64
+ return outer
65
+
66
+
67
+ _RUN_ID_RE = re.compile(r"^[A-Za-z0-9_.\-]+/[A-Za-z0-9_.\-]+/[0-9]+$")
68
+ _EMITTER_RE = re.compile(r"^[A-Za-z0-9_.\-]+$")
69
+ _MAX_BODY = 32 * 1024 * 1024
70
+ #: Once decompressed (``jgm forward`` sends at most ~4 MB of JSONL per request; a gzip bomb
71
+ #: inflates ~1000x).
72
+ _MAX_INFLATED = 64 * 1024 * 1024
73
+
74
+
75
+ def _gunzip(raw: bytes, limit: int) -> bytes:
76
+ """Decompress, reading at most ``limit + 1`` bytes: memory stays bounded whatever the ratio."""
77
+ with gzip.GzipFile(fileobj=io.BytesIO(raw)) as gz:
78
+ return gz.read(limit + 1)
79
+
80
+
81
+ def _build_app(store: Store, token: str = "", ingest_token: str = "", ingest_dir: str = "") -> Any:
82
+ try:
83
+ from fastapi import Depends, FastAPI, HTTPException, Query, Request
84
+ from fastapi.responses import StreamingResponse
85
+ except ImportError as e: # pragma: no cover
86
+ raise SystemExit('the API needs fastapi and uvicorn: pip install "jobgpumonitor-server[api]"') from e
87
+
88
+ app = FastAPI(title="jobgpumonitor-server", version="0.1.0")
89
+
90
+ async def auth(request: Request) -> None:
91
+ if not token:
92
+ return
93
+ got = request.headers.get("authorization") or ""
94
+ if not hmac.compare_digest(got.encode(), f"Bearer {token}".encode()):
95
+ raise HTTPException(status_code=401, detail="bad token", headers={"WWW-Authenticate": "Bearer"})
96
+
97
+ async def ingest_auth(request: Request) -> None:
98
+ if not ingest_token:
99
+ raise HTTPException(status_code=503, detail="ingest disabled: set [api] ingest_token")
100
+ got = request.headers.get("authorization") or ""
101
+ if not hmac.compare_digest(got.encode(), f"Bearer {ingest_token}".encode()):
102
+ raise HTTPException(status_code=401, detail="bad ingest token", headers={"WWW-Authenticate": "Bearer"})
103
+
104
+ @app.post("/ingest", dependencies=[Depends(ingest_auth)])
105
+ async def ingest(request: Request) -> Dict[str, Any]:
106
+ """Accept a JSON list of events (optionally gzip-encoded) and append them to the local
107
+ event directory, where the engine picks them up exactly like locally written files."""
108
+ try:
109
+ declared = int(request.headers.get("content-length") or 0)
110
+ except ValueError:
111
+ declared = 0
112
+ if declared > _MAX_BODY:
113
+ raise HTTPException(status_code=413, detail="body too large")
114
+ raw = await request.body()
115
+ if len(raw) > _MAX_BODY:
116
+ raise HTTPException(status_code=413, detail="body too large")
117
+ if request.headers.get("content-encoding", "").lower() == "gzip":
118
+ try:
119
+ raw = _gunzip(raw, _MAX_INFLATED)
120
+ except (OSError, EOFError, zlib.error) as e:
121
+ raise HTTPException(status_code=400, detail="bad gzip") from e
122
+ if len(raw) > _MAX_INFLATED:
123
+ raise HTTPException(status_code=413, detail="decompressed body too large")
124
+ try:
125
+ events = json.loads(raw.decode("utf-8"))
126
+ except (ValueError, UnicodeDecodeError) as e:
127
+ raise HTTPException(status_code=400, detail="body must be a JSON list of events") from e
128
+ if not isinstance(events, list):
129
+ raise HTTPException(status_code=400, detail="body must be a JSON list of events")
130
+ accepted, rejected = append_events(ingest_dir, events)
131
+ return {"accepted": accepted, "rejected": rejected}
132
+
133
+ @app.api_route("/health", methods=["GET", "HEAD"])
134
+ def health() -> Dict[str, Any]:
135
+ return {"ok": True, "ts": time.time(), "active": len(store.active_runs())}
136
+
137
+ @app.get("/runs", dependencies=[Depends(auth)])
138
+ def runs(phase: Optional[str] = None, cluster: Optional[str] = None, limit: int = Query(100, le=1000)) -> List[Dict[str, Any]]:
139
+ return store.list_runs(phase=phase, limit=limit, cluster=cluster)
140
+
141
+ @app.get("/alerts", dependencies=[Depends(auth)])
142
+ def alerts(limit: int = Query(100, le=1000), run_id: Optional[str] = None) -> List[Dict[str, Any]]:
143
+ return store.list_alerts(limit=limit, run_id=run_id)
144
+
145
+ @app.get("/runs/{run_id:path}/events", dependencies=[Depends(auth)])
146
+ def events(run_id: str, after: int = 0, limit: int = Query(1000, le=10000), types: Optional[str] = None) -> List[Dict[str, Any]]:
147
+ return store.events(run_id, after_id=after, limit=limit, types=types.split(",") if types else None)
148
+
149
+ @app.get("/runs/{run_id:path}/logs", dependencies=[Depends(auth)])
150
+ def logs(run_id: str, stream: Optional[str] = None, tail: int = Query(200_000, le=4_000_000)) -> Any:
151
+ """Without ``stream``: the streams available. With it: the (tail of the) reconstructed file."""
152
+ if not stream:
153
+ return store.log_streams(run_id)
154
+ doc = store.get_log(run_id, stream, tail=tail)
155
+ if doc is None:
156
+ raise HTTPException(status_code=404, detail="no such log")
157
+ return doc
158
+
159
+ @app.get("/runs/{run_id:path}/stream", dependencies=[Depends(auth)])
160
+ async def stream(run_id: str, after: int = 0) -> Any:
161
+ async def gen():
162
+ last = after
163
+ while True:
164
+ batch = store.events(run_id, after_id=last, limit=500)
165
+ for e in batch:
166
+ last = e["_id"]
167
+ yield f"id: {last}\nevent: {e['type']}\ndata: {json.dumps(e, separators=(',', ':'))}\n\n"
168
+ if not batch:
169
+ yield ": keepalive\n\n"
170
+ await asyncio.sleep(2)
171
+
172
+ return StreamingResponse(gen(), media_type="text/event-stream")
173
+
174
+ @app.get("/runs/{run_id:path}", dependencies=[Depends(auth)])
175
+ def run(run_id: str) -> Dict[str, Any]:
176
+ doc = store.get_run(run_id)
177
+ if doc is None:
178
+ raise HTTPException(status_code=404, detail="unknown run")
179
+ return doc
180
+
181
+ return app
182
+
183
+
184
+ def serve_api(store: Store, host: str, port: int, token: str = "", prefix: str = "", ingest_token: str = "", ingest_dir: str = "") -> None: # pragma: no cover
185
+ import uvicorn
186
+
187
+ uvicorn.run(create_app(store, token, prefix, ingest_token, ingest_dir), host=host, port=port, log_level="warning")