jobgpumonitor-server 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jobgpumonitor_server-0.4.0/.github/workflows/ci.yml +19 -0
- jobgpumonitor_server-0.4.0/.github/workflows/release.yml +47 -0
- jobgpumonitor_server-0.4.0/.gitignore +14 -0
- jobgpumonitor_server-0.4.0/PKG-INFO +104 -0
- jobgpumonitor_server-0.4.0/README.md +81 -0
- jobgpumonitor_server-0.4.0/deploy/README.md +8 -0
- jobgpumonitor_server-0.4.0/deploy/jobgpumonitor-server.example.service +15 -0
- jobgpumonitor_server-0.4.0/deploy/update.example.bat +11 -0
- jobgpumonitor_server-0.4.0/pyproject.toml +42 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/__init__.py +3 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/api.py +187 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/cli.py +196 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/config.py +145 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/engine.py +127 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/ingest.py +93 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/model.py +310 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/notify.py +112 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/rules.py +237 -0
- jobgpumonitor_server-0.4.0/src/jobgpumonitor_server/store.py +320 -0
- jobgpumonitor_server-0.4.0/tests/test_server.py +567 -0
- jobgpumonitor_server-0.4.0/uv.lock +689 -0
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
on:
|
|
3
|
+
push:
|
|
4
|
+
branches: [main]
|
|
5
|
+
pull_request:
|
|
6
|
+
jobs:
|
|
7
|
+
test:
|
|
8
|
+
runs-on: ubuntu-latest
|
|
9
|
+
strategy:
|
|
10
|
+
matrix:
|
|
11
|
+
python: ["3.9", "3.12"]
|
|
12
|
+
steps:
|
|
13
|
+
- uses: actions/checkout@v4
|
|
14
|
+
- uses: actions/setup-python@v5
|
|
15
|
+
with:
|
|
16
|
+
python-version: ${{ matrix.python }}
|
|
17
|
+
- run: pip install ruff && ruff check src tests
|
|
18
|
+
- run: pip install -e ".[dev]"
|
|
19
|
+
- run: pytest -q -p no:warnings
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
name: Release
|
|
2
|
+
|
|
3
|
+
# Tag a commit `vX.Y.Z` (matching the version in pyproject.toml) to build and publish to PyPI.
|
|
4
|
+
# Uses PyPI trusted publishing: register this repo/workflow once at
|
|
5
|
+
# https://pypi.org/manage/account/publishing/ (no API token needed).
|
|
6
|
+
|
|
7
|
+
on:
|
|
8
|
+
push:
|
|
9
|
+
tags: ["v*"]
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
build:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: actions/setup-python@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: "3.12"
|
|
19
|
+
- run: pip install build
|
|
20
|
+
- name: check tag matches package version
|
|
21
|
+
run: |
|
|
22
|
+
v=$(python -c "import tomllib;print(tomllib.load(open('pyproject.toml','rb'))['project']['version'])")
|
|
23
|
+
test "v$v" = "${GITHUB_REF_NAME}" || { echo "tag ${GITHUB_REF_NAME} != v$v"; exit 1; }
|
|
24
|
+
- run: python -m build
|
|
25
|
+
- uses: actions/upload-artifact@v4
|
|
26
|
+
with:
|
|
27
|
+
name: dist
|
|
28
|
+
path: dist/
|
|
29
|
+
|
|
30
|
+
publish:
|
|
31
|
+
needs: build
|
|
32
|
+
runs-on: ubuntu-latest
|
|
33
|
+
environment: pypi
|
|
34
|
+
permissions:
|
|
35
|
+
id-token: write
|
|
36
|
+
contents: write
|
|
37
|
+
steps:
|
|
38
|
+
- uses: actions/download-artifact@v4
|
|
39
|
+
with:
|
|
40
|
+
name: dist
|
|
41
|
+
path: dist/
|
|
42
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
43
|
+
- name: GitHub release
|
|
44
|
+
uses: softprops/action-gh-release@v2
|
|
45
|
+
with:
|
|
46
|
+
files: dist/*
|
|
47
|
+
generate_release_notes: true
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: jobgpumonitor-server
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Consumer for jobgpumonitor events: run state, alert rules, notifications (ntfy, Telegram, webhook) and a small API.
|
|
5
|
+
Project-URL: Homepage, https://github.com/marcpinet/jobgpumonitor-server
|
|
6
|
+
Author: Marc Pinet
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
Keywords: gpu,hpc,monitoring,notifications,ntfy,oar,slurm
|
|
9
|
+
Requires-Python: >=3.9
|
|
10
|
+
Provides-Extra: api
|
|
11
|
+
Requires-Dist: fastapi>=0.110; extra == 'api'
|
|
12
|
+
Requires-Dist: uvicorn>=0.27; extra == 'api'
|
|
13
|
+
Provides-Extra: dev
|
|
14
|
+
Requires-Dist: fastapi>=0.110; extra == 'dev'
|
|
15
|
+
Requires-Dist: httpx>=0.27; extra == 'dev'
|
|
16
|
+
Requires-Dist: jobgpumonitor>=0.1.0; extra == 'dev'
|
|
17
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
18
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
19
|
+
Requires-Dist: tomli>=2; (python_version < '3.11') and extra == 'dev'
|
|
20
|
+
Provides-Extra: toml
|
|
21
|
+
Requires-Dist: tomli>=2; (python_version < '3.11') and extra == 'toml'
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# jobgpumonitor-server
|
|
25
|
+
|
|
26
|
+
The consumer side of [jobgpumonitor](https://github.com/marcpinet/jobgpumonitor): reads the
|
|
27
|
+
JSONL events written by the emitter and the scheduler probe, keeps one document per run,
|
|
28
|
+
sends notifications (ntfy, Telegram, webhook) and serves a small read-only API.
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
jobgpumonitor (in the job) ──┐
|
|
32
|
+
├──> $JGM_DIR/runs/**.jsonl ──> jgmd serve ──> ntfy / Telegram / webhook
|
|
33
|
+
jgm scheduler (login node) ──┘ └──> SQLite ──> HTTP API
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## Quick start (login node)
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
pip install "jobgpumonitor-server[api]"
|
|
40
|
+
jgmd init # writes ~/.config/jgm-server/config.toml
|
|
41
|
+
$EDITOR ~/.config/jgm-server/config.toml # put an ntfy topic or a Telegram bot in [notify.*]
|
|
42
|
+
jgmd notify-test # phone should buzz
|
|
43
|
+
jgmd serve --api # keep it in tmux / systemd --user
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Then, still on the login node, `jgm scheduler` from the emitter package so that OOM,
|
|
47
|
+
time-outs, preemptions and the queue are reported too.
|
|
48
|
+
|
|
49
|
+
## What you get notified about
|
|
50
|
+
|
|
51
|
+
| Alert | When |
|
|
52
|
+
|---|---|
|
|
53
|
+
| 🚀 started | the job leaves the queue (with node, GPUs, time queued, time limit) |
|
|
54
|
+
| ✅ finished | clean end: duration, last metrics, GPU utilisation and idle share, peak memory |
|
|
55
|
+
| ❌ failed / OOM / timed out / cancelled / preempted | with the traceback or the scheduler's reason, exit code, last stderr lines, path of the `.out` file |
|
|
56
|
+
| ⚠️ correction | the scheduler's verdict contradicts what the process reported |
|
|
57
|
+
| ⏱ will not finish in time | tqdm ETA overshoots the job deadline by more than 2 minutes |
|
|
58
|
+
| 🧠 memory | above 90 % of the cgroup / requested memory |
|
|
59
|
+
| 🥱 GPU idle | every visible GPU under 5 % for 15 minutes |
|
|
60
|
+
| 💀 stopped reporting | no heartbeat for 3 intervals while the scheduler still says RUNNING (15 min when nothing at all arrives from the cluster: the agent or the proxy is then the likelier culprit) |
|
|
61
|
+
| 🔄 reporting again | heartbeats are back after a 💀; the next silence is reported again |
|
|
62
|
+
| ⚠️ correction | the scheduler's real verdict arrives after an `UNKNOWN_ENDED` (accounting was down) |
|
|
63
|
+
|
|
64
|
+
Every alert fires once per run and is recorded, so restarting the server never re-sends.
|
|
65
|
+
A channel that fails (ntfy down, no network) gets the alert again later, with backoff, for
|
|
66
|
+
up to 6 hours; the other channels are not sent it twice. Raw events are kept
|
|
67
|
+
`keep_events_days` after they are received (not after they were emitted).
|
|
68
|
+
|
|
69
|
+
## API
|
|
70
|
+
|
|
71
|
+
`jgmd serve --api` (needs the `[api]` extra) exposes on `127.0.0.1:21834`:
|
|
72
|
+
|
|
73
|
+
```
|
|
74
|
+
GET /health
|
|
75
|
+
GET /runs?phase=running
|
|
76
|
+
GET /runs/<cluster>/<job>/<restart>
|
|
77
|
+
GET /runs/<cluster>/<job>/<restart>/events?after=0&types=metric.log,progress.update
|
|
78
|
+
GET /runs/<cluster>/<job>/<restart>/stream # server-sent events
|
|
79
|
+
GET /runs/<cluster>/<job>/<restart>/logs?stream=stdout # the job's .out/.err, live
|
|
80
|
+
GET /alerts
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Interactive docs at `/docs`. From your laptop: `ssh -L 21834:localhost:21834 cluster`.
|
|
84
|
+
|
|
85
|
+
## CLI
|
|
86
|
+
|
|
87
|
+
```
|
|
88
|
+
jgmd serve [--once] [--api] ingest, rules, notifications
|
|
89
|
+
jgmd runs [--phase running] table of runs
|
|
90
|
+
jgmd show <run_id | job id> full run document (--events N for raw events)
|
|
91
|
+
jgmd alerts what was sent, and through which channel
|
|
92
|
+
jgmd notify-test send a test message
|
|
93
|
+
jgmd init write the example config
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Configuration lives in `~/.config/jgm-server/config.toml` (see `jgmd init`); `JGMD_DIRS`,
|
|
97
|
+
`JGMD_NTFY_TOPIC`, `JGMD_TELEGRAM_TOKEN` / `JGMD_TELEGRAM_CHAT_ID`, `JGMD_WEBHOOK_URL` work
|
|
98
|
+
without a file.
|
|
99
|
+
|
|
100
|
+
## Development
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
uv venv && uv pip install -e ".[dev]" && uv run pytest
|
|
104
|
+
```
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# jobgpumonitor-server
|
|
2
|
+
|
|
3
|
+
The consumer side of [jobgpumonitor](https://github.com/marcpinet/jobgpumonitor): reads the
|
|
4
|
+
JSONL events written by the emitter and the scheduler probe, keeps one document per run,
|
|
5
|
+
sends notifications (ntfy, Telegram, webhook) and serves a small read-only API.
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
jobgpumonitor (in the job) ──┐
|
|
9
|
+
├──> $JGM_DIR/runs/**.jsonl ──> jgmd serve ──> ntfy / Telegram / webhook
|
|
10
|
+
jgm scheduler (login node) ──┘ └──> SQLite ──> HTTP API
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Quick start (login node)
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
pip install "jobgpumonitor-server[api]"
|
|
17
|
+
jgmd init # writes ~/.config/jgm-server/config.toml
|
|
18
|
+
$EDITOR ~/.config/jgm-server/config.toml # put an ntfy topic or a Telegram bot in [notify.*]
|
|
19
|
+
jgmd notify-test # phone should buzz
|
|
20
|
+
jgmd serve --api # keep it in tmux / systemd --user
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Then, still on the login node, `jgm scheduler` from the emitter package so that OOM,
|
|
24
|
+
time-outs, preemptions and the queue are reported too.
|
|
25
|
+
|
|
26
|
+
## What you get notified about
|
|
27
|
+
|
|
28
|
+
| Alert | When |
|
|
29
|
+
|---|---|
|
|
30
|
+
| 🚀 started | the job leaves the queue (with node, GPUs, time queued, time limit) |
|
|
31
|
+
| ✅ finished | clean end: duration, last metrics, GPU utilisation and idle share, peak memory |
|
|
32
|
+
| ❌ failed / OOM / timed out / cancelled / preempted | with the traceback or the scheduler's reason, exit code, last stderr lines, path of the `.out` file |
|
|
33
|
+
| ⚠️ correction | the scheduler's verdict contradicts what the process reported |
|
|
34
|
+
| ⏱ will not finish in time | tqdm ETA overshoots the job deadline by more than 2 minutes |
|
|
35
|
+
| 🧠 memory | above 90 % of the cgroup / requested memory |
|
|
36
|
+
| 🥱 GPU idle | every visible GPU under 5 % for 15 minutes |
|
|
37
|
+
| 💀 stopped reporting | no heartbeat for 3 intervals while the scheduler still says RUNNING (15 min when nothing at all arrives from the cluster: the agent or the proxy is then the likelier culprit) |
|
|
38
|
+
| 🔄 reporting again | heartbeats are back after a 💀; the next silence is reported again |
|
|
39
|
+
| ⚠️ correction | the scheduler's real verdict arrives after an `UNKNOWN_ENDED` (accounting was down) |
|
|
40
|
+
|
|
41
|
+
Every alert fires once per run and is recorded, so restarting the server never re-sends.
|
|
42
|
+
A channel that fails (ntfy down, no network) gets the alert again later, with backoff, for
|
|
43
|
+
up to 6 hours; the other channels are not sent it twice. Raw events are kept
|
|
44
|
+
`keep_events_days` after they are received (not after they were emitted).
|
|
45
|
+
|
|
46
|
+
## API
|
|
47
|
+
|
|
48
|
+
`jgmd serve --api` (needs the `[api]` extra) exposes on `127.0.0.1:21834`:
|
|
49
|
+
|
|
50
|
+
```
|
|
51
|
+
GET /health
|
|
52
|
+
GET /runs?phase=running
|
|
53
|
+
GET /runs/<cluster>/<job>/<restart>
|
|
54
|
+
GET /runs/<cluster>/<job>/<restart>/events?after=0&types=metric.log,progress.update
|
|
55
|
+
GET /runs/<cluster>/<job>/<restart>/stream # server-sent events
|
|
56
|
+
GET /runs/<cluster>/<job>/<restart>/logs?stream=stdout # the job's .out/.err, live
|
|
57
|
+
GET /alerts
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Interactive docs at `/docs`. From your laptop: `ssh -L 21834:localhost:21834 cluster`.
|
|
61
|
+
|
|
62
|
+
## CLI
|
|
63
|
+
|
|
64
|
+
```
|
|
65
|
+
jgmd serve [--once] [--api] ingest, rules, notifications
|
|
66
|
+
jgmd runs [--phase running] table of runs
|
|
67
|
+
jgmd show <run_id | job id> full run document (--events N for raw events)
|
|
68
|
+
jgmd alerts what was sent, and through which channel
|
|
69
|
+
jgmd notify-test send a test message
|
|
70
|
+
jgmd init write the example config
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
Configuration lives in `~/.config/jgm-server/config.toml` (see `jgmd init`); `JGMD_DIRS`,
|
|
74
|
+
`JGMD_NTFY_TOPIC`, `JGMD_TELEGRAM_TOKEN` / `JGMD_TELEGRAM_CHAT_ID`, `JGMD_WEBHOOK_URL` work
|
|
75
|
+
without a file.
|
|
76
|
+
|
|
77
|
+
## Development
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
uv venv && uv pip install -e ".[dev]" && uv run pytest
|
|
81
|
+
```
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# Deploying on a Linux box (systemd)
|
|
2
|
+
|
|
3
|
+
1. Copy `jobgpumonitor-server.example.service` to `../jobgpumonitor-server.service`, replace `USER`.
|
|
4
|
+
2. Copy `update.example.bat` to `../update.bat`, set `HOST`. Both copies are gitignored.
|
|
5
|
+
3. Run `update.bat` from Windows (or the same ssh lines from any shell). It pulls, installs into `.venv`,
|
|
6
|
+
installs the unit and (re)starts the service. The API listens on `127.0.0.1:21834`; put your
|
|
7
|
+
reverse proxy in front of it.
|
|
8
|
+
4. Configuration lives in `~/.config/jgm-server/config.toml` on the host (`jgmd init` writes an example).
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
[Unit]
|
|
2
|
+
Description=jobgpumonitor server (run state, alerts, API)
|
|
3
|
+
After=network.target
|
|
4
|
+
|
|
5
|
+
[Service]
|
|
6
|
+
User=USER
|
|
7
|
+
Type=simple
|
|
8
|
+
WorkingDirectory=/home/USER/jobgpumonitor-server/
|
|
9
|
+
ExecStart=/home/USER/jobgpumonitor-server/.venv/bin/jgmd serve --api
|
|
10
|
+
Restart=always
|
|
11
|
+
RestartSec=5
|
|
12
|
+
Environment=PYTHONUNBUFFERED=1
|
|
13
|
+
|
|
14
|
+
[Install]
|
|
15
|
+
WantedBy=multi-user.target
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
@echo off
|
|
2
|
+
:: Deploy to a Linux host over SSH (edit HOST). update.bat itself is gitignored: copy this file.
|
|
3
|
+
set HOST=USER@your-host.example
|
|
4
|
+
ssh %HOST% "cd jobgpumonitor-server && git pull"
|
|
5
|
+
ssh %HOST% "sudo systemctl stop jobgpumonitor-server"
|
|
6
|
+
ssh %HOST% "cd jobgpumonitor-server && [ ! -d .venv ] && python3 -m venv .venv"
|
|
7
|
+
ssh %HOST% "cd jobgpumonitor-server && source .venv/bin/activate && pip install --upgrade pip && pip install -e '.[api,toml]'"
|
|
8
|
+
scp jobgpumonitor-server.service %HOST%:jobgpumonitor-server/
|
|
9
|
+
ssh %HOST% "sudo cp jobgpumonitor-server/jobgpumonitor-server.service /etc/systemd/system/"
|
|
10
|
+
ssh %HOST% "sudo systemctl enable jobgpumonitor-server && sudo systemctl daemon-reload && sudo systemctl start jobgpumonitor-server"
|
|
11
|
+
ssh %HOST% "sleep 2 && systemctl is-active jobgpumonitor-server && curl -s http://127.0.0.1:21834/health"
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.25"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "jobgpumonitor-server"
|
|
7
|
+
version = "0.4.0"
|
|
8
|
+
description = "Consumer for jobgpumonitor events: run state, alert rules, notifications (ntfy, Telegram, webhook) and a small API."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
authors = [{ name = "Marc Pinet" }]
|
|
13
|
+
keywords = ["slurm", "oar", "hpc", "gpu", "monitoring", "notifications", "ntfy"]
|
|
14
|
+
dependencies = []
|
|
15
|
+
|
|
16
|
+
[project.optional-dependencies]
|
|
17
|
+
toml = ["tomli>=2; python_version < '3.11'"]
|
|
18
|
+
api = ["fastapi>=0.110", "uvicorn>=0.27"]
|
|
19
|
+
dev = ["pytest>=8", "ruff>=0.6", "jobgpumonitor>=0.1.0", "httpx>=0.27", "fastapi>=0.110", "tomli>=2; python_version < '3.11'"]
|
|
20
|
+
|
|
21
|
+
[project.scripts]
|
|
22
|
+
jgmd = "jobgpumonitor_server.cli:main"
|
|
23
|
+
|
|
24
|
+
[project.urls]
|
|
25
|
+
Homepage = "https://github.com/marcpinet/jobgpumonitor-server"
|
|
26
|
+
|
|
27
|
+
[tool.hatch.build.targets.wheel]
|
|
28
|
+
packages = ["src/jobgpumonitor_server"]
|
|
29
|
+
|
|
30
|
+
[tool.pytest.ini_options]
|
|
31
|
+
testpaths = ["tests"]
|
|
32
|
+
addopts = "-q"
|
|
33
|
+
|
|
34
|
+
[tool.ruff]
|
|
35
|
+
line-length = 110
|
|
36
|
+
target-version = "py39"
|
|
37
|
+
src = ["src", "tests"]
|
|
38
|
+
|
|
39
|
+
[tool.ruff.lint]
|
|
40
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
41
|
+
ignore = ["E501", "UP006", "UP035", "UP007", "UP045", "B008"]
|
|
42
|
+
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""Read-only HTTP API over the store (optional: ``pip install "jobgpumonitor-server[api]"``).
|
|
2
|
+
|
|
3
|
+
GET /health
|
|
4
|
+
GET /runs?phase=running&cluster=…&limit=100
|
|
5
|
+
GET /runs/{run_id} (run_id contains slashes: /runs/marcel-c3/8224458/0)
|
|
6
|
+
GET /runs/{run_id}/events?after=0&limit=1000&types=run.heartbeat,metric.log
|
|
7
|
+
GET /runs/{run_id}/stream server-sent events, new events as they are ingested
|
|
8
|
+
GET /alerts?limit=100
|
|
9
|
+
GET /runs/{run_id}/logs[?stream=stdout&tail=200000] live stdout/stderr rebuilt from log.chunk events
|
|
10
|
+
POST /ingest JSON list of events (gzip ok), bearer = ingest_token
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import asyncio
|
|
14
|
+
import gzip
|
|
15
|
+
import hmac
|
|
16
|
+
import io
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
import re
|
|
20
|
+
import time
|
|
21
|
+
import zlib
|
|
22
|
+
from typing import Any, Dict, List, Optional
|
|
23
|
+
|
|
24
|
+
from .store import Store
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def append_events(base_dir: str, events: List[Any]) -> "tuple[int, int]":
|
|
28
|
+
"""Append well-formed events to ``<base_dir>/runs/<run_id>/<emitter>.jsonl``. Returns (accepted, rejected)."""
|
|
29
|
+
accepted = rejected = 0
|
|
30
|
+
handles: Dict[str, Any] = {}
|
|
31
|
+
try:
|
|
32
|
+
for ev in events:
|
|
33
|
+
if not (isinstance(ev, dict) and isinstance(ev.get("run_id"), str) and isinstance(ev.get("emitter"), str)
|
|
34
|
+
and isinstance(ev.get("type"), str) and isinstance(ev.get("data"), dict)
|
|
35
|
+
and _RUN_ID_RE.match(ev["run_id"]) and _EMITTER_RE.match(ev["emitter"])
|
|
36
|
+
and not any(p in (".", "..") for p in [*ev["run_id"].split("/"), ev["emitter"]])):
|
|
37
|
+
rejected += 1
|
|
38
|
+
continue
|
|
39
|
+
path = os.path.join(base_dir, "runs", *ev["run_id"].split("/"), ev["emitter"] + ".jsonl")
|
|
40
|
+
fh = handles.get(path)
|
|
41
|
+
if fh is None:
|
|
42
|
+
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
43
|
+
fh = handles[path] = open(path, "a", encoding="utf-8")
|
|
44
|
+
fh.write(json.dumps(ev, separators=(",", ":"), ensure_ascii=False) + "\n")
|
|
45
|
+
accepted += 1
|
|
46
|
+
finally:
|
|
47
|
+
for fh in handles.values():
|
|
48
|
+
fh.flush()
|
|
49
|
+
fh.close()
|
|
50
|
+
return accepted, rejected
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def create_app(store: Store, token: str = "", prefix: str = "", ingest_token: str = "", ingest_dir: str = "") -> Any:
|
|
54
|
+
"""Build the app. With ``prefix`` (e.g. ``/jgm``) every route is served under that path,
|
|
55
|
+
which lets a reverse proxy expose the API as a sub-path of an existing host."""
|
|
56
|
+
app = _build_app(store, token, ingest_token, ingest_dir)
|
|
57
|
+
prefix = "/" + prefix.strip("/") if prefix and prefix.strip("/") else ""
|
|
58
|
+
if not prefix:
|
|
59
|
+
return app
|
|
60
|
+
from fastapi import FastAPI
|
|
61
|
+
|
|
62
|
+
outer = FastAPI(docs_url=None, redoc_url=None, openapi_url=None)
|
|
63
|
+
outer.mount(prefix, app)
|
|
64
|
+
return outer
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
_RUN_ID_RE = re.compile(r"^[A-Za-z0-9_.\-]+/[A-Za-z0-9_.\-]+/[0-9]+$")
|
|
68
|
+
_EMITTER_RE = re.compile(r"^[A-Za-z0-9_.\-]+$")
|
|
69
|
+
_MAX_BODY = 32 * 1024 * 1024
|
|
70
|
+
#: Once decompressed (``jgm forward`` sends at most ~4 MB of JSONL per request; a gzip bomb
|
|
71
|
+
#: inflates ~1000x).
|
|
72
|
+
_MAX_INFLATED = 64 * 1024 * 1024
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _gunzip(raw: bytes, limit: int) -> bytes:
|
|
76
|
+
"""Decompress, reading at most ``limit + 1`` bytes: memory stays bounded whatever the ratio."""
|
|
77
|
+
with gzip.GzipFile(fileobj=io.BytesIO(raw)) as gz:
|
|
78
|
+
return gz.read(limit + 1)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _build_app(store: Store, token: str = "", ingest_token: str = "", ingest_dir: str = "") -> Any:
|
|
82
|
+
try:
|
|
83
|
+
from fastapi import Depends, FastAPI, HTTPException, Query, Request
|
|
84
|
+
from fastapi.responses import StreamingResponse
|
|
85
|
+
except ImportError as e: # pragma: no cover
|
|
86
|
+
raise SystemExit('the API needs fastapi and uvicorn: pip install "jobgpumonitor-server[api]"') from e
|
|
87
|
+
|
|
88
|
+
app = FastAPI(title="jobgpumonitor-server", version="0.1.0")
|
|
89
|
+
|
|
90
|
+
async def auth(request: Request) -> None:
|
|
91
|
+
if not token:
|
|
92
|
+
return
|
|
93
|
+
got = request.headers.get("authorization") or ""
|
|
94
|
+
if not hmac.compare_digest(got.encode(), f"Bearer {token}".encode()):
|
|
95
|
+
raise HTTPException(status_code=401, detail="bad token", headers={"WWW-Authenticate": "Bearer"})
|
|
96
|
+
|
|
97
|
+
async def ingest_auth(request: Request) -> None:
|
|
98
|
+
if not ingest_token:
|
|
99
|
+
raise HTTPException(status_code=503, detail="ingest disabled: set [api] ingest_token")
|
|
100
|
+
got = request.headers.get("authorization") or ""
|
|
101
|
+
if not hmac.compare_digest(got.encode(), f"Bearer {ingest_token}".encode()):
|
|
102
|
+
raise HTTPException(status_code=401, detail="bad ingest token", headers={"WWW-Authenticate": "Bearer"})
|
|
103
|
+
|
|
104
|
+
@app.post("/ingest", dependencies=[Depends(ingest_auth)])
|
|
105
|
+
async def ingest(request: Request) -> Dict[str, Any]:
|
|
106
|
+
"""Accept a JSON list of events (optionally gzip-encoded) and append them to the local
|
|
107
|
+
event directory, where the engine picks them up exactly like locally written files."""
|
|
108
|
+
try:
|
|
109
|
+
declared = int(request.headers.get("content-length") or 0)
|
|
110
|
+
except ValueError:
|
|
111
|
+
declared = 0
|
|
112
|
+
if declared > _MAX_BODY:
|
|
113
|
+
raise HTTPException(status_code=413, detail="body too large")
|
|
114
|
+
raw = await request.body()
|
|
115
|
+
if len(raw) > _MAX_BODY:
|
|
116
|
+
raise HTTPException(status_code=413, detail="body too large")
|
|
117
|
+
if request.headers.get("content-encoding", "").lower() == "gzip":
|
|
118
|
+
try:
|
|
119
|
+
raw = _gunzip(raw, _MAX_INFLATED)
|
|
120
|
+
except (OSError, EOFError, zlib.error) as e:
|
|
121
|
+
raise HTTPException(status_code=400, detail="bad gzip") from e
|
|
122
|
+
if len(raw) > _MAX_INFLATED:
|
|
123
|
+
raise HTTPException(status_code=413, detail="decompressed body too large")
|
|
124
|
+
try:
|
|
125
|
+
events = json.loads(raw.decode("utf-8"))
|
|
126
|
+
except (ValueError, UnicodeDecodeError) as e:
|
|
127
|
+
raise HTTPException(status_code=400, detail="body must be a JSON list of events") from e
|
|
128
|
+
if not isinstance(events, list):
|
|
129
|
+
raise HTTPException(status_code=400, detail="body must be a JSON list of events")
|
|
130
|
+
accepted, rejected = append_events(ingest_dir, events)
|
|
131
|
+
return {"accepted": accepted, "rejected": rejected}
|
|
132
|
+
|
|
133
|
+
@app.api_route("/health", methods=["GET", "HEAD"])
|
|
134
|
+
def health() -> Dict[str, Any]:
|
|
135
|
+
return {"ok": True, "ts": time.time(), "active": len(store.active_runs())}
|
|
136
|
+
|
|
137
|
+
@app.get("/runs", dependencies=[Depends(auth)])
|
|
138
|
+
def runs(phase: Optional[str] = None, cluster: Optional[str] = None, limit: int = Query(100, le=1000)) -> List[Dict[str, Any]]:
|
|
139
|
+
return store.list_runs(phase=phase, limit=limit, cluster=cluster)
|
|
140
|
+
|
|
141
|
+
@app.get("/alerts", dependencies=[Depends(auth)])
|
|
142
|
+
def alerts(limit: int = Query(100, le=1000), run_id: Optional[str] = None) -> List[Dict[str, Any]]:
|
|
143
|
+
return store.list_alerts(limit=limit, run_id=run_id)
|
|
144
|
+
|
|
145
|
+
@app.get("/runs/{run_id:path}/events", dependencies=[Depends(auth)])
|
|
146
|
+
def events(run_id: str, after: int = 0, limit: int = Query(1000, le=10000), types: Optional[str] = None) -> List[Dict[str, Any]]:
|
|
147
|
+
return store.events(run_id, after_id=after, limit=limit, types=types.split(",") if types else None)
|
|
148
|
+
|
|
149
|
+
@app.get("/runs/{run_id:path}/logs", dependencies=[Depends(auth)])
|
|
150
|
+
def logs(run_id: str, stream: Optional[str] = None, tail: int = Query(200_000, le=4_000_000)) -> Any:
|
|
151
|
+
"""Without ``stream``: the streams available. With it: the (tail of the) reconstructed file."""
|
|
152
|
+
if not stream:
|
|
153
|
+
return store.log_streams(run_id)
|
|
154
|
+
doc = store.get_log(run_id, stream, tail=tail)
|
|
155
|
+
if doc is None:
|
|
156
|
+
raise HTTPException(status_code=404, detail="no such log")
|
|
157
|
+
return doc
|
|
158
|
+
|
|
159
|
+
@app.get("/runs/{run_id:path}/stream", dependencies=[Depends(auth)])
|
|
160
|
+
async def stream(run_id: str, after: int = 0) -> Any:
|
|
161
|
+
async def gen():
|
|
162
|
+
last = after
|
|
163
|
+
while True:
|
|
164
|
+
batch = store.events(run_id, after_id=last, limit=500)
|
|
165
|
+
for e in batch:
|
|
166
|
+
last = e["_id"]
|
|
167
|
+
yield f"id: {last}\nevent: {e['type']}\ndata: {json.dumps(e, separators=(',', ':'))}\n\n"
|
|
168
|
+
if not batch:
|
|
169
|
+
yield ": keepalive\n\n"
|
|
170
|
+
await asyncio.sleep(2)
|
|
171
|
+
|
|
172
|
+
return StreamingResponse(gen(), media_type="text/event-stream")
|
|
173
|
+
|
|
174
|
+
@app.get("/runs/{run_id:path}", dependencies=[Depends(auth)])
|
|
175
|
+
def run(run_id: str) -> Dict[str, Any]:
|
|
176
|
+
doc = store.get_run(run_id)
|
|
177
|
+
if doc is None:
|
|
178
|
+
raise HTTPException(status_code=404, detail="unknown run")
|
|
179
|
+
return doc
|
|
180
|
+
|
|
181
|
+
return app
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def serve_api(store: Store, host: str, port: int, token: str = "", prefix: str = "", ingest_token: str = "", ingest_dir: str = "") -> None: # pragma: no cover
|
|
185
|
+
import uvicorn
|
|
186
|
+
|
|
187
|
+
uvicorn.run(create_app(store, token, prefix, ingest_token, ingest_dir), host=host, port=port, log_level="warning")
|