expr-tracker 0.2.5__tar.gz → 0.2.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/PKG-INFO +2 -6
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/README.md +1 -2
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/design.md +6 -3
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/getting-started.md +4 -1
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/alerts.md +2 -2
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/distributed.md +6 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/spans.md +8 -6
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/index.md +1 -1
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/reference/api.md +1 -1
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/pyproject.toml +0 -3
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/backends/__init__.py +34 -38
- expr_tracker-0.2.7/src/expr_tracker/alerts/backends/cards.py +102 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/spans.py +5 -1
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_alert_backends.py +24 -21
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_distributed.py +22 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_examples.py +1 -1
- expr_tracker-0.2.7/tests/test_lark.py +380 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_span_plugins.py +46 -13
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/uv.lock +1 -61
- expr_tracker-0.2.5/tests/test_lark_live.py +0 -269
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/.github/workflows/docs.yaml +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/.github/workflows/release.yaml +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/.gitignore +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/LICENSE +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/architecture.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/examples.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/artifacts.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/backends.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/cli.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/history.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/logging.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/guide/streams.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/reference/configuration.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/docs/reference/expressions.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/examples/README.md +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/examples/alert_rules.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/examples/checkpoints.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/examples/early_stopping.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/examples/multiprocess_pipeline.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/examples/profile_step.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/examples/quickstart.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/mkdocs.yml +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/__init__.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/_compat.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/__init__.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/backends/base.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/dispatch.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/engine.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/expr/__init__.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/expr/eval.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/expr/functions.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/expr/lexer.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/expr/nodes.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/expr/parser.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/expr/rule.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/alerts/models.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/artifacts.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/cli.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/encoders.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/history/__init__.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/history/codec.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/history/frame.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/history/naming.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/history/reader.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/history/series.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/history/store.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/history/writer.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/plugins.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/py.typed +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/run.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/summary.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/trace.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/tracker.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/src/expr_tracker/types.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/conftest.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_alert_delivery.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_alert_dispatch.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_alert_engine.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_alert_models.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_alert_routing.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_artifacts.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_benchmark.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_cache.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_cli.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_correctness.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_e2e.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_expr_builder.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_expr_eval.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_expr_functions.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_expr_parser.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_expr_properties.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_failure_modes.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_features.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_frame_codec_summary.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_history.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_hot_paths.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_integration.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_perf.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_public_surfaces.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_review_regressions.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_rule_lifecycle.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_run_backends.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_scenarios.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_spans.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_step_commit.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_streams.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_stress.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_trace.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_trackio.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_value_encoding.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_wandb.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_writer_buffer.py +0 -0
- {expr_tracker-0.2.5 → expr_tracker-0.2.7}/tests/test_writer_durability.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: expr_tracker
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.7
|
|
4
4
|
Summary: Local-first experiment tracking with queryable history and expression-based alerts on your training metrics
|
|
5
5
|
Project-URL: Homepage, https://hspk.github.io/expr_tracker/
|
|
6
6
|
Project-URL: Documentation, https://hspk.github.io/expr_tracker/
|
|
@@ -31,13 +31,10 @@ Provides-Extra: all
|
|
|
31
31
|
Requires-Dist: nvidia-ml-py>=12.0; extra == 'all'
|
|
32
32
|
Requires-Dist: pandas>=1.5; extra == 'all'
|
|
33
33
|
Requires-Dist: polars>=0.20; extra == 'all'
|
|
34
|
-
Requires-Dist: slark>=0.1.28; extra == 'all'
|
|
35
34
|
Requires-Dist: trackio>=0.4.0; extra == 'all'
|
|
36
35
|
Requires-Dist: wandb>=0.21.0; extra == 'all'
|
|
37
36
|
Provides-Extra: gpu
|
|
38
37
|
Requires-Dist: nvidia-ml-py>=12.0; extra == 'gpu'
|
|
39
|
-
Provides-Extra: lark
|
|
40
|
-
Requires-Dist: slark>=0.1.28; extra == 'lark'
|
|
41
38
|
Provides-Extra: pandas
|
|
42
39
|
Requires-Dist: pandas>=1.5; extra == 'pandas'
|
|
43
40
|
Provides-Extra: polars
|
|
@@ -92,7 +89,6 @@ et.history(-1, output_type="pd") # everything, as a DataFrame
|
|
|
92
89
|
uv add expr_tracker # local-first: click, loguru, pydantic only
|
|
93
90
|
uv add "expr_tracker[wandb]" # mirror to Weights & Biases
|
|
94
91
|
uv add "expr_tracker[trackio]" # mirror to trackio
|
|
95
|
-
uv add "expr_tracker[lark]" # Feishu/Lark alert channel
|
|
96
92
|
uv add "expr_tracker[pandas]" # history(output_type="pandas")
|
|
97
93
|
uv add "expr_tracker[all]" # everything
|
|
98
94
|
```
|
|
@@ -180,7 +176,7 @@ uv run ruff format src tests
|
|
|
180
176
|
| `test_distributed.py` | rank shards, `alert_on_rank`, real multi-process runs |
|
|
181
177
|
| `test_wandb.py` | real wandb in offline mode: parameter mapping, step alignment, artifacts |
|
|
182
178
|
| `test_trackio.py` | trackio contract, resume mapping, real end-to-end |
|
|
183
|
-
| `
|
|
179
|
+
| `test_lark.py` | Lark card construction; real delivery when `ET_LARK_TEST_WEBHOOK` is set |
|
|
184
180
|
| `test_stress.py` (`slow`) | 100k-row writes, concurrency, cache thrash, write-failure recovery |
|
|
185
181
|
| `test_benchmark.py` (`benchmark`) | throughput, tail latency, query cost, memory stability |
|
|
186
182
|
|
|
@@ -42,7 +42,6 @@ et.history(-1, output_type="pd") # everything, as a DataFrame
|
|
|
42
42
|
uv add expr_tracker # local-first: click, loguru, pydantic only
|
|
43
43
|
uv add "expr_tracker[wandb]" # mirror to Weights & Biases
|
|
44
44
|
uv add "expr_tracker[trackio]" # mirror to trackio
|
|
45
|
-
uv add "expr_tracker[lark]" # Feishu/Lark alert channel
|
|
46
45
|
uv add "expr_tracker[pandas]" # history(output_type="pandas")
|
|
47
46
|
uv add "expr_tracker[all]" # everything
|
|
48
47
|
```
|
|
@@ -130,7 +129,7 @@ uv run ruff format src tests
|
|
|
130
129
|
| `test_distributed.py` | rank shards, `alert_on_rank`, real multi-process runs |
|
|
131
130
|
| `test_wandb.py` | real wandb in offline mode: parameter mapping, step alignment, artifacts |
|
|
132
131
|
| `test_trackio.py` | trackio contract, resume mapping, real end-to-end |
|
|
133
|
-
| `
|
|
132
|
+
| `test_lark.py` | Lark card construction; real delivery when `ET_LARK_TEST_WEBHOOK` is set |
|
|
134
133
|
| `test_stress.py` (`slow`) | 100k-row writes, concurrency, cache thrash, write-failure recovery |
|
|
135
134
|
| `test_benchmark.py` (`benchmark`) | throughput, tail latency, query cost, memory stability |
|
|
136
135
|
|
|
@@ -375,9 +375,12 @@ AlertRule(name, condition, level="warning", title=None,
|
|
|
375
375
|
timeout.
|
|
376
376
|
- **Only send failures are swallowed**; configuration errors raise at configuration
|
|
377
377
|
time.
|
|
378
|
-
- **Backends**: `lark
|
|
379
|
-
|
|
380
|
-
|
|
378
|
+
- **Backends**: `lark`, `slack`, `dingtalk`, `wecom` and `webhook` (a generic
|
|
379
|
+
template) use stdlib `urllib`; `email` uses stdlib `smtplib` — **every channel is
|
|
380
|
+
dependency-free**. `register_backend()` extends the set. The Lark card layout is
|
|
381
|
+
built as plain dicts in `backends/cards.py` rather than pulled in with a client
|
|
382
|
+
library: it is one JSON body, and a whole HTTP stack to shape it was not a trade
|
|
383
|
+
worth making.
|
|
381
384
|
- **Configuration precedence**: `init(alert=)` > `configure_alert()` >
|
|
382
385
|
`ET_ALERT_CONFIG` file > environment (`ET_LARK_WEBHOOK_URL`, legacy `WEBHOOK_URL`)
|
|
383
386
|
> defaults.
|
|
@@ -12,10 +12,13 @@ Only `click`, `loguru` and `pydantic` are required. Everything else is an extra:
|
|
|
12
12
|
| --- | --- |
|
|
13
13
|
| `wandb` | mirror metrics to Weights & Biases |
|
|
14
14
|
| `trackio` | mirror metrics to trackio |
|
|
15
|
-
| `lark` | Feishu/Lark alert channel |
|
|
16
15
|
| `pandas` / `polars` | `history(output_type=...)` frames |
|
|
16
|
+
| `gpu` | `GpuStats` span plugin, via NVML |
|
|
17
17
|
| `all` | all of the above |
|
|
18
18
|
|
|
19
|
+
Alert channels need nothing: every one of them, Lark included, is built on the
|
|
20
|
+
standard library.
|
|
21
|
+
|
|
19
22
|
A missing extra is reported with the exact install command; it never crashes a run.
|
|
20
23
|
|
|
21
24
|
## A complete run
|
|
@@ -90,8 +90,8 @@ et.alert(title="done", text="training finished", level="info", channels=["oncall
|
|
|
90
90
|
```
|
|
91
91
|
|
|
92
92
|
Built-in types: `lark`, `slack`, `dingtalk`, `wecom`, `webhook` (a generic JSON
|
|
93
|
-
template), `email`, `callable`. All
|
|
94
|
-
your own with `register_backend()`.
|
|
93
|
+
template), `email`, `callable`. All of them use only the standard library, so no
|
|
94
|
+
channel needs an extra. Add your own with `register_backend()`.
|
|
95
95
|
|
|
96
96
|
### Email
|
|
97
97
|
|
|
@@ -73,6 +73,12 @@ together with `group`, which both backends understand:
|
|
|
73
73
|
| rank 2 | `sft-1-rank2` | `sft-1` |
|
|
74
74
|
| rank 2 of stream `data` | `sft-1-data-rank2` | `sft-1` |
|
|
75
75
|
|
|
76
|
+
`rank_aware` does not affect this. It decides whether the *local* file is
|
|
77
|
+
sharded; the remote id is decided by `backend_on_rank`. They stay independent on
|
|
78
|
+
purpose — `rank_aware=False` says you serialise the writes to one file yourself,
|
|
79
|
+
and there is no equivalent lock between two processes' wandb clients, so a rank
|
|
80
|
+
that reports still needs an id of its own.
|
|
81
|
+
|
|
76
82
|
## Typical setup
|
|
77
83
|
|
|
78
84
|
```python
|
|
@@ -158,7 +158,7 @@ await asyncio.gather(work("a"), work("b"))
|
|
|
158
158
|
|
|
159
159
|
## Printing a span as it runs
|
|
160
160
|
|
|
161
|
-
Pass `print_fn` and the span announces itself, indented
|
|
161
|
+
Pass `print_fn` and the span announces itself, indented two spaces per level:
|
|
162
162
|
|
|
163
163
|
```python
|
|
164
164
|
with et.span("step", print_fn=print):
|
|
@@ -169,10 +169,10 @@ with et.span("step", print_fn=print):
|
|
|
169
169
|
|
|
170
170
|
```
|
|
171
171
|
-> step 16:41:38
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
172
|
+
-> forward 16:41:38
|
|
173
|
+
-> attention 16:41:38
|
|
174
|
+
<- attention 16:41:38 3.074ms
|
|
175
|
+
<- forward 16:41:38 5.481ms
|
|
176
176
|
<- step 16:41:38 9.641ms
|
|
177
177
|
```
|
|
178
178
|
|
|
@@ -188,7 +188,9 @@ and interleave their lines; each is still indented by its own depth, and the
|
|
|
188
188
|
`->` and `<-` markers pair them up.
|
|
189
189
|
|
|
190
190
|
Any callable taking one string works — `print`, `logger.info`, a list's `append`
|
|
191
|
-
in tests.
|
|
191
|
+
in tests. The indent is spaces rather than a tab so that it survives a log
|
|
192
|
+
prefix: a terminal measures tab stops from the start of the line, so behind
|
|
193
|
+
`... | INFO | ` the first level would collapse to whatever is left of the stop. It is called on the thread that ran the span, and an exception in it is
|
|
192
194
|
logged and swallowed.
|
|
193
195
|
|
|
194
196
|
Set it for the whole run instead of per span with `span_print_fn`:
|
|
@@ -36,7 +36,7 @@ or an unserialisable value degrades with a warning; none of them can stop traini
|
|
|
36
36
|
|
|
37
37
|
```bash
|
|
38
38
|
uv add expr_tracker # local-first, three small dependencies
|
|
39
|
-
uv add "expr_tracker[all]" # + wandb, trackio,
|
|
39
|
+
uv add "expr_tracker[all]" # + wandb, trackio, pandas, polars, gpu
|
|
40
40
|
```
|
|
41
41
|
|
|
42
42
|
## Next
|
|
@@ -177,7 +177,7 @@ span.metrics # what the plugins measured
|
|
|
177
177
|
`<path>/duration_ms`, `<path>/count` and one key per plugin metric to the open
|
|
178
178
|
row, and appends the full record to `spans.jsonl`.
|
|
179
179
|
|
|
180
|
-
`print_fn(line)` announces the span's start and end, indented
|
|
180
|
+
`print_fn(line)` announces the span's start and end, indented two spaces per level.
|
|
181
181
|
`plugins` measure a resource across the span; both are inherited by child spans
|
|
182
182
|
and default to the run's `span_print_fn` and `span_plugins`.
|
|
183
183
|
|
|
@@ -40,14 +40,12 @@ dynamic = ["version"]
|
|
|
40
40
|
|
|
41
41
|
[project.optional-dependencies]
|
|
42
42
|
# Local-first by default: remote backends and rich frames are opt-in.
|
|
43
|
-
lark = ["slark>=0.1.28"]
|
|
44
43
|
wandb = ["wandb>=0.21.0"]
|
|
45
44
|
trackio = ["trackio>=0.4.0"]
|
|
46
45
|
pandas = ["pandas>=1.5"]
|
|
47
46
|
polars = ["polars>=0.20"]
|
|
48
47
|
gpu = ["nvidia-ml-py>=12.0"]
|
|
49
48
|
all = [
|
|
50
|
-
"slark>=0.1.28",
|
|
51
49
|
"wandb>=0.21.0",
|
|
52
50
|
"trackio>=0.4.0",
|
|
53
51
|
"pandas>=1.5",
|
|
@@ -72,7 +70,6 @@ docs = [
|
|
|
72
70
|
]
|
|
73
71
|
dev = [
|
|
74
72
|
"ipykernel>=6.30.1",
|
|
75
|
-
"slark>=0.1.28",
|
|
76
73
|
"wandb>=0.21.0",
|
|
77
74
|
"trackio>=0.4.0",
|
|
78
75
|
"numpy>=1.24",
|
|
@@ -6,7 +6,7 @@ import json
|
|
|
6
6
|
import smtplib
|
|
7
7
|
from email.message import EmailMessage
|
|
8
8
|
|
|
9
|
-
from ..models import AlertLevel, AlertMessage
|
|
9
|
+
from ..models import AlertLevel, AlertMessage
|
|
10
10
|
from .base import (
|
|
11
11
|
AlertBackend,
|
|
12
12
|
SendError,
|
|
@@ -16,6 +16,7 @@ from .base import (
|
|
|
16
16
|
render_html,
|
|
17
17
|
render_text,
|
|
18
18
|
)
|
|
19
|
+
from .cards import build_card, card_payload
|
|
19
20
|
|
|
20
21
|
LEVEL_EMOJI = {
|
|
21
22
|
AlertLevel.DEBUG: "🔍",
|
|
@@ -56,13 +57,22 @@ class UrlBackend(AlertBackend):
|
|
|
56
57
|
self.url, payload, self.timeout, self.config.options.get("headers")
|
|
57
58
|
)
|
|
58
59
|
|
|
59
|
-
def
|
|
60
|
-
"""
|
|
61
|
-
|
|
60
|
+
def post_reply(self, payload: dict) -> dict:
|
|
61
|
+
"""Post, and decode the reply these APIs put their verdict in.
|
|
62
|
+
|
|
63
|
+
An unreadable body is not a failure: the POST itself succeeded, and some
|
|
64
|
+
proxies answer 200 with nothing at all.
|
|
65
|
+
"""
|
|
66
|
+
body = self.post(payload) # outside the try: a failed POST must raise
|
|
62
67
|
try:
|
|
63
68
|
result = json.loads(body)
|
|
64
69
|
except Exception:
|
|
65
|
-
return
|
|
70
|
+
return {}
|
|
71
|
+
return result if isinstance(result, dict) else {}
|
|
72
|
+
|
|
73
|
+
def post_checked(self, payload: dict):
|
|
74
|
+
"""DingTalk and WeCom report failures via errcode inside an HTTP 200 body."""
|
|
75
|
+
result = self.post_reply(payload)
|
|
66
76
|
code = result.get("errcode")
|
|
67
77
|
if code:
|
|
68
78
|
raise SendError(
|
|
@@ -73,45 +83,31 @@ class UrlBackend(AlertBackend):
|
|
|
73
83
|
|
|
74
84
|
|
|
75
85
|
class LarkBackend(UrlBackend):
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
def __init__(self, config: ChannelConfig):
|
|
79
|
-
super().__init__(config)
|
|
80
|
-
self._client = None
|
|
86
|
+
"""Feishu/Lark bot webhook, as an interactive card."""
|
|
81
87
|
|
|
82
|
-
|
|
83
|
-
if self._client is None:
|
|
84
|
-
from slark import Lark
|
|
88
|
+
type = "lark"
|
|
85
89
|
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
return self._client
|
|
90
|
+
# Lark answers HTTP 200 and puts the verdict in the body
|
|
91
|
+
RETRYABLE_CODES = frozenset({9499, 11232, 19024})
|
|
89
92
|
|
|
90
93
|
def send(self, message: AlertMessage):
|
|
91
94
|
title = f"{LEVEL_EMOJI.get(message.level, '')} {message.title}".strip()
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
95
|
+
card = build_card(
|
|
96
|
+
render_text(message),
|
|
97
|
+
title,
|
|
98
|
+
message.subtitle,
|
|
99
|
+
message.traceback,
|
|
100
|
+
failed=message.level >= AlertLevel.ERROR,
|
|
101
|
+
)
|
|
102
|
+
result = self.post_reply(card_payload(card))
|
|
103
|
+
# Success is code 0, or StatusCode 0 from the older bot endpoint
|
|
104
|
+
code = result.get("code") or result.get("StatusCode") or 0
|
|
105
|
+
if code:
|
|
106
|
+
detail = result.get("msg") or result.get("StatusMessage")
|
|
97
107
|
raise SendError(
|
|
98
|
-
f
|
|
99
|
-
retryable=
|
|
100
|
-
)
|
|
101
|
-
try:
|
|
102
|
-
if message.level >= AlertLevel.ERROR:
|
|
103
|
-
client.webhook.post_error_card(
|
|
104
|
-
msg=text,
|
|
105
|
-
traceback=message.traceback or "",
|
|
106
|
-
title=title,
|
|
107
|
-
subtitle=subtitle,
|
|
108
|
-
)
|
|
109
|
-
else:
|
|
110
|
-
client.webhook.post_success_card(
|
|
111
|
-
msg=text, title=title, subtitle=subtitle
|
|
112
|
-
)
|
|
113
|
-
except Exception as e:
|
|
114
|
-
raise SendError(f"Lark webhook failed: {e}") from e
|
|
108
|
+
f"Lark rejected the message: code={code} msg={detail!r}",
|
|
109
|
+
retryable=code in self.RETRYABLE_CODES,
|
|
110
|
+
)
|
|
115
111
|
|
|
116
112
|
|
|
117
113
|
class SlackBackend(UrlBackend):
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Lark (Feishu) interactive message cards.
|
|
2
|
+
|
|
3
|
+
The card layout is the one ``slark`` builds, reproduced here as plain dicts so
|
|
4
|
+
the Lark channel needs no third-party client: a webhook post is one JSON body
|
|
5
|
+
over stdlib HTTP, which is what every other channel already does.
|
|
6
|
+
|
|
7
|
+
Reference: https://open.feishu.cn/document/server-docs/im-v1/message-card
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import time
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
GREEN = "green"
|
|
16
|
+
RED = "red"
|
|
17
|
+
YES_ICON = "yes_filled"
|
|
18
|
+
ERROR_ICON = "error_filled"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _heading(content: str) -> dict[str, Any]:
|
|
22
|
+
"""A section heading. Lark has no heading element, hence the wrapping."""
|
|
23
|
+
return {
|
|
24
|
+
"tag": "column_set",
|
|
25
|
+
"flex_mode": "none",
|
|
26
|
+
"horizontal_spacing": "default",
|
|
27
|
+
"horizontal_align": "left",
|
|
28
|
+
"background_style": "default",
|
|
29
|
+
"columns": [
|
|
30
|
+
{
|
|
31
|
+
"tag": "column",
|
|
32
|
+
"background_style": "default",
|
|
33
|
+
"elements": [
|
|
34
|
+
{
|
|
35
|
+
"tag": "div",
|
|
36
|
+
"text": {
|
|
37
|
+
"tag": "plain_text",
|
|
38
|
+
"content": content,
|
|
39
|
+
"text_size": "heading",
|
|
40
|
+
"text_align": "left",
|
|
41
|
+
"text_color": "default",
|
|
42
|
+
},
|
|
43
|
+
}
|
|
44
|
+
],
|
|
45
|
+
"width": "auto",
|
|
46
|
+
"weight": 1,
|
|
47
|
+
"vertical_align": "top",
|
|
48
|
+
"vertical_spacing": "default",
|
|
49
|
+
}
|
|
50
|
+
],
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _code_block(content: str, language: str = "txt") -> dict[str, Any]:
|
|
55
|
+
return {
|
|
56
|
+
"tag": "markdown",
|
|
57
|
+
"content": f"```{language}\n{content}\n```",
|
|
58
|
+
"text_align": "left",
|
|
59
|
+
"text_size": "normal",
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _header(title: str, subtitle: str, template: str, icon: str) -> dict[str, Any]:
|
|
64
|
+
return {
|
|
65
|
+
"title": {"tag": "plain_text", "content": title},
|
|
66
|
+
"subtitle": {"tag": "plain_text", "content": subtitle},
|
|
67
|
+
"template": template,
|
|
68
|
+
"ud_icon": {"token": icon},
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def build_card(
|
|
73
|
+
message: str,
|
|
74
|
+
title: str,
|
|
75
|
+
subtitle: str | None = None,
|
|
76
|
+
traceback: str | None = None,
|
|
77
|
+
*,
|
|
78
|
+
failed: bool = False,
|
|
79
|
+
) -> dict[str, Any]:
|
|
80
|
+
"""One card: a coloured header, the message, and a traceback if there is one."""
|
|
81
|
+
elements: list[dict[str, Any]] = [
|
|
82
|
+
_heading("Message"),
|
|
83
|
+
_code_block(message),
|
|
84
|
+
]
|
|
85
|
+
if traceback and traceback.strip():
|
|
86
|
+
# Skipped when absent: most alerts are metric conditions, not crashes,
|
|
87
|
+
# and an empty "Traceback" code block on every one of them is noise
|
|
88
|
+
elements += [_heading("Traceback"), _code_block(traceback.strip(), "")]
|
|
89
|
+
return {
|
|
90
|
+
"header": _header(
|
|
91
|
+
title,
|
|
92
|
+
subtitle if subtitle is not None else time.strftime("%Y-%m-%d %H:%M:%S"),
|
|
93
|
+
RED if failed else GREEN,
|
|
94
|
+
ERROR_ICON if failed else YES_ICON,
|
|
95
|
+
),
|
|
96
|
+
"elements": elements,
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def card_payload(card: dict[str, Any]) -> dict[str, Any]:
|
|
101
|
+
"""The webhook envelope a card has to travel in."""
|
|
102
|
+
return {"msg_type": "interactive", "card": card}
|
|
@@ -26,6 +26,10 @@ _STACK: ContextVar[tuple[Span, ...]] = ContextVar("et_span_stack", default=())
|
|
|
26
26
|
DURATION_SUFFIX = "duration_ms"
|
|
27
27
|
COUNT_SUFFIX = "count"
|
|
28
28
|
RESERVED_METRICS = frozenset({DURATION_SUFFIX, COUNT_SUFFIX})
|
|
29
|
+
# Spaces, not a tab: a terminal measures tab stops from the start of the line, so
|
|
30
|
+
# behind a log prefix the first level collapses to whatever is left of the stop.
|
|
31
|
+
# Two spaces render the same width wherever the line begins.
|
|
32
|
+
INDENT = " "
|
|
29
33
|
|
|
30
34
|
|
|
31
35
|
_WARNED: set[tuple[str, str, str]] = set()
|
|
@@ -180,7 +184,7 @@ class Span:
|
|
|
180
184
|
def _announce(self, marker: str, duration_ms: float | None):
|
|
181
185
|
if self.print_fn is None:
|
|
182
186
|
return
|
|
183
|
-
indent =
|
|
187
|
+
indent = INDENT * self.print_depth
|
|
184
188
|
stamp = time.strftime("%H:%M:%S", time.localtime())
|
|
185
189
|
if duration_ms is None:
|
|
186
190
|
line = f"{indent}{marker} {self.name} {stamp}"
|
|
@@ -75,27 +75,30 @@ def test_custom_headers_are_forwarded(captured):
|
|
|
75
75
|
assert captured[0]["headers"] == {"X-Token": "t"}
|
|
76
76
|
|
|
77
77
|
|
|
78
|
-
def
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
78
|
+
def test_lark_sends_an_interactive_card(captured):
|
|
79
|
+
build("lark").send(message(level="info"))
|
|
80
|
+
payload = captured[0]["payload"]
|
|
81
|
+
assert payload["msg_type"] == "interactive"
|
|
82
|
+
assert payload["card"]["header"]["title"]["content"].endswith("Title")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@pytest.mark.parametrize(
|
|
86
|
+
("level", "template"),
|
|
87
|
+
[("info", "green"), ("warning", "green"), ("error", "red"), ("critical", "red")],
|
|
88
|
+
)
|
|
89
|
+
def test_the_lark_card_colour_follows_the_level(captured, level, template):
|
|
90
|
+
build("lark").send(message(level=level))
|
|
91
|
+
assert captured[0]["payload"]["card"]["header"]["template"] == template
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def test_a_lark_card_carries_the_traceback(captured):
|
|
95
|
+
build("lark").send(message(level="error", traceback="tb here"))
|
|
96
|
+
blocks = [
|
|
97
|
+
e["content"]
|
|
98
|
+
for e in captured[0]["payload"]["card"]["elements"]
|
|
99
|
+
if e["tag"] == "markdown"
|
|
100
|
+
]
|
|
101
|
+
assert any("tb here" in block for block in blocks)
|
|
99
102
|
|
|
100
103
|
|
|
101
104
|
def test_callable_backend_requires_handler():
|
|
@@ -456,3 +456,25 @@ def test_streams_and_ranks_both_reach_the_backend_name(rank, tmp_path):
|
|
|
456
456
|
call = backend.inits[0]
|
|
457
457
|
assert call["id"] == "job-data-rank3"
|
|
458
458
|
assert call["group"] == "job" and call["job_type"] == "data"
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def test_the_backend_id_does_not_follow_rank_aware(rank, tmp_path):
|
|
462
|
+
"""Local sharding and the remote id are separate concerns.
|
|
463
|
+
|
|
464
|
+
rank_aware=False promises you serialise writes to one file. Two processes'
|
|
465
|
+
wandb clients have no such lock, so a reporting rank still needs its own id.
|
|
466
|
+
"""
|
|
467
|
+
backend = RecordingBackend()
|
|
468
|
+
rank(2)
|
|
469
|
+
run = Run(
|
|
470
|
+
project="dist",
|
|
471
|
+
name="job",
|
|
472
|
+
dir=str(tmp_path),
|
|
473
|
+
backends=[backend],
|
|
474
|
+
backend_on_rank=None,
|
|
475
|
+
rank_aware=False,
|
|
476
|
+
)
|
|
477
|
+
run.log({"loss": 1.0})
|
|
478
|
+
run.finish()
|
|
479
|
+
assert backend.inits[0]["id"] == "job-rank2"
|
|
480
|
+
assert run.history.log_fp.name == "metrics.jsonl" # unsharded, as asked
|
|
@@ -342,7 +342,7 @@ def test_printing_spans_is_opt_in(profile_step, tmp_path, capsys):
|
|
|
342
342
|
assert "-> step" not in capsys.readouterr().out
|
|
343
343
|
profile_step.main([*argv(**{"--steps": 2, "--dir": tmp_path}), "--print-spans"])
|
|
344
344
|
printed = capsys.readouterr().out
|
|
345
|
-
assert "-> step" in printed and "\
|
|
345
|
+
assert "-> step" in printed and "\n -> data" in printed
|
|
346
346
|
|
|
347
347
|
|
|
348
348
|
# ------------------------------------------------------------------ early_stopping
|