pyattacker 0.3.0__tar.gz → 0.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyattacker-0.3.0 → pyattacker-0.3.1}/CHANGELOG.md +18 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/PKG-INFO +11 -5
- {pyattacker-0.3.0 → pyattacker-0.3.1}/README.md +10 -4
- {pyattacker-0.3.0 → pyattacker-0.3.1}/README.zh-CN.md +8 -2
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/cli.md +8 -2
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/design.md +10 -2
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/reference.md +43 -3
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/tutorial.md +5 -2
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/cli.md +8 -2
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/design.md +8 -2
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/reference.md +38 -2
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/tutorial.md +5 -2
- pyattacker-0.3.1/examples/live_metrics.py +72 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/pyproject.toml +1 -1
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/__init__.py +3 -1
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/errors.py +4 -1
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/monitor.py +27 -2
- pyattacker-0.3.1/src/pyattacker/reported_metrics.py +72 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/runner.py +48 -1
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/server.py +70 -9
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/base.py +7 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/memory.py +25 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/sqlite.py +55 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/writebehind.py +17 -1
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/task.py +12 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_monitor.py +1 -1
- pyattacker-0.3.1/tests/test_reported_metrics.py +193 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_server.py +14 -10
- {pyattacker-0.3.0 → pyattacker-0.3.1}/uv.lock +1 -1
- {pyattacker-0.3.0 → pyattacker-0.3.1}/.github/workflows/ci.yml +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/.github/workflows/release.yml +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/.gitignore +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/LICENSE +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/benchmark.md +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/releasing.md +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/benchmark.md +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/releasing.md +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/data.jsonl +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/README.md +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/README.zh-CN.md +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/__init__.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/backend.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/demo.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/pipelines.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/README.md +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/README.zh-CN.md +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pa_demo_plugin/__init__.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pa_demo_plugin/algorithms.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pa_demo_plugin/codecs.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pa_demo_plugin/tasks.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pyproject.toml +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_tasks.yaml +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/qa_eval.yaml +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/quickstart.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/sharded.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/__main__.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/algorithm.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/artifact.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/backends.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/__init__.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/clock.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/harness.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/metrics.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/provider.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/report.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/scenario.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/cli.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/declarative.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/export.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/handoff.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/history.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/merge.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/pipeline.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/plugins.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/resource.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/scheduler.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/shard.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/__init__.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/visits.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/tasks/__init__.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/helpers.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_algorithms.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_artifact.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_backends.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_backward.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_cli.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_clock.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_harness.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_provider.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_report.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_scenario.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_cli.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_declarative.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_docs_examples.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_docs_facts.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_docs_i18n.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_errors.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_export.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_handoff.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_lease_safety.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_m2.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_optional_yaml.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_packaging.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_pipeline.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_plugins.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_resume_cursor.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_resume_identity.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_retry_policy.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_runner.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_scheduler.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_shard.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_store.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_subprocess_lifecycle.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_task_overrides.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_tasks.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_tutorial.py +0 -0
- {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_worker_liveness.py +0 -0
|
@@ -4,6 +4,24 @@ All notable changes to this project are documented here. The format follows
|
|
|
4
4
|
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to
|
|
5
5
|
[Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
6
6
|
|
|
7
|
+
## [0.3.1] — 2026-09-19
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
* **Application-reported live metrics.** `Runner.report_metric()` and `TaskContext.report_metric()`
|
|
12
|
+
publish latest values scoped to a run or pipeline. Applications can observe committed pipeline
|
|
13
|
+
completions with `on_pipeline_finished(runner, record, artifact)` and compute their own accuracy or
|
|
14
|
+
other cross-pipeline measures. SQLite and in-memory stores persist the reports; `/metrics`, the HTML
|
|
15
|
+
dashboard, and terminal `watch` display them. See `examples/live_metrics.py` for a restart-aware
|
|
16
|
+
accuracy monitor. Metric calculation remains application-owned.
|
|
17
|
+
|
|
18
|
+
### Fixed
|
|
19
|
+
|
|
20
|
+
* **Consistent monitoring scope.** With no explicit run ID, the dashboard, JSON endpoints and terminal
|
|
21
|
+
`watch` all select the latest started run. The dashboard uses that same ID for statistics, metrics,
|
|
22
|
+
pipelines and events, including when only pipeline-scoped values were reported. `run_id=all` requests
|
|
23
|
+
aggregate operational statistics without mixing in a single run's application metrics.
|
|
24
|
+
|
|
7
25
|
## [0.3.0] — 2026-09-19
|
|
8
26
|
|
|
9
27
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pyattacker
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.1
|
|
4
4
|
Summary: Artifact-centric, resumable async task orchestration: pipeline / task / resource pool.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Hazer-BJTU/pyattacker
|
|
6
6
|
Project-URL: Repository, https://github.com/Hazer-BJTU/pyattacker
|
|
@@ -284,8 +284,8 @@ with Runner(store=":memory:", concurrency=4) as runner:
|
|
|
284
284
|
* **It is a durable checkpoint.** The jump is committed atomically (source task, attempt, entry artifact,
|
|
285
285
|
ledger row and cursor together), so a killed process resumes **at the target** with the recorded entry
|
|
286
286
|
state and does not re-run the source task. On a control-enabled pipeline `n_tasks_done` is a *position*,
|
|
287
|
-
not a progress count: skipped stations have no task rows. `
|
|
288
|
-
|
|
287
|
+
not a progress count: skipped stations have no task rows. `watch` counts commits in the selected
|
|
288
|
+
run by default (`--run-id all` selects all history); `/pipelines.handoffs` counts active execution records and `handoffs_historical` counts all ledger
|
|
289
289
|
records. A resume can use an earlier run's active handoff while recording no new handoffs. Pipeline
|
|
290
290
|
exports include ledger identity and the watermark to distinguish the scopes.
|
|
291
291
|
* **Opt-in and inert.** Without the `control` block nothing changes — not one row, not one counter, and not a
|
|
@@ -427,13 +427,15 @@ keeps everything in the store.
|
|
|
427
427
|
|
|
428
428
|
```bash
|
|
429
429
|
uv run pyattacker serve runs/qa.db # http://127.0.0.1:8787
|
|
430
|
-
# / dashboard /stats /events /pipelines /resources /errors (JSON)
|
|
430
|
+
# / dashboard /stats /metrics /events /pipelines /resources /errors (JSON)
|
|
431
431
|
```
|
|
432
432
|
|
|
433
433
|
It opens a fresh read-only connection per request, so it runs happily beside a live run. It has **no
|
|
434
434
|
authentication** and binds to loopback: there is no artifact route, but it does expose what your run recorded —
|
|
435
435
|
event `data` and stored error messages — so treat it as a debug view over your own data and do not put it on a
|
|
436
436
|
public interface without your own proxy in front.
|
|
437
|
+
Applications can publish live values with `runner.report_metric("accuracy", value, display="percent")`;
|
|
438
|
+
the framework displays them without calculating the metric. See [`examples/live_metrics.py`](examples/live_metrics.py).
|
|
437
439
|
|
|
438
440
|
**Branching inside a step** — `fanout(a, b)` runs several tasks on the same input concurrently and
|
|
439
441
|
returns `{task_name: value}`. Retry granularity becomes the group, which is the honest price of not
|
|
@@ -519,6 +521,11 @@ every known tradeoff with its reason.
|
|
|
519
521
|
|
|
520
522
|
## Status
|
|
521
523
|
|
|
524
|
+
**0.3.1 — application-reported live metrics.** Applications can publish values such as accuracy to the
|
|
525
|
+
read-only monitoring dashboard through `Runner.report_metric()` or `TaskContext.report_metric()`. A completion
|
|
526
|
+
callback receives committed pipeline results; the application owns the calculation. The dashboard and
|
|
527
|
+
`watch` now follow one run consistently by default. See [`examples/live_metrics.py`](examples/live_metrics.py).
|
|
528
|
+
|
|
522
529
|
**0.3.0 — advanced control flow (opt-in), and the documentation in Chinese.** Tasks can now
|
|
523
530
|
[hand off](#advanced-handoffs-opt-in): return a `Handoff` to skip declared stations or finish the pipeline
|
|
524
531
|
early, recorded in a durable ledger that recovery resumes from. It is opt-in and inert — a pipeline without a
|
|
@@ -579,4 +586,3 @@ extracted and executed by `tests/test_tutorial.py`, so documentation that rots f
|
|
|
579
586
|
## License
|
|
580
587
|
|
|
581
588
|
MIT — see [LICENSE](https://github.com/Hazer-BJTU/pyattacker/blob/main/LICENSE).
|
|
582
|
-
|
|
@@ -259,8 +259,8 @@ with Runner(store=":memory:", concurrency=4) as runner:
|
|
|
259
259
|
* **It is a durable checkpoint.** The jump is committed atomically (source task, attempt, entry artifact,
|
|
260
260
|
ledger row and cursor together), so a killed process resumes **at the target** with the recorded entry
|
|
261
261
|
state and does not re-run the source task. On a control-enabled pipeline `n_tasks_done` is a *position*,
|
|
262
|
-
not a progress count: skipped stations have no task rows. `
|
|
263
|
-
|
|
262
|
+
not a progress count: skipped stations have no task rows. `watch` counts commits in the selected
|
|
263
|
+
run by default (`--run-id all` selects all history); `/pipelines.handoffs` counts active execution records and `handoffs_historical` counts all ledger
|
|
264
264
|
records. A resume can use an earlier run's active handoff while recording no new handoffs. Pipeline
|
|
265
265
|
exports include ledger identity and the watermark to distinguish the scopes.
|
|
266
266
|
* **Opt-in and inert.** Without the `control` block nothing changes — not one row, not one counter, and not a
|
|
@@ -402,13 +402,15 @@ keeps everything in the store.
|
|
|
402
402
|
|
|
403
403
|
```bash
|
|
404
404
|
uv run pyattacker serve runs/qa.db # http://127.0.0.1:8787
|
|
405
|
-
# / dashboard /stats /events /pipelines /resources /errors (JSON)
|
|
405
|
+
# / dashboard /stats /metrics /events /pipelines /resources /errors (JSON)
|
|
406
406
|
```
|
|
407
407
|
|
|
408
408
|
It opens a fresh read-only connection per request, so it runs happily beside a live run. It has **no
|
|
409
409
|
authentication** and binds to loopback: there is no artifact route, but it does expose what your run recorded —
|
|
410
410
|
event `data` and stored error messages — so treat it as a debug view over your own data and do not put it on a
|
|
411
411
|
public interface without your own proxy in front.
|
|
412
|
+
Applications can publish live values with `runner.report_metric("accuracy", value, display="percent")`;
|
|
413
|
+
the framework displays them without calculating the metric. See [`examples/live_metrics.py`](examples/live_metrics.py).
|
|
412
414
|
|
|
413
415
|
**Branching inside a step** — `fanout(a, b)` runs several tasks on the same input concurrently and
|
|
414
416
|
returns `{task_name: value}`. Retry granularity becomes the group, which is the honest price of not
|
|
@@ -494,6 +496,11 @@ every known tradeoff with its reason.
|
|
|
494
496
|
|
|
495
497
|
## Status
|
|
496
498
|
|
|
499
|
+
**0.3.1 — application-reported live metrics.** Applications can publish values such as accuracy to the
|
|
500
|
+
read-only monitoring dashboard through `Runner.report_metric()` or `TaskContext.report_metric()`. A completion
|
|
501
|
+
callback receives committed pipeline results; the application owns the calculation. The dashboard and
|
|
502
|
+
`watch` now follow one run consistently by default. See [`examples/live_metrics.py`](examples/live_metrics.py).
|
|
503
|
+
|
|
497
504
|
**0.3.0 — advanced control flow (opt-in), and the documentation in Chinese.** Tasks can now
|
|
498
505
|
[hand off](#advanced-handoffs-opt-in): return a `Handoff` to skip declared stations or finish the pipeline
|
|
499
506
|
early, recorded in a durable ledger that recovery resumes from. It is opt-in and inert — a pipeline without a
|
|
@@ -554,4 +561,3 @@ extracted and executed by `tests/test_tutorial.py`, so documentation that rots f
|
|
|
554
561
|
## License
|
|
555
562
|
|
|
556
563
|
MIT — see [LICENSE](https://github.com/Hazer-BJTU/pyattacker/blob/main/LICENSE).
|
|
557
|
-
|
|
@@ -215,7 +215,7 @@ with Runner(store=":memory:", concurrency=4) as runner:
|
|
|
215
215
|
|
|
216
216
|
* **边是显式声明的,不是猜的。** 走了没声明的边,或者一个没有 `control` 块的流水线里出现了 `Handoff`,都是致命配置错误——不重试,也不会静默跳转。目标步骤必须在来源步骤之后;最后一步发起 `end` 会被拒绝(因为没意义)。
|
|
217
217
|
* **交接是一种处置结果,不是失败。** 它是个返回值,所以重试策略根本看不到它,任务里写的 `except Exception:` 也吞不掉它。租约的归还和成功时完全一样,被取消或超时的尝试压根走不到返回这一步。
|
|
218
|
-
* **它是持久化的检查点。** 跳转是原子提交的(源任务、尝试、入口产物、账本记录、游标一起提交),进程被杀了会在目标步骤恢复,带着记录好的入口状态,不会重跑源任务。开了控制流的流水线里,`n_tasks_done` 是个*位置*,不是进度计数——被跳过的步骤没有任务记录。`
|
|
218
|
+
* **它是持久化的检查点。** 跳转是原子提交的(源任务、尝试、入口产物、账本记录、游标一起提交),进程被杀了会在目标步骤恢复,带着记录好的入口状态,不会重跑源任务。开了控制流的流水线里,`n_tasks_done` 是个*位置*,不是进度计数——被跳过的步骤没有任务记录。`watch` 默认统计选定运行的提交数(`--run-id all` 查看全部历史);`/pipelines.handoffs` 统计活跃执行记录,`handoffs_historical` 统计全部账本记录。续跑时可以用之前某次运行的活跃交接,同时不记录新的交接。导出的流水线数据里会带上账本身份和水位标记,用来区分这些范围。
|
|
219
219
|
* **不开就完全没影响。** 没有 `control` 块的流水线,一行数据、一个计数器、`spec_digest` 的一个字节都不会变。
|
|
220
220
|
* **往回跳需要单独声明。** `Handoff.rewind(target, value)` 把作者指定的状态送回更早的步骤;`Handoff.retry_all()` 从最初的种子重新开始。需要声明 `control.rewind` / `control.retry_all`,还要设一个有限的 `control.max_handoffs`。可选的 `HistoryArtifact` 载荷可以做显式快照和恢复;普通字典的控制权完全在作者手里。访问记录(visits)和精确的产物出现记录会保留历史,保证恢复安全。API、预算和恢复边界见 [reference → 进阶:反向遍历](https://github.com/Hazer-BJTU/pyattacker/blob/main/docs/zh-CN/reference.md#进阶反向遍历rewindretry-allvisits),可运行的示例见 [tutorial 第 16–17 步](https://github.com/Hazer-BJTU/pyattacker/blob/main/docs/zh-CN/tutorial.md#第-16-步--高级用回退和全部重试重新生成)。
|
|
221
221
|
|
|
@@ -322,10 +322,12 @@ uv run pyattacker run -c examples/qa_eval.yaml --artifact-backend file:///data/b
|
|
|
322
322
|
|
|
323
323
|
```bash
|
|
324
324
|
uv run pyattacker serve runs/qa.db # http://127.0.0.1:8787
|
|
325
|
-
# / dashboard /stats /events /pipelines /resources /errors (JSON)
|
|
325
|
+
# / dashboard /stats /metrics /events /pipelines /resources /errors (JSON)
|
|
326
326
|
```
|
|
327
327
|
|
|
328
328
|
每个请求都新开一个只读连接,所以可以和正在跑的进程并存。**没有认证**,只绑回环地址——它没有 artifact 路由,但会暴露这次运行记下来的东西:事件的 `data`、存下来的错误信息。当调试工具用就好,别裸暴露到公网。
|
|
329
|
+
应用可用 `runner.report_metric("accuracy", value, display="percent")` 汇报实时指标;框架只展示,
|
|
330
|
+
不计算指标。参见 [`examples/live_metrics.py`](examples/live_metrics.py)。
|
|
329
331
|
|
|
330
332
|
**步骤内分支**——`fanout(a, b)` 拿同一份输入并发跑多个任务,返回 `{task_name: value}`。重试粒度变成整个组,这是不把流水线做成 DAG 的代价。组是 Runner 看到的唯一规格,所以子任务们对 `resource`、`algorithm`、`timeout_s` 达成一致时,这些参数从子任务继承(此时 `timeout_s` 限制整个组)。
|
|
331
333
|
|
|
@@ -389,6 +391,10 @@ uv run pyattacker run -c examples/qa_eval.yaml --limit 40
|
|
|
389
391
|
|
|
390
392
|
## 当前状态
|
|
391
393
|
|
|
394
|
+
**0.3.1——应用汇报的实时指标。** 应用可用 `Runner.report_metric()` 或 `TaskContext.report_metric()`
|
|
395
|
+
把准确率等数值汇报到只读监控面板。完成回调提供已提交的流水线结果,指标计算仍由应用负责。
|
|
396
|
+
面板和 `watch` 默认统一跟随同一个运行。参见 [`examples/live_metrics.py`](examples/live_metrics.py)。
|
|
397
|
+
|
|
392
398
|
**0.3.0——高级控制流(可选),以及中文文档。** 任务现在可以做[交接](README.zh-CN.md#进阶交接可选启用):返回一个 `Handoff` 跳过声明的后续步骤或提前收尾,记录在持久化账本里,续跑时从账本接着走。需要手动开启,不开完全没影响——没有 `control` 块的流水线不写新数据,`spec_digest` 逐字节不变。1.0 之前标记为实验性。反向遍历现在支持声明式回退和全量重试,带 visits 和可选的载荷历史。整套文档也提供了简体中文版([`README.zh-CN.md`](README.zh-CN.md)、[`docs/zh-CN/`](https://github.com/Hazer-BJTU/pyattacker/tree/main/docs/zh-CN)),由 CI 保持同步。升级时有一个改名要知道:`MergedReport.events_total` 现在叫 `source_events_total`,因为它是合并报告里**唯一**不去重、按源库原始累加的计数(旧名字仍可作为废弃别名使用)。
|
|
393
399
|
|
|
394
400
|
**0.2.0——基准测试、更严格的身份校验、三处正确性修复。** 新增 `pyattacker bench`:一个模拟接口方的世界,用一组指标对比各获取算法,而不是一个加权分数([`docs/benchmark.md`](https://github.com/Hazer-BJTU/pyattacker/blob/main/docs/zh-CN/benchmark.md))。新增 `Retrying.decide`:把重试决策暴露为策略上的方法。新增分页的整类读取(`pyattacker.store.iter_*`,由可选的 `PagedStore` 扩展提供),导出大存储不再需要全量加载。0.1.x 已经实现了 M0–M4 计划的全部内容:内核、持久化和任务级恢复、重试和错误分类、带 7 种获取算法的资源池、延迟延续和 write-behind 批处理、分片和合并报告、三种格式五种导出形状、入口点插件、外部产物后端、fan-out 辅助函数、HTTP 监控端点。
|
|
@@ -165,10 +165,13 @@ derives it from the event log, scoped to the pipelines in the report.
|
|
|
165
165
|
pyattacker watch STORE [--run-id ID] [--interval S] [--iterations N] [--no-clear]
|
|
166
166
|
```
|
|
167
167
|
|
|
168
|
+
Without `--run-id`, `watch` follows the latest started run. Use `--run-id all` for aggregate
|
|
169
|
+
operational statistics across the store; application-reported metrics are omitted in that view.
|
|
170
|
+
|
|
168
171
|
A read-only connection to the same SQLite file, so it runs beside a live run (WAL allows one writer and many
|
|
169
172
|
readers). Shows the pipeline state distribution, latency percentiles, per-pool `active/capacity`,
|
|
170
173
|
`ready/degraded/dead`, how many pipelines are waiting or parked, and recent errors. A run with handoffs also
|
|
171
|
-
shows `handoffs=N` (commits in the selected
|
|
174
|
+
shows `handoffs=N` (commits in the selected run, or all history with `--run-id all`; a resume may reuse earlier active handoffs) on its attempts line — on a control-enabled pipeline the cursor is a position, not a
|
|
172
175
|
progress count, so that number is what explains a short task list. `--iterations N` makes it
|
|
173
176
|
exit on its own, which is what you want in a script.
|
|
174
177
|
|
|
@@ -201,7 +204,10 @@ an export of a store that is still being written does and does not guarantee.
|
|
|
201
204
|
pyattacker serve STORE [--host HOST] [--port PORT] [--run-id ID]
|
|
202
205
|
```
|
|
203
206
|
|
|
204
|
-
|
|
207
|
+
Without `--run-id`, the dashboard follows the latest started run. Use `--run-id all` for an
|
|
208
|
+
aggregate view, or open `/?run_id=ID` to inspect a specific run.
|
|
209
|
+
|
|
210
|
+
`/` is a small auto-refreshing dashboard; `/stats`, `/metrics`, `/events`, `/pipelines`, `/resources`, `/errors` are JSON.
|
|
205
211
|
A fresh read-only connection per request means it is safe beside a live run.
|
|
206
212
|
|
|
207
213
|
**It has no authentication, and there is no artifact route.** What it exposes is what the run recorded —
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
**English** | [简体中文](zh-CN/design.md)
|
|
4
4
|
|
|
5
|
-
> Version: 0.3.
|
|
5
|
+
> Version: 0.3.1 — M0–M5 complete, including the opt-in advanced control flow of §4.8 (forward handoffs and
|
|
6
6
|
> backward traversal), which stays experimental until 1.0. See "Implemented / Left for later" in Section 9.
|
|
7
7
|
> In one sentence: **an async task orchestration framework centered on the artifact, using the pipeline as the unit of completion, and the resource pool as the only shared surface.**
|
|
8
8
|
> It does not touch the network, does not do reduction, and does not do DAG scheduling — it is only responsible for "running tens of thousands of mutually independent pipelines to completion, reliably, recoverably, and observably".
|
|
@@ -28,6 +28,11 @@ framework to step in:
|
|
|
28
28
|
2. Write a **sink pipeline**: use one task to publish the results onto some resource/bus, where subscribers
|
|
29
29
|
consume them — reduction becomes something you build yourself out of framework primitives.
|
|
30
30
|
|
|
31
|
+
An application can now also publish its own latest values through `Runner.report_metric()`. This is
|
|
32
|
+
transport and presentation of an already computed value; it does not change the reduction boundary.
|
|
33
|
+
The value is scoped to a run (optionally a pipeline), upserted synchronously, and read by the
|
|
34
|
+
read-only `/metrics` endpoint. The runner's store remains the sole writer.
|
|
35
|
+
|
|
31
36
|
---
|
|
32
37
|
|
|
33
38
|
## 2. Core Invariants
|
|
@@ -314,7 +319,7 @@ short, and with a fake clock the pump advances virtual time instantly (so tests
|
|
|
314
319
|
is not a checkpoint. A `SIGKILL` can therefore lose the last batch of history while every checkpoint stays
|
|
315
320
|
intact; `--no-write-behind` trades throughput for an immediate commit per attempt.
|
|
316
321
|
|
|
317
|
-
### 4.6 Monitoring:
|
|
322
|
+
### 4.6 Monitoring: Operational State and Application Reports
|
|
318
323
|
|
|
319
324
|
```python
|
|
320
325
|
snapshot = runner.stats() # live in-process snapshot
|
|
@@ -324,6 +329,9 @@ snapshot = runner.stats() # live in-process snapshot
|
|
|
324
329
|
The panel shows: pipeline state distribution, p95/max latency, per-task counts, and for each pool its
|
|
325
330
|
`active/capacity`, `ready/degraded/dead`, `waiting`, throughput and leak counters, and recent errors.
|
|
326
331
|
It is also usable programmatically: `monitor.render_snapshot(stats)` / `monitor.watch(store)`.
|
|
332
|
+
Application-reported values appear separately from operational counters. The application may
|
|
333
|
+
observe committed pipeline completions and publish accuracy, but it owns deduplication and recovery
|
|
334
|
+
of its accumulator from durable artifacts.
|
|
327
335
|
|
|
328
336
|
### 4.7 Completion Is Counted, Worker Lifetime Is Supervised
|
|
329
337
|
|
|
@@ -232,6 +232,7 @@ A selector matching **no** resource waits forever under the default `wait` algor
|
|
|
232
232
|
| `ctx.held_leases()` | leases this attempt currently holds |
|
|
233
233
|
| `ctx.reclaim_now()` | force-release everything held; synchronous, uninterruptible |
|
|
234
234
|
| `ctx.emit(kind, **data)` | write your own event into the run's event stream |
|
|
235
|
+
| `ctx.report_metric(name, value, *, label="", display="number")` | publish a latest value scoped to this pipeline |
|
|
235
236
|
|
|
236
237
|
```python
|
|
237
238
|
@task("adaptive", resource="apis")
|
|
@@ -739,6 +740,7 @@ the `with` block, or reopen the file afterwards with `open_store`.
|
|
|
739
740
|
| `run(specs, *, resume=False, **overrides)` | run to completion, returns a `RunReport`. Wraps `run_async` in `asyncio.run` |
|
|
740
741
|
| `await run_async(specs, *, resume=False, **overrides)` | same, inside an existing event loop |
|
|
741
742
|
| `stats()` | live snapshot; safe to call mid-run |
|
|
743
|
+
| `report_metric(name, value, *, label="", display="number", pipeline_id=None)` | publish an application-defined latest value |
|
|
742
744
|
| `stop(reason="user")` | ask the run to stop gracefully: stop admitting, drain what is in flight |
|
|
743
745
|
| `stopping` | whether a stop is in progress |
|
|
744
746
|
| `run_id` | the current run's id |
|
|
@@ -2165,6 +2167,45 @@ runner.run(template.map(jsonl_source("dataset.jsonl", limit=500)))
|
|
|
2165
2167
|
|
|
2166
2168
|
## Monitoring
|
|
2167
2169
|
|
|
2170
|
+
### Application-reported metrics
|
|
2171
|
+
|
|
2172
|
+
An application can publish its latest experiment values while a run is active. Pyattacker stores
|
|
2173
|
+
and displays the values; the application computes them. For a runnable accuracy example, see
|
|
2174
|
+
[`examples/live_metrics.py`](../examples/live_metrics.py).
|
|
2175
|
+
|
|
2176
|
+
```python
|
|
2177
|
+
runner.report_metric("evaluated", completed, label="Evaluated")
|
|
2178
|
+
runner.report_metric("accuracy", correct / completed, label="Accuracy", display="percent")
|
|
2179
|
+
# Inside a task, ctx.report_metric("phase", "scoring", display="text")
|
|
2180
|
+
```
|
|
2181
|
+
|
|
2182
|
+
`Runner(..., on_pipeline_finished=callback)` calls `callback(runner, record, artifact)` after a pipeline's
|
|
2183
|
+
terminal state is stored. `artifact` is its final `Artifact` for success and `None` for failure or
|
|
2184
|
+
interruption. The passed `runner` provides `report_metric()` and can decode a retained artifact with
|
|
2185
|
+
`runner.registry.load(artifact.encoded())`. The callback
|
|
2186
|
+
runs in the scheduler thread and should finish quickly. Exceptions in it are recorded as
|
|
2187
|
+
`monitor.callback_failed` and do not change the pipeline outcome. A process crash may miss or replay
|
|
2188
|
+
the callback, so applications should deduplicate by pipeline ID and rebuild from stored final
|
|
2189
|
+
artifacts when needed. Resumed runs have a new run ID and their reported values have a separate scope.
|
|
2190
|
+
|
|
2191
|
+
`report_metric(name, value, *, label="", display="number", pipeline_id=None)` accepts a string,
|
|
2192
|
+
boolean, or finite number. `display` is `number`, `percent` (a numeric fraction, displayed as a
|
|
2193
|
+
percentage), or `text` (a string). Repeating the same name in the same run and scope replaces the
|
|
2194
|
+
previous value. `ctx.report_metric(...)` uses the current pipeline as its scope. Reports are
|
|
2195
|
+
synchronous state writes, including when event write-behind is enabled. The feature is optional for
|
|
2196
|
+
third-party stores: reporting on a store without it raises `StoreFeatureUnsupported`.
|
|
2197
|
+
|
|
2198
|
+
`/metrics?run_id=...` returns `{"run_id": ..., "rows": [{"run_id", "pipeline_id", "name",
|
|
2199
|
+
"value", "label", "display", "updated_at"}, ...]}`. Add `pipeline_id=...` to read a pipeline's
|
|
2200
|
+
reported values; `/pipelines` also includes a `reported_metrics` array in each row. With no run ID,
|
|
2201
|
+
`read_snapshot()`, `watch`, and every `StatsServer` endpoint select the latest started run, including
|
|
2202
|
+
runs that reported only pipeline-scoped values. The HTML dashboard resolves that run through `/stats`
|
|
2203
|
+
and passes its ID to `/metrics`, `/pipelines`, and `/events` for a consistent refresh. Use
|
|
2204
|
+
`?run_id=all` (or `--run-id all` for `watch`/`serve`) for aggregate operational data; that view
|
|
2205
|
+
does not show application metrics from an arbitrarily chosen run. The HTML dashboard shows
|
|
2206
|
+
run-level values as cards; the terminal `watch` view shows them too. The HTTP server remains read-only,
|
|
2207
|
+
and metric writes use the active Runner's store connection.
|
|
2208
|
+
|
|
2168
2209
|
### `runner.stats()`
|
|
2169
2210
|
|
|
2170
2211
|
The in-process live snapshot. Safe to call mid-run.
|
|
@@ -2197,9 +2238,9 @@ with StatsServer("runs/qa.db", port=8787) as server:
|
|
|
2197
2238
|
| Endpoint | Returns |
|
|
2198
2239
|
|---|---|
|
|
2199
2240
|
| `/` | a small auto-refreshing dashboard |
|
|
2200
|
-
| `/stats`, `/events`, `/pipelines`, `/resources`, `/errors` | JSON |
|
|
2241
|
+
| `/stats`, `/metrics`, `/events`, `/pipelines`, `/resources`, `/errors` | JSON |
|
|
2201
2242
|
|
|
2202
|
-
`/stats` carries `handoffs_total` (commits by the selected run, or all commits
|
|
2243
|
+
`/stats` carries `handoffs_total` (commits by the selected run, or all commits with `run_id=all`), and each `/pipelines` row carries a
|
|
2203
2244
|
`handoffs` count of active-execution records, `handoffs_historical` count of all records and
|
|
2204
2245
|
`handoff_floor`, next to `n_tasks_done`/`n_tasks_total` — on a control-enabled pipeline those two are a
|
|
2205
2246
|
**position** in the chain, not a count of tasks that ran, so a non-zero `handoffs` is what says "this
|
|
@@ -2227,4 +2268,3 @@ command line.
|
|
|
2227
2268
|
| [`docs/cli.md`](cli.md) | commands, flags, exit codes, config file format |
|
|
2228
2269
|
| [`docs/design.md`](design.md) | the model, the invariants, and the tradeoffs behind these APIs |
|
|
2229
2270
|
| [`examples/`](../examples) | complete programs, including a measured comparison of pipeline shapes |
|
|
2230
|
-
|
|
@@ -1603,11 +1603,14 @@ blobs on disk: 3 files, [11, 27, 1536] bytes
|
|
|
1603
1603
|
[`examples/plugin_package/`](../examples/plugin_package/README.md).
|
|
1604
1604
|
|
|
1605
1605
|
**Monitoring a run in progress.** `pyattacker watch runs/qa.db` gives you a terminal view from a second
|
|
1606
|
-
process, and `pyattacker serve runs/qa.db` an HTTP dashboard plus JSON at `/stats`, `/events`, `/pipelines`,
|
|
1606
|
+
process, and `pyattacker serve runs/qa.db` an HTTP dashboard plus JSON at `/stats`, `/metrics`, `/events`, `/pipelines`,
|
|
1607
1607
|
`/resources` and `/errors`. Both open read-only connections, so they are safe beside a live run. The HTTP
|
|
1608
1608
|
endpoint has no authentication and serves your payloads — keep it on loopback.
|
|
1609
|
+
For live experiment accuracy, run [`examples/live_metrics.py`](../examples/live_metrics.py). Its
|
|
1610
|
+
completion callback decodes final artifacts, deduplicates by pipeline ID, and reports the current
|
|
1611
|
+
accuracy through `runner.report_metric()`; the dashboard displays the value as it changes.
|
|
1609
1612
|
* **Monitoring**: `pyattacker serve runs/qa.db` is a zero-dependency read-only HTTP view (`/`,
|
|
1610
|
-
`/stats`, `/events`, `/pipelines`, `/resources`, `/errors`) that opens a fresh connection per request,
|
|
1613
|
+
`/stats`, `/metrics`, `/events`, `/pipelines`, `/resources`, `/errors`) that opens a fresh connection per request,
|
|
1611
1614
|
so it runs happily beside a live run. It binds to loopback and has no authentication — treat it as a
|
|
1612
1615
|
debug view, not a dashboard.
|
|
1613
1616
|
|
|
@@ -128,7 +128,10 @@ pyattacker report STORE [STORE ...] [--run-id ID] [--errors N] [--json] [--artif
|
|
|
128
128
|
pyattacker watch STORE [--run-id ID] [--interval S] [--iterations N] [--no-clear]
|
|
129
129
|
```
|
|
130
130
|
|
|
131
|
-
|
|
131
|
+
不指定 `--run-id` 时,`watch` 跟随最近启动的运行。使用 `--run-id all` 可查看全库运行统计;
|
|
132
|
+
该视图不显示某个运行的应用汇报指标。
|
|
133
|
+
|
|
134
|
+
对同一个 SQLite 文件开只读连接,所以能和正在跑的流水线并存(WAL 允许一个写入者多个读取者)。显示流水线状态分布、延迟百分位、每个资源池的 `active/capacity`、`ready/degraded/dead`、多少流水线在等待或挂起、最近的错误。开了交接的运行还会在 attempts 行显示 `handoffs=N`(选定运行的提交数,使用 `--run-id all` 则是全部历史;续跑可能复用更早的活跃交接)——开了控制流的流水线,游标是位置不是进度计数,所以任务列表很短的时候就是这个数字在起作用。`--iterations N` 让它自己退出,脚本里用很方便。
|
|
132
135
|
|
|
133
136
|
## `export` —— 导出记录
|
|
134
137
|
|
|
@@ -154,7 +157,10 @@ pyattacker export STORE [STORE ...] OUTPUT [--rows SHAPE] [--format FMT] [--run-
|
|
|
154
157
|
pyattacker serve STORE [--host HOST] [--port PORT] [--run-id ID]
|
|
155
158
|
```
|
|
156
159
|
|
|
157
|
-
|
|
160
|
+
不指定 `--run-id` 时,面板跟随最近启动的运行。使用 `--run-id all` 可查看全库视图,
|
|
161
|
+
也可打开 `/?run_id=ID` 查看指定运行。
|
|
162
|
+
|
|
163
|
+
`/` 是个自动刷新的小仪表盘;`/stats`、`/metrics`、`/events`、`/pipelines`、`/resources`、`/errors` 返回 JSON。每个请求都新开一个只读连接,所以和正在跑的流水线并存是安全的。
|
|
158
164
|
|
|
159
165
|
**没有认证,也没有 artifact 路由。** 它暴露的是这次运行记下来的内容——事件的 `data` 原样呈现,存下来的 `error_message` 字段也在——所以默认只绑回环地址。要绑别的地址?前面自己加个代理。
|
|
160
166
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
[English](../design.md) | **简体中文**
|
|
4
4
|
|
|
5
|
-
> 版本:0.3.
|
|
5
|
+
> 版本:0.3.1——M0–M5 已完成,含 §4.8 那套可选启用的高级控制流(正向交接与反向遍历),1.0 之前标实验性。
|
|
6
6
|
> 见第 9 节"已实现 / 留待以后"。
|
|
7
7
|
> 一句话概括:**一个以 artifact 为中心的异步任务编排框架,用 pipeline 作为完成单位,把 resource pool 作为唯一的共享面。**
|
|
8
8
|
> 它不碰网络、不做归约、也不做 DAG 调度——只管一件事:"把几万条彼此独立的 pipeline 可靠、可恢复、可观测地跑到完成"。
|
|
@@ -24,6 +24,10 @@
|
|
|
24
24
|
1. 导出 artifact(`export` / `report.export_jsonl`),在外面自己算;
|
|
25
25
|
2. 写一条**汇聚流水线**(sink pipeline):用一个任务把结果发到某个资源/总线上,订阅者消费——归约就变成你用框架原语自己拼的东西。
|
|
26
26
|
|
|
27
|
+
应用还可以通过 `Runner.report_metric()` 汇报自己算好的最新值。这只是传输和展示,不改变
|
|
28
|
+
语义归约的边界。数值按运行(可选按流水线)划分作用域,同步覆盖写入,并由只读的
|
|
29
|
+
`/metrics` 端点读取。Runner 的存储连接仍是唯一写入者。
|
|
30
|
+
|
|
27
31
|
---
|
|
28
32
|
|
|
29
33
|
## 2. 核心不变量
|
|
@@ -233,7 +237,7 @@ REVOKED ◀── explicit revoke / revoked from within a task
|
|
|
233
237
|
|
|
234
238
|
**只追加的事实按批写。** `WriteBehindStore` 缓冲尝试和事件,按批刷(大小阈值、时间间隔、任何读 API、运行心跳、运行结束)。状态写入——`pipelines`、`tasks`、`artifacts`——永远直接落盘,因为还没落盘的检查点不算检查点。所以 `SIGKILL` 可能丢最后一批历史,所有检查点完好;`--no-write-behind` 用吞吐换每次尝试立即提交。
|
|
235
239
|
|
|
236
|
-
### 4.6
|
|
240
|
+
### 4.6 监控:运行状态与应用汇报
|
|
237
241
|
|
|
238
242
|
```python
|
|
239
243
|
snapshot = runner.stats() # live in-process snapshot
|
|
@@ -241,6 +245,8 @@ snapshot = runner.stats() # live in-process snapshot
|
|
|
241
245
|
```
|
|
242
246
|
|
|
243
247
|
面板显示:流水线状态分布、p95/最大延迟、各任务计数,每个资源池的 `active/capacity`、`ready/degraded/dead`、`waiting`、吞吐和泄漏计数器,最近的错误。也可以编程用:`monitor.render_snapshot(stats)` / `monitor.watch(store)`。
|
|
248
|
+
应用汇报值与运行计数分开展示。应用可以观察已提交的流水线终态并汇报准确率,但对流水线
|
|
249
|
+
去重、以及从持久化产物恢复累计结果都由应用负责。
|
|
244
250
|
|
|
245
251
|
### 4.7 完成以计数为准,worker 生命周期受监督
|
|
246
252
|
|
|
@@ -215,6 +215,7 @@ async with ctx.acquire(where=lambda r: r.options["ctx_len"] >= 32000) as lease:
|
|
|
215
215
|
| `ctx.held_leases()` | 这次尝试当前持有的租约 |
|
|
216
216
|
| `ctx.reclaim_now()` | 强制归还当前持有的全部租约;同步、不可中断 |
|
|
217
217
|
| `ctx.emit(kind, **data)` | 把你自己的事件写进这次运行的事件流 |
|
|
218
|
+
| `ctx.report_metric(name, value, *, label="", display="number")` | 汇报当前流水线作用域的最新值 |
|
|
218
219
|
|
|
219
220
|
```python
|
|
220
221
|
@task("adaptive", resource="apis")
|
|
@@ -677,6 +678,7 @@ with Runner(store="runs/qa.db", pools=[pool], concurrency=64) as runner:
|
|
|
677
678
|
| `run(specs, *, resume=False, **overrides)` | 跑至完成,返回 `RunReport`。在 `asyncio.run` 里包 `run_async` |
|
|
678
679
|
| `await run_async(specs, *, resume=False, **overrides)` | 同上,但在已有的事件循环里 |
|
|
679
680
|
| `stats()` | 实时快照;运行中途安全调 |
|
|
681
|
+
| `report_metric(name, value, *, label="", display="number", pipeline_id=None)` | 汇报应用计算的最新值 |
|
|
680
682
|
| `stop(reason="user")` | 请求运行优雅停:不再接新活,排空在途工作 |
|
|
681
683
|
| `stopping` | 是否在停 |
|
|
682
684
|
| `run_id` | 当前运行的 id |
|
|
@@ -2085,6 +2087,40 @@ runner.run(template.map(jsonl_source("dataset.jsonl", limit=500)))
|
|
|
2085
2087
|
|
|
2086
2088
|
## 监控
|
|
2087
2089
|
|
|
2090
|
+
### 应用汇报的指标
|
|
2091
|
+
|
|
2092
|
+
应用可以在运行期间汇报实验指标的最新值。Pyattacker 只负责保存和展示,指标由应用计算。
|
|
2093
|
+
可运行的准确率示例见 [`examples/live_metrics.py`](../../examples/live_metrics.py)。
|
|
2094
|
+
|
|
2095
|
+
```python
|
|
2096
|
+
runner.report_metric("evaluated", completed, label="Evaluated")
|
|
2097
|
+
runner.report_metric("accuracy", correct / completed, label="Accuracy", display="percent")
|
|
2098
|
+
# Inside a task, ctx.report_metric("phase", "scoring", display="text")
|
|
2099
|
+
```
|
|
2100
|
+
|
|
2101
|
+
`Runner(..., on_pipeline_finished=callback)` 在流水线终态持久化后调用
|
|
2102
|
+
`callback(runner, record, artifact)`。成功时传入最终 `Artifact`,失败或中断时传入 `None`。
|
|
2103
|
+
传入的 `runner` 可用于 `report_metric()`,也可以用
|
|
2104
|
+
`runner.registry.load(artifact.encoded())` 解码保存的产物。回调在调度线程中运行,
|
|
2105
|
+
应尽快返回。回调异常记录为 `monitor.callback_failed`,不会改变流水线结果。进程崩溃可能
|
|
2106
|
+
导致回调遗漏或重放,因此应用应按 pipeline ID 去重,并在需要时从持久化的最终产物重建
|
|
2107
|
+
统计。恢复运行使用新的 run ID,汇报值也属于新的作用域。
|
|
2108
|
+
|
|
2109
|
+
`report_metric(name, value, *, label="", display="number", pipeline_id=None)` 接受字符串、
|
|
2110
|
+
布尔值或有限数字。`display` 可为 `number`、`percent`(传入比例,展示为百分比)或
|
|
2111
|
+
`text`(字符串)。同一运行、同一作用域中的同名指标会覆盖旧值。`ctx.report_metric(...)`
|
|
2112
|
+
自动使用当前 pipeline 作为作用域。即使启用事件延迟写入,指标仍同步写入。第三方存储
|
|
2113
|
+
可以选择支持此功能;对不支持的存储汇报时抛出 `StoreFeatureUnsupported`。
|
|
2114
|
+
|
|
2115
|
+
`/metrics?run_id=...` 返回 `{"run_id": ..., "rows": [{"run_id", "pipeline_id", "name",
|
|
2116
|
+
"value", "label", "display", "updated_at"}, ...]}`。加上 `pipeline_id=...` 可读取该流水线的
|
|
2117
|
+
自定义值;`/pipelines` 的每行也包含 `reported_metrics` 数组。未指定 run ID 时,
|
|
2118
|
+
`read_snapshot()`、`watch` 和 `StatsServer` 的各端点统一选择最近启动的运行,包括只汇报
|
|
2119
|
+
流水线级值的运行。HTML 面板先从 `/stats` 确定 run ID,再用同一个 ID 查询 `/metrics`、
|
|
2120
|
+
`/pipelines` 和 `/events`。需要全库运行统计时可用 `?run_id=all`(`watch`/`serve` 使用
|
|
2121
|
+
`--run-id all`);全库视图不会任意挑一个运行的自定义指标来展示。HTML 面板将运行级值显示为卡片;终端
|
|
2122
|
+
`watch` 也会展示。HTTP 服务仍只读,指标通过正在运行的 Runner 的存储连接写入。
|
|
2123
|
+
|
|
2088
2124
|
### `runner.stats()`
|
|
2089
2125
|
|
|
2090
2126
|
进程内实时快照。运行中途安全调。
|
|
@@ -2117,9 +2153,9 @@ with StatsServer("runs/qa.db", port=8787) as server:
|
|
|
2117
2153
|
| 端点 | 返回 |
|
|
2118
2154
|
|---|---|
|
|
2119
2155
|
| `/` | 一个小巧的自动刷新仪表盘 |
|
|
2120
|
-
| `/stats`, `/events`, `/pipelines`, `/resources`, `/errors` | JSON |
|
|
2156
|
+
| `/stats`, `/metrics`, `/events`, `/pipelines`, `/resources`, `/errors` | JSON |
|
|
2121
2157
|
|
|
2122
|
-
`/stats` 带 `handoffs_total
|
|
2158
|
+
`/stats` 带 `handoffs_total`(所选运行的提交数,`run_id=all` 则为全部提交),每行 `/pipelines` 带
|
|
2123
2159
|
`handoffs`(活动执行记录的计数)、`handoffs_historical`(全部记录的计数)和
|
|
2124
2160
|
`handoff_floor`,紧挨着 `n_tasks_done`/`n_tasks_total`——开控制流的流水线上,这两者是链中的
|
|
2125
2161
|
**位置**,不是已跑任务数,所以非零的 `handoffs` 才说明"这条
|
|
@@ -1468,9 +1468,12 @@ blobs on disk: 3 files, [11, 27, 1536] bytes
|
|
|
1468
1468
|
* **工件后端**决定载荷字节放哪:`None`/`"inline"` 留在数据库里,`"null"` 丢掉字节但留摘要,`"file:///data/blobs"`(或 `{"kind": "file", "root": ..., "min_bytes": 262144}`)溢到内容寻址的文件里,读取时再水合回来。上面的 `min_bytes=0` 什么都溢出去,所以工件行显示的是引用而不是字节。
|
|
1469
1469
|
* **插件**是 `importlib.metadata` 入口点,分四组:`pyattacker.tasks`、`pyattacker.algorithms`、`pyattacker.codecs` 和 `pyattacker.stores`(按 URI scheme 索引,`store = "s3://bucket/runs.db"` 就能用)。内置项先解析,导入时报错的插件记下来不致命——`pyattacker plugins` 两者都列。完整示例包在 [`examples/plugin_package/`](../../examples/plugin_package/README.zh-CN.md)。
|
|
1470
1470
|
|
|
1471
|
-
**监控正在跑的运行。** `pyattacker watch runs/qa.db` 让你从第二个进程看终端视图,`pyattacker serve runs/qa.db` 提供 HTTP 仪表盘,在 `/stats`、`/events`、`/pipelines`、`/resources` 和 `/errors` 提供 JSON。两个都开只读连接,能和正在跑的任务并排用。HTTP 端点没认证,会把你的载荷给出来——只绑回环地址。
|
|
1471
|
+
**监控正在跑的运行。** `pyattacker watch runs/qa.db` 让你从第二个进程看终端视图,`pyattacker serve runs/qa.db` 提供 HTTP 仪表盘,在 `/stats`、`/metrics`、`/events`、`/pipelines`、`/resources` 和 `/errors` 提供 JSON。两个都开只读连接,能和正在跑的任务并排用。HTTP 端点没认证,会把你的载荷给出来——只绑回环地址。
|
|
1472
|
+
需要实时实验准确率时,运行 [`examples/live_metrics.py`](../../examples/live_metrics.py)。示例的
|
|
1473
|
+
完成回调解码最终产物、按 pipeline ID 去重,再通过 `runner.report_metric()` 汇报当前
|
|
1474
|
+
准确率;面板会随着汇报更新。
|
|
1472
1475
|
* **监控**:`pyattacker serve runs/qa.db` 是零依赖的只读 HTTP 视图(`/`、
|
|
1473
|
-
`/stats`、`/events`、`/pipelines`、`/resources`、`/errors`),每个请求开个新连接,和正在跑的任务并行无碍。绑回环地址、没认证——当调试视图用,别当仪表盘用。
|
|
1476
|
+
`/stats`、`/metrics`、`/events`、`/pipelines`、`/resources`、`/errors`),每个请求开个新连接,和正在跑的任务并行无碍。绑回环地址、没认证——当调试视图用,别当仪表盘用。
|
|
1474
1477
|
|
|
1475
1478
|
---
|
|
1476
1479
|
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Live, application-owned accuracy on the HTML dashboard.
|
|
2
|
+
|
|
3
|
+
Run ``uv run python examples/live_metrics.py`` and open the printed URL while it runs.
|
|
4
|
+
The application reads final artifacts, owns the accumulator, and only reports its
|
|
5
|
+
latest values to pyattacker. The database can be reopened to rebuild the accumulator.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import asyncio
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from pyattacker import Runner, SqliteStore, StatsServer, pipeline, task
|
|
14
|
+
from pyattacker.artifact import DEFAULT_REGISTRY
|
|
15
|
+
from pyattacker.store import iter_pipelines
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@task("live.judge")
|
|
19
|
+
async def judge(seed: dict) -> dict:
|
|
20
|
+
await asyncio.sleep(0.1)
|
|
21
|
+
return {"correct": seed["answer"] == seed["expected"]}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def rebuild(db: str) -> dict[str, bool]:
|
|
25
|
+
"""Rebuild application state from durable final artifacts after a restart."""
|
|
26
|
+
results: dict[str, bool] = {}
|
|
27
|
+
store = SqliteStore(db, read_only=True)
|
|
28
|
+
try:
|
|
29
|
+
for record in iter_pipelines(store, state="succeeded"):
|
|
30
|
+
artifact = next((item for item in store.artifacts(record.pipeline_id) if item.is_final), None)
|
|
31
|
+
if artifact is not None and artifact.available:
|
|
32
|
+
results[record.pipeline_id] = bool(DEFAULT_REGISTRY.load(artifact.encoded())["correct"])
|
|
33
|
+
finally:
|
|
34
|
+
store.close()
|
|
35
|
+
return results
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class AccuracyMonitor:
|
|
39
|
+
"""Own the cross-pipeline reduction; Runner supplies committed completions."""
|
|
40
|
+
|
|
41
|
+
def __init__(self, results: dict[str, bool]) -> None:
|
|
42
|
+
self.results = results
|
|
43
|
+
|
|
44
|
+
def __call__(self, runner: Runner, record, artifact) -> None:
|
|
45
|
+
if artifact is not None and artifact.available:
|
|
46
|
+
self.results[record.pipeline_id] = bool(runner.registry.load(artifact.encoded())["correct"])
|
|
47
|
+
elif artifact is None:
|
|
48
|
+
self.results.pop(record.pipeline_id, None)
|
|
49
|
+
# A failed pipeline does not enter this example's accuracy denominator.
|
|
50
|
+
evaluated = len(self.results)
|
|
51
|
+
correct = sum(self.results.values())
|
|
52
|
+
runner.report_metric("evaluated", evaluated, label="Evaluated")
|
|
53
|
+
runner.report_metric("correct", correct, label="Correct")
|
|
54
|
+
if evaluated:
|
|
55
|
+
runner.report_metric("accuracy", correct / evaluated,
|
|
56
|
+
label="Accuracy", display="percent")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def main() -> None:
|
|
60
|
+
db = "runs/live_metrics.db"
|
|
61
|
+
Path(db).parent.mkdir(exist_ok=True)
|
|
62
|
+
monitor = AccuracyMonitor(rebuild(db) if Path(db).exists() else {})
|
|
63
|
+
|
|
64
|
+
with Runner(store=db, on_pipeline_finished=monitor, retry_succeeded=True,
|
|
65
|
+
handle_signals=False) as runner, StatsServer(db, port=0) as server:
|
|
66
|
+
print(f"Monitor: {server.url}")
|
|
67
|
+
rows = ({"answer": i, "expected": i if i % 3 else i + 1} for i in range(100))
|
|
68
|
+
runner.run(pipeline("live-eval", judge).map(rows))
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
if __name__ == "__main__":
|
|
72
|
+
main()
|
|
@@ -77,6 +77,7 @@ from .history import HistoryArtifact
|
|
|
77
77
|
from .merge import MergedReport, merge_reports
|
|
78
78
|
from .pipeline import Chain, PipelineSpec, PipelineTemplate, compute_spec_digest, pipeline, with_retry
|
|
79
79
|
from .plugins import PLUGINS, PluginRegistry, list_plugins
|
|
80
|
+
from .reported_metrics import ReportedMetric
|
|
80
81
|
from .resource import Bus, Lease, Pool, Resource, ResourceEvent, ResourceState
|
|
81
82
|
from .runner import RunConfig, Runner, RunReport
|
|
82
83
|
from .server import StatsServer
|
|
@@ -105,7 +106,7 @@ from .tasks import (
|
|
|
105
106
|
write_jsonl,
|
|
106
107
|
)
|
|
107
108
|
|
|
108
|
-
__version__ = "0.3.
|
|
109
|
+
__version__ = "0.3.1"
|
|
109
110
|
|
|
110
111
|
__all__ = [
|
|
111
112
|
"__version__",
|
|
@@ -134,6 +135,7 @@ __all__ = [
|
|
|
134
135
|
"Handoff",
|
|
135
136
|
"HistoryArtifact",
|
|
136
137
|
"Artifact",
|
|
138
|
+
"ReportedMetric",
|
|
137
139
|
"Codec",
|
|
138
140
|
"CodecRegistry",
|
|
139
141
|
"JsonCodec",
|
|
@@ -143,7 +143,7 @@ class StoreUnavailable(PyAttackerError):
|
|
|
143
143
|
|
|
144
144
|
|
|
145
145
|
class StoreFeatureUnsupported(PyAttackerError):
|
|
146
|
-
"""The store
|
|
146
|
+
"""The store lacks a requested optional feature, or uses a newer feature level.
|
|
147
147
|
|
|
148
148
|
A store records the highest on-disk feature level it has reached (``store/visits.py``), and a
|
|
149
149
|
binary that does not know that level refuses to operate on it: reading or writing a
|
|
@@ -151,6 +151,9 @@ class StoreFeatureUnsupported(PyAttackerError):
|
|
|
151
151
|
occurrence, not merely display incomplete data. Upgrade the package to open the store; back it
|
|
152
152
|
up with SQLite's own backup API (``sqlite3 .backup``) or a file copy taken while no writer is
|
|
153
153
|
active (see ``docs/reference.md``, "Advanced: backward traversal").
|
|
154
|
+
|
|
155
|
+
An older third-party store also raises this error when a caller requests an optional
|
|
156
|
+
capability such as application-reported metrics that it does not implement.
|
|
154
157
|
"""
|
|
155
158
|
|
|
156
159
|
|