pyattacker 0.3.0__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. {pyattacker-0.3.0 → pyattacker-0.3.1}/CHANGELOG.md +18 -0
  2. {pyattacker-0.3.0 → pyattacker-0.3.1}/PKG-INFO +11 -5
  3. {pyattacker-0.3.0 → pyattacker-0.3.1}/README.md +10 -4
  4. {pyattacker-0.3.0 → pyattacker-0.3.1}/README.zh-CN.md +8 -2
  5. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/cli.md +8 -2
  6. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/design.md +10 -2
  7. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/reference.md +43 -3
  8. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/tutorial.md +5 -2
  9. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/cli.md +8 -2
  10. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/design.md +8 -2
  11. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/reference.md +38 -2
  12. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/tutorial.md +5 -2
  13. pyattacker-0.3.1/examples/live_metrics.py +72 -0
  14. {pyattacker-0.3.0 → pyattacker-0.3.1}/pyproject.toml +1 -1
  15. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/__init__.py +3 -1
  16. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/errors.py +4 -1
  17. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/monitor.py +27 -2
  18. pyattacker-0.3.1/src/pyattacker/reported_metrics.py +72 -0
  19. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/runner.py +48 -1
  20. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/server.py +70 -9
  21. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/base.py +7 -0
  22. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/memory.py +25 -0
  23. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/sqlite.py +55 -0
  24. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/writebehind.py +17 -1
  25. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/task.py +12 -0
  26. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_monitor.py +1 -1
  27. pyattacker-0.3.1/tests/test_reported_metrics.py +193 -0
  28. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_server.py +14 -10
  29. {pyattacker-0.3.0 → pyattacker-0.3.1}/uv.lock +1 -1
  30. {pyattacker-0.3.0 → pyattacker-0.3.1}/.github/workflows/ci.yml +0 -0
  31. {pyattacker-0.3.0 → pyattacker-0.3.1}/.github/workflows/release.yml +0 -0
  32. {pyattacker-0.3.0 → pyattacker-0.3.1}/.gitignore +0 -0
  33. {pyattacker-0.3.0 → pyattacker-0.3.1}/LICENSE +0 -0
  34. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/benchmark.md +0 -0
  35. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/releasing.md +0 -0
  36. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/benchmark.md +0 -0
  37. {pyattacker-0.3.0 → pyattacker-0.3.1}/docs/zh-CN/releasing.md +0 -0
  38. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/data.jsonl +0 -0
  39. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/README.md +0 -0
  40. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/README.zh-CN.md +0 -0
  41. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/__init__.py +0 -0
  42. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/backend.py +0 -0
  43. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/demo.py +0 -0
  44. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/llm_eval/pipelines.py +0 -0
  45. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/README.md +0 -0
  46. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/README.zh-CN.md +0 -0
  47. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pa_demo_plugin/__init__.py +0 -0
  48. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pa_demo_plugin/algorithms.py +0 -0
  49. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pa_demo_plugin/codecs.py +0 -0
  50. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pa_demo_plugin/tasks.py +0 -0
  51. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_package/pyproject.toml +0 -0
  52. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/plugin_tasks.yaml +0 -0
  53. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/qa_eval.yaml +0 -0
  54. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/quickstart.py +0 -0
  55. {pyattacker-0.3.0 → pyattacker-0.3.1}/examples/sharded.py +0 -0
  56. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/__main__.py +0 -0
  57. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/algorithm.py +0 -0
  58. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/artifact.py +0 -0
  59. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/backends.py +0 -0
  60. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/__init__.py +0 -0
  61. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/clock.py +0 -0
  62. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/harness.py +0 -0
  63. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/metrics.py +0 -0
  64. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/provider.py +0 -0
  65. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/report.py +0 -0
  66. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/benchmark/scenario.py +0 -0
  67. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/cli.py +0 -0
  68. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/declarative.py +0 -0
  69. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/export.py +0 -0
  70. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/handoff.py +0 -0
  71. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/history.py +0 -0
  72. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/merge.py +0 -0
  73. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/pipeline.py +0 -0
  74. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/plugins.py +0 -0
  75. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/resource.py +0 -0
  76. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/scheduler.py +0 -0
  77. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/shard.py +0 -0
  78. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/__init__.py +0 -0
  79. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/store/visits.py +0 -0
  80. {pyattacker-0.3.0 → pyattacker-0.3.1}/src/pyattacker/tasks/__init__.py +0 -0
  81. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/helpers.py +0 -0
  82. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_algorithms.py +0 -0
  83. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_artifact.py +0 -0
  84. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_backends.py +0 -0
  85. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_backward.py +0 -0
  86. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_cli.py +0 -0
  87. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_clock.py +0 -0
  88. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_harness.py +0 -0
  89. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_provider.py +0 -0
  90. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_report.py +0 -0
  91. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_benchmark_scenario.py +0 -0
  92. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_cli.py +0 -0
  93. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_declarative.py +0 -0
  94. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_docs_examples.py +0 -0
  95. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_docs_facts.py +0 -0
  96. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_docs_i18n.py +0 -0
  97. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_errors.py +0 -0
  98. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_export.py +0 -0
  99. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_handoff.py +0 -0
  100. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_lease_safety.py +0 -0
  101. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_m2.py +0 -0
  102. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_optional_yaml.py +0 -0
  103. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_packaging.py +0 -0
  104. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_pipeline.py +0 -0
  105. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_plugins.py +0 -0
  106. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_resume_cursor.py +0 -0
  107. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_resume_identity.py +0 -0
  108. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_retry_policy.py +0 -0
  109. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_runner.py +0 -0
  110. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_scheduler.py +0 -0
  111. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_shard.py +0 -0
  112. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_store.py +0 -0
  113. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_subprocess_lifecycle.py +0 -0
  114. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_task_overrides.py +0 -0
  115. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_tasks.py +0 -0
  116. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_tutorial.py +0 -0
  117. {pyattacker-0.3.0 → pyattacker-0.3.1}/tests/test_worker_liveness.py +0 -0
@@ -4,6 +4,24 @@ All notable changes to this project are documented here. The format follows
4
4
  [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to
5
5
  [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
6
6
 
7
+ ## [0.3.1] — 2026-09-19
8
+
9
+ ### Added
10
+
11
+ * **Application-reported live metrics.** `Runner.report_metric()` and `TaskContext.report_metric()`
12
+ publish latest values scoped to a run or pipeline. Applications can observe committed pipeline
13
+ completions with `on_pipeline_finished(runner, record, artifact)` and compute their own accuracy or
14
+ other cross-pipeline measures. SQLite and in-memory stores persist the reports; `/metrics`, the HTML
15
+ dashboard, and terminal `watch` display them. See `examples/live_metrics.py` for a restart-aware
16
+ accuracy monitor. Metric calculation remains application-owned.
17
+
18
+ ### Fixed
19
+
20
+ * **Consistent monitoring scope.** With no explicit run ID, the dashboard, JSON endpoints and terminal
21
+ `watch` all select the latest started run. The dashboard uses that same ID for statistics, metrics,
22
+ pipelines and events, including when only pipeline-scoped values were reported. `run_id=all` requests
23
+ aggregate operational statistics without mixing in a single run's application metrics.
24
+
7
25
  ## [0.3.0] — 2026-09-19
8
26
 
9
27
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pyattacker
3
- Version: 0.3.0
3
+ Version: 0.3.1
4
4
  Summary: Artifact-centric, resumable async task orchestration: pipeline / task / resource pool.
5
5
  Project-URL: Homepage, https://github.com/Hazer-BJTU/pyattacker
6
6
  Project-URL: Repository, https://github.com/Hazer-BJTU/pyattacker
@@ -284,8 +284,8 @@ with Runner(store=":memory:", concurrency=4) as runner:
284
284
  * **It is a durable checkpoint.** The jump is committed atomically (source task, attempt, entry artifact,
285
285
  ledger row and cursor together), so a killed process resumes **at the target** with the recorded entry
286
286
  state and does not re-run the source task. On a control-enabled pipeline `n_tasks_done` is a *position*,
287
- not a progress count: skipped stations have no task rows. `report`/`watch` count commits in the selected
288
- scope (one run when filtered, all history otherwise); `/pipelines.handoffs` counts active execution records and `handoffs_historical` counts all ledger
287
+ not a progress count: skipped stations have no task rows. `watch` counts commits in the selected
288
+ run by default (`--run-id all` selects all history); `/pipelines.handoffs` counts active execution records and `handoffs_historical` counts all ledger
289
289
  records. A resume can use an earlier run's active handoff while recording no new handoffs. Pipeline
290
290
  exports include ledger identity and the watermark to distinguish the scopes.
291
291
  * **Opt-in and inert.** Without the `control` block nothing changes — not one row, not one counter, and not a
@@ -427,13 +427,15 @@ keeps everything in the store.
427
427
 
428
428
  ```bash
429
429
  uv run pyattacker serve runs/qa.db # http://127.0.0.1:8787
430
- # / dashboard /stats /events /pipelines /resources /errors (JSON)
430
+ # / dashboard /stats /metrics /events /pipelines /resources /errors (JSON)
431
431
  ```
432
432
 
433
433
  It opens a fresh read-only connection per request, so it runs happily beside a live run. It has **no
434
434
  authentication** and binds to loopback: there is no artifact route, but it does expose what your run recorded —
435
435
  event `data` and stored error messages — so treat it as a debug view over your own data and do not put it on a
436
436
  public interface without your own proxy in front.
437
+ Applications can publish live values with `runner.report_metric("accuracy", value, display="percent")`;
438
+ the framework displays them without calculating the metric. See [`examples/live_metrics.py`](examples/live_metrics.py).
437
439
 
438
440
  **Branching inside a step** — `fanout(a, b)` runs several tasks on the same input concurrently and
439
441
  returns `{task_name: value}`. Retry granularity becomes the group, which is the honest price of not
@@ -519,6 +521,11 @@ every known tradeoff with its reason.
519
521
 
520
522
  ## Status
521
523
 
524
+ **0.3.1 — application-reported live metrics.** Applications can publish values such as accuracy to the
525
+ read-only monitoring dashboard through `Runner.report_metric()` or `TaskContext.report_metric()`. A completion
526
+ callback receives committed pipeline results; the application owns the calculation. The dashboard and
527
+ `watch` now follow one run consistently by default. See [`examples/live_metrics.py`](examples/live_metrics.py).
528
+
522
529
  **0.3.0 — advanced control flow (opt-in), and the documentation in Chinese.** Tasks can now
523
530
  [hand off](#advanced-handoffs-opt-in): return a `Handoff` to skip declared stations or finish the pipeline
524
531
  early, recorded in a durable ledger that recovery resumes from. It is opt-in and inert — a pipeline without a
@@ -579,4 +586,3 @@ extracted and executed by `tests/test_tutorial.py`, so documentation that rots f
579
586
  ## License
580
587
 
581
588
  MIT — see [LICENSE](https://github.com/Hazer-BJTU/pyattacker/blob/main/LICENSE).
582
-
@@ -259,8 +259,8 @@ with Runner(store=":memory:", concurrency=4) as runner:
259
259
  * **It is a durable checkpoint.** The jump is committed atomically (source task, attempt, entry artifact,
260
260
  ledger row and cursor together), so a killed process resumes **at the target** with the recorded entry
261
261
  state and does not re-run the source task. On a control-enabled pipeline `n_tasks_done` is a *position*,
262
- not a progress count: skipped stations have no task rows. `report`/`watch` count commits in the selected
263
- scope (one run when filtered, all history otherwise); `/pipelines.handoffs` counts active execution records and `handoffs_historical` counts all ledger
262
+ not a progress count: skipped stations have no task rows. `watch` counts commits in the selected
263
+ run by default (`--run-id all` selects all history); `/pipelines.handoffs` counts active execution records and `handoffs_historical` counts all ledger
264
264
  records. A resume can use an earlier run's active handoff while recording no new handoffs. Pipeline
265
265
  exports include ledger identity and the watermark to distinguish the scopes.
266
266
  * **Opt-in and inert.** Without the `control` block nothing changes — not one row, not one counter, and not a
@@ -402,13 +402,15 @@ keeps everything in the store.
402
402
 
403
403
  ```bash
404
404
  uv run pyattacker serve runs/qa.db # http://127.0.0.1:8787
405
- # / dashboard /stats /events /pipelines /resources /errors (JSON)
405
+ # / dashboard /stats /metrics /events /pipelines /resources /errors (JSON)
406
406
  ```
407
407
 
408
408
  It opens a fresh read-only connection per request, so it runs happily beside a live run. It has **no
409
409
  authentication** and binds to loopback: there is no artifact route, but it does expose what your run recorded —
410
410
  event `data` and stored error messages — so treat it as a debug view over your own data and do not put it on a
411
411
  public interface without your own proxy in front.
412
+ Applications can publish live values with `runner.report_metric("accuracy", value, display="percent")`;
413
+ the framework displays them without calculating the metric. See [`examples/live_metrics.py`](examples/live_metrics.py).
412
414
 
413
415
  **Branching inside a step** — `fanout(a, b)` runs several tasks on the same input concurrently and
414
416
  returns `{task_name: value}`. Retry granularity becomes the group, which is the honest price of not
@@ -494,6 +496,11 @@ every known tradeoff with its reason.
494
496
 
495
497
  ## Status
496
498
 
499
+ **0.3.1 — application-reported live metrics.** Applications can publish values such as accuracy to the
500
+ read-only monitoring dashboard through `Runner.report_metric()` or `TaskContext.report_metric()`. A completion
501
+ callback receives committed pipeline results; the application owns the calculation. The dashboard and
502
+ `watch` now follow one run consistently by default. See [`examples/live_metrics.py`](examples/live_metrics.py).
503
+
497
504
  **0.3.0 — advanced control flow (opt-in), and the documentation in Chinese.** Tasks can now
498
505
  [hand off](#advanced-handoffs-opt-in): return a `Handoff` to skip declared stations or finish the pipeline
499
506
  early, recorded in a durable ledger that recovery resumes from. It is opt-in and inert — a pipeline without a
@@ -554,4 +561,3 @@ extracted and executed by `tests/test_tutorial.py`, so documentation that rots f
554
561
  ## License
555
562
 
556
563
  MIT — see [LICENSE](https://github.com/Hazer-BJTU/pyattacker/blob/main/LICENSE).
557
-
@@ -215,7 +215,7 @@ with Runner(store=":memory:", concurrency=4) as runner:
215
215
 
216
216
  * **边是显式声明的,不是猜的。** 走了没声明的边,或者一个没有 `control` 块的流水线里出现了 `Handoff`,都是致命配置错误——不重试,也不会静默跳转。目标步骤必须在来源步骤之后;最后一步发起 `end` 会被拒绝(因为没意义)。
217
217
  * **交接是一种处置结果,不是失败。** 它是个返回值,所以重试策略根本看不到它,任务里写的 `except Exception:` 也吞不掉它。租约的归还和成功时完全一样,被取消或超时的尝试压根走不到返回这一步。
218
- * **它是持久化的检查点。** 跳转是原子提交的(源任务、尝试、入口产物、账本记录、游标一起提交),进程被杀了会在目标步骤恢复,带着记录好的入口状态,不会重跑源任务。开了控制流的流水线里,`n_tasks_done` 是个*位置*,不是进度计数——被跳过的步骤没有任务记录。`report`/`watch` 统计选定范围内的提交数(过滤时是单次运行,否则是全部历史);`/pipelines.handoffs` 统计活跃执行记录,`handoffs_historical` 统计全部账本记录。续跑时可以用之前某次运行的活跃交接,同时不记录新的交接。导出的流水线数据里会带上账本身份和水位标记,用来区分这些范围。
218
+ * **它是持久化的检查点。** 跳转是原子提交的(源任务、尝试、入口产物、账本记录、游标一起提交),进程被杀了会在目标步骤恢复,带着记录好的入口状态,不会重跑源任务。开了控制流的流水线里,`n_tasks_done` 是个*位置*,不是进度计数——被跳过的步骤没有任务记录。`watch` 默认统计选定运行的提交数(`--run-id all` 查看全部历史);`/pipelines.handoffs` 统计活跃执行记录,`handoffs_historical` 统计全部账本记录。续跑时可以用之前某次运行的活跃交接,同时不记录新的交接。导出的流水线数据里会带上账本身份和水位标记,用来区分这些范围。
219
219
  * **不开就完全没影响。** 没有 `control` 块的流水线,一行数据、一个计数器、`spec_digest` 的一个字节都不会变。
220
220
  * **往回跳需要单独声明。** `Handoff.rewind(target, value)` 把作者指定的状态送回更早的步骤;`Handoff.retry_all()` 从最初的种子重新开始。需要声明 `control.rewind` / `control.retry_all`,还要设一个有限的 `control.max_handoffs`。可选的 `HistoryArtifact` 载荷可以做显式快照和恢复;普通字典的控制权完全在作者手里。访问记录(visits)和精确的产物出现记录会保留历史,保证恢复安全。API、预算和恢复边界见 [reference → 进阶:反向遍历](https://github.com/Hazer-BJTU/pyattacker/blob/main/docs/zh-CN/reference.md#进阶反向遍历rewindretry-allvisits),可运行的示例见 [tutorial 第 16–17 步](https://github.com/Hazer-BJTU/pyattacker/blob/main/docs/zh-CN/tutorial.md#第-16-步--高级用回退和全部重试重新生成)。
221
221
 
@@ -322,10 +322,12 @@ uv run pyattacker run -c examples/qa_eval.yaml --artifact-backend file:///data/b
322
322
 
323
323
  ```bash
324
324
  uv run pyattacker serve runs/qa.db # http://127.0.0.1:8787
325
- # / dashboard /stats /events /pipelines /resources /errors (JSON)
325
+ # / dashboard /stats /metrics /events /pipelines /resources /errors (JSON)
326
326
  ```
327
327
 
328
328
  每个请求都新开一个只读连接,所以可以和正在跑的进程并存。**没有认证**,只绑回环地址——它没有 artifact 路由,但会暴露这次运行记下来的东西:事件的 `data`、存下来的错误信息。当调试工具用就好,别裸暴露到公网。
329
+ 应用可用 `runner.report_metric("accuracy", value, display="percent")` 汇报实时指标;框架只展示,
330
+ 不计算指标。参见 [`examples/live_metrics.py`](examples/live_metrics.py)。
329
331
 
330
332
  **步骤内分支**——`fanout(a, b)` 拿同一份输入并发跑多个任务,返回 `{task_name: value}`。重试粒度变成整个组,这是不把流水线做成 DAG 的代价。组是 Runner 看到的唯一规格,所以子任务们对 `resource`、`algorithm`、`timeout_s` 达成一致时,这些参数从子任务继承(此时 `timeout_s` 限制整个组)。
331
333
 
@@ -389,6 +391,10 @@ uv run pyattacker run -c examples/qa_eval.yaml --limit 40
389
391
 
390
392
  ## 当前状态
391
393
 
394
+ **0.3.1——应用汇报的实时指标。** 应用可用 `Runner.report_metric()` 或 `TaskContext.report_metric()`
395
+ 把准确率等数值汇报到只读监控面板。完成回调提供已提交的流水线结果,指标计算仍由应用负责。
396
+ 面板和 `watch` 默认统一跟随同一个运行。参见 [`examples/live_metrics.py`](examples/live_metrics.py)。
397
+
392
398
  **0.3.0——高级控制流(可选),以及中文文档。** 任务现在可以做[交接](README.zh-CN.md#进阶交接可选启用):返回一个 `Handoff` 跳过声明的后续步骤或提前收尾,记录在持久化账本里,续跑时从账本接着走。需要手动开启,不开完全没影响——没有 `control` 块的流水线不写新数据,`spec_digest` 逐字节不变。1.0 之前标记为实验性。反向遍历现在支持声明式回退和全量重试,带 visits 和可选的载荷历史。整套文档也提供了简体中文版([`README.zh-CN.md`](README.zh-CN.md)、[`docs/zh-CN/`](https://github.com/Hazer-BJTU/pyattacker/tree/main/docs/zh-CN)),由 CI 保持同步。升级时有一个改名要知道:`MergedReport.events_total` 现在叫 `source_events_total`,因为它是合并报告里**唯一**不去重、按源库原始累加的计数(旧名字仍可作为废弃别名使用)。
393
399
 
394
400
  **0.2.0——基准测试、更严格的身份校验、三处正确性修复。** 新增 `pyattacker bench`:一个模拟接口方的世界,用一组指标对比各获取算法,而不是一个加权分数([`docs/benchmark.md`](https://github.com/Hazer-BJTU/pyattacker/blob/main/docs/zh-CN/benchmark.md))。新增 `Retrying.decide`:把重试决策暴露为策略上的方法。新增分页的整类读取(`pyattacker.store.iter_*`,由可选的 `PagedStore` 扩展提供),导出大存储不再需要全量加载。0.1.x 已经实现了 M0–M4 计划的全部内容:内核、持久化和任务级恢复、重试和错误分类、带 7 种获取算法的资源池、延迟延续和 write-behind 批处理、分片和合并报告、三种格式五种导出形状、入口点插件、外部产物后端、fan-out 辅助函数、HTTP 监控端点。
@@ -165,10 +165,13 @@ derives it from the event log, scoped to the pipelines in the report.
165
165
  pyattacker watch STORE [--run-id ID] [--interval S] [--iterations N] [--no-clear]
166
166
  ```
167
167
 
168
+ Without `--run-id`, `watch` follows the latest started run. Use `--run-id all` for aggregate
169
+ operational statistics across the store; application-reported metrics are omitted in that view.
170
+
168
171
  A read-only connection to the same SQLite file, so it runs beside a live run (WAL allows one writer and many
169
172
  readers). Shows the pipeline state distribution, latency percentiles, per-pool `active/capacity`,
170
173
  `ready/degraded/dead`, how many pipelines are waiting or parked, and recent errors. A run with handoffs also
171
- shows `handoffs=N` (commits in the selected scope, all history without a run filter; a resume may reuse earlier active handoffs) on its attempts line — on a control-enabled pipeline the cursor is a position, not a
174
+ shows `handoffs=N` (commits in the selected run, or all history with `--run-id all`; a resume may reuse earlier active handoffs) on its attempts line — on a control-enabled pipeline the cursor is a position, not a
172
175
  progress count, so that number is what explains a short task list. `--iterations N` makes it
173
176
  exit on its own, which is what you want in a script.
174
177
 
@@ -201,7 +204,10 @@ an export of a store that is still being written does and does not guarantee.
201
204
  pyattacker serve STORE [--host HOST] [--port PORT] [--run-id ID]
202
205
  ```
203
206
 
204
- `/` is a small auto-refreshing dashboard; `/stats`, `/events`, `/pipelines`, `/resources`, `/errors` are JSON.
207
+ Without `--run-id`, the dashboard follows the latest started run. Use `--run-id all` for an
208
+ aggregate view, or open `/?run_id=ID` to inspect a specific run.
209
+
210
+ `/` is a small auto-refreshing dashboard; `/stats`, `/metrics`, `/events`, `/pipelines`, `/resources`, `/errors` are JSON.
205
211
  A fresh read-only connection per request means it is safe beside a live run.
206
212
 
207
213
  **It has no authentication, and there is no artifact route.** What it exposes is what the run recorded —
@@ -2,7 +2,7 @@
2
2
 
3
3
  **English** | [简体中文](zh-CN/design.md)
4
4
 
5
- > Version: 0.3.0 — M0–M5 complete, including the opt-in advanced control flow of §4.8 (forward handoffs and
5
+ > Version: 0.3.1 — M0–M5 complete, including the opt-in advanced control flow of §4.8 (forward handoffs and
6
6
  > backward traversal), which stays experimental until 1.0. See "Implemented / Left for later" in Section 9.
7
7
  > In one sentence: **an async task orchestration framework centered on the artifact, using the pipeline as the unit of completion, and the resource pool as the only shared surface.**
8
8
  > It does not touch the network, does not do reduction, and does not do DAG scheduling — it is only responsible for "running tens of thousands of mutually independent pipelines to completion, reliably, recoverably, and observably".
@@ -28,6 +28,11 @@ framework to step in:
28
28
  2. Write a **sink pipeline**: use one task to publish the results onto some resource/bus, where subscribers
29
29
  consume them — reduction becomes something you build yourself out of framework primitives.
30
30
 
31
+ An application can now also publish its own latest values through `Runner.report_metric()`. This is
32
+ transport and presentation of an already computed value; it does not change the reduction boundary.
33
+ The value is scoped to a run (optionally a pipeline), upserted synchronously, and read by the
34
+ read-only `/metrics` endpoint. The runner's store remains the sole writer.
35
+
31
36
  ---
32
37
 
33
38
  ## 2. Core Invariants
@@ -314,7 +319,7 @@ short, and with a fake clock the pump advances virtual time instantly (so tests
314
319
  is not a checkpoint. A `SIGKILL` can therefore lose the last batch of history while every checkpoint stays
315
320
  intact; `--no-write-behind` trades throughput for an immediate commit per attempt.
316
321
 
317
- ### 4.6 Monitoring: Cares About Traffic and Blocking, Not About Metrics
322
+ ### 4.6 Monitoring: Operational State and Application Reports
318
323
 
319
324
  ```python
320
325
  snapshot = runner.stats() # live in-process snapshot
@@ -324,6 +329,9 @@ snapshot = runner.stats() # live in-process snapshot
324
329
  The panel shows: pipeline state distribution, p95/max latency, per-task counts, and for each pool its
325
330
  `active/capacity`, `ready/degraded/dead`, `waiting`, throughput and leak counters, and recent errors.
326
331
  It is also usable programmatically: `monitor.render_snapshot(stats)` / `monitor.watch(store)`.
332
+ Application-reported values appear separately from operational counters. The application may
333
+ observe committed pipeline completions and publish accuracy, but it owns deduplication and recovery
334
+ of its accumulator from durable artifacts.
327
335
 
328
336
  ### 4.7 Completion Is Counted, Worker Lifetime Is Supervised
329
337
 
@@ -232,6 +232,7 @@ A selector matching **no** resource waits forever under the default `wait` algor
232
232
  | `ctx.held_leases()` | leases this attempt currently holds |
233
233
  | `ctx.reclaim_now()` | force-release everything held; synchronous, uninterruptible |
234
234
  | `ctx.emit(kind, **data)` | write your own event into the run's event stream |
235
+ | `ctx.report_metric(name, value, *, label="", display="number")` | publish a latest value scoped to this pipeline |
235
236
 
236
237
  ```python
237
238
  @task("adaptive", resource="apis")
@@ -739,6 +740,7 @@ the `with` block, or reopen the file afterwards with `open_store`.
739
740
  | `run(specs, *, resume=False, **overrides)` | run to completion, returns a `RunReport`. Wraps `run_async` in `asyncio.run` |
740
741
  | `await run_async(specs, *, resume=False, **overrides)` | same, inside an existing event loop |
741
742
  | `stats()` | live snapshot; safe to call mid-run |
743
+ | `report_metric(name, value, *, label="", display="number", pipeline_id=None)` | publish an application-defined latest value |
742
744
  | `stop(reason="user")` | ask the run to stop gracefully: stop admitting, drain what is in flight |
743
745
  | `stopping` | whether a stop is in progress |
744
746
  | `run_id` | the current run's id |
@@ -2165,6 +2167,45 @@ runner.run(template.map(jsonl_source("dataset.jsonl", limit=500)))
2165
2167
 
2166
2168
  ## Monitoring
2167
2169
 
2170
+ ### Application-reported metrics
2171
+
2172
+ An application can publish its latest experiment values while a run is active. Pyattacker stores
2173
+ and displays the values; the application computes them. For a runnable accuracy example, see
2174
+ [`examples/live_metrics.py`](../examples/live_metrics.py).
2175
+
2176
+ ```python
2177
+ runner.report_metric("evaluated", completed, label="Evaluated")
2178
+ runner.report_metric("accuracy", correct / completed, label="Accuracy", display="percent")
2179
+ # Inside a task, ctx.report_metric("phase", "scoring", display="text")
2180
+ ```
2181
+
2182
+ `Runner(..., on_pipeline_finished=callback)` calls `callback(runner, record, artifact)` after a pipeline's
2183
+ terminal state is stored. `artifact` is its final `Artifact` for success and `None` for failure or
2184
+ interruption. The passed `runner` provides `report_metric()` and can decode a retained artifact with
2185
+ `runner.registry.load(artifact.encoded())`. The callback
2186
+ runs in the scheduler thread and should finish quickly. Exceptions in it are recorded as
2187
+ `monitor.callback_failed` and do not change the pipeline outcome. A process crash may miss or replay
2188
+ the callback, so applications should deduplicate by pipeline ID and rebuild from stored final
2189
+ artifacts when needed. Resumed runs have a new run ID and their reported values have a separate scope.
2190
+
2191
+ `report_metric(name, value, *, label="", display="number", pipeline_id=None)` accepts a string,
2192
+ boolean, or finite number. `display` is `number`, `percent` (a numeric fraction, displayed as a
2193
+ percentage), or `text` (a string). Repeating the same name in the same run and scope replaces the
2194
+ previous value. `ctx.report_metric(...)` uses the current pipeline as its scope. Reports are
2195
+ synchronous state writes, including when event write-behind is enabled. The feature is optional for
2196
+ third-party stores: reporting on a store without it raises `StoreFeatureUnsupported`.
2197
+
2198
+ `/metrics?run_id=...` returns `{"run_id": ..., "rows": [{"run_id", "pipeline_id", "name",
2199
+ "value", "label", "display", "updated_at"}, ...]}`. Add `pipeline_id=...` to read a pipeline's
2200
+ reported values; `/pipelines` also includes a `reported_metrics` array in each row. With no run ID,
2201
+ `read_snapshot()`, `watch`, and every `StatsServer` endpoint select the latest started run, including
2202
+ runs that reported only pipeline-scoped values. The HTML dashboard resolves that run through `/stats`
2203
+ and passes its ID to `/metrics`, `/pipelines`, and `/events` for a consistent refresh. Use
2204
+ `?run_id=all` (or `--run-id all` for `watch`/`serve`) for aggregate operational data; that view
2205
+ does not show application metrics from an arbitrarily chosen run. The HTML dashboard shows
2206
+ run-level values as cards; the terminal `watch` view shows them too. The HTTP server remains read-only,
2207
+ and metric writes use the active Runner's store connection.
2208
+
2168
2209
  ### `runner.stats()`
2169
2210
 
2170
2211
  The in-process live snapshot. Safe to call mid-run.
@@ -2197,9 +2238,9 @@ with StatsServer("runs/qa.db", port=8787) as server:
2197
2238
  | Endpoint | Returns |
2198
2239
  |---|---|
2199
2240
  | `/` | a small auto-refreshing dashboard |
2200
- | `/stats`, `/events`, `/pipelines`, `/resources`, `/errors` | JSON |
2241
+ | `/stats`, `/metrics`, `/events`, `/pipelines`, `/resources`, `/errors` | JSON |
2201
2242
 
2202
- `/stats` carries `handoffs_total` (commits by the selected run, or all commits without a run filter), and each `/pipelines` row carries a
2243
+ `/stats` carries `handoffs_total` (commits by the selected run, or all commits with `run_id=all`), and each `/pipelines` row carries a
2203
2244
  `handoffs` count of active-execution records, `handoffs_historical` count of all records and
2204
2245
  `handoff_floor`, next to `n_tasks_done`/`n_tasks_total` — on a control-enabled pipeline those two are a
2205
2246
  **position** in the chain, not a count of tasks that ran, so a non-zero `handoffs` is what says "this
@@ -2227,4 +2268,3 @@ command line.
2227
2268
  | [`docs/cli.md`](cli.md) | commands, flags, exit codes, config file format |
2228
2269
  | [`docs/design.md`](design.md) | the model, the invariants, and the tradeoffs behind these APIs |
2229
2270
  | [`examples/`](../examples) | complete programs, including a measured comparison of pipeline shapes |
2230
-
@@ -1603,11 +1603,14 @@ blobs on disk: 3 files, [11, 27, 1536] bytes
1603
1603
  [`examples/plugin_package/`](../examples/plugin_package/README.md).
1604
1604
 
1605
1605
  **Monitoring a run in progress.** `pyattacker watch runs/qa.db` gives you a terminal view from a second
1606
- process, and `pyattacker serve runs/qa.db` an HTTP dashboard plus JSON at `/stats`, `/events`, `/pipelines`,
1606
+ process, and `pyattacker serve runs/qa.db` an HTTP dashboard plus JSON at `/stats`, `/metrics`, `/events`, `/pipelines`,
1607
1607
  `/resources` and `/errors`. Both open read-only connections, so they are safe beside a live run. The HTTP
1608
1608
  endpoint has no authentication and serves your payloads — keep it on loopback.
1609
+ For live experiment accuracy, run [`examples/live_metrics.py`](../examples/live_metrics.py). Its
1610
+ completion callback decodes final artifacts, deduplicates by pipeline ID, and reports the current
1611
+ accuracy through `runner.report_metric()`; the dashboard displays the value as it changes.
1609
1612
  * **Monitoring**: `pyattacker serve runs/qa.db` is a zero-dependency read-only HTTP view (`/`,
1610
- `/stats`, `/events`, `/pipelines`, `/resources`, `/errors`) that opens a fresh connection per request,
1613
+ `/stats`, `/metrics`, `/events`, `/pipelines`, `/resources`, `/errors`) that opens a fresh connection per request,
1611
1614
  so it runs happily beside a live run. It binds to loopback and has no authentication — treat it as a
1612
1615
  debug view, not a dashboard.
1613
1616
 
@@ -128,7 +128,10 @@ pyattacker report STORE [STORE ...] [--run-id ID] [--errors N] [--json] [--artif
128
128
  pyattacker watch STORE [--run-id ID] [--interval S] [--iterations N] [--no-clear]
129
129
  ```
130
130
 
131
- 对同一个 SQLite 文件开只读连接,所以能和正在跑的流水线并存(WAL 允许一个写入者多个读取者)。显示流水线状态分布、延迟百分位、每个资源池的 `active/capacity`、`ready/degraded/dead`、多少流水线在等待或挂起、最近的错误。开了交接的运行还会在 attempts 行显示 `handoffs=N`(选定范围内的提交数,不带 run 过滤就是全部历史;续跑可能复用更早的活跃交接)——开了控制流的流水线,游标是位置不是进度计数,所以任务列表很短的时候就是这个数字在起作用。`--iterations N` 让它自己退出,脚本里用很方便。
131
+ 不指定 `--run-id` 时,`watch` 跟随最近启动的运行。使用 `--run-id all` 可查看全库运行统计;
132
+ 该视图不显示某个运行的应用汇报指标。
133
+
134
+ 对同一个 SQLite 文件开只读连接,所以能和正在跑的流水线并存(WAL 允许一个写入者多个读取者)。显示流水线状态分布、延迟百分位、每个资源池的 `active/capacity`、`ready/degraded/dead`、多少流水线在等待或挂起、最近的错误。开了交接的运行还会在 attempts 行显示 `handoffs=N`(选定运行的提交数,使用 `--run-id all` 则是全部历史;续跑可能复用更早的活跃交接)——开了控制流的流水线,游标是位置不是进度计数,所以任务列表很短的时候就是这个数字在起作用。`--iterations N` 让它自己退出,脚本里用很方便。
132
135
 
133
136
  ## `export` —— 导出记录
134
137
 
@@ -154,7 +157,10 @@ pyattacker export STORE [STORE ...] OUTPUT [--rows SHAPE] [--format FMT] [--run-
154
157
  pyattacker serve STORE [--host HOST] [--port PORT] [--run-id ID]
155
158
  ```
156
159
 
157
- `/` 是个自动刷新的小仪表盘;`/stats`、`/events`、`/pipelines`、`/resources`、`/errors` 返回 JSON。每个请求都新开一个只读连接,所以和正在跑的流水线并存是安全的。
160
+ 不指定 `--run-id` 时,面板跟随最近启动的运行。使用 `--run-id all` 可查看全库视图,
161
+ 也可打开 `/?run_id=ID` 查看指定运行。
162
+
163
+ `/` 是个自动刷新的小仪表盘;`/stats`、`/metrics`、`/events`、`/pipelines`、`/resources`、`/errors` 返回 JSON。每个请求都新开一个只读连接,所以和正在跑的流水线并存是安全的。
158
164
 
159
165
  **没有认证,也没有 artifact 路由。** 它暴露的是这次运行记下来的内容——事件的 `data` 原样呈现,存下来的 `error_message` 字段也在——所以默认只绑回环地址。要绑别的地址?前面自己加个代理。
160
166
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  [English](../design.md) | **简体中文**
4
4
 
5
- > 版本:0.3.0——M0–M5 已完成,含 §4.8 那套可选启用的高级控制流(正向交接与反向遍历),1.0 之前标实验性。
5
+ > 版本:0.3.1——M0–M5 已完成,含 §4.8 那套可选启用的高级控制流(正向交接与反向遍历),1.0 之前标实验性。
6
6
  > 见第 9 节"已实现 / 留待以后"。
7
7
  > 一句话概括:**一个以 artifact 为中心的异步任务编排框架,用 pipeline 作为完成单位,把 resource pool 作为唯一的共享面。**
8
8
  > 它不碰网络、不做归约、也不做 DAG 调度——只管一件事:"把几万条彼此独立的 pipeline 可靠、可恢复、可观测地跑到完成"。
@@ -24,6 +24,10 @@
24
24
  1. 导出 artifact(`export` / `report.export_jsonl`),在外面自己算;
25
25
  2. 写一条**汇聚流水线**(sink pipeline):用一个任务把结果发到某个资源/总线上,订阅者消费——归约就变成你用框架原语自己拼的东西。
26
26
 
27
+ 应用还可以通过 `Runner.report_metric()` 汇报自己算好的最新值。这只是传输和展示,不改变
28
+ 语义归约的边界。数值按运行(可选按流水线)划分作用域,同步覆盖写入,并由只读的
29
+ `/metrics` 端点读取。Runner 的存储连接仍是唯一写入者。
30
+
27
31
  ---
28
32
 
29
33
  ## 2. 核心不变量
@@ -233,7 +237,7 @@ REVOKED ◀── explicit revoke / revoked from within a task
233
237
 
234
238
  **只追加的事实按批写。** `WriteBehindStore` 缓冲尝试和事件,按批刷(大小阈值、时间间隔、任何读 API、运行心跳、运行结束)。状态写入——`pipelines`、`tasks`、`artifacts`——永远直接落盘,因为还没落盘的检查点不算检查点。所以 `SIGKILL` 可能丢最后一批历史,所有检查点完好;`--no-write-behind` 用吞吐换每次尝试立即提交。
235
239
 
236
- ### 4.6 监控:看流量和阻塞,不看指标
240
+ ### 4.6 监控:运行状态与应用汇报
237
241
 
238
242
  ```python
239
243
  snapshot = runner.stats() # live in-process snapshot
@@ -241,6 +245,8 @@ snapshot = runner.stats() # live in-process snapshot
241
245
  ```
242
246
 
243
247
  面板显示:流水线状态分布、p95/最大延迟、各任务计数,每个资源池的 `active/capacity`、`ready/degraded/dead`、`waiting`、吞吐和泄漏计数器,最近的错误。也可以编程用:`monitor.render_snapshot(stats)` / `monitor.watch(store)`。
248
+ 应用汇报值与运行计数分开展示。应用可以观察已提交的流水线终态并汇报准确率,但对流水线
249
+ 去重、以及从持久化产物恢复累计结果都由应用负责。
244
250
 
245
251
  ### 4.7 完成以计数为准,worker 生命周期受监督
246
252
 
@@ -215,6 +215,7 @@ async with ctx.acquire(where=lambda r: r.options["ctx_len"] >= 32000) as lease:
215
215
  | `ctx.held_leases()` | 这次尝试当前持有的租约 |
216
216
  | `ctx.reclaim_now()` | 强制归还当前持有的全部租约;同步、不可中断 |
217
217
  | `ctx.emit(kind, **data)` | 把你自己的事件写进这次运行的事件流 |
218
+ | `ctx.report_metric(name, value, *, label="", display="number")` | 汇报当前流水线作用域的最新值 |
218
219
 
219
220
  ```python
220
221
  @task("adaptive", resource="apis")
@@ -677,6 +678,7 @@ with Runner(store="runs/qa.db", pools=[pool], concurrency=64) as runner:
677
678
  | `run(specs, *, resume=False, **overrides)` | 跑至完成,返回 `RunReport`。在 `asyncio.run` 里包 `run_async` |
678
679
  | `await run_async(specs, *, resume=False, **overrides)` | 同上,但在已有的事件循环里 |
679
680
  | `stats()` | 实时快照;运行中途安全调 |
681
+ | `report_metric(name, value, *, label="", display="number", pipeline_id=None)` | 汇报应用计算的最新值 |
680
682
  | `stop(reason="user")` | 请求运行优雅停:不再接新活,排空在途工作 |
681
683
  | `stopping` | 是否在停 |
682
684
  | `run_id` | 当前运行的 id |
@@ -2085,6 +2087,40 @@ runner.run(template.map(jsonl_source("dataset.jsonl", limit=500)))
2085
2087
 
2086
2088
  ## 监控
2087
2089
 
2090
+ ### 应用汇报的指标
2091
+
2092
+ 应用可以在运行期间汇报实验指标的最新值。Pyattacker 只负责保存和展示,指标由应用计算。
2093
+ 可运行的准确率示例见 [`examples/live_metrics.py`](../../examples/live_metrics.py)。
2094
+
2095
+ ```python
2096
+ runner.report_metric("evaluated", completed, label="Evaluated")
2097
+ runner.report_metric("accuracy", correct / completed, label="Accuracy", display="percent")
2098
+ # Inside a task, ctx.report_metric("phase", "scoring", display="text")
2099
+ ```
2100
+
2101
+ `Runner(..., on_pipeline_finished=callback)` 在流水线终态持久化后调用
2102
+ `callback(runner, record, artifact)`。成功时传入最终 `Artifact`,失败或中断时传入 `None`。
2103
+ 传入的 `runner` 可用于 `report_metric()`,也可以用
2104
+ `runner.registry.load(artifact.encoded())` 解码保存的产物。回调在调度线程中运行,
2105
+ 应尽快返回。回调异常记录为 `monitor.callback_failed`,不会改变流水线结果。进程崩溃可能
2106
+ 导致回调遗漏或重放,因此应用应按 pipeline ID 去重,并在需要时从持久化的最终产物重建
2107
+ 统计。恢复运行使用新的 run ID,汇报值也属于新的作用域。
2108
+
2109
+ `report_metric(name, value, *, label="", display="number", pipeline_id=None)` 接受字符串、
2110
+ 布尔值或有限数字。`display` 可为 `number`、`percent`(传入比例,展示为百分比)或
2111
+ `text`(字符串)。同一运行、同一作用域中的同名指标会覆盖旧值。`ctx.report_metric(...)`
2112
+ 自动使用当前 pipeline 作为作用域。即使启用事件延迟写入,指标仍同步写入。第三方存储
2113
+ 可以选择支持此功能;对不支持的存储汇报时抛出 `StoreFeatureUnsupported`。
2114
+
2115
+ `/metrics?run_id=...` 返回 `{"run_id": ..., "rows": [{"run_id", "pipeline_id", "name",
2116
+ "value", "label", "display", "updated_at"}, ...]}`。加上 `pipeline_id=...` 可读取该流水线的
2117
+ 自定义值;`/pipelines` 的每行也包含 `reported_metrics` 数组。未指定 run ID 时,
2118
+ `read_snapshot()`、`watch` 和 `StatsServer` 的各端点统一选择最近启动的运行,包括只汇报
2119
+ 流水线级值的运行。HTML 面板先从 `/stats` 确定 run ID,再用同一个 ID 查询 `/metrics`、
2120
+ `/pipelines` 和 `/events`。需要全库运行统计时可用 `?run_id=all`(`watch`/`serve` 使用
2121
+ `--run-id all`);全库视图不会任意挑一个运行的自定义指标来展示。HTML 面板将运行级值显示为卡片;终端
2122
+ `watch` 也会展示。HTTP 服务仍只读,指标通过正在运行的 Runner 的存储连接写入。
2123
+
2088
2124
  ### `runner.stats()`
2089
2125
 
2090
2126
  进程内实时快照。运行中途安全调。
@@ -2117,9 +2153,9 @@ with StatsServer("runs/qa.db", port=8787) as server:
2117
2153
  | 端点 | 返回 |
2118
2154
  |---|---|
2119
2155
  | `/` | 一个小巧的自动刷新仪表盘 |
2120
- | `/stats`, `/events`, `/pipelines`, `/resources`, `/errors` | JSON |
2156
+ | `/stats`, `/metrics`, `/events`, `/pipelines`, `/resources`, `/errors` | JSON |
2121
2157
 
2122
- `/stats` 带 `handoffs_total`(所选运行的提交数,没运行过滤器则全部提交),每行 `/pipelines` 带
2158
+ `/stats` 带 `handoffs_total`(所选运行的提交数,`run_id=all` 则为全部提交),每行 `/pipelines` 带
2123
2159
  `handoffs`(活动执行记录的计数)、`handoffs_historical`(全部记录的计数)和
2124
2160
  `handoff_floor`,紧挨着 `n_tasks_done`/`n_tasks_total`——开控制流的流水线上,这两者是链中的
2125
2161
  **位置**,不是已跑任务数,所以非零的 `handoffs` 才说明"这条
@@ -1468,9 +1468,12 @@ blobs on disk: 3 files, [11, 27, 1536] bytes
1468
1468
  * **工件后端**决定载荷字节放哪:`None`/`"inline"` 留在数据库里,`"null"` 丢掉字节但留摘要,`"file:///data/blobs"`(或 `{"kind": "file", "root": ..., "min_bytes": 262144}`)溢到内容寻址的文件里,读取时再水合回来。上面的 `min_bytes=0` 什么都溢出去,所以工件行显示的是引用而不是字节。
1469
1469
  * **插件**是 `importlib.metadata` 入口点,分四组:`pyattacker.tasks`、`pyattacker.algorithms`、`pyattacker.codecs` 和 `pyattacker.stores`(按 URI scheme 索引,`store = "s3://bucket/runs.db"` 就能用)。内置项先解析,导入时报错的插件记下来不致命——`pyattacker plugins` 两者都列。完整示例包在 [`examples/plugin_package/`](../../examples/plugin_package/README.zh-CN.md)。
1470
1470
 
1471
- **监控正在跑的运行。** `pyattacker watch runs/qa.db` 让你从第二个进程看终端视图,`pyattacker serve runs/qa.db` 提供 HTTP 仪表盘,在 `/stats`、`/events`、`/pipelines`、`/resources` 和 `/errors` 提供 JSON。两个都开只读连接,能和正在跑的任务并排用。HTTP 端点没认证,会把你的载荷给出来——只绑回环地址。
1471
+ **监控正在跑的运行。** `pyattacker watch runs/qa.db` 让你从第二个进程看终端视图,`pyattacker serve runs/qa.db` 提供 HTTP 仪表盘,在 `/stats`、`/metrics`、`/events`、`/pipelines`、`/resources` 和 `/errors` 提供 JSON。两个都开只读连接,能和正在跑的任务并排用。HTTP 端点没认证,会把你的载荷给出来——只绑回环地址。
1472
+ 需要实时实验准确率时,运行 [`examples/live_metrics.py`](../../examples/live_metrics.py)。示例的
1473
+ 完成回调解码最终产物、按 pipeline ID 去重,再通过 `runner.report_metric()` 汇报当前
1474
+ 准确率;面板会随着汇报更新。
1472
1475
  * **监控**:`pyattacker serve runs/qa.db` 是零依赖的只读 HTTP 视图(`/`、
1473
- `/stats`、`/events`、`/pipelines`、`/resources`、`/errors`),每个请求开个新连接,和正在跑的任务并行无碍。绑回环地址、没认证——当调试视图用,别当仪表盘用。
1476
+ `/stats`、`/metrics`、`/events`、`/pipelines`、`/resources`、`/errors`),每个请求开个新连接,和正在跑的任务并行无碍。绑回环地址、没认证——当调试视图用,别当仪表盘用。
1474
1477
 
1475
1478
  ---
1476
1479
 
@@ -0,0 +1,72 @@
1
+ """Live, application-owned accuracy on the HTML dashboard.
2
+
3
+ Run ``uv run python examples/live_metrics.py`` and open the printed URL while it runs.
4
+ The application reads final artifacts, owns the accumulator, and only reports its
5
+ latest values to pyattacker. The database can be reopened to rebuild the accumulator.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import asyncio
11
+ from pathlib import Path
12
+
13
+ from pyattacker import Runner, SqliteStore, StatsServer, pipeline, task
14
+ from pyattacker.artifact import DEFAULT_REGISTRY
15
+ from pyattacker.store import iter_pipelines
16
+
17
+
18
+ @task("live.judge")
19
+ async def judge(seed: dict) -> dict:
20
+ await asyncio.sleep(0.1)
21
+ return {"correct": seed["answer"] == seed["expected"]}
22
+
23
+
24
+ def rebuild(db: str) -> dict[str, bool]:
25
+ """Rebuild application state from durable final artifacts after a restart."""
26
+ results: dict[str, bool] = {}
27
+ store = SqliteStore(db, read_only=True)
28
+ try:
29
+ for record in iter_pipelines(store, state="succeeded"):
30
+ artifact = next((item for item in store.artifacts(record.pipeline_id) if item.is_final), None)
31
+ if artifact is not None and artifact.available:
32
+ results[record.pipeline_id] = bool(DEFAULT_REGISTRY.load(artifact.encoded())["correct"])
33
+ finally:
34
+ store.close()
35
+ return results
36
+
37
+
38
+ class AccuracyMonitor:
39
+ """Own the cross-pipeline reduction; Runner supplies committed completions."""
40
+
41
+ def __init__(self, results: dict[str, bool]) -> None:
42
+ self.results = results
43
+
44
+ def __call__(self, runner: Runner, record, artifact) -> None:
45
+ if artifact is not None and artifact.available:
46
+ self.results[record.pipeline_id] = bool(runner.registry.load(artifact.encoded())["correct"])
47
+ elif artifact is None:
48
+ self.results.pop(record.pipeline_id, None)
49
+ # A failed pipeline does not enter this example's accuracy denominator.
50
+ evaluated = len(self.results)
51
+ correct = sum(self.results.values())
52
+ runner.report_metric("evaluated", evaluated, label="Evaluated")
53
+ runner.report_metric("correct", correct, label="Correct")
54
+ if evaluated:
55
+ runner.report_metric("accuracy", correct / evaluated,
56
+ label="Accuracy", display="percent")
57
+
58
+
59
+ def main() -> None:
60
+ db = "runs/live_metrics.db"
61
+ Path(db).parent.mkdir(exist_ok=True)
62
+ monitor = AccuracyMonitor(rebuild(db) if Path(db).exists() else {})
63
+
64
+ with Runner(store=db, on_pipeline_finished=monitor, retry_succeeded=True,
65
+ handle_signals=False) as runner, StatsServer(db, port=0) as server:
66
+ print(f"Monitor: {server.url}")
67
+ rows = ({"answer": i, "expected": i if i % 3 else i + 1} for i in range(100))
68
+ runner.run(pipeline("live-eval", judge).map(rows))
69
+
70
+
71
+ if __name__ == "__main__":
72
+ main()
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyattacker"
3
- version = "0.3.0"
3
+ version = "0.3.1"
4
4
  description = "Artifact-centric, resumable async task orchestration: pipeline / task / resource pool."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -77,6 +77,7 @@ from .history import HistoryArtifact
77
77
  from .merge import MergedReport, merge_reports
78
78
  from .pipeline import Chain, PipelineSpec, PipelineTemplate, compute_spec_digest, pipeline, with_retry
79
79
  from .plugins import PLUGINS, PluginRegistry, list_plugins
80
+ from .reported_metrics import ReportedMetric
80
81
  from .resource import Bus, Lease, Pool, Resource, ResourceEvent, ResourceState
81
82
  from .runner import RunConfig, Runner, RunReport
82
83
  from .server import StatsServer
@@ -105,7 +106,7 @@ from .tasks import (
105
106
  write_jsonl,
106
107
  )
107
108
 
108
- __version__ = "0.3.0"
109
+ __version__ = "0.3.1"
109
110
 
110
111
  __all__ = [
111
112
  "__version__",
@@ -134,6 +135,7 @@ __all__ = [
134
135
  "Handoff",
135
136
  "HistoryArtifact",
136
137
  "Artifact",
138
+ "ReportedMetric",
137
139
  "Codec",
138
140
  "CodecRegistry",
139
141
  "JsonCodec",
@@ -143,7 +143,7 @@ class StoreUnavailable(PyAttackerError):
143
143
 
144
144
 
145
145
  class StoreFeatureUnsupported(PyAttackerError):
146
- """The store was written by a newer pyattacker and this build cannot interpret it.
146
+ """The store lacks a requested optional feature, or uses a newer feature level.
147
147
 
148
148
  A store records the highest on-disk feature level it has reached (``store/visits.py``), and a
149
149
  binary that does not know that level refuses to operate on it: reading or writing a
@@ -151,6 +151,9 @@ class StoreFeatureUnsupported(PyAttackerError):
151
151
  occurrence, not merely display incomplete data. Upgrade the package to open the store; back it
152
152
  up with SQLite's own backup API (``sqlite3 .backup``) or a file copy taken while no writer is
153
153
  active (see ``docs/reference.md``, "Advanced: backward traversal").
154
+
155
+ An older third-party store also raises this error when a caller requests an optional
156
+ capability such as application-reported metrics that it does not implement.
154
157
  """
155
158
 
156
159