workstreams-cli 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/PKG-INFO +100 -23
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/README.md +99 -22
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/pyproject.toml +1 -1
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/__init__.py +16 -1
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/cli.py +80 -0
- workstreams_cli-0.6.0/src/workstreams/confidence.py +196 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams_cli.egg-info/PKG-INFO +100 -23
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams_cli.egg-info/SOURCES.txt +2 -0
- workstreams_cli-0.6.0/tests/test_confidence.py +185 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/setup.cfg +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/config.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/dashboard.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/event_log.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/manager.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/models.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/multiplexer/__init__.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/multiplexer/base.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/multiplexer/tmux.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/multiplexer/tmux_compatible.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/multiplexer/zellij.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/notifier.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/py.typed +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/subagent_client.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams_cli.egg-info/dependency_links.txt +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams_cli.egg-info/entry_points.txt +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams_cli.egg-info/requires.txt +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams_cli.egg-info/top_level.txt +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/tests/test_config.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/tests/test_event_log.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/tests/test_models.py +0 -0
- {workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/tests/test_notifier.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: workstreams-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Visually dispatch coding-agent work to subagents in real terminal windows and monitor it in one dashboard - for any coding agent (Claude Code, Codex, OpenCode, Qwen Code, Hermes, Cline, and more).
|
|
5
5
|
Author: Dream-Pixels-Forge
|
|
6
6
|
License: MIT
|
|
@@ -45,6 +45,14 @@ Key capabilities:
|
|
|
45
45
|
- **Cross-terminal notifications** — desktop notifications (Linux `notify-send`, macOS `osascript`) plus a shared `notifications.jsonl` that other terminals can poll
|
|
46
46
|
- **Agent-agnostic** — no vendor lock-in. `dispatch` and `work` send arbitrary shell commands to panes, so it works with whatever agent binary you can run from a shell
|
|
47
47
|
|
|
48
|
+
## ⭐ If workstreams helps you ship faster, star the repo
|
|
49
|
+
|
|
50
|
+
If you found this useful, a GitHub star helps other developers discover it.
|
|
51
|
+
|
|
52
|
+
[](https://github.com/Dream-Pixels-Forge/workstreams-cli)
|
|
53
|
+
|
|
54
|
+
⭐ **Star this repo:** [github.com/Dream-Pixels-Forge/workstreams-cli](https://github.com/Dream-Pixels-Forge/workstreams-cli)
|
|
55
|
+
|
|
48
56
|
## Why this exists
|
|
49
57
|
|
|
50
58
|
Coding agents increasingly support "subagents" that run in the background of the main agent's process. That means: no visibility (you can't watch them), no isolation (they share one working tree and one set of installed dependencies), no way to run several in parallel on independent branches, and no shared event stream you can watch from your main terminal.
|
|
@@ -61,6 +69,7 @@ Coding agents increasingly support "subagents" that run in the background of the
|
|
|
61
69
|
- [Configuration (.workstreams.yaml)](#configuration-workstreamsyaml)
|
|
62
70
|
- [How the Multiplexers Work](#how-the-multiplexers-work)
|
|
63
71
|
- [Subagent Event System (Python API + CLI)](#subagent-event-system)
|
|
72
|
+
- [Confidence Scoring](#confidence-scoring)
|
|
64
73
|
- [Environment Variables](#environment-variables)
|
|
65
74
|
- [Exit Codes](#exit-codes)
|
|
66
75
|
- [Data Locations](#data-locations)
|
|
@@ -68,6 +77,7 @@ Coding agents increasingly support "subagents" that run in the background of the
|
|
|
68
77
|
- [CI/CD Integration](#cicd-integration)
|
|
69
78
|
- [Troubleshooting](#troubleshooting)
|
|
70
79
|
- [Best Practices](#best-practices)
|
|
80
|
+
- [Architecture](#architecture)
|
|
71
81
|
- [Contributing & License](#contributing--license)
|
|
72
82
|
|
|
73
83
|
---
|
|
@@ -104,7 +114,7 @@ pip install -e ".[yaml,dev]" # dev extras add pytest
|
|
|
104
114
|
Verify:
|
|
105
115
|
|
|
106
116
|
```bash
|
|
107
|
-
workstreams --version # -> workstreams 0.
|
|
117
|
+
workstreams --version # -> workstreams 0.6.0
|
|
108
118
|
```
|
|
109
119
|
|
|
110
120
|
> **Note:** every command also accepts `--json` to emit machine-readable output (where supported), which coding agents can parse. All read-side commands work without a multiplexer installed; only `start`/`dispatch`/`work`/`attach` need one.
|
|
@@ -435,6 +445,67 @@ Events are appended to `~/.workstreams/<project>/events.jsonl` using `O_APPEND`
|
|
|
435
445
|
|
|
436
446
|
---
|
|
437
447
|
|
|
448
|
+
## Confidence Scoring
|
|
449
|
+
|
|
450
|
+
Subagents are sometimes wrong. Confidence scoring lets each subagent **self-rate the quality of its result on a 0–10 scale** so the main agent (or a CI gate) can decide whether to accept, re-dispatch, or escalate. It is the difference between "subagent said it's done" and "subagent is 9/10 sure it actually delivered the right result".
|
|
451
|
+
|
|
452
|
+
- The score rides in the event's `data.confidence` field — no new storage required.
|
|
453
|
+
- A **gate** (`workstreams confidence --min-score N`) exits `0` when the latest score is at/above the threshold and `1` otherwise, so it drops straight into CI or a `--wait` loop.
|
|
454
|
+
- Scores are clamped to `[0, 10]`; a misbehaving subagent can't emit `100`.
|
|
455
|
+
|
|
456
|
+
### Emitting a score
|
|
457
|
+
|
|
458
|
+
```bash
|
|
459
|
+
# a subagent reports it finished, 9/10 confident
|
|
460
|
+
workstreams event completed --project myproj --workstream 1 \
|
|
461
|
+
--subagent claude-code --issue 42 \
|
|
462
|
+
--message "All tests green, edge cases covered" \
|
|
463
|
+
--confidence 9
|
|
464
|
+
|
|
465
|
+
# or via --data JSON
|
|
466
|
+
workstreams event completed --project myproj --workstream 1 \
|
|
467
|
+
--subagent codex --issue 42 --message "Done" --data '{"confidence": 7}'
|
|
468
|
+
|
|
469
|
+
# Python API (in a subagent script)
|
|
470
|
+
from workstreams import subagent_report
|
|
471
|
+
subagent_report("myproj", 1, "claude-code", 42, "completed", "Done",
|
|
472
|
+
data={"confidence": 9})
|
|
473
|
+
```
|
|
474
|
+
|
|
475
|
+
### Reading / gating on scores
|
|
476
|
+
|
|
477
|
+
```bash
|
|
478
|
+
# human-readable summary + gate (exit 0 if latest >= 8)
|
|
479
|
+
workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 8
|
|
480
|
+
|
|
481
|
+
# machine-readable
|
|
482
|
+
workstreams confidence --project myproj --workstream 1 --json
|
|
483
|
+
|
|
484
|
+
# list every confidence-bearing event, oldest first
|
|
485
|
+
workstreams confidence --project myproj --workstream 1 --records
|
|
486
|
+
```
|
|
487
|
+
|
|
488
|
+
Example output:
|
|
489
|
+
|
|
490
|
+
```
|
|
491
|
+
Confidence for ws1 #42: PASS
|
|
492
|
+
latest: 9.0/10 (completed from claude-code)
|
|
493
|
+
average: 7.0/10 over 2 report(s)
|
|
494
|
+
threshold: 8.0/10 -> All tests green, edge cases covered
|
|
495
|
+
```
|
|
496
|
+
|
|
497
|
+
### Using it as a CI / re-dispatch gate
|
|
498
|
+
|
|
499
|
+
```bash
|
|
500
|
+
# block the merge until the subagent's latest self-rating clears the bar
|
|
501
|
+
workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 9 \
|
|
502
|
+
|| { echo "confidence too low, re-dispatching"; workstreams dispatch ...; }
|
|
503
|
+
```
|
|
504
|
+
|
|
505
|
+
> Scores are self-reported. Treat them as a *signal*, not a proof — pair them with real test coverage. A 10/10 that shipped a broken build is still a broken build.
|
|
506
|
+
|
|
507
|
+
---
|
|
508
|
+
|
|
438
509
|
## Environment Variables
|
|
439
510
|
|
|
440
511
|
All are overridable in the config file / CLI; env vars are a fallback when neither is set.
|
|
@@ -588,28 +659,34 @@ workstreams logs --workstream 1 --lines 50
|
|
|
588
659
|
|
|
589
660
|
---
|
|
590
661
|
|
|
591
|
-
##
|
|
662
|
+
## Architecture
|
|
592
663
|
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
664
|
+
`workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary.
|
|
665
|
+
|
|
666
|
+

|
|
667
|
+
|
|
668
|
+
The diagram above shows the full data flow: the CLI routes each command to `WorkstreamsManager`, which talks to a pluggable multiplexer backend to place agents into visible terminal panes, while every lane's subagent writes JSONL events back to a shared, concurrent-safe log that the live dashboard re-renders on a timer. The core modules:
|
|
669
|
+
|
|
670
|
+
- **`cli.py`** — argparse entry point; every command accepts `--project` and `--json`
|
|
671
|
+
- **`manager.py`** — `WorkstreamsManager` orchestrates init/start/dispatch/work/sync/pr/merge/run/logs/events
|
|
672
|
+
- **`config.py`** — loads/saves `.workstreams.yaml` (or JSON) with no-pyyaml fallback; resolution order: CLI flag > yaml > env > default
|
|
673
|
+
- **`multiplexer/`** — pluggable backends behind `MultiplexerBase` (`tmux.py`, `zellij.py`, plus tmux-compatible wrappers for `nami`/`lmux`/`wmux`/`herdr`)
|
|
674
|
+
- **`event_log.py` + `subagent_client.py`** — shared cross-process JSONL event stream and the Python API emitters
|
|
675
|
+
- **`notifier.py`** — desktop notifications (`notify-send`/`osascript`/PowerShell) plus a `notifications.jsonl` queue
|
|
676
|
+
- **`confidence.py`** — aggregates subagent self-rated 0–10 quality scores for accept / re-dispatch gating
|
|
677
|
+
- **`dashboard.py`** — the live ANSI TUI that polls events and re-renders every 2s
|
|
678
|
+
- **`models.py`** — dataclasses (`WorkstreamConfig`, `WorkstreamsConfig`, `WorkstreamStatus`, `SubagentEvent`) shared across all of the above
|
|
679
|
+
|
|
680
|
+
How a run flows end-to-end:
|
|
681
|
+
|
|
682
|
+
1. **`init`** creates the lanes: for each workstream it adds a git worktree/branch (`ws/N`) plus a `worktrees/N/` directory, then writes `.workstreams.yaml` in the repo root.
|
|
683
|
+
2. **`start`** asks the chosen multiplexer backend (tmux by default) to open a detached, auto-named session `workstreams-<project>`, one window/tab per lane, each holding a persistent shell.
|
|
684
|
+
3. **`dispatch` / `work` / `run`** target a specific lane's pane (`<session>:<window-name>` for tmux windows, pane index for tiled) and send a command/prompt into it, emitting a `started` event to the shared log and firing a notification.
|
|
685
|
+
4. **Any** process — the agent inside a pane, a CI job, a script, the main terminal — appends JSONL events (`started`, `progress`, `completed`, `failed`, `error`, `done`) to `events.jsonl` using `O_APPEND` plus a short lock that self-heals stale locks after 10s, so concurrent writers never corrupt the stream.
|
|
686
|
+
5. **`monitor`** (the dashboard) and **`events`** read that stream back and re-render every 2s; `--wait` on `dispatch`/`work` blocks until a terminal event arrives (4h safety timeout).
|
|
687
|
+
6. **`sync` / `pr` / `merge` / `workstream cleanup`** close the loop: rebase/merge the lane, push and open a PR via `gh`, merge it, and prune the finished worktree.
|
|
688
|
+
|
|
689
|
+
Read-side commands (`status`, `events`, `logs`, `monitor`) need no multiplexer; only `start`/`dispatch`/`work`/`attach` require one to be installed.
|
|
613
690
|
|
|
614
691
|
---
|
|
615
692
|
|
|
@@ -15,6 +15,14 @@ Key capabilities:
|
|
|
15
15
|
- **Cross-terminal notifications** — desktop notifications (Linux `notify-send`, macOS `osascript`) plus a shared `notifications.jsonl` that other terminals can poll
|
|
16
16
|
- **Agent-agnostic** — no vendor lock-in. `dispatch` and `work` send arbitrary shell commands to panes, so it works with whatever agent binary you can run from a shell
|
|
17
17
|
|
|
18
|
+
## ⭐ If workstreams helps you ship faster, star the repo
|
|
19
|
+
|
|
20
|
+
If you found this useful, a GitHub star helps other developers discover it.
|
|
21
|
+
|
|
22
|
+
[](https://github.com/Dream-Pixels-Forge/workstreams-cli)
|
|
23
|
+
|
|
24
|
+
⭐ **Star this repo:** [github.com/Dream-Pixels-Forge/workstreams-cli](https://github.com/Dream-Pixels-Forge/workstreams-cli)
|
|
25
|
+
|
|
18
26
|
## Why this exists
|
|
19
27
|
|
|
20
28
|
Coding agents increasingly support "subagents" that run in the background of the main agent's process. That means: no visibility (you can't watch them), no isolation (they share one working tree and one set of installed dependencies), no way to run several in parallel on independent branches, and no shared event stream you can watch from your main terminal.
|
|
@@ -31,6 +39,7 @@ Coding agents increasingly support "subagents" that run in the background of the
|
|
|
31
39
|
- [Configuration (.workstreams.yaml)](#configuration-workstreamsyaml)
|
|
32
40
|
- [How the Multiplexers Work](#how-the-multiplexers-work)
|
|
33
41
|
- [Subagent Event System (Python API + CLI)](#subagent-event-system)
|
|
42
|
+
- [Confidence Scoring](#confidence-scoring)
|
|
34
43
|
- [Environment Variables](#environment-variables)
|
|
35
44
|
- [Exit Codes](#exit-codes)
|
|
36
45
|
- [Data Locations](#data-locations)
|
|
@@ -38,6 +47,7 @@ Coding agents increasingly support "subagents" that run in the background of the
|
|
|
38
47
|
- [CI/CD Integration](#cicd-integration)
|
|
39
48
|
- [Troubleshooting](#troubleshooting)
|
|
40
49
|
- [Best Practices](#best-practices)
|
|
50
|
+
- [Architecture](#architecture)
|
|
41
51
|
- [Contributing & License](#contributing--license)
|
|
42
52
|
|
|
43
53
|
---
|
|
@@ -74,7 +84,7 @@ pip install -e ".[yaml,dev]" # dev extras add pytest
|
|
|
74
84
|
Verify:
|
|
75
85
|
|
|
76
86
|
```bash
|
|
77
|
-
workstreams --version # -> workstreams 0.
|
|
87
|
+
workstreams --version # -> workstreams 0.6.0
|
|
78
88
|
```
|
|
79
89
|
|
|
80
90
|
> **Note:** every command also accepts `--json` to emit machine-readable output (where supported), which coding agents can parse. All read-side commands work without a multiplexer installed; only `start`/`dispatch`/`work`/`attach` need one.
|
|
@@ -405,6 +415,67 @@ Events are appended to `~/.workstreams/<project>/events.jsonl` using `O_APPEND`
|
|
|
405
415
|
|
|
406
416
|
---
|
|
407
417
|
|
|
418
|
+
## Confidence Scoring
|
|
419
|
+
|
|
420
|
+
Subagents are sometimes wrong. Confidence scoring lets each subagent **self-rate the quality of its result on a 0–10 scale** so the main agent (or a CI gate) can decide whether to accept, re-dispatch, or escalate. It is the difference between "subagent said it's done" and "subagent is 9/10 sure it actually delivered the right result".
|
|
421
|
+
|
|
422
|
+
- The score rides in the event's `data.confidence` field — no new storage required.
|
|
423
|
+
- A **gate** (`workstreams confidence --min-score N`) exits `0` when the latest score is at/above the threshold and `1` otherwise, so it drops straight into CI or a `--wait` loop.
|
|
424
|
+
- Scores are clamped to `[0, 10]`; a misbehaving subagent can't emit `100`.
|
|
425
|
+
|
|
426
|
+
### Emitting a score
|
|
427
|
+
|
|
428
|
+
```bash
|
|
429
|
+
# a subagent reports it finished, 9/10 confident
|
|
430
|
+
workstreams event completed --project myproj --workstream 1 \
|
|
431
|
+
--subagent claude-code --issue 42 \
|
|
432
|
+
--message "All tests green, edge cases covered" \
|
|
433
|
+
--confidence 9
|
|
434
|
+
|
|
435
|
+
# or via --data JSON
|
|
436
|
+
workstreams event completed --project myproj --workstream 1 \
|
|
437
|
+
--subagent codex --issue 42 --message "Done" --data '{"confidence": 7}'
|
|
438
|
+
|
|
439
|
+
# Python API (in a subagent script)
|
|
440
|
+
from workstreams import subagent_report
|
|
441
|
+
subagent_report("myproj", 1, "claude-code", 42, "completed", "Done",
|
|
442
|
+
data={"confidence": 9})
|
|
443
|
+
```
|
|
444
|
+
|
|
445
|
+
### Reading / gating on scores
|
|
446
|
+
|
|
447
|
+
```bash
|
|
448
|
+
# human-readable summary + gate (exit 0 if latest >= 8)
|
|
449
|
+
workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 8
|
|
450
|
+
|
|
451
|
+
# machine-readable
|
|
452
|
+
workstreams confidence --project myproj --workstream 1 --json
|
|
453
|
+
|
|
454
|
+
# list every confidence-bearing event, oldest first
|
|
455
|
+
workstreams confidence --project myproj --workstream 1 --records
|
|
456
|
+
```
|
|
457
|
+
|
|
458
|
+
Example output:
|
|
459
|
+
|
|
460
|
+
```
|
|
461
|
+
Confidence for ws1 #42: PASS
|
|
462
|
+
latest: 9.0/10 (completed from claude-code)
|
|
463
|
+
average: 7.0/10 over 2 report(s)
|
|
464
|
+
threshold: 8.0/10 -> All tests green, edge cases covered
|
|
465
|
+
```
|
|
466
|
+
|
|
467
|
+
### Using it as a CI / re-dispatch gate
|
|
468
|
+
|
|
469
|
+
```bash
|
|
470
|
+
# block the merge until the subagent's latest self-rating clears the bar
|
|
471
|
+
workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 9 \
|
|
472
|
+
|| { echo "confidence too low, re-dispatching"; workstreams dispatch ...; }
|
|
473
|
+
```
|
|
474
|
+
|
|
475
|
+
> Scores are self-reported. Treat them as a *signal*, not a proof — pair them with real test coverage. A 10/10 that shipped a broken build is still a broken build.
|
|
476
|
+
|
|
477
|
+
---
|
|
478
|
+
|
|
408
479
|
## Environment Variables
|
|
409
480
|
|
|
410
481
|
All are overridable in the config file / CLI; env vars are a fallback when neither is set.
|
|
@@ -558,28 +629,34 @@ workstreams logs --workstream 1 --lines 50
|
|
|
558
629
|
|
|
559
630
|
---
|
|
560
631
|
|
|
561
|
-
##
|
|
632
|
+
## Architecture
|
|
562
633
|
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
634
|
+
`workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary.
|
|
635
|
+
|
|
636
|
+

|
|
637
|
+
|
|
638
|
+
The diagram above shows the full data flow: the CLI routes each command to `WorkstreamsManager`, which talks to a pluggable multiplexer backend to place agents into visible terminal panes, while every lane's subagent writes JSONL events back to a shared, concurrent-safe log that the live dashboard re-renders on a timer. The core modules:
|
|
639
|
+
|
|
640
|
+
- **`cli.py`** — argparse entry point; every command accepts `--project` and `--json`
|
|
641
|
+
- **`manager.py`** — `WorkstreamsManager` orchestrates init/start/dispatch/work/sync/pr/merge/run/logs/events
|
|
642
|
+
- **`config.py`** — loads/saves `.workstreams.yaml` (or JSON) with no-pyyaml fallback; resolution order: CLI flag > yaml > env > default
|
|
643
|
+
- **`multiplexer/`** — pluggable backends behind `MultiplexerBase` (`tmux.py`, `zellij.py`, plus tmux-compatible wrappers for `nami`/`lmux`/`wmux`/`herdr`)
|
|
644
|
+
- **`event_log.py` + `subagent_client.py`** — shared cross-process JSONL event stream and the Python API emitters
|
|
645
|
+
- **`notifier.py`** — desktop notifications (`notify-send`/`osascript`/PowerShell) plus a `notifications.jsonl` queue
|
|
646
|
+
- **`confidence.py`** — aggregates subagent self-rated 0–10 quality scores for accept / re-dispatch gating
|
|
647
|
+
- **`dashboard.py`** — the live ANSI TUI that polls events and re-renders every 2s
|
|
648
|
+
- **`models.py`** — dataclasses (`WorkstreamConfig`, `WorkstreamsConfig`, `WorkstreamStatus`, `SubagentEvent`) shared across all of the above
|
|
649
|
+
|
|
650
|
+
How a run flows end-to-end:
|
|
651
|
+
|
|
652
|
+
1. **`init`** creates the lanes: for each workstream it adds a git worktree/branch (`ws/N`) plus a `worktrees/N/` directory, then writes `.workstreams.yaml` in the repo root.
|
|
653
|
+
2. **`start`** asks the chosen multiplexer backend (tmux by default) to open a detached, auto-named session `workstreams-<project>`, one window/tab per lane, each holding a persistent shell.
|
|
654
|
+
3. **`dispatch` / `work` / `run`** target a specific lane's pane (`<session>:<window-name>` for tmux windows, pane index for tiled) and send a command/prompt into it, emitting a `started` event to the shared log and firing a notification.
|
|
655
|
+
4. **Any** process — the agent inside a pane, a CI job, a script, the main terminal — appends JSONL events (`started`, `progress`, `completed`, `failed`, `error`, `done`) to `events.jsonl` using `O_APPEND` plus a short lock that self-heals stale locks after 10s, so concurrent writers never corrupt the stream.
|
|
656
|
+
5. **`monitor`** (the dashboard) and **`events`** read that stream back and re-render every 2s; `--wait` on `dispatch`/`work` blocks until a terminal event arrives (4h safety timeout).
|
|
657
|
+
6. **`sync` / `pr` / `merge` / `workstream cleanup`** close the loop: rebase/merge the lane, push and open a PR via `gh`, merge it, and prune the finished worktree.
|
|
658
|
+
|
|
659
|
+
Read-side commands (`status`, `events`, `logs`, `monitor`) need no multiplexer; only `start`/`dispatch`/`work`/`attach` require one to be installed.
|
|
583
660
|
|
|
584
661
|
---
|
|
585
662
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "workstreams-cli"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.6.0"
|
|
8
8
|
description = "Visually dispatch coding-agent work to subagents in real terminal windows and monitor it in one dashboard - for any coding agent (Claude Code, Codex, OpenCode, Qwen Code, Hermes, Cline, and more)."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -33,8 +33,16 @@ from .subagent_client import (
|
|
|
33
33
|
subagent_done,
|
|
34
34
|
)
|
|
35
35
|
from .multiplexer import MultiplexerBase, TmuxMultiplexer, ZellijMultiplexer, get_multiplexer
|
|
36
|
+
from .confidence import (
|
|
37
|
+
ConfidenceRecord,
|
|
38
|
+
ConfidenceSummary,
|
|
39
|
+
get_confidence,
|
|
40
|
+
confidence_records,
|
|
41
|
+
clamp_score,
|
|
42
|
+
extract_score,
|
|
43
|
+
)
|
|
36
44
|
|
|
37
|
-
__version__ = "0.
|
|
45
|
+
__version__ = "0.6.0"
|
|
38
46
|
|
|
39
47
|
__all__ = [
|
|
40
48
|
# models
|
|
@@ -68,4 +76,11 @@ __all__ = [
|
|
|
68
76
|
"TmuxMultiplexer",
|
|
69
77
|
"ZellijMultiplexer",
|
|
70
78
|
"get_multiplexer",
|
|
79
|
+
# confidence scoring
|
|
80
|
+
"ConfidenceRecord",
|
|
81
|
+
"ConfidenceSummary",
|
|
82
|
+
"get_confidence",
|
|
83
|
+
"confidence_records",
|
|
84
|
+
"clamp_score",
|
|
85
|
+
"extract_score",
|
|
71
86
|
]
|
|
@@ -208,8 +208,18 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
208
208
|
event.add_argument("--issue", type=int, default=0)
|
|
209
209
|
event.add_argument("--message", default="")
|
|
210
210
|
event.add_argument("--data", help="JSON object of extra data (e.g. '{\"files\": 3}')")
|
|
211
|
+
event.add_argument("--confidence", type=float, default=None, help="Self-rated result quality 0-10 (stored in data.confidence)")
|
|
211
212
|
event.add_argument("--json", action="store_true")
|
|
212
213
|
|
|
214
|
+
# confidence (aggregate / gate on subagent self-ratings)
|
|
215
|
+
confidence = common_parent("confidence")
|
|
216
|
+
confidence.add_argument("--workstream", type=int, help="Restrict to one workstream")
|
|
217
|
+
confidence.add_argument("--issue", type=int, help="Restrict to one issue")
|
|
218
|
+
confidence.add_argument("--subagent", help="Filter by subagent name")
|
|
219
|
+
confidence.add_argument("--since", type=int, default=43200, help="Minutes back (default: 30 days)")
|
|
220
|
+
confidence.add_argument("--min-score", type=float, default=8.0, help="Pass threshold for the gate (default: 8.0)")
|
|
221
|
+
confidence.add_argument("--records", action="store_true", help="List individual confidence-bearing events")
|
|
222
|
+
|
|
213
223
|
# notify
|
|
214
224
|
notify = common_parent("notify")
|
|
215
225
|
notify.add_argument("--title", required=True)
|
|
@@ -412,6 +422,8 @@ def _cmd_event(args) -> int:
|
|
|
412
422
|
except json.JSONDecodeError:
|
|
413
423
|
print(f"Invalid --data JSON: {args.data}", file=sys.stderr)
|
|
414
424
|
return 2
|
|
425
|
+
if getattr(args, "confidence", None) is not None:
|
|
426
|
+
data["confidence"] = args.confidence
|
|
415
427
|
ok = subagent_report(
|
|
416
428
|
project=args.project,
|
|
417
429
|
workstream_id=args.workstream,
|
|
@@ -431,6 +443,73 @@ def _cmd_event(args) -> int:
|
|
|
431
443
|
return 0 if ok else 1
|
|
432
444
|
|
|
433
445
|
|
|
446
|
+
def _cmd_confidence(args) -> int:
|
|
447
|
+
"""Aggregate / gate on subagent self-rated confidence scores."""
|
|
448
|
+
from .confidence import get_confidence, confidence_records, clamp_score
|
|
449
|
+
|
|
450
|
+
# clamp the threshold so a bogus --min-score can't break the gate
|
|
451
|
+
min_score = clamp_score(args.min_score)
|
|
452
|
+
if min_score is None:
|
|
453
|
+
print(f"Invalid --min-score: {args.min_score}", file=sys.stderr)
|
|
454
|
+
return 2
|
|
455
|
+
|
|
456
|
+
# The event log is keyed by project, but confidence read commands are
|
|
457
|
+
# project-scoped like the rest of the read side; fall back to .workstreams.yaml
|
|
458
|
+
# if no --project was given.
|
|
459
|
+
import os
|
|
460
|
+
from .config import load_config
|
|
461
|
+
project = getattr(args, "project", None)
|
|
462
|
+
if not project:
|
|
463
|
+
cfg = load_config(Path.cwd())
|
|
464
|
+
project = cfg.project
|
|
465
|
+
|
|
466
|
+
if args.records:
|
|
467
|
+
records = confidence_records(
|
|
468
|
+
project=project,
|
|
469
|
+
workstream_id=args.workstream,
|
|
470
|
+
issue=args.issue,
|
|
471
|
+
subagent=args.subagent,
|
|
472
|
+
since_minutes=args.since,
|
|
473
|
+
)
|
|
474
|
+
if args.json:
|
|
475
|
+
print(json.dumps([r.to_dict() for r in records], indent=2, default=str))
|
|
476
|
+
return 0
|
|
477
|
+
if not records:
|
|
478
|
+
print(f"(no confidence reports in last {args.since} min)")
|
|
479
|
+
return 0
|
|
480
|
+
for r in records:
|
|
481
|
+
ts = r.timestamp[11:19]
|
|
482
|
+
print(f"[{ts}] [{r.subagent}] ws{r.workstream_id} #{r.issue} - {r.event_type}: {r.score}/10 {r.message}")
|
|
483
|
+
return 0
|
|
484
|
+
|
|
485
|
+
summary = get_confidence(
|
|
486
|
+
project=project,
|
|
487
|
+
workstream_id=args.workstream,
|
|
488
|
+
issue=args.issue,
|
|
489
|
+
subagent=args.subagent,
|
|
490
|
+
since_minutes=args.since,
|
|
491
|
+
min_score=min_score,
|
|
492
|
+
)
|
|
493
|
+
if args.json:
|
|
494
|
+
print(json.dumps(summary.to_dict(), indent=2, default=str))
|
|
495
|
+
return 0 if summary.passed else 1
|
|
496
|
+
|
|
497
|
+
scope = f"ws{args.workstream}" if args.workstream else "project"
|
|
498
|
+
if args.issue:
|
|
499
|
+
scope += f" #{args.issue}"
|
|
500
|
+
if summary.n_reports == 0:
|
|
501
|
+
print(f"No confidence reports for {scope} yet.")
|
|
502
|
+
return 1
|
|
503
|
+
avg = f"{summary.avg_score:.1f}" if summary.avg_score is not None else "n/a"
|
|
504
|
+
latest = f"{summary.latest_score:.1f}" if summary.latest_score is not None else "n/a"
|
|
505
|
+
mark = "PASS" if summary.passed else "FAIL"
|
|
506
|
+
print(f"Confidence for {scope}: {mark}")
|
|
507
|
+
print(f" latest: {latest}/10 ({summary.latest_event_type} from {summary.latest_subagent})")
|
|
508
|
+
print(f" average: {avg}/10 over {summary.n_reports} report(s)")
|
|
509
|
+
print(f" threshold: {min_score}/10 -> {summary.latest_message}")
|
|
510
|
+
return 0 if summary.passed else 1
|
|
511
|
+
|
|
512
|
+
|
|
434
513
|
def _cmd_notify(args) -> int:
|
|
435
514
|
manager = _load_manager(args)
|
|
436
515
|
manager.notify(args.title, args.message, args.urgency)
|
|
@@ -472,6 +551,7 @@ HANDLERS = {
|
|
|
472
551
|
"tail": _cmd_tail,
|
|
473
552
|
"events": _cmd_events,
|
|
474
553
|
"event": _cmd_event,
|
|
554
|
+
"confidence": _cmd_confidence,
|
|
475
555
|
"notify": _cmd_notify,
|
|
476
556
|
"assign": _cmd_assign,
|
|
477
557
|
"sync": _cmd_sync,
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""Confidence scoring — let subagents self-rate result quality.
|
|
2
|
+
|
|
3
|
+
A subagent can attach a confidence score (0-10) to any event, signalling
|
|
4
|
+
how sure it is that the delivered result meets the required bar. The main
|
|
5
|
+
agent (or a CI gate) can then query the latest / aggregate confidence and
|
|
6
|
+
decide whether to accept, re-dispatch, or escalate.
|
|
7
|
+
|
|
8
|
+
# subagent reports a high-confidence completion
|
|
9
|
+
workstreams event completed --project p --workstream 1 --subagent claude-code \
|
|
10
|
+
--issue 42 --message "All tests green" --confidence 9
|
|
11
|
+
|
|
12
|
+
# main agent / CI gate: pass only if latest confidence >= 8
|
|
13
|
+
workstreams confidence --project p --workstream 1 --issue 42 --min-score 8
|
|
14
|
+
# exit 0 -> gate passed, exit 1 -> below threshold
|
|
15
|
+
|
|
16
|
+
Scores are read straight from the event log's `data.confidence` field, so
|
|
17
|
+
no new storage is required.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from dataclasses import dataclass, field, asdict
|
|
23
|
+
from typing import Any, Dict, List, Optional
|
|
24
|
+
|
|
25
|
+
from .event_log import get_event_log
|
|
26
|
+
from .models import SubagentEvent
|
|
27
|
+
|
|
28
|
+
# Confidence scale is 0-10. Anything at/above this is treated as a
|
|
29
|
+
# "high confidence" result by default.
|
|
30
|
+
DEFAULT_MIN_SCORE = 8
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def clamp_score(value: Any) -> Optional[float]:
|
|
34
|
+
"""Normalise a confidence value to a float in [0, 10], or None.
|
|
35
|
+
|
|
36
|
+
Accepts ints, floats, and numeric strings. Values are clamped to
|
|
37
|
+
[0, 10] so a misbehaving subagent can't emit e.g. 100.
|
|
38
|
+
"""
|
|
39
|
+
if value is None:
|
|
40
|
+
return None
|
|
41
|
+
try:
|
|
42
|
+
score = float(value)
|
|
43
|
+
except (TypeError, ValueError):
|
|
44
|
+
return None
|
|
45
|
+
if score < 0:
|
|
46
|
+
return 0.0
|
|
47
|
+
if score > 10:
|
|
48
|
+
return 10.0
|
|
49
|
+
return score
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class ConfidenceRecord:
|
|
54
|
+
"""One confidence-bearing event, normalised for display."""
|
|
55
|
+
|
|
56
|
+
workstream_id: int
|
|
57
|
+
subagent: str
|
|
58
|
+
issue: int
|
|
59
|
+
event_type: str
|
|
60
|
+
score: Optional[float]
|
|
61
|
+
message: str
|
|
62
|
+
timestamp: str
|
|
63
|
+
data: Dict[str, Any] = field(default_factory=dict)
|
|
64
|
+
|
|
65
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
66
|
+
return asdict(self)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class ConfidenceSummary:
|
|
71
|
+
"""Aggregated confidence view for a workstream (optionally per issue)."""
|
|
72
|
+
|
|
73
|
+
workstream_id: int
|
|
74
|
+
latest_score: Optional[float]
|
|
75
|
+
latest_event_type: str
|
|
76
|
+
latest_subagent: str
|
|
77
|
+
latest_message: str
|
|
78
|
+
avg_score: Optional[float]
|
|
79
|
+
n_reports: int
|
|
80
|
+
passed: bool
|
|
81
|
+
min_score: float
|
|
82
|
+
|
|
83
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
84
|
+
return asdict(self)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def extract_score(event: SubagentEvent) -> Optional[float]:
|
|
88
|
+
"""Pull a confidence score out of an event (data.confidence preferred)."""
|
|
89
|
+
score = event.data.get("confidence")
|
|
90
|
+
if score is None:
|
|
91
|
+
# also accept top-level 'score' in data for convenience
|
|
92
|
+
score = event.data.get("score")
|
|
93
|
+
return clamp_score(score)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def get_confidence(
|
|
97
|
+
project: str,
|
|
98
|
+
workstream_id: Optional[int] = None,
|
|
99
|
+
issue: Optional[int] = None,
|
|
100
|
+
subagent: Optional[str] = None,
|
|
101
|
+
since_minutes: int = 60 * 24 * 30, # 30 days
|
|
102
|
+
min_score: float = DEFAULT_MIN_SCORE,
|
|
103
|
+
) -> ConfidenceSummary:
|
|
104
|
+
"""Aggregate confidence for a workstream (and optional issue/subagent).
|
|
105
|
+
|
|
106
|
+
Only events that carry a confidence score are counted. `passed` is True
|
|
107
|
+
when the *latest* score is at or above `min_score` and at least one
|
|
108
|
+
scored report exists.
|
|
109
|
+
"""
|
|
110
|
+
from datetime import datetime, UTC, timedelta
|
|
111
|
+
|
|
112
|
+
event_log = get_event_log(project)
|
|
113
|
+
since = datetime.now(UTC) - timedelta(minutes=since_minutes)
|
|
114
|
+
|
|
115
|
+
events = event_log.get_events(
|
|
116
|
+
workstream_id=workstream_id,
|
|
117
|
+
since=since,
|
|
118
|
+
subagent=subagent,
|
|
119
|
+
)
|
|
120
|
+
if issue is not None:
|
|
121
|
+
events = [e for e in events if e.issue == issue]
|
|
122
|
+
|
|
123
|
+
scored = [e for e in events if extract_score(e) is not None]
|
|
124
|
+
|
|
125
|
+
latest_score: Optional[float] = None
|
|
126
|
+
latest_event_type = ""
|
|
127
|
+
latest_subagent = ""
|
|
128
|
+
latest_message = ""
|
|
129
|
+
total = 0.0
|
|
130
|
+
n = 0
|
|
131
|
+
|
|
132
|
+
for e in scored:
|
|
133
|
+
s = extract_score(e)
|
|
134
|
+
total += s or 0.0
|
|
135
|
+
n += 1
|
|
136
|
+
# events are oldest->newest, so the last scored event is the latest
|
|
137
|
+
latest_score = s
|
|
138
|
+
latest_event_type = e.event_type
|
|
139
|
+
latest_subagent = e.subagent
|
|
140
|
+
latest_message = e.message
|
|
141
|
+
|
|
142
|
+
avg_score = (total / n) if n else None
|
|
143
|
+
passed = (
|
|
144
|
+
latest_score is not None
|
|
145
|
+
and latest_score >= min_score
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
return ConfidenceSummary(
|
|
149
|
+
workstream_id=workstream_id or 0,
|
|
150
|
+
latest_score=latest_score,
|
|
151
|
+
latest_event_type=latest_event_type,
|
|
152
|
+
latest_subagent=latest_subagent,
|
|
153
|
+
latest_message=latest_message,
|
|
154
|
+
avg_score=avg_score,
|
|
155
|
+
n_reports=n,
|
|
156
|
+
passed=passed,
|
|
157
|
+
min_score=min_score,
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def confidence_records(
|
|
162
|
+
project: str,
|
|
163
|
+
workstream_id: Optional[int] = None,
|
|
164
|
+
issue: Optional[int] = None,
|
|
165
|
+
subagent: Optional[str] = None,
|
|
166
|
+
since_minutes: int = 60 * 24 * 30,
|
|
167
|
+
) -> List[ConfidenceRecord]:
|
|
168
|
+
"""Return all confidence-bearing events, oldest first."""
|
|
169
|
+
from datetime import datetime, UTC, timedelta
|
|
170
|
+
|
|
171
|
+
event_log = get_event_log(project)
|
|
172
|
+
since = datetime.now(UTC) - timedelta(minutes=since_minutes)
|
|
173
|
+
events = event_log.get_events(
|
|
174
|
+
workstream_id=workstream_id, since=since, subagent=subagent
|
|
175
|
+
)
|
|
176
|
+
if issue is not None:
|
|
177
|
+
events = [e for e in events if e.issue == issue]
|
|
178
|
+
|
|
179
|
+
out: List[ConfidenceRecord] = []
|
|
180
|
+
for e in events:
|
|
181
|
+
s = extract_score(e)
|
|
182
|
+
if s is None:
|
|
183
|
+
continue
|
|
184
|
+
out.append(
|
|
185
|
+
ConfidenceRecord(
|
|
186
|
+
workstream_id=e.workstream_id,
|
|
187
|
+
subagent=e.subagent,
|
|
188
|
+
issue=e.issue,
|
|
189
|
+
event_type=e.event_type,
|
|
190
|
+
score=s,
|
|
191
|
+
message=e.message,
|
|
192
|
+
timestamp=e.timestamp,
|
|
193
|
+
data=e.data,
|
|
194
|
+
)
|
|
195
|
+
)
|
|
196
|
+
return out
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: workstreams-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Visually dispatch coding-agent work to subagents in real terminal windows and monitor it in one dashboard - for any coding agent (Claude Code, Codex, OpenCode, Qwen Code, Hermes, Cline, and more).
|
|
5
5
|
Author: Dream-Pixels-Forge
|
|
6
6
|
License: MIT
|
|
@@ -45,6 +45,14 @@ Key capabilities:
|
|
|
45
45
|
- **Cross-terminal notifications** — desktop notifications (Linux `notify-send`, macOS `osascript`) plus a shared `notifications.jsonl` that other terminals can poll
|
|
46
46
|
- **Agent-agnostic** — no vendor lock-in. `dispatch` and `work` send arbitrary shell commands to panes, so it works with whatever agent binary you can run from a shell
|
|
47
47
|
|
|
48
|
+
## ⭐ If workstreams helps you ship faster, star the repo
|
|
49
|
+
|
|
50
|
+
If you found this useful, a GitHub star helps other developers discover it.
|
|
51
|
+
|
|
52
|
+
[](https://github.com/Dream-Pixels-Forge/workstreams-cli)
|
|
53
|
+
|
|
54
|
+
⭐ **Star this repo:** [github.com/Dream-Pixels-Forge/workstreams-cli](https://github.com/Dream-Pixels-Forge/workstreams-cli)
|
|
55
|
+
|
|
48
56
|
## Why this exists
|
|
49
57
|
|
|
50
58
|
Coding agents increasingly support "subagents" that run in the background of the main agent's process. That means: no visibility (you can't watch them), no isolation (they share one working tree and one set of installed dependencies), no way to run several in parallel on independent branches, and no shared event stream you can watch from your main terminal.
|
|
@@ -61,6 +69,7 @@ Coding agents increasingly support "subagents" that run in the background of the
|
|
|
61
69
|
- [Configuration (.workstreams.yaml)](#configuration-workstreamsyaml)
|
|
62
70
|
- [How the Multiplexers Work](#how-the-multiplexers-work)
|
|
63
71
|
- [Subagent Event System (Python API + CLI)](#subagent-event-system)
|
|
72
|
+
- [Confidence Scoring](#confidence-scoring)
|
|
64
73
|
- [Environment Variables](#environment-variables)
|
|
65
74
|
- [Exit Codes](#exit-codes)
|
|
66
75
|
- [Data Locations](#data-locations)
|
|
@@ -68,6 +77,7 @@ Coding agents increasingly support "subagents" that run in the background of the
|
|
|
68
77
|
- [CI/CD Integration](#cicd-integration)
|
|
69
78
|
- [Troubleshooting](#troubleshooting)
|
|
70
79
|
- [Best Practices](#best-practices)
|
|
80
|
+
- [Architecture](#architecture)
|
|
71
81
|
- [Contributing & License](#contributing--license)
|
|
72
82
|
|
|
73
83
|
---
|
|
@@ -104,7 +114,7 @@ pip install -e ".[yaml,dev]" # dev extras add pytest
|
|
|
104
114
|
Verify:
|
|
105
115
|
|
|
106
116
|
```bash
|
|
107
|
-
workstreams --version # -> workstreams 0.
|
|
117
|
+
workstreams --version # -> workstreams 0.6.0
|
|
108
118
|
```
|
|
109
119
|
|
|
110
120
|
> **Note:** every command also accepts `--json` to emit machine-readable output (where supported), which coding agents can parse. All read-side commands work without a multiplexer installed; only `start`/`dispatch`/`work`/`attach` need one.
|
|
@@ -435,6 +445,67 @@ Events are appended to `~/.workstreams/<project>/events.jsonl` using `O_APPEND`
|
|
|
435
445
|
|
|
436
446
|
---
|
|
437
447
|
|
|
448
|
+
## Confidence Scoring
|
|
449
|
+
|
|
450
|
+
Subagents are sometimes wrong. Confidence scoring lets each subagent **self-rate the quality of its result on a 0–10 scale** so the main agent (or a CI gate) can decide whether to accept, re-dispatch, or escalate. It is the difference between "subagent said it's done" and "subagent is 9/10 sure it actually delivered the right result".
|
|
451
|
+
|
|
452
|
+
- The score rides in the event's `data.confidence` field — no new storage required.
|
|
453
|
+
- A **gate** (`workstreams confidence --min-score N`) exits `0` when the latest score is at/above the threshold and `1` otherwise, so it drops straight into CI or a `--wait` loop.
|
|
454
|
+
- Scores are clamped to `[0, 10]`; a misbehaving subagent can't emit `100`.
|
|
455
|
+
|
|
456
|
+
### Emitting a score
|
|
457
|
+
|
|
458
|
+
```bash
|
|
459
|
+
# a subagent reports it finished, 9/10 confident
|
|
460
|
+
workstreams event completed --project myproj --workstream 1 \
|
|
461
|
+
--subagent claude-code --issue 42 \
|
|
462
|
+
--message "All tests green, edge cases covered" \
|
|
463
|
+
--confidence 9
|
|
464
|
+
|
|
465
|
+
# or via --data JSON
|
|
466
|
+
workstreams event completed --project myproj --workstream 1 \
|
|
467
|
+
--subagent codex --issue 42 --message "Done" --data '{"confidence": 7}'
|
|
468
|
+
|
|
469
|
+
# Python API (in a subagent script)
|
|
470
|
+
from workstreams import subagent_report
|
|
471
|
+
subagent_report("myproj", 1, "claude-code", 42, "completed", "Done",
|
|
472
|
+
data={"confidence": 9})
|
|
473
|
+
```
|
|
474
|
+
|
|
475
|
+
### Reading / gating on scores
|
|
476
|
+
|
|
477
|
+
```bash
|
|
478
|
+
# human-readable summary + gate (exit 0 if latest >= 8)
|
|
479
|
+
workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 8
|
|
480
|
+
|
|
481
|
+
# machine-readable
|
|
482
|
+
workstreams confidence --project myproj --workstream 1 --json
|
|
483
|
+
|
|
484
|
+
# list every confidence-bearing event, oldest first
|
|
485
|
+
workstreams confidence --project myproj --workstream 1 --records
|
|
486
|
+
```
|
|
487
|
+
|
|
488
|
+
Example output:
|
|
489
|
+
|
|
490
|
+
```
|
|
491
|
+
Confidence for ws1 #42: PASS
|
|
492
|
+
latest: 9.0/10 (completed from claude-code)
|
|
493
|
+
average: 7.0/10 over 2 report(s)
|
|
494
|
+
threshold: 8.0/10 -> All tests green, edge cases covered
|
|
495
|
+
```
|
|
496
|
+
|
|
497
|
+
### Using it as a CI / re-dispatch gate
|
|
498
|
+
|
|
499
|
+
```bash
|
|
500
|
+
# block the merge until the subagent's latest self-rating clears the bar
|
|
501
|
+
workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 9 \
|
|
502
|
+
|| { echo "confidence too low, re-dispatching"; workstreams dispatch ...; }
|
|
503
|
+
```
|
|
504
|
+
|
|
505
|
+
> Scores are self-reported. Treat them as a *signal*, not a proof — pair them with real test coverage. A 10/10 that shipped a broken build is still a broken build.
|
|
506
|
+
|
|
507
|
+
---
|
|
508
|
+
|
|
438
509
|
## Environment Variables
|
|
439
510
|
|
|
440
511
|
All are overridable in the config file / CLI; env vars are a fallback when neither is set.
|
|
@@ -588,28 +659,34 @@ workstreams logs --workstream 1 --lines 50
|
|
|
588
659
|
|
|
589
660
|
---
|
|
590
661
|
|
|
591
|
-
##
|
|
662
|
+
## Architecture
|
|
592
663
|
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
664
|
+
`workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary.
|
|
665
|
+
|
|
666
|
+

|
|
667
|
+
|
|
668
|
+
The diagram above shows the full data flow: the CLI routes each command to `WorkstreamsManager`, which talks to a pluggable multiplexer backend to place agents into visible terminal panes, while every lane's subagent writes JSONL events back to a shared, concurrent-safe log that the live dashboard re-renders on a timer. The core modules:
|
|
669
|
+
|
|
670
|
+
- **`cli.py`** — argparse entry point; every command accepts `--project` and `--json`
|
|
671
|
+
- **`manager.py`** — `WorkstreamsManager` orchestrates init/start/dispatch/work/sync/pr/merge/run/logs/events
|
|
672
|
+
- **`config.py`** — loads/saves `.workstreams.yaml` (or JSON) with no-pyyaml fallback; resolution order: CLI flag > yaml > env > default
|
|
673
|
+
- **`multiplexer/`** — pluggable backends behind `MultiplexerBase` (`tmux.py`, `zellij.py`, plus tmux-compatible wrappers for `nami`/`lmux`/`wmux`/`herdr`)
|
|
674
|
+
- **`event_log.py` + `subagent_client.py`** — shared cross-process JSONL event stream and the Python API emitters
|
|
675
|
+
- **`notifier.py`** — desktop notifications (`notify-send`/`osascript`/PowerShell) plus a `notifications.jsonl` queue
|
|
676
|
+
- **`confidence.py`** — aggregates subagent self-rated 0–10 quality scores for accept / re-dispatch gating
|
|
677
|
+
- **`dashboard.py`** — the live ANSI TUI that polls events and re-renders every 2s
|
|
678
|
+
- **`models.py`** — dataclasses (`WorkstreamConfig`, `WorkstreamsConfig`, `WorkstreamStatus`, `SubagentEvent`) shared across all of the above
|
|
679
|
+
|
|
680
|
+
How a run flows end-to-end:
|
|
681
|
+
|
|
682
|
+
1. **`init`** creates the lanes: for each workstream it adds a git worktree/branch (`ws/N`) plus a `worktrees/N/` directory, then writes `.workstreams.yaml` in the repo root.
|
|
683
|
+
2. **`start`** asks the chosen multiplexer backend (tmux by default) to open a detached, auto-named session `workstreams-<project>`, one window/tab per lane, each holding a persistent shell.
|
|
684
|
+
3. **`dispatch` / `work` / `run`** target a specific lane's pane (`<session>:<window-name>` for tmux windows, pane index for tiled) and send a command/prompt into it, emitting a `started` event to the shared log and firing a notification.
|
|
685
|
+
4. **Any** process — the agent inside a pane, a CI job, a script, the main terminal — appends JSONL events (`started`, `progress`, `completed`, `failed`, `error`, `done`) to `events.jsonl` using `O_APPEND` plus a short lock that self-heals stale locks after 10s, so concurrent writers never corrupt the stream.
|
|
686
|
+
5. **`monitor`** (the dashboard) and **`events`** read that stream back and re-render every 2s; `--wait` on `dispatch`/`work` blocks until a terminal event arrives (4h safety timeout).
|
|
687
|
+
6. **`sync` / `pr` / `merge` / `workstream cleanup`** close the loop: rebase/merge the lane, push and open a PR via `gh`, merge it, and prune the finished worktree.
|
|
688
|
+
|
|
689
|
+
Read-side commands (`status`, `events`, `logs`, `monitor`) need no multiplexer; only `start`/`dispatch`/`work`/`attach` require one to be installed.
|
|
613
690
|
|
|
614
691
|
---
|
|
615
692
|
|
|
@@ -2,6 +2,7 @@ README.md
|
|
|
2
2
|
pyproject.toml
|
|
3
3
|
src/workstreams/__init__.py
|
|
4
4
|
src/workstreams/cli.py
|
|
5
|
+
src/workstreams/confidence.py
|
|
5
6
|
src/workstreams/config.py
|
|
6
7
|
src/workstreams/dashboard.py
|
|
7
8
|
src/workstreams/event_log.py
|
|
@@ -21,6 +22,7 @@ src/workstreams_cli.egg-info/dependency_links.txt
|
|
|
21
22
|
src/workstreams_cli.egg-info/entry_points.txt
|
|
22
23
|
src/workstreams_cli.egg-info/requires.txt
|
|
23
24
|
src/workstreams_cli.egg-info/top_level.txt
|
|
25
|
+
tests/test_confidence.py
|
|
24
26
|
tests/test_config.py
|
|
25
27
|
tests/test_event_log.py
|
|
26
28
|
tests/test_models.py
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""Tests for confidence scoring (0-10 self-rated result quality)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
import pytest
|
|
11
|
+
|
|
12
|
+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))
|
|
13
|
+
|
|
14
|
+
from workstreams.confidence import (
|
|
15
|
+
clamp_score,
|
|
16
|
+
confidence_records,
|
|
17
|
+
extract_score,
|
|
18
|
+
get_confidence,
|
|
19
|
+
)
|
|
20
|
+
from workstreams.models import SubagentEvent
|
|
21
|
+
from workstreams.event_log import EventLog
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _mk_event(score=None, subagent="claude-code", issue=7, event_type="completed", ws=1):
|
|
25
|
+
data = {}
|
|
26
|
+
if score is not None:
|
|
27
|
+
data["confidence"] = score
|
|
28
|
+
return SubagentEvent(
|
|
29
|
+
workstream_id=ws,
|
|
30
|
+
subagent=subagent,
|
|
31
|
+
issue=issue,
|
|
32
|
+
event_type=event_type,
|
|
33
|
+
message="msg",
|
|
34
|
+
data=data,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
# ---------------------------------------------------------------------------
|
|
39
|
+
# clamp_score
|
|
40
|
+
# ---------------------------------------------------------------------------
|
|
41
|
+
|
|
42
|
+
def test_clamp_score_none():
|
|
43
|
+
assert clamp_score(None) is None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_clamp_score_valid_range():
|
|
47
|
+
assert clamp_score(5) == 5.0
|
|
48
|
+
assert clamp_score("7.5") == 7.5
|
|
49
|
+
assert clamp_score(0) == 0.0
|
|
50
|
+
assert clamp_score(10) == 10.0
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_clamp_score_clamps_out_of_range():
|
|
54
|
+
assert clamp_score(-3) == 0.0
|
|
55
|
+
assert clamp_score(100) == 10.0
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_clamp_score_garbage_is_none():
|
|
59
|
+
assert clamp_score("not-a-number") is None
|
|
60
|
+
assert clamp_score([1, 2]) is None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# ---------------------------------------------------------------------------
|
|
64
|
+
# extract_score
|
|
65
|
+
# ---------------------------------------------------------------------------
|
|
66
|
+
|
|
67
|
+
def test_extract_score_reads_confidence():
|
|
68
|
+
assert extract_score(_mk_event(score=8)) == 8.0
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_extract_score_reads_score_key_too():
|
|
72
|
+
e = SubagentEvent(workstream_id=1, subagent="x", issue=1, event_type="completed", message="", data={"score": 6})
|
|
73
|
+
assert extract_score(e) == 6.0
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def test_extract_score_missing_returns_none():
|
|
77
|
+
assert extract_score(_mk_event(score=None)) is None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
# ---------------------------------------------------------------------------
|
|
81
|
+
# get_confidence aggregation
|
|
82
|
+
# ---------------------------------------------------------------------------
|
|
83
|
+
|
|
84
|
+
def test_get_confidence_pass(tmp_path, monkeypatch):
|
|
85
|
+
monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
|
|
86
|
+
from workstreams import event_log
|
|
87
|
+
event_log._log_cache.clear()
|
|
88
|
+
|
|
89
|
+
log = EventLog("p", log_dir=tmp_path / "p")
|
|
90
|
+
log.append(_mk_event(score=5))
|
|
91
|
+
log.append(_mk_event(score=9))
|
|
92
|
+
|
|
93
|
+
summary = get_confidence("p", workstream_id=1, issue=7, min_score=8)
|
|
94
|
+
assert summary.passed is True
|
|
95
|
+
assert summary.latest_score == 9.0
|
|
96
|
+
assert summary.n_reports == 2
|
|
97
|
+
assert summary.avg_score == pytest.approx(7.0)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def test_get_confidence_fail(tmp_path, monkeypatch):
|
|
101
|
+
monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
|
|
102
|
+
from workstreams import event_log
|
|
103
|
+
event_log._log_cache.clear()
|
|
104
|
+
|
|
105
|
+
log = EventLog("p", log_dir=tmp_path / "p")
|
|
106
|
+
log.append(_mk_event(score=3))
|
|
107
|
+
|
|
108
|
+
summary = get_confidence("p", workstream_id=1, issue=7, min_score=8)
|
|
109
|
+
assert summary.passed is False
|
|
110
|
+
assert summary.latest_score == 3.0
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def test_get_confidence_no_reports(tmp_path, monkeypatch):
|
|
114
|
+
monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
|
|
115
|
+
from workstreams import event_log
|
|
116
|
+
event_log._log_cache.clear()
|
|
117
|
+
|
|
118
|
+
log = EventLog("p", log_dir=tmp_path / "p")
|
|
119
|
+
log.append(SubagentEvent(workstream_id=1, subagent="x", issue=7, event_type="started", message="", data={}))
|
|
120
|
+
|
|
121
|
+
summary = get_confidence("p", workstream_id=1, issue=7, min_score=8)
|
|
122
|
+
assert summary.n_reports == 0
|
|
123
|
+
assert summary.passed is False
|
|
124
|
+
assert summary.latest_score is None
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def test_get_confidence_filters_unscored_events(tmp_path, monkeypatch):
|
|
128
|
+
monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
|
|
129
|
+
from workstreams import event_log
|
|
130
|
+
event_log._log_cache.clear()
|
|
131
|
+
|
|
132
|
+
log = EventLog("p", log_dir=tmp_path / "p")
|
|
133
|
+
log.append(_mk_event(score=None)) # no confidence -> ignored
|
|
134
|
+
log.append(_mk_event(score=10, subagent="codex"))
|
|
135
|
+
|
|
136
|
+
summary = get_confidence("p", workstream_id=1, issue=7, min_score=8)
|
|
137
|
+
assert summary.n_reports == 1
|
|
138
|
+
assert summary.latest_subagent == "codex"
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def test_get_confidence_issue_filter(tmp_path, monkeypatch):
|
|
142
|
+
monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
|
|
143
|
+
from workstreams import event_log
|
|
144
|
+
event_log._log_cache.clear()
|
|
145
|
+
|
|
146
|
+
log = EventLog("p", log_dir=tmp_path / "p")
|
|
147
|
+
log.append(SubagentEvent(workstream_id=1, subagent="a", issue=1, event_type="completed", message="", data={"confidence": 2}))
|
|
148
|
+
log.append(SubagentEvent(workstream_id=1, subagent="a", issue=2, event_type="completed", message="", data={"confidence": 9}))
|
|
149
|
+
|
|
150
|
+
s1 = get_confidence("p", workstream_id=1, issue=1, min_score=8)
|
|
151
|
+
s2 = get_confidence("p", workstream_id=1, issue=2, min_score=8)
|
|
152
|
+
assert s1.latest_score == 2.0 and s1.passed is False
|
|
153
|
+
assert s2.latest_score == 9.0 and s2.passed is True
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
# ---------------------------------------------------------------------------
|
|
157
|
+
# confidence_records
|
|
158
|
+
# ---------------------------------------------------------------------------
|
|
159
|
+
|
|
160
|
+
def test_confidence_records_oldest_first(tmp_path, monkeypatch):
|
|
161
|
+
monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
|
|
162
|
+
from workstreams import event_log
|
|
163
|
+
event_log._log_cache.clear()
|
|
164
|
+
|
|
165
|
+
log = EventLog("p", log_dir=tmp_path / "p")
|
|
166
|
+
log.append(_mk_event(score=1, subagent="a"))
|
|
167
|
+
log.append(_mk_event(score=5, subagent="b"))
|
|
168
|
+
log.append(_mk_event(score=9, subagent="c"))
|
|
169
|
+
|
|
170
|
+
records = confidence_records("p", workstream_id=1, issue=7)
|
|
171
|
+
assert [r.subagent for r in records] == ["a", "b", "c"]
|
|
172
|
+
assert [r.score for r in records] == [1.0, 5.0, 9.0]
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def test_confidence_records_excludes_unscored(tmp_path, monkeypatch):
|
|
176
|
+
monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
|
|
177
|
+
from workstreams import event_log
|
|
178
|
+
event_log._log_cache.clear()
|
|
179
|
+
|
|
180
|
+
log = EventLog("p", log_dir=tmp_path / "p")
|
|
181
|
+
log.append(_mk_event(score=None, subagent="no-score"))
|
|
182
|
+
log.append(_mk_event(score=7, subagent="scored"))
|
|
183
|
+
|
|
184
|
+
records = confidence_records("p", workstream_id=1, issue=7)
|
|
185
|
+
assert [r.subagent for r in records] == ["scored"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams/multiplexer/tmux_compatible.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams_cli.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{workstreams_cli-0.5.0 → workstreams_cli-0.6.0}/src/workstreams_cli.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|