workstreams-cli 0.5.1__tar.gz → 0.6.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/PKG-INFO +86 -52
  2. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/README.md +85 -51
  3. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/pyproject.toml +1 -1
  4. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/__init__.py +16 -1
  5. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/cli.py +80 -0
  6. workstreams_cli-0.6.1/src/workstreams/confidence.py +196 -0
  7. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/config.py +15 -6
  8. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/multiplexer/tmux.py +6 -0
  9. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams_cli.egg-info/PKG-INFO +86 -52
  10. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams_cli.egg-info/SOURCES.txt +2 -0
  11. workstreams_cli-0.6.1/tests/test_confidence.py +185 -0
  12. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/tests/test_config.py +34 -0
  13. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/setup.cfg +0 -0
  14. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/dashboard.py +0 -0
  15. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/event_log.py +0 -0
  16. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/manager.py +0 -0
  17. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/models.py +0 -0
  18. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/multiplexer/__init__.py +0 -0
  19. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/multiplexer/base.py +0 -0
  20. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/multiplexer/tmux_compatible.py +0 -0
  21. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/multiplexer/zellij.py +0 -0
  22. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/notifier.py +0 -0
  23. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/py.typed +0 -0
  24. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams/subagent_client.py +0 -0
  25. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams_cli.egg-info/dependency_links.txt +0 -0
  26. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams_cli.egg-info/entry_points.txt +0 -0
  27. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams_cli.egg-info/requires.txt +0 -0
  28. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/src/workstreams_cli.egg-info/top_level.txt +0 -0
  29. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/tests/test_event_log.py +0 -0
  30. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/tests/test_models.py +0 -0
  31. {workstreams_cli-0.5.1 → workstreams_cli-0.6.1}/tests/test_notifier.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: workstreams-cli
3
- Version: 0.5.1
3
+ Version: 0.6.1
4
4
  Summary: Visually dispatch coding-agent work to subagents in real terminal windows and monitor it in one dashboard - for any coding agent (Claude Code, Codex, OpenCode, Qwen Code, Hermes, Cline, and more).
5
5
  Author: Dream-Pixels-Forge
6
6
  License: MIT
@@ -45,6 +45,14 @@ Key capabilities:
45
45
  - **Cross-terminal notifications** — desktop notifications (Linux `notify-send`, macOS `osascript`) plus a shared `notifications.jsonl` that other terminals can poll
46
46
  - **Agent-agnostic** — no vendor lock-in. `dispatch` and `work` send arbitrary shell commands to panes, so it works with whatever agent binary you can run from a shell
47
47
 
48
+ ## ⭐ If workstreams helps you ship faster, star the repo
49
+
50
+ If you found this useful, a GitHub star helps other developers discover it.
51
+
52
+ [![Star on GitHub](https://img.shields.io/github/stars/Dream-Pixels-Forge/workstreams-cli?style=social)](https://github.com/Dream-Pixels-Forge/workstreams-cli)
53
+
54
+ ⭐ **Star this repo:** [github.com/Dream-Pixels-Forge/workstreams-cli](https://github.com/Dream-Pixels-Forge/workstreams-cli)
55
+
48
56
  ## Why this exists
49
57
 
50
58
  Coding agents increasingly support "subagents" that run in the background of the main agent's process. That means: no visibility (you can't watch them), no isolation (they share one working tree and one set of installed dependencies), no way to run several in parallel on independent branches, and no shared event stream you can watch from your main terminal.
@@ -61,6 +69,7 @@ Coding agents increasingly support "subagents" that run in the background of the
61
69
  - [Configuration (.workstreams.yaml)](#configuration-workstreamsyaml)
62
70
  - [How the Multiplexers Work](#how-the-multiplexers-work)
63
71
  - [Subagent Event System (Python API + CLI)](#subagent-event-system)
72
+ - [Confidence Scoring](#confidence-scoring)
64
73
  - [Environment Variables](#environment-variables)
65
74
  - [Exit Codes](#exit-codes)
66
75
  - [Data Locations](#data-locations)
@@ -105,7 +114,7 @@ pip install -e ".[yaml,dev]" # dev extras add pytest
105
114
  Verify:
106
115
 
107
116
  ```bash
108
- workstreams --version # -> workstreams 0.5.1
117
+ workstreams --version # -> workstreams 0.6.1
109
118
  ```
110
119
 
111
120
  > **Note:** every command also accepts `--json` to emit machine-readable output (where supported), which coding agents can parse. All read-side commands work without a multiplexer installed; only `start`/`dispatch`/`work`/`attach` need one.
@@ -436,6 +445,67 @@ Events are appended to `~/.workstreams/<project>/events.jsonl` using `O_APPEND`
436
445
 
437
446
  ---
438
447
 
448
+ ## Confidence Scoring
449
+
450
+ Subagents are sometimes wrong. Confidence scoring lets each subagent **self-rate the quality of its result on a 0–10 scale** so the main agent (or a CI gate) can decide whether to accept, re-dispatch, or escalate. It is the difference between "subagent said it's done" and "subagent is 9/10 sure it actually delivered the right result".
451
+
452
+ - The score rides in the event's `data.confidence` field — no new storage required.
453
+ - A **gate** (`workstreams confidence --min-score N`) exits `0` when the latest score is at/above the threshold and `1` otherwise, so it drops straight into CI or a `--wait` loop.
454
+ - Scores are clamped to `[0, 10]`; a misbehaving subagent can't emit `100`.
455
+
456
+ ### Emitting a score
457
+
458
+ ```bash
459
+ # a subagent reports it finished, 9/10 confident
460
+ workstreams event completed --project myproj --workstream 1 \
461
+ --subagent claude-code --issue 42 \
462
+ --message "All tests green, edge cases covered" \
463
+ --confidence 9
464
+
465
+ # or via --data JSON
466
+ workstreams event completed --project myproj --workstream 1 \
467
+ --subagent codex --issue 42 --message "Done" --data '{"confidence": 7}'
468
+
469
+ # Python API (in a subagent script)
470
+ from workstreams import subagent_report
471
+ subagent_report("myproj", 1, "claude-code", 42, "completed", "Done",
472
+ data={"confidence": 9})
473
+ ```
474
+
475
+ ### Reading / gating on scores
476
+
477
+ ```bash
478
+ # human-readable summary + gate (exit 0 if latest >= 8)
479
+ workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 8
480
+
481
+ # machine-readable
482
+ workstreams confidence --project myproj --workstream 1 --json
483
+
484
+ # list every confidence-bearing event, oldest first
485
+ workstreams confidence --project myproj --workstream 1 --records
486
+ ```
487
+
488
+ Example output:
489
+
490
+ ```
491
+ Confidence for ws1 #42: PASS
492
+ latest: 9.0/10 (completed from claude-code)
493
+ average: 7.0/10 over 2 report(s)
494
+ threshold: 8.0/10 -> All tests green, edge cases covered
495
+ ```
496
+
497
+ ### Using it as a CI / re-dispatch gate
498
+
499
+ ```bash
500
+ # block the merge until the subagent's latest self-rating clears the bar
501
+ workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 9 \
502
+ || { echo "confidence too low, re-dispatching"; workstreams dispatch ...; }
503
+ ```
504
+
505
+ > Scores are self-reported. Treat them as a *signal*, not a proof — pair them with real test coverage. A 10/10 that shipped a broken build is still a broken build.
506
+
507
+ ---
508
+
439
509
  ## Environment Variables
440
510
 
441
511
  All are overridable in the config file / CLI; env vars are a fallback when neither is set.
@@ -591,57 +661,21 @@ workstreams logs --workstream 1 --lines 50
591
661
 
592
662
  ## Architecture
593
663
 
594
- `workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary. The core parts and how they fit together:
664
+ `workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary.
595
665
 
596
- ```
597
- you / your main terminal / a CI job / a cron
598
- │ CLI (argparse)
599
- ▼
600
- ┌─────────────────────────────────────────────────────────┐
601
- │ cli.py (entry: workstreams) │
602
- └─────────────────────────────────────────────────────────┘
603
- │ command routing │ config resolution (flag > yaml > env > default)
604
- ▼ ▼
605
- ┌──────────────────────┐ ┌──────────────────────┐
606
- │ WorkstreamsManager │ │ config.py │
607
- │ (manager.py) │◄──│ .workstreams.yaml / │
608
- │ init·start·dispatch │ │ .json load+save, │
609
- │ work·sync·pr·merge· │ │ no-pyyaml fallback │
610
- │ run·logs·events │ └──────────────────────┘
611
- └──────┬───────────────┘
612
- │
613
- ┌───┴──────────────────────────────────────────────┐
614
- │ │
615
- ▼ ▼
616
- ┌────────────────────────────┐ ┌────────────────────────────┐
617
- │ multiplexer/ (backends) │ │ event_log.py + │
618
- │ tmux.py / zellij.py │ │ subagent_client.py │
619
- │ MultiplexerBase: │ │ (Python API emitters) │
620
- │ create_session, send_ │ │ │
621
- │ command, list_sessions… │ │ NOTIFIER │
622
- └─────────────┬──────────────┘ │ (notifier.py): desktop │
623
- │ send-keys / tabs │ notify-send/osascript + │
624
- ▼ │ notifications.jsonl │
625
- visible terminal panes ◄──────────┤ │
626
- (one per workstream lane) └────────────────────────────┘
627
- │ each pane runs its own agent (claude, codex, …)
628
- │ agent appends JSONL events back
629
- ▼
630
- ┌──────────────────────────────────────────────────────────────┐
631
- │ ~/.workstreams/<project>/ │
632
- │ events.jsonl shared, concurrent-safe event stream │
633
- │ notifications.jsonl cross-terminal notification queue │
634
- └──────────────────────────────────────────────────────────────┘
635
- ▲
636
- │ re-render every 2s (configurable)
637
- ┌────────────────────────────┐
638
- │ dashboard.py │
639
- │ LiveDashboard: ANSI TUI │ ◄── the only read-side that polls events live
640
- └────────────────────────────┘
641
-
642
- models.py defines the dataclasses (WorkstreamConfig, WorkstreamsConfig,
643
- WorkstreamStatus, SubagentEvent) shared across all of the above.
644
- ```
666
+ ![workstreams architecture](https://github.com/Dream-Pixels-Forge/workstreams-cli/raw/main/assets/workstreams-architecture.webp)
667
+
668
+ The diagram above shows the full data flow: the CLI routes each command to `WorkstreamsManager`, which talks to a pluggable multiplexer backend to place agents into visible terminal panes, while every lane's subagent writes JSONL events back to a shared, concurrent-safe log that the live dashboard re-renders on a timer. The core modules:
669
+
670
+ - **`cli.py`** — argparse entry point; every command accepts `--project` and `--json`
671
+ - **`manager.py`** — `WorkstreamsManager` orchestrates init/start/dispatch/work/sync/pr/merge/run/logs/events
672
+ - **`config.py`** — loads/saves `.workstreams.yaml` (or JSON) with no-pyyaml fallback; resolution order: CLI flag > yaml > env > default
673
+ - **`multiplexer/`** — pluggable backends behind `MultiplexerBase` (`tmux.py`, `zellij.py`, plus tmux-compatible wrappers for `nami`/`lmux`/`wmux`/`herdr`)
674
+ - **`event_log.py` + `subagent_client.py`** — shared cross-process JSONL event stream and the Python API emitters
675
+ - **`notifier.py`** — desktop notifications (`notify-send`/`osascript`/PowerShell) plus a `notifications.jsonl` queue
676
+ - **`confidence.py`** — aggregates subagent self-rated 0–10 quality scores for accept / re-dispatch gating
677
+ - **`dashboard.py`** — the live ANSI TUI that polls events and re-renders every 2s
678
+ - **`models.py`** — dataclasses (`WorkstreamConfig`, `WorkstreamsConfig`, `WorkstreamStatus`, `SubagentEvent`) shared across all of the above
645
679
 
646
680
  How a run flows end-to-end:
647
681
 
@@ -15,6 +15,14 @@ Key capabilities:
15
15
  - **Cross-terminal notifications** — desktop notifications (Linux `notify-send`, macOS `osascript`) plus a shared `notifications.jsonl` that other terminals can poll
16
16
  - **Agent-agnostic** — no vendor lock-in. `dispatch` and `work` send arbitrary shell commands to panes, so it works with whatever agent binary you can run from a shell
17
17
 
18
+ ## ⭐ If workstreams helps you ship faster, star the repo
19
+
20
+ If you found this useful, a GitHub star helps other developers discover it.
21
+
22
+ [![Star on GitHub](https://img.shields.io/github/stars/Dream-Pixels-Forge/workstreams-cli?style=social)](https://github.com/Dream-Pixels-Forge/workstreams-cli)
23
+
24
+ ⭐ **Star this repo:** [github.com/Dream-Pixels-Forge/workstreams-cli](https://github.com/Dream-Pixels-Forge/workstreams-cli)
25
+
18
26
  ## Why this exists
19
27
 
20
28
  Coding agents increasingly support "subagents" that run in the background of the main agent's process. That means: no visibility (you can't watch them), no isolation (they share one working tree and one set of installed dependencies), no way to run several in parallel on independent branches, and no shared event stream you can watch from your main terminal.
@@ -31,6 +39,7 @@ Coding agents increasingly support "subagents" that run in the background of the
31
39
  - [Configuration (.workstreams.yaml)](#configuration-workstreamsyaml)
32
40
  - [How the Multiplexers Work](#how-the-multiplexers-work)
33
41
  - [Subagent Event System (Python API + CLI)](#subagent-event-system)
42
+ - [Confidence Scoring](#confidence-scoring)
34
43
  - [Environment Variables](#environment-variables)
35
44
  - [Exit Codes](#exit-codes)
36
45
  - [Data Locations](#data-locations)
@@ -75,7 +84,7 @@ pip install -e ".[yaml,dev]" # dev extras add pytest
75
84
  Verify:
76
85
 
77
86
  ```bash
78
- workstreams --version # -> workstreams 0.5.1
87
+ workstreams --version # -> workstreams 0.6.1
79
88
  ```
80
89
 
81
90
  > **Note:** every command also accepts `--json` to emit machine-readable output (where supported), which coding agents can parse. All read-side commands work without a multiplexer installed; only `start`/`dispatch`/`work`/`attach` need one.
@@ -406,6 +415,67 @@ Events are appended to `~/.workstreams/<project>/events.jsonl` using `O_APPEND`
406
415
 
407
416
  ---
408
417
 
418
+ ## Confidence Scoring
419
+
420
+ Subagents are sometimes wrong. Confidence scoring lets each subagent **self-rate the quality of its result on a 0–10 scale** so the main agent (or a CI gate) can decide whether to accept, re-dispatch, or escalate. It is the difference between "subagent said it's done" and "subagent is 9/10 sure it actually delivered the right result".
421
+
422
+ - The score rides in the event's `data.confidence` field — no new storage required.
423
+ - A **gate** (`workstreams confidence --min-score N`) exits `0` when the latest score is at/above the threshold and `1` otherwise, so it drops straight into CI or a `--wait` loop.
424
+ - Scores are clamped to `[0, 10]`; a misbehaving subagent can't emit `100`.
425
+
426
+ ### Emitting a score
427
+
428
+ ```bash
429
+ # a subagent reports it finished, 9/10 confident
430
+ workstreams event completed --project myproj --workstream 1 \
431
+ --subagent claude-code --issue 42 \
432
+ --message "All tests green, edge cases covered" \
433
+ --confidence 9
434
+
435
+ # or via --data JSON
436
+ workstreams event completed --project myproj --workstream 1 \
437
+ --subagent codex --issue 42 --message "Done" --data '{"confidence": 7}'
438
+
439
+ # Python API (in a subagent script)
440
+ from workstreams import subagent_report
441
+ subagent_report("myproj", 1, "claude-code", 42, "completed", "Done",
442
+ data={"confidence": 9})
443
+ ```
444
+
445
+ ### Reading / gating on scores
446
+
447
+ ```bash
448
+ # human-readable summary + gate (exit 0 if latest >= 8)
449
+ workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 8
450
+
451
+ # machine-readable
452
+ workstreams confidence --project myproj --workstream 1 --json
453
+
454
+ # list every confidence-bearing event, oldest first
455
+ workstreams confidence --project myproj --workstream 1 --records
456
+ ```
457
+
458
+ Example output:
459
+
460
+ ```
461
+ Confidence for ws1 #42: PASS
462
+ latest: 9.0/10 (completed from claude-code)
463
+ average: 7.0/10 over 2 report(s)
464
+ threshold: 8.0/10 -> All tests green, edge cases covered
465
+ ```
466
+
467
+ ### Using it as a CI / re-dispatch gate
468
+
469
+ ```bash
470
+ # block the merge until the subagent's latest self-rating clears the bar
471
+ workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 9 \
472
+ || { echo "confidence too low, re-dispatching"; workstreams dispatch ...; }
473
+ ```
474
+
475
+ > Scores are self-reported. Treat them as a *signal*, not a proof — pair them with real test coverage. A 10/10 that shipped a broken build is still a broken build.
476
+
477
+ ---
478
+
409
479
  ## Environment Variables
410
480
 
411
481
  All are overridable in the config file / CLI; env vars are a fallback when neither is set.
@@ -561,57 +631,21 @@ workstreams logs --workstream 1 --lines 50
561
631
 
562
632
  ## Architecture
563
633
 
564
- `workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary. The core parts and how they fit together:
634
+ `workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary.
565
635
 
566
- ```
567
- you / your main terminal / a CI job / a cron
568
- │ CLI (argparse)
569
- ▼
570
- ┌─────────────────────────────────────────────────────────┐
571
- │ cli.py (entry: workstreams) │
572
- └─────────────────────────────────────────────────────────┘
573
- │ command routing │ config resolution (flag > yaml > env > default)
574
- ▼ ▼
575
- ┌──────────────────────┐ ┌──────────────────────┐
576
- │ WorkstreamsManager │ │ config.py │
577
- │ (manager.py) │◄──│ .workstreams.yaml / │
578
- │ init·start·dispatch │ │ .json load+save, │
579
- │ work·sync·pr·merge· │ │ no-pyyaml fallback │
580
- │ run·logs·events │ └──────────────────────┘
581
- └──────┬───────────────┘
582
- │
583
- ┌───┴──────────────────────────────────────────────┐
584
- │ │
585
- ▼ ▼
586
- ┌────────────────────────────┐ ┌────────────────────────────┐
587
- │ multiplexer/ (backends) │ │ event_log.py + │
588
- │ tmux.py / zellij.py │ │ subagent_client.py │
589
- │ MultiplexerBase: │ │ (Python API emitters) │
590
- │ create_session, send_ │ │ │
591
- │ command, list_sessions… │ │ NOTIFIER │
592
- └─────────────┬──────────────┘ │ (notifier.py): desktop │
593
- │ send-keys / tabs │ notify-send/osascript + │
594
- ▼ │ notifications.jsonl │
595
- visible terminal panes ◄──────────┤ │
596
- (one per workstream lane) └────────────────────────────┘
597
- │ each pane runs its own agent (claude, codex, …)
598
- │ agent appends JSONL events back
599
- ▼
600
- ┌──────────────────────────────────────────────────────────────┐
601
- │ ~/.workstreams/<project>/ │
602
- │ events.jsonl shared, concurrent-safe event stream │
603
- │ notifications.jsonl cross-terminal notification queue │
604
- └──────────────────────────────────────────────────────────────┘
605
- ▲
606
- │ re-render every 2s (configurable)
607
- ┌────────────────────────────┐
608
- │ dashboard.py │
609
- │ LiveDashboard: ANSI TUI │ ◄── the only read-side that polls events live
610
- └────────────────────────────┘
611
-
612
- models.py defines the dataclasses (WorkstreamConfig, WorkstreamsConfig,
613
- WorkstreamStatus, SubagentEvent) shared across all of the above.
614
- ```
636
+ ![workstreams architecture](https://github.com/Dream-Pixels-Forge/workstreams-cli/raw/main/assets/workstreams-architecture.webp)
637
+
638
+ The diagram above shows the full data flow: the CLI routes each command to `WorkstreamsManager`, which talks to a pluggable multiplexer backend to place agents into visible terminal panes, while every lane's subagent writes JSONL events back to a shared, concurrent-safe log that the live dashboard re-renders on a timer. The core modules:
639
+
640
+ - **`cli.py`** — argparse entry point; every command accepts `--project` and `--json`
641
+ - **`manager.py`** — `WorkstreamsManager` orchestrates init/start/dispatch/work/sync/pr/merge/run/logs/events
642
+ - **`config.py`** — loads/saves `.workstreams.yaml` (or JSON) with no-pyyaml fallback; resolution order: CLI flag > yaml > env > default
643
+ - **`multiplexer/`** — pluggable backends behind `MultiplexerBase` (`tmux.py`, `zellij.py`, plus tmux-compatible wrappers for `nami`/`lmux`/`wmux`/`herdr`)
644
+ - **`event_log.py` + `subagent_client.py`** — shared cross-process JSONL event stream and the Python API emitters
645
+ - **`notifier.py`** — desktop notifications (`notify-send`/`osascript`/PowerShell) plus a `notifications.jsonl` queue
646
+ - **`confidence.py`** — aggregates subagent self-rated 0–10 quality scores for accept / re-dispatch gating
647
+ - **`dashboard.py`** — the live ANSI TUI that polls events and re-renders every 2s
648
+ - **`models.py`** — dataclasses (`WorkstreamConfig`, `WorkstreamsConfig`, `WorkstreamStatus`, `SubagentEvent`) shared across all of the above
615
649
 
616
650
  How a run flows end-to-end:
617
651
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "workstreams-cli"
7
- version = "0.5.1"
7
+ version = "0.6.1"
8
8
  description = "Visually dispatch coding-agent work to subagents in real terminal windows and monitor it in one dashboard - for any coding agent (Claude Code, Codex, OpenCode, Qwen Code, Hermes, Cline, and more)."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -33,8 +33,16 @@ from .subagent_client import (
33
33
  subagent_done,
34
34
  )
35
35
  from .multiplexer import MultiplexerBase, TmuxMultiplexer, ZellijMultiplexer, get_multiplexer
36
+ from .confidence import (
37
+ ConfidenceRecord,
38
+ ConfidenceSummary,
39
+ get_confidence,
40
+ confidence_records,
41
+ clamp_score,
42
+ extract_score,
43
+ )
36
44
 
37
- __version__ = "0.5.1"
45
+ __version__ = "0.6.1"
38
46
 
39
47
  __all__ = [
40
48
  # models
@@ -68,4 +76,11 @@ __all__ = [
68
76
  "TmuxMultiplexer",
69
77
  "ZellijMultiplexer",
70
78
  "get_multiplexer",
79
+ # confidence scoring
80
+ "ConfidenceRecord",
81
+ "ConfidenceSummary",
82
+ "get_confidence",
83
+ "confidence_records",
84
+ "clamp_score",
85
+ "extract_score",
71
86
  ]
@@ -208,8 +208,18 @@ def build_parser() -> argparse.ArgumentParser:
208
208
  event.add_argument("--issue", type=int, default=0)
209
209
  event.add_argument("--message", default="")
210
210
  event.add_argument("--data", help="JSON object of extra data (e.g. '{\"files\": 3}')")
211
+ event.add_argument("--confidence", type=float, default=None, help="Self-rated result quality 0-10 (stored in data.confidence)")
211
212
  event.add_argument("--json", action="store_true")
212
213
 
214
+ # confidence (aggregate / gate on subagent self-ratings)
215
+ confidence = common_parent("confidence")
216
+ confidence.add_argument("--workstream", type=int, help="Restrict to one workstream")
217
+ confidence.add_argument("--issue", type=int, help="Restrict to one issue")
218
+ confidence.add_argument("--subagent", help="Filter by subagent name")
219
+ confidence.add_argument("--since", type=int, default=43200, help="Minutes back (default: 30 days)")
220
+ confidence.add_argument("--min-score", type=float, default=8.0, help="Pass threshold for the gate (default: 8.0)")
221
+ confidence.add_argument("--records", action="store_true", help="List individual confidence-bearing events")
222
+
213
223
  # notify
214
224
  notify = common_parent("notify")
215
225
  notify.add_argument("--title", required=True)
@@ -412,6 +422,8 @@ def _cmd_event(args) -> int:
412
422
  except json.JSONDecodeError:
413
423
  print(f"Invalid --data JSON: {args.data}", file=sys.stderr)
414
424
  return 2
425
+ if getattr(args, "confidence", None) is not None:
426
+ data["confidence"] = args.confidence
415
427
  ok = subagent_report(
416
428
  project=args.project,
417
429
  workstream_id=args.workstream,
@@ -431,6 +443,73 @@ def _cmd_event(args) -> int:
431
443
  return 0 if ok else 1
432
444
 
433
445
 
446
+ def _cmd_confidence(args) -> int:
447
+ """Aggregate / gate on subagent self-rated confidence scores."""
448
+ from .confidence import get_confidence, confidence_records, clamp_score
449
+
450
+ # clamp the threshold so a bogus --min-score can't break the gate
451
+ min_score = clamp_score(args.min_score)
452
+ if min_score is None:
453
+ print(f"Invalid --min-score: {args.min_score}", file=sys.stderr)
454
+ return 2
455
+
456
+ # The event log is keyed by project, but confidence read commands are
457
+ # project-scoped like the rest of the read side; fall back to .workstreams.yaml
458
+ # if no --project was given.
459
+ import os
460
+ from .config import load_config
461
+ project = getattr(args, "project", None)
462
+ if not project:
463
+ cfg = load_config(Path.cwd())
464
+ project = cfg.project
465
+
466
+ if args.records:
467
+ records = confidence_records(
468
+ project=project,
469
+ workstream_id=args.workstream,
470
+ issue=args.issue,
471
+ subagent=args.subagent,
472
+ since_minutes=args.since,
473
+ )
474
+ if args.json:
475
+ print(json.dumps([r.to_dict() for r in records], indent=2, default=str))
476
+ return 0
477
+ if not records:
478
+ print(f"(no confidence reports in last {args.since} min)")
479
+ return 0
480
+ for r in records:
481
+ ts = r.timestamp[11:19]
482
+ print(f"[{ts}] [{r.subagent}] ws{r.workstream_id} #{r.issue} - {r.event_type}: {r.score}/10 {r.message}")
483
+ return 0
484
+
485
+ summary = get_confidence(
486
+ project=project,
487
+ workstream_id=args.workstream,
488
+ issue=args.issue,
489
+ subagent=args.subagent,
490
+ since_minutes=args.since,
491
+ min_score=min_score,
492
+ )
493
+ if args.json:
494
+ print(json.dumps(summary.to_dict(), indent=2, default=str))
495
+ return 0 if summary.passed else 1
496
+
497
+ scope = f"ws{args.workstream}" if args.workstream else "project"
498
+ if args.issue:
499
+ scope += f" #{args.issue}"
500
+ if summary.n_reports == 0:
501
+ print(f"No confidence reports for {scope} yet.")
502
+ return 1
503
+ avg = f"{summary.avg_score:.1f}" if summary.avg_score is not None else "n/a"
504
+ latest = f"{summary.latest_score:.1f}" if summary.latest_score is not None else "n/a"
505
+ mark = "PASS" if summary.passed else "FAIL"
506
+ print(f"Confidence for {scope}: {mark}")
507
+ print(f" latest: {latest}/10 ({summary.latest_event_type} from {summary.latest_subagent})")
508
+ print(f" average: {avg}/10 over {summary.n_reports} report(s)")
509
+ print(f" threshold: {min_score}/10 -> {summary.latest_message}")
510
+ return 0 if summary.passed else 1
511
+
512
+
434
513
  def _cmd_notify(args) -> int:
435
514
  manager = _load_manager(args)
436
515
  manager.notify(args.title, args.message, args.urgency)
@@ -472,6 +551,7 @@ HANDLERS = {
472
551
  "tail": _cmd_tail,
473
552
  "events": _cmd_events,
474
553
  "event": _cmd_event,
554
+ "confidence": _cmd_confidence,
475
555
  "notify": _cmd_notify,
476
556
  "assign": _cmd_assign,
477
557
  "sync": _cmd_sync,
@@ -0,0 +1,196 @@
1
+ """Confidence scoring — let subagents self-rate result quality.
2
+
3
+ A subagent can attach a confidence score (0-10) to any event, signalling
4
+ how sure it is that the delivered result meets the required bar. The main
5
+ agent (or a CI gate) can then query the latest / aggregate confidence and
6
+ decide whether to accept, re-dispatch, or escalate.
7
+
8
+ # subagent reports a high-confidence completion
9
+ workstreams event completed --project p --workstream 1 --subagent claude-code \
10
+ --issue 42 --message "All tests green" --confidence 9
11
+
12
+ # main agent / CI gate: pass only if latest confidence >= 8
13
+ workstreams confidence --project p --workstream 1 --issue 42 --min-score 8
14
+ # exit 0 -> gate passed, exit 1 -> below threshold
15
+
16
+ Scores are read straight from the event log's `data.confidence` field, so
17
+ no new storage is required.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from dataclasses import dataclass, field, asdict
23
+ from typing import Any, Dict, List, Optional
24
+
25
+ from .event_log import get_event_log
26
+ from .models import SubagentEvent
27
+
28
+ # Confidence scale is 0-10. Anything at/above this is treated as a
29
+ # "high confidence" result by default.
30
+ DEFAULT_MIN_SCORE = 8
31
+
32
+
33
+ def clamp_score(value: Any) -> Optional[float]:
34
+ """Normalise a confidence value to a float in [0, 10], or None.
35
+
36
+ Accepts ints, floats, and numeric strings. Values are clamped to
37
+ [0, 10] so a misbehaving subagent can't emit e.g. 100.
38
+ """
39
+ if value is None:
40
+ return None
41
+ try:
42
+ score = float(value)
43
+ except (TypeError, ValueError):
44
+ return None
45
+ if score < 0:
46
+ return 0.0
47
+ if score > 10:
48
+ return 10.0
49
+ return score
50
+
51
+
52
+ @dataclass
53
+ class ConfidenceRecord:
54
+ """One confidence-bearing event, normalised for display."""
55
+
56
+ workstream_id: int
57
+ subagent: str
58
+ issue: int
59
+ event_type: str
60
+ score: Optional[float]
61
+ message: str
62
+ timestamp: str
63
+ data: Dict[str, Any] = field(default_factory=dict)
64
+
65
+ def to_dict(self) -> Dict[str, Any]:
66
+ return asdict(self)
67
+
68
+
69
+ @dataclass
70
+ class ConfidenceSummary:
71
+ """Aggregated confidence view for a workstream (optionally per issue)."""
72
+
73
+ workstream_id: int
74
+ latest_score: Optional[float]
75
+ latest_event_type: str
76
+ latest_subagent: str
77
+ latest_message: str
78
+ avg_score: Optional[float]
79
+ n_reports: int
80
+ passed: bool
81
+ min_score: float
82
+
83
+ def to_dict(self) -> Dict[str, Any]:
84
+ return asdict(self)
85
+
86
+
87
+ def extract_score(event: SubagentEvent) -> Optional[float]:
88
+ """Pull a confidence score out of an event (data.confidence preferred)."""
89
+ score = event.data.get("confidence")
90
+ if score is None:
91
+ # also accept top-level 'score' in data for convenience
92
+ score = event.data.get("score")
93
+ return clamp_score(score)
94
+
95
+
96
+ def get_confidence(
97
+ project: str,
98
+ workstream_id: Optional[int] = None,
99
+ issue: Optional[int] = None,
100
+ subagent: Optional[str] = None,
101
+ since_minutes: int = 60 * 24 * 30, # 30 days
102
+ min_score: float = DEFAULT_MIN_SCORE,
103
+ ) -> ConfidenceSummary:
104
+ """Aggregate confidence for a workstream (and optional issue/subagent).
105
+
106
+ Only events that carry a confidence score are counted. `passed` is True
107
+ when the *latest* score is at or above `min_score` and at least one
108
+ scored report exists.
109
+ """
110
+ from datetime import datetime, UTC, timedelta
111
+
112
+ event_log = get_event_log(project)
113
+ since = datetime.now(UTC) - timedelta(minutes=since_minutes)
114
+
115
+ events = event_log.get_events(
116
+ workstream_id=workstream_id,
117
+ since=since,
118
+ subagent=subagent,
119
+ )
120
+ if issue is not None:
121
+ events = [e for e in events if e.issue == issue]
122
+
123
+ scored = [e for e in events if extract_score(e) is not None]
124
+
125
+ latest_score: Optional[float] = None
126
+ latest_event_type = ""
127
+ latest_subagent = ""
128
+ latest_message = ""
129
+ total = 0.0
130
+ n = 0
131
+
132
+ for e in scored:
133
+ s = extract_score(e)
134
+ total += s or 0.0
135
+ n += 1
136
+ # events are oldest->newest, so the last scored event is the latest
137
+ latest_score = s
138
+ latest_event_type = e.event_type
139
+ latest_subagent = e.subagent
140
+ latest_message = e.message
141
+
142
+ avg_score = (total / n) if n else None
143
+ passed = (
144
+ latest_score is not None
145
+ and latest_score >= min_score
146
+ )
147
+
148
+ return ConfidenceSummary(
149
+ workstream_id=workstream_id or 0,
150
+ latest_score=latest_score,
151
+ latest_event_type=latest_event_type,
152
+ latest_subagent=latest_subagent,
153
+ latest_message=latest_message,
154
+ avg_score=avg_score,
155
+ n_reports=n,
156
+ passed=passed,
157
+ min_score=min_score,
158
+ )
159
+
160
+
161
+ def confidence_records(
162
+ project: str,
163
+ workstream_id: Optional[int] = None,
164
+ issue: Optional[int] = None,
165
+ subagent: Optional[str] = None,
166
+ since_minutes: int = 60 * 24 * 30,
167
+ ) -> List[ConfidenceRecord]:
168
+ """Return all confidence-bearing events, oldest first."""
169
+ from datetime import datetime, UTC, timedelta
170
+
171
+ event_log = get_event_log(project)
172
+ since = datetime.now(UTC) - timedelta(minutes=since_minutes)
173
+ events = event_log.get_events(
174
+ workstream_id=workstream_id, since=since, subagent=subagent
175
+ )
176
+ if issue is not None:
177
+ events = [e for e in events if e.issue == issue]
178
+
179
+ out: List[ConfidenceRecord] = []
180
+ for e in events:
181
+ s = extract_score(e)
182
+ if s is None:
183
+ continue
184
+ out.append(
185
+ ConfidenceRecord(
186
+ workstream_id=e.workstream_id,
187
+ subagent=e.subagent,
188
+ issue=e.issue,
189
+ event_type=e.event_type,
190
+ score=s,
191
+ message=e.message,
192
+ timestamp=e.timestamp,
193
+ data=e.data,
194
+ )
195
+ )
196
+ return out
@@ -47,19 +47,28 @@ def load_config(
47
47
  """
48
48
  base_path = (base_path or Path.cwd()).resolve()
49
49
  config_file = config_path_in(base_path)
50
+ # Fall back to the JSON config if YAML is not present (no-pyyaml installs)
51
+ if not config_file.exists():
52
+ json_fallback = config_file.with_suffix(".json")
53
+ if json_fallback.exists():
54
+ config_file = json_fallback
50
55
  data: Dict[str, Any] = {}
51
56
  if config_file.exists():
52
57
  try:
53
58
  text = config_file.read_text()
54
- parsed = _yaml_load(text)
55
- if parsed is not None:
56
- data = parsed
59
+ if config_file.suffix == ".json":
60
+ parsed = json.loads(text)
61
+ data = parsed if isinstance(parsed, dict) else {}
57
62
  else:
58
- # Fallback parser for simple YAML (no pyyaml installed)
59
- parsed = _simple_yaml_load(text)
63
+ parsed = _yaml_load(text)
60
64
  if parsed is not None:
61
65
  data = parsed
62
- except OSError:
66
+ else:
67
+ # Fallback parser for simple YAML (no pyyaml installed)
68
+ parsed = _simple_yaml_load(text)
69
+ if parsed is not None:
70
+ data = parsed
71
+ except (OSError, json.JSONDecodeError):
63
72
  data = {}
64
73
 
65
74
  workstreams = [
@@ -180,7 +180,13 @@ class TmuxMultiplexer(MultiplexerBase):
180
180
  flush=True,
181
181
  )
182
182
  return False
183
+ # Try the named window first; if tmux can't resolve that target
184
+ # (stale name, or a session created outside workstreams), fall back
185
+ # to the numeric window index so dispatches don't silently no-op.
183
186
  result = self._tmux("send-keys", "-t", target, command, "Enter")
187
+ if result.returncode != 0:
188
+ fallback = f"{self.session}:{workstream_id - 1}"
189
+ result = self._tmux("send-keys", "-t", fallback, command, "Enter")
184
190
  return result.returncode == 0
185
191
 
186
192
  def capture(self, workstream_id: int, lines: int = 20) -> str:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: workstreams-cli
3
- Version: 0.5.1
3
+ Version: 0.6.1
4
4
  Summary: Visually dispatch coding-agent work to subagents in real terminal windows and monitor it in one dashboard - for any coding agent (Claude Code, Codex, OpenCode, Qwen Code, Hermes, Cline, and more).
5
5
  Author: Dream-Pixels-Forge
6
6
  License: MIT
@@ -45,6 +45,14 @@ Key capabilities:
45
45
  - **Cross-terminal notifications** — desktop notifications (Linux `notify-send`, macOS `osascript`) plus a shared `notifications.jsonl` that other terminals can poll
46
46
  - **Agent-agnostic** — no vendor lock-in. `dispatch` and `work` send arbitrary shell commands to panes, so it works with whatever agent binary you can run from a shell
47
47
 
48
+ ## ⭐ If workstreams helps you ship faster, star the repo
49
+
50
+ If you found this useful, a GitHub star helps other developers discover it.
51
+
52
+ [![Star on GitHub](https://img.shields.io/github/stars/Dream-Pixels-Forge/workstreams-cli?style=social)](https://github.com/Dream-Pixels-Forge/workstreams-cli)
53
+
54
+ ⭐ **Star this repo:** [github.com/Dream-Pixels-Forge/workstreams-cli](https://github.com/Dream-Pixels-Forge/workstreams-cli)
55
+
48
56
  ## Why this exists
49
57
 
50
58
  Coding agents increasingly support "subagents" that run in the background of the main agent's process. That means: no visibility (you can't watch them), no isolation (they share one working tree and one set of installed dependencies), no way to run several in parallel on independent branches, and no shared event stream you can watch from your main terminal.
@@ -61,6 +69,7 @@ Coding agents increasingly support "subagents" that run in the background of the
61
69
  - [Configuration (.workstreams.yaml)](#configuration-workstreamsyaml)
62
70
  - [How the Multiplexers Work](#how-the-multiplexers-work)
63
71
  - [Subagent Event System (Python API + CLI)](#subagent-event-system)
72
+ - [Confidence Scoring](#confidence-scoring)
64
73
  - [Environment Variables](#environment-variables)
65
74
  - [Exit Codes](#exit-codes)
66
75
  - [Data Locations](#data-locations)
@@ -105,7 +114,7 @@ pip install -e ".[yaml,dev]" # dev extras add pytest
105
114
  Verify:
106
115
 
107
116
  ```bash
108
- workstreams --version # -> workstreams 0.5.1
117
+ workstreams --version # -> workstreams 0.6.1
109
118
  ```
110
119
 
111
120
  > **Note:** every command also accepts `--json` to emit machine-readable output (where supported), which coding agents can parse. All read-side commands work without a multiplexer installed; only `start`/`dispatch`/`work`/`attach` need one.
@@ -436,6 +445,67 @@ Events are appended to `~/.workstreams/<project>/events.jsonl` using `O_APPEND`
436
445
 
437
446
  ---
438
447
 
448
+ ## Confidence Scoring
449
+
450
+ Subagents are sometimes wrong. Confidence scoring lets each subagent **self-rate the quality of its result on a 0–10 scale** so the main agent (or a CI gate) can decide whether to accept, re-dispatch, or escalate. It is the difference between "subagent said it's done" and "subagent is 9/10 sure it actually delivered the right result".
451
+
452
+ - The score rides in the event's `data.confidence` field — no new storage required.
453
+ - A **gate** (`workstreams confidence --min-score N`) exits `0` when the latest score is at/above the threshold and `1` otherwise, so it drops straight into CI or a `--wait` loop.
454
+ - Scores are clamped to `[0, 10]`; a misbehaving subagent can't emit `100`.
455
+
456
+ ### Emitting a score
457
+
458
+ ```bash
459
+ # a subagent reports it finished, 9/10 confident
460
+ workstreams event completed --project myproj --workstream 1 \
461
+ --subagent claude-code --issue 42 \
462
+ --message "All tests green, edge cases covered" \
463
+ --confidence 9
464
+
465
+ # or via --data JSON
466
+ workstreams event completed --project myproj --workstream 1 \
467
+ --subagent codex --issue 42 --message "Done" --data '{"confidence": 7}'
468
+
469
+ # Python API (in a subagent script)
470
+ from workstreams import subagent_report
471
+ subagent_report("myproj", 1, "claude-code", 42, "completed", "Done",
472
+ data={"confidence": 9})
473
+ ```
474
+
475
+ ### Reading / gating on scores
476
+
477
+ ```bash
478
+ # human-readable summary + gate (exit 0 if latest >= 8)
479
+ workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 8
480
+
481
+ # machine-readable
482
+ workstreams confidence --project myproj --workstream 1 --json
483
+
484
+ # list every confidence-bearing event, oldest first
485
+ workstreams confidence --project myproj --workstream 1 --records
486
+ ```
487
+
488
+ Example output:
489
+
490
+ ```
491
+ Confidence for ws1 #42: PASS
492
+ latest: 9.0/10 (completed from claude-code)
493
+ average: 7.0/10 over 2 report(s)
494
+ threshold: 8.0/10 -> All tests green, edge cases covered
495
+ ```
496
+
497
+ ### Using it as a CI / re-dispatch gate
498
+
499
+ ```bash
500
+ # block the merge until the subagent's latest self-rating clears the bar
501
+ workstreams confidence --project myproj --workstream 1 --issue 42 --min-score 9 \
502
+ || { echo "confidence too low, re-dispatching"; workstreams dispatch ...; }
503
+ ```
504
+
505
+ > Scores are self-reported. Treat them as a *signal*, not a proof — pair them with real test coverage. A 10/10 that shipped a broken build is still a broken build.
506
+
507
+ ---
508
+
439
509
  ## Environment Variables
440
510
 
441
511
  All are overridable in the config file / CLI; env vars are a fallback when neither is set.
@@ -591,57 +661,21 @@ workstreams logs --workstream 1 --lines 50
591
661
 
592
662
  ## Architecture
593
663
 
594
- `workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary. The core parts and how they fit together:
664
+ `workstreams` is a thin orchestration layer that sits between you, your git repo, a terminal multiplexer, and any coding agent binary.
595
665
 
596
- ```
597
- you / your main terminal / a CI job / a cron
598
- │ CLI (argparse)
599
- ▼
600
- ┌─────────────────────────────────────────────────────────┐
601
- │ cli.py (entry: workstreams) │
602
- └─────────────────────────────────────────────────────────┘
603
- │ command routing │ config resolution (flag > yaml > env > default)
604
- ▼ ▼
605
- ┌──────────────────────┐ ┌──────────────────────┐
606
- │ WorkstreamsManager │ │ config.py │
607
- │ (manager.py) │◄──│ .workstreams.yaml / │
608
- │ init·start·dispatch │ │ .json load+save, │
609
- │ work·sync·pr·merge· │ │ no-pyyaml fallback │
610
- │ run·logs·events │ └──────────────────────┘
611
- └──────┬───────────────┘
612
- │
613
- ┌───┴──────────────────────────────────────────────┐
614
- │ │
615
- ▼ ▼
616
- ┌────────────────────────────┐ ┌────────────────────────────┐
617
- │ multiplexer/ (backends) │ │ event_log.py + │
618
- │ tmux.py / zellij.py │ │ subagent_client.py │
619
- │ MultiplexerBase: │ │ (Python API emitters) │
620
- │ create_session, send_ │ │ │
621
- │ command, list_sessions… │ │ NOTIFIER │
622
- └─────────────┬──────────────┘ │ (notifier.py): desktop │
623
- │ send-keys / tabs │ notify-send/osascript + │
624
- ▼ │ notifications.jsonl │
625
- visible terminal panes ◄──────────┤ │
626
- (one per workstream lane) └────────────────────────────┘
627
- │ each pane runs its own agent (claude, codex, …)
628
- │ agent appends JSONL events back
629
- ▼
630
- ┌──────────────────────────────────────────────────────────────┐
631
- │ ~/.workstreams/<project>/ │
632
- │ events.jsonl shared, concurrent-safe event stream │
633
- │ notifications.jsonl cross-terminal notification queue │
634
- └──────────────────────────────────────────────────────────────┘
635
- ▲
636
- │ re-render every 2s (configurable)
637
- ┌────────────────────────────┐
638
- │ dashboard.py │
639
- │ LiveDashboard: ANSI TUI │ ◄── the only read-side that polls events live
640
- └────────────────────────────┘
641
-
642
- models.py defines the dataclasses (WorkstreamConfig, WorkstreamsConfig,
643
- WorkstreamStatus, SubagentEvent) shared across all of the above.
644
- ```
666
+ ![workstreams architecture](https://github.com/Dream-Pixels-Forge/workstreams-cli/raw/main/assets/workstreams-architecture.webp)
667
+
668
+ The diagram above shows the full data flow: the CLI routes each command to `WorkstreamsManager`, which talks to a pluggable multiplexer backend to place agents into visible terminal panes, while every lane's subagent writes JSONL events back to a shared, concurrent-safe log that the live dashboard re-renders on a timer. The core modules:
669
+
670
+ - **`cli.py`** — argparse entry point; every command accepts `--project` and `--json`
671
+ - **`manager.py`** — `WorkstreamsManager` orchestrates init/start/dispatch/work/sync/pr/merge/run/logs/events
672
+ - **`config.py`** — loads/saves `.workstreams.yaml` (or JSON) with no-pyyaml fallback; resolution order: CLI flag > yaml > env > default
673
+ - **`multiplexer/`** — pluggable backends behind `MultiplexerBase` (`tmux.py`, `zellij.py`, plus tmux-compatible wrappers for `nami`/`lmux`/`wmux`/`herdr`)
674
+ - **`event_log.py` + `subagent_client.py`** — shared cross-process JSONL event stream and the Python API emitters
675
+ - **`notifier.py`** — desktop notifications (`notify-send`/`osascript`/PowerShell) plus a `notifications.jsonl` queue
676
+ - **`confidence.py`** — aggregates subagent self-rated 0–10 quality scores for accept / re-dispatch gating
677
+ - **`dashboard.py`** — the live ANSI TUI that polls events and re-renders every 2s
678
+ - **`models.py`** — dataclasses (`WorkstreamConfig`, `WorkstreamsConfig`, `WorkstreamStatus`, `SubagentEvent`) shared across all of the above
645
679
 
646
680
  How a run flows end-to-end:
647
681
 
@@ -2,6 +2,7 @@ README.md
2
2
  pyproject.toml
3
3
  src/workstreams/__init__.py
4
4
  src/workstreams/cli.py
5
+ src/workstreams/confidence.py
5
6
  src/workstreams/config.py
6
7
  src/workstreams/dashboard.py
7
8
  src/workstreams/event_log.py
@@ -21,6 +22,7 @@ src/workstreams_cli.egg-info/dependency_links.txt
21
22
  src/workstreams_cli.egg-info/entry_points.txt
22
23
  src/workstreams_cli.egg-info/requires.txt
23
24
  src/workstreams_cli.egg-info/top_level.txt
25
+ tests/test_confidence.py
24
26
  tests/test_config.py
25
27
  tests/test_event_log.py
26
28
  tests/test_models.py
@@ -0,0 +1,185 @@
1
+ """Tests for confidence scoring (0-10 self-rated result quality)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import sys
8
+ from pathlib import Path
9
+
10
+ import pytest
11
+
12
+ sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))
13
+
14
+ from workstreams.confidence import (
15
+ clamp_score,
16
+ confidence_records,
17
+ extract_score,
18
+ get_confidence,
19
+ )
20
+ from workstreams.models import SubagentEvent
21
+ from workstreams.event_log import EventLog
22
+
23
+
24
+ def _mk_event(score=None, subagent="claude-code", issue=7, event_type="completed", ws=1):
25
+ data = {}
26
+ if score is not None:
27
+ data["confidence"] = score
28
+ return SubagentEvent(
29
+ workstream_id=ws,
30
+ subagent=subagent,
31
+ issue=issue,
32
+ event_type=event_type,
33
+ message="msg",
34
+ data=data,
35
+ )
36
+
37
+
38
+ # ---------------------------------------------------------------------------
39
+ # clamp_score
40
+ # ---------------------------------------------------------------------------
41
+
42
+ def test_clamp_score_none():
43
+ assert clamp_score(None) is None
44
+
45
+
46
+ def test_clamp_score_valid_range():
47
+ assert clamp_score(5) == 5.0
48
+ assert clamp_score("7.5") == 7.5
49
+ assert clamp_score(0) == 0.0
50
+ assert clamp_score(10) == 10.0
51
+
52
+
53
+ def test_clamp_score_clamps_out_of_range():
54
+ assert clamp_score(-3) == 0.0
55
+ assert clamp_score(100) == 10.0
56
+
57
+
58
+ def test_clamp_score_garbage_is_none():
59
+ assert clamp_score("not-a-number") is None
60
+ assert clamp_score([1, 2]) is None
61
+
62
+
63
+ # ---------------------------------------------------------------------------
64
+ # extract_score
65
+ # ---------------------------------------------------------------------------
66
+
67
+ def test_extract_score_reads_confidence():
68
+ assert extract_score(_mk_event(score=8)) == 8.0
69
+
70
+
71
+ def test_extract_score_reads_score_key_too():
72
+ e = SubagentEvent(workstream_id=1, subagent="x", issue=1, event_type="completed", message="", data={"score": 6})
73
+ assert extract_score(e) == 6.0
74
+
75
+
76
+ def test_extract_score_missing_returns_none():
77
+ assert extract_score(_mk_event(score=None)) is None
78
+
79
+
80
+ # ---------------------------------------------------------------------------
81
+ # get_confidence aggregation
82
+ # ---------------------------------------------------------------------------
83
+
84
+ def test_get_confidence_pass(tmp_path, monkeypatch):
85
+ monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
86
+ from workstreams import event_log
87
+ event_log._log_cache.clear()
88
+
89
+ log = EventLog("p", log_dir=tmp_path / "p")
90
+ log.append(_mk_event(score=5))
91
+ log.append(_mk_event(score=9))
92
+
93
+ summary = get_confidence("p", workstream_id=1, issue=7, min_score=8)
94
+ assert summary.passed is True
95
+ assert summary.latest_score == 9.0
96
+ assert summary.n_reports == 2
97
+ assert summary.avg_score == pytest.approx(7.0)
98
+
99
+
100
+ def test_get_confidence_fail(tmp_path, monkeypatch):
101
+ monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
102
+ from workstreams import event_log
103
+ event_log._log_cache.clear()
104
+
105
+ log = EventLog("p", log_dir=tmp_path / "p")
106
+ log.append(_mk_event(score=3))
107
+
108
+ summary = get_confidence("p", workstream_id=1, issue=7, min_score=8)
109
+ assert summary.passed is False
110
+ assert summary.latest_score == 3.0
111
+
112
+
113
+ def test_get_confidence_no_reports(tmp_path, monkeypatch):
114
+ monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
115
+ from workstreams import event_log
116
+ event_log._log_cache.clear()
117
+
118
+ log = EventLog("p", log_dir=tmp_path / "p")
119
+ log.append(SubagentEvent(workstream_id=1, subagent="x", issue=7, event_type="started", message="", data={}))
120
+
121
+ summary = get_confidence("p", workstream_id=1, issue=7, min_score=8)
122
+ assert summary.n_reports == 0
123
+ assert summary.passed is False
124
+ assert summary.latest_score is None
125
+
126
+
127
+ def test_get_confidence_filters_unscored_events(tmp_path, monkeypatch):
128
+ monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
129
+ from workstreams import event_log
130
+ event_log._log_cache.clear()
131
+
132
+ log = EventLog("p", log_dir=tmp_path / "p")
133
+ log.append(_mk_event(score=None)) # no confidence -> ignored
134
+ log.append(_mk_event(score=10, subagent="codex"))
135
+
136
+ summary = get_confidence("p", workstream_id=1, issue=7, min_score=8)
137
+ assert summary.n_reports == 1
138
+ assert summary.latest_subagent == "codex"
139
+
140
+
141
+ def test_get_confidence_issue_filter(tmp_path, monkeypatch):
142
+ monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
143
+ from workstreams import event_log
144
+ event_log._log_cache.clear()
145
+
146
+ log = EventLog("p", log_dir=tmp_path / "p")
147
+ log.append(SubagentEvent(workstream_id=1, subagent="a", issue=1, event_type="completed", message="", data={"confidence": 2}))
148
+ log.append(SubagentEvent(workstream_id=1, subagent="a", issue=2, event_type="completed", message="", data={"confidence": 9}))
149
+
150
+ s1 = get_confidence("p", workstream_id=1, issue=1, min_score=8)
151
+ s2 = get_confidence("p", workstream_id=1, issue=2, min_score=8)
152
+ assert s1.latest_score == 2.0 and s1.passed is False
153
+ assert s2.latest_score == 9.0 and s2.passed is True
154
+
155
+
156
+ # ---------------------------------------------------------------------------
157
+ # confidence_records
158
+ # ---------------------------------------------------------------------------
159
+
160
+ def test_confidence_records_oldest_first(tmp_path, monkeypatch):
161
+ monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
162
+ from workstreams import event_log
163
+ event_log._log_cache.clear()
164
+
165
+ log = EventLog("p", log_dir=tmp_path / "p")
166
+ log.append(_mk_event(score=1, subagent="a"))
167
+ log.append(_mk_event(score=5, subagent="b"))
168
+ log.append(_mk_event(score=9, subagent="c"))
169
+
170
+ records = confidence_records("p", workstream_id=1, issue=7)
171
+ assert [r.subagent for r in records] == ["a", "b", "c"]
172
+ assert [r.score for r in records] == [1.0, 5.0, 9.0]
173
+
174
+
175
+ def test_confidence_records_excludes_unscored(tmp_path, monkeypatch):
176
+ monkeypatch.setenv("WORKSTREAMS_DATA_DIR", str(tmp_path))
177
+ from workstreams import event_log
178
+ event_log._log_cache.clear()
179
+
180
+ log = EventLog("p", log_dir=tmp_path / "p")
181
+ log.append(_mk_event(score=None, subagent="no-score"))
182
+ log.append(_mk_event(score=7, subagent="scored"))
183
+
184
+ records = confidence_records("p", workstream_id=1, issue=7)
185
+ assert [r.subagent for r in records] == ["scored"]
@@ -67,6 +67,40 @@ def test_load_nonexistent_returns_default_project(tmp_path):
67
67
  assert loaded.multiplexer == "tmux"
68
68
 
69
69
 
70
+ def test_load_json_when_no_yaml_present(tmp_path):
71
+ """Regression: when the on-disk config is .json (no pyyaml), load_config
72
+ must still read it back instead of silently defaulting to 0 workstreams.
73
+
74
+ This is the path that caused `start` to create a misnamed session with a
75
+ `wsNone` window: the workstream list was empty because the JSON file was
76
+ never consulted."""
77
+ cfg = _make_config(n=2)
78
+ project_dir = tmp_path / "jsonproj"
79
+ project_dir.mkdir()
80
+ # Save with pyyaml (writes .yaml), then remove the .yaml so only .json
81
+ # semantics remain — emulate an install without pyyaml.
82
+ saved = save_config(cfg, project_dir)
83
+ json_fallback = project_dir / ".workstreams.json"
84
+ if saved.suffix == ".yaml":
85
+ # Convert: write the equivalent JSON and drop the YAML so the loader
86
+ # is forced down the .json branch.
87
+ import json as _json
88
+ data = _json.loads(_json.dumps(cfg.to_dict()))
89
+ json_fallback.write_text(_json.dumps(data, indent=2) + "\n")
90
+ saved.unlink()
91
+ else:
92
+ json_fallback = saved
93
+
94
+ loaded = load_config(project_dir, "testproj")
95
+ assert loaded is not None
96
+ assert loaded.project == "testproj"
97
+ assert len(loaded.workstreams) == 2, (
98
+ f"expected 2 workstreams from JSON config, got {len(loaded.workstreams)}"
99
+ )
100
+ assert loaded.workstreams[0].name == "ws1"
101
+ assert loaded.workstreams[1].branch == "ws/2"
102
+
103
+
70
104
  def test_load_from_json_fallback(tmp_path):
71
105
  """When pyyaml is unavailable the config is saved as .json and must still load."""
72
106
  cfg = _make_config(n=1)