plan_driven 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +74 -1
- data/README.md +437 -37
- data/app/controllers/plan_driven/wizard/application_controller.rb +59 -0
- data/app/controllers/plan_driven/wizard/configuration_controller.rb +62 -0
- data/app/controllers/plan_driven/wizard/jobs_controller.rb +16 -0
- data/app/controllers/plan_driven/wizard/plans_controller.rb +113 -0
- data/app/views/layouts/plan_driven/wizard/application.html.erb +151 -0
- data/app/views/plan_driven/wizard/configuration/show.html.erb +104 -0
- data/app/views/plan_driven/wizard/plans/_agents.html.erb +44 -0
- data/app/views/plan_driven/wizard/plans/_approve.html.erb +40 -0
- data/app/views/plan_driven/wizard/plans/_finish.html.erb +31 -0
- data/app/views/plan_driven/wizard/plans/_plan.html.erb +52 -0
- data/app/views/plan_driven/wizard/plans/_tickets.html.erb +42 -0
- data/app/views/plan_driven/wizard/plans/index.html.erb +31 -0
- data/app/views/plan_driven/wizard/plans/new.html.erb +21 -0
- data/app/views/plan_driven/wizard/plans/show.html.erb +37 -0
- data/app/views/plan_driven/wizard/plans/statistics.html.erb +94 -0
- data/config/routes.rb +17 -0
- data/exe/plan-driven +1 -0
- data/lib/generators/plan_driven/install_generator.rb +7 -0
- data/lib/generators/plan_driven/templates/plan_driven.rb +10 -1
- data/lib/plan_driven/charts.rb +252 -0
- data/lib/plan_driven/cli/config_commands.rb +60 -0
- data/lib/plan_driven/cli/plan_commands.rb +6 -2
- data/lib/plan_driven/cli/setup_commands.rb +10 -0
- data/lib/plan_driven/cli/ticket_commands.rb +64 -3
- data/lib/plan_driven/cli/ui.rb +41 -2
- data/lib/plan_driven/cli.rb +21 -4
- data/lib/plan_driven/configuration.rb +31 -3
- data/lib/plan_driven/connections.rb +69 -0
- data/lib/plan_driven/cursor_agents.rb +10 -2
- data/lib/plan_driven/delivery.rb +23 -5
- data/lib/plan_driven/evidence.rb +11 -4
- data/lib/plan_driven/github.rb +4 -0
- data/lib/plan_driven/guards/migration_guard.rb +86 -9
- data/lib/plan_driven/guards/ticket_guard.rb +1 -2
- data/lib/plan_driven/interview.rb +161 -0
- data/lib/plan_driven/local_agents.rb +254 -0
- data/lib/plan_driven/renderer/html.rb +34 -5
- data/lib/plan_driven/renderer/markdown.rb +67 -3
- data/lib/plan_driven/renderer/style.css +21 -0
- data/lib/plan_driven/renderer.rb +11 -4
- data/lib/plan_driven/statistics.rb +195 -0
- data/lib/plan_driven/template.rb +7 -2
- data/lib/plan_driven/usage.rb +114 -0
- data/lib/plan_driven/version.rb +1 -1
- data/lib/plan_driven/wizard/engine.rb +19 -0
- data/lib/plan_driven/wizard.rb +245 -0
- data/lib/plan_driven.rb +8 -1
- metadata +32 -6
data/README.md
CHANGED
|
@@ -6,11 +6,16 @@
|
|
|
6
6
|
[](#rails-and-ruby-support)
|
|
7
7
|
[](#rails-and-ruby-support)
|
|
8
8
|
|
|
9
|
-
**From implementation plan to merged, tested pull requests, driven from the terminal.**
|
|
9
|
+
**From implementation plan to merged, tested pull requests, driven from the browser or the terminal.**
|
|
10
10
|
|
|
11
|
-
`plan_driven` runs a Rails team's delivery process
|
|
12
|
-
the writing and your team making the decisions.
|
|
13
|
-
|
|
11
|
+
`plan_driven` runs a Rails team's delivery process inside your Rails app, with AI agents doing
|
|
12
|
+
the writing and your team making the decisions. Drive it the way you prefer: click through the
|
|
13
|
+
[wizard in the browser](#the-browser-wizard), mounted at `/plan_driven` in development, or type
|
|
14
|
+
the same steps in the [terminal](#commands). Every button in the wizard runs one `plan-driven`
|
|
15
|
+
command and shows it to you, so both are the same process, with the same rules and the same
|
|
16
|
+
audit trail, and you can switch between them at any step.
|
|
17
|
+
|
|
18
|
+
A short interview becomes an implementation plan grounded in your real schema and code. Guards written in Ruby check the
|
|
14
19
|
plan, you read it and approve it. The approved plan becomes tickets, each ticket goes to a
|
|
15
20
|
Cursor cloud agent that opens a pull request, and only the pull requests you approve are
|
|
16
21
|
merged. Acceptance criteria map to Cucumber scenarios, so the delivery report shows which
|
|
@@ -24,10 +29,50 @@ Zagreb. [Need Rails engineers?](#about-rubycode)
|
|
|
24
29
|
|
|
25
30
|
## Watch it deliver a feature
|
|
26
31
|
|
|
27
|
-
[](https://github.com/blaz1988/plan-driven/releases/download/v0.1.0/plan-driven-wizard.mp4)
|
|
33
|
+
|
|
34
|
+
**[▶ Watch the wizard demo](https://github.com/blaz1988/plan-driven/releases/download/v0.1.0/plan-driven-wizard.mp4)**
|
|
35
|
+
(12 minutes, narrated, with captions). One feature, comments on events, is delivered from
|
|
36
|
+
the [browser wizard](#the-browser-wizard) in [Gather](https://github.com/blaz1988/gather):
|
|
37
|
+
the plan drafted and one section redrafted, six tickets as issues
|
|
38
|
+
([#34](https://github.com/blaz1988/gather/issues/34) to
|
|
39
|
+
[#39](https://github.com/blaz1988/gather/issues/39)), six agents and six pull requests
|
|
40
|
+
([#40](https://github.com/blaz1988/gather/pull/40) to
|
|
41
|
+
[#45](https://github.com/blaz1988/gather/pull/45)), one round of feedback, and 42 of 42
|
|
42
|
+
acceptance criteria proven by a passing scenario. After every click the video zooms into the
|
|
43
|
+
wizard's terminal panel, which shows the `plan-driven` command that ran and its output. The
|
|
44
|
+
wizard ships with 0.2.0.
|
|
45
|
+
|
|
46
|
+
<details>
|
|
47
|
+
<summary>Chapters</summary>
|
|
48
|
+
|
|
49
|
+
| Time | Chapter |
|
|
50
|
+
| ---: | --- |
|
|
51
|
+
| 0:00 | What plan_driven is |
|
|
52
|
+
| 0:24 | Installing it, and how the wizard works |
|
|
53
|
+
| 1:13 | `doctor`, from the wizard |
|
|
54
|
+
| 1:48 | The interview as a form, and the draft |
|
|
55
|
+
| 2:43 | Reading the plan, redrafting a section, submitting and approving |
|
|
56
|
+
| 3:51 | Tickets drafted and approved, as GitHub issues |
|
|
57
|
+
| 4:37 | Starting the agents |
|
|
58
|
+
| 5:18 | Refreshing, and `review` |
|
|
59
|
+
| 6:01 | Reading the first pull request |
|
|
60
|
+
| 6:46 | Approving and merging, and the next agent starts |
|
|
61
|
+
| 7:43 | Feedback to the agent on T3 |
|
|
62
|
+
| 8:24 | Every ticket merged |
|
|
63
|
+
| 8:53 | Evidence (42 of 42), tokens, and the delivery report |
|
|
64
|
+
| 9:47 | Reading the delivery report |
|
|
65
|
+
| 10:11 | The feature in the app |
|
|
66
|
+
| 10:46 | The same commands in a terminal |
|
|
67
|
+
| 11:22 | Recap |
|
|
68
|
+
|
|
69
|
+
</details>
|
|
70
|
+
|
|
71
|
+
### The CLI deep dive
|
|
28
72
|
|
|
29
|
-
**[▶ Watch the demo](https://github.com/blaz1988/plan-driven/releases/download/v0.1.0/plan-driven-demo.mp4)**
|
|
30
|
-
(
|
|
73
|
+
**[▶ Watch the CLI demo](https://github.com/blaz1988/plan-driven/releases/download/v0.1.0/plan-driven-demo.mp4)**
|
|
74
|
+
(28 minutes, narrated, with captions), for everything the wizard runs, typed in a terminal.
|
|
75
|
+
One feature, RSVPs with a waitlist, goes from an
|
|
31
76
|
idea to production code in a new Rails 8 app,
|
|
32
77
|
[Gather](https://github.com/blaz1988/gather). Nothing in it is staged: the plan, the five
|
|
33
78
|
tickets, the five pull requests ([#8](https://github.com/blaz1988/gather/pull/8) to
|
|
@@ -44,18 +89,20 @@ repository. The planner and the five agents ran on Claude Opus 5.5 through Curso
|
|
|
44
89
|
| 2:49 | `doctor`: keys, repository and the Cursor connection |
|
|
45
90
|
| 3:15 | The interview |
|
|
46
91
|
| 4:14 | Reading the implementation plan |
|
|
47
|
-
| 6:05 | Deciding the open questions
|
|
48
|
-
|
|
|
49
|
-
| 9:
|
|
50
|
-
| 10:
|
|
51
|
-
|
|
|
52
|
-
| 13:
|
|
53
|
-
|
|
|
54
|
-
|
|
|
55
|
-
|
|
|
56
|
-
|
|
|
57
|
-
| 23:
|
|
58
|
-
| 24:
|
|
92
|
+
| 6:05 | Deciding the open questions with `redraft` |
|
|
93
|
+
| 6:50 | Every section can be changed: Database changes redrafted and edited in vim (PD-2) |
|
|
94
|
+
| 9:51 | Submitting and approving the plan |
|
|
95
|
+
| 10:23 | Tickets: drafted, steered to five, read and approved |
|
|
96
|
+
| 12:44 | The tickets as GitHub issues |
|
|
97
|
+
| 13:15 | What the agent is told, and starting the first agent |
|
|
98
|
+
| 14:10 | Reading the first pull request, `review`, `approve-pr` and `merge` |
|
|
99
|
+
| 16:21 | The model and its rules (T2) |
|
|
100
|
+
| 18:25 | The RSVP card, tried on the branch before merging (T3) |
|
|
101
|
+
| 20:44 | Feedback: a behind-main warning and a refactor (T5) |
|
|
102
|
+
| 23:15 | The organizer's attendee list (T4) |
|
|
103
|
+
| 24:50 | Evidence: 33 of 33 acceptance criteria, and the delivery report |
|
|
104
|
+
| 26:04 | The finished feature in the app |
|
|
105
|
+
| 27:15 | Recap |
|
|
59
106
|
|
|
60
107
|
</details>
|
|
61
108
|
|
|
@@ -65,11 +112,16 @@ repository. The planner and the five agents ran on Claude Opus 5.5 through Curso
|
|
|
65
112
|
- [Requirements](#requirements)
|
|
66
113
|
- [Installation](#installation)
|
|
67
114
|
- [Getting your app ready](#getting-your-app-ready)
|
|
115
|
+
- [The browser wizard](#the-browser-wizard)
|
|
68
116
|
- [Walkthrough: one feature from idea to merged](#walkthrough-one-feature-from-idea-to-merged)
|
|
69
117
|
- [Guards](#guards)
|
|
70
118
|
- [What the agent is told](#what-the-agent-is-told)
|
|
71
119
|
- [Commands](#commands)
|
|
72
120
|
- [Configuration](#configuration)
|
|
121
|
+
- [Choosing the coding agents](#choosing-the-coding-agents)
|
|
122
|
+
- [Tokens and cost](#tokens-and-cost)
|
|
123
|
+
- [Statistics](#statistics)
|
|
124
|
+
- [How it compares](#how-it-compares)
|
|
73
125
|
- [Keys](#keys)
|
|
74
126
|
- [Working as a team](#working-as-a-team)
|
|
75
127
|
- [Troubleshooting](#troubleshooting)
|
|
@@ -93,7 +145,7 @@ repository. The planner and the five agents ran on Claude Opus 5.5 through Curso
|
|
|
93
145
|
plan-driven approve-tickets ─▶ tickets approved GitHub issues created
|
|
94
146
|
│
|
|
95
147
|
▼
|
|
96
|
-
plan-driven develop one
|
|
148
|
+
plan-driven develop one agent per ready ticket (Cursor cloud, or a local CLI), one PR each
|
|
97
149
|
plan-driven status agent finished ─▶ PR open
|
|
98
150
|
plan-driven review PrGuard: scope, specs, Cucumber scenarios, CI, up to date
|
|
99
151
|
plan-driven feedback the same agent pushes a fix to the same PR
|
|
@@ -102,7 +154,7 @@ repository. The planner and the five agents ran on Claude Opus 5.5 through Curso
|
|
|
102
154
|
│
|
|
103
155
|
▼
|
|
104
156
|
plan-driven evidence Cucumber results mapped to acceptance criteria
|
|
105
|
-
plan-driven report docs/plans/pd-1-…/delivery-report.pdf
|
|
157
|
+
plan-driven report docs/plans/pd-1-…/delivery-report.pdf, with tokens and cost
|
|
106
158
|
```
|
|
107
159
|
|
|
108
160
|
Phases are stored in your application's database, so a plan can't skip a step. Tickets can't
|
|
@@ -119,9 +171,12 @@ model as a list to fix, and a plan that still fails isn't accepted.
|
|
|
119
171
|
|
|
120
172
|
- Ruby 3.1+ and Rails 7.0+ (see [support](#rails-and-ruby-support)).
|
|
121
173
|
- The application on GitHub, with CI running on pull requests.
|
|
122
|
-
-
|
|
123
|
-
|
|
124
|
-
|
|
174
|
+
- Coding agents, one of (see [Choosing the coding agents](#choosing-the-coding-agents)):
|
|
175
|
+
- **Cursor cloud agents** (the default): a [Cursor](https://cursor.com) account with the
|
|
176
|
+
GitHub integration connected to that repository, and a Cursor API key (Cursor dashboard,
|
|
177
|
+
Integrations).
|
|
178
|
+
- **A local agent CLI** (`agent_provider :local`): Claude Code, Codex, the Cursor CLI or any
|
|
179
|
+
command that edits files in its working directory, plus `git` push access to the repository.
|
|
125
180
|
- A model for drafting plans and tickets, one of:
|
|
126
181
|
- **Cursor** (`llm_provider :cursor`): any model on your Cursor account, Claude Opus 5.5 by
|
|
127
182
|
default. Needs Node 22.13+ and the Cursor SDK. No other LLM key.
|
|
@@ -204,7 +259,7 @@ into your application. See [Keys](#keys).
|
|
|
204
259
|
```
|
|
205
260
|
$ bin/plan-driven doctor
|
|
206
261
|
|
|
207
|
-
plan-driven 0.
|
|
262
|
+
plan-driven 0.2.0
|
|
208
263
|
✓ Rails application: plan_driven tables present
|
|
209
264
|
LLM: cursor/claude-opus-5-5
|
|
210
265
|
✓ Cursor SDK: Node v24.21.0, @cursor/sdk found
|
|
@@ -263,6 +318,81 @@ config.extra_context = "Gather lists community events. An event has an organizer
|
|
|
263
318
|
**Who approves.** `plan_approvals` and `ticket_approvals` list the roles that must sign off,
|
|
264
319
|
for example `%w[review qa devops director]`. One person can hold every role on a small team.
|
|
265
320
|
|
|
321
|
+
## The browser wizard
|
|
322
|
+
|
|
323
|
+
Not everyone wants to drive a delivery from the terminal. The install generator mounts a
|
|
324
|
+
wizard in your app, in development only:
|
|
325
|
+
|
|
326
|
+
```ruby
|
|
327
|
+
# config/routes.rb
|
|
328
|
+
mount PlanDriven::Wizard::Engine, at: "/plan_driven" if Rails.env.development?
|
|
329
|
+
```
|
|
330
|
+
|
|
331
|
+
Start the app and open `http://localhost:3000/plan_driven`. It walks a plan through the same
|
|
332
|
+
five steps, with Back and Next: **Plan** (the interview, then read, edit or redraft any section
|
|
333
|
+
and submit), **Approve**, **Tickets**, **Agents & PRs** and **Proof & report**. A step opens
|
|
334
|
+
once the plan has reached it.
|
|
335
|
+
|
|
336
|
+
The wizard is a front end for the CLI, not a second implementation. Every button runs one
|
|
337
|
+
`plan-driven` command in the background, and the panel on the right shows that command and its
|
|
338
|
+
output as it runs, exactly as you'd see it in a terminal:
|
|
339
|
+
|
|
340
|
+

|
|
341
|
+
|
|
342
|
+
```
|
|
343
|
+
$ bin/plan-driven edit PD-3 database_changes --from tmp/plan_driven/wizard/sections/PD-3-database_changes-1f2e.md --yes
|
|
344
|
+
✓ Database changes updated; PD-3 is now revision 2 (draft)
|
|
345
|
+
✓ All checks passed
|
|
346
|
+
```
|
|
347
|
+
|
|
348
|
+
So anything done in the browser can be repeated, scripted or reviewed from the terminal, and
|
|
349
|
+
the audit trail is the same either way. A few things to know:
|
|
350
|
+
|
|
351
|
+
- Only a fixed list of commands can run, built from the form fields as an argument list, never
|
|
352
|
+
through a shell. A key pasted on the Configuration page goes to `plan-driven connect` on
|
|
353
|
+
stdin, so it's never in the command line, the panel or the logs.
|
|
354
|
+
- It answers local requests only, and only in development. `config.wizard_enabled = true`
|
|
355
|
+
turns it on in another environment, still for local requests only.
|
|
356
|
+
- "Acting as" at the top sets `PLAN_DRIVEN_ACTOR` for the commands it runs, so approvals are
|
|
357
|
+
recorded under the name you give; it defaults to your git identity.
|
|
358
|
+
- Each run is kept in `tmp/plan_driven/wizard/`: the command, its output and its exit status.
|
|
359
|
+
|
|
360
|
+
### Configuration: connections and the interview
|
|
361
|
+
|
|
362
|
+
**Configuration**, at the top of every page, has two parts.
|
|
363
|
+
|
|
364
|
+
**Connections** shows which services this app's configuration uses (Cursor for cloud agents
|
|
365
|
+
or drafting, OpenAI or Anthropic for drafting, GitHub for issues and pull requests), whether
|
|
366
|
+
each one has a key, and where the key comes from. Paste a key and click Connect:
|
|
367
|
+
`plan-driven connect cursor` checks it with the service first (for Cursor, the account it
|
|
368
|
+
belongs to) and only then stores it in `~/.plan_driven/config`. A refused key is never stored.
|
|
369
|
+
**Check every connection** runs `doctor`. The same works in the terminal:
|
|
370
|
+
|
|
371
|
+
```
|
|
372
|
+
$ bin/plan-driven connect cursor
|
|
373
|
+
Cursor key:
|
|
374
|
+
Checking the key with Cursor...
|
|
375
|
+
✓ Cursor: connected as ana@example.com
|
|
376
|
+
stored in ~/.plan_driven/config (0600), never in the app
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
Which model drafts and which agents write the code are still set in the initializer (see
|
|
380
|
+
[Choosing the coding agents](#choosing-the-coding-agents)); the page shows the current choice.
|
|
381
|
+
|
|
382
|
+
**Interview questions** lists what `plan-driven new` and the New plan form ask. Change a
|
|
383
|
+
question's title or wording, make it required or optional, add your own questions, or put one
|
|
384
|
+
back to the default. Each Save runs `plan-driven question`:
|
|
385
|
+
|
|
386
|
+
```
|
|
387
|
+
$ bin/plan-driven question success_metric --title "Success metric" --ask "How will we know it worked?" --optional
|
|
388
|
+
✓ Question success_metric added
|
|
389
|
+
```
|
|
390
|
+
|
|
391
|
+
The changes are written to `config/plan_driven/interview.yml` in your app. Commit it, and the
|
|
392
|
+
whole team gets the same interview, in the wizard and in the terminal. An added question goes
|
|
393
|
+
to the model with the other answers, and it's a section of the plan under the group you pick.
|
|
394
|
+
Drafted sections (Database changes, Risks...) belong to the model and can't be changed here.
|
|
395
|
+
|
|
266
396
|
## Walkthrough: one feature from idea to merged
|
|
267
397
|
|
|
268
398
|
This is the run from the demo video, in [Gather](https://github.com/blaz1988/gather), with the
|
|
@@ -319,8 +449,22 @@ column it cites exists.
|
|
|
319
449
|
|
|
320
450
|

|
|
321
451
|
|
|
322
|
-
|
|
323
|
-
|
|
452
|
+
**Every section of the plan can be changed**, not only Outstanding questions: What, Why,
|
|
453
|
+
Database changes, Application changes, Risks, Testing, any of them (the keys are listed under
|
|
454
|
+
[Commands](#commands)). The draft is the model's proposal, and your team has the final say.
|
|
455
|
+
There are two ways to change a section:
|
|
456
|
+
|
|
457
|
+
- `edit PLAN SECTION` opens the section in `$VISUAL` or `$EDITOR` as Markdown. Use it for exact
|
|
458
|
+
changes: a column, a name, a step, a test case.
|
|
459
|
+
- `redraft PLAN SECTION "instruction"` has the model rewrite only that section, following your
|
|
460
|
+
instruction. It reads the rest of the plan and the code while it does.
|
|
461
|
+
|
|
462
|
+
Either way the change becomes a new revision, the plan goes back to draft, the guards run again
|
|
463
|
+
and the Markdown, HTML and PDF are written again. `plan-driven log PLAN` lists every change.
|
|
464
|
+
Change the plan through these commands, not by editing `plan.md`: the database is the source,
|
|
465
|
+
and the files are rendered from it.
|
|
466
|
+
|
|
467
|
+
`redraft` is also the quickest way to record decisions:
|
|
324
468
|
|
|
325
469
|
```
|
|
326
470
|
$ bin/plan-driven redraft PD-1 outstanding_questions "Record my answers under Decided and keep only
|
|
@@ -337,6 +481,80 @@ Redrafting outstanding_questions...
|
|
|
337
481
|
|
|
338
482
|

|
|
339
483
|
|
|
484
|
+
#### Example: changing the database design
|
|
485
|
+
|
|
486
|
+
The agent's draft is a starting point, and the database design is where teams most often
|
|
487
|
+
disagree with it. In Gather's second plan, PD-2 (event categories), the model suggested a string
|
|
488
|
+
column on `events`:
|
|
489
|
+
|
|
490
|
+
```
|
|
491
|
+
$ bin/plan-driven show PD-2 --section database_changes
|
|
492
|
+
### Step 1: Expand (migration `AddCategoryToEvents`)
|
|
493
|
+
|
|
494
|
+
On table `events`:
|
|
495
|
+
- Add column `category`: type `string`, **nullable**, default `'meetup'`.
|
|
496
|
+
- Add composite index `index_events_on_category_and_starts_at` on `[:category, :starts_at]`. ...
|
|
497
|
+
```
|
|
498
|
+
|
|
499
|
+
The team wanted a table instead. `redraft` rewrites the section to that design, keeping the
|
|
500
|
+
expand and contract steps:
|
|
501
|
+
|
|
502
|
+
```
|
|
503
|
+
$ bin/plan-driven redraft PD-2 database_changes "Use a categories table instead of a string column:
|
|
504
|
+
name and slug, seeded with meetup, workshop, talk and conference, and a category_id reference on events."
|
|
505
|
+
Redrafting database_changes...
|
|
506
|
+
Categories live in their own `categories` table, and each event points to one of them through
|
|
507
|
+
`events.category_id`. ...
|
|
508
|
+
### Step 1: Expand
|
|
509
|
+
#### Migration `CreateCategories`
|
|
510
|
+
New table `categories`:
|
|
511
|
+
- `name` string, **not null**, no default. ...
|
|
512
|
+
- `slug` string, **not null**, no default. ...
|
|
513
|
+
...
|
|
514
|
+
### Step 5: Contract (migration `EnforceCategoryOnEvents`)
|
|
515
|
+
- `change_column_null :events, :category_id, false`.
|
|
516
|
+
✓ All checks passed
|
|
517
|
+
```
|
|
518
|
+
|
|
519
|
+
A small change doesn't need the model. `edit` opens the section in your editor. Here, a `color`
|
|
520
|
+
column is added to the new table, in the column list, the migration and the resulting schema:
|
|
521
|
+
|
|
522
|
+
```
|
|
523
|
+
$ EDITOR=vim bin/plan-driven edit PD-2 database_changes
|
|
524
|
+
✓ Database changes updated; PD-2 is now revision 3 (draft)
|
|
525
|
+
✓ All checks passed
|
|
526
|
+
```
|
|
527
|
+
|
|
528
|
+

|
|
529
|
+
|
|
530
|
+
The guards check your edit the same way they check the model's draft (see [Guards](#guards)),
|
|
531
|
+
and anything that fails is listed right after you save. `submit` refuses a plan with errors.
|
|
532
|
+
|
|
533
|
+
Other sections that depend on the change follow the same way. The model reads the whole plan,
|
|
534
|
+
so it picks up the new table and the `color` column:
|
|
535
|
+
|
|
536
|
+
```
|
|
537
|
+
$ bin/plan-driven redraft PD-2 application_changes "Follow the new Database changes: a Category model,
|
|
538
|
+
events.category_id instead of an enum, and the category colour on the card badge."
|
|
539
|
+
...
|
|
540
|
+
✓ All checks passed
|
|
541
|
+
|
|
542
|
+
$ bin/plan-driven submit PD-2
|
|
543
|
+
✓ PD-2 revision 4 is in review
|
|
544
|
+
|
|
545
|
+
$ bin/plan-driven log PD-2
|
|
546
|
+
When Event Ticket By Details
|
|
547
|
+
2026-09-30 10:07 plan.drafted Ivan Blažević <ivan...> model=cursor/claude-opus-5-5 ...
|
|
548
|
+
2026-09-30 10:10 plan.revised Ivan Blažević <ivan...> sections=database_changes
|
|
549
|
+
2026-09-30 10:16 plan.revised Ivan Blažević <ivan...> sections=database_changes
|
|
550
|
+
...
|
|
551
|
+
```
|
|
552
|
+
|
|
553
|
+
You can change a plan in draft, in review and after it's approved. An approval belongs to a
|
|
554
|
+
revision, so a changed plan must be approved again. Once its tickets are drafted, the plan is
|
|
555
|
+
locked, because the tickets and pull requests were built from it. Changes after that go into a
|
|
556
|
+
follow-up plan.
|
|
557
|
+
|
|
340
558
|
### 3. Submit and approve
|
|
341
559
|
|
|
342
560
|
```
|
|
@@ -505,8 +723,9 @@ $ bin/plan-driven report PD-1
|
|
|
505
723
|
|
|
506
724
|
`evidence` runs the plan's scenarios on your machine and stores the result with the commit it
|
|
507
725
|
ran on; `--from cucumber.json` imports a run from CI instead. The delivery report lists each
|
|
508
|
-
ticket with its pull request, merge commit and approver,
|
|
509
|
-
|
|
726
|
+
ticket with its pull request, merge commit and approver, the [statistics](#statistics) with
|
|
727
|
+
their charts, then every acceptance criterion with the scenario that proves it, marked passed
|
|
728
|
+
or failed in colour, the guard findings, every approval and the full timeline. Commit
|
|
510
729
|
`docs/plans/` with it, and the plan and its proof stay next to the code.
|
|
511
730
|
|
|
512
731
|

|
|
@@ -530,8 +749,12 @@ An error blocks the next step and a warning is shown and recorded.
|
|
|
530
749
|
|
|
531
750
|
**MigrationGuard** reads the database changes:
|
|
532
751
|
|
|
533
|
-
- removing or renaming a column or table in one step is an error
|
|
534
|
-
|
|
752
|
+
- removing or renaming a column or table in one step is an error. Each change is judged on its
|
|
753
|
+
own, sentence by sentence and line by line in migration code: it's accepted only when that
|
|
754
|
+
sentence stages it (`ignored_columns`, a later release, after the backfill) or when it sits
|
|
755
|
+
under a contract, cleanup or later step. Mentioning "expand" somewhere else in the section
|
|
756
|
+
doesn't excuse it. Headings, negated sentences ("No column is removed"), rollback notes and
|
|
757
|
+
tables the plan itself creates don't count as removals;
|
|
535
758
|
- NOT NULL on an existing table without a default or backfill is a warning;
|
|
536
759
|
- on PostgreSQL, an index that isn't built concurrently is a warning.
|
|
537
760
|
|
|
@@ -623,14 +846,19 @@ key such as `PD-1`, and `PLAN/TICKET` is a ticket such as `PD-1/T3`.
|
|
|
623
846
|
| `evidence PLAN [--from FILE]` | Run or import Cucumber results |
|
|
624
847
|
| `report PLAN` | Write the delivery report |
|
|
625
848
|
| `log PLAN` | The audit trail |
|
|
849
|
+
| `usage PLAN` | Tokens, time and cost per step and per agent run |
|
|
850
|
+
| `stats PLAN` | Where the time went: phases, agents and people, each ticket |
|
|
851
|
+
| `questions` | The interview's questions, and which ones the team changed or added |
|
|
852
|
+
| `question KEY [--title T] [--ask Q] [--group G] [--required \| --optional] [--remove]` | Change or add an interview question, or put it back |
|
|
626
853
|
| `configure` | Store keys in `~/.plan_driven/config` |
|
|
627
|
-
| `
|
|
854
|
+
| `connect SERVICE` | Check a key with `cursor`, `openai`, `anthropic` or `github`, then store it |
|
|
855
|
+
| `doctor` | Check keys, repository, PDF browser, and the Cursor connection or local agent command |
|
|
628
856
|
|
|
629
857
|
Section keys for `show --section`, `edit` and `redraft`: `what`, `why`, `where`, `who`,
|
|
630
858
|
`when`, `background`, `existing_data_structure`, `architecture`, `database_changes`,
|
|
631
859
|
`application_changes`, `infrastructure_changes`, `out_of_scope`, `risks`, `performance`,
|
|
632
|
-
`security`, `monitoring`, `outstanding_questions` and `testing
|
|
633
|
-
lists them.
|
|
860
|
+
`security`, `monitoring`, `outstanding_questions` and `testing`, plus any question the team
|
|
861
|
+
added. `edit PLAN` without a section lists them.
|
|
634
862
|
|
|
635
863
|
`--yes` skips confirmations, for scripts. `merge` still needs the ticket key typed unless
|
|
636
864
|
`--yes` is given.
|
|
@@ -657,11 +885,17 @@ PlanDriven.configure do |config|
|
|
|
657
885
|
config.plan_approvals = %w[review] # e.g. %w[review qa devops director]
|
|
658
886
|
config.ticket_approvals = %w[review]
|
|
659
887
|
|
|
660
|
-
#
|
|
661
|
-
config.
|
|
888
|
+
# Coding agents
|
|
889
|
+
config.agent_provider = :cursor # or :local, see "Choosing the coding agents"
|
|
890
|
+
config.agent_command = nil # :local only, e.g. "claude -p --permission-mode acceptEdits --output-format json"
|
|
891
|
+
config.agent_timeout = 3600 # :local only; seconds before a run is stopped
|
|
892
|
+
config.agent_model = nil # :cursor; nil uses your Cursor default
|
|
662
893
|
config.base_branch = "main"
|
|
663
894
|
config.max_parallel_agents = 3
|
|
664
|
-
config.skip_reviewer_request = false # true: the agent doesn't request you as reviewer
|
|
895
|
+
config.skip_reviewer_request = false # :cursor; true: the agent doesn't request you as reviewer
|
|
896
|
+
|
|
897
|
+
# Tokens and cost: dollars per million tokens, by model id. None ship with the gem.
|
|
898
|
+
config.token_prices = {} # { "model-id" => { input: 3.0, output: 15.0, cache_write: 3.75, cache_read: 0.3 } }
|
|
665
899
|
|
|
666
900
|
# GitHub
|
|
667
901
|
config.github_repository = nil # "owner/name"; read from the origin remote when nil
|
|
@@ -695,6 +929,172 @@ plan and can't change a file.
|
|
|
695
929
|
|
|
696
930
|
`config.template` replaces the plan's sections if your template differs.
|
|
697
931
|
|
|
932
|
+
## Choosing the coding agents
|
|
933
|
+
|
|
934
|
+
Every ticket goes to one agent, which works on its own branch and opens one pull request. The
|
|
935
|
+
rest of the workflow is the same whichever agents you use: `review`, `feedback`, `approve-pr`
|
|
936
|
+
and `merge` see only the pull request.
|
|
937
|
+
|
|
938
|
+
**Cursor cloud agents** (`agent_provider :cursor`, the default) run on Cursor-hosted machines
|
|
939
|
+
against a fresh clone of the repository, so nothing runs on your laptop and several tickets
|
|
940
|
+
can run at once. `config.agent_model` picks the model.
|
|
941
|
+
|
|
942
|
+
**A local agent CLI** (`agent_provider :local`) runs a command on your machine. Each ticket
|
|
943
|
+
gets its own git worktree and branch under `tmp/plan_driven/agents`, so tickets don't touch
|
|
944
|
+
your working copy or each other. The command gets the same prompt a cloud agent gets, on stdin,
|
|
945
|
+
or wherever the command says `{prompt_file}`. When it exits cleanly, plan_driven commits what
|
|
946
|
+
it left, pushes the branch and opens the pull request, using the description the agent wrote
|
|
947
|
+
to `PR_DESCRIPTION.md`. Feedback runs the command again in the same worktree and pushes to the
|
|
948
|
+
same pull request. After the merge, the worktree and the local branch are removed.
|
|
949
|
+
|
|
950
|
+
```ruby
|
|
951
|
+
config.agent_provider = :local
|
|
952
|
+
|
|
953
|
+
# Claude Code
|
|
954
|
+
config.agent_command = "claude -p --permission-mode acceptEdits --output-format json"
|
|
955
|
+
# Codex
|
|
956
|
+
config.agent_command = "codex exec --full-auto -"
|
|
957
|
+
# The Cursor CLI
|
|
958
|
+
config.agent_command = 'cursor-agent -p --force --output-format json "$(cat {prompt_file})"'
|
|
959
|
+
```
|
|
960
|
+
|
|
961
|
+
A local agent runs with your permissions and your shell, so give it only the tools it needs to
|
|
962
|
+
edit and to run the test suite, and read the pull request as carefully as a cloud agent's.
|
|
963
|
+
Runs longer than `config.agent_timeout` (an hour by default) are stopped. The Cursor CLI setup
|
|
964
|
+
is the one tested end to end; flags change between CLI versions, so check your CLI's `--help`.
|
|
965
|
+
`plan-driven doctor` checks that the command is on the `PATH`.
|
|
966
|
+
|
|
967
|
+
## Tokens and cost
|
|
968
|
+
|
|
969
|
+
Every model call and every agent run is recorded with the tokens it used: drafting the plan,
|
|
970
|
+
each `redraft`, drafting the tickets, and each agent run and follow-up, with its duration.
|
|
971
|
+
`plan-driven usage PD-1` prints them, and the delivery report has a Tokens and cost table.
|
|
972
|
+
|
|
973
|
+
Prices change and differ per account, so the gem ships none: put what your provider charges in
|
|
974
|
+
`config.token_prices` (dollars per million tokens, with separate cache prices), and the report
|
|
975
|
+
shows dollars next to the tokens. Without a price you still get the tokens and the time.
|
|
976
|
+
|
|
977
|
+
Cursor reports tokens for every cloud agent run. A local CLI's tokens are recorded when it
|
|
978
|
+
prints them the way Claude Code's `--output-format json` does; otherwise only the time is.
|
|
979
|
+
|
|
980
|
+
For scale, these are the five cloud agents from the demo, read back from Cursor's usage API
|
|
981
|
+
(T4 and T5 include their feedback runs):
|
|
982
|
+
|
|
983
|
+
| Ticket | Runs | Output tokens | Cache reads | Total tokens |
|
|
984
|
+
| --- | ---: | ---: | ---: | ---: |
|
|
985
|
+
| T1 migration | 1 | 13,861 | 738,595 | 791,861 |
|
|
986
|
+
| T2 model rules | 1 | 21,037 | 1,519,826 | 1,596,409 |
|
|
987
|
+
| T3 RSVP card | 1 | 16,611 | 1,161,514 | 1,242,945 |
|
|
988
|
+
| T4 attendee list | 2 | 10,957 | 1,083,671 | 1,149,274 |
|
|
989
|
+
| T5 seats on the index | 2 | 12,725 | 1,154,333 | 1,248,462 |
|
|
990
|
+
| **Total** | 7 | **75,191** | **5,657,939** | **6,028,951** |
|
|
991
|
+
|
|
992
|
+
94% of the tokens are cache reads, which cost a fraction of fresh input, and only 75 thousand
|
|
993
|
+
are code and text the agents wrote. Most of an agent's tokens go into reading the codebase, so
|
|
994
|
+
a small, conventional one is cheaper to work on.
|
|
995
|
+
|
|
996
|
+
## Statistics
|
|
997
|
+
|
|
998
|
+
Where did the time go? Was it the agents writing code, or the pull requests waiting for a
|
|
999
|
+
person? The same numbers are in three places:
|
|
1000
|
+
|
|
1001
|
+
- **In the wizard:** every plan has a Statistics page, linked under its title and from the
|
|
1002
|
+
Proof & report step. In development that's `http://localhost:3000/plan_driven/plans/PD-1/statistics`.
|
|
1003
|
+
- **In the terminal:** `bin/plan-driven stats PD-1`.
|
|
1004
|
+
- **In the delivery report:** a Statistics section with the same charts, in the Markdown, the
|
|
1005
|
+
HTML and the PDF.
|
|
1006
|
+
|
|
1007
|
+

|
|
1008
|
+
|
|
1009
|
+
This is PD-3 from the demo: six tickets delivered in 1 h 20 min, 42 of 42 criteria proven.
|
|
1010
|
+
|
|
1011
|
+
- **Cards:** idea to delivery, development time, criteria proven, pull requests approved the
|
|
1012
|
+
first time, the agents' share of the work, and tokens.
|
|
1013
|
+
- **Acceptance criteria, merged and proven:** a burn-up against the plan's scope. The blue
|
|
1014
|
+
line rises as each ticket merges with its criteria, and the green line rises when an
|
|
1015
|
+
evidence run proves them. Every run is a dot, red when it failed. Here, two runs failed
|
|
1016
|
+
around 14:50 and the third proved all 42.
|
|
1017
|
+
- **Where the time went, ticket by ticket:** one row per ticket on a shared clock. Grey is
|
|
1018
|
+
queued, waiting for the tickets it depends on. Blue is an agent coding, amber is the pull
|
|
1019
|
+
request waiting for review, purple is an agent fixing feedback, and green is approved but
|
|
1020
|
+
not merged. T3 has one round of feedback, and each ticket waited for the one before it.
|
|
1021
|
+
- **Agents and people:** how the time tickets were worked on splits. Here agents took 91% of
|
|
1022
|
+
it and reviews took 7%. On a plan where the donut is mostly amber, the bottleneck is review,
|
|
1023
|
+
not code.
|
|
1024
|
+
- **Ticket by ticket, and the phases:** the same times as a table, with the estimate and the
|
|
1025
|
+
review rounds, and how long planning, tickets, development and proof took.
|
|
1026
|
+
|
|
1027
|
+
```
|
|
1028
|
+
$ bin/plan-driven stats PD-3
|
|
1029
|
+
PD-3 Comments on events: statistics
|
|
1030
|
+
Planning 2 min
|
|
1031
|
+
Tickets 2 min
|
|
1032
|
+
Development 1 h 20 min
|
|
1033
|
+
Proof 10 min
|
|
1034
|
+
Idea to delivery 1 h 25 min
|
|
1035
|
+
Tickets merged 6 of 6, 5 approved the first time
|
|
1036
|
+
✓ 42 of 42 acceptance criteria proven
|
|
1037
|
+
|
|
1038
|
+
Where the time went while tickets were worked on (agents 91%):
|
|
1039
|
+
Agent coding 1 h 2 min ████████████████████ 79%
|
|
1040
|
+
Waiting for review 5 min ██ 7%
|
|
1041
|
+
Agent fixing feedback 9 min ███ 12%
|
|
1042
|
+
Approved, not merged 1 min █ 2%
|
|
1043
|
+
|
|
1044
|
+
# Est Queued Agent Review Fixes Merge Rounds Total
|
|
1045
|
+
T1 2 53s 10 min 4 min - 23s 0 16 min
|
|
1046
|
+
T2 2 17 min 10 min 8s - 12s 0 10 min
|
|
1047
|
+
T3 3 28 min 11 min 28s 9 min 10s 1 21 min
|
|
1048
|
+
...
|
|
1049
|
+
```
|
|
1050
|
+
|
|
1051
|
+
### How it's worked out
|
|
1052
|
+
|
|
1053
|
+
Nothing is estimated, and no model is asked. Every number is the time between two events that
|
|
1054
|
+
plan-driven already records in the audit trail:
|
|
1055
|
+
|
|
1056
|
+
| From | To | Counts as |
|
|
1057
|
+
| --- | --- | --- |
|
|
1058
|
+
| `tickets.approved` | `ticket.agent_started` | Queued |
|
|
1059
|
+
| `ticket.agent_started` | `ticket.pr_opened` | Agent coding |
|
|
1060
|
+
| `ticket.pr_opened` | `ticket.pr_approved` or `ticket.changes_requested` | Waiting for review |
|
|
1061
|
+
| `ticket.changes_requested` | the next `ticket.pr_opened` | Agent fixing feedback |
|
|
1062
|
+
| `ticket.pr_approved` | `ticket.merged` | Approved, not merged |
|
|
1063
|
+
|
|
1064
|
+
The phases run from `plan.drafted` to the last `plan.approved` (planning), then to
|
|
1065
|
+
`tickets.approved` (tickets), then to `plan.delivered` (development), then to the first
|
|
1066
|
+
passing evidence run (proof). A plan still in development is counted up to now. A plan whose
|
|
1067
|
+
tickets aren't approved yet shows only its planning time.
|
|
1068
|
+
|
|
1069
|
+
The charts are SVG drawn in Ruby, with no JavaScript and nothing to install. `report` writes
|
|
1070
|
+
them next to `delivery-report.md` (`statistics-burnup.svg`, `statistics-timeline.svg`,
|
|
1071
|
+
`statistics-time.svg`, `statistics-proof.svg`), so GitHub shows them in the Markdown, and it
|
|
1072
|
+
inlines them in the HTML and the PDF so both stand alone.
|
|
1073
|
+
|
|
1074
|
+
The report also marks every acceptance criterion's result in colour: a green, red, amber or
|
|
1075
|
+
grey pill with a matching edge on its row, and failed rows tinted red. On GitHub, the Markdown
|
|
1076
|
+
shows ✅, ❌, ⏸️ or ⚠️ instead.
|
|
1077
|
+
|
|
1078
|
+

|
|
1079
|
+
|
|
1080
|
+
## How it compares
|
|
1081
|
+
|
|
1082
|
+
plan_driven sits next to spec-driven tools such as GitHub's Spec Kit and Kiro, which also start
|
|
1083
|
+
from a written spec before an agent writes code. As we understand them, those are
|
|
1084
|
+
language-agnostic and focus on producing the spec, the design and the task list for an agent to
|
|
1085
|
+
follow. plan_driven is narrower and goes further on the Rails side:
|
|
1086
|
+
|
|
1087
|
+
- the plan is drafted from your Rails schema, models and routes, and Existing Data Structure is
|
|
1088
|
+
checked against them;
|
|
1089
|
+
- the rules are Ruby code that blocks the next step (expand and contract, ticket size and
|
|
1090
|
+
order, pull request scope, CI), rather than guidance in a prompt;
|
|
1091
|
+
- phases and approvals are stored in your database, per revision, with an audit trail;
|
|
1092
|
+
- each acceptance criterion is mapped to a Cucumber scenario, and the delivery report shows
|
|
1093
|
+
the proof, the approvals and the cost.
|
|
1094
|
+
|
|
1095
|
+
If your stack isn't Rails, or you only want a spec for a single agent session, a general tool
|
|
1096
|
+
is the better fit.
|
|
1097
|
+
|
|
698
1098
|
## Keys
|
|
699
1099
|
|
|
700
1100
|
| Key | Used for | Environment |
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module PlanDriven
|
|
4
|
+
module Wizard
|
|
5
|
+
# The wizard runs commands on this machine, so it only answers local requests in development
|
|
6
|
+
# (or wherever config.wizard_enabled says).
|
|
7
|
+
class ApplicationController < ActionController::Base
|
|
8
|
+
protect_from_forgery with: :exception
|
|
9
|
+
layout "plan_driven/wizard/application"
|
|
10
|
+
before_action :local_only!
|
|
11
|
+
helper_method :acting_as, :current_job
|
|
12
|
+
|
|
13
|
+
helper do
|
|
14
|
+
# A button that runs one command for the plan on the page, e.g. run_button("Submit", "submit").
|
|
15
|
+
def run_button(label, action, fields = {}, primary: false, disabled: false)
|
|
16
|
+
button_to label, run_plan_path(@plan.key),
|
|
17
|
+
params: fields.merge(do: action, step: @step), disabled: disabled,
|
|
18
|
+
class: primary ? "primary" : nil, form: { style: "display:inline" }
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def markdown(text)
|
|
22
|
+
PlanDriven::Renderer::HTML.convert(text).html_safe
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def guard_list(report)
|
|
26
|
+
safe_join([
|
|
27
|
+
(tag.ul(safe_join(report.errors.map { |e| tag.li(e) }), class: "errors") if report.errors.any?),
|
|
28
|
+
(tag.ul(safe_join(report.warnings.map { |w| tag.li(w) }), class: "warnings") if report.warnings.any?),
|
|
29
|
+
(tag.p("#{report.passes.size} checks passed.", class: "muted") if report.passes.any?)
|
|
30
|
+
].compact)
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
private
|
|
35
|
+
|
|
36
|
+
def local_only!
|
|
37
|
+
enabled = PlanDriven.configuration.wizard_enabled
|
|
38
|
+
enabled = Rails.env.development? if enabled.nil?
|
|
39
|
+
head :forbidden unless enabled && request.local?
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def acting_as
|
|
43
|
+
session[:plan_driven_actor].presence || PlanDriven.actor
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def current_job
|
|
47
|
+
return @current_job if defined?(@current_job)
|
|
48
|
+
|
|
49
|
+
@current_job = params[:job].present? ? Job.find(params[:job]) : nil
|
|
50
|
+
rescue ArgumentError
|
|
51
|
+
@current_job = nil
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def start(argv, stdin: nil)
|
|
55
|
+
Job.start(argv, stdin: stdin, actor: session[:plan_driven_actor])
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|