@miphamai/cli 0.25.0 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/standard/codebase-design.SKILL.md +189 -0
- package/skills/standard/domain-modeling.SKILL.md +129 -0
- package/skills/standard/grill-with-docs.SKILL.md +199 -0
- package/skills/standard/to-spec.SKILL.md +138 -0
- package/skills/standard/triage.SKILL.md +155 -0
- package/src/core/instructions.ts +40 -0
- package/src/core/permission.ts +65 -10
- package/src/tools/agent/enter-plan.ts +5 -5
- package/src/tools/agent/exit-plan.ts +36 -27
- package/src/tools/exec/bash.ts +113 -0
package/package.json
CHANGED
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: codebase-design
|
|
3
|
+
description: Deep module design principles for designing or improving module interfaces. Use when designing a new module, refactoring an existing one, or finding deepening opportunities in the codebase.
|
|
4
|
+
version: 1.0.0
|
|
5
|
+
user-invocable: true
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- Read
|
|
8
|
+
- Glob
|
|
9
|
+
- Grep
|
|
10
|
+
- Edit
|
|
11
|
+
- Write
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
# Codebase Design — Deep Module Principles
|
|
15
|
+
|
|
16
|
+
Based on John Ousterhout's "A Philosophy of Software Design." The core idea: the greatest single factor in software complexity is the depth of modules — how much functionality they provide relative to the size of their interface.
|
|
17
|
+
|
|
18
|
+
## When to Use
|
|
19
|
+
|
|
20
|
+
- Designing a new module, API, or component
|
|
21
|
+
- Reviewing existing code for design quality
|
|
22
|
+
- Deciding where to split or join modules
|
|
23
|
+
- User asks: "is this well-designed?", "where should this go?", "how should I structure this?"
|
|
24
|
+
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
## Core Concepts
|
|
28
|
+
|
|
29
|
+
### Deep vs Shallow Modules
|
|
30
|
+
|
|
31
|
+
```
|
|
32
|
+
Deep Module (good): Shallow Module (bad):
|
|
33
|
+
┌─────────────────┐ ┌─────────────────┐
|
|
34
|
+
│ small interface │ │ large interface │
|
|
35
|
+
│ ┌─────────────┐ │ │ (many params, │
|
|
36
|
+
│ │ │ │ │ complex setup) │
|
|
37
|
+
│ │ large │ │ ├─────────────────┤
|
|
38
|
+
│ │ implementation│ │ small │
|
|
39
|
+
│ │ │ │ │ implementation │
|
|
40
|
+
│ └─────────────┘ │ │ (just passes │
|
|
41
|
+
└─────────────────┘ │ through) │
|
|
42
|
+
└─────────────────┘
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
**Deep**: Unix file I/O — 5 syscalls (`open`, `read`, `write`, `lseek`, `close`), incredibly powerful implementation.
|
|
46
|
+
|
|
47
|
+
**Shallow**: A function that takes 12 parameters, validates 3 of them, then calls another function. Interface cost > implementation value.
|
|
48
|
+
|
|
49
|
+
### The Rule of Deep Modules
|
|
50
|
+
|
|
51
|
+
> The interface should be as small as possible while providing as much functionality as possible.
|
|
52
|
+
|
|
53
|
+
- **Cost** = interface complexity (parameters, configuration, setup required)
|
|
54
|
+
- **Benefit** = functionality provided (what the caller no longer needs to worry about)
|
|
55
|
+
- **Depth** = Benefit / Cost
|
|
56
|
+
|
|
57
|
+
---
|
|
58
|
+
|
|
59
|
+
## The 5 Design Checks
|
|
60
|
+
|
|
61
|
+
When evaluating a module design, run through these:
|
|
62
|
+
|
|
63
|
+
### Check 1: Interface Size
|
|
64
|
+
|
|
65
|
+
Count the effective parameters:
|
|
66
|
+
|
|
67
|
+
- Required parameters + optional parameters with non-trivial defaults
|
|
68
|
+
- Configuration methods that MUST be called before use
|
|
69
|
+
- Implicit dependencies (global state, env vars, singletons)
|
|
70
|
+
|
|
71
|
+
**Red flag**: > 4 effective parameters → the module may be too shallow.
|
|
72
|
+
|
|
73
|
+
**Fix**: Bundle related parameters into a config object. Or split the module.
|
|
74
|
+
|
|
75
|
+
### Check 2: Information Hiding
|
|
76
|
+
|
|
77
|
+
Does the module expose information that callers don't need?
|
|
78
|
+
|
|
79
|
+
- Internal data structures leaked through the interface
|
|
80
|
+
- Implementation details exposed via parameter types
|
|
81
|
+
- Error types that reveal internal architecture
|
|
82
|
+
|
|
83
|
+
**Red flag**: Callers import types they don't use directly.
|
|
84
|
+
|
|
85
|
+
**Fix**: Define a public API type layer. Return opaque handles instead of raw data.
|
|
86
|
+
|
|
87
|
+
### Check 3: Abstraction Quality
|
|
88
|
+
|
|
89
|
+
Does the module represent a single, coherent idea?
|
|
90
|
+
|
|
91
|
+
- Can you describe what it does in one sentence without "and"?
|
|
92
|
+
- Would a new team member guess where to find this functionality?
|
|
93
|
+
- If you remove the module, does exactly one concept go missing?
|
|
94
|
+
|
|
95
|
+
**Red flag**: Module name contains "and", "Utils", "Common", "Helpers".
|
|
96
|
+
|
|
97
|
+
**Fix**: Split by concept. `UserService` + `EmailService` instead of `UserAndEmailUtils`.
|
|
98
|
+
|
|
99
|
+
### Check 4: General-Purpose vs Special-Purpose
|
|
100
|
+
|
|
101
|
+
Is the module solving the general case or a specific use case?
|
|
102
|
+
|
|
103
|
+
- Would the interface work if requirements changed slightly?
|
|
104
|
+
- Are there hardcoded assumptions that could be parameters?
|
|
105
|
+
- Is the module useful in contexts other than its creator imagined?
|
|
106
|
+
|
|
107
|
+
**Red flag**: Module only works for one specific call site.
|
|
108
|
+
|
|
109
|
+
**Fix**: Make the specific case a thin wrapper around the general case. The general module is deep; the wrapper is shallow (and that's fine — wrappers are allowed to be shallow).
|
|
110
|
+
|
|
111
|
+
### Check 5: Seam Placement
|
|
112
|
+
|
|
113
|
+
Where you split modules matters as much as what they do.
|
|
114
|
+
|
|
115
|
+
- Does the split happen at a natural boundary?
|
|
116
|
+
- Are there circular dependencies across the seam?
|
|
117
|
+
- Can each side be tested independently?
|
|
118
|
+
|
|
119
|
+
**Red flag**: Circular imports, or modules that are always imported together.
|
|
120
|
+
|
|
121
|
+
**Fix**: Use dependency inversion. Define interfaces at the seam, not implementations.
|
|
122
|
+
|
|
123
|
+
---
|
|
124
|
+
|
|
125
|
+
## Finding Deepening Opportunities
|
|
126
|
+
|
|
127
|
+
Scan the codebase for these patterns:
|
|
128
|
+
|
|
129
|
+
### Shallow Pass-Through
|
|
130
|
+
|
|
131
|
+
```typescript
|
|
132
|
+
// Shallow — just delegates with no added value
|
|
133
|
+
function getUser(id: string) {
|
|
134
|
+
return db.findUser(id)
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// Deep — handles errors, caching, authorization in one call
|
|
138
|
+
function getUser(id: string, ctx: RequestContext) {
|
|
139
|
+
const cached = cache.get(`user:${id}`)
|
|
140
|
+
if (cached) return cached
|
|
141
|
+
ctx.auth.assertCanRead('user', id)
|
|
142
|
+
const user = db.findUser(id)
|
|
143
|
+
if (!user) throw new NotFoundError('User', id)
|
|
144
|
+
cache.set(`user:${id}`, user)
|
|
145
|
+
return user
|
|
146
|
+
}
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
### Temporal Decomposition
|
|
150
|
+
|
|
151
|
+
When a module's methods must be called in a specific order, the interface is too wide.
|
|
152
|
+
|
|
153
|
+
```typescript
|
|
154
|
+
// Shallow — caller manages lifecycle
|
|
155
|
+
const conn = new Connection()
|
|
156
|
+
conn.open()
|
|
157
|
+
conn.authenticate(token)
|
|
158
|
+
conn.send(data)
|
|
159
|
+
conn.close()
|
|
160
|
+
|
|
161
|
+
// Deep — module manages lifecycle
|
|
162
|
+
const conn = await Connection.create(token)
|
|
163
|
+
conn.send(data)
|
|
164
|
+
// clean up automatically
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
### Overexposure
|
|
168
|
+
|
|
169
|
+
When internal types leak through the public API:
|
|
170
|
+
|
|
171
|
+
```typescript
|
|
172
|
+
// Shallow — exposes ORM internals
|
|
173
|
+
interface UserService {
|
|
174
|
+
findUser(id: string): Promise<PrismaUser | null> // ❌ PrismaUser is internal
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// Deep — owns its types
|
|
178
|
+
interface UserService {
|
|
179
|
+
findUser(id: string): Promise<User | null> // ✅ User is a domain type
|
|
180
|
+
}
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
---
|
|
184
|
+
|
|
185
|
+
## Integration With Mipham Code
|
|
186
|
+
|
|
187
|
+
- **code-review**: This skill fills the architecture dimension that code-review's 7 dimensions don't cover. Use `/code-review` for correctness/security/perf; use `/codebase-design` for interface depth/abstraction quality/seam placement.
|
|
188
|
+
- **domain-modeling**: Good domain modeling makes deep modules easier — the CONTEXT.md glossary defines the concepts that modules should represent.
|
|
189
|
+
- **Critical Thinking Layer**: The counter-example search applies directly: "what would break if I changed the implementation of this module?"
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: domain-modeling
|
|
3
|
+
description: Build and sharpen a project's domain model. Use when the user wants to pin down domain terminology or a ubiquitous language, record an architectural decision, or when another skill needs to maintain the domain model.
|
|
4
|
+
version: 1.0.0
|
|
5
|
+
user-invocable: true
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- Read
|
|
8
|
+
- Write
|
|
9
|
+
- Edit
|
|
10
|
+
- Glob
|
|
11
|
+
- Grep
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
# Domain Modeling — Continuous Shared Language
|
|
15
|
+
|
|
16
|
+
Actively build and sharpen the project's domain model as you work. This is the _active_ discipline — challenging terms, inventing edge-case scenarios, and writing the glossary and decisions down the moment they crystallize. (Merely _reading_ `CONTEXT.md` for vocabulary is not this skill — that's a one-line habit any skill can do. This skill is for when you're changing the model, not just consuming it.)
|
|
17
|
+
|
|
18
|
+
## File Structure
|
|
19
|
+
|
|
20
|
+
```
|
|
21
|
+
/
|
|
22
|
+
├── CONTEXT.md ← shared language glossary
|
|
23
|
+
├── docs/
|
|
24
|
+
│ └── adr/
|
|
25
|
+
│ ├── 0001-slug.md ← architectural decisions
|
|
26
|
+
│ └── 0002-slug.md
|
|
27
|
+
└── src/
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Create files lazily — only when you have something to write.
|
|
31
|
+
|
|
32
|
+
**Multiple contexts**: If a `CONTEXT-MAP.md` exists, read it to find which context the current topic relates to.
|
|
33
|
+
|
|
34
|
+
---
|
|
35
|
+
|
|
36
|
+
## During the Session
|
|
37
|
+
|
|
38
|
+
### Challenge Against the Glossary
|
|
39
|
+
|
|
40
|
+
When the user uses a term that conflicts with existing language in `CONTEXT.md`, call it out immediately:
|
|
41
|
+
|
|
42
|
+
> "Your glossary defines 'cancellation' as X, but you seem to mean Y — which is it?"
|
|
43
|
+
|
|
44
|
+
### Sharpen Fuzzy Language
|
|
45
|
+
|
|
46
|
+
When the user uses vague or overloaded terms, propose a precise canonical term:
|
|
47
|
+
|
|
48
|
+
> "You're saying 'account' — do you mean the Customer or the User? Those are different things."
|
|
49
|
+
|
|
50
|
+
### Discuss Concrete Scenarios
|
|
51
|
+
|
|
52
|
+
When domain relationships are discussed, stress-test them with specific scenarios. Invent scenarios that probe edge cases and force precision about boundaries between concepts.
|
|
53
|
+
|
|
54
|
+
### Cross-Reference With Code
|
|
55
|
+
|
|
56
|
+
When the user states how something works, check whether the code agrees. Surface contradictions:
|
|
57
|
+
|
|
58
|
+
> "Your code cancels entire Orders, but you just said partial cancellation is possible — which is right?"
|
|
59
|
+
|
|
60
|
+
### Update CONTEXT.md Inline
|
|
61
|
+
|
|
62
|
+
When a term is resolved, update `CONTEXT.md` right there. Don't batch — capture as they happen.
|
|
63
|
+
|
|
64
|
+
### Offer ADRs Sparingly
|
|
65
|
+
|
|
66
|
+
Only create an ADR when ALL three are true:
|
|
67
|
+
|
|
68
|
+
1. **Hard to reverse** — changing your mind later has real cost
|
|
69
|
+
2. **Surprising without context** — a future reader would wonder "why?"
|
|
70
|
+
3. **The result of a real trade-off** — there were genuine alternatives
|
|
71
|
+
|
|
72
|
+
---
|
|
73
|
+
|
|
74
|
+
## CONTEXT.md Format
|
|
75
|
+
|
|
76
|
+
```markdown
|
|
77
|
+
# {Context Name}
|
|
78
|
+
|
|
79
|
+
{One or two sentence description of what this context is and why it exists.}
|
|
80
|
+
|
|
81
|
+
## Language
|
|
82
|
+
|
|
83
|
+
**{Term}**:
|
|
84
|
+
{One or two sentence definition of what it IS.}
|
|
85
|
+
_Avoid_: {alternative terms that should not be used}
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
### Rules
|
|
89
|
+
|
|
90
|
+
- **Be opinionated.** Pick the best term, ban the rest.
|
|
91
|
+
- **Keep definitions tight.** One or two sentences max.
|
|
92
|
+
- **Only domain-specific terms.** Not general programming concepts.
|
|
93
|
+
- **Group under subheadings** when natural clusters emerge.
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## ADR Format
|
|
98
|
+
|
|
99
|
+
```markdown
|
|
100
|
+
# {Short title of the decision}
|
|
101
|
+
|
|
102
|
+
{1-3 sentences: context, decision, and why.}
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Number sequentially (`docs/adr/0001-slug.md`, `0002-slug.md`, ...).
|
|
106
|
+
|
|
107
|
+
Optional sections (only when they add value):
|
|
108
|
+
|
|
109
|
+
- **Status** frontmatter: `proposed | accepted | deprecated | superseded by ADR-NNNN`
|
|
110
|
+
- **Considered Options**: rejected alternatives worth remembering
|
|
111
|
+
- **Consequences**: non-obvious downstream effects
|
|
112
|
+
|
|
113
|
+
### When an ADR Qualifies
|
|
114
|
+
|
|
115
|
+
- Architecture shape (monorepo, event sourcing, microservices)
|
|
116
|
+
- Integration patterns between contexts
|
|
117
|
+
- Technology choices with lock-in (database, message bus, auth)
|
|
118
|
+
- Boundary and scope decisions ("X owns Y, Z references by ID only")
|
|
119
|
+
- Deliberate deviations from convention
|
|
120
|
+
- Constraints not visible in code (compliance, latency SLA)
|
|
121
|
+
- Rejected alternatives when non-obvious (stops someone suggesting it again in 6 months)
|
|
122
|
+
|
|
123
|
+
---
|
|
124
|
+
|
|
125
|
+
## Integration With Mipham Code
|
|
126
|
+
|
|
127
|
+
- **Memory System**: Domain terms discovered through this skill persist to project memory
|
|
128
|
+
- **grill-with-docs**: For initial domain establishment, use `/grill-with-docs`. This skill handles ongoing maintenance
|
|
129
|
+
- **Critical Thinking Layer**: Apply counter-example search to domain definitions — "does this definition hold for all edge cases?"
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: grill-with-docs
|
|
3
|
+
description: A relentless interview to sharpen a plan or design, creating CONTEXT.md (shared language) and ADRs (architectural decisions) as we go. Use before any non-trivial implementation to align on requirements and terminology.
|
|
4
|
+
version: 1.0.0
|
|
5
|
+
user-invocable: true
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- Read
|
|
8
|
+
- Write
|
|
9
|
+
- Edit
|
|
10
|
+
- Bash
|
|
11
|
+
- Glob
|
|
12
|
+
- Grep
|
|
13
|
+
- WebSearch
|
|
14
|
+
- WebFetch
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
# Grill With Docs — Deep Requirements Alignment
|
|
18
|
+
|
|
19
|
+
Inspired by Matt Pocock's `grill-with-docs` and `domain-modeling` skills. Before writing code, run a structured interview to align on requirements, establish shared language, and record architectural decisions.
|
|
20
|
+
|
|
21
|
+
## When to Use
|
|
22
|
+
|
|
23
|
+
- Before any non-trivial feature implementation
|
|
24
|
+
- When requirements are fuzzy ("make it faster", "add X")
|
|
25
|
+
- When you need to establish project terminology
|
|
26
|
+
- When architectural decisions need to be recorded
|
|
27
|
+
- User says: "plan X", "design Y", "what should we do about Z"
|
|
28
|
+
|
|
29
|
+
## When NOT to Use
|
|
30
|
+
|
|
31
|
+
- Trivial bug fixes with clear expected behavior
|
|
32
|
+
- One-line changes
|
|
33
|
+
- Tasks where the requirements are already crystal clear
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## The Interview Flow
|
|
38
|
+
|
|
39
|
+
### Phase 1: Understand the Intent
|
|
40
|
+
|
|
41
|
+
Start by understanding what the user actually wants. Don't ask "what should I build?" — ask about their goal.
|
|
42
|
+
|
|
43
|
+
**Core Questions:**
|
|
44
|
+
|
|
45
|
+
1. What problem are you solving? (Not what feature you're building)
|
|
46
|
+
2. Who is this for? (End user, developer, internal tool?)
|
|
47
|
+
3. What does success look like? (How will you know when it's done?)
|
|
48
|
+
4. What's the deadline or priority context?
|
|
49
|
+
|
|
50
|
+
**Anti-pattern**: Jumping to implementation questions ("Do you want REST or GraphQL?") before understanding the problem.
|
|
51
|
+
|
|
52
|
+
### Phase 2: Sharpen the Language
|
|
53
|
+
|
|
54
|
+
Identify vague or overloaded terms and pin them down **immediately**. This is the single highest-leverage activity — shared language reduces token waste and prevents misunderstandings.
|
|
55
|
+
|
|
56
|
+
**Technique: The Canonical Term**
|
|
57
|
+
|
|
58
|
+
- When the user uses multiple words for the same thing, pick one as canonical
|
|
59
|
+
- List rejected alternatives under `_Avoid_`
|
|
60
|
+
- Be opinionated — the glossary is prescriptive, not descriptive
|
|
61
|
+
|
|
62
|
+
```
|
|
63
|
+
User: "We need a way for users to save articles for later."
|
|
64
|
+
You: "Let's pin that down. 'Save for later' could mean bookmarking, or a reading list, or offline download. Which one?"
|
|
65
|
+
User: "Like a reading list — they can come back to it."
|
|
66
|
+
You: "Got it. Let's call it a **Reading List**. Avoid 'bookmark', 'save', 'favorites'."
|
|
67
|
+
→ Write to CONTEXT.md immediately.
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
**Technique: The Boundary Test**
|
|
71
|
+
|
|
72
|
+
- When a term is proposed, test its boundaries with edge cases
|
|
73
|
+
- "Does X include Y? What about Z?"
|
|
74
|
+
|
|
75
|
+
**Technique: The Code Cross-Reference**
|
|
76
|
+
|
|
77
|
+
- When the user describes how something works, check if existing code agrees
|
|
78
|
+
- Surface contradictions immediately
|
|
79
|
+
|
|
80
|
+
### Phase 3: Probe Edge Cases
|
|
81
|
+
|
|
82
|
+
Before accepting any requirement, stress-test it with edge cases.
|
|
83
|
+
|
|
84
|
+
**Edge Case Inventory:**
|
|
85
|
+
|
|
86
|
+
- **Empty state**: What does the user see when there's nothing yet?
|
|
87
|
+
- **Error state**: What happens when things go wrong?
|
|
88
|
+
- **Extreme values**: What about 0? What about 10,000?
|
|
89
|
+
- **Concurrency**: What if two people do this at the same time?
|
|
90
|
+
- **Permissions**: Who can do this? Who cannot?
|
|
91
|
+
- **Scale**: What changes at 10x the current volume?
|
|
92
|
+
|
|
93
|
+
**Technique: The 5 Whys**
|
|
94
|
+
When a requirement seems odd, dig deeper:
|
|
95
|
+
|
|
96
|
+
```
|
|
97
|
+
User: "We need real-time updates."
|
|
98
|
+
You: "Why real-time?"
|
|
99
|
+
User: "Because users need to see changes immediately."
|
|
100
|
+
You: "Why do they need to see changes immediately?"
|
|
101
|
+
User: "Because they're collaborating on the same document."
|
|
102
|
+
→ Now you know the REAL requirement is collaboration, not real-time.
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
### Phase 4: Make Architecture Decisions
|
|
106
|
+
|
|
107
|
+
When a design decision meets ALL three criteria, offer to record it as an ADR:
|
|
108
|
+
|
|
109
|
+
1. **Hard to reverse** — changing your mind later has real cost
|
|
110
|
+
2. **Surprising without context** — a future reader would wonder "why?"
|
|
111
|
+
3. **The result of a real trade-off** — there were genuine alternatives
|
|
112
|
+
|
|
113
|
+
**What qualifies for an ADR:**
|
|
114
|
+
|
|
115
|
+
- Architecture shape (monorepo vs polyrepo, event sourcing vs CRUD)
|
|
116
|
+
- Integration patterns between contexts
|
|
117
|
+
- Technology choices with lock-in (database, message bus, auth provider)
|
|
118
|
+
- Deliberate deviations from convention ("we use raw SQL because...")
|
|
119
|
+
- Constraints not visible in code ("we can't use X because compliance")
|
|
120
|
+
|
|
121
|
+
**ADR Format** (write to `docs/adr/NNNN-slug.md`):
|
|
122
|
+
|
|
123
|
+
```markdown
|
|
124
|
+
# {Short title of the decision}
|
|
125
|
+
|
|
126
|
+
{1-3 sentences: context, decision, and why.}
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Only add optional sections (Status, Considered Options, Consequences) when they add genuine value. Most ADRs are a single paragraph.
|
|
130
|
+
|
|
131
|
+
### Phase 5: Write the CONTEXT.md
|
|
132
|
+
|
|
133
|
+
After the interview, synthesize everything into `CONTEXT.md`.
|
|
134
|
+
|
|
135
|
+
**Format** (`CONTEXT.md` at project root):
|
|
136
|
+
|
|
137
|
+
```markdown
|
|
138
|
+
# {Project Name} Context
|
|
139
|
+
|
|
140
|
+
{One or two sentence description of the project domain.}
|
|
141
|
+
|
|
142
|
+
## Language
|
|
143
|
+
|
|
144
|
+
**{Term}**:
|
|
145
|
+
{One or two sentence definition of what it IS.}
|
|
146
|
+
_Avoid_: {alternative terms that should not be used}
|
|
147
|
+
|
|
148
|
+
## Decisions
|
|
149
|
+
|
|
150
|
+
- [ADR 0001: {Title}](docs/adr/0001-slug.md) — {one-line summary}
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
**Rules:**
|
|
154
|
+
|
|
155
|
+
- Be opinionated — pick the best term, ban the rest
|
|
156
|
+
- Only include domain-specific terms (not general programming concepts)
|
|
157
|
+
- Keep definitions tight — one or two sentences
|
|
158
|
+
- Update inline during the conversation, don't batch
|
|
159
|
+
- CONTEXT.md is a glossary, NOT a spec or implementation plan
|
|
160
|
+
|
|
161
|
+
---
|
|
162
|
+
|
|
163
|
+
## During the Conversation
|
|
164
|
+
|
|
165
|
+
### DO
|
|
166
|
+
|
|
167
|
+
- Challenge the user when they use vague terms — "What do you mean by 'fast'?"
|
|
168
|
+
- Propose canonical terms and write them down immediately
|
|
169
|
+
- Invent edge cases and probe boundaries
|
|
170
|
+
- Offer ADRs sparingly (only when all 3 criteria are met)
|
|
171
|
+
- Cross-reference with existing code if available
|
|
172
|
+
- Call out contradictions between what the user says and what the code does
|
|
173
|
+
|
|
174
|
+
### DON'T
|
|
175
|
+
|
|
176
|
+
- Rush to implementation questions before understanding the problem
|
|
177
|
+
- Write ADRs for trivial decisions
|
|
178
|
+
- Let fuzzy language slide — pin it down now or pay later
|
|
179
|
+
- Treat CONTEXT.md as a spec or scratch pad
|
|
180
|
+
- Ask yes/no questions when open-ended ones would reveal more
|
|
181
|
+
|
|
182
|
+
---
|
|
183
|
+
|
|
184
|
+
## Output
|
|
185
|
+
|
|
186
|
+
After the interview, the user should have:
|
|
187
|
+
|
|
188
|
+
1. **CONTEXT.md** — shared language glossary (created or updated)
|
|
189
|
+
2. **ADRs** (if needed) — architectural decisions in `docs/adr/`
|
|
190
|
+
3. **Clear requirements** — edge cases explored, assumptions surfaced
|
|
191
|
+
4. **Shared understanding** — you and the user now mean the same thing by the same words
|
|
192
|
+
|
|
193
|
+
---
|
|
194
|
+
|
|
195
|
+
## Integration with Mipham Code
|
|
196
|
+
|
|
197
|
+
- **Memory System**: Key terms go to project memory for persistence across sessions
|
|
198
|
+
- **Critical Thinking Layer**: Apply the 5-dimension self-check (evidence standard, equivalence verification, counter-example search, confidence calibration, depth check) to your own interview questions
|
|
199
|
+
- **Workflow**: For complex projects, the output of this skill feeds directly into `/implement`
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: to-spec
|
|
3
|
+
description: Turn a conversation into a structured specification document. Use after a grill-with-docs session or any requirements discussion to capture decisions in a durable, shareable format.
|
|
4
|
+
version: 1.0.0
|
|
5
|
+
user-invocable: true
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- Read
|
|
8
|
+
- Write
|
|
9
|
+
- Edit
|
|
10
|
+
- Bash
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
# To Spec — Conversation → Specification
|
|
14
|
+
|
|
15
|
+
Turn the output of a requirements discussion into a structured specification document. This is the bridge between `/grill-with-docs` (alignment) and `/triage` (task decomposition).
|
|
16
|
+
|
|
17
|
+
## When to Use
|
|
18
|
+
|
|
19
|
+
- After a `/grill-with-docs` session — capture what was decided
|
|
20
|
+
- After any requirements discussion — before starting implementation
|
|
21
|
+
- User asks: "write this up", "create a spec", "document the plan"
|
|
22
|
+
- Before handing off work to another session or person
|
|
23
|
+
|
|
24
|
+
## When NOT to Use
|
|
25
|
+
|
|
26
|
+
- The requirements are a single sentence and obvious
|
|
27
|
+
- You're in the middle of a grill session — finish the interview first
|
|
28
|
+
- The scope is so small that the spec would be longer than the implementation
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Spec Format
|
|
33
|
+
|
|
34
|
+
Write to `docs/specs/YYYY-MM-DD-slug.md`:
|
|
35
|
+
|
|
36
|
+
```markdown
|
|
37
|
+
---
|
|
38
|
+
status: draft | approved | implemented
|
|
39
|
+
created: 2026-08-10
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
# {Title}
|
|
43
|
+
|
|
44
|
+
## Problem
|
|
45
|
+
|
|
46
|
+
{What problem are we solving? Why now? 1-3 sentences.}
|
|
47
|
+
|
|
48
|
+
## Scope
|
|
49
|
+
|
|
50
|
+
### In Scope
|
|
51
|
+
|
|
52
|
+
- {What we're building}
|
|
53
|
+
|
|
54
|
+
### Out of Scope (Explicit)
|
|
55
|
+
|
|
56
|
+
- {What we're NOT building — prevents scope creep}
|
|
57
|
+
|
|
58
|
+
## Requirements
|
|
59
|
+
|
|
60
|
+
### Functional
|
|
61
|
+
|
|
62
|
+
- **{Requirement}**: {Description}. Acceptance: {measurable criterion}.
|
|
63
|
+
|
|
64
|
+
### Non-Functional
|
|
65
|
+
|
|
66
|
+
- **Performance**: {latency, throughput targets}
|
|
67
|
+
- **Security**: {auth, data protection, threat model}
|
|
68
|
+
- **Scale**: {expected volume, growth projections}
|
|
69
|
+
|
|
70
|
+
## Design Decisions
|
|
71
|
+
|
|
72
|
+
- **Decision**: {What we decided}. Because: {why}. Alternatives considered: {options + reasons rejected}.
|
|
73
|
+
|
|
74
|
+
## Domain Model
|
|
75
|
+
|
|
76
|
+
{Key terms and their definitions — from CONTEXT.md or the grill session.}
|
|
77
|
+
|
|
78
|
+
## Edge Cases
|
|
79
|
+
|
|
80
|
+
- **{Scenario}**: {Expected behavior}
|
|
81
|
+
- **{Scenario}**: {Expected behavior}
|
|
82
|
+
|
|
83
|
+
## Open Questions
|
|
84
|
+
|
|
85
|
+
- {Question} — {who needs to answer / when needed}
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
---
|
|
89
|
+
|
|
90
|
+
## The Spec Workflow
|
|
91
|
+
|
|
92
|
+
### Step 1: Extract from Conversation
|
|
93
|
+
|
|
94
|
+
Scan the conversation history for:
|
|
95
|
+
|
|
96
|
+
- Decisions made (explicit and implicit)
|
|
97
|
+
- Terms defined (candidates for CONTEXT.md)
|
|
98
|
+
- Edge cases discussed
|
|
99
|
+
- Alternatives rejected (and why)
|
|
100
|
+
- Open questions that remain
|
|
101
|
+
|
|
102
|
+
### Step 2: Fill Gaps
|
|
103
|
+
|
|
104
|
+
For each gap you find:
|
|
105
|
+
|
|
106
|
+
- Edge cases not discussed → flag as Open Questions
|
|
107
|
+
- Terms used but not defined → propose definitions
|
|
108
|
+
- Assumptions not stated → make them explicit
|
|
109
|
+
|
|
110
|
+
### Step 3: Validate with User
|
|
111
|
+
|
|
112
|
+
Present the spec and ask:
|
|
113
|
+
|
|
114
|
+
1. "Does this match your understanding?"
|
|
115
|
+
2. "What's missing?"
|
|
116
|
+
3. "What's wrong?"
|
|
117
|
+
4. "What surprised you?"
|
|
118
|
+
|
|
119
|
+
### Step 4: Feed Into Triage
|
|
120
|
+
|
|
121
|
+
Once approved, the spec's functional requirements become tickets in `/triage`. Non-functional requirements become acceptance criteria.
|
|
122
|
+
|
|
123
|
+
---
|
|
124
|
+
|
|
125
|
+
## Anti-Patterns
|
|
126
|
+
|
|
127
|
+
- **Waterfall trap**: Don't try to spec everything upfront. Spec the next increment. Specs are living documents, not contracts.
|
|
128
|
+
- **Premature detail**: Don't spec API signatures or DB schemas in the spec — those are implementation details.
|
|
129
|
+
- **Vague acceptance**: "Works well" is not acceptance criteria. "Returns 200 with valid JWT within 500ms" is.
|
|
130
|
+
|
|
131
|
+
---
|
|
132
|
+
|
|
133
|
+
## Integration With Mipham Code
|
|
134
|
+
|
|
135
|
+
- **grill-with-docs**: Input — the grill session produces the raw material
|
|
136
|
+
- **triage**: Output — the spec feeds into ticket decomposition
|
|
137
|
+
- **domain-modeling**: Terms discovered during spec writing go to CONTEXT.md
|
|
138
|
+
- **Memory System**: The spec file persists as project reference across sessions
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: triage
|
|
3
|
+
description: Structured task decomposition and tracking across sessions. Use for breaking complex plans into trackable tickets with dependency graphs, checking task status, or continuing work from a previous session.
|
|
4
|
+
version: 1.0.0
|
|
5
|
+
user-invocable: true
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- Read
|
|
8
|
+
- Write
|
|
9
|
+
- Edit
|
|
10
|
+
- Bash
|
|
11
|
+
- Glob
|
|
12
|
+
- Grep
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
# Triage — Cross-Session Task Tracking
|
|
16
|
+
|
|
17
|
+
Turn plans into trackable tickets with dependency management. Inspired by Matt Pocock's `triage` + `to-tickets` + `wayfinder` skills, consolidated into one Mipham Code skill.
|
|
18
|
+
|
|
19
|
+
## When to Use
|
|
20
|
+
|
|
21
|
+
- Breaking a large plan into actionable tickets
|
|
22
|
+
- Tracking work across multiple sessions
|
|
23
|
+
- User asks: "what's next?", "where did I leave off?", "what's the status?"
|
|
24
|
+
- Complex tasks with dependencies between them
|
|
25
|
+
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
## The Ticket Format
|
|
29
|
+
|
|
30
|
+
Tickets live in `.mipham/tickets/` as individual Markdown files:
|
|
31
|
+
|
|
32
|
+
```markdown
|
|
33
|
+
---
|
|
34
|
+
id: T-001
|
|
35
|
+
title: Add user authentication
|
|
36
|
+
status: in-progress
|
|
37
|
+
priority: P0
|
|
38
|
+
depends_on: []
|
|
39
|
+
blocks: [T-003]
|
|
40
|
+
created: 2026-08-10
|
|
41
|
+
tags:
|
|
42
|
+
- auth
|
|
43
|
+
- backend
|
|
44
|
+
---
|
|
45
|
+
|
|
46
|
+
## Description
|
|
47
|
+
|
|
48
|
+
Add JWT-based authentication with refresh token rotation.
|
|
49
|
+
|
|
50
|
+
## Acceptance Criteria
|
|
51
|
+
|
|
52
|
+
- [ ] Login endpoint returns access + refresh tokens
|
|
53
|
+
- [ ] Refresh endpoint rotates tokens
|
|
54
|
+
- [ ] Invalid tokens return 401
|
|
55
|
+
- [ ] Rate limiting on login attempts
|
|
56
|
+
|
|
57
|
+
## Notes
|
|
58
|
+
|
|
59
|
+
- OAuth not in scope for T-001 (punted to T-005)
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
### Status Values
|
|
63
|
+
|
|
64
|
+
| Status | Meaning |
|
|
65
|
+
| ------------- | ------------------------------------------ |
|
|
66
|
+
| `backlog` | Not yet planned for any session |
|
|
67
|
+
| `planned` | Scoped and ready to work |
|
|
68
|
+
| `in-progress` | Currently being worked on |
|
|
69
|
+
| `review` | Implementation done, awaiting verification |
|
|
70
|
+
| `done` | Verified and merged |
|
|
71
|
+
| `blocked` | Cannot proceed due to dependency |
|
|
72
|
+
| `wontfix` | Decided not to do |
|
|
73
|
+
|
|
74
|
+
---
|
|
75
|
+
|
|
76
|
+
## The Triage Workflow
|
|
77
|
+
|
|
78
|
+
### Phase 1: Decompose (Plan → Tickets)
|
|
79
|
+
|
|
80
|
+
Given a plan or feature request:
|
|
81
|
+
|
|
82
|
+
1. **Identify the smallest independently-valuable units of work**
|
|
83
|
+
- Each ticket should deliver value on its own
|
|
84
|
+
- If a ticket requires 3+ files touched, it's probably too big
|
|
85
|
+
- If a ticket can be done in < 15 minutes, it's probably too small
|
|
86
|
+
|
|
87
|
+
2. **Map dependencies**
|
|
88
|
+
- What must be done first? (hard dependency)
|
|
89
|
+
- What would be easier after something else? (soft dependency)
|
|
90
|
+
- What blocks other work? (reverse dependency)
|
|
91
|
+
|
|
92
|
+
3. **Assign priorities**
|
|
93
|
+
- **P0**: Blocks other work, must do first
|
|
94
|
+
- **P1**: High value, should do soon
|
|
95
|
+
- **P2**: Nice to have, can defer
|
|
96
|
+
- **P3**: Optional, do if time permits
|
|
97
|
+
|
|
98
|
+
4. **Write acceptance criteria**
|
|
99
|
+
- Specific, testable, unambiguous
|
|
100
|
+
- "Login works" is bad. "POST /auth/login with valid credentials returns 200 + JWT" is good.
|
|
101
|
+
|
|
102
|
+
### Phase 2: Status Check
|
|
103
|
+
|
|
104
|
+
When the user asks "what's next?" or "what's the status?":
|
|
105
|
+
|
|
106
|
+
1. Read `.mipham/tickets/` directory
|
|
107
|
+
2. Report:
|
|
108
|
+
- Currently in-progress tickets
|
|
109
|
+
- Blocked tickets (and what's blocking them)
|
|
110
|
+
- Next unblocked P0/P1 tickets ready to work
|
|
111
|
+
- Recently completed tickets (for context)
|
|
112
|
+
|
|
113
|
+
### Phase 3: Session Handoff
|
|
114
|
+
|
|
115
|
+
When starting a new session, check for continuity:
|
|
116
|
+
|
|
117
|
+
1. Read the previous session's context from the session store
|
|
118
|
+
2. Check ticket statuses — any that were `in-progress` last session?
|
|
119
|
+
3. Present: "Last session you were working on T-004 (Add rate limiting). Continue from there, or start on T-007 (API docs) which is next in the P1 queue?"
|
|
120
|
+
|
|
121
|
+
### Phase 4: Ticket Lifecycle
|
|
122
|
+
|
|
123
|
+
When working on a ticket:
|
|
124
|
+
|
|
125
|
+
- Mark it `in-progress` when you start
|
|
126
|
+
- Mark it `review` when implementation is done
|
|
127
|
+
- Mark it `done` after verification (tests pass, typecheck clean)
|
|
128
|
+
- If you discover new dependencies, add them to `blocks`/`depends_on`
|
|
129
|
+
|
|
130
|
+
---
|
|
131
|
+
|
|
132
|
+
## Dependency Graph
|
|
133
|
+
|
|
134
|
+
For tickets with complex dependencies, generate a visual summary:
|
|
135
|
+
|
|
136
|
+
```
|
|
137
|
+
T-001 (Auth) ──blocks──→ T-003 (Dashboard)
|
|
138
|
+
│ │
|
|
139
|
+
└──blocks──→ T-002 (API) ─┘
|
|
140
|
+
│
|
|
141
|
+
└──soft-dep──→ T-004 (Rate Limiting)
|
|
142
|
+
|
|
143
|
+
Ready to work: T-001 (no dependencies)
|
|
144
|
+
Blocked: T-002 (waiting on T-001), T-003 (waiting on T-001, T-002)
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
---
|
|
148
|
+
|
|
149
|
+
## Integration With Mipham Code
|
|
150
|
+
|
|
151
|
+
- **Session Store**: Ticket status persists across sessions via `.mipham/tickets/`
|
|
152
|
+
- **Memory System**: Active tickets are loaded as project memory for context
|
|
153
|
+
- **grill-with-docs**: The output of a grill session feeds directly into ticket decomposition
|
|
154
|
+
- **Background Agents**: Long-running work on a ticket can be spawned as a background agent
|
|
155
|
+
- **Critical Thinking Layer**: When decomposing, ask "what's the smallest thing that delivers value?" — don't over-decompose
|
package/src/core/instructions.ts
CHANGED
|
@@ -71,6 +71,46 @@ export class InstructionsLoader {
|
|
|
71
71
|
parts.push(this.skillsReminder)
|
|
72
72
|
}
|
|
73
73
|
|
|
74
|
+
// Inject critical thinking self-check layer (for analysis/comparison tasks)
|
|
75
|
+
parts.push(`## Critical Thinking Self-Check
|
|
76
|
+
|
|
77
|
+
Before delivering any analysis, comparison, evaluation, or "X vs Y"
|
|
78
|
+
report, run this checklist internally:
|
|
79
|
+
|
|
80
|
+
### 1. Evidence Standard
|
|
81
|
+
- Every factual claim MUST cite a specific source (file path, URL, line number)
|
|
82
|
+
- If you cannot cite a source, label the claim as [推断] (inference) or [待验证] (unverified)
|
|
83
|
+
- Numbers (counts, percentages, download stats) require cross-validation from a second source
|
|
84
|
+
|
|
85
|
+
### 2. Equivalence Verification
|
|
86
|
+
- When you claim "A is equivalent to B" or "X has been merged from Y",
|
|
87
|
+
compare their ACTUAL implementation, not just their names or descriptions
|
|
88
|
+
- If you haven't read both implementations, say "appears similar at the
|
|
89
|
+
description level; implementation equivalence not verified"
|
|
90
|
+
|
|
91
|
+
### 3. Counter-Example Search
|
|
92
|
+
- For each major conclusion, find at least 1 counter-example or edge case
|
|
93
|
+
- If you cannot find one, state that explicitly: "No counter-example found
|
|
94
|
+
within the examined scope"
|
|
95
|
+
- When comparing two systems, ask: "What does X do that Y CANNOT do?"
|
|
96
|
+
(and vice versa) — don't just list overlaps
|
|
97
|
+
|
|
98
|
+
### 4. Confidence Calibration
|
|
99
|
+
- Label each conclusion with confidence: [高] [中] [低]
|
|
100
|
+
- [高] = verified from source code or primary documentation
|
|
101
|
+
- [中] = inferred from description but not implementation-verified
|
|
102
|
+
- [低] = speculative, based on naming convention or surface similarity
|
|
103
|
+
|
|
104
|
+
### 5. Depth Check
|
|
105
|
+
- If your analysis is based ONLY on file names and description fields,
|
|
106
|
+
you are doing surface analysis — state this limitation upfront
|
|
107
|
+
- To reach depth: read at least one implementation file per comparison target
|
|
108
|
+
- Ask: "What would a domain expert notice that I'm missing?"
|
|
109
|
+
|
|
110
|
+
These checks are not optional for analysis tasks. Apply them before
|
|
111
|
+
presenting conclusions, and surface any [低] confidence findings
|
|
112
|
+
explicitly rather than burying them.`)
|
|
113
|
+
|
|
74
114
|
// Inject workflow auto-generation guidance
|
|
75
115
|
parts.push(`## Workflow Auto-Generation
|
|
76
116
|
|
package/src/core/permission.ts
CHANGED
|
@@ -9,6 +9,53 @@ import type { PermissionRuleEntry } from '../shared/index.ts'
|
|
|
9
9
|
import { matchBashRule, compileRule } from './permission-rules'
|
|
10
10
|
import { loadPermissionConfig, nextMode, clampMode, MODE_CYCLE } from './permission-config'
|
|
11
11
|
|
|
12
|
+
/**
|
|
13
|
+
* Check if a Bash command is a "verification-only" command that should be
|
|
14
|
+
* auto-approved in acceptEdits mode. These are non-destructive read/check
|
|
15
|
+
* operations that form the core of the vibe coding edit→test→fix loop.
|
|
16
|
+
*/
|
|
17
|
+
function isVerificationCommand(input: Record<string, unknown>): boolean {
|
|
18
|
+
const cmd = (input.command as string) || ''
|
|
19
|
+
// Patterns for verification-only commands (no side effects on codebase)
|
|
20
|
+
const verifyPatterns = [
|
|
21
|
+
/\bpnpm\s+test\b/, // test runner
|
|
22
|
+
/\bpnpm\s+t\b/, // shorthand test
|
|
23
|
+
/\bpnpm\s+typecheck\b/, // type checking
|
|
24
|
+
/\bpnpm\s+lint\b/, // linting
|
|
25
|
+
/\bpnpm\s+format:check\b/, // format check
|
|
26
|
+
/\bnpm\s+test\b/, // npm test
|
|
27
|
+
/\bnpm\s+run\s+test\b/, // npm run test
|
|
28
|
+
/\bvitest\b/, // vitest runner
|
|
29
|
+
/\bjest\b/, // jest runner
|
|
30
|
+
/\btsc\s+(?!init)/, // TypeScript compiler (not tsc init)
|
|
31
|
+
/\btsc\s+--noEmit\b/, // type check only
|
|
32
|
+
/\beslint\b/, // eslint
|
|
33
|
+
/\bprettier\s+--check\b/, // prettier check
|
|
34
|
+
/\bpytest\b/, // python test runner
|
|
35
|
+
/\bruff\s+check\b/, // python linter
|
|
36
|
+
/\bcargo\s+test\b/, // rust test
|
|
37
|
+
/\bcargo\s+check\b/, // rust check
|
|
38
|
+
/\bgo\s+test\b/, // go test
|
|
39
|
+
/\bgo\s+vet\b/, // go vet
|
|
40
|
+
/\bmake\s+test\b/, // make test
|
|
41
|
+
/\bgit\s+status\b/, // git status (read-only)
|
|
42
|
+
/\bgit\s+diff\b/, // git diff (read-only)
|
|
43
|
+
/\bgit\s+log\b/, // git log (read-only)
|
|
44
|
+
/\bgit\s+branch\b/, // git branch (read-only)
|
|
45
|
+
/\bls\b/, // list files
|
|
46
|
+
/\bcat\b/, // read file
|
|
47
|
+
/\bhead\b/, // read file start
|
|
48
|
+
/\btail\b/, // read file end
|
|
49
|
+
/\bwhich\b/, // find binary
|
|
50
|
+
/\becho\b/, // print text
|
|
51
|
+
/\bnode\s+-v\b/, // node version
|
|
52
|
+
/\bpython\s+--version\b/, // python version
|
|
53
|
+
/\bwhoami\b/, // current user
|
|
54
|
+
/\bpwd\b/, // current directory
|
|
55
|
+
]
|
|
56
|
+
return verifyPatterns.some((p) => p.test(cmd))
|
|
57
|
+
}
|
|
58
|
+
|
|
12
59
|
const VALID_MODES: Set<string> = new Set<string>(MODE_CYCLE)
|
|
13
60
|
|
|
14
61
|
export class PermissionSystem {
|
|
@@ -276,7 +323,7 @@ export class PermissionSystem {
|
|
|
276
323
|
}
|
|
277
324
|
|
|
278
325
|
// 5. Mode baseline
|
|
279
|
-
const baseline = this.modeBaseline(tool)
|
|
326
|
+
const baseline = this.modeBaseline(tool, input)
|
|
280
327
|
if (baseline !== 'mode-baseline') {
|
|
281
328
|
this.checkCache.set(cacheKey, baseline)
|
|
282
329
|
return baseline
|
|
@@ -323,21 +370,29 @@ export class PermissionSystem {
|
|
|
323
370
|
return rule.pattern === tool.name || rule.compiled.test(tool.name)
|
|
324
371
|
}
|
|
325
372
|
|
|
326
|
-
private modeBaseline(
|
|
373
|
+
private modeBaseline(
|
|
374
|
+
tool: ToolDefinition,
|
|
375
|
+
input?: Record<string, unknown>,
|
|
376
|
+
): PermissionLevel | 'mode-baseline' {
|
|
327
377
|
switch (this.mode) {
|
|
328
378
|
case 'default':
|
|
329
379
|
// Delegate to tool.permission (backward compat)
|
|
330
380
|
return 'mode-baseline'
|
|
331
381
|
|
|
332
382
|
case 'acceptEdits':
|
|
333
|
-
// Reads + file edits free; Bash
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
:
|
|
339
|
-
|
|
340
|
-
|
|
383
|
+
// Reads + file edits free; Bash auto-approved for verification commands
|
|
384
|
+
if (tool.category === 'file' && tool.name !== 'Bash') {
|
|
385
|
+
return 'bypass'
|
|
386
|
+
}
|
|
387
|
+
if (tool.name === 'Bash') {
|
|
388
|
+
// Vibe coding fix: auto-approve verification commands
|
|
389
|
+
// so the edit→test→fix loop isn't interrupted by permission prompts
|
|
390
|
+
if (input && isVerificationCommand(input)) {
|
|
391
|
+
return 'bypass'
|
|
392
|
+
}
|
|
393
|
+
return 'ask'
|
|
394
|
+
}
|
|
395
|
+
return 'ask'
|
|
341
396
|
|
|
342
397
|
case 'plan':
|
|
343
398
|
// Only reads, no writes or executes
|
|
@@ -80,12 +80,12 @@ export const enterPlanModeTool: ToolDefinition = {
|
|
|
80
80
|
' ⚠️ All other tools — require confirmation',
|
|
81
81
|
'',
|
|
82
82
|
'Design your approach, explore the codebase, then:',
|
|
83
|
-
' • Use ExitPlanMode to submit your plan for
|
|
84
|
-
' •
|
|
83
|
+
' • Use ExitPlanMode to submit your plan for user review',
|
|
84
|
+
' • Present your plan and ask for explicit approval',
|
|
85
|
+
' • The user must say "approved" before you switch to acceptEdits mode',
|
|
85
86
|
'',
|
|
86
|
-
'
|
|
87
|
-
'
|
|
88
|
-
' approved: false → revert to default mode',
|
|
87
|
+
'⚠️ You CANNOT self-approve your plan. The user must explicitly confirm.',
|
|
88
|
+
' After ExitPlanMode, wait for user approval before making code changes.',
|
|
89
89
|
].join('\n'),
|
|
90
90
|
}
|
|
91
91
|
},
|
|
@@ -1,53 +1,62 @@
|
|
|
1
|
+
import { readFileSync } from 'node:fs'
|
|
2
|
+
import { join } from 'node:path'
|
|
1
3
|
import type { ToolDefinition } from '../../shared/index.ts'
|
|
2
4
|
|
|
3
5
|
export const exitPlanModeTool: ToolDefinition = {
|
|
4
6
|
name: 'ExitPlanMode',
|
|
5
7
|
description:
|
|
6
|
-
'Exit plan mode and
|
|
7
|
-
'
|
|
8
|
-
'
|
|
9
|
-
'
|
|
8
|
+
'Exit plan mode and present your plan for user approval. ' +
|
|
9
|
+
'This tool does NOT switch to implementation mode — the user must explicitly approve first. ' +
|
|
10
|
+
'After calling this, present your plan and ask the user to confirm. ' +
|
|
11
|
+
'The user can approve by saying "approved" or "/approve", or by cycling to acceptEdits mode with Shift+Tab.',
|
|
10
12
|
category: 'agent',
|
|
11
13
|
permission: 'auto',
|
|
12
14
|
parameters: {
|
|
13
15
|
type: 'object',
|
|
14
16
|
properties: {
|
|
15
|
-
|
|
16
|
-
type: '
|
|
17
|
+
planFile: {
|
|
18
|
+
type: 'string',
|
|
17
19
|
description:
|
|
18
|
-
'
|
|
20
|
+
'Path to the plan file you wrote (e.g., .mipham/plans/plan-2026-08-10T12-00-00.md). If omitted, the most recent plan file is used.',
|
|
19
21
|
},
|
|
20
22
|
},
|
|
21
|
-
required: [
|
|
23
|
+
required: [],
|
|
22
24
|
},
|
|
23
|
-
async execute(params,
|
|
24
|
-
const
|
|
25
|
+
async execute(params, ctx) {
|
|
26
|
+
const planDir = join(ctx.cwd, '.mipham', 'plans')
|
|
27
|
+
const planFile = (params.planFile as string) || ''
|
|
25
28
|
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
'✓ Exiting plan mode.',
|
|
33
|
-
'✓ Switching to acceptEdits mode — reads and file edits are auto-approved.',
|
|
34
|
-
'',
|
|
35
|
-
'You can now implement the plan. The plan file is in .mipham/plans/.',
|
|
36
|
-
'',
|
|
37
|
-
'Use Shift+Tab to cycle permission modes if you need to change.',
|
|
38
|
-
].join('\n'),
|
|
29
|
+
// Try to read the plan to confirm it exists
|
|
30
|
+
let planContent = ''
|
|
31
|
+
try {
|
|
32
|
+
const planPath = planFile || ''
|
|
33
|
+
if (planPath) {
|
|
34
|
+
planContent = readFileSync(planPath, 'utf-8')
|
|
39
35
|
}
|
|
36
|
+
} catch {
|
|
37
|
+
// Plan file not found — still exit plan mode
|
|
40
38
|
}
|
|
41
39
|
|
|
42
40
|
return {
|
|
43
41
|
success: true,
|
|
44
42
|
content: [
|
|
45
|
-
'── Plan
|
|
43
|
+
'── Plan Ready for Review ──',
|
|
46
44
|
'',
|
|
47
|
-
'✓
|
|
45
|
+
'✓ Exiting plan mode.',
|
|
46
|
+
'✓ Plan file saved. Present your plan to the user now.',
|
|
48
47
|
'',
|
|
49
|
-
'
|
|
50
|
-
'
|
|
48
|
+
'⚠️ IMPORTANT: You are still in limited permission mode.',
|
|
49
|
+
' The user must explicitly approve before you can make changes.',
|
|
50
|
+
'',
|
|
51
|
+
'Next steps:',
|
|
52
|
+
' 1. Present your plan to the user (summarize key decisions)',
|
|
53
|
+
' 2. Ask: "Does this plan look good? Reply approved to begin."',
|
|
54
|
+
' 3. Wait for the user to explicitly say "approved" or "/approve"',
|
|
55
|
+
' 4. Only then switch to acceptEdits mode (Shift+Tab or user action)',
|
|
56
|
+
'',
|
|
57
|
+
'DO NOT start implementing until the user explicitly approves.',
|
|
58
|
+
'DO NOT call ExitPlanMode with approved:true — that parameter no longer exists.',
|
|
59
|
+
planContent ? `\n── Plan Content (for reference) ──\n\n${planContent.slice(0, 3000)}` : '',
|
|
51
60
|
].join('\n'),
|
|
52
61
|
}
|
|
53
62
|
},
|
package/src/tools/exec/bash.ts
CHANGED
|
@@ -185,6 +185,105 @@ export function detectViolations(stderr: string): string[] {
|
|
|
185
185
|
return violations
|
|
186
186
|
}
|
|
187
187
|
|
|
188
|
+
/**
|
|
189
|
+
* Vibe coding: Parse stderr output for error locations (file path + line + column).
|
|
190
|
+
* Supports common tool formats: TypeScript, ESLint, pytest, Rust, Go, Prettier, etc.
|
|
191
|
+
* Returns up to 10 unique locations sorted by file then line.
|
|
192
|
+
*/
|
|
193
|
+
interface ErrorLocation {
|
|
194
|
+
file: string
|
|
195
|
+
line: number
|
|
196
|
+
col?: number
|
|
197
|
+
message?: string
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
function parseErrorLocations(stderr: string): ErrorLocation[] {
|
|
201
|
+
const locations: ErrorLocation[] = []
|
|
202
|
+
|
|
203
|
+
// TypeScript / ESLint / Prettier: path(line,col): message
|
|
204
|
+
// e.g., src/foo.ts(42,10): error TS2304: Cannot find name 'foo'
|
|
205
|
+
const tsPattern = /([^\s(]+)\((\d+),(\d+)\):\s*(.+)/g
|
|
206
|
+
let match: RegExpExecArray | null
|
|
207
|
+
let m: RegExpExecArray | null
|
|
208
|
+
while ((m = tsPattern.exec(stderr)) !== null) {
|
|
209
|
+
locations.push({
|
|
210
|
+
file: m[1]!,
|
|
211
|
+
line: parseInt(m[2]!),
|
|
212
|
+
col: parseInt(m[3]!),
|
|
213
|
+
message: (m[4] || '').slice(0, 120),
|
|
214
|
+
})
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// pytest / Python: path:line: message
|
|
218
|
+
// e.g., tests/test_foo.py:42: AssertionError: ...
|
|
219
|
+
const pyPattern = /([^\s:]+\.py):(\d+):\s*(.+)/g
|
|
220
|
+
while ((match = pyPattern.exec(stderr)) !== null) {
|
|
221
|
+
const pm = match
|
|
222
|
+
locations.push({
|
|
223
|
+
file: pm[1]!,
|
|
224
|
+
line: parseInt(pm[2]!),
|
|
225
|
+
message: (pm[3] || '').slice(0, 120),
|
|
226
|
+
})
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
// Rust: --> path:line:col
|
|
230
|
+
// e.g., --> src/main.rs:42:10
|
|
231
|
+
const rustPattern = /-->\s*([^\s:]+):(\d+):(\d+)/g
|
|
232
|
+
while ((match = rustPattern.exec(stderr)) !== null) {
|
|
233
|
+
const rm = match
|
|
234
|
+
locations.push({
|
|
235
|
+
file: rm[1]!,
|
|
236
|
+
line: parseInt(rm[2]!),
|
|
237
|
+
col: parseInt(rm[3]!),
|
|
238
|
+
})
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
// Go: path:line:col: message
|
|
242
|
+
// e.g., ./main.go:42:10: undefined: foo
|
|
243
|
+
const goPattern = /([^\s:]+\.go):(\d+):(\d+):\s*(.+)/g
|
|
244
|
+
while ((match = goPattern.exec(stderr)) !== null) {
|
|
245
|
+
const gm2 = match
|
|
246
|
+
locations.push({
|
|
247
|
+
file: gm2[1]!,
|
|
248
|
+
line: parseInt(gm2[2]!),
|
|
249
|
+
col: parseInt(gm2[3]!),
|
|
250
|
+
message: (gm2[4] || '').slice(0, 120),
|
|
251
|
+
})
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
// Generic: path:line (any file extension)
|
|
255
|
+
// e.g., src/foo.ts:42
|
|
256
|
+
const genericPattern = /([^\s:]+\.[a-zA-Z]{1,6}):(\d+)\b/g
|
|
257
|
+
while ((match = genericPattern.exec(stderr)) !== null) {
|
|
258
|
+
const gm = match
|
|
259
|
+
const file = gm[1]!
|
|
260
|
+
// Skip if we already have this exact location from a more specific pattern
|
|
261
|
+
const alreadyHave = locations.some((l) => l.file === file && l.line === parseInt(gm[2]!))
|
|
262
|
+
if (!alreadyHave) {
|
|
263
|
+
locations.push({ file, line: parseInt(match[2]!) })
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
// Deduplicate and sort: same file+line → keep first
|
|
268
|
+
const seen = new Set<string>()
|
|
269
|
+
const unique: ErrorLocation[] = []
|
|
270
|
+
for (const loc of locations) {
|
|
271
|
+
const key = `${loc.file}:${loc.line}`
|
|
272
|
+
if (!seen.has(key)) {
|
|
273
|
+
seen.add(key)
|
|
274
|
+
unique.push(loc)
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// Sort by file path then line number
|
|
279
|
+
unique.sort((a, b) => {
|
|
280
|
+
const fileCmp = a.file.localeCompare(b.file)
|
|
281
|
+
return fileCmp !== 0 ? fileCmp : a.line - b.line
|
|
282
|
+
})
|
|
283
|
+
|
|
284
|
+
return unique.slice(0, 10)
|
|
285
|
+
}
|
|
286
|
+
|
|
188
287
|
export const bashTool: ToolDefinition = {
|
|
189
288
|
name: 'Bash',
|
|
190
289
|
description:
|
|
@@ -283,6 +382,20 @@ export const bashTool: ToolDefinition = {
|
|
|
283
382
|
if (violations.length > 0) {
|
|
284
383
|
errorContent += '\n\n── Sandbox Violations ──\n' + violations.join('\n')
|
|
285
384
|
}
|
|
385
|
+
// Vibe coding fix: auto-parse error locations from stderr
|
|
386
|
+
const errorLocations = parseErrorLocations(rawStderr)
|
|
387
|
+
if (errorLocations.length > 0) {
|
|
388
|
+
errorContent +=
|
|
389
|
+
'\n\n── Error Locations (for quick fix) ──\n' +
|
|
390
|
+
errorLocations
|
|
391
|
+
.map(
|
|
392
|
+
(l) =>
|
|
393
|
+
` ${l.file}:${l.line}` +
|
|
394
|
+
(l.col ? `:${l.col}` : '') +
|
|
395
|
+
(l.message ? ` — ${l.message}` : ''),
|
|
396
|
+
)
|
|
397
|
+
.join('\n')
|
|
398
|
+
}
|
|
286
399
|
return {
|
|
287
400
|
success: false,
|
|
288
401
|
content: errorContent,
|