@miphamai/cli 0.25.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@miphamai/cli",
3
- "version": "0.25.0",
3
+ "version": "0.27.0",
4
4
  "description": "Mipham Code — Multi-model open-core intelligent coding terminal by MiphamAI",
5
5
  "keywords": [
6
6
  "ai",
@@ -0,0 +1,189 @@
1
+ ---
2
+ name: codebase-design
3
+ description: Deep module design principles for designing or improving module interfaces. Use when designing a new module, refactoring an existing one, or finding deepening opportunities in the codebase.
4
+ version: 1.0.0
5
+ user-invocable: true
6
+ allowed-tools:
7
+ - Read
8
+ - Glob
9
+ - Grep
10
+ - Edit
11
+ - Write
12
+ ---
13
+
14
+ # Codebase Design — Deep Module Principles
15
+
16
+ Based on John Ousterhout's "A Philosophy of Software Design." The core idea: the greatest single factor in software complexity is the depth of modules — how much functionality they provide relative to the size of their interface.
17
+
18
+ ## When to Use
19
+
20
+ - Designing a new module, API, or component
21
+ - Reviewing existing code for design quality
22
+ - Deciding where to split or join modules
23
+ - User asks: "is this well-designed?", "where should this go?", "how should I structure this?"
24
+
25
+ ---
26
+
27
+ ## Core Concepts
28
+
29
+ ### Deep vs Shallow Modules
30
+
31
+ ```
32
+ Deep Module (good): Shallow Module (bad):
33
+ ┌─────────────────┐ ┌─────────────────┐
34
+ │ small interface │ │ large interface │
35
+ │ ┌─────────────┐ │ │ (many params, │
36
+ │ │ │ │ │ complex setup) │
37
+ │ │ large │ │ ├─────────────────┤
38
+ │ │ implementation│ │ small │
39
+ │ │ │ │ │ implementation │
40
+ │ └─────────────┘ │ │ (just passes │
41
+ └─────────────────┘ │ through) │
42
+ └─────────────────┘
43
+ ```
44
+
45
+ **Deep**: Unix file I/O — 5 syscalls (`open`, `read`, `write`, `lseek`, `close`), incredibly powerful implementation.
46
+
47
+ **Shallow**: A function that takes 12 parameters, validates 3 of them, then calls another function. Interface cost > implementation value.
48
+
49
+ ### The Rule of Deep Modules
50
+
51
+ > The interface should be as small as possible while providing as much functionality as possible.
52
+
53
+ - **Cost** = interface complexity (parameters, configuration, setup required)
54
+ - **Benefit** = functionality provided (what the caller no longer needs to worry about)
55
+ - **Depth** = Benefit / Cost
56
+
57
+ ---
58
+
59
+ ## The 5 Design Checks
60
+
61
+ When evaluating a module design, run through these:
62
+
63
+ ### Check 1: Interface Size
64
+
65
+ Count the effective parameters:
66
+
67
+ - Required parameters + optional parameters with non-trivial defaults
68
+ - Configuration methods that MUST be called before use
69
+ - Implicit dependencies (global state, env vars, singletons)
70
+
71
+ **Red flag**: > 4 effective parameters → the module may be too shallow.
72
+
73
+ **Fix**: Bundle related parameters into a config object. Or split the module.
74
+
75
+ ### Check 2: Information Hiding
76
+
77
+ Does the module expose information that callers don't need?
78
+
79
+ - Internal data structures leaked through the interface
80
+ - Implementation details exposed via parameter types
81
+ - Error types that reveal internal architecture
82
+
83
+ **Red flag**: Callers import types they don't use directly.
84
+
85
+ **Fix**: Define a public API type layer. Return opaque handles instead of raw data.
86
+
87
+ ### Check 3: Abstraction Quality
88
+
89
+ Does the module represent a single, coherent idea?
90
+
91
+ - Can you describe what it does in one sentence without "and"?
92
+ - Would a new team member guess where to find this functionality?
93
+ - If you remove the module, does exactly one concept go missing?
94
+
95
+ **Red flag**: Module name contains "and", "Utils", "Common", "Helpers".
96
+
97
+ **Fix**: Split by concept. `UserService` + `EmailService` instead of `UserAndEmailUtils`.
98
+
99
+ ### Check 4: General-Purpose vs Special-Purpose
100
+
101
+ Is the module solving the general case or a specific use case?
102
+
103
+ - Would the interface work if requirements changed slightly?
104
+ - Are there hardcoded assumptions that could be parameters?
105
+ - Is the module useful in contexts other than its creator imagined?
106
+
107
+ **Red flag**: Module only works for one specific call site.
108
+
109
+ **Fix**: Make the specific case a thin wrapper around the general case. The general module is deep; the wrapper is shallow (and that's fine — wrappers are allowed to be shallow).
110
+
111
+ ### Check 5: Seam Placement
112
+
113
+ Where you split modules matters as much as what they do.
114
+
115
+ - Does the split happen at a natural boundary?
116
+ - Are there circular dependencies across the seam?
117
+ - Can each side be tested independently?
118
+
119
+ **Red flag**: Circular imports, or modules that are always imported together.
120
+
121
+ **Fix**: Use dependency inversion. Define interfaces at the seam, not implementations.
122
+
123
+ ---
124
+
125
+ ## Finding Deepening Opportunities
126
+
127
+ Scan the codebase for these patterns:
128
+
129
+ ### Shallow Pass-Through
130
+
131
+ ```typescript
132
+ // Shallow — just delegates with no added value
133
+ function getUser(id: string) {
134
+ return db.findUser(id)
135
+ }
136
+
137
+ // Deep — handles errors, caching, authorization in one call
138
+ function getUser(id: string, ctx: RequestContext) {
139
+ const cached = cache.get(`user:${id}`)
140
+ if (cached) return cached
141
+ ctx.auth.assertCanRead('user', id)
142
+ const user = db.findUser(id)
143
+ if (!user) throw new NotFoundError('User', id)
144
+ cache.set(`user:${id}`, user)
145
+ return user
146
+ }
147
+ ```
148
+
149
+ ### Temporal Decomposition
150
+
151
+ When a module's methods must be called in a specific order, the interface is too wide.
152
+
153
+ ```typescript
154
+ // Shallow — caller manages lifecycle
155
+ const conn = new Connection()
156
+ conn.open()
157
+ conn.authenticate(token)
158
+ conn.send(data)
159
+ conn.close()
160
+
161
+ // Deep — module manages lifecycle
162
+ const conn = await Connection.create(token)
163
+ conn.send(data)
164
+ // clean up automatically
165
+ ```
166
+
167
+ ### Overexposure
168
+
169
+ When internal types leak through the public API:
170
+
171
+ ```typescript
172
+ // Shallow — exposes ORM internals
173
+ interface UserService {
174
+ findUser(id: string): Promise<PrismaUser | null> // ❌ PrismaUser is internal
175
+ }
176
+
177
+ // Deep — owns its types
178
+ interface UserService {
179
+ findUser(id: string): Promise<User | null> // ✅ User is a domain type
180
+ }
181
+ ```
182
+
183
+ ---
184
+
185
+ ## Integration With Mipham Code
186
+
187
+ - **code-review**: This skill fills the architecture dimension that code-review's 7 dimensions don't cover. Use `/code-review` for correctness/security/perf; use `/codebase-design` for interface depth/abstraction quality/seam placement.
188
+ - **domain-modeling**: Good domain modeling makes deep modules easier — the CONTEXT.md glossary defines the concepts that modules should represent.
189
+ - **Critical Thinking Layer**: The counter-example search applies directly: "what would break if I changed the implementation of this module?"
@@ -0,0 +1,129 @@
1
+ ---
2
+ name: domain-modeling
3
+ description: Build and sharpen a project's domain model. Use when the user wants to pin down domain terminology or a ubiquitous language, record an architectural decision, or when another skill needs to maintain the domain model.
4
+ version: 1.0.0
5
+ user-invocable: true
6
+ allowed-tools:
7
+ - Read
8
+ - Write
9
+ - Edit
10
+ - Glob
11
+ - Grep
12
+ ---
13
+
14
+ # Domain Modeling — Continuous Shared Language
15
+
16
+ Actively build and sharpen the project's domain model as you work. This is the _active_ discipline — challenging terms, inventing edge-case scenarios, and writing the glossary and decisions down the moment they crystallize. (Merely _reading_ `CONTEXT.md` for vocabulary is not this skill — that's a one-line habit any skill can do. This skill is for when you're changing the model, not just consuming it.)
17
+
18
+ ## File Structure
19
+
20
+ ```
21
+ /
22
+ ├── CONTEXT.md ← shared language glossary
23
+ ├── docs/
24
+ │ └── adr/
25
+ │ ├── 0001-slug.md ← architectural decisions
26
+ │ └── 0002-slug.md
27
+ └── src/
28
+ ```
29
+
30
+ Create files lazily — only when you have something to write.
31
+
32
+ **Multiple contexts**: If a `CONTEXT-MAP.md` exists, read it to find which context the current topic relates to.
33
+
34
+ ---
35
+
36
+ ## During the Session
37
+
38
+ ### Challenge Against the Glossary
39
+
40
+ When the user uses a term that conflicts with existing language in `CONTEXT.md`, call it out immediately:
41
+
42
+ > "Your glossary defines 'cancellation' as X, but you seem to mean Y — which is it?"
43
+
44
+ ### Sharpen Fuzzy Language
45
+
46
+ When the user uses vague or overloaded terms, propose a precise canonical term:
47
+
48
+ > "You're saying 'account' — do you mean the Customer or the User? Those are different things."
49
+
50
+ ### Discuss Concrete Scenarios
51
+
52
+ When domain relationships are discussed, stress-test them with specific scenarios. Invent scenarios that probe edge cases and force precision about boundaries between concepts.
53
+
54
+ ### Cross-Reference With Code
55
+
56
+ When the user states how something works, check whether the code agrees. Surface contradictions:
57
+
58
+ > "Your code cancels entire Orders, but you just said partial cancellation is possible — which is right?"
59
+
60
+ ### Update CONTEXT.md Inline
61
+
62
+ When a term is resolved, update `CONTEXT.md` right there. Don't batch — capture as they happen.
63
+
64
+ ### Offer ADRs Sparingly
65
+
66
+ Only create an ADR when ALL three are true:
67
+
68
+ 1. **Hard to reverse** — changing your mind later has real cost
69
+ 2. **Surprising without context** — a future reader would wonder "why?"
70
+ 3. **The result of a real trade-off** — there were genuine alternatives
71
+
72
+ ---
73
+
74
+ ## CONTEXT.md Format
75
+
76
+ ```markdown
77
+ # {Context Name}
78
+
79
+ {One or two sentence description of what this context is and why it exists.}
80
+
81
+ ## Language
82
+
83
+ **{Term}**:
84
+ {One or two sentence definition of what it IS.}
85
+ _Avoid_: {alternative terms that should not be used}
86
+ ```
87
+
88
+ ### Rules
89
+
90
+ - **Be opinionated.** Pick the best term, ban the rest.
91
+ - **Keep definitions tight.** One or two sentences max.
92
+ - **Only domain-specific terms.** Not general programming concepts.
93
+ - **Group under subheadings** when natural clusters emerge.
94
+
95
+ ---
96
+
97
+ ## ADR Format
98
+
99
+ ```markdown
100
+ # {Short title of the decision}
101
+
102
+ {1-3 sentences: context, decision, and why.}
103
+ ```
104
+
105
+ Number sequentially (`docs/adr/0001-slug.md`, `0002-slug.md`, ...).
106
+
107
+ Optional sections (only when they add value):
108
+
109
+ - **Status** frontmatter: `proposed | accepted | deprecated | superseded by ADR-NNNN`
110
+ - **Considered Options**: rejected alternatives worth remembering
111
+ - **Consequences**: non-obvious downstream effects
112
+
113
+ ### When an ADR Qualifies
114
+
115
+ - Architecture shape (monorepo, event sourcing, microservices)
116
+ - Integration patterns between contexts
117
+ - Technology choices with lock-in (database, message bus, auth)
118
+ - Boundary and scope decisions ("X owns Y, Z references by ID only")
119
+ - Deliberate deviations from convention
120
+ - Constraints not visible in code (compliance, latency SLA)
121
+ - Rejected alternatives when non-obvious (stops someone suggesting it again in 6 months)
122
+
123
+ ---
124
+
125
+ ## Integration With Mipham Code
126
+
127
+ - **Memory System**: Domain terms discovered through this skill persist to project memory
128
+ - **grill-with-docs**: For initial domain establishment, use `/grill-with-docs`. This skill handles ongoing maintenance
129
+ - **Critical Thinking Layer**: Apply counter-example search to domain definitions — "does this definition hold for all edge cases?"
@@ -0,0 +1,199 @@
1
+ ---
2
+ name: grill-with-docs
3
+ description: A relentless interview to sharpen a plan or design, creating CONTEXT.md (shared language) and ADRs (architectural decisions) as we go. Use before any non-trivial implementation to align on requirements and terminology.
4
+ version: 1.0.0
5
+ user-invocable: true
6
+ allowed-tools:
7
+ - Read
8
+ - Write
9
+ - Edit
10
+ - Bash
11
+ - Glob
12
+ - Grep
13
+ - WebSearch
14
+ - WebFetch
15
+ ---
16
+
17
+ # Grill With Docs — Deep Requirements Alignment
18
+
19
+ Inspired by Matt Pocock's `grill-with-docs` and `domain-modeling` skills. Before writing code, run a structured interview to align on requirements, establish shared language, and record architectural decisions.
20
+
21
+ ## When to Use
22
+
23
+ - Before any non-trivial feature implementation
24
+ - When requirements are fuzzy ("make it faster", "add X")
25
+ - When you need to establish project terminology
26
+ - When architectural decisions need to be recorded
27
+ - User says: "plan X", "design Y", "what should we do about Z"
28
+
29
+ ## When NOT to Use
30
+
31
+ - Trivial bug fixes with clear expected behavior
32
+ - One-line changes
33
+ - Tasks where the requirements are already crystal clear
34
+
35
+ ---
36
+
37
+ ## The Interview Flow
38
+
39
+ ### Phase 1: Understand the Intent
40
+
41
+ Start by understanding what the user actually wants. Don't ask "what should I build?" — ask about their goal.
42
+
43
+ **Core Questions:**
44
+
45
+ 1. What problem are you solving? (Not what feature you're building)
46
+ 2. Who is this for? (End user, developer, internal tool?)
47
+ 3. What does success look like? (How will you know when it's done?)
48
+ 4. What's the deadline or priority context?
49
+
50
+ **Anti-pattern**: Jumping to implementation questions ("Do you want REST or GraphQL?") before understanding the problem.
51
+
52
+ ### Phase 2: Sharpen the Language
53
+
54
+ Identify vague or overloaded terms and pin them down **immediately**. This is the single highest-leverage activity — shared language reduces token waste and prevents misunderstandings.
55
+
56
+ **Technique: The Canonical Term**
57
+
58
+ - When the user uses multiple words for the same thing, pick one as canonical
59
+ - List rejected alternatives under `_Avoid_`
60
+ - Be opinionated — the glossary is prescriptive, not descriptive
61
+
62
+ ```
63
+ User: "We need a way for users to save articles for later."
64
+ You: "Let's pin that down. 'Save for later' could mean bookmarking, or a reading list, or offline download. Which one?"
65
+ User: "Like a reading list — they can come back to it."
66
+ You: "Got it. Let's call it a **Reading List**. Avoid 'bookmark', 'save', 'favorites'."
67
+ → Write to CONTEXT.md immediately.
68
+ ```
69
+
70
+ **Technique: The Boundary Test**
71
+
72
+ - When a term is proposed, test its boundaries with edge cases
73
+ - "Does X include Y? What about Z?"
74
+
75
+ **Technique: The Code Cross-Reference**
76
+
77
+ - When the user describes how something works, check if existing code agrees
78
+ - Surface contradictions immediately
79
+
80
+ ### Phase 3: Probe Edge Cases
81
+
82
+ Before accepting any requirement, stress-test it with edge cases.
83
+
84
+ **Edge Case Inventory:**
85
+
86
+ - **Empty state**: What does the user see when there's nothing yet?
87
+ - **Error state**: What happens when things go wrong?
88
+ - **Extreme values**: What about 0? What about 10,000?
89
+ - **Concurrency**: What if two people do this at the same time?
90
+ - **Permissions**: Who can do this? Who cannot?
91
+ - **Scale**: What changes at 10x the current volume?
92
+
93
+ **Technique: The 5 Whys**
94
+ When a requirement seems odd, dig deeper:
95
+
96
+ ```
97
+ User: "We need real-time updates."
98
+ You: "Why real-time?"
99
+ User: "Because users need to see changes immediately."
100
+ You: "Why do they need to see changes immediately?"
101
+ User: "Because they're collaborating on the same document."
102
+ → Now you know the REAL requirement is collaboration, not real-time.
103
+ ```
104
+
105
+ ### Phase 4: Make Architecture Decisions
106
+
107
+ When a design decision meets ALL three criteria, offer to record it as an ADR:
108
+
109
+ 1. **Hard to reverse** — changing your mind later has real cost
110
+ 2. **Surprising without context** — a future reader would wonder "why?"
111
+ 3. **The result of a real trade-off** — there were genuine alternatives
112
+
113
+ **What qualifies for an ADR:**
114
+
115
+ - Architecture shape (monorepo vs polyrepo, event sourcing vs CRUD)
116
+ - Integration patterns between contexts
117
+ - Technology choices with lock-in (database, message bus, auth provider)
118
+ - Deliberate deviations from convention ("we use raw SQL because...")
119
+ - Constraints not visible in code ("we can't use X because compliance")
120
+
121
+ **ADR Format** (write to `docs/adr/NNNN-slug.md`):
122
+
123
+ ```markdown
124
+ # {Short title of the decision}
125
+
126
+ {1-3 sentences: context, decision, and why.}
127
+ ```
128
+
129
+ Only add optional sections (Status, Considered Options, Consequences) when they add genuine value. Most ADRs are a single paragraph.
130
+
131
+ ### Phase 5: Write the CONTEXT.md
132
+
133
+ After the interview, synthesize everything into `CONTEXT.md`.
134
+
135
+ **Format** (`CONTEXT.md` at project root):
136
+
137
+ ```markdown
138
+ # {Project Name} Context
139
+
140
+ {One or two sentence description of the project domain.}
141
+
142
+ ## Language
143
+
144
+ **{Term}**:
145
+ {One or two sentence definition of what it IS.}
146
+ _Avoid_: {alternative terms that should not be used}
147
+
148
+ ## Decisions
149
+
150
+ - [ADR 0001: {Title}](docs/adr/0001-slug.md) — {one-line summary}
151
+ ```
152
+
153
+ **Rules:**
154
+
155
+ - Be opinionated — pick the best term, ban the rest
156
+ - Only include domain-specific terms (not general programming concepts)
157
+ - Keep definitions tight — one or two sentences
158
+ - Update inline during the conversation, don't batch
159
+ - CONTEXT.md is a glossary, NOT a spec or implementation plan
160
+
161
+ ---
162
+
163
+ ## During the Conversation
164
+
165
+ ### DO
166
+
167
+ - Challenge the user when they use vague terms — "What do you mean by 'fast'?"
168
+ - Propose canonical terms and write them down immediately
169
+ - Invent edge cases and probe boundaries
170
+ - Offer ADRs sparingly (only when all 3 criteria are met)
171
+ - Cross-reference with existing code if available
172
+ - Call out contradictions between what the user says and what the code does
173
+
174
+ ### DON'T
175
+
176
+ - Rush to implementation questions before understanding the problem
177
+ - Write ADRs for trivial decisions
178
+ - Let fuzzy language slide — pin it down now or pay later
179
+ - Treat CONTEXT.md as a spec or scratch pad
180
+ - Ask yes/no questions when open-ended ones would reveal more
181
+
182
+ ---
183
+
184
+ ## Output
185
+
186
+ After the interview, the user should have:
187
+
188
+ 1. **CONTEXT.md** — shared language glossary (created or updated)
189
+ 2. **ADRs** (if needed) — architectural decisions in `docs/adr/`
190
+ 3. **Clear requirements** — edge cases explored, assumptions surfaced
191
+ 4. **Shared understanding** — you and the user now mean the same thing by the same words
192
+
193
+ ---
194
+
195
+ ## Integration with Mipham Code
196
+
197
+ - **Memory System**: Key terms go to project memory for persistence across sessions
198
+ - **Critical Thinking Layer**: Apply the 5-dimension self-check (evidence standard, equivalence verification, counter-example search, confidence calibration, depth check) to your own interview questions
199
+ - **Workflow**: For complex projects, the output of this skill feeds directly into `/implement`
@@ -0,0 +1,138 @@
1
+ ---
2
+ name: to-spec
3
+ description: Turn a conversation into a structured specification document. Use after a grill-with-docs session or any requirements discussion to capture decisions in a durable, shareable format.
4
+ version: 1.0.0
5
+ user-invocable: true
6
+ allowed-tools:
7
+ - Read
8
+ - Write
9
+ - Edit
10
+ - Bash
11
+ ---
12
+
13
+ # To Spec — Conversation → Specification
14
+
15
+ Turn the output of a requirements discussion into a structured specification document. This is the bridge between `/grill-with-docs` (alignment) and `/triage` (task decomposition).
16
+
17
+ ## When to Use
18
+
19
+ - After a `/grill-with-docs` session — capture what was decided
20
+ - After any requirements discussion — before starting implementation
21
+ - User asks: "write this up", "create a spec", "document the plan"
22
+ - Before handing off work to another session or person
23
+
24
+ ## When NOT to Use
25
+
26
+ - The requirements are a single sentence and obvious
27
+ - You're in the middle of a grill session — finish the interview first
28
+ - The scope is so small that the spec would be longer than the implementation
29
+
30
+ ---
31
+
32
+ ## Spec Format
33
+
34
+ Write to `docs/specs/YYYY-MM-DD-slug.md`:
35
+
36
+ ```markdown
37
+ ---
38
+ status: draft | approved | implemented
39
+ created: 2026-08-10
40
+ ---
41
+
42
+ # {Title}
43
+
44
+ ## Problem
45
+
46
+ {What problem are we solving? Why now? 1-3 sentences.}
47
+
48
+ ## Scope
49
+
50
+ ### In Scope
51
+
52
+ - {What we're building}
53
+
54
+ ### Out of Scope (Explicit)
55
+
56
+ - {What we're NOT building — prevents scope creep}
57
+
58
+ ## Requirements
59
+
60
+ ### Functional
61
+
62
+ - **{Requirement}**: {Description}. Acceptance: {measurable criterion}.
63
+
64
+ ### Non-Functional
65
+
66
+ - **Performance**: {latency, throughput targets}
67
+ - **Security**: {auth, data protection, threat model}
68
+ - **Scale**: {expected volume, growth projections}
69
+
70
+ ## Design Decisions
71
+
72
+ - **Decision**: {What we decided}. Because: {why}. Alternatives considered: {options + reasons rejected}.
73
+
74
+ ## Domain Model
75
+
76
+ {Key terms and their definitions — from CONTEXT.md or the grill session.}
77
+
78
+ ## Edge Cases
79
+
80
+ - **{Scenario}**: {Expected behavior}
81
+ - **{Scenario}**: {Expected behavior}
82
+
83
+ ## Open Questions
84
+
85
+ - {Question} — {who needs to answer / when needed}
86
+ ```
87
+
88
+ ---
89
+
90
+ ## The Spec Workflow
91
+
92
+ ### Step 1: Extract from Conversation
93
+
94
+ Scan the conversation history for:
95
+
96
+ - Decisions made (explicit and implicit)
97
+ - Terms defined (candidates for CONTEXT.md)
98
+ - Edge cases discussed
99
+ - Alternatives rejected (and why)
100
+ - Open questions that remain
101
+
102
+ ### Step 2: Fill Gaps
103
+
104
+ For each gap you find:
105
+
106
+ - Edge cases not discussed → flag as Open Questions
107
+ - Terms used but not defined → propose definitions
108
+ - Assumptions not stated → make them explicit
109
+
110
+ ### Step 3: Validate with User
111
+
112
+ Present the spec and ask:
113
+
114
+ 1. "Does this match your understanding?"
115
+ 2. "What's missing?"
116
+ 3. "What's wrong?"
117
+ 4. "What surprised you?"
118
+
119
+ ### Step 4: Feed Into Triage
120
+
121
+ Once approved, the spec's functional requirements become tickets in `/triage`. Non-functional requirements become acceptance criteria.
122
+
123
+ ---
124
+
125
+ ## Anti-Patterns
126
+
127
+ - **Waterfall trap**: Don't try to spec everything upfront. Spec the next increment. Specs are living documents, not contracts.
128
+ - **Premature detail**: Don't spec API signatures or DB schemas in the spec — those are implementation details.
129
+ - **Vague acceptance**: "Works well" is not acceptance criteria. "Returns 200 with valid JWT within 500ms" is.
130
+
131
+ ---
132
+
133
+ ## Integration With Mipham Code
134
+
135
+ - **grill-with-docs**: Input — the grill session produces the raw material
136
+ - **triage**: Output — the spec feeds into ticket decomposition
137
+ - **domain-modeling**: Terms discovered during spec writing go to CONTEXT.md
138
+ - **Memory System**: The spec file persists as project reference across sessions
@@ -0,0 +1,155 @@
1
+ ---
2
+ name: triage
3
+ description: Structured task decomposition and tracking across sessions. Use for breaking complex plans into trackable tickets with dependency graphs, checking task status, or continuing work from a previous session.
4
+ version: 1.0.0
5
+ user-invocable: true
6
+ allowed-tools:
7
+ - Read
8
+ - Write
9
+ - Edit
10
+ - Bash
11
+ - Glob
12
+ - Grep
13
+ ---
14
+
15
+ # Triage — Cross-Session Task Tracking
16
+
17
+ Turn plans into trackable tickets with dependency management. Inspired by Matt Pocock's `triage` + `to-tickets` + `wayfinder` skills, consolidated into one Mipham Code skill.
18
+
19
+ ## When to Use
20
+
21
+ - Breaking a large plan into actionable tickets
22
+ - Tracking work across multiple sessions
23
+ - User asks: "what's next?", "where did I leave off?", "what's the status?"
24
+ - Complex tasks with dependencies between them
25
+
26
+ ---
27
+
28
+ ## The Ticket Format
29
+
30
+ Tickets live in `.mipham/tickets/` as individual Markdown files:
31
+
32
+ ```markdown
33
+ ---
34
+ id: T-001
35
+ title: Add user authentication
36
+ status: in-progress
37
+ priority: P0
38
+ depends_on: []
39
+ blocks: [T-003]
40
+ created: 2026-08-10
41
+ tags:
42
+ - auth
43
+ - backend
44
+ ---
45
+
46
+ ## Description
47
+
48
+ Add JWT-based authentication with refresh token rotation.
49
+
50
+ ## Acceptance Criteria
51
+
52
+ - [ ] Login endpoint returns access + refresh tokens
53
+ - [ ] Refresh endpoint rotates tokens
54
+ - [ ] Invalid tokens return 401
55
+ - [ ] Rate limiting on login attempts
56
+
57
+ ## Notes
58
+
59
+ - OAuth not in scope for T-001 (punted to T-005)
60
+ ```
61
+
62
+ ### Status Values
63
+
64
+ | Status | Meaning |
65
+ | ------------- | ------------------------------------------ |
66
+ | `backlog` | Not yet planned for any session |
67
+ | `planned` | Scoped and ready to work |
68
+ | `in-progress` | Currently being worked on |
69
+ | `review` | Implementation done, awaiting verification |
70
+ | `done` | Verified and merged |
71
+ | `blocked` | Cannot proceed due to dependency |
72
+ | `wontfix` | Decided not to do |
73
+
74
+ ---
75
+
76
+ ## The Triage Workflow
77
+
78
+ ### Phase 1: Decompose (Plan → Tickets)
79
+
80
+ Given a plan or feature request:
81
+
82
+ 1. **Identify the smallest independently-valuable units of work**
83
+ - Each ticket should deliver value on its own
84
+ - If a ticket requires 3+ files touched, it's probably too big
85
+ - If a ticket can be done in < 15 minutes, it's probably too small
86
+
87
+ 2. **Map dependencies**
88
+ - What must be done first? (hard dependency)
89
+ - What would be easier after something else? (soft dependency)
90
+ - What blocks other work? (reverse dependency)
91
+
92
+ 3. **Assign priorities**
93
+ - **P0**: Blocks other work, must do first
94
+ - **P1**: High value, should do soon
95
+ - **P2**: Nice to have, can defer
96
+ - **P3**: Optional, do if time permits
97
+
98
+ 4. **Write acceptance criteria**
99
+ - Specific, testable, unambiguous
100
+ - "Login works" is bad. "POST /auth/login with valid credentials returns 200 + JWT" is good.
101
+
102
+ ### Phase 2: Status Check
103
+
104
+ When the user asks "what's next?" or "what's the status?":
105
+
106
+ 1. Read `.mipham/tickets/` directory
107
+ 2. Report:
108
+ - Currently in-progress tickets
109
+ - Blocked tickets (and what's blocking them)
110
+ - Next unblocked P0/P1 tickets ready to work
111
+ - Recently completed tickets (for context)
112
+
113
+ ### Phase 3: Session Handoff
114
+
115
+ When starting a new session, check for continuity:
116
+
117
+ 1. Read the previous session's context from the session store
118
+ 2. Check ticket statuses — any that were `in-progress` last session?
119
+ 3. Present: "Last session you were working on T-004 (Add rate limiting). Continue from there, or start on T-007 (API docs) which is next in the P1 queue?"
120
+
121
+ ### Phase 4: Ticket Lifecycle
122
+
123
+ When working on a ticket:
124
+
125
+ - Mark it `in-progress` when you start
126
+ - Mark it `review` when implementation is done
127
+ - Mark it `done` after verification (tests pass, typecheck clean)
128
+ - If you discover new dependencies, add them to `blocks`/`depends_on`
129
+
130
+ ---
131
+
132
+ ## Dependency Graph
133
+
134
+ For tickets with complex dependencies, generate a visual summary:
135
+
136
+ ```
137
+ T-001 (Auth) ──blocks──→ T-003 (Dashboard)
138
+ │ │
139
+ └──blocks──→ T-002 (API) ─┘
140
+ │
141
+ └──soft-dep──→ T-004 (Rate Limiting)
142
+
143
+ Ready to work: T-001 (no dependencies)
144
+ Blocked: T-002 (waiting on T-001), T-003 (waiting on T-001, T-002)
145
+ ```
146
+
147
+ ---
148
+
149
+ ## Integration With Mipham Code
150
+
151
+ - **Session Store**: Ticket status persists across sessions via `.mipham/tickets/`
152
+ - **Memory System**: Active tickets are loaded as project memory for context
153
+ - **grill-with-docs**: The output of a grill session feeds directly into ticket decomposition
154
+ - **Background Agents**: Long-running work on a ticket can be spawned as a background agent
155
+ - **Critical Thinking Layer**: When decomposing, ask "what's the smallest thing that delivers value?" — don't over-decompose
@@ -71,6 +71,46 @@ export class InstructionsLoader {
71
71
  parts.push(this.skillsReminder)
72
72
  }
73
73
 
74
+ // Inject critical thinking self-check layer (for analysis/comparison tasks)
75
+ parts.push(`## Critical Thinking Self-Check
76
+
77
+ Before delivering any analysis, comparison, evaluation, or "X vs Y"
78
+ report, run this checklist internally:
79
+
80
+ ### 1. Evidence Standard
81
+ - Every factual claim MUST cite a specific source (file path, URL, line number)
82
+ - If you cannot cite a source, label the claim as [推断] (inference) or [待验证] (unverified)
83
+ - Numbers (counts, percentages, download stats) require cross-validation from a second source
84
+
85
+ ### 2. Equivalence Verification
86
+ - When you claim "A is equivalent to B" or "X has been merged from Y",
87
+ compare their ACTUAL implementation, not just their names or descriptions
88
+ - If you haven't read both implementations, say "appears similar at the
89
+ description level; implementation equivalence not verified"
90
+
91
+ ### 3. Counter-Example Search
92
+ - For each major conclusion, find at least 1 counter-example or edge case
93
+ - If you cannot find one, state that explicitly: "No counter-example found
94
+ within the examined scope"
95
+ - When comparing two systems, ask: "What does X do that Y CANNOT do?"
96
+ (and vice versa) — don't just list overlaps
97
+
98
+ ### 4. Confidence Calibration
99
+ - Label each conclusion with confidence: [高] [中] [低]
100
+ - [高] = verified from source code or primary documentation
101
+ - [中] = inferred from description but not implementation-verified
102
+ - [低] = speculative, based on naming convention or surface similarity
103
+
104
+ ### 5. Depth Check
105
+ - If your analysis is based ONLY on file names and description fields,
106
+ you are doing surface analysis — state this limitation upfront
107
+ - To reach depth: read at least one implementation file per comparison target
108
+ - Ask: "What would a domain expert notice that I'm missing?"
109
+
110
+ These checks are not optional for analysis tasks. Apply them before
111
+ presenting conclusions, and surface any [低] confidence findings
112
+ explicitly rather than burying them.`)
113
+
74
114
  // Inject workflow auto-generation guidance
75
115
  parts.push(`## Workflow Auto-Generation
76
116
 
@@ -9,6 +9,53 @@ import type { PermissionRuleEntry } from '../shared/index.ts'
9
9
  import { matchBashRule, compileRule } from './permission-rules'
10
10
  import { loadPermissionConfig, nextMode, clampMode, MODE_CYCLE } from './permission-config'
11
11
 
12
+ /**
13
+ * Check if a Bash command is a "verification-only" command that should be
14
+ * auto-approved in acceptEdits mode. These are non-destructive read/check
15
+ * operations that form the core of the vibe coding edit→test→fix loop.
16
+ */
17
+ function isVerificationCommand(input: Record<string, unknown>): boolean {
18
+ const cmd = (input.command as string) || ''
19
+ // Patterns for verification-only commands (no side effects on codebase)
20
+ const verifyPatterns = [
21
+ /\bpnpm\s+test\b/, // test runner
22
+ /\bpnpm\s+t\b/, // shorthand test
23
+ /\bpnpm\s+typecheck\b/, // type checking
24
+ /\bpnpm\s+lint\b/, // linting
25
+ /\bpnpm\s+format:check\b/, // format check
26
+ /\bnpm\s+test\b/, // npm test
27
+ /\bnpm\s+run\s+test\b/, // npm run test
28
+ /\bvitest\b/, // vitest runner
29
+ /\bjest\b/, // jest runner
30
+ /\btsc\s+(?!init)/, // TypeScript compiler (not tsc init)
31
+ /\btsc\s+--noEmit\b/, // type check only
32
+ /\beslint\b/, // eslint
33
+ /\bprettier\s+--check\b/, // prettier check
34
+ /\bpytest\b/, // python test runner
35
+ /\bruff\s+check\b/, // python linter
36
+ /\bcargo\s+test\b/, // rust test
37
+ /\bcargo\s+check\b/, // rust check
38
+ /\bgo\s+test\b/, // go test
39
+ /\bgo\s+vet\b/, // go vet
40
+ /\bmake\s+test\b/, // make test
41
+ /\bgit\s+status\b/, // git status (read-only)
42
+ /\bgit\s+diff\b/, // git diff (read-only)
43
+ /\bgit\s+log\b/, // git log (read-only)
44
+ /\bgit\s+branch\b/, // git branch (read-only)
45
+ /\bls\b/, // list files
46
+ /\bcat\b/, // read file
47
+ /\bhead\b/, // read file start
48
+ /\btail\b/, // read file end
49
+ /\bwhich\b/, // find binary
50
+ /\becho\b/, // print text
51
+ /\bnode\s+-v\b/, // node version
52
+ /\bpython\s+--version\b/, // python version
53
+ /\bwhoami\b/, // current user
54
+ /\bpwd\b/, // current directory
55
+ ]
56
+ return verifyPatterns.some((p) => p.test(cmd))
57
+ }
58
+
12
59
  const VALID_MODES: Set<string> = new Set<string>(MODE_CYCLE)
13
60
 
14
61
  export class PermissionSystem {
@@ -276,7 +323,7 @@ export class PermissionSystem {
276
323
  }
277
324
 
278
325
  // 5. Mode baseline
279
- const baseline = this.modeBaseline(tool)
326
+ const baseline = this.modeBaseline(tool, input)
280
327
  if (baseline !== 'mode-baseline') {
281
328
  this.checkCache.set(cacheKey, baseline)
282
329
  return baseline
@@ -323,21 +370,29 @@ export class PermissionSystem {
323
370
  return rule.pattern === tool.name || rule.compiled.test(tool.name)
324
371
  }
325
372
 
326
- private modeBaseline(tool: ToolDefinition): PermissionLevel | 'mode-baseline' {
373
+ private modeBaseline(
374
+ tool: ToolDefinition,
375
+ input?: Record<string, unknown>,
376
+ ): PermissionLevel | 'mode-baseline' {
327
377
  switch (this.mode) {
328
378
  case 'default':
329
379
  // Delegate to tool.permission (backward compat)
330
380
  return 'mode-baseline'
331
381
 
332
382
  case 'acceptEdits':
333
- // Reads + file edits free; Bash requires approval
334
- return tool.category === 'file'
335
- ? ['Bash'].includes(tool.name)
336
- ? 'ask'
337
- : 'bypass'
338
- : tool.name === 'Bash'
339
- ? 'ask'
340
- : 'ask'
383
+ // Reads + file edits free; Bash auto-approved for verification commands
384
+ if (tool.category === 'file' && tool.name !== 'Bash') {
385
+ return 'bypass'
386
+ }
387
+ if (tool.name === 'Bash') {
388
+ // Vibe coding fix: auto-approve verification commands
389
+ // so the edit→test→fix loop isn't interrupted by permission prompts
390
+ if (input && isVerificationCommand(input)) {
391
+ return 'bypass'
392
+ }
393
+ return 'ask'
394
+ }
395
+ return 'ask'
341
396
 
342
397
  case 'plan':
343
398
  // Only reads, no writes or executes
@@ -80,12 +80,12 @@ export const enterPlanModeTool: ToolDefinition = {
80
80
  ' ⚠️ All other tools — require confirmation',
81
81
  '',
82
82
  'Design your approach, explore the codebase, then:',
83
- ' • Use ExitPlanMode to submit your plan for approval',
84
- ' • The user will review and approve before code changes begin',
83
+ ' • Use ExitPlanMode to submit your plan for user review',
84
+ ' • Present your plan and ask for explicit approval',
85
+ ' • The user must say "approved" before you switch to acceptEdits mode',
85
86
  '',
86
- 'ExitPlanMode parameters:',
87
- ' approved: true → switch to acceptEdits mode (code changes allowed)',
88
- ' approved: false → revert to default mode',
87
+ '⚠️ You CANNOT self-approve your plan. The user must explicitly confirm.',
88
+ ' After ExitPlanMode, wait for user approval before making code changes.',
89
89
  ].join('\n'),
90
90
  }
91
91
  },
@@ -1,53 +1,62 @@
1
+ import { readFileSync } from 'node:fs'
2
+ import { join } from 'node:path'
1
3
  import type { ToolDefinition } from '../../shared/index.ts'
2
4
 
3
5
  export const exitPlanModeTool: ToolDefinition = {
4
6
  name: 'ExitPlanMode',
5
7
  description:
6
- 'Exit plan mode and submit your plan for approval. ' +
7
- 'Set approved: true to switch to acceptEdits mode (code changes allowed). ' +
8
- 'Set approved: false to revert to default mode. ' +
9
- 'Use this after you have finished designing your approach in plan mode.',
8
+ 'Exit plan mode and present your plan for user approval. ' +
9
+ 'This tool does NOT switch to implementation mode — the user must explicitly approve first. ' +
10
+ 'After calling this, present your plan and ask the user to confirm. ' +
11
+ 'The user can approve by saying "approved" or "/approve", or by cycling to acceptEdits mode with Shift+Tab.',
10
12
  category: 'agent',
11
13
  permission: 'auto',
12
14
  parameters: {
13
15
  type: 'object',
14
16
  properties: {
15
- approved: {
16
- type: 'boolean',
17
+ planFile: {
18
+ type: 'string',
17
19
  description:
18
- 'Whether the user approved the plan. true → switch to acceptEdits mode. false → revert to default.',
20
+ 'Path to the plan file you wrote (e.g., .mipham/plans/plan-2026-08-10T12-00-00.md). If omitted, the most recent plan file is used.',
19
21
  },
20
22
  },
21
- required: ['approved'],
23
+ required: [],
22
24
  },
23
- async execute(params, _ctx) {
24
- const approved = params.approved === true
25
+ async execute(params, ctx) {
26
+ const planDir = join(ctx.cwd, '.mipham', 'plans')
27
+ const planFile = (params.planFile as string) || ''
25
28
 
26
- if (approved) {
27
- return {
28
- success: true,
29
- content: [
30
- '── Plan Approved ──',
31
- '',
32
- '✓ Exiting plan mode.',
33
- '✓ Switching to acceptEdits mode — reads and file edits are auto-approved.',
34
- '',
35
- 'You can now implement the plan. The plan file is in .mipham/plans/.',
36
- '',
37
- 'Use Shift+Tab to cycle permission modes if you need to change.',
38
- ].join('\n'),
29
+ // Try to read the plan to confirm it exists
30
+ let planContent = ''
31
+ try {
32
+ const planPath = planFile || ''
33
+ if (planPath) {
34
+ planContent = readFileSync(planPath, 'utf-8')
39
35
  }
36
+ } catch {
37
+ // Plan file not found — still exit plan mode
40
38
  }
41
39
 
42
40
  return {
43
41
  success: true,
44
42
  content: [
45
- '── Plan Mode Exited ──',
43
+ '── Plan Ready for Review ──',
46
44
  '',
47
- '✓ Returning to default permission mode.',
45
+ '✓ Exiting plan mode.',
46
+ '✓ Plan file saved. Present your plan to the user now.',
48
47
  '',
49
- 'No code changes were made. The plan file is preserved in .mipham/plans/.',
50
- 'Use EnterPlanMode again when ready to resume planning.',
48
+ '⚠️ IMPORTANT: You are still in limited permission mode.',
49
+ ' The user must explicitly approve before you can make changes.',
50
+ '',
51
+ 'Next steps:',
52
+ ' 1. Present your plan to the user (summarize key decisions)',
53
+ ' 2. Ask: "Does this plan look good? Reply approved to begin."',
54
+ ' 3. Wait for the user to explicitly say "approved" or "/approve"',
55
+ ' 4. Only then switch to acceptEdits mode (Shift+Tab or user action)',
56
+ '',
57
+ 'DO NOT start implementing until the user explicitly approves.',
58
+ 'DO NOT call ExitPlanMode with approved:true — that parameter no longer exists.',
59
+ planContent ? `\n── Plan Content (for reference) ──\n\n${planContent.slice(0, 3000)}` : '',
51
60
  ].join('\n'),
52
61
  }
53
62
  },
@@ -185,6 +185,105 @@ export function detectViolations(stderr: string): string[] {
185
185
  return violations
186
186
  }
187
187
 
188
+ /**
189
+ * Vibe coding: Parse stderr output for error locations (file path + line + column).
190
+ * Supports common tool formats: TypeScript, ESLint, pytest, Rust, Go, Prettier, etc.
191
+ * Returns up to 10 unique locations sorted by file then line.
192
+ */
193
+ interface ErrorLocation {
194
+ file: string
195
+ line: number
196
+ col?: number
197
+ message?: string
198
+ }
199
+
200
+ function parseErrorLocations(stderr: string): ErrorLocation[] {
201
+ const locations: ErrorLocation[] = []
202
+
203
+ // TypeScript / ESLint / Prettier: path(line,col): message
204
+ // e.g., src/foo.ts(42,10): error TS2304: Cannot find name 'foo'
205
+ const tsPattern = /([^\s(]+)\((\d+),(\d+)\):\s*(.+)/g
206
+ let match: RegExpExecArray | null
207
+ let m: RegExpExecArray | null
208
+ while ((m = tsPattern.exec(stderr)) !== null) {
209
+ locations.push({
210
+ file: m[1]!,
211
+ line: parseInt(m[2]!),
212
+ col: parseInt(m[3]!),
213
+ message: (m[4] || '').slice(0, 120),
214
+ })
215
+ }
216
+
217
+ // pytest / Python: path:line: message
218
+ // e.g., tests/test_foo.py:42: AssertionError: ...
219
+ const pyPattern = /([^\s:]+\.py):(\d+):\s*(.+)/g
220
+ while ((match = pyPattern.exec(stderr)) !== null) {
221
+ const pm = match
222
+ locations.push({
223
+ file: pm[1]!,
224
+ line: parseInt(pm[2]!),
225
+ message: (pm[3] || '').slice(0, 120),
226
+ })
227
+ }
228
+
229
+ // Rust: --> path:line:col
230
+ // e.g., --> src/main.rs:42:10
231
+ const rustPattern = /-->\s*([^\s:]+):(\d+):(\d+)/g
232
+ while ((match = rustPattern.exec(stderr)) !== null) {
233
+ const rm = match
234
+ locations.push({
235
+ file: rm[1]!,
236
+ line: parseInt(rm[2]!),
237
+ col: parseInt(rm[3]!),
238
+ })
239
+ }
240
+
241
+ // Go: path:line:col: message
242
+ // e.g., ./main.go:42:10: undefined: foo
243
+ const goPattern = /([^\s:]+\.go):(\d+):(\d+):\s*(.+)/g
244
+ while ((match = goPattern.exec(stderr)) !== null) {
245
+ const gm2 = match
246
+ locations.push({
247
+ file: gm2[1]!,
248
+ line: parseInt(gm2[2]!),
249
+ col: parseInt(gm2[3]!),
250
+ message: (gm2[4] || '').slice(0, 120),
251
+ })
252
+ }
253
+
254
+ // Generic: path:line (any file extension)
255
+ // e.g., src/foo.ts:42
256
+ const genericPattern = /([^\s:]+\.[a-zA-Z]{1,6}):(\d+)\b/g
257
+ while ((match = genericPattern.exec(stderr)) !== null) {
258
+ const gm = match
259
+ const file = gm[1]!
260
+ // Skip if we already have this exact location from a more specific pattern
261
+ const alreadyHave = locations.some((l) => l.file === file && l.line === parseInt(gm[2]!))
262
+ if (!alreadyHave) {
263
+ locations.push({ file, line: parseInt(match[2]!) })
264
+ }
265
+ }
266
+
267
+ // Deduplicate and sort: same file+line → keep first
268
+ const seen = new Set<string>()
269
+ const unique: ErrorLocation[] = []
270
+ for (const loc of locations) {
271
+ const key = `${loc.file}:${loc.line}`
272
+ if (!seen.has(key)) {
273
+ seen.add(key)
274
+ unique.push(loc)
275
+ }
276
+ }
277
+
278
+ // Sort by file path then line number
279
+ unique.sort((a, b) => {
280
+ const fileCmp = a.file.localeCompare(b.file)
281
+ return fileCmp !== 0 ? fileCmp : a.line - b.line
282
+ })
283
+
284
+ return unique.slice(0, 10)
285
+ }
286
+
188
287
  export const bashTool: ToolDefinition = {
189
288
  name: 'Bash',
190
289
  description:
@@ -283,6 +382,20 @@ export const bashTool: ToolDefinition = {
283
382
  if (violations.length > 0) {
284
383
  errorContent += '\n\n── Sandbox Violations ──\n' + violations.join('\n')
285
384
  }
385
+ // Vibe coding fix: auto-parse error locations from stderr
386
+ const errorLocations = parseErrorLocations(rawStderr)
387
+ if (errorLocations.length > 0) {
388
+ errorContent +=
389
+ '\n\n── Error Locations (for quick fix) ──\n' +
390
+ errorLocations
391
+ .map(
392
+ (l) =>
393
+ ` ${l.file}:${l.line}` +
394
+ (l.col ? `:${l.col}` : '') +
395
+ (l.message ? ` — ${l.message}` : ''),
396
+ )
397
+ .join('\n')
398
+ }
286
399
  return {
287
400
  success: false,
288
401
  content: errorContent,