@markus-global/cli 0.8.4 → 0.8.5-rc.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/agent.js +9 -9
- package/dist/commands/agent.js.map +1 -1
- package/dist/commands/doctor.d.ts +3 -1
- package/dist/commands/doctor.d.ts.map +1 -1
- package/dist/commands/doctor.js +27 -1
- package/dist/commands/doctor.js.map +1 -1
- package/dist/commands/models.d.ts.map +1 -1
- package/dist/commands/models.js +6 -7
- package/dist/commands/models.js.map +1 -1
- package/dist/commands/project.d.ts +3 -0
- package/dist/commands/project.d.ts.map +1 -0
- package/dist/commands/project.js +25 -0
- package/dist/commands/project.js.map +1 -0
- package/dist/commands/requirement.d.ts +3 -0
- package/dist/commands/requirement.d.ts.map +1 -0
- package/dist/commands/requirement.js +34 -0
- package/dist/commands/requirement.js.map +1 -0
- package/dist/commands/start.d.ts.map +1 -1
- package/dist/commands/start.js +42 -1
- package/dist/commands/start.js.map +1 -1
- package/dist/commands/task.d.ts +3 -0
- package/dist/commands/task.d.ts.map +1 -0
- package/dist/commands/task.js +110 -0
- package/dist/commands/task.js.map +1 -0
- package/dist/index.js +8 -0
- package/dist/index.js.map +1 -1
- package/dist/markus.mjs +3770 -965
- package/dist/output.d.ts +3 -1
- package/dist/output.d.ts.map +1 -1
- package/dist/output.js +34 -3
- package/dist/output.js.map +1 -1
- package/dist/web-ui/assets/arc-azDa9rNQ.js +1 -0
- package/dist/web-ui/assets/architectureDiagram-3BPJPVTR-CWoGp8TB.js +36 -0
- package/dist/web-ui/assets/blockDiagram-GPEHLZMM-C2Tq3zqo.js +132 -0
- package/dist/web-ui/assets/c4Diagram-AAUBKEIU-C2tj98or.js +10 -0
- package/dist/web-ui/assets/channel-D0Q-P9rQ.js +1 -0
- package/dist/web-ui/assets/chunk-2J33WTMH-DMhlyS99.js +1 -0
- package/dist/web-ui/assets/chunk-4BX2VUAB-C8hL0QFv.js +1 -0
- package/dist/web-ui/assets/chunk-55IACEB6-BPCe4caz.js +1 -0
- package/dist/web-ui/assets/chunk-727SXJPM-C5EAjSrN.js +206 -0
- package/dist/web-ui/assets/chunk-AQP2D5EJ-BLSz7iPE.js +231 -0
- package/dist/web-ui/assets/chunk-FMBD7UC4-Sk4yLzwq.js +15 -0
- package/dist/web-ui/assets/chunk-ND2GUHAM-DbuQgWyn.js +1 -0
- package/dist/web-ui/assets/chunk-QZHKN3VN-REM6PaDE.js +1 -0
- package/dist/web-ui/assets/classDiagram-4FO5ZUOK-BcOPwdcC.js +1 -0
- package/dist/web-ui/assets/classDiagram-v2-Q7XG4LA2-BcOPwdcC.js +1 -0
- package/dist/web-ui/assets/cose-bilkent-S5V4N54A-Chu2Y9EC.js +1 -0
- package/dist/web-ui/assets/cytoscape.esm-D3_iZ_3b.js +321 -0
- package/dist/web-ui/assets/dagre-BM42HDAG-BGQGbUMF.js +4 -0
- package/dist/web-ui/assets/defaultLocale-DX6XiGOO.js +1 -0
- package/dist/web-ui/assets/diagram-2AECGRRQ-DnZ1SQGN.js +43 -0
- package/dist/web-ui/assets/diagram-5GNKFQAL-B0S37NyM.js +10 -0
- package/dist/web-ui/assets/diagram-KO2AKTUF-BxDwUDuY.js +3 -0
- package/dist/web-ui/assets/diagram-LMA3HP47-5UC8M7Iq.js +24 -0
- package/dist/web-ui/assets/diagram-OG6HWLK6-C-eMeMRG.js +24 -0
- package/dist/web-ui/assets/erDiagram-TEJ5UH35-CVWhzHKv.js +85 -0
- package/dist/web-ui/assets/flowDiagram-I6XJVG4X-CMG-a-kh.js +162 -0
- package/dist/web-ui/assets/ganttDiagram-6RSMTGT7-DbVJ6VGB.js +292 -0
- package/dist/web-ui/assets/gitGraphDiagram-PVQCEYII-DcCuYR-R.js +106 -0
- package/dist/web-ui/assets/graph--OzhPTMs.js +1 -0
- package/dist/web-ui/assets/index-PVrVcpcl.css +1 -0
- package/dist/web-ui/assets/index-zJq4U9RT.js +776 -0
- package/dist/web-ui/assets/infoDiagram-5YYISTIA-CaY7gJ4a.js +2 -0
- package/dist/web-ui/assets/init-Gi6I4Gst.js +1 -0
- package/dist/web-ui/assets/ishikawaDiagram-YF4QCWOH-l4_2NV1P.js +70 -0
- package/dist/web-ui/assets/journeyDiagram-JHISSGLW-BVeQNwa5.js +139 -0
- package/dist/web-ui/assets/kanban-definition-UN3LZRKU-CtvPOV3r.js +89 -0
- package/dist/web-ui/assets/layout-SsrduOYp.js +1 -0
- package/dist/web-ui/assets/linear-B0DfGdNc.js +1 -0
- package/dist/web-ui/assets/mermaid.core-Bz3avYM5.js +303 -0
- package/dist/web-ui/assets/mindmap-definition-RKZ34NQL-1X-u7gPH.js +96 -0
- package/dist/web-ui/assets/ordinal-Cboi1Yqb.js +1 -0
- package/dist/web-ui/assets/pieDiagram-4H26LBE5-BO8LpJ1H.js +30 -0
- package/dist/web-ui/assets/plantuml-DezRDxd4.js +357 -0
- package/dist/web-ui/assets/quadrantDiagram-W4KKPZXB-BBYmPM7O.js +7 -0
- package/dist/web-ui/assets/requirementDiagram-4Y6WPE33-CUU8gZny.js +84 -0
- package/dist/web-ui/assets/sankeyDiagram-5OEKKPKP-k6GjcALi.js +40 -0
- package/dist/web-ui/assets/sequenceDiagram-3UESZ5HK-CScOE6Nf.js +162 -0
- package/dist/web-ui/assets/stateDiagram-AJRCARHV-BcvHRBZl.js +1 -0
- package/dist/web-ui/assets/stateDiagram-v2-BHNVJYJU-p321ujvX.js +1 -0
- package/dist/web-ui/assets/timeline-definition-PNZ67QCA-D7tGfjR6.js +120 -0
- package/dist/web-ui/assets/vennDiagram-CIIHVFJN-AU7MqjmN.js +34 -0
- package/dist/web-ui/assets/viz-global-C_AyN6D9.js +9 -0
- package/dist/web-ui/assets/wardley-L42UT6IY-DKQmSXOS.js +161 -0
- package/dist/web-ui/assets/wardleyDiagram-YWT4CUSO-DVMv24j_.js +78 -0
- package/dist/web-ui/assets/xychartDiagram-2RQKCTM6-CGQCKCak.js +7 -0
- package/dist/web-ui/index.html +2 -2
- package/package.json +2 -1
- package/templates/roles/SHARED.md +113 -8
- package/templates/roles/ai-engineer/ROLE.md +35 -0
- package/templates/roles/ai-engineer/agent.json +1 -1
- package/templates/roles/architect/ROLE.md +15 -0
- package/templates/roles/architect/agent.json +1 -1
- package/templates/roles/content-writer/HEARTBEAT.md +29 -0
- package/templates/roles/content-writer/POLICIES.md +30 -0
- package/templates/roles/content-writer/ROLE.md +235 -19
- package/templates/roles/data-engineer/ROLE.md +29 -0
- package/templates/roles/data-engineer/agent.json +1 -1
- package/templates/roles/developer/HEARTBEAT.md +25 -7
- package/templates/roles/developer/POLICIES.md +24 -6
- package/templates/roles/developer/ROLE.md +335 -55
- package/templates/roles/devops/HEARTBEAT.md +30 -0
- package/templates/roles/devops/POLICIES.md +30 -0
- package/templates/roles/devops/ROLE.md +126 -20
- package/templates/roles/org-manager/ROLE.md +15 -0
- package/templates/roles/product-manager/POLICIES.md +29 -0
- package/templates/roles/product-manager/ROLE.md +126 -17
- package/templates/roles/project-manager/HEARTBEAT.md +30 -0
- package/templates/roles/project-manager/POLICIES.md +29 -0
- package/templates/roles/project-manager/ROLE.md +18 -0
- package/templates/roles/qa-engineer/HEARTBEAT.md +29 -0
- package/templates/roles/qa-engineer/POLICIES.md +29 -0
- package/templates/roles/qa-engineer/ROLE.md +133 -26
- package/templates/roles/research-assistant/HEARTBEAT.md +29 -0
- package/templates/roles/research-assistant/POLICIES.md +29 -0
- package/templates/roles/research-assistant/ROLE.md +310 -48
- package/templates/roles/reviewer/POLICIES.md +29 -0
- package/templates/roles/reviewer/ROLE.md +49 -0
- package/templates/roles/scrum-master/ROLE.md +6 -0
- package/templates/roles/skill-architect/HEARTBEAT.md +29 -0
- package/templates/roles/skill-architect/POLICIES.md +29 -0
- package/templates/roles/skill-architect/ROLE.md +267 -20
- package/templates/roles/sre/agent.json +1 -1
- package/templates/roles/tech-writer/HEARTBEAT.md +29 -0
- package/templates/roles/tech-writer/POLICIES.md +28 -0
- package/templates/roles/tech-writer/ROLE.md +258 -21
- package/templates/skills/claude-code/SKILL.md +239 -0
- package/templates/skills/claude-code/skill.json +17 -0
- package/templates/skills/codex/SKILL.md +217 -0
- package/templates/skills/codex/skill.json +17 -0
- package/templates/skills/coding-tools/SKILL.md +300 -0
- package/templates/skills/coding-tools/skill.json +17 -0
- package/templates/skills/cursor-agent/SKILL.md +262 -0
- package/templates/skills/cursor-agent/skill.json +17 -0
- package/templates/skills/feishu-interaction/SKILL.md +103 -0
- package/templates/skills/feishu-interaction/skill.json +26 -0
- package/templates/skills/self-evolution/SKILL.md +31 -0
- package/templates/teams/content-team/NORMS.md +17 -0
- package/templates/teams/dev-squad/NORMS.md +26 -0
- package/templates/teams/dev-squad/team.json +4 -4
- package/templates/teams/engineering-pod/NORMS.md +33 -0
- package/templates/teams/engineering-pod/team.json +4 -4
- package/templates/teams/research-lab/NORMS.md +15 -0
- package/templates/teams/startup-team/NORMS.md +17 -0
- package/templates/teams/startup-team/team.json +1 -1
- package/dist/web-ui/assets/index-CZL1VHgy.css +0 -1
- package/dist/web-ui/assets/index-DZjXJ0HZ.js +0 -724
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: coding-tools
|
|
3
|
+
description: Use external coding tools (Claude Code, Codex, Cursor) via invoke_coding_tool and coding_tool_apply to implement, debug, and refactor code
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Coding Tools
|
|
7
|
+
|
|
8
|
+
Markus can delegate hands-on coding work to external CLI tools. You have two built-in tools for this workflow:
|
|
9
|
+
|
|
10
|
+
| Tool | Purpose |
|
|
11
|
+
|---|---|
|
|
12
|
+
| `invoke_coding_tool` | Run Claude Code, Codex, or Cursor Agent against a repository with a prompt |
|
|
13
|
+
| `coding_tool_apply` | Commit the changes produced by a coding tool session |
|
|
14
|
+
|
|
15
|
+
Use these when the task requires substantial code changes across multiple files, when you want a specialized coding agent to explore a codebase, or when direct editing would be slower or less reliable than delegating to a dedicated tool.
|
|
16
|
+
|
|
17
|
+
## Available Coding Tools
|
|
18
|
+
|
|
19
|
+
| Tool name | CLI binary | Best for |
|
|
20
|
+
|---|---|---|
|
|
21
|
+
| `claude-code` | `claude` | Complex refactors, multi-file changes, deep exploration, architecture-level edits |
|
|
22
|
+
| `codex` | `codex` | Quick fixes, targeted patches, scripted automation, fast iteration |
|
|
23
|
+
| `cursor-agent` | `cursor` | IDE-heavy work, projects with `.cursor/rules`, repo-specific conventions |
|
|
24
|
+
|
|
25
|
+
Check availability with `markus doctor` before relying on a tool. If a tool is not installed, the handler returns an `installHint`.
|
|
26
|
+
|
|
27
|
+
## When to Use External Tools vs Write Code Directly
|
|
28
|
+
|
|
29
|
+
**Use external coding tools when:**
|
|
30
|
+
|
|
31
|
+
- The change spans many files or requires broad codebase exploration
|
|
32
|
+
- You need an autonomous agent to iterate (edit → test → fix) inside the repo
|
|
33
|
+
- The task is well-scoped with clear acceptance criteria but large implementation surface
|
|
34
|
+
- You are coordinating work and want a specialist tool to execute while you review results
|
|
35
|
+
|
|
36
|
+
**Write code directly (file_edit, shell_execute) when:**
|
|
37
|
+
|
|
38
|
+
- The change is small — a few lines in one or two files
|
|
39
|
+
- You already know exactly what to change and where
|
|
40
|
+
- You need fine-grained control over every edit
|
|
41
|
+
- The task is configuration, documentation, or non-code work
|
|
42
|
+
|
|
43
|
+
**Rule of thumb:** If you would open an IDE and spend 15+ minutes navigating and editing, prefer `invoke_coding_tool`. If you can describe the exact diff in one paragraph, edit directly.
|
|
44
|
+
|
|
45
|
+
## How to Invoke a Coding Tool
|
|
46
|
+
|
|
47
|
+
```
|
|
48
|
+
invoke_coding_tool({
|
|
49
|
+
tool: "claude-code", // or "codex" | "cursor-agent"
|
|
50
|
+
prompt: "<clear instruction>",
|
|
51
|
+
workdir: "/absolute/path/to/repo",
|
|
52
|
+
task_id: "<optional-task-id>" // injects task context when provided
|
|
53
|
+
})
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
### Prompt Best Practices
|
|
57
|
+
|
|
58
|
+
Write prompts as if briefing a senior engineer who has never seen the ticket:
|
|
59
|
+
|
|
60
|
+
1. **Goal** — What outcome is required? Link to acceptance criteria.
|
|
61
|
+
2. **Scope** — Which directories, modules, or files are in/out of scope.
|
|
62
|
+
3. **Constraints** — Coding standards, test requirements, patterns to follow or avoid.
|
|
63
|
+
4. **Verification** — How to confirm success (tests to run, commands, expected behavior).
|
|
64
|
+
5. **Context** — Upstream dependencies, related PRs, or prior attempts.
|
|
65
|
+
|
|
66
|
+
Keep prompts focused. One coding tool invocation = one coherent unit of work. Split large tasks into sequential invocations rather than one mega-prompt.
|
|
67
|
+
|
|
68
|
+
### Task Context Injection
|
|
69
|
+
|
|
70
|
+
When you pass `task_id`, Markus fetches full task context (requirement, project, upstream/downstream dependencies) and injects it into the tool's working directory:
|
|
71
|
+
|
|
72
|
+
- **Claude Code** → `CLAUDE.md`
|
|
73
|
+
- **Cursor Agent** → `.cursor/rules/markus-task.mdc`
|
|
74
|
+
- **Codex** → `.agent_context/task_context.md`
|
|
75
|
+
|
|
76
|
+
The injected context also includes progress-reporting instructions for the Markus CLI. Always pass `task_id` when working on an assigned task.
|
|
77
|
+
|
|
78
|
+
### Reading Results
|
|
79
|
+
|
|
80
|
+
The tool returns JSON:
|
|
81
|
+
|
|
82
|
+
```json
|
|
83
|
+
{
|
|
84
|
+
"status": "success",
|
|
85
|
+
"sessionId": "...",
|
|
86
|
+
"tool": "claude-code",
|
|
87
|
+
"result": {
|
|
88
|
+
"success": true,
|
|
89
|
+
"summary": "...",
|
|
90
|
+
"diffStats": { "filesChanged": 3, "additions": 42, "deletions": 7 },
|
|
91
|
+
"modifiedFiles": ["src/foo.ts"],
|
|
92
|
+
"testResult": { "passed": 10, "failed": 0, "success": true }
|
|
93
|
+
},
|
|
94
|
+
"cost": { "estimatedCostUsd": 0.12, "inputTokens": 5000, "outputTokens": 1200 }
|
|
95
|
+
}
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
**Always review before applying:**
|
|
99
|
+
|
|
100
|
+
1. Read `result.summary` and `modifiedFiles`
|
|
101
|
+
2. Check `result.testResult` if present
|
|
102
|
+
3. If needed, inspect the repo with `git diff` via `shell_execute` in `workdir`
|
|
103
|
+
4. If results are incomplete, iterate with a follow-up `invoke_coding_tool` call referencing what still needs fixing
|
|
104
|
+
|
|
105
|
+
## Applying Changes
|
|
106
|
+
|
|
107
|
+
After reviewing and approving the tool's work:
|
|
108
|
+
|
|
109
|
+
```
|
|
110
|
+
coding_tool_apply({
|
|
111
|
+
session_id: "<sessionId from invoke_coding_tool>",
|
|
112
|
+
workdir: "/absolute/path/to/repo",
|
|
113
|
+
commit_message: "feat: implement user auth middleware"
|
|
114
|
+
})
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
This stages all changes and creates a git commit. If there are no changes, it returns success with `filesChanged: 0`.
|
|
118
|
+
|
|
119
|
+
**Do not apply blindly.** Verify tests pass and the diff matches expectations first.
|
|
120
|
+
|
|
121
|
+
## Choosing the Right Tool
|
|
122
|
+
|
|
123
|
+
| Scenario | Recommended tool | Why |
|
|
124
|
+
|---|---|---|
|
|
125
|
+
| Large refactor across packages | `claude-code` | Strong multi-turn reasoning, `--max-turns 50`, stream-json progress |
|
|
126
|
+
| One-file bug fix or typo | `codex` | Fast, `exec --full-auto`, minimal overhead |
|
|
127
|
+
| Repo with `.cursor/rules` | `cursor-agent` | Reads project rules natively |
|
|
128
|
+
| Need cost/token visibility | `claude-code` | Reports tokens and USD in stream-json result events |
|
|
129
|
+
| Long-running exploratory task | `claude-code` | Best at sustained codebase navigation |
|
|
130
|
+
| CI/automation-friendly run | `codex` | Designed for non-interactive full-auto mode |
|
|
131
|
+
|
|
132
|
+
When unsure, start with `claude-code` for complexity and `codex` for speed. Switch tools if the first attempt stalls or produces poor results.
|
|
133
|
+
|
|
134
|
+
## Reporting Progress via Markus CLI
|
|
135
|
+
|
|
136
|
+
Keep the task board updated while coding tools run. Use `shell_execute`:
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
markus task progress <task-id> -t "Claude Code implementing auth middleware" --percent 40
|
|
140
|
+
markus task note <task-id> -t "Coding tool modified 3 files, running tests"
|
|
141
|
+
markus task context <task-id> # refresh full context if needed
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
Report progress at these milestones:
|
|
145
|
+
|
|
146
|
+
- Before invoking a coding tool (what you're delegating)
|
|
147
|
+
- After the tool completes (summary + file count)
|
|
148
|
+
- After applying changes (commit created)
|
|
149
|
+
- On failure (error details + retry plan)
|
|
150
|
+
|
|
151
|
+
Coding tools also receive these CLI instructions in their injected context file when `task_id` is provided.
|
|
152
|
+
|
|
153
|
+
## Error Handling and Retry Strategies
|
|
154
|
+
|
|
155
|
+
### Tool Not Installed
|
|
156
|
+
|
|
157
|
+
```json
|
|
158
|
+
{ "error": "Claude Code is not installed. npm install -g @anthropic-ai/claude-code", "installHint": "..." }
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
**Action:** Try a different installed tool, or report the blocker via `task note` and escalate.
|
|
162
|
+
|
|
163
|
+
### Execution Failed or Incomplete
|
|
164
|
+
|
|
165
|
+
1. Read `result.error` and `result.rawOutput` (truncated)
|
|
166
|
+
2. Check whether partial changes exist in `workdir` (`git status`, `git diff`)
|
|
167
|
+
3. Retry with a **narrower prompt** that references the failure:
|
|
168
|
+
- "The previous attempt failed because X. Fix only Y in file Z."
|
|
169
|
+
4. After 2 failed attempts with the same tool, switch to a different tool or fall back to direct editing
|
|
170
|
+
|
|
171
|
+
### Timeout
|
|
172
|
+
|
|
173
|
+
Long tasks may hit configured `timeoutMs`. Split the work into smaller invocations with explicit checkpoints.
|
|
174
|
+
|
|
175
|
+
### Apply Failures
|
|
176
|
+
|
|
177
|
+
If `coding_tool_apply` fails (merge conflict, git error):
|
|
178
|
+
|
|
179
|
+
1. Inspect `git status` in `workdir`
|
|
180
|
+
2. Resolve conflicts manually with `file_edit`
|
|
181
|
+
3. Commit manually via `shell_execute` if needed
|
|
182
|
+
4. Document what happened in a task note
|
|
183
|
+
|
|
184
|
+
### Quality Verification
|
|
185
|
+
|
|
186
|
+
When `result.testResult` shows failures, **do not apply**. Re-invoke with:
|
|
187
|
+
|
|
188
|
+
```
|
|
189
|
+
The following tests failed: <output>. Fix the failures without changing unrelated code.
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
## Pre-Apply Quality Protocol
|
|
193
|
+
|
|
194
|
+
Before calling `coding_tool_apply`, verify ALL of the following:
|
|
195
|
+
|
|
196
|
+
1. **Tests pass**: `result.testResult.success` is true, or you have explicitly accepted known failures with justification
|
|
197
|
+
2. **Diff is clean**: Review `result.modifiedFiles` — no unexpected files were changed
|
|
198
|
+
3. **Scope compliance**: Changes are within the task scope — flag any out-of-scope modifications
|
|
199
|
+
4. **No regressions**: If a broader test suite exists, run it via `shell_execute` before applying
|
|
200
|
+
|
|
201
|
+
If any check fails, re-invoke with a targeted prompt referencing the specific issue rather than applying and fixing later.
|
|
202
|
+
|
|
203
|
+
## Workflow Summary
|
|
204
|
+
|
|
205
|
+
```
|
|
206
|
+
1. Scope the work → write a clear prompt
|
|
207
|
+
2. invoke_coding_tool({ tool, prompt, workdir, task_id })
|
|
208
|
+
3. Review result (summary, diff, tests)
|
|
209
|
+
4. Iterate if needed (step 2 with refined prompt)
|
|
210
|
+
5. coding_tool_apply({ session_id, workdir, commit_message })
|
|
211
|
+
6. Report progress via markus task progress/note
|
|
212
|
+
7. Submit task for review
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
## Model, Mode, and Effort Overrides
|
|
216
|
+
|
|
217
|
+
The `invoke_coding_tool` handler accepts per-invocation overrides:
|
|
218
|
+
|
|
219
|
+
```
|
|
220
|
+
invoke_coding_tool({
|
|
221
|
+
tool: "claude-code",
|
|
222
|
+
prompt: "...",
|
|
223
|
+
model: "opus", // optional: override the user's default model
|
|
224
|
+
mode: "plan", // optional: tool-specific execution mode
|
|
225
|
+
effort: "high", // optional: reasoning effort level
|
|
226
|
+
approved: true // required when approvalRequired is enabled for this tool
|
|
227
|
+
})
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
### Model selection strategy
|
|
231
|
+
|
|
232
|
+
- If the user has set a `defaultModel` in settings, respect it unless you have a specific reason to override
|
|
233
|
+
- When overriding, explain why in your progress notes (e.g., "Using Opus for complex architecture task")
|
|
234
|
+
- For simple tasks, prefer cheaper models; for complex multi-file work, consider more capable models
|
|
235
|
+
|
|
236
|
+
### Mode chaining pattern
|
|
237
|
+
|
|
238
|
+
Many tasks benefit from a multi-step approach:
|
|
239
|
+
|
|
240
|
+
1. **Plan first** — `invoke_coding_tool({ tool: "claude-code", mode: "plan", prompt: "Analyze the codebase and create an implementation plan for..." })` or `invoke_coding_tool({ tool: "cursor-agent", mode: "plan", ... })`
|
|
241
|
+
2. **Execute** — Follow up with the default agent mode (no `mode` override) using the plan output as context
|
|
242
|
+
3. **Review/ask** — If results need refinement, use `mode: "ask"` (Cursor) for targeted questions
|
|
243
|
+
|
|
244
|
+
### Approval workflow
|
|
245
|
+
|
|
246
|
+
If a tool has `approvalRequired: true`, the handler returns an `approval_required` error. You must:
|
|
247
|
+
|
|
248
|
+
1. Call `request_user_approval` explaining what you want to do and why
|
|
249
|
+
2. After receiving approval, retry with `approved: true`
|
|
250
|
+
|
|
251
|
+
**Voluntarily request approval** (even without `approvalRequired`) when:
|
|
252
|
+
|
|
253
|
+
- You're about to use an expensive model (Opus, gpt-5.5) for a long-running task
|
|
254
|
+
- The task scope is unclear and might consume significant resources
|
|
255
|
+
- The user's cost profile suggests caution (check the tool-specific skill for guidance)
|
|
256
|
+
|
|
257
|
+
## Cost-Aware Practices
|
|
258
|
+
|
|
259
|
+
Cost awareness is your responsibility as the agent. The UI provides enforceable controls (default model, budget cap for Claude Code, approval toggle), but strategic cost optimization is up to you.
|
|
260
|
+
|
|
261
|
+
### General principles
|
|
262
|
+
|
|
263
|
+
- **Start cheap, escalate if needed** — Try the default or cheaper model first. Only reach for expensive models when the cheaper attempt fails or the task clearly requires deep reasoning
|
|
264
|
+
- **Scope tightly** — One focused invocation is cheaper than a broad one that explores unnecessarily
|
|
265
|
+
- **Monitor results** — After each invocation, check `cost.estimatedCostUsd` (if available) and note it in progress updates
|
|
266
|
+
- **Split rather than retry** — If a task partially completes, split the remainder into a new focused invocation rather than re-running the entire thing
|
|
267
|
+
|
|
268
|
+
### Cost visibility varies by tool
|
|
269
|
+
|
|
270
|
+
| Tool | Cost data available? |
|
|
271
|
+
|---|---|
|
|
272
|
+
| `claude-code` | Yes — tokens and USD in every result |
|
|
273
|
+
| `codex` | Limited — no structured cost data from CLI |
|
|
274
|
+
| `cursor-agent` | Limited — may include tokens in result events |
|
|
275
|
+
|
|
276
|
+
When cost data is unavailable, estimate by task duration and complexity. Report what you know.
|
|
277
|
+
|
|
278
|
+
## Related Skills
|
|
279
|
+
|
|
280
|
+
For tool-specific details, activate the matching skill:
|
|
281
|
+
|
|
282
|
+
- **claude-code** — CLI flags, stream-json, CLAUDE.md, cost tracking, model/effort guidance
|
|
283
|
+
- **codex** — full-auto mode, AGENTS.md, sandbox behavior, model/effort guidance
|
|
284
|
+
- **cursor-agent** — agent mode, `.cursor/rules`, working directory setup, Auto vs Max Mode
|
|
285
|
+
|
|
286
|
+
Use `discover_tools({ name: ["claude-code"] })` to load a tool-specific skill before invoking that tool.
|
|
287
|
+
|
|
288
|
+
## Rules
|
|
289
|
+
|
|
290
|
+
- **DO** pass `task_id` for assigned tasks
|
|
291
|
+
- **DO** review diffs and test results before applying
|
|
292
|
+
- **DO** split large work into focused invocations
|
|
293
|
+
- **DO** report progress to the task board
|
|
294
|
+
- **DO** respect the user's `defaultModel` setting unless you have a clear reason to override
|
|
295
|
+
- **DO** voluntarily request user approval before expensive operations
|
|
296
|
+
- **DO NOT** apply changes when tests fail unless explicitly acceptable
|
|
297
|
+
- **DO NOT** run coding tools outside the project's repository path
|
|
298
|
+
- **DO NOT** retry the same broad prompt more than twice — refine or switch tools
|
|
299
|
+
- **DO NOT** default to expensive models when cheaper ones can handle the task
|
|
300
|
+
- **NEVER** call coding tool CLIs (`cursor`, `claude`, `codex`) directly via `shell_execute`. Always use `invoke_coding_tool` — it handles binary resolution, argument building, context injection, streaming, cost tracking, and session management. Direct shell calls bypass all of this and will likely use wrong arguments.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
{
|
|
2
|
+
"type": "skill",
|
|
3
|
+
"name": "coding-tools",
|
|
4
|
+
"displayName": "Coding Tools",
|
|
5
|
+
"version": "1.0.0",
|
|
6
|
+
"description": "Use external coding tools (Claude Code, Codex, Cursor) to implement, debug, and refactor code",
|
|
7
|
+
"author": "markus",
|
|
8
|
+
"category": "development",
|
|
9
|
+
"tags": ["coding", "tools", "claude-code", "codex", "cursor", "programming"],
|
|
10
|
+
"i18n": {
|
|
11
|
+
"zh-CN": {
|
|
12
|
+
"displayName": "编程工具",
|
|
13
|
+
"description": "使用外部编程工具(Claude Code、Codex、Cursor)来实现、调试和重构代码"
|
|
14
|
+
}
|
|
15
|
+
},
|
|
16
|
+
"skill": { "skillFile": "SKILL.md" }
|
|
17
|
+
}
|
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: cursor-agent
|
|
3
|
+
description: Use the Cursor CLI agent mode for IDE-integrated coding with project rules and repo-aware edits
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Cursor Agent
|
|
7
|
+
|
|
8
|
+
Cursor Agent (`cursor` binary) runs the Cursor IDE's agent mode from the command line. Markus invokes it via `invoke_coding_tool({ tool: "cursor-agent", ... })`. Use it when the project has Cursor-specific configuration (`.cursor/rules`) or when IDE-integrated, project-aware editing is the best fit.
|
|
9
|
+
|
|
10
|
+
## Installation
|
|
11
|
+
|
|
12
|
+
Install Cursor from [https://cursor.sh](https://cursor.sh), then enable the shell command:
|
|
13
|
+
|
|
14
|
+
1. Open Cursor → Command Palette → "Install 'cursor' command in PATH"
|
|
15
|
+
2. Verify: `cursor --version`
|
|
16
|
+
|
|
17
|
+
Check availability with `markus doctor`. Requires authentication via `cursor agent login` (browser-based) or `CURSOR_API_KEY` env var.
|
|
18
|
+
|
|
19
|
+
## How Markus Invokes Cursor Agent
|
|
20
|
+
|
|
21
|
+
Markus runs Cursor in CLI agent mode:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
cursor agent --print --output-format stream-json --workspace <workdir> --trust --force "<prompt>"
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
| Flag | Purpose |
|
|
28
|
+
|---|---|
|
|
29
|
+
| `--print` | Non-interactive mode |
|
|
30
|
+
| `--output-format stream-json` | Emits structured JSON events for progress parsing |
|
|
31
|
+
| `--workspace <workdir>` | Specifies the repo root for project context |
|
|
32
|
+
| `--trust --force` | Auto-trusts the workspace without interactive prompts |
|
|
33
|
+
|
|
34
|
+
Additional args can be configured via `CodingToolConfig.defaultArgs`.
|
|
35
|
+
|
|
36
|
+
## .cursor/rules Context Files
|
|
37
|
+
|
|
38
|
+
Cursor projects use **`.cursor/rules/`** for persistent agent instructions. Rule files (`.mdc` or `.md`) define:
|
|
39
|
+
|
|
40
|
+
- Coding standards and naming conventions
|
|
41
|
+
- Architecture decisions and patterns to follow
|
|
42
|
+
- Test requirements and deployment procedures
|
|
43
|
+
- Files or areas agents should not modify
|
|
44
|
+
|
|
45
|
+
Cursor Agent reads these rules automatically when operating in a project directory. Well-maintained rules significantly improve output quality.
|
|
46
|
+
|
|
47
|
+
### Markus Task Context Injection
|
|
48
|
+
|
|
49
|
+
When you pass `task_id` to `invoke_coding_tool`, Markus writes task-specific context to:
|
|
50
|
+
|
|
51
|
+
```
|
|
52
|
+
.cursor/rules/markus-task.mdc
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
This file contains the full Markus task context (title, description, subtasks, dependencies, progress-reporting CLI commands). It sits alongside your project's permanent rules and takes effect for the coding session.
|
|
56
|
+
|
|
57
|
+
**Important:**
|
|
58
|
+
|
|
59
|
+
- Markus creates the `.cursor/rules/` directory if it doesn't exist
|
|
60
|
+
- The `markus-task.mdc` file is regenerated each invocation — do not store permanent rules there
|
|
61
|
+
- Keep long-lived project rules in separate files (e.g., `coding-standards.mdc`, `architecture.mdc`)
|
|
62
|
+
|
|
63
|
+
### Example Project Rules Structure
|
|
64
|
+
|
|
65
|
+
```
|
|
66
|
+
.cursor/rules/
|
|
67
|
+
├── coding-standards.mdc # permanent — naming, formatting, test requirements
|
|
68
|
+
├── architecture.mdc # permanent — module boundaries, patterns
|
|
69
|
+
└── markus-task.mdc # ephemeral — injected by Markus per task
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Working Directory Considerations
|
|
73
|
+
|
|
74
|
+
The `workdir` parameter is critical for Cursor Agent:
|
|
75
|
+
|
|
76
|
+
### Use the Repository Root
|
|
77
|
+
|
|
78
|
+
Always pass the **absolute path to the git repository root** as `workdir`:
|
|
79
|
+
|
|
80
|
+
```
|
|
81
|
+
workdir: "/Users/agent/projects/my-app"
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Cursor Agent resolves `.cursor/rules/` relative to this directory. Pointing at a subdirectory may cause rules to be missed.
|
|
85
|
+
|
|
86
|
+
### Multi-Repo Projects
|
|
87
|
+
|
|
88
|
+
If the task spans multiple repositories, invoke Cursor Agent **once per repository** with repo-specific prompts:
|
|
89
|
+
|
|
90
|
+
```
|
|
91
|
+
invoke_coding_tool({
|
|
92
|
+
tool: "cursor-agent",
|
|
93
|
+
prompt: "Implement the API client changes described in the task context.",
|
|
94
|
+
workdir: "/path/to/backend-repo",
|
|
95
|
+
task_id: "task-789"
|
|
96
|
+
})
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
### Git State
|
|
100
|
+
|
|
101
|
+
Ensure the working directory is a clean git checkout (or at least understand existing uncommitted changes). Cursor Agent edits files in-place. Review with `git status` and `git diff` before applying.
|
|
102
|
+
|
|
103
|
+
### Environment Variables
|
|
104
|
+
|
|
105
|
+
Markus injects these env vars during context injection:
|
|
106
|
+
|
|
107
|
+
| Variable | Purpose |
|
|
108
|
+
|---|---|
|
|
109
|
+
| `MARKUS_API_URL` | API server for task operations |
|
|
110
|
+
| `MARKUS_TASK_ID` | Current task ID |
|
|
111
|
+
| `MARKUS_CLI` | Path to the markus CLI binary |
|
|
112
|
+
|
|
113
|
+
Cursor Agent subprocesses can use these if the injected context references CLI commands for progress reporting.
|
|
114
|
+
|
|
115
|
+
## Usage Patterns
|
|
116
|
+
|
|
117
|
+
### Rule-Driven Feature Implementation
|
|
118
|
+
|
|
119
|
+
```
|
|
120
|
+
invoke_coding_tool({
|
|
121
|
+
tool: "cursor-agent",
|
|
122
|
+
prompt: "Implement the user settings page following the patterns in .cursor/rules/frontend.mdc. Use existing component library imports. Add unit tests.",
|
|
123
|
+
workdir: "/path/to/frontend-repo",
|
|
124
|
+
task_id: "task-789"
|
|
125
|
+
})
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
### Convention-Heavy Refactor
|
|
129
|
+
|
|
130
|
+
```
|
|
131
|
+
invoke_coding_tool({
|
|
132
|
+
tool: "cursor-agent",
|
|
133
|
+
prompt: "Migrate all API routes in src/routes/ to the new error-handling pattern defined in .cursor/rules/api-conventions.mdc. Update tests accordingly.",
|
|
134
|
+
workdir: "/path/to/repo",
|
|
135
|
+
task_id: "task-789"
|
|
136
|
+
})
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
### Monorepo Package Work
|
|
140
|
+
|
|
141
|
+
```
|
|
142
|
+
invoke_coding_tool({
|
|
143
|
+
tool: "cursor-agent",
|
|
144
|
+
prompt: "Add the coding-tools skill templates under templates/skills/. Follow the existing skill.json + SKILL.md pattern from templates/skills/self-evolution/.",
|
|
145
|
+
workdir: "/path/to/markus-monorepo",
|
|
146
|
+
task_id: "task-789"
|
|
147
|
+
})
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
## Model and Mode Selection
|
|
151
|
+
|
|
152
|
+
Cursor Agent supports per-invocation model and mode overrides:
|
|
153
|
+
|
|
154
|
+
```
|
|
155
|
+
invoke_coding_tool({
|
|
156
|
+
tool: "cursor-agent",
|
|
157
|
+
prompt: "...",
|
|
158
|
+
model: "auto", // optional: CLI-specific model (auto, composer-2.5, etc.)
|
|
159
|
+
mode: "plan", // optional: plan | ask
|
|
160
|
+
})
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
### Important: CLI Model Limitations
|
|
164
|
+
|
|
165
|
+
The Cursor `agent` CLI only accepts a limited set of CLI-specific model identifiers: `auto`, `composer-2.5`, etc. **Cursor Cloud API models** (Claude, GPT families) **are not available via the CLI** — they only work via the REST API.
|
|
166
|
+
|
|
167
|
+
The Settings UI only shows CLI-compatible models to prevent selection errors.
|
|
168
|
+
|
|
169
|
+
### Critical cost fact: Auto Mode vs Max Mode
|
|
170
|
+
|
|
171
|
+
**Not specifying `--model` uses Auto mode** — this is unlimited on paid Cursor plans and very cheap (or free). The moment you specify a model name, Cursor enters **Max Mode**, which bills per-token from the monthly credit pool.
|
|
172
|
+
|
|
173
|
+
| Mode | Cost | When to use |
|
|
174
|
+
|---|---|---|
|
|
175
|
+
| Auto (no `--model`) | Free / unlimited on paid plans | Default for most tasks |
|
|
176
|
+
| Max Mode (explicit `--model`) | Per-token billing, varies by model | Only when task needs specific model capabilities |
|
|
177
|
+
|
|
178
|
+
**Warning:** Opus in Max Mode consumes credits approximately 20x faster than Auto mode. **Only specify `--model` when the task genuinely needs a specific model's capabilities.** For most tasks, Auto mode is the correct and cost-effective choice.
|
|
179
|
+
|
|
180
|
+
### Mode guidance
|
|
181
|
+
|
|
182
|
+
- `plan` — Analyze architecture first, don't make changes yet. Good as a first step for complex tasks.
|
|
183
|
+
- `ask` — Answer questions about the codebase without making changes.
|
|
184
|
+
- Default (no mode) — Full agent mode, makes edits. Use for implementation.
|
|
185
|
+
|
|
186
|
+
### Mode chaining pattern
|
|
187
|
+
|
|
188
|
+
For complex tasks:
|
|
189
|
+
|
|
190
|
+
1. `mode: "plan"` — "Analyze the codebase and propose an implementation plan for..."
|
|
191
|
+
2. Default agent mode — "Implement the plan: [paste plan from step 1]"
|
|
192
|
+
3. `mode: "ask"` — "Review the changes and identify any issues"
|
|
193
|
+
|
|
194
|
+
### Cost strategy
|
|
195
|
+
|
|
196
|
+
- **Default to no model override** (Auto mode) — it handles most tasks well and is free/cheap
|
|
197
|
+
- Only specify a model when Auto mode produces poor results or the task clearly needs specific capabilities
|
|
198
|
+
- **Voluntarily call `request_user_approval` before specifying expensive models** (Opus) since it enables Max Mode billing
|
|
199
|
+
|
|
200
|
+
## When to Choose Cursor Agent
|
|
201
|
+
|
|
202
|
+
| Choose Cursor Agent | Choose something else |
|
|
203
|
+
|---|---|
|
|
204
|
+
| Project has rich `.cursor/rules/` | No Cursor config, generic repo → `claude-code` |
|
|
205
|
+
| Team standardized on Cursor conventions | Need stream-json cost tracking → `claude-code` |
|
|
206
|
+
| IDE-style project-aware edits | Quick one-file fix → `codex` |
|
|
207
|
+
| Frontend/UI work with component rules | Deep backend exploration → `claude-code` |
|
|
208
|
+
|
|
209
|
+
## Error Handling
|
|
210
|
+
|
|
211
|
+
Cursor Agent may include some token data in result events. Evaluate results by:
|
|
212
|
+
|
|
213
|
+
1. `result.success` and `result.summary`
|
|
214
|
+
2. `result.modifiedFiles` — do they match expected scope?
|
|
215
|
+
3. `result.testResult` from quality verification
|
|
216
|
+
4. Manual `git diff` review in `workdir`
|
|
217
|
+
|
|
218
|
+
If Cursor Agent ignores project rules, verify:
|
|
219
|
+
|
|
220
|
+
- `workdir` points to the repo root (where `.cursor/rules/` lives)
|
|
221
|
+
- Rule files use correct `.mdc` format and are not empty
|
|
222
|
+
- The prompt explicitly references relevant rule files
|
|
223
|
+
|
|
224
|
+
If output quality is poor after 2 attempts, switch to `claude-code` with equivalent prompt and CLAUDE.md context.
|
|
225
|
+
|
|
226
|
+
## Best Practices
|
|
227
|
+
|
|
228
|
+
- Maintain `.cursor/rules/` with clear, actionable project conventions
|
|
229
|
+
- Always pass absolute repo root as `workdir`
|
|
230
|
+
- Pass `task_id` so `markus-task.mdc` carries task context
|
|
231
|
+
- Reference specific rule files in prompts when multiple rules exist
|
|
232
|
+
- Review diffs carefully — Cursor Agent may interpret rules differently than expected
|
|
233
|
+
- **Default to Auto mode (no model override) for cost-effectiveness**
|
|
234
|
+
- Only specify models when Auto mode is insufficient
|
|
235
|
+
|
|
236
|
+
## Mode Chaining for Quality
|
|
237
|
+
|
|
238
|
+
Use Cursor Agent modes strategically:
|
|
239
|
+
1. **Plan mode**: `invoke_coding_tool({ tool: "cursor-agent", mode: "plan", prompt: "..." })` — analyze before implementing
|
|
240
|
+
2. **Agent mode** (default): Full implementation
|
|
241
|
+
3. **Ask mode**: `invoke_coding_tool({ tool: "cursor-agent", mode: "ask", prompt: "..." })` — targeted questions about the codebase
|
|
242
|
+
|
|
243
|
+
For complex tasks, always plan first. Review the plan before proceeding to implementation.
|
|
244
|
+
|
|
245
|
+
## Project Rules Integration
|
|
246
|
+
|
|
247
|
+
Cursor Agent reads `.cursor/rules/` automatically. When `task_id` is provided, Markus injects task context into `.cursor/rules/markus-task.mdc`. This means Cursor Agent automatically understands:
|
|
248
|
+
- Task requirements and acceptance criteria
|
|
249
|
+
- Project conventions and constraints
|
|
250
|
+
- Upstream/downstream dependencies
|
|
251
|
+
|
|
252
|
+
## Rules
|
|
253
|
+
|
|
254
|
+
- **DO** use when the project has Cursor rules configured
|
|
255
|
+
- **DO** pass the repository root as `workdir`
|
|
256
|
+
- **DO** keep permanent rules separate from `markus-task.mdc`
|
|
257
|
+
- **DO** default to Auto mode (no `--model`) for cost savings
|
|
258
|
+
- **DO** request approval before using expensive models in Max Mode
|
|
259
|
+
- **DO NOT** store permanent instructions in `markus-task.mdc`
|
|
260
|
+
- **DO NOT** use for repos with no Cursor configuration unless other tools are unavailable
|
|
261
|
+
- **DO NOT** apply changes without reviewing diffs against project rules
|
|
262
|
+
- **DO NOT** specify `--model` by default — it triggers Max Mode billing
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
{
|
|
2
|
+
"type": "skill",
|
|
3
|
+
"name": "cursor-agent",
|
|
4
|
+
"displayName": "Cursor Agent",
|
|
5
|
+
"version": "1.0.0",
|
|
6
|
+
"description": "Use the Cursor CLI agent mode for IDE-integrated coding, rule-driven workflows, and project-aware edits",
|
|
7
|
+
"author": "markus",
|
|
8
|
+
"category": "development",
|
|
9
|
+
"tags": ["coding", "cursor", "ide", "agent"],
|
|
10
|
+
"i18n": {
|
|
11
|
+
"zh-CN": {
|
|
12
|
+
"displayName": "Cursor Agent",
|
|
13
|
+
"description": "使用 Cursor CLI 代理模式进行 IDE 集成编程、规则驱动的工作流和项目感知编辑"
|
|
14
|
+
}
|
|
15
|
+
},
|
|
16
|
+
"skill": { "skillFile": "SKILL.md" }
|
|
17
|
+
}
|