@markus-global/cli 0.8.4 → 0.8.5-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/dist/commands/agent.js +9 -9
  2. package/dist/commands/agent.js.map +1 -1
  3. package/dist/commands/doctor.d.ts +3 -1
  4. package/dist/commands/doctor.d.ts.map +1 -1
  5. package/dist/commands/doctor.js +27 -1
  6. package/dist/commands/doctor.js.map +1 -1
  7. package/dist/commands/models.d.ts.map +1 -1
  8. package/dist/commands/models.js +6 -7
  9. package/dist/commands/models.js.map +1 -1
  10. package/dist/commands/project.d.ts +3 -0
  11. package/dist/commands/project.d.ts.map +1 -0
  12. package/dist/commands/project.js +25 -0
  13. package/dist/commands/project.js.map +1 -0
  14. package/dist/commands/requirement.d.ts +3 -0
  15. package/dist/commands/requirement.d.ts.map +1 -0
  16. package/dist/commands/requirement.js +34 -0
  17. package/dist/commands/requirement.js.map +1 -0
  18. package/dist/commands/start.d.ts.map +1 -1
  19. package/dist/commands/start.js +42 -1
  20. package/dist/commands/start.js.map +1 -1
  21. package/dist/commands/task.d.ts +3 -0
  22. package/dist/commands/task.d.ts.map +1 -0
  23. package/dist/commands/task.js +110 -0
  24. package/dist/commands/task.js.map +1 -0
  25. package/dist/index.js +8 -0
  26. package/dist/index.js.map +1 -1
  27. package/dist/markus.mjs +3770 -965
  28. package/dist/output.d.ts +3 -1
  29. package/dist/output.d.ts.map +1 -1
  30. package/dist/output.js +34 -3
  31. package/dist/output.js.map +1 -1
  32. package/dist/web-ui/assets/arc-azDa9rNQ.js +1 -0
  33. package/dist/web-ui/assets/architectureDiagram-3BPJPVTR-CWoGp8TB.js +36 -0
  34. package/dist/web-ui/assets/blockDiagram-GPEHLZMM-C2Tq3zqo.js +132 -0
  35. package/dist/web-ui/assets/c4Diagram-AAUBKEIU-C2tj98or.js +10 -0
  36. package/dist/web-ui/assets/channel-D0Q-P9rQ.js +1 -0
  37. package/dist/web-ui/assets/chunk-2J33WTMH-DMhlyS99.js +1 -0
  38. package/dist/web-ui/assets/chunk-4BX2VUAB-C8hL0QFv.js +1 -0
  39. package/dist/web-ui/assets/chunk-55IACEB6-BPCe4caz.js +1 -0
  40. package/dist/web-ui/assets/chunk-727SXJPM-C5EAjSrN.js +206 -0
  41. package/dist/web-ui/assets/chunk-AQP2D5EJ-BLSz7iPE.js +231 -0
  42. package/dist/web-ui/assets/chunk-FMBD7UC4-Sk4yLzwq.js +15 -0
  43. package/dist/web-ui/assets/chunk-ND2GUHAM-DbuQgWyn.js +1 -0
  44. package/dist/web-ui/assets/chunk-QZHKN3VN-REM6PaDE.js +1 -0
  45. package/dist/web-ui/assets/classDiagram-4FO5ZUOK-BcOPwdcC.js +1 -0
  46. package/dist/web-ui/assets/classDiagram-v2-Q7XG4LA2-BcOPwdcC.js +1 -0
  47. package/dist/web-ui/assets/cose-bilkent-S5V4N54A-Chu2Y9EC.js +1 -0
  48. package/dist/web-ui/assets/cytoscape.esm-D3_iZ_3b.js +321 -0
  49. package/dist/web-ui/assets/dagre-BM42HDAG-BGQGbUMF.js +4 -0
  50. package/dist/web-ui/assets/defaultLocale-DX6XiGOO.js +1 -0
  51. package/dist/web-ui/assets/diagram-2AECGRRQ-DnZ1SQGN.js +43 -0
  52. package/dist/web-ui/assets/diagram-5GNKFQAL-B0S37NyM.js +10 -0
  53. package/dist/web-ui/assets/diagram-KO2AKTUF-BxDwUDuY.js +3 -0
  54. package/dist/web-ui/assets/diagram-LMA3HP47-5UC8M7Iq.js +24 -0
  55. package/dist/web-ui/assets/diagram-OG6HWLK6-C-eMeMRG.js +24 -0
  56. package/dist/web-ui/assets/erDiagram-TEJ5UH35-CVWhzHKv.js +85 -0
  57. package/dist/web-ui/assets/flowDiagram-I6XJVG4X-CMG-a-kh.js +162 -0
  58. package/dist/web-ui/assets/ganttDiagram-6RSMTGT7-DbVJ6VGB.js +292 -0
  59. package/dist/web-ui/assets/gitGraphDiagram-PVQCEYII-DcCuYR-R.js +106 -0
  60. package/dist/web-ui/assets/graph--OzhPTMs.js +1 -0
  61. package/dist/web-ui/assets/index-PVrVcpcl.css +1 -0
  62. package/dist/web-ui/assets/index-zJq4U9RT.js +776 -0
  63. package/dist/web-ui/assets/infoDiagram-5YYISTIA-CaY7gJ4a.js +2 -0
  64. package/dist/web-ui/assets/init-Gi6I4Gst.js +1 -0
  65. package/dist/web-ui/assets/ishikawaDiagram-YF4QCWOH-l4_2NV1P.js +70 -0
  66. package/dist/web-ui/assets/journeyDiagram-JHISSGLW-BVeQNwa5.js +139 -0
  67. package/dist/web-ui/assets/kanban-definition-UN3LZRKU-CtvPOV3r.js +89 -0
  68. package/dist/web-ui/assets/layout-SsrduOYp.js +1 -0
  69. package/dist/web-ui/assets/linear-B0DfGdNc.js +1 -0
  70. package/dist/web-ui/assets/mermaid.core-Bz3avYM5.js +303 -0
  71. package/dist/web-ui/assets/mindmap-definition-RKZ34NQL-1X-u7gPH.js +96 -0
  72. package/dist/web-ui/assets/ordinal-Cboi1Yqb.js +1 -0
  73. package/dist/web-ui/assets/pieDiagram-4H26LBE5-BO8LpJ1H.js +30 -0
  74. package/dist/web-ui/assets/plantuml-DezRDxd4.js +357 -0
  75. package/dist/web-ui/assets/quadrantDiagram-W4KKPZXB-BBYmPM7O.js +7 -0
  76. package/dist/web-ui/assets/requirementDiagram-4Y6WPE33-CUU8gZny.js +84 -0
  77. package/dist/web-ui/assets/sankeyDiagram-5OEKKPKP-k6GjcALi.js +40 -0
  78. package/dist/web-ui/assets/sequenceDiagram-3UESZ5HK-CScOE6Nf.js +162 -0
  79. package/dist/web-ui/assets/stateDiagram-AJRCARHV-BcvHRBZl.js +1 -0
  80. package/dist/web-ui/assets/stateDiagram-v2-BHNVJYJU-p321ujvX.js +1 -0
  81. package/dist/web-ui/assets/timeline-definition-PNZ67QCA-D7tGfjR6.js +120 -0
  82. package/dist/web-ui/assets/vennDiagram-CIIHVFJN-AU7MqjmN.js +34 -0
  83. package/dist/web-ui/assets/viz-global-C_AyN6D9.js +9 -0
  84. package/dist/web-ui/assets/wardley-L42UT6IY-DKQmSXOS.js +161 -0
  85. package/dist/web-ui/assets/wardleyDiagram-YWT4CUSO-DVMv24j_.js +78 -0
  86. package/dist/web-ui/assets/xychartDiagram-2RQKCTM6-CGQCKCak.js +7 -0
  87. package/dist/web-ui/index.html +2 -2
  88. package/package.json +2 -1
  89. package/templates/roles/SHARED.md +113 -8
  90. package/templates/roles/ai-engineer/ROLE.md +35 -0
  91. package/templates/roles/ai-engineer/agent.json +1 -1
  92. package/templates/roles/architect/ROLE.md +15 -0
  93. package/templates/roles/architect/agent.json +1 -1
  94. package/templates/roles/content-writer/HEARTBEAT.md +29 -0
  95. package/templates/roles/content-writer/POLICIES.md +30 -0
  96. package/templates/roles/content-writer/ROLE.md +235 -19
  97. package/templates/roles/data-engineer/ROLE.md +29 -0
  98. package/templates/roles/data-engineer/agent.json +1 -1
  99. package/templates/roles/developer/HEARTBEAT.md +25 -7
  100. package/templates/roles/developer/POLICIES.md +24 -6
  101. package/templates/roles/developer/ROLE.md +335 -55
  102. package/templates/roles/devops/HEARTBEAT.md +30 -0
  103. package/templates/roles/devops/POLICIES.md +30 -0
  104. package/templates/roles/devops/ROLE.md +126 -20
  105. package/templates/roles/org-manager/ROLE.md +15 -0
  106. package/templates/roles/product-manager/POLICIES.md +29 -0
  107. package/templates/roles/product-manager/ROLE.md +126 -17
  108. package/templates/roles/project-manager/HEARTBEAT.md +30 -0
  109. package/templates/roles/project-manager/POLICIES.md +29 -0
  110. package/templates/roles/project-manager/ROLE.md +18 -0
  111. package/templates/roles/qa-engineer/HEARTBEAT.md +29 -0
  112. package/templates/roles/qa-engineer/POLICIES.md +29 -0
  113. package/templates/roles/qa-engineer/ROLE.md +133 -26
  114. package/templates/roles/research-assistant/HEARTBEAT.md +29 -0
  115. package/templates/roles/research-assistant/POLICIES.md +29 -0
  116. package/templates/roles/research-assistant/ROLE.md +310 -48
  117. package/templates/roles/reviewer/POLICIES.md +29 -0
  118. package/templates/roles/reviewer/ROLE.md +49 -0
  119. package/templates/roles/scrum-master/ROLE.md +6 -0
  120. package/templates/roles/skill-architect/HEARTBEAT.md +29 -0
  121. package/templates/roles/skill-architect/POLICIES.md +29 -0
  122. package/templates/roles/skill-architect/ROLE.md +267 -20
  123. package/templates/roles/sre/agent.json +1 -1
  124. package/templates/roles/tech-writer/HEARTBEAT.md +29 -0
  125. package/templates/roles/tech-writer/POLICIES.md +28 -0
  126. package/templates/roles/tech-writer/ROLE.md +258 -21
  127. package/templates/skills/claude-code/SKILL.md +239 -0
  128. package/templates/skills/claude-code/skill.json +17 -0
  129. package/templates/skills/codex/SKILL.md +217 -0
  130. package/templates/skills/codex/skill.json +17 -0
  131. package/templates/skills/coding-tools/SKILL.md +300 -0
  132. package/templates/skills/coding-tools/skill.json +17 -0
  133. package/templates/skills/cursor-agent/SKILL.md +262 -0
  134. package/templates/skills/cursor-agent/skill.json +17 -0
  135. package/templates/skills/feishu-interaction/SKILL.md +103 -0
  136. package/templates/skills/feishu-interaction/skill.json +26 -0
  137. package/templates/skills/self-evolution/SKILL.md +31 -0
  138. package/templates/teams/content-team/NORMS.md +17 -0
  139. package/templates/teams/dev-squad/NORMS.md +26 -0
  140. package/templates/teams/dev-squad/team.json +4 -4
  141. package/templates/teams/engineering-pod/NORMS.md +33 -0
  142. package/templates/teams/engineering-pod/team.json +4 -4
  143. package/templates/teams/research-lab/NORMS.md +15 -0
  144. package/templates/teams/startup-team/NORMS.md +17 -0
  145. package/templates/teams/startup-team/team.json +1 -1
  146. package/dist/web-ui/assets/index-CZL1VHgy.css +0 -1
  147. package/dist/web-ui/assets/index-DZjXJ0HZ.js +0 -724
@@ -0,0 +1,300 @@
1
+ ---
2
+ name: coding-tools
3
+ description: Use external coding tools (Claude Code, Codex, Cursor) via invoke_coding_tool and coding_tool_apply to implement, debug, and refactor code
4
+ ---
5
+
6
+ # Coding Tools
7
+
8
+ Markus can delegate hands-on coding work to external CLI tools. You have two built-in tools for this workflow:
9
+
10
+ | Tool | Purpose |
11
+ |---|---|
12
+ | `invoke_coding_tool` | Run Claude Code, Codex, or Cursor Agent against a repository with a prompt |
13
+ | `coding_tool_apply` | Commit the changes produced by a coding tool session |
14
+
15
+ Use these when the task requires substantial code changes across multiple files, when you want a specialized coding agent to explore a codebase, or when direct editing would be slower or less reliable than delegating to a dedicated tool.
16
+
17
+ ## Available Coding Tools
18
+
19
+ | Tool name | CLI binary | Best for |
20
+ |---|---|---|
21
+ | `claude-code` | `claude` | Complex refactors, multi-file changes, deep exploration, architecture-level edits |
22
+ | `codex` | `codex` | Quick fixes, targeted patches, scripted automation, fast iteration |
23
+ | `cursor-agent` | `cursor` | IDE-heavy work, projects with `.cursor/rules`, repo-specific conventions |
24
+
25
+ Check availability with `markus doctor` before relying on a tool. If a tool is not installed, the handler returns an `installHint`.
26
+
27
+ ## When to Use External Tools vs Write Code Directly
28
+
29
+ **Use external coding tools when:**
30
+
31
+ - The change spans many files or requires broad codebase exploration
32
+ - You need an autonomous agent to iterate (edit → test → fix) inside the repo
33
+ - The task is well-scoped with clear acceptance criteria but large implementation surface
34
+ - You are coordinating work and want a specialist tool to execute while you review results
35
+
36
+ **Write code directly (file_edit, shell_execute) when:**
37
+
38
+ - The change is small — a few lines in one or two files
39
+ - You already know exactly what to change and where
40
+ - You need fine-grained control over every edit
41
+ - The task is configuration, documentation, or non-code work
42
+
43
+ **Rule of thumb:** If you would open an IDE and spend 15+ minutes navigating and editing, prefer `invoke_coding_tool`. If you can describe the exact diff in one paragraph, edit directly.
44
+
45
+ ## How to Invoke a Coding Tool
46
+
47
+ ```
48
+ invoke_coding_tool({
49
+ tool: "claude-code", // or "codex" | "cursor-agent"
50
+ prompt: "<clear instruction>",
51
+ workdir: "/absolute/path/to/repo",
52
+ task_id: "<optional-task-id>" // injects task context when provided
53
+ })
54
+ ```
55
+
56
+ ### Prompt Best Practices
57
+
58
+ Write prompts as if briefing a senior engineer who has never seen the ticket:
59
+
60
+ 1. **Goal** — What outcome is required? Link to acceptance criteria.
61
+ 2. **Scope** — Which directories, modules, or files are in/out of scope.
62
+ 3. **Constraints** — Coding standards, test requirements, patterns to follow or avoid.
63
+ 4. **Verification** — How to confirm success (tests to run, commands, expected behavior).
64
+ 5. **Context** — Upstream dependencies, related PRs, or prior attempts.
65
+
66
+ Keep prompts focused. One coding tool invocation = one coherent unit of work. Split large tasks into sequential invocations rather than one mega-prompt.
67
+
68
+ ### Task Context Injection
69
+
70
+ When you pass `task_id`, Markus fetches full task context (requirement, project, upstream/downstream dependencies) and injects it into the tool's working directory:
71
+
72
+ - **Claude Code** → `CLAUDE.md`
73
+ - **Cursor Agent** → `.cursor/rules/markus-task.mdc`
74
+ - **Codex** → `.agent_context/task_context.md`
75
+
76
+ The injected context also includes progress-reporting instructions for the Markus CLI. Always pass `task_id` when working on an assigned task.
77
+
78
+ ### Reading Results
79
+
80
+ The tool returns JSON:
81
+
82
+ ```json
83
+ {
84
+ "status": "success",
85
+ "sessionId": "...",
86
+ "tool": "claude-code",
87
+ "result": {
88
+ "success": true,
89
+ "summary": "...",
90
+ "diffStats": { "filesChanged": 3, "additions": 42, "deletions": 7 },
91
+ "modifiedFiles": ["src/foo.ts"],
92
+ "testResult": { "passed": 10, "failed": 0, "success": true }
93
+ },
94
+ "cost": { "estimatedCostUsd": 0.12, "inputTokens": 5000, "outputTokens": 1200 }
95
+ }
96
+ ```
97
+
98
+ **Always review before applying:**
99
+
100
+ 1. Read `result.summary` and `modifiedFiles`
101
+ 2. Check `result.testResult` if present
102
+ 3. If needed, inspect the repo with `git diff` via `shell_execute` in `workdir`
103
+ 4. If results are incomplete, iterate with a follow-up `invoke_coding_tool` call referencing what still needs fixing
104
+
105
+ ## Applying Changes
106
+
107
+ After reviewing and approving the tool's work:
108
+
109
+ ```
110
+ coding_tool_apply({
111
+ session_id: "<sessionId from invoke_coding_tool>",
112
+ workdir: "/absolute/path/to/repo",
113
+ commit_message: "feat: implement user auth middleware"
114
+ })
115
+ ```
116
+
117
+ This stages all changes and creates a git commit. If there are no changes, it returns success with `filesChanged: 0`.
118
+
119
+ **Do not apply blindly.** Verify tests pass and the diff matches expectations first.
120
+
121
+ ## Choosing the Right Tool
122
+
123
+ | Scenario | Recommended tool | Why |
124
+ |---|---|---|
125
+ | Large refactor across packages | `claude-code` | Strong multi-turn reasoning, `--max-turns 50`, stream-json progress |
126
+ | One-file bug fix or typo | `codex` | Fast, `exec --full-auto`, minimal overhead |
127
+ | Repo with `.cursor/rules` | `cursor-agent` | Reads project rules natively |
128
+ | Need cost/token visibility | `claude-code` | Reports tokens and USD in stream-json result events |
129
+ | Long-running exploratory task | `claude-code` | Best at sustained codebase navigation |
130
+ | CI/automation-friendly run | `codex` | Designed for non-interactive full-auto mode |
131
+
132
+ When unsure, start with `claude-code` for complexity and `codex` for speed. Switch tools if the first attempt stalls or produces poor results.
133
+
134
+ ## Reporting Progress via Markus CLI
135
+
136
+ Keep the task board updated while coding tools run. Use `shell_execute`:
137
+
138
+ ```bash
139
+ markus task progress <task-id> -t "Claude Code implementing auth middleware" --percent 40
140
+ markus task note <task-id> -t "Coding tool modified 3 files, running tests"
141
+ markus task context <task-id> # refresh full context if needed
142
+ ```
143
+
144
+ Report progress at these milestones:
145
+
146
+ - Before invoking a coding tool (what you're delegating)
147
+ - After the tool completes (summary + file count)
148
+ - After applying changes (commit created)
149
+ - On failure (error details + retry plan)
150
+
151
+ Coding tools also receive these CLI instructions in their injected context file when `task_id` is provided.
152
+
153
+ ## Error Handling and Retry Strategies
154
+
155
+ ### Tool Not Installed
156
+
157
+ ```json
158
+ { "error": "Claude Code is not installed. npm install -g @anthropic-ai/claude-code", "installHint": "..." }
159
+ ```
160
+
161
+ **Action:** Try a different installed tool, or report the blocker via `task note` and escalate.
162
+
163
+ ### Execution Failed or Incomplete
164
+
165
+ 1. Read `result.error` and `result.rawOutput` (truncated)
166
+ 2. Check whether partial changes exist in `workdir` (`git status`, `git diff`)
167
+ 3. Retry with a **narrower prompt** that references the failure:
168
+ - "The previous attempt failed because X. Fix only Y in file Z."
169
+ 4. After 2 failed attempts with the same tool, switch to a different tool or fall back to direct editing
170
+
171
+ ### Timeout
172
+
173
+ Long tasks may hit configured `timeoutMs`. Split the work into smaller invocations with explicit checkpoints.
174
+
175
+ ### Apply Failures
176
+
177
+ If `coding_tool_apply` fails (merge conflict, git error):
178
+
179
+ 1. Inspect `git status` in `workdir`
180
+ 2. Resolve conflicts manually with `file_edit`
181
+ 3. Commit manually via `shell_execute` if needed
182
+ 4. Document what happened in a task note
183
+
184
+ ### Quality Verification
185
+
186
+ When `result.testResult` shows failures, **do not apply**. Re-invoke with:
187
+
188
+ ```
189
+ The following tests failed: <output>. Fix the failures without changing unrelated code.
190
+ ```
191
+
192
+ ## Pre-Apply Quality Protocol
193
+
194
+ Before calling `coding_tool_apply`, verify ALL of the following:
195
+
196
+ 1. **Tests pass**: `result.testResult.success` is true, or you have explicitly accepted known failures with justification
197
+ 2. **Diff is clean**: Review `result.modifiedFiles` — no unexpected files were changed
198
+ 3. **Scope compliance**: Changes are within the task scope — flag any out-of-scope modifications
199
+ 4. **No regressions**: If a broader test suite exists, run it via `shell_execute` before applying
200
+
201
+ If any check fails, re-invoke with a targeted prompt referencing the specific issue rather than applying and fixing later.
202
+
203
+ ## Workflow Summary
204
+
205
+ ```
206
+ 1. Scope the work → write a clear prompt
207
+ 2. invoke_coding_tool({ tool, prompt, workdir, task_id })
208
+ 3. Review result (summary, diff, tests)
209
+ 4. Iterate if needed (step 2 with refined prompt)
210
+ 5. coding_tool_apply({ session_id, workdir, commit_message })
211
+ 6. Report progress via markus task progress/note
212
+ 7. Submit task for review
213
+ ```
214
+
215
+ ## Model, Mode, and Effort Overrides
216
+
217
+ The `invoke_coding_tool` handler accepts per-invocation overrides:
218
+
219
+ ```
220
+ invoke_coding_tool({
221
+ tool: "claude-code",
222
+ prompt: "...",
223
+ model: "opus", // optional: override the user's default model
224
+ mode: "plan", // optional: tool-specific execution mode
225
+ effort: "high", // optional: reasoning effort level
226
+ approved: true // required when approvalRequired is enabled for this tool
227
+ })
228
+ ```
229
+
230
+ ### Model selection strategy
231
+
232
+ - If the user has set a `defaultModel` in settings, respect it unless you have a specific reason to override
233
+ - When overriding, explain why in your progress notes (e.g., "Using Opus for complex architecture task")
234
+ - For simple tasks, prefer cheaper models; for complex multi-file work, consider more capable models
235
+
236
+ ### Mode chaining pattern
237
+
238
+ Many tasks benefit from a multi-step approach:
239
+
240
+ 1. **Plan first** — `invoke_coding_tool({ tool: "claude-code", mode: "plan", prompt: "Analyze the codebase and create an implementation plan for..." })` or `invoke_coding_tool({ tool: "cursor-agent", mode: "plan", ... })`
241
+ 2. **Execute** — Follow up with the default agent mode (no `mode` override) using the plan output as context
242
+ 3. **Review/ask** — If results need refinement, use `mode: "ask"` (Cursor) for targeted questions
243
+
244
+ ### Approval workflow
245
+
246
+ If a tool has `approvalRequired: true`, the handler returns an `approval_required` error. You must:
247
+
248
+ 1. Call `request_user_approval` explaining what you want to do and why
249
+ 2. After receiving approval, retry with `approved: true`
250
+
251
+ **Voluntarily request approval** (even without `approvalRequired`) when:
252
+
253
+ - You're about to use an expensive model (Opus, gpt-5.5) for a long-running task
254
+ - The task scope is unclear and might consume significant resources
255
+ - The user's cost profile suggests caution (check the tool-specific skill for guidance)
256
+
257
+ ## Cost-Aware Practices
258
+
259
+ Cost awareness is your responsibility as the agent. The UI provides enforceable controls (default model, budget cap for Claude Code, approval toggle), but strategic cost optimization is up to you.
260
+
261
+ ### General principles
262
+
263
+ - **Start cheap, escalate if needed** — Try the default or cheaper model first. Only reach for expensive models when the cheaper attempt fails or the task clearly requires deep reasoning
264
+ - **Scope tightly** — One focused invocation is cheaper than a broad one that explores unnecessarily
265
+ - **Monitor results** — After each invocation, check `cost.estimatedCostUsd` (if available) and note it in progress updates
266
+ - **Split rather than retry** — If a task partially completes, split the remainder into a new focused invocation rather than re-running the entire thing
267
+
268
+ ### Cost visibility varies by tool
269
+
270
+ | Tool | Cost data available? |
271
+ |---|---|
272
+ | `claude-code` | Yes — tokens and USD in every result |
273
+ | `codex` | Limited — no structured cost data from CLI |
274
+ | `cursor-agent` | Limited — may include tokens in result events |
275
+
276
+ When cost data is unavailable, estimate by task duration and complexity. Report what you know.
277
+
278
+ ## Related Skills
279
+
280
+ For tool-specific details, activate the matching skill:
281
+
282
+ - **claude-code** — CLI flags, stream-json, CLAUDE.md, cost tracking, model/effort guidance
283
+ - **codex** — full-auto mode, AGENTS.md, sandbox behavior, model/effort guidance
284
+ - **cursor-agent** — agent mode, `.cursor/rules`, working directory setup, Auto vs Max Mode
285
+
286
+ Use `discover_tools({ name: ["claude-code"] })` to load a tool-specific skill before invoking that tool.
287
+
288
+ ## Rules
289
+
290
+ - **DO** pass `task_id` for assigned tasks
291
+ - **DO** review diffs and test results before applying
292
+ - **DO** split large work into focused invocations
293
+ - **DO** report progress to the task board
294
+ - **DO** respect the user's `defaultModel` setting unless you have a clear reason to override
295
+ - **DO** voluntarily request user approval before expensive operations
296
+ - **DO NOT** apply changes when tests fail unless explicitly acceptable
297
+ - **DO NOT** run coding tools outside the project's repository path
298
+ - **DO NOT** retry the same broad prompt more than twice — refine or switch tools
299
+ - **DO NOT** default to expensive models when cheaper ones can handle the task
300
+ - **NEVER** call coding tool CLIs (`cursor`, `claude`, `codex`) directly via `shell_execute`. Always use `invoke_coding_tool` — it handles binary resolution, argument building, context injection, streaming, cost tracking, and session management. Direct shell calls bypass all of this and will likely use wrong arguments.
@@ -0,0 +1,17 @@
1
+ {
2
+ "type": "skill",
3
+ "name": "coding-tools",
4
+ "displayName": "Coding Tools",
5
+ "version": "1.0.0",
6
+ "description": "Use external coding tools (Claude Code, Codex, Cursor) to implement, debug, and refactor code",
7
+ "author": "markus",
8
+ "category": "development",
9
+ "tags": ["coding", "tools", "claude-code", "codex", "cursor", "programming"],
10
+ "i18n": {
11
+ "zh-CN": {
12
+ "displayName": "编程工具",
13
+ "description": "使用外部编程工具(Claude Code、Codex、Cursor)来实现、调试和重构代码"
14
+ }
15
+ },
16
+ "skill": { "skillFile": "SKILL.md" }
17
+ }
@@ -0,0 +1,262 @@
1
+ ---
2
+ name: cursor-agent
3
+ description: Use the Cursor CLI agent mode for IDE-integrated coding with project rules and repo-aware edits
4
+ ---
5
+
6
+ # Cursor Agent
7
+
8
+ Cursor Agent (`cursor` binary) runs the Cursor IDE's agent mode from the command line. Markus invokes it via `invoke_coding_tool({ tool: "cursor-agent", ... })`. Use it when the project has Cursor-specific configuration (`.cursor/rules`) or when IDE-integrated, project-aware editing is the best fit.
9
+
10
+ ## Installation
11
+
12
+ Install Cursor from [https://cursor.sh](https://cursor.sh), then enable the shell command:
13
+
14
+ 1. Open Cursor → Command Palette → "Install 'cursor' command in PATH"
15
+ 2. Verify: `cursor --version`
16
+
17
+ Check availability with `markus doctor`. Requires authentication via `cursor agent login` (browser-based) or `CURSOR_API_KEY` env var.
18
+
19
+ ## How Markus Invokes Cursor Agent
20
+
21
+ Markus runs Cursor in CLI agent mode:
22
+
23
+ ```bash
24
+ cursor agent --print --output-format stream-json --workspace <workdir> --trust --force "<prompt>"
25
+ ```
26
+
27
+ | Flag | Purpose |
28
+ |---|---|
29
+ | `--print` | Non-interactive mode |
30
+ | `--output-format stream-json` | Emits structured JSON events for progress parsing |
31
+ | `--workspace <workdir>` | Specifies the repo root for project context |
32
+ | `--trust --force` | Auto-trusts the workspace without interactive prompts |
33
+
34
+ Additional args can be configured via `CodingToolConfig.defaultArgs`.
35
+
36
+ ## .cursor/rules Context Files
37
+
38
+ Cursor projects use **`.cursor/rules/`** for persistent agent instructions. Rule files (`.mdc` or `.md`) define:
39
+
40
+ - Coding standards and naming conventions
41
+ - Architecture decisions and patterns to follow
42
+ - Test requirements and deployment procedures
43
+ - Files or areas agents should not modify
44
+
45
+ Cursor Agent reads these rules automatically when operating in a project directory. Well-maintained rules significantly improve output quality.
46
+
47
+ ### Markus Task Context Injection
48
+
49
+ When you pass `task_id` to `invoke_coding_tool`, Markus writes task-specific context to:
50
+
51
+ ```
52
+ .cursor/rules/markus-task.mdc
53
+ ```
54
+
55
+ This file contains the full Markus task context (title, description, subtasks, dependencies, progress-reporting CLI commands). It sits alongside your project's permanent rules and takes effect for the coding session.
56
+
57
+ **Important:**
58
+
59
+ - Markus creates the `.cursor/rules/` directory if it doesn't exist
60
+ - The `markus-task.mdc` file is regenerated each invocation — do not store permanent rules there
61
+ - Keep long-lived project rules in separate files (e.g., `coding-standards.mdc`, `architecture.mdc`)
62
+
63
+ ### Example Project Rules Structure
64
+
65
+ ```
66
+ .cursor/rules/
67
+ ├── coding-standards.mdc # permanent — naming, formatting, test requirements
68
+ ├── architecture.mdc # permanent — module boundaries, patterns
69
+ └── markus-task.mdc # ephemeral — injected by Markus per task
70
+ ```
71
+
72
+ ## Working Directory Considerations
73
+
74
+ The `workdir` parameter is critical for Cursor Agent:
75
+
76
+ ### Use the Repository Root
77
+
78
+ Always pass the **absolute path to the git repository root** as `workdir`:
79
+
80
+ ```
81
+ workdir: "/Users/agent/projects/my-app"
82
+ ```
83
+
84
+ Cursor Agent resolves `.cursor/rules/` relative to this directory. Pointing at a subdirectory may cause rules to be missed.
85
+
86
+ ### Multi-Repo Projects
87
+
88
+ If the task spans multiple repositories, invoke Cursor Agent **once per repository** with repo-specific prompts:
89
+
90
+ ```
91
+ invoke_coding_tool({
92
+ tool: "cursor-agent",
93
+ prompt: "Implement the API client changes described in the task context.",
94
+ workdir: "/path/to/backend-repo",
95
+ task_id: "task-789"
96
+ })
97
+ ```
98
+
99
+ ### Git State
100
+
101
+ Ensure the working directory is a clean git checkout (or at least understand existing uncommitted changes). Cursor Agent edits files in-place. Review with `git status` and `git diff` before applying.
102
+
103
+ ### Environment Variables
104
+
105
+ Markus injects these env vars during context injection:
106
+
107
+ | Variable | Purpose |
108
+ |---|---|
109
+ | `MARKUS_API_URL` | API server for task operations |
110
+ | `MARKUS_TASK_ID` | Current task ID |
111
+ | `MARKUS_CLI` | Path to the markus CLI binary |
112
+
113
+ Cursor Agent subprocesses can use these if the injected context references CLI commands for progress reporting.
114
+
115
+ ## Usage Patterns
116
+
117
+ ### Rule-Driven Feature Implementation
118
+
119
+ ```
120
+ invoke_coding_tool({
121
+ tool: "cursor-agent",
122
+ prompt: "Implement the user settings page following the patterns in .cursor/rules/frontend.mdc. Use existing component library imports. Add unit tests.",
123
+ workdir: "/path/to/frontend-repo",
124
+ task_id: "task-789"
125
+ })
126
+ ```
127
+
128
+ ### Convention-Heavy Refactor
129
+
130
+ ```
131
+ invoke_coding_tool({
132
+ tool: "cursor-agent",
133
+ prompt: "Migrate all API routes in src/routes/ to the new error-handling pattern defined in .cursor/rules/api-conventions.mdc. Update tests accordingly.",
134
+ workdir: "/path/to/repo",
135
+ task_id: "task-789"
136
+ })
137
+ ```
138
+
139
+ ### Monorepo Package Work
140
+
141
+ ```
142
+ invoke_coding_tool({
143
+ tool: "cursor-agent",
144
+ prompt: "Add the coding-tools skill templates under templates/skills/. Follow the existing skill.json + SKILL.md pattern from templates/skills/self-evolution/.",
145
+ workdir: "/path/to/markus-monorepo",
146
+ task_id: "task-789"
147
+ })
148
+ ```
149
+
150
+ ## Model and Mode Selection
151
+
152
+ Cursor Agent supports per-invocation model and mode overrides:
153
+
154
+ ```
155
+ invoke_coding_tool({
156
+ tool: "cursor-agent",
157
+ prompt: "...",
158
+ model: "auto", // optional: CLI-specific model (auto, composer-2.5, etc.)
159
+ mode: "plan", // optional: plan | ask
160
+ })
161
+ ```
162
+
163
+ ### Important: CLI Model Limitations
164
+
165
+ The Cursor `agent` CLI only accepts a limited set of CLI-specific model identifiers: `auto`, `composer-2.5`, etc. **Cursor Cloud API models** (Claude, GPT families) **are not available via the CLI** — they only work via the REST API.
166
+
167
+ The Settings UI only shows CLI-compatible models to prevent selection errors.
168
+
169
+ ### Critical cost fact: Auto Mode vs Max Mode
170
+
171
+ **Not specifying `--model` uses Auto mode** — this is unlimited on paid Cursor plans and very cheap (or free). The moment you specify a model name, Cursor enters **Max Mode**, which bills per-token from the monthly credit pool.
172
+
173
+ | Mode | Cost | When to use |
174
+ |---|---|---|
175
+ | Auto (no `--model`) | Free / unlimited on paid plans | Default for most tasks |
176
+ | Max Mode (explicit `--model`) | Per-token billing, varies by model | Only when task needs specific model capabilities |
177
+
178
+ **Warning:** Opus in Max Mode consumes credits approximately 20x faster than Auto mode. **Only specify `--model` when the task genuinely needs a specific model's capabilities.** For most tasks, Auto mode is the correct and cost-effective choice.
179
+
180
+ ### Mode guidance
181
+
182
+ - `plan` — Analyze architecture first, don't make changes yet. Good as a first step for complex tasks.
183
+ - `ask` — Answer questions about the codebase without making changes.
184
+ - Default (no mode) — Full agent mode, makes edits. Use for implementation.
185
+
186
+ ### Mode chaining pattern
187
+
188
+ For complex tasks:
189
+
190
+ 1. `mode: "plan"` — "Analyze the codebase and propose an implementation plan for..."
191
+ 2. Default agent mode — "Implement the plan: [paste plan from step 1]"
192
+ 3. `mode: "ask"` — "Review the changes and identify any issues"
193
+
194
+ ### Cost strategy
195
+
196
+ - **Default to no model override** (Auto mode) — it handles most tasks well and is free/cheap
197
+ - Only specify a model when Auto mode produces poor results or the task clearly needs specific capabilities
198
+ - **Voluntarily call `request_user_approval` before specifying expensive models** (Opus) since it enables Max Mode billing
199
+
200
+ ## When to Choose Cursor Agent
201
+
202
+ | Choose Cursor Agent | Choose something else |
203
+ |---|---|
204
+ | Project has rich `.cursor/rules/` | No Cursor config, generic repo → `claude-code` |
205
+ | Team standardized on Cursor conventions | Need stream-json cost tracking → `claude-code` |
206
+ | IDE-style project-aware edits | Quick one-file fix → `codex` |
207
+ | Frontend/UI work with component rules | Deep backend exploration → `claude-code` |
208
+
209
+ ## Error Handling
210
+
211
+ Cursor Agent may include some token data in result events. Evaluate results by:
212
+
213
+ 1. `result.success` and `result.summary`
214
+ 2. `result.modifiedFiles` — do they match expected scope?
215
+ 3. `result.testResult` from quality verification
216
+ 4. Manual `git diff` review in `workdir`
217
+
218
+ If Cursor Agent ignores project rules, verify:
219
+
220
+ - `workdir` points to the repo root (where `.cursor/rules/` lives)
221
+ - Rule files use correct `.mdc` format and are not empty
222
+ - The prompt explicitly references relevant rule files
223
+
224
+ If output quality is poor after 2 attempts, switch to `claude-code` with equivalent prompt and CLAUDE.md context.
225
+
226
+ ## Best Practices
227
+
228
+ - Maintain `.cursor/rules/` with clear, actionable project conventions
229
+ - Always pass absolute repo root as `workdir`
230
+ - Pass `task_id` so `markus-task.mdc` carries task context
231
+ - Reference specific rule files in prompts when multiple rules exist
232
+ - Review diffs carefully — Cursor Agent may interpret rules differently than expected
233
+ - **Default to Auto mode (no model override) for cost-effectiveness**
234
+ - Only specify models when Auto mode is insufficient
235
+
236
+ ## Mode Chaining for Quality
237
+
238
+ Use Cursor Agent modes strategically:
239
+ 1. **Plan mode**: `invoke_coding_tool({ tool: "cursor-agent", mode: "plan", prompt: "..." })` — analyze before implementing
240
+ 2. **Agent mode** (default): Full implementation
241
+ 3. **Ask mode**: `invoke_coding_tool({ tool: "cursor-agent", mode: "ask", prompt: "..." })` — targeted questions about the codebase
242
+
243
+ For complex tasks, always plan first. Review the plan before proceeding to implementation.
244
+
245
+ ## Project Rules Integration
246
+
247
+ Cursor Agent reads `.cursor/rules/` automatically. When `task_id` is provided, Markus injects task context into `.cursor/rules/markus-task.mdc`. This means Cursor Agent automatically understands:
248
+ - Task requirements and acceptance criteria
249
+ - Project conventions and constraints
250
+ - Upstream/downstream dependencies
251
+
252
+ ## Rules
253
+
254
+ - **DO** use when the project has Cursor rules configured
255
+ - **DO** pass the repository root as `workdir`
256
+ - **DO** keep permanent rules separate from `markus-task.mdc`
257
+ - **DO** default to Auto mode (no `--model`) for cost savings
258
+ - **DO** request approval before using expensive models in Max Mode
259
+ - **DO NOT** store permanent instructions in `markus-task.mdc`
260
+ - **DO NOT** use for repos with no Cursor configuration unless other tools are unavailable
261
+ - **DO NOT** apply changes without reviewing diffs against project rules
262
+ - **DO NOT** specify `--model` by default — it triggers Max Mode billing
@@ -0,0 +1,17 @@
1
+ {
2
+ "type": "skill",
3
+ "name": "cursor-agent",
4
+ "displayName": "Cursor Agent",
5
+ "version": "1.0.0",
6
+ "description": "Use the Cursor CLI agent mode for IDE-integrated coding, rule-driven workflows, and project-aware edits",
7
+ "author": "markus",
8
+ "category": "development",
9
+ "tags": ["coding", "cursor", "ide", "agent"],
10
+ "i18n": {
11
+ "zh-CN": {
12
+ "displayName": "Cursor Agent",
13
+ "description": "使用 Cursor CLI 代理模式进行 IDE 集成编程、规则驱动的工作流和项目感知编辑"
14
+ }
15
+ },
16
+ "skill": { "skillFile": "SKILL.md" }
17
+ }