@markus-global/cli 0.8.4 → 0.8.5-rc.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/agent.js +9 -9
- package/dist/commands/agent.js.map +1 -1
- package/dist/commands/doctor.d.ts +3 -1
- package/dist/commands/doctor.d.ts.map +1 -1
- package/dist/commands/doctor.js +27 -1
- package/dist/commands/doctor.js.map +1 -1
- package/dist/commands/models.d.ts.map +1 -1
- package/dist/commands/models.js +6 -7
- package/dist/commands/models.js.map +1 -1
- package/dist/commands/project.d.ts +3 -0
- package/dist/commands/project.d.ts.map +1 -0
- package/dist/commands/project.js +25 -0
- package/dist/commands/project.js.map +1 -0
- package/dist/commands/requirement.d.ts +3 -0
- package/dist/commands/requirement.d.ts.map +1 -0
- package/dist/commands/requirement.js +34 -0
- package/dist/commands/requirement.js.map +1 -0
- package/dist/commands/start.d.ts.map +1 -1
- package/dist/commands/start.js +42 -1
- package/dist/commands/start.js.map +1 -1
- package/dist/commands/task.d.ts +3 -0
- package/dist/commands/task.d.ts.map +1 -0
- package/dist/commands/task.js +110 -0
- package/dist/commands/task.js.map +1 -0
- package/dist/index.js +8 -0
- package/dist/index.js.map +1 -1
- package/dist/markus.mjs +3770 -965
- package/dist/output.d.ts +3 -1
- package/dist/output.d.ts.map +1 -1
- package/dist/output.js +34 -3
- package/dist/output.js.map +1 -1
- package/dist/web-ui/assets/arc-azDa9rNQ.js +1 -0
- package/dist/web-ui/assets/architectureDiagram-3BPJPVTR-CWoGp8TB.js +36 -0
- package/dist/web-ui/assets/blockDiagram-GPEHLZMM-C2Tq3zqo.js +132 -0
- package/dist/web-ui/assets/c4Diagram-AAUBKEIU-C2tj98or.js +10 -0
- package/dist/web-ui/assets/channel-D0Q-P9rQ.js +1 -0
- package/dist/web-ui/assets/chunk-2J33WTMH-DMhlyS99.js +1 -0
- package/dist/web-ui/assets/chunk-4BX2VUAB-C8hL0QFv.js +1 -0
- package/dist/web-ui/assets/chunk-55IACEB6-BPCe4caz.js +1 -0
- package/dist/web-ui/assets/chunk-727SXJPM-C5EAjSrN.js +206 -0
- package/dist/web-ui/assets/chunk-AQP2D5EJ-BLSz7iPE.js +231 -0
- package/dist/web-ui/assets/chunk-FMBD7UC4-Sk4yLzwq.js +15 -0
- package/dist/web-ui/assets/chunk-ND2GUHAM-DbuQgWyn.js +1 -0
- package/dist/web-ui/assets/chunk-QZHKN3VN-REM6PaDE.js +1 -0
- package/dist/web-ui/assets/classDiagram-4FO5ZUOK-BcOPwdcC.js +1 -0
- package/dist/web-ui/assets/classDiagram-v2-Q7XG4LA2-BcOPwdcC.js +1 -0
- package/dist/web-ui/assets/cose-bilkent-S5V4N54A-Chu2Y9EC.js +1 -0
- package/dist/web-ui/assets/cytoscape.esm-D3_iZ_3b.js +321 -0
- package/dist/web-ui/assets/dagre-BM42HDAG-BGQGbUMF.js +4 -0
- package/dist/web-ui/assets/defaultLocale-DX6XiGOO.js +1 -0
- package/dist/web-ui/assets/diagram-2AECGRRQ-DnZ1SQGN.js +43 -0
- package/dist/web-ui/assets/diagram-5GNKFQAL-B0S37NyM.js +10 -0
- package/dist/web-ui/assets/diagram-KO2AKTUF-BxDwUDuY.js +3 -0
- package/dist/web-ui/assets/diagram-LMA3HP47-5UC8M7Iq.js +24 -0
- package/dist/web-ui/assets/diagram-OG6HWLK6-C-eMeMRG.js +24 -0
- package/dist/web-ui/assets/erDiagram-TEJ5UH35-CVWhzHKv.js +85 -0
- package/dist/web-ui/assets/flowDiagram-I6XJVG4X-CMG-a-kh.js +162 -0
- package/dist/web-ui/assets/ganttDiagram-6RSMTGT7-DbVJ6VGB.js +292 -0
- package/dist/web-ui/assets/gitGraphDiagram-PVQCEYII-DcCuYR-R.js +106 -0
- package/dist/web-ui/assets/graph--OzhPTMs.js +1 -0
- package/dist/web-ui/assets/index-PVrVcpcl.css +1 -0
- package/dist/web-ui/assets/index-zJq4U9RT.js +776 -0
- package/dist/web-ui/assets/infoDiagram-5YYISTIA-CaY7gJ4a.js +2 -0
- package/dist/web-ui/assets/init-Gi6I4Gst.js +1 -0
- package/dist/web-ui/assets/ishikawaDiagram-YF4QCWOH-l4_2NV1P.js +70 -0
- package/dist/web-ui/assets/journeyDiagram-JHISSGLW-BVeQNwa5.js +139 -0
- package/dist/web-ui/assets/kanban-definition-UN3LZRKU-CtvPOV3r.js +89 -0
- package/dist/web-ui/assets/layout-SsrduOYp.js +1 -0
- package/dist/web-ui/assets/linear-B0DfGdNc.js +1 -0
- package/dist/web-ui/assets/mermaid.core-Bz3avYM5.js +303 -0
- package/dist/web-ui/assets/mindmap-definition-RKZ34NQL-1X-u7gPH.js +96 -0
- package/dist/web-ui/assets/ordinal-Cboi1Yqb.js +1 -0
- package/dist/web-ui/assets/pieDiagram-4H26LBE5-BO8LpJ1H.js +30 -0
- package/dist/web-ui/assets/plantuml-DezRDxd4.js +357 -0
- package/dist/web-ui/assets/quadrantDiagram-W4KKPZXB-BBYmPM7O.js +7 -0
- package/dist/web-ui/assets/requirementDiagram-4Y6WPE33-CUU8gZny.js +84 -0
- package/dist/web-ui/assets/sankeyDiagram-5OEKKPKP-k6GjcALi.js +40 -0
- package/dist/web-ui/assets/sequenceDiagram-3UESZ5HK-CScOE6Nf.js +162 -0
- package/dist/web-ui/assets/stateDiagram-AJRCARHV-BcvHRBZl.js +1 -0
- package/dist/web-ui/assets/stateDiagram-v2-BHNVJYJU-p321ujvX.js +1 -0
- package/dist/web-ui/assets/timeline-definition-PNZ67QCA-D7tGfjR6.js +120 -0
- package/dist/web-ui/assets/vennDiagram-CIIHVFJN-AU7MqjmN.js +34 -0
- package/dist/web-ui/assets/viz-global-C_AyN6D9.js +9 -0
- package/dist/web-ui/assets/wardley-L42UT6IY-DKQmSXOS.js +161 -0
- package/dist/web-ui/assets/wardleyDiagram-YWT4CUSO-DVMv24j_.js +78 -0
- package/dist/web-ui/assets/xychartDiagram-2RQKCTM6-CGQCKCak.js +7 -0
- package/dist/web-ui/index.html +2 -2
- package/package.json +2 -1
- package/templates/roles/SHARED.md +113 -8
- package/templates/roles/ai-engineer/ROLE.md +35 -0
- package/templates/roles/ai-engineer/agent.json +1 -1
- package/templates/roles/architect/ROLE.md +15 -0
- package/templates/roles/architect/agent.json +1 -1
- package/templates/roles/content-writer/HEARTBEAT.md +29 -0
- package/templates/roles/content-writer/POLICIES.md +30 -0
- package/templates/roles/content-writer/ROLE.md +235 -19
- package/templates/roles/data-engineer/ROLE.md +29 -0
- package/templates/roles/data-engineer/agent.json +1 -1
- package/templates/roles/developer/HEARTBEAT.md +25 -7
- package/templates/roles/developer/POLICIES.md +24 -6
- package/templates/roles/developer/ROLE.md +335 -55
- package/templates/roles/devops/HEARTBEAT.md +30 -0
- package/templates/roles/devops/POLICIES.md +30 -0
- package/templates/roles/devops/ROLE.md +126 -20
- package/templates/roles/org-manager/ROLE.md +15 -0
- package/templates/roles/product-manager/POLICIES.md +29 -0
- package/templates/roles/product-manager/ROLE.md +126 -17
- package/templates/roles/project-manager/HEARTBEAT.md +30 -0
- package/templates/roles/project-manager/POLICIES.md +29 -0
- package/templates/roles/project-manager/ROLE.md +18 -0
- package/templates/roles/qa-engineer/HEARTBEAT.md +29 -0
- package/templates/roles/qa-engineer/POLICIES.md +29 -0
- package/templates/roles/qa-engineer/ROLE.md +133 -26
- package/templates/roles/research-assistant/HEARTBEAT.md +29 -0
- package/templates/roles/research-assistant/POLICIES.md +29 -0
- package/templates/roles/research-assistant/ROLE.md +310 -48
- package/templates/roles/reviewer/POLICIES.md +29 -0
- package/templates/roles/reviewer/ROLE.md +49 -0
- package/templates/roles/scrum-master/ROLE.md +6 -0
- package/templates/roles/skill-architect/HEARTBEAT.md +29 -0
- package/templates/roles/skill-architect/POLICIES.md +29 -0
- package/templates/roles/skill-architect/ROLE.md +267 -20
- package/templates/roles/sre/agent.json +1 -1
- package/templates/roles/tech-writer/HEARTBEAT.md +29 -0
- package/templates/roles/tech-writer/POLICIES.md +28 -0
- package/templates/roles/tech-writer/ROLE.md +258 -21
- package/templates/skills/claude-code/SKILL.md +239 -0
- package/templates/skills/claude-code/skill.json +17 -0
- package/templates/skills/codex/SKILL.md +217 -0
- package/templates/skills/codex/skill.json +17 -0
- package/templates/skills/coding-tools/SKILL.md +300 -0
- package/templates/skills/coding-tools/skill.json +17 -0
- package/templates/skills/cursor-agent/SKILL.md +262 -0
- package/templates/skills/cursor-agent/skill.json +17 -0
- package/templates/skills/feishu-interaction/SKILL.md +103 -0
- package/templates/skills/feishu-interaction/skill.json +26 -0
- package/templates/skills/self-evolution/SKILL.md +31 -0
- package/templates/teams/content-team/NORMS.md +17 -0
- package/templates/teams/dev-squad/NORMS.md +26 -0
- package/templates/teams/dev-squad/team.json +4 -4
- package/templates/teams/engineering-pod/NORMS.md +33 -0
- package/templates/teams/engineering-pod/team.json +4 -4
- package/templates/teams/research-lab/NORMS.md +15 -0
- package/templates/teams/startup-team/NORMS.md +17 -0
- package/templates/teams/startup-team/team.json +1 -1
- package/dist/web-ui/assets/index-CZL1VHgy.css +0 -1
- package/dist/web-ui/assets/index-DZjXJ0HZ.js +0 -724
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: claude-code
|
|
3
|
+
description: Use the Claude Code CLI for complex refactors, multi-file changes, and sustained codebase exploration
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Claude Code
|
|
7
|
+
|
|
8
|
+
Claude Code (`claude` binary) is Anthropic's agentic coding CLI. Markus invokes it via `invoke_coding_tool({ tool: "claude-code", ... })`. Use it for complex, multi-turn coding tasks where deep reasoning and broad file exploration are needed.
|
|
9
|
+
|
|
10
|
+
## Installation
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
npm install -g @anthropic-ai/claude-code
|
|
14
|
+
claude --version
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Verify with `markus doctor`. Requires authentication via one of: `ANTHROPIC_API_KEY` env var, `ANTHROPIC_BASE_URL` for custom endpoints, or interactive `claude` login.
|
|
18
|
+
|
|
19
|
+
## How Markus Invokes Claude Code
|
|
20
|
+
|
|
21
|
+
Markus runs Claude Code in non-interactive print mode with structured streaming output:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
claude --print --output-format stream-json --verbose --max-turns 50 --permission-mode bypassPermissions "<prompt>"
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
| Flag | Purpose |
|
|
28
|
+
|---|---|
|
|
29
|
+
| `--print` | Non-interactive mode — runs to completion without user input |
|
|
30
|
+
| `--output-format stream-json` | Emits structured JSON events for progress parsing |
|
|
31
|
+
| `--verbose` | Detailed progress output |
|
|
32
|
+
| `--max-turns 50` | Allows up to 50 agent turns for complex tasks |
|
|
33
|
+
| `--permission-mode bypassPermissions` | Auto-approves all file edits and commands — required because `--print` mode has no interactive stdin for approval prompts |
|
|
34
|
+
|
|
35
|
+
Additional args can be configured per-deployment via `CodingToolConfig.defaultArgs`.
|
|
36
|
+
|
|
37
|
+
## Stream-JSON Output
|
|
38
|
+
|
|
39
|
+
Each stdout line is a JSON event. Key event types:
|
|
40
|
+
|
|
41
|
+
| Event type | Meaning |
|
|
42
|
+
|---|---|
|
|
43
|
+
| `assistant` with `text` content | Progress message / reasoning summary |
|
|
44
|
+
| `assistant` with `tool_use` | File edit, shell command, or other tool invocation |
|
|
45
|
+
| `result` | Final outcome with cost and token data |
|
|
46
|
+
|
|
47
|
+
The `result` event includes:
|
|
48
|
+
|
|
49
|
+
- `result` — Final summary text
|
|
50
|
+
- `input_tokens`, `output_tokens` — Token usage
|
|
51
|
+
- `cache_read_tokens`, `cache_write_tokens` — Prompt cache stats
|
|
52
|
+
- `cost_usd` — Estimated cost in USD
|
|
53
|
+
|
|
54
|
+
Markus parses these into progress events (`file_edit`, `progress`, `completed`) and extracts cost reports automatically.
|
|
55
|
+
|
|
56
|
+
## CLAUDE.md Context File
|
|
57
|
+
|
|
58
|
+
When you pass `task_id` to `invoke_coding_tool`, Markus writes a `CLAUDE.md` file in the repository root before invoking Claude Code. This file contains:
|
|
59
|
+
|
|
60
|
+
- Task title, description, status, and priority
|
|
61
|
+
- Subtasks, notes, and deliverables
|
|
62
|
+
- Requirement and project context
|
|
63
|
+
- Upstream/downstream dependency summaries
|
|
64
|
+
- Markus CLI commands for reporting progress
|
|
65
|
+
|
|
66
|
+
Claude Code reads `CLAUDE.md` automatically as project context. **Do not delete or overwrite it** during a session — it is regenerated each invocation.
|
|
67
|
+
|
|
68
|
+
If the repo already has a permanent `CLAUDE.md`, Markus overwrites it for the session. Consider restoring project-level content after the task if needed.
|
|
69
|
+
|
|
70
|
+
## Usage Patterns
|
|
71
|
+
|
|
72
|
+
### Complex Refactor
|
|
73
|
+
|
|
74
|
+
```
|
|
75
|
+
invoke_coding_tool({
|
|
76
|
+
tool: "claude-code",
|
|
77
|
+
prompt: "Refactor the auth module to use dependency injection. Move AuthService to src/services/, update all imports, keep existing test behavior. Run the test suite when done.",
|
|
78
|
+
workdir: "/path/to/repo",
|
|
79
|
+
task_id: "task-123"
|
|
80
|
+
})
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
### Debug and Fix
|
|
84
|
+
|
|
85
|
+
```
|
|
86
|
+
invoke_coding_tool({
|
|
87
|
+
tool: "claude-code",
|
|
88
|
+
prompt: "Tests in packages/core/test/auth.test.ts are failing with 'token expired'. Find the root cause in src/auth/ and fix without changing the public API. Show which tests pass after the fix.",
|
|
89
|
+
workdir: "/path/to/repo",
|
|
90
|
+
task_id: "task-123"
|
|
91
|
+
})
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
### Explore Then Implement
|
|
95
|
+
|
|
96
|
+
For unfamiliar codebases, ask Claude Code to explore first:
|
|
97
|
+
|
|
98
|
+
```
|
|
99
|
+
"Read the codebase structure under src/coding-tools/. Then implement the feature described in the task context. Start by listing the files you plan to modify."
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## Retry Strategies for Long Tasks
|
|
103
|
+
|
|
104
|
+
Claude Code supports up to 50 turns, but long tasks can still stall or partially complete.
|
|
105
|
+
|
|
106
|
+
### If the session completes but work is incomplete
|
|
107
|
+
|
|
108
|
+
Re-invoke with explicit remaining scope:
|
|
109
|
+
|
|
110
|
+
```
|
|
111
|
+
"Previous session modified src/foo.ts and src/bar.ts but did not update tests. Complete the test coverage for the changes in src/foo.ts. Do not re-modify files that are already correct."
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
### If the session fails or times out
|
|
115
|
+
|
|
116
|
+
1. Check `git status` in `workdir` for partial changes
|
|
117
|
+
2. Either apply partial work with `coding_tool_apply` or discard with `git checkout -- .`
|
|
118
|
+
3. Retry with a **smaller scope** — one module or one feature at a time
|
|
119
|
+
|
|
120
|
+
### If Claude Code loops or over-edits
|
|
121
|
+
|
|
122
|
+
Add constraints to the prompt:
|
|
123
|
+
|
|
124
|
+
```
|
|
125
|
+
"Modify ONLY files under src/handlers/. Do not touch tests, config, or unrelated packages."
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
### Escalation after 2 retries
|
|
129
|
+
|
|
130
|
+
Switch to `codex` for a targeted fix, or edit directly with `file_edit`.
|
|
131
|
+
|
|
132
|
+
## Model and Effort Selection
|
|
133
|
+
|
|
134
|
+
Claude Code supports per-invocation model and effort overrides:
|
|
135
|
+
|
|
136
|
+
```
|
|
137
|
+
invoke_coding_tool({
|
|
138
|
+
tool: "claude-code",
|
|
139
|
+
prompt: "...",
|
|
140
|
+
model: "sonnet", // default: tool's own default (usually sonnet)
|
|
141
|
+
effort: "medium", // low | medium | high | xhigh | max
|
|
142
|
+
})
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
### Model guidance
|
|
146
|
+
|
|
147
|
+
| Model | Best for | Cost note |
|
|
148
|
+
|---|---|---|
|
|
149
|
+
| `haiku` | Simple subagent tasks, quick checks | Cheapest option |
|
|
150
|
+
| `sonnet` | Most coding work — default choice | Good balance of cost and capability |
|
|
151
|
+
| `opus` | Complex architecture, multi-file refactors, unfamiliar codebases | ~15x more expensive per token than Sonnet |
|
|
152
|
+
| `fable` | Creative or documentation tasks | Specialized |
|
|
153
|
+
|
|
154
|
+
**Strategy:** Start with `sonnet` (or the user's `defaultModel`). Only use `opus` when:
|
|
155
|
+
|
|
156
|
+
- Sonnet attempt failed or produced poor results
|
|
157
|
+
- The task involves complex reasoning across 10+ files
|
|
158
|
+
- Architecture decisions or trade-off analysis is required
|
|
159
|
+
|
|
160
|
+
**Warning:** Opus is approximately 15x more expensive than Sonnet per token. A task that costs $0.30 with Sonnet could cost $4.50 with Opus. **Voluntarily call `request_user_approval` before using Opus for any task expected to run more than a few turns.**
|
|
161
|
+
|
|
162
|
+
### Effort guidance
|
|
163
|
+
|
|
164
|
+
- `low` — Simple edits, typo fixes, config changes
|
|
165
|
+
- `medium` — Standard development tasks (default)
|
|
166
|
+
- `high` — Complex reasoning, multi-step problem solving
|
|
167
|
+
|
|
168
|
+
### Budget cap
|
|
169
|
+
|
|
170
|
+
If the user has set `maxBudgetPerSessionUsd`, Markus passes it as `--max-budget-usd` to Claude Code. This is a **hard limit enforced by Claude Code** — the session terminates if the budget is reached. Claude Code is the only tool with this enforced budget mechanism.
|
|
171
|
+
|
|
172
|
+
When working under a budget cap:
|
|
173
|
+
- Prefer `sonnet` over `opus` to stay within budget
|
|
174
|
+
- Split large tasks so each invocation stays within the per-session limit
|
|
175
|
+
- Monitor `cost.estimatedCostUsd` in results to gauge remaining budget capacity
|
|
176
|
+
|
|
177
|
+
## Cost Awareness
|
|
178
|
+
|
|
179
|
+
Claude Code is the most capable but potentially most expensive coding tool. It provides the **best cost visibility** — every result includes token counts and USD estimates:
|
|
180
|
+
|
|
181
|
+
```json
|
|
182
|
+
{
|
|
183
|
+
"cost": {
|
|
184
|
+
"estimatedCostUsd": 0.45,
|
|
185
|
+
"inputTokens": 85000,
|
|
186
|
+
"outputTokens": 12000,
|
|
187
|
+
"cacheReadTokens": 40000,
|
|
188
|
+
"source": "tool_output"
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
**Cost-saving practices:**
|
|
194
|
+
|
|
195
|
+
- Scope prompts tightly — avoid "refactor everything"
|
|
196
|
+
- Split large tasks into sequential focused invocations
|
|
197
|
+
- Use `codex` for trivial fixes instead of Claude Code
|
|
198
|
+
- Leverage prompt caching (repeated context in CLAUDE.md is cache-friendly)
|
|
199
|
+
- Review `cost.estimatedCostUsd` before chaining multiple invocations
|
|
200
|
+
- Use `effort: "low"` for simple edits, `effort: "high"` only for complex reasoning
|
|
201
|
+
- Prefer `sonnet` — only escalate to `opus` when justified
|
|
202
|
+
|
|
203
|
+
Report unusually high costs (> $1 per invocation) in a task note for visibility.
|
|
204
|
+
|
|
205
|
+
## Best Practices
|
|
206
|
+
|
|
207
|
+
- Write prompts with explicit file boundaries and test commands
|
|
208
|
+
- Always pass `task_id` so CLAUDE.md carries full task context
|
|
209
|
+
- Watch progress output for `file_edit` events to track what's changing
|
|
210
|
+
- Verify `result.testResult` before calling `coding_tool_apply`
|
|
211
|
+
- Prefer Claude Code when the task requires reading 5+ files to understand context
|
|
212
|
+
- Check `cost.estimatedCostUsd` after each invocation and factor it into your next decision
|
|
213
|
+
|
|
214
|
+
## Quality Verification Loop
|
|
215
|
+
|
|
216
|
+
For complex tasks, chain Claude Code invocations:
|
|
217
|
+
1. **Plan**: `invoke_coding_tool({ tool: "claude-code", mode: "plan", prompt: "Analyze and plan..." })`
|
|
218
|
+
2. **Implement**: `invoke_coding_tool({ tool: "claude-code", prompt: "Implement the plan..." })`
|
|
219
|
+
3. **Verify**: Check `result.testResult`. If failures exist, re-invoke with: "Fix these test failures: <output>"
|
|
220
|
+
4. **Apply**: Only after tests pass — `coding_tool_apply({ session_id, commit_message })`
|
|
221
|
+
|
|
222
|
+
Never apply changes from a session where tests failed.
|
|
223
|
+
|
|
224
|
+
## Cost Management
|
|
225
|
+
|
|
226
|
+
- Start with the default model. Only escalate to more expensive models for tasks that demonstrably need deeper reasoning.
|
|
227
|
+
- Check `cost.estimatedCostUsd` after each invocation and note it in task progress.
|
|
228
|
+
- For exploratory work, set a mental budget — if cost exceeds expectations, split into smaller focused invocations.
|
|
229
|
+
|
|
230
|
+
## Rules
|
|
231
|
+
|
|
232
|
+
- **DO** use for multi-file refactors and exploratory coding
|
|
233
|
+
- **DO** monitor token/cost data in the response
|
|
234
|
+
- **DO** break very large tasks into sequential invocations
|
|
235
|
+
- **DO** start with `sonnet` and only escalate to `opus` when needed
|
|
236
|
+
- **DO** request user approval before using `opus` for long tasks
|
|
237
|
+
- **DO NOT** use for one-line fixes — use `codex` or direct edit instead
|
|
238
|
+
- **DO NOT** ignore failed tests in the result — iterate until they pass
|
|
239
|
+
- **DO NOT** use `opus` by default — its cost can surprise users
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
{
|
|
2
|
+
"type": "skill",
|
|
3
|
+
"name": "claude-code",
|
|
4
|
+
"displayName": "Claude Code",
|
|
5
|
+
"version": "1.0.0",
|
|
6
|
+
"description": "Use the Claude Code CLI for complex refactors, multi-file changes, and deep codebase exploration",
|
|
7
|
+
"author": "markus",
|
|
8
|
+
"category": "development",
|
|
9
|
+
"tags": ["coding", "claude-code", "anthropic", "refactoring"],
|
|
10
|
+
"i18n": {
|
|
11
|
+
"zh-CN": {
|
|
12
|
+
"displayName": "Claude Code",
|
|
13
|
+
"description": "使用 Claude Code CLI 进行复杂重构、多文件修改和深度代码库探索"
|
|
14
|
+
}
|
|
15
|
+
},
|
|
16
|
+
"skill": { "skillFile": "SKILL.md" }
|
|
17
|
+
}
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: codex
|
|
3
|
+
description: Use the OpenAI Codex CLI for quick fixes, targeted edits, and non-interactive automation
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Codex
|
|
7
|
+
|
|
8
|
+
Codex (`codex` binary) is OpenAI's agentic coding CLI. Markus invokes it via `invoke_coding_tool({ tool: "codex", ... })`. Use it for fast, focused changes where speed and non-interactive automation matter more than deep multi-turn exploration.
|
|
9
|
+
|
|
10
|
+
## Installation
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
npm install -g @openai/codex
|
|
14
|
+
codex --version
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Verify with `markus doctor`. Requires authentication via `codex login` or `CODEX_API_KEY` env var for non-interactive mode.
|
|
18
|
+
|
|
19
|
+
## How Markus Invokes Codex
|
|
20
|
+
|
|
21
|
+
Markus runs Codex in fully automated, non-interactive mode:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
codex exec --full-auto --json --skip-git-repo-check "<prompt>"
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
| Flag | Purpose |
|
|
28
|
+
|---|---|
|
|
29
|
+
| `exec --full-auto` | Non-interactive mode — auto-approves all file edits and shell commands |
|
|
30
|
+
| `--json` | Emits JSONL events for structured progress parsing |
|
|
31
|
+
| `--skip-git-repo-check` | Allows running outside strict git repo requirements |
|
|
32
|
+
|
|
33
|
+
Additional args can be configured via `CodingToolConfig.defaultArgs`.
|
|
34
|
+
|
|
35
|
+
## Full-Auto Approval Mode
|
|
36
|
+
|
|
37
|
+
In a Markus agent session, there is no human at the terminal to approve Codex actions. The `exec --full-auto` mode is essential:
|
|
38
|
+
|
|
39
|
+
- Codex can edit files and run commands without prompting
|
|
40
|
+
- All actions happen within the sandbox (see below)
|
|
41
|
+
- If Codex would normally ask "Allow this edit?", it proceeds automatically
|
|
42
|
+
|
|
43
|
+
**Note:** `OPENAI_BASE_URL` is deprecated and no longer supported by Codex CLI. Custom endpoint configuration should use `~/.codex/config.toml`.
|
|
44
|
+
|
|
45
|
+
**Implication:** Write precise prompts with clear scope boundaries. Codex will act autonomously on whatever the prompt authorizes.
|
|
46
|
+
|
|
47
|
+
## AGENTS.md Context File
|
|
48
|
+
|
|
49
|
+
Codex reads project-level instruction files to understand repo conventions. The standard file is **`AGENTS.md`** in the repository root — a markdown file describing:
|
|
50
|
+
|
|
51
|
+
- Project structure and architecture
|
|
52
|
+
- Coding conventions and patterns
|
|
53
|
+
- Test commands and CI expectations
|
|
54
|
+
- Areas that are off-limits or require caution
|
|
55
|
+
|
|
56
|
+
If the repo already has `AGENTS.md`, Codex uses it automatically. Ensure it stays accurate for the project.
|
|
57
|
+
|
|
58
|
+
### Markus Task Context Injection
|
|
59
|
+
|
|
60
|
+
When you pass `task_id` to `invoke_coding_tool`, Markus additionally writes task-specific context to:
|
|
61
|
+
|
|
62
|
+
```
|
|
63
|
+
.agent_context/task_context.md
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
This file contains the full Markus task context (title, description, dependencies, progress-reporting CLI commands). Codex can read it during execution alongside any existing `AGENTS.md`.
|
|
67
|
+
|
|
68
|
+
**Best practice:** Keep permanent project guidance in `AGENTS.md`. Task-specific instructions come from Markus injection — do not manually duplicate task details into `AGENTS.md`.
|
|
69
|
+
|
|
70
|
+
## Sandbox Behavior
|
|
71
|
+
|
|
72
|
+
Codex runs in a sandboxed environment that restricts what the agent can access:
|
|
73
|
+
|
|
74
|
+
- File edits are scoped to the working directory (`workdir`)
|
|
75
|
+
- Network access may be limited depending on Codex configuration
|
|
76
|
+
- Shell commands run within sandbox constraints
|
|
77
|
+
|
|
78
|
+
**Implications for prompts:**
|
|
79
|
+
|
|
80
|
+
- Specify the exact files or directories to modify
|
|
81
|
+
- Include the test command to run (e.g., `pnpm test packages/core`)
|
|
82
|
+
- Do not assume Codex can reach external APIs unless sandbox allows it
|
|
83
|
+
- If a task requires installing new dependencies, mention it explicitly in the prompt
|
|
84
|
+
|
|
85
|
+
If Codex fails due to sandbox restrictions, note the error and either adjust the prompt to work within constraints or switch to `claude-code` for less restrictive execution.
|
|
86
|
+
|
|
87
|
+
## Usage Patterns
|
|
88
|
+
|
|
89
|
+
### Quick Bug Fix
|
|
90
|
+
|
|
91
|
+
```
|
|
92
|
+
invoke_coding_tool({
|
|
93
|
+
tool: "codex",
|
|
94
|
+
prompt: "Fix the off-by-one error in src/utils/pagination.ts line 42. The page size should default to 20, not 21. Run tests in that package after fixing.",
|
|
95
|
+
workdir: "/path/to/repo",
|
|
96
|
+
task_id: "task-456"
|
|
97
|
+
})
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
### Targeted Feature Addition
|
|
101
|
+
|
|
102
|
+
```
|
|
103
|
+
invoke_coding_tool({
|
|
104
|
+
tool: "codex",
|
|
105
|
+
prompt: "Add a --json flag to the task list command in packages/cli/src/commands/task.ts. Follow the existing output pattern used by other commands. Add a test case.",
|
|
106
|
+
workdir: "/path/to/repo",
|
|
107
|
+
task_id: "task-456"
|
|
108
|
+
})
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### Config or Script Update
|
|
112
|
+
|
|
113
|
+
```
|
|
114
|
+
invoke_coding_tool({
|
|
115
|
+
tool: "codex",
|
|
116
|
+
prompt: "Update the GitHub Actions workflow in .github/workflows/test.yml to add a matrix entry for Node 22. Do not change other jobs.",
|
|
117
|
+
workdir: "/path/to/repo",
|
|
118
|
+
task_id: "task-456"
|
|
119
|
+
})
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## When to Choose Codex vs Other Tools
|
|
123
|
+
|
|
124
|
+
| Choose Codex | Choose something else |
|
|
125
|
+
|---|---|
|
|
126
|
+
| Single-file or few-file fix | Multi-package refactor → `claude-code` |
|
|
127
|
+
| Clear, narrow prompt | Exploratory "figure out how this works" → `claude-code` |
|
|
128
|
+
| Speed is priority | Need token/cost reporting → `claude-code` |
|
|
129
|
+
| Repo has good `AGENTS.md` | Heavy `.cursor/rules` setup → `cursor-agent` |
|
|
130
|
+
|
|
131
|
+
## Model and Effort Selection
|
|
132
|
+
|
|
133
|
+
Codex supports per-invocation model and effort overrides:
|
|
134
|
+
|
|
135
|
+
```
|
|
136
|
+
invoke_coding_tool({
|
|
137
|
+
tool: "codex",
|
|
138
|
+
prompt: "...",
|
|
139
|
+
model: "gpt-5-codex", // default if not specified
|
|
140
|
+
effort: "medium", // sets CODEX_REASONING_EFFORT env var
|
|
141
|
+
})
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
### Model guidance
|
|
145
|
+
|
|
146
|
+
| Model | Best for | Cost note |
|
|
147
|
+
|---|---|---|
|
|
148
|
+
| `gpt-5.4-mini` | Trivial fixes, typos, config | Cheapest option |
|
|
149
|
+
| `gpt-5-codex` | Standard coding work | Cost-effective default for coding |
|
|
150
|
+
| `gpt-5.5` | Complex reasoning, architecture | ~4x more expensive than gpt-5-codex |
|
|
151
|
+
|
|
152
|
+
**Strategy:** `gpt-5-codex` is the right default for most Codex work. Only use `gpt-5.5` when the task involves complex reasoning that simpler models fail at. Prefer `gpt-5.4-mini` for trivial, low-risk changes.
|
|
153
|
+
|
|
154
|
+
### Effort levels
|
|
155
|
+
|
|
156
|
+
- `minimal` / `low` — Simple fixes with minimal reasoning
|
|
157
|
+
- `medium` — Standard development (default)
|
|
158
|
+
- `high` / `xhigh` — Complex problem solving, only when needed
|
|
159
|
+
|
|
160
|
+
### Cost note
|
|
161
|
+
|
|
162
|
+
Codex does not expose structured cost data through Markus. Estimate cost by:
|
|
163
|
+
- Task complexity and expected duration
|
|
164
|
+
- Model choice (gpt-5.5 is ~4x more expensive)
|
|
165
|
+
- Number of turns the agent takes
|
|
166
|
+
|
|
167
|
+
**Voluntarily call `request_user_approval` before using `gpt-5.5` for tasks that might run long.**
|
|
168
|
+
|
|
169
|
+
## Error Handling
|
|
170
|
+
|
|
171
|
+
Focus on result quality:
|
|
172
|
+
|
|
173
|
+
1. Check `result.success` and `result.summary`
|
|
174
|
+
2. Review `result.modifiedFiles` — should match expected scope
|
|
175
|
+
3. Inspect `result.testResult` if quality verification ran
|
|
176
|
+
4. On failure, read `result.error` and retry with a narrower prompt
|
|
177
|
+
|
|
178
|
+
If Codex modifies unexpected files, discard changes (`git checkout -- .` in `workdir`) and re-invoke with explicit file boundaries:
|
|
179
|
+
|
|
180
|
+
```
|
|
181
|
+
"Modify ONLY packages/cli/src/commands/task.ts. Do not touch any other files."
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
## Best Practices
|
|
185
|
+
|
|
186
|
+
- Keep prompts short and specific — Codex excels at targeted tasks
|
|
187
|
+
- Ensure `AGENTS.md` exists for project conventions (create or update if missing)
|
|
188
|
+
- Always pass `task_id` for task context injection
|
|
189
|
+
- Verify changes with `git diff` before `coding_tool_apply`
|
|
190
|
+
- Use `gpt-5-codex` as the default — escalate only when justified
|
|
191
|
+
- `exec --full-auto` is handled by Markus — do not try to run Codex interactively from an agent
|
|
192
|
+
|
|
193
|
+
## Full-Auto Mode Best Practices
|
|
194
|
+
|
|
195
|
+
Codex runs in `--full-auto` mode by default, which means it will make changes without asking for confirmation. This makes quality verification especially important:
|
|
196
|
+
|
|
197
|
+
1. **Scope tightly**: Write precise prompts that describe exactly what to change and what NOT to change
|
|
198
|
+
2. **Verify before applying**: Always check `result.diffStats` and `result.modifiedFiles` before `coding_tool_apply`
|
|
199
|
+
3. **Run tests**: If Codex doesn't run tests automatically, run them yourself via `shell_execute` before applying
|
|
200
|
+
|
|
201
|
+
## When to Choose Codex
|
|
202
|
+
|
|
203
|
+
- Quick, targeted fixes (one file, clear problem)
|
|
204
|
+
- Scripted automation (generate boilerplate, rename across files)
|
|
205
|
+
- CI-friendly operations (no interactive prompts needed)
|
|
206
|
+
- When speed matters more than deep reasoning
|
|
207
|
+
|
|
208
|
+
## Rules
|
|
209
|
+
|
|
210
|
+
- **DO** use for quick fixes and well-scoped edits
|
|
211
|
+
- **DO** maintain an accurate `AGENTS.md` in project repos
|
|
212
|
+
- **DO** set explicit file boundaries in prompts
|
|
213
|
+
- **DO** default to `gpt-5-codex` for cost-effectiveness
|
|
214
|
+
- **DO NOT** use for large exploratory refactors — use `claude-code`
|
|
215
|
+
- **DO NOT** assume network or install permissions — check sandbox errors
|
|
216
|
+
- **DO NOT** apply changes that touch files outside the stated scope
|
|
217
|
+
- **DO NOT** use `gpt-5.5` by default — its cost is ~4x higher
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
{
|
|
2
|
+
"type": "skill",
|
|
3
|
+
"name": "codex",
|
|
4
|
+
"displayName": "Codex",
|
|
5
|
+
"version": "1.0.0",
|
|
6
|
+
"description": "Use the OpenAI Codex CLI for quick fixes, targeted edits, and non-interactive automation",
|
|
7
|
+
"author": "markus",
|
|
8
|
+
"category": "development",
|
|
9
|
+
"tags": ["coding", "codex", "openai", "automation"],
|
|
10
|
+
"i18n": {
|
|
11
|
+
"zh-CN": {
|
|
12
|
+
"displayName": "Codex",
|
|
13
|
+
"description": "使用 OpenAI Codex CLI 进行快速修复、定向编辑和非交互式自动化"
|
|
14
|
+
}
|
|
15
|
+
},
|
|
16
|
+
"skill": { "skillFile": "SKILL.md" }
|
|
17
|
+
}
|