@markus-global/cli 0.8.4 → 0.8.5-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/dist/commands/agent.js +9 -9
  2. package/dist/commands/agent.js.map +1 -1
  3. package/dist/commands/doctor.d.ts +3 -1
  4. package/dist/commands/doctor.d.ts.map +1 -1
  5. package/dist/commands/doctor.js +27 -1
  6. package/dist/commands/doctor.js.map +1 -1
  7. package/dist/commands/models.d.ts.map +1 -1
  8. package/dist/commands/models.js +6 -7
  9. package/dist/commands/models.js.map +1 -1
  10. package/dist/commands/project.d.ts +3 -0
  11. package/dist/commands/project.d.ts.map +1 -0
  12. package/dist/commands/project.js +25 -0
  13. package/dist/commands/project.js.map +1 -0
  14. package/dist/commands/requirement.d.ts +3 -0
  15. package/dist/commands/requirement.d.ts.map +1 -0
  16. package/dist/commands/requirement.js +34 -0
  17. package/dist/commands/requirement.js.map +1 -0
  18. package/dist/commands/start.d.ts.map +1 -1
  19. package/dist/commands/start.js +42 -1
  20. package/dist/commands/start.js.map +1 -1
  21. package/dist/commands/task.d.ts +3 -0
  22. package/dist/commands/task.d.ts.map +1 -0
  23. package/dist/commands/task.js +110 -0
  24. package/dist/commands/task.js.map +1 -0
  25. package/dist/index.js +8 -0
  26. package/dist/index.js.map +1 -1
  27. package/dist/markus.mjs +3770 -965
  28. package/dist/output.d.ts +3 -1
  29. package/dist/output.d.ts.map +1 -1
  30. package/dist/output.js +34 -3
  31. package/dist/output.js.map +1 -1
  32. package/dist/web-ui/assets/arc-azDa9rNQ.js +1 -0
  33. package/dist/web-ui/assets/architectureDiagram-3BPJPVTR-CWoGp8TB.js +36 -0
  34. package/dist/web-ui/assets/blockDiagram-GPEHLZMM-C2Tq3zqo.js +132 -0
  35. package/dist/web-ui/assets/c4Diagram-AAUBKEIU-C2tj98or.js +10 -0
  36. package/dist/web-ui/assets/channel-D0Q-P9rQ.js +1 -0
  37. package/dist/web-ui/assets/chunk-2J33WTMH-DMhlyS99.js +1 -0
  38. package/dist/web-ui/assets/chunk-4BX2VUAB-C8hL0QFv.js +1 -0
  39. package/dist/web-ui/assets/chunk-55IACEB6-BPCe4caz.js +1 -0
  40. package/dist/web-ui/assets/chunk-727SXJPM-C5EAjSrN.js +206 -0
  41. package/dist/web-ui/assets/chunk-AQP2D5EJ-BLSz7iPE.js +231 -0
  42. package/dist/web-ui/assets/chunk-FMBD7UC4-Sk4yLzwq.js +15 -0
  43. package/dist/web-ui/assets/chunk-ND2GUHAM-DbuQgWyn.js +1 -0
  44. package/dist/web-ui/assets/chunk-QZHKN3VN-REM6PaDE.js +1 -0
  45. package/dist/web-ui/assets/classDiagram-4FO5ZUOK-BcOPwdcC.js +1 -0
  46. package/dist/web-ui/assets/classDiagram-v2-Q7XG4LA2-BcOPwdcC.js +1 -0
  47. package/dist/web-ui/assets/cose-bilkent-S5V4N54A-Chu2Y9EC.js +1 -0
  48. package/dist/web-ui/assets/cytoscape.esm-D3_iZ_3b.js +321 -0
  49. package/dist/web-ui/assets/dagre-BM42HDAG-BGQGbUMF.js +4 -0
  50. package/dist/web-ui/assets/defaultLocale-DX6XiGOO.js +1 -0
  51. package/dist/web-ui/assets/diagram-2AECGRRQ-DnZ1SQGN.js +43 -0
  52. package/dist/web-ui/assets/diagram-5GNKFQAL-B0S37NyM.js +10 -0
  53. package/dist/web-ui/assets/diagram-KO2AKTUF-BxDwUDuY.js +3 -0
  54. package/dist/web-ui/assets/diagram-LMA3HP47-5UC8M7Iq.js +24 -0
  55. package/dist/web-ui/assets/diagram-OG6HWLK6-C-eMeMRG.js +24 -0
  56. package/dist/web-ui/assets/erDiagram-TEJ5UH35-CVWhzHKv.js +85 -0
  57. package/dist/web-ui/assets/flowDiagram-I6XJVG4X-CMG-a-kh.js +162 -0
  58. package/dist/web-ui/assets/ganttDiagram-6RSMTGT7-DbVJ6VGB.js +292 -0
  59. package/dist/web-ui/assets/gitGraphDiagram-PVQCEYII-DcCuYR-R.js +106 -0
  60. package/dist/web-ui/assets/graph--OzhPTMs.js +1 -0
  61. package/dist/web-ui/assets/index-PVrVcpcl.css +1 -0
  62. package/dist/web-ui/assets/index-zJq4U9RT.js +776 -0
  63. package/dist/web-ui/assets/infoDiagram-5YYISTIA-CaY7gJ4a.js +2 -0
  64. package/dist/web-ui/assets/init-Gi6I4Gst.js +1 -0
  65. package/dist/web-ui/assets/ishikawaDiagram-YF4QCWOH-l4_2NV1P.js +70 -0
  66. package/dist/web-ui/assets/journeyDiagram-JHISSGLW-BVeQNwa5.js +139 -0
  67. package/dist/web-ui/assets/kanban-definition-UN3LZRKU-CtvPOV3r.js +89 -0
  68. package/dist/web-ui/assets/layout-SsrduOYp.js +1 -0
  69. package/dist/web-ui/assets/linear-B0DfGdNc.js +1 -0
  70. package/dist/web-ui/assets/mermaid.core-Bz3avYM5.js +303 -0
  71. package/dist/web-ui/assets/mindmap-definition-RKZ34NQL-1X-u7gPH.js +96 -0
  72. package/dist/web-ui/assets/ordinal-Cboi1Yqb.js +1 -0
  73. package/dist/web-ui/assets/pieDiagram-4H26LBE5-BO8LpJ1H.js +30 -0
  74. package/dist/web-ui/assets/plantuml-DezRDxd4.js +357 -0
  75. package/dist/web-ui/assets/quadrantDiagram-W4KKPZXB-BBYmPM7O.js +7 -0
  76. package/dist/web-ui/assets/requirementDiagram-4Y6WPE33-CUU8gZny.js +84 -0
  77. package/dist/web-ui/assets/sankeyDiagram-5OEKKPKP-k6GjcALi.js +40 -0
  78. package/dist/web-ui/assets/sequenceDiagram-3UESZ5HK-CScOE6Nf.js +162 -0
  79. package/dist/web-ui/assets/stateDiagram-AJRCARHV-BcvHRBZl.js +1 -0
  80. package/dist/web-ui/assets/stateDiagram-v2-BHNVJYJU-p321ujvX.js +1 -0
  81. package/dist/web-ui/assets/timeline-definition-PNZ67QCA-D7tGfjR6.js +120 -0
  82. package/dist/web-ui/assets/vennDiagram-CIIHVFJN-AU7MqjmN.js +34 -0
  83. package/dist/web-ui/assets/viz-global-C_AyN6D9.js +9 -0
  84. package/dist/web-ui/assets/wardley-L42UT6IY-DKQmSXOS.js +161 -0
  85. package/dist/web-ui/assets/wardleyDiagram-YWT4CUSO-DVMv24j_.js +78 -0
  86. package/dist/web-ui/assets/xychartDiagram-2RQKCTM6-CGQCKCak.js +7 -0
  87. package/dist/web-ui/index.html +2 -2
  88. package/package.json +2 -1
  89. package/templates/roles/SHARED.md +113 -8
  90. package/templates/roles/ai-engineer/ROLE.md +35 -0
  91. package/templates/roles/ai-engineer/agent.json +1 -1
  92. package/templates/roles/architect/ROLE.md +15 -0
  93. package/templates/roles/architect/agent.json +1 -1
  94. package/templates/roles/content-writer/HEARTBEAT.md +29 -0
  95. package/templates/roles/content-writer/POLICIES.md +30 -0
  96. package/templates/roles/content-writer/ROLE.md +235 -19
  97. package/templates/roles/data-engineer/ROLE.md +29 -0
  98. package/templates/roles/data-engineer/agent.json +1 -1
  99. package/templates/roles/developer/HEARTBEAT.md +25 -7
  100. package/templates/roles/developer/POLICIES.md +24 -6
  101. package/templates/roles/developer/ROLE.md +335 -55
  102. package/templates/roles/devops/HEARTBEAT.md +30 -0
  103. package/templates/roles/devops/POLICIES.md +30 -0
  104. package/templates/roles/devops/ROLE.md +126 -20
  105. package/templates/roles/org-manager/ROLE.md +15 -0
  106. package/templates/roles/product-manager/POLICIES.md +29 -0
  107. package/templates/roles/product-manager/ROLE.md +126 -17
  108. package/templates/roles/project-manager/HEARTBEAT.md +30 -0
  109. package/templates/roles/project-manager/POLICIES.md +29 -0
  110. package/templates/roles/project-manager/ROLE.md +18 -0
  111. package/templates/roles/qa-engineer/HEARTBEAT.md +29 -0
  112. package/templates/roles/qa-engineer/POLICIES.md +29 -0
  113. package/templates/roles/qa-engineer/ROLE.md +133 -26
  114. package/templates/roles/research-assistant/HEARTBEAT.md +29 -0
  115. package/templates/roles/research-assistant/POLICIES.md +29 -0
  116. package/templates/roles/research-assistant/ROLE.md +310 -48
  117. package/templates/roles/reviewer/POLICIES.md +29 -0
  118. package/templates/roles/reviewer/ROLE.md +49 -0
  119. package/templates/roles/scrum-master/ROLE.md +6 -0
  120. package/templates/roles/skill-architect/HEARTBEAT.md +29 -0
  121. package/templates/roles/skill-architect/POLICIES.md +29 -0
  122. package/templates/roles/skill-architect/ROLE.md +267 -20
  123. package/templates/roles/sre/agent.json +1 -1
  124. package/templates/roles/tech-writer/HEARTBEAT.md +29 -0
  125. package/templates/roles/tech-writer/POLICIES.md +28 -0
  126. package/templates/roles/tech-writer/ROLE.md +258 -21
  127. package/templates/skills/claude-code/SKILL.md +239 -0
  128. package/templates/skills/claude-code/skill.json +17 -0
  129. package/templates/skills/codex/SKILL.md +217 -0
  130. package/templates/skills/codex/skill.json +17 -0
  131. package/templates/skills/coding-tools/SKILL.md +300 -0
  132. package/templates/skills/coding-tools/skill.json +17 -0
  133. package/templates/skills/cursor-agent/SKILL.md +262 -0
  134. package/templates/skills/cursor-agent/skill.json +17 -0
  135. package/templates/skills/feishu-interaction/SKILL.md +103 -0
  136. package/templates/skills/feishu-interaction/skill.json +26 -0
  137. package/templates/skills/self-evolution/SKILL.md +31 -0
  138. package/templates/teams/content-team/NORMS.md +17 -0
  139. package/templates/teams/dev-squad/NORMS.md +26 -0
  140. package/templates/teams/dev-squad/team.json +4 -4
  141. package/templates/teams/engineering-pod/NORMS.md +33 -0
  142. package/templates/teams/engineering-pod/team.json +4 -4
  143. package/templates/teams/research-lab/NORMS.md +15 -0
  144. package/templates/teams/startup-team/NORMS.md +17 -0
  145. package/templates/teams/startup-team/team.json +1 -1
  146. package/dist/web-ui/assets/index-CZL1VHgy.css +0 -1
  147. package/dist/web-ui/assets/index-DZjXJ0HZ.js +0 -724
@@ -0,0 +1,239 @@
1
+ ---
2
+ name: claude-code
3
+ description: Use the Claude Code CLI for complex refactors, multi-file changes, and sustained codebase exploration
4
+ ---
5
+
6
+ # Claude Code
7
+
8
+ Claude Code (`claude` binary) is Anthropic's agentic coding CLI. Markus invokes it via `invoke_coding_tool({ tool: "claude-code", ... })`. Use it for complex, multi-turn coding tasks where deep reasoning and broad file exploration are needed.
9
+
10
+ ## Installation
11
+
12
+ ```bash
13
+ npm install -g @anthropic-ai/claude-code
14
+ claude --version
15
+ ```
16
+
17
+ Verify with `markus doctor`. Requires authentication via one of: `ANTHROPIC_API_KEY` env var, `ANTHROPIC_BASE_URL` for custom endpoints, or interactive `claude` login.
18
+
19
+ ## How Markus Invokes Claude Code
20
+
21
+ Markus runs Claude Code in non-interactive print mode with structured streaming output:
22
+
23
+ ```bash
24
+ claude --print --output-format stream-json --verbose --max-turns 50 --permission-mode bypassPermissions "<prompt>"
25
+ ```
26
+
27
+ | Flag | Purpose |
28
+ |---|---|
29
+ | `--print` | Non-interactive mode — runs to completion without user input |
30
+ | `--output-format stream-json` | Emits structured JSON events for progress parsing |
31
+ | `--verbose` | Detailed progress output |
32
+ | `--max-turns 50` | Allows up to 50 agent turns for complex tasks |
33
+ | `--permission-mode bypassPermissions` | Auto-approves all file edits and commands — required because `--print` mode has no interactive stdin for approval prompts |
34
+
35
+ Additional args can be configured per-deployment via `CodingToolConfig.defaultArgs`.
36
+
37
+ ## Stream-JSON Output
38
+
39
+ Each stdout line is a JSON event. Key event types:
40
+
41
+ | Event type | Meaning |
42
+ |---|---|
43
+ | `assistant` with `text` content | Progress message / reasoning summary |
44
+ | `assistant` with `tool_use` | File edit, shell command, or other tool invocation |
45
+ | `result` | Final outcome with cost and token data |
46
+
47
+ The `result` event includes:
48
+
49
+ - `result` — Final summary text
50
+ - `input_tokens`, `output_tokens` — Token usage
51
+ - `cache_read_tokens`, `cache_write_tokens` — Prompt cache stats
52
+ - `cost_usd` — Estimated cost in USD
53
+
54
+ Markus parses these into progress events (`file_edit`, `progress`, `completed`) and extracts cost reports automatically.
55
+
56
+ ## CLAUDE.md Context File
57
+
58
+ When you pass `task_id` to `invoke_coding_tool`, Markus writes a `CLAUDE.md` file in the repository root before invoking Claude Code. This file contains:
59
+
60
+ - Task title, description, status, and priority
61
+ - Subtasks, notes, and deliverables
62
+ - Requirement and project context
63
+ - Upstream/downstream dependency summaries
64
+ - Markus CLI commands for reporting progress
65
+
66
+ Claude Code reads `CLAUDE.md` automatically as project context. **Do not delete or overwrite it** during a session — it is regenerated each invocation.
67
+
68
+ If the repo already has a permanent `CLAUDE.md`, Markus overwrites it for the session. Consider restoring project-level content after the task if needed.
69
+
70
+ ## Usage Patterns
71
+
72
+ ### Complex Refactor
73
+
74
+ ```
75
+ invoke_coding_tool({
76
+ tool: "claude-code",
77
+ prompt: "Refactor the auth module to use dependency injection. Move AuthService to src/services/, update all imports, keep existing test behavior. Run the test suite when done.",
78
+ workdir: "/path/to/repo",
79
+ task_id: "task-123"
80
+ })
81
+ ```
82
+
83
+ ### Debug and Fix
84
+
85
+ ```
86
+ invoke_coding_tool({
87
+ tool: "claude-code",
88
+ prompt: "Tests in packages/core/test/auth.test.ts are failing with 'token expired'. Find the root cause in src/auth/ and fix without changing the public API. Show which tests pass after the fix.",
89
+ workdir: "/path/to/repo",
90
+ task_id: "task-123"
91
+ })
92
+ ```
93
+
94
+ ### Explore Then Implement
95
+
96
+ For unfamiliar codebases, ask Claude Code to explore first:
97
+
98
+ ```
99
+ "Read the codebase structure under src/coding-tools/. Then implement the feature described in the task context. Start by listing the files you plan to modify."
100
+ ```
101
+
102
+ ## Retry Strategies for Long Tasks
103
+
104
+ Claude Code supports up to 50 turns, but long tasks can still stall or partially complete.
105
+
106
+ ### If the session completes but work is incomplete
107
+
108
+ Re-invoke with explicit remaining scope:
109
+
110
+ ```
111
+ "Previous session modified src/foo.ts and src/bar.ts but did not update tests. Complete the test coverage for the changes in src/foo.ts. Do not re-modify files that are already correct."
112
+ ```
113
+
114
+ ### If the session fails or times out
115
+
116
+ 1. Check `git status` in `workdir` for partial changes
117
+ 2. Either apply partial work with `coding_tool_apply` or discard with `git checkout -- .`
118
+ 3. Retry with a **smaller scope** — one module or one feature at a time
119
+
120
+ ### If Claude Code loops or over-edits
121
+
122
+ Add constraints to the prompt:
123
+
124
+ ```
125
+ "Modify ONLY files under src/handlers/. Do not touch tests, config, or unrelated packages."
126
+ ```
127
+
128
+ ### Escalation after 2 retries
129
+
130
+ Switch to `codex` for a targeted fix, or edit directly with `file_edit`.
131
+
132
+ ## Model and Effort Selection
133
+
134
+ Claude Code supports per-invocation model and effort overrides:
135
+
136
+ ```
137
+ invoke_coding_tool({
138
+ tool: "claude-code",
139
+ prompt: "...",
140
+ model: "sonnet", // default: tool's own default (usually sonnet)
141
+ effort: "medium", // low | medium | high | xhigh | max
142
+ })
143
+ ```
144
+
145
+ ### Model guidance
146
+
147
+ | Model | Best for | Cost note |
148
+ |---|---|---|
149
+ | `haiku` | Simple subagent tasks, quick checks | Cheapest option |
150
+ | `sonnet` | Most coding work — default choice | Good balance of cost and capability |
151
+ | `opus` | Complex architecture, multi-file refactors, unfamiliar codebases | ~15x more expensive per token than Sonnet |
152
+ | `fable` | Creative or documentation tasks | Specialized |
153
+
154
+ **Strategy:** Start with `sonnet` (or the user's `defaultModel`). Only use `opus` when:
155
+
156
+ - Sonnet attempt failed or produced poor results
157
+ - The task involves complex reasoning across 10+ files
158
+ - Architecture decisions or trade-off analysis is required
159
+
160
+ **Warning:** Opus is approximately 15x more expensive than Sonnet per token. A task that costs $0.30 with Sonnet could cost $4.50 with Opus. **Voluntarily call `request_user_approval` before using Opus for any task expected to run more than a few turns.**
161
+
162
+ ### Effort guidance
163
+
164
+ - `low` — Simple edits, typo fixes, config changes
165
+ - `medium` — Standard development tasks (default)
166
+ - `high` — Complex reasoning, multi-step problem solving
167
+
168
+ ### Budget cap
169
+
170
+ If the user has set `maxBudgetPerSessionUsd`, Markus passes it as `--max-budget-usd` to Claude Code. This is a **hard limit enforced by Claude Code** — the session terminates if the budget is reached. Claude Code is the only tool with this enforced budget mechanism.
171
+
172
+ When working under a budget cap:
173
+ - Prefer `sonnet` over `opus` to stay within budget
174
+ - Split large tasks so each invocation stays within the per-session limit
175
+ - Monitor `cost.estimatedCostUsd` in results to gauge remaining budget capacity
176
+
177
+ ## Cost Awareness
178
+
179
+ Claude Code is the most capable but potentially most expensive coding tool. It provides the **best cost visibility** — every result includes token counts and USD estimates:
180
+
181
+ ```json
182
+ {
183
+ "cost": {
184
+ "estimatedCostUsd": 0.45,
185
+ "inputTokens": 85000,
186
+ "outputTokens": 12000,
187
+ "cacheReadTokens": 40000,
188
+ "source": "tool_output"
189
+ }
190
+ }
191
+ ```
192
+
193
+ **Cost-saving practices:**
194
+
195
+ - Scope prompts tightly — avoid "refactor everything"
196
+ - Split large tasks into sequential focused invocations
197
+ - Use `codex` for trivial fixes instead of Claude Code
198
+ - Leverage prompt caching (repeated context in CLAUDE.md is cache-friendly)
199
+ - Review `cost.estimatedCostUsd` before chaining multiple invocations
200
+ - Use `effort: "low"` for simple edits, `effort: "high"` only for complex reasoning
201
+ - Prefer `sonnet` — only escalate to `opus` when justified
202
+
203
+ Report unusually high costs (> $1 per invocation) in a task note for visibility.
204
+
205
+ ## Best Practices
206
+
207
+ - Write prompts with explicit file boundaries and test commands
208
+ - Always pass `task_id` so CLAUDE.md carries full task context
209
+ - Watch progress output for `file_edit` events to track what's changing
210
+ - Verify `result.testResult` before calling `coding_tool_apply`
211
+ - Prefer Claude Code when the task requires reading 5+ files to understand context
212
+ - Check `cost.estimatedCostUsd` after each invocation and factor it into your next decision
213
+
214
+ ## Quality Verification Loop
215
+
216
+ For complex tasks, chain Claude Code invocations:
217
+ 1. **Plan**: `invoke_coding_tool({ tool: "claude-code", mode: "plan", prompt: "Analyze and plan..." })`
218
+ 2. **Implement**: `invoke_coding_tool({ tool: "claude-code", prompt: "Implement the plan..." })`
219
+ 3. **Verify**: Check `result.testResult`. If failures exist, re-invoke with: "Fix these test failures: <output>"
220
+ 4. **Apply**: Only after tests pass — `coding_tool_apply({ session_id, commit_message })`
221
+
222
+ Never apply changes from a session where tests failed.
223
+
224
+ ## Cost Management
225
+
226
+ - Start with the default model. Only escalate to more expensive models for tasks that demonstrably need deeper reasoning.
227
+ - Check `cost.estimatedCostUsd` after each invocation and note it in task progress.
228
+ - For exploratory work, set a mental budget — if cost exceeds expectations, split into smaller focused invocations.
229
+
230
+ ## Rules
231
+
232
+ - **DO** use for multi-file refactors and exploratory coding
233
+ - **DO** monitor token/cost data in the response
234
+ - **DO** break very large tasks into sequential invocations
235
+ - **DO** start with `sonnet` and only escalate to `opus` when needed
236
+ - **DO** request user approval before using `opus` for long tasks
237
+ - **DO NOT** use for one-line fixes — use `codex` or direct edit instead
238
+ - **DO NOT** ignore failed tests in the result — iterate until they pass
239
+ - **DO NOT** use `opus` by default — its cost can surprise users
@@ -0,0 +1,17 @@
1
+ {
2
+ "type": "skill",
3
+ "name": "claude-code",
4
+ "displayName": "Claude Code",
5
+ "version": "1.0.0",
6
+ "description": "Use the Claude Code CLI for complex refactors, multi-file changes, and deep codebase exploration",
7
+ "author": "markus",
8
+ "category": "development",
9
+ "tags": ["coding", "claude-code", "anthropic", "refactoring"],
10
+ "i18n": {
11
+ "zh-CN": {
12
+ "displayName": "Claude Code",
13
+ "description": "使用 Claude Code CLI 进行复杂重构、多文件修改和深度代码库探索"
14
+ }
15
+ },
16
+ "skill": { "skillFile": "SKILL.md" }
17
+ }
@@ -0,0 +1,217 @@
1
+ ---
2
+ name: codex
3
+ description: Use the OpenAI Codex CLI for quick fixes, targeted edits, and non-interactive automation
4
+ ---
5
+
6
+ # Codex
7
+
8
+ Codex (`codex` binary) is OpenAI's agentic coding CLI. Markus invokes it via `invoke_coding_tool({ tool: "codex", ... })`. Use it for fast, focused changes where speed and non-interactive automation matter more than deep multi-turn exploration.
9
+
10
+ ## Installation
11
+
12
+ ```bash
13
+ npm install -g @openai/codex
14
+ codex --version
15
+ ```
16
+
17
+ Verify with `markus doctor`. Requires authentication via `codex login` or `CODEX_API_KEY` env var for non-interactive mode.
18
+
19
+ ## How Markus Invokes Codex
20
+
21
+ Markus runs Codex in fully automated, non-interactive mode:
22
+
23
+ ```bash
24
+ codex exec --full-auto --json --skip-git-repo-check "<prompt>"
25
+ ```
26
+
27
+ | Flag | Purpose |
28
+ |---|---|
29
+ | `exec --full-auto` | Non-interactive mode — auto-approves all file edits and shell commands |
30
+ | `--json` | Emits JSONL events for structured progress parsing |
31
+ | `--skip-git-repo-check` | Allows running outside strict git repo requirements |
32
+
33
+ Additional args can be configured via `CodingToolConfig.defaultArgs`.
34
+
35
+ ## Full-Auto Approval Mode
36
+
37
+ In a Markus agent session, there is no human at the terminal to approve Codex actions. The `exec --full-auto` mode is essential:
38
+
39
+ - Codex can edit files and run commands without prompting
40
+ - All actions happen within the sandbox (see below)
41
+ - If Codex would normally ask "Allow this edit?", it proceeds automatically
42
+
43
+ **Note:** `OPENAI_BASE_URL` is deprecated and no longer supported by Codex CLI. Custom endpoint configuration should use `~/.codex/config.toml`.
44
+
45
+ **Implication:** Write precise prompts with clear scope boundaries. Codex will act autonomously on whatever the prompt authorizes.
46
+
47
+ ## AGENTS.md Context File
48
+
49
+ Codex reads project-level instruction files to understand repo conventions. The standard file is **`AGENTS.md`** in the repository root — a markdown file describing:
50
+
51
+ - Project structure and architecture
52
+ - Coding conventions and patterns
53
+ - Test commands and CI expectations
54
+ - Areas that are off-limits or require caution
55
+
56
+ If the repo already has `AGENTS.md`, Codex uses it automatically. Ensure it stays accurate for the project.
57
+
58
+ ### Markus Task Context Injection
59
+
60
+ When you pass `task_id` to `invoke_coding_tool`, Markus additionally writes task-specific context to:
61
+
62
+ ```
63
+ .agent_context/task_context.md
64
+ ```
65
+
66
+ This file contains the full Markus task context (title, description, dependencies, progress-reporting CLI commands). Codex can read it during execution alongside any existing `AGENTS.md`.
67
+
68
+ **Best practice:** Keep permanent project guidance in `AGENTS.md`. Task-specific instructions come from Markus injection — do not manually duplicate task details into `AGENTS.md`.
69
+
70
+ ## Sandbox Behavior
71
+
72
+ Codex runs in a sandboxed environment that restricts what the agent can access:
73
+
74
+ - File edits are scoped to the working directory (`workdir`)
75
+ - Network access may be limited depending on Codex configuration
76
+ - Shell commands run within sandbox constraints
77
+
78
+ **Implications for prompts:**
79
+
80
+ - Specify the exact files or directories to modify
81
+ - Include the test command to run (e.g., `pnpm test packages/core`)
82
+ - Do not assume Codex can reach external APIs unless sandbox allows it
83
+ - If a task requires installing new dependencies, mention it explicitly in the prompt
84
+
85
+ If Codex fails due to sandbox restrictions, note the error and either adjust the prompt to work within constraints or switch to `claude-code` for less restrictive execution.
86
+
87
+ ## Usage Patterns
88
+
89
+ ### Quick Bug Fix
90
+
91
+ ```
92
+ invoke_coding_tool({
93
+ tool: "codex",
94
+ prompt: "Fix the off-by-one error in src/utils/pagination.ts line 42. The page size should default to 20, not 21. Run tests in that package after fixing.",
95
+ workdir: "/path/to/repo",
96
+ task_id: "task-456"
97
+ })
98
+ ```
99
+
100
+ ### Targeted Feature Addition
101
+
102
+ ```
103
+ invoke_coding_tool({
104
+ tool: "codex",
105
+ prompt: "Add a --json flag to the task list command in packages/cli/src/commands/task.ts. Follow the existing output pattern used by other commands. Add a test case.",
106
+ workdir: "/path/to/repo",
107
+ task_id: "task-456"
108
+ })
109
+ ```
110
+
111
+ ### Config or Script Update
112
+
113
+ ```
114
+ invoke_coding_tool({
115
+ tool: "codex",
116
+ prompt: "Update the GitHub Actions workflow in .github/workflows/test.yml to add a matrix entry for Node 22. Do not change other jobs.",
117
+ workdir: "/path/to/repo",
118
+ task_id: "task-456"
119
+ })
120
+ ```
121
+
122
+ ## When to Choose Codex vs Other Tools
123
+
124
+ | Choose Codex | Choose something else |
125
+ |---|---|
126
+ | Single-file or few-file fix | Multi-package refactor → `claude-code` |
127
+ | Clear, narrow prompt | Exploratory "figure out how this works" → `claude-code` |
128
+ | Speed is priority | Need token/cost reporting → `claude-code` |
129
+ | Repo has good `AGENTS.md` | Heavy `.cursor/rules` setup → `cursor-agent` |
130
+
131
+ ## Model and Effort Selection
132
+
133
+ Codex supports per-invocation model and effort overrides:
134
+
135
+ ```
136
+ invoke_coding_tool({
137
+ tool: "codex",
138
+ prompt: "...",
139
+ model: "gpt-5-codex", // default if not specified
140
+ effort: "medium", // sets CODEX_REASONING_EFFORT env var
141
+ })
142
+ ```
143
+
144
+ ### Model guidance
145
+
146
+ | Model | Best for | Cost note |
147
+ |---|---|---|
148
+ | `gpt-5.4-mini` | Trivial fixes, typos, config | Cheapest option |
149
+ | `gpt-5-codex` | Standard coding work | Cost-effective default for coding |
150
+ | `gpt-5.5` | Complex reasoning, architecture | ~4x more expensive than gpt-5-codex |
151
+
152
+ **Strategy:** `gpt-5-codex` is the right default for most Codex work. Only use `gpt-5.5` when the task involves complex reasoning that simpler models fail at. Prefer `gpt-5.4-mini` for trivial, low-risk changes.
153
+
154
+ ### Effort levels
155
+
156
+ - `minimal` / `low` — Simple fixes with minimal reasoning
157
+ - `medium` — Standard development (default)
158
+ - `high` / `xhigh` — Complex problem solving, only when needed
159
+
160
+ ### Cost note
161
+
162
+ Codex does not expose structured cost data through Markus. Estimate cost by:
163
+ - Task complexity and expected duration
164
+ - Model choice (gpt-5.5 is ~4x more expensive)
165
+ - Number of turns the agent takes
166
+
167
+ **Voluntarily call `request_user_approval` before using `gpt-5.5` for tasks that might run long.**
168
+
169
+ ## Error Handling
170
+
171
+ Focus on result quality:
172
+
173
+ 1. Check `result.success` and `result.summary`
174
+ 2. Review `result.modifiedFiles` — should match expected scope
175
+ 3. Inspect `result.testResult` if quality verification ran
176
+ 4. On failure, read `result.error` and retry with a narrower prompt
177
+
178
+ If Codex modifies unexpected files, discard changes (`git checkout -- .` in `workdir`) and re-invoke with explicit file boundaries:
179
+
180
+ ```
181
+ "Modify ONLY packages/cli/src/commands/task.ts. Do not touch any other files."
182
+ ```
183
+
184
+ ## Best Practices
185
+
186
+ - Keep prompts short and specific — Codex excels at targeted tasks
187
+ - Ensure `AGENTS.md` exists for project conventions (create or update if missing)
188
+ - Always pass `task_id` for task context injection
189
+ - Verify changes with `git diff` before `coding_tool_apply`
190
+ - Use `gpt-5-codex` as the default — escalate only when justified
191
+ - `exec --full-auto` is handled by Markus — do not try to run Codex interactively from an agent
192
+
193
+ ## Full-Auto Mode Best Practices
194
+
195
+ Codex runs in `--full-auto` mode by default, which means it will make changes without asking for confirmation. This makes quality verification especially important:
196
+
197
+ 1. **Scope tightly**: Write precise prompts that describe exactly what to change and what NOT to change
198
+ 2. **Verify before applying**: Always check `result.diffStats` and `result.modifiedFiles` before `coding_tool_apply`
199
+ 3. **Run tests**: If Codex doesn't run tests automatically, run them yourself via `shell_execute` before applying
200
+
201
+ ## When to Choose Codex
202
+
203
+ - Quick, targeted fixes (one file, clear problem)
204
+ - Scripted automation (generate boilerplate, rename across files)
205
+ - CI-friendly operations (no interactive prompts needed)
206
+ - When speed matters more than deep reasoning
207
+
208
+ ## Rules
209
+
210
+ - **DO** use for quick fixes and well-scoped edits
211
+ - **DO** maintain an accurate `AGENTS.md` in project repos
212
+ - **DO** set explicit file boundaries in prompts
213
+ - **DO** default to `gpt-5-codex` for cost-effectiveness
214
+ - **DO NOT** use for large exploratory refactors — use `claude-code`
215
+ - **DO NOT** assume network or install permissions — check sandbox errors
216
+ - **DO NOT** apply changes that touch files outside the stated scope
217
+ - **DO NOT** use `gpt-5.5` by default — its cost is ~4x higher
@@ -0,0 +1,17 @@
1
+ {
2
+ "type": "skill",
3
+ "name": "codex",
4
+ "displayName": "Codex",
5
+ "version": "1.0.0",
6
+ "description": "Use the OpenAI Codex CLI for quick fixes, targeted edits, and non-interactive automation",
7
+ "author": "markus",
8
+ "category": "development",
9
+ "tags": ["coding", "codex", "openai", "automation"],
10
+ "i18n": {
11
+ "zh-CN": {
12
+ "displayName": "Codex",
13
+ "description": "使用 OpenAI Codex CLI 进行快速修复、定向编辑和非交互式自动化"
14
+ }
15
+ },
16
+ "skill": { "skillFile": "SKILL.md" }
17
+ }