corecoder 0.5.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {corecoder-0.5.0 → corecoder-0.6.0}/PKG-INFO +57 -23
  2. {corecoder-0.5.0 → corecoder-0.6.0}/README.md +56 -22
  3. {corecoder-0.5.0 → corecoder-0.6.0}/README_CN.md +57 -23
  4. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/__init__.py +1 -1
  5. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/agent.py +38 -5
  6. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/cli.py +33 -4
  7. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/config.py +15 -17
  8. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/demo.py +28 -12
  9. corecoder-0.6.0/corecoder/hooks.py +85 -0
  10. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/llm.py +12 -106
  11. corecoder-0.6.0/corecoder/mcp.py +208 -0
  12. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/prompt.py +8 -0
  13. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/__init__.py +0 -7
  14. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/agent.py +2 -1
  15. {corecoder-0.5.0 → corecoder-0.6.0}/pyproject.toml +1 -1
  16. corecoder-0.6.0/tests/conftest.py +11 -0
  17. {corecoder-0.5.0 → corecoder-0.6.0}/tests/test_core.py +1 -1
  18. {corecoder-0.5.0 → corecoder-0.6.0}/tests/test_demo.py +2 -2
  19. corecoder-0.6.0/tests/test_hooks.py +145 -0
  20. corecoder-0.6.0/tests/test_mcp.py +198 -0
  21. {corecoder-0.5.0 → corecoder-0.6.0}/tests/test_permissions.py +3 -2
  22. corecoder-0.6.0/tests/test_plan_mode.py +100 -0
  23. {corecoder-0.5.0 → corecoder-0.6.0}/tests/test_tools.py +2 -1
  24. {corecoder-0.5.0 → corecoder-0.6.0}/.github/workflows/ci.yml +0 -0
  25. {corecoder-0.5.0 → corecoder-0.6.0}/.github/workflows/publish.yml +0 -0
  26. {corecoder-0.5.0 → corecoder-0.6.0}/.gitignore +0 -0
  27. {corecoder-0.5.0 → corecoder-0.6.0}/LICENSE +0 -0
  28. {corecoder-0.5.0 → corecoder-0.6.0}/article/00-index.md +0 -0
  29. {corecoder-0.5.0 → corecoder-0.6.0}/article/00-index_EN.md +0 -0
  30. {corecoder-0.5.0 → corecoder-0.6.0}/article/01-the-loop.md +0 -0
  31. {corecoder-0.5.0 → corecoder-0.6.0}/article/01-the-loop_EN.md +0 -0
  32. {corecoder-0.5.0 → corecoder-0.6.0}/article/02-tools.md +0 -0
  33. {corecoder-0.5.0 → corecoder-0.6.0}/article/02-tools_EN.md +0 -0
  34. {corecoder-0.5.0 → corecoder-0.6.0}/article/03-llm-and-cost.md +0 -0
  35. {corecoder-0.5.0 → corecoder-0.6.0}/article/03-llm-and-cost_EN.md +0 -0
  36. {corecoder-0.5.0 → corecoder-0.6.0}/article/04-context.md +0 -0
  37. {corecoder-0.5.0 → corecoder-0.6.0}/article/04-context_EN.md +0 -0
  38. {corecoder-0.5.0 → corecoder-0.6.0}/article/05-parallel-and-subagents.md +0 -0
  39. {corecoder-0.5.0 → corecoder-0.6.0}/article/05-parallel-and-subagents_EN.md +0 -0
  40. {corecoder-0.5.0 → corecoder-0.6.0}/article/06-session-and-cli.md +0 -0
  41. {corecoder-0.5.0 → corecoder-0.6.0}/article/06-session-and-cli_EN.md +0 -0
  42. {corecoder-0.5.0 → corecoder-0.6.0}/article/07-build-your-own.md +0 -0
  43. {corecoder-0.5.0 → corecoder-0.6.0}/article/07-build-your-own_EN.md +0 -0
  44. {corecoder-0.5.0 → corecoder-0.6.0}/assets/demo.png +0 -0
  45. {corecoder-0.5.0 → corecoder-0.6.0}/assets/demo_en.png +0 -0
  46. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/__main__.py +0 -0
  47. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/checkpoints.py +0 -0
  48. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/context.py +0 -0
  49. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/permissions.py +0 -0
  50. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/session.py +0 -0
  51. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/base.py +0 -0
  52. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/bash.py +0 -0
  53. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/edit.py +0 -0
  54. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/glob_tool.py +0 -0
  55. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/grep.py +0 -0
  56. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/read.py +0 -0
  57. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/todo.py +0 -0
  58. {corecoder-0.5.0 → corecoder-0.6.0}/corecoder/tools/write.py +0 -0
  59. {corecoder-0.5.0 → corecoder-0.6.0}/tests/__init__.py +0 -0
  60. {corecoder-0.5.0 → corecoder-0.6.0}/tests/test_checkpoints.py +0 -0
  61. {corecoder-0.5.0 → corecoder-0.6.0}/tests/test_litellm.py +0 -0
  62. {corecoder-0.5.0 → corecoder-0.6.0}/tests/test_session.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: corecoder
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: Minimal AI coding agent (~1,000 lines of Python) inspired by Claude Code. Works with any LLM. (formerly NanoCoder)
5
5
  Project-URL: Homepage, https://github.com/he-yufeng/CoreCoder
6
6
  Project-URL: Repository, https://github.com/he-yufeng/CoreCoder
@@ -37,7 +37,7 @@ Description-Content-Type: text/markdown
37
37
 
38
38
  # CoreCoder
39
39
 
40
- **The nanoGPT of coding agents. 1,217 lines of pure Python — understand how a coding agent actually works, then fork your own.**
40
+ **The nanoGPT of coding agents. 1,161 lines of pure Python: understand how a coding agent actually works, then fork your own.**
41
41
 
42
42
  *learn from it · fork it · ship something better*
43
43
 
@@ -47,7 +47,7 @@ Description-Content-Type: text/markdown
47
47
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
48
48
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
49
49
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
50
- [![engine](https://img.shields.io/badge/engine-1217_LoC-blue)](article/00-index_EN.md)
50
+ [![engine](https://img.shields.io/badge/engine-1161_LoC-blue)](article/00-index_EN.md)
51
51
  [![essays](https://img.shields.io/badge/source--reading-8_bilingual-orange)](article/00-index_EN.md)
52
52
 
53
53
  </div>
@@ -60,7 +60,7 @@ Description-Content-Type: text/markdown
60
60
 
61
61
  | | CoreCoder | Claude Code | aider | nanoGPT |
62
62
  |---|---|---|---|---|
63
- | Lines of code | ~1,217 engine / 2,107 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
63
+ | Lines of code | ~1,161 engine / 2,384 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
64
64
  | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
65
65
  | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
66
66
  | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
@@ -71,9 +71,9 @@ The nanoGPT column is there as a reference point: minimal, readable, but it teac
71
71
 
72
72
  I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
73
73
 
74
- The engine (loop, model interface, context, tools, sessions) is 1,217 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 22 files: 2,107 physical lines, 1,697 net, every one short enough to read in a single sitting.
74
+ The engine (loop, model interface, context, tools, sessions) is 1,161 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 24 files: 2,384 physical lines, 1,931 net, every one short enough to read in a single sitting.
75
75
 
76
- And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first. 119 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
76
+ And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first. 146 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
77
77
 
78
78
  The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
79
79
 
@@ -120,27 +120,29 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
120
120
 
121
121
  ```
122
122
  corecoder/
123
- ├── agent.py agent loop + parallel tool exec 180 lines ← start here
124
- ├── llm.py streaming client + retry + cost 336 lines
123
+ ├── agent.py agent loop + parallel tool exec 213 lines ← start here
124
+ ├── llm.py streaming client + retry + cost 267 lines
125
125
  ├── context.py three-tier context compaction 210 lines
126
126
  ├── session.py save / resume + path-traversal guard 97 lines
127
127
  ├── permissions.py consent for mutating tools 48 lines
128
- ├── prompt.py system prompt 33 lines
129
- ├── cli.py REPL + slash commands + one-shot 317 lines
130
- ├── config.py env-var config 57 lines
128
+ ├── hooks.py Pre/PostToolUse shell hooks 85 lines
129
+ ├── mcp.py MCP stdio client for external tools 208 lines
130
+ ├── prompt.py system prompt 41 lines
131
+ ├── cli.py REPL + slash commands + one-shot 346 lines
132
+ ├── config.py env-var config 55 lines
131
133
  └── tools/
132
- ├── bash.py shell + dangerous-command gate + cd 127 lines
133
- ├── edit.py unique-match search/replace + diff 92 lines
134
- ├── grep.py content search 79 lines
135
- ├── glob_tool.py filename matching 47 lines
136
- ├── read.py file read 53 lines
137
- ├── write.py file write 38 lines
134
+ ├── bash.py shell + dangerous-command gate + cd 131 lines
135
+ ├── edit.py unique-match search/replace + diff 96 lines
136
+ ├── grep.py content search 93 lines
137
+ ├── glob_tool.py filename matching 52 lines
138
+ ├── read.py file read 56 lines
139
+ ├── write.py file write 43 lines
138
140
  ├── todo.py agent-maintained task checklist 79 lines
139
- ├── agent.py sub-agent spawning 63 lines
141
+ ├── agent.py sub-agent spawning 64 lines
140
142
  └── base.py tool base class 27 lines
141
143
  ```
142
144
 
143
- Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
145
+ Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core. If `~/.corecoder/mcp.json` exists, its MCP servers join the eight as extra `mcp__*` tools; the MCP section below covers it.
144
146
 
145
147
  ## A `while` loop is the whole agent
146
148
 
@@ -190,9 +192,9 @@ I also wrote a bilingual source-reading series, one intro plus seven parts, each
190
192
 
191
193
  Once you understand it, the natural next step is to fork. Getting started doesn't take much:
192
194
 
193
- - **Swap in a model you actually use.** It's the two env vars from above; `llm.py` (336 lines) is the entry point for all provider adaptation.
195
+ - **Swap in a model you actually use.** It's the two env vars from above; `llm.py` (267 lines) is the entry point for all provider adaptation.
194
196
  - **Add a tool of your own.** Write a new file against the tool base class in `tools/base.py` (27 lines): run tests, fetch a page, call an LSP, whatever. The end of the second essay walks you through your first one by hand.
195
- - **Rewrite the system prompt.** `prompt.py` is all of 33 lines; change one line and you'll watch the agent's temperament shift. It's the cheapest "change one thing, see a result" in the whole project.
197
+ - **Rewrite the system prompt.** `prompt.py` is all of 41 lines; change one line and you'll watch the agent's temperament shift. It's the cheapest "change one thing, see a result" in the whole project.
196
198
  - **Import it as a library.** The top level exports `Agent`, `LLM`, and `Config`, ready to embed in your own program:
197
199
 
198
200
  ```python
@@ -207,7 +209,7 @@ Going deeper, the directions are out in the open too. None of the following is i
207
209
  - **The dangerous-command blocking in bash is just a regex blacklist.** It guards against slips, not a security sandbox. Facing untrusted input means reaching for seccomp or container isolation. This is the hardest of the four; it goes all the way down to the syscall and isolation layer.
208
210
  - **Retry is only exponential backoff.** No fallback model, no hard dollar budget. Follow `llm.py` down and add a fallback model chain plus a stop-on-over-budget gate; the change stays mostly inside that one file.
209
211
  - **Sub-agents only run the plainest synchronous execution.** Make it async or a streaming executor and you close the exact gap the fifth essay identifies between this and how production agents stream execution.
210
- - **No MCP, no RAG.** Wire up MCP to give it the external tool ecosystem, or add retrieval-based code location for big repos. Both are real ways to grow from a minimal core into your own stronger agent.
212
+ - **No RAG, and the MCP client speaks tools only.** Retrieval-based code location for big repos is still open, and `mcp.py` leaves resources and prompts unimplemented on purpose. Either one is a real way to grow from a minimal core into your own stronger agent.
211
213
 
212
214
  The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
213
215
 
@@ -221,6 +223,7 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
221
223
  /tokens token usage and cost estimate
222
224
  /diff files changed this session
223
225
  /undo revert the most recent file change
226
+ /plan toggle plan mode (read-only, then a plan to approve)
224
227
  /save /sessions save / list sessions
225
228
  quit / exit exit (Ctrl+C cancels the current round)
226
229
  ```
@@ -235,6 +238,37 @@ Read-only tools (`read_file`, `glob`, `grep`, `todo_write`) run the moment the m
235
238
  - In one-shot mode (`-p`) there is nobody to ask, so a mutating call is refused on the spot and the refusal goes back to the model as an ordinary tool result: the loop never hangs on input that can't arrive. Pass `--yes` to approve everything up front (scripts, CI).
236
239
  - The decision itself is pure logic in `permissions.py`, with the terminal only supplying the prompt callback. You can unit-test consent without a TTY, or reuse the layer in your own embedding.
237
240
 
241
+ ## Plan mode
242
+
243
+ `/plan` toggles plan mode in the REPL. While it's on, the prompt shows `(plan)` and every mutating call (writes, edits, bash, MCP tools, sub-agents) is refused on the spot: the refusal goes back to the model as an ordinary tool result, telling it to keep investigating read-only and present a numbered plan instead. When the plan looks right, `approve` (or `/plan` again) hands control back and the agent executes. Mechanically it is one flag on the `Agent` plus one refusal branch ahead of the consent gate, which itself stays untouched; there is no plan file and nothing is remembered between sessions.
244
+
245
+ ## Hooks
246
+
247
+ Drop a `hooks.json` under `~/.corecoder` and your own shell commands run around every tool call, the same idea as Claude Code's hooks:
248
+
249
+ ```json
250
+ {
251
+ "PreToolUse": [{"matcher": "bash", "command": "cat >> ~/.corecoder/audit.jsonl"}],
252
+ "PostToolUse": [{"matcher": "*", "command": "cat >> ~/.corecoder/trace.jsonl"}]
253
+ }
254
+ ```
255
+
256
+ Each hook gets the call as JSON on stdin (`tool_name`, `tool_input`; post hooks also get `tool_response`). The matcher is an exact tool name; empty or `*` fires on every tool. A pre hook can veto the call with exit code 2, and its stderr travels back to the model as the reason so it can route around the block. Post hooks only observe and can never block. A hook that errors or runs past ten seconds is skipped with a warning: hooks assist the loop, they never get to kill it. The whole mechanism is `hooks.py`, and the REPL banner shows how many hooks loaded.
257
+
258
+ ## MCP servers
259
+
260
+ Drop a `mcp.json` under `~/.corecoder` and tools from any MCP server join the agent over stdio, the same config shape as Claude Code's:
261
+
262
+ ```json
263
+ {
264
+ "mcpServers": {
265
+ "filesystem": {"command": "npx", "args": ["-y", "@modelcontextprotocol/server-filesystem", "/tmp"]}
266
+ }
267
+ }
268
+ ```
269
+
270
+ Each configured server starts as a subprocess at launch, handshakes, and lists its tools; every one is registered as `mcp__<server>__<tool>`, so hook matchers and the consent gate treat it exactly like a built-in. MCP tools stay out of the read-only set, meaning the agent asks before running one. The handshake gets fifteen seconds, a call gets sixty, and a server that dies or never answers fails that one call as an ordinary tool result instead of killing the loop. The client speaks the tools slice of the protocol (initialize, tools/list, tools/call) and nothing else, which keeps the whole thing inside `mcp.py` at about 200 lines. With no `mcp.json` there is no MCP and nothing changes.
271
+
238
272
  ## Related Projects
239
273
 
240
274
  If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
@@ -247,7 +281,7 @@ If working through CoreCoder was useful, here are a few other tools I've built a
247
281
 
248
282
  ## Contributing / License
249
283
 
250
- Before you send anything, run `pytest tests/ -q` (119 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
284
+ Before you send anything, run `pytest tests/ -q` (146 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
251
285
 
252
286
  ---
253
287
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  # CoreCoder
4
4
 
5
- **The nanoGPT of coding agents. 1,217 lines of pure Python — understand how a coding agent actually works, then fork your own.**
5
+ **The nanoGPT of coding agents. 1,161 lines of pure Python: understand how a coding agent actually works, then fork your own.**
6
6
 
7
7
  *learn from it · fork it · ship something better*
8
8
 
@@ -12,7 +12,7 @@
12
12
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
13
13
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
14
14
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
15
- [![engine](https://img.shields.io/badge/engine-1217_LoC-blue)](article/00-index_EN.md)
15
+ [![engine](https://img.shields.io/badge/engine-1161_LoC-blue)](article/00-index_EN.md)
16
16
  [![essays](https://img.shields.io/badge/source--reading-8_bilingual-orange)](article/00-index_EN.md)
17
17
 
18
18
  </div>
@@ -25,7 +25,7 @@
25
25
 
26
26
  | | CoreCoder | Claude Code | aider | nanoGPT |
27
27
  |---|---|---|---|---|
28
- | Lines of code | ~1,217 engine / 2,107 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
28
+ | Lines of code | ~1,161 engine / 2,384 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
29
29
  | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
30
30
  | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
31
31
  | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
@@ -36,9 +36,9 @@ The nanoGPT column is there as a reference point: minimal, readable, but it teac
36
36
 
37
37
  I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
38
38
 
39
- The engine (loop, model interface, context, tools, sessions) is 1,217 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 22 files: 2,107 physical lines, 1,697 net, every one short enough to read in a single sitting.
39
+ The engine (loop, model interface, context, tools, sessions) is 1,161 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 24 files: 2,384 physical lines, 1,931 net, every one short enough to read in a single sitting.
40
40
 
41
- And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first. 119 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
41
+ And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first. 146 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
42
42
 
43
43
  The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
44
44
 
@@ -85,27 +85,29 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
85
85
 
86
86
  ```
87
87
  corecoder/
88
- ├── agent.py agent loop + parallel tool exec 180 lines ← start here
89
- ├── llm.py streaming client + retry + cost 336 lines
88
+ ├── agent.py agent loop + parallel tool exec 213 lines ← start here
89
+ ├── llm.py streaming client + retry + cost 267 lines
90
90
  ├── context.py three-tier context compaction 210 lines
91
91
  ├── session.py save / resume + path-traversal guard 97 lines
92
92
  ├── permissions.py consent for mutating tools 48 lines
93
- ├── prompt.py system prompt 33 lines
94
- ├── cli.py REPL + slash commands + one-shot 317 lines
95
- ├── config.py env-var config 57 lines
93
+ ├── hooks.py Pre/PostToolUse shell hooks 85 lines
94
+ ├── mcp.py MCP stdio client for external tools 208 lines
95
+ ├── prompt.py system prompt 41 lines
96
+ ├── cli.py REPL + slash commands + one-shot 346 lines
97
+ ├── config.py env-var config 55 lines
96
98
  └── tools/
97
- ├── bash.py shell + dangerous-command gate + cd 127 lines
98
- ├── edit.py unique-match search/replace + diff 92 lines
99
- ├── grep.py content search 79 lines
100
- ├── glob_tool.py filename matching 47 lines
101
- ├── read.py file read 53 lines
102
- ├── write.py file write 38 lines
99
+ ├── bash.py shell + dangerous-command gate + cd 131 lines
100
+ ├── edit.py unique-match search/replace + diff 96 lines
101
+ ├── grep.py content search 93 lines
102
+ ├── glob_tool.py filename matching 52 lines
103
+ ├── read.py file read 56 lines
104
+ ├── write.py file write 43 lines
103
105
  ├── todo.py agent-maintained task checklist 79 lines
104
- ├── agent.py sub-agent spawning 63 lines
106
+ ├── agent.py sub-agent spawning 64 lines
105
107
  └── base.py tool base class 27 lines
106
108
  ```
107
109
 
108
- Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
110
+ Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core. If `~/.corecoder/mcp.json` exists, its MCP servers join the eight as extra `mcp__*` tools; the MCP section below covers it.
109
111
 
110
112
  ## A `while` loop is the whole agent
111
113
 
@@ -155,9 +157,9 @@ I also wrote a bilingual source-reading series, one intro plus seven parts, each
155
157
 
156
158
  Once you understand it, the natural next step is to fork. Getting started doesn't take much:
157
159
 
158
- - **Swap in a model you actually use.** It's the two env vars from above; `llm.py` (336 lines) is the entry point for all provider adaptation.
160
+ - **Swap in a model you actually use.** It's the two env vars from above; `llm.py` (267 lines) is the entry point for all provider adaptation.
159
161
  - **Add a tool of your own.** Write a new file against the tool base class in `tools/base.py` (27 lines): run tests, fetch a page, call an LSP, whatever. The end of the second essay walks you through your first one by hand.
160
- - **Rewrite the system prompt.** `prompt.py` is all of 33 lines; change one line and you'll watch the agent's temperament shift. It's the cheapest "change one thing, see a result" in the whole project.
162
+ - **Rewrite the system prompt.** `prompt.py` is all of 41 lines; change one line and you'll watch the agent's temperament shift. It's the cheapest "change one thing, see a result" in the whole project.
161
163
  - **Import it as a library.** The top level exports `Agent`, `LLM`, and `Config`, ready to embed in your own program:
162
164
 
163
165
  ```python
@@ -172,7 +174,7 @@ Going deeper, the directions are out in the open too. None of the following is i
172
174
  - **The dangerous-command blocking in bash is just a regex blacklist.** It guards against slips, not a security sandbox. Facing untrusted input means reaching for seccomp or container isolation. This is the hardest of the four; it goes all the way down to the syscall and isolation layer.
173
175
  - **Retry is only exponential backoff.** No fallback model, no hard dollar budget. Follow `llm.py` down and add a fallback model chain plus a stop-on-over-budget gate; the change stays mostly inside that one file.
174
176
  - **Sub-agents only run the plainest synchronous execution.** Make it async or a streaming executor and you close the exact gap the fifth essay identifies between this and how production agents stream execution.
175
- - **No MCP, no RAG.** Wire up MCP to give it the external tool ecosystem, or add retrieval-based code location for big repos. Both are real ways to grow from a minimal core into your own stronger agent.
177
+ - **No RAG, and the MCP client speaks tools only.** Retrieval-based code location for big repos is still open, and `mcp.py` leaves resources and prompts unimplemented on purpose. Either one is a real way to grow from a minimal core into your own stronger agent.
176
178
 
177
179
  The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
178
180
 
@@ -186,6 +188,7 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
186
188
  /tokens token usage and cost estimate
187
189
  /diff files changed this session
188
190
  /undo revert the most recent file change
191
+ /plan toggle plan mode (read-only, then a plan to approve)
189
192
  /save /sessions save / list sessions
190
193
  quit / exit exit (Ctrl+C cancels the current round)
191
194
  ```
@@ -200,6 +203,37 @@ Read-only tools (`read_file`, `glob`, `grep`, `todo_write`) run the moment the m
200
203
  - In one-shot mode (`-p`) there is nobody to ask, so a mutating call is refused on the spot and the refusal goes back to the model as an ordinary tool result: the loop never hangs on input that can't arrive. Pass `--yes` to approve everything up front (scripts, CI).
201
204
  - The decision itself is pure logic in `permissions.py`, with the terminal only supplying the prompt callback. You can unit-test consent without a TTY, or reuse the layer in your own embedding.
202
205
 
206
+ ## Plan mode
207
+
208
+ `/plan` toggles plan mode in the REPL. While it's on, the prompt shows `(plan)` and every mutating call (writes, edits, bash, MCP tools, sub-agents) is refused on the spot: the refusal goes back to the model as an ordinary tool result, telling it to keep investigating read-only and present a numbered plan instead. When the plan looks right, `approve` (or `/plan` again) hands control back and the agent executes. Mechanically it is one flag on the `Agent` plus one refusal branch ahead of the consent gate, which itself stays untouched; there is no plan file and nothing is remembered between sessions.
209
+
210
+ ## Hooks
211
+
212
+ Drop a `hooks.json` under `~/.corecoder` and your own shell commands run around every tool call, the same idea as Claude Code's hooks:
213
+
214
+ ```json
215
+ {
216
+ "PreToolUse": [{"matcher": "bash", "command": "cat >> ~/.corecoder/audit.jsonl"}],
217
+ "PostToolUse": [{"matcher": "*", "command": "cat >> ~/.corecoder/trace.jsonl"}]
218
+ }
219
+ ```
220
+
221
+ Each hook gets the call as JSON on stdin (`tool_name`, `tool_input`; post hooks also get `tool_response`). The matcher is an exact tool name; empty or `*` fires on every tool. A pre hook can veto the call with exit code 2, and its stderr travels back to the model as the reason so it can route around the block. Post hooks only observe and can never block. A hook that errors or runs past ten seconds is skipped with a warning: hooks assist the loop, they never get to kill it. The whole mechanism is `hooks.py`, and the REPL banner shows how many hooks loaded.
222
+
223
+ ## MCP servers
224
+
225
+ Drop a `mcp.json` under `~/.corecoder` and tools from any MCP server join the agent over stdio, the same config shape as Claude Code's:
226
+
227
+ ```json
228
+ {
229
+ "mcpServers": {
230
+ "filesystem": {"command": "npx", "args": ["-y", "@modelcontextprotocol/server-filesystem", "/tmp"]}
231
+ }
232
+ }
233
+ ```
234
+
235
+ Each configured server starts as a subprocess at launch, handshakes, and lists its tools; every one is registered as `mcp__<server>__<tool>`, so hook matchers and the consent gate treat it exactly like a built-in. MCP tools stay out of the read-only set, meaning the agent asks before running one. The handshake gets fifteen seconds, a call gets sixty, and a server that dies or never answers fails that one call as an ordinary tool result instead of killing the loop. The client speaks the tools slice of the protocol (initialize, tools/list, tools/call) and nothing else, which keeps the whole thing inside `mcp.py` at about 200 lines. With no `mcp.json` there is no MCP and nothing changes.
236
+
203
237
  ## Related Projects
204
238
 
205
239
  If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
@@ -212,7 +246,7 @@ If working through CoreCoder was useful, here are a few other tools I've built a
212
246
 
213
247
  ## Contributing / License
214
248
 
215
- Before you send anything, run `pytest tests/ -q` (119 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
249
+ Before you send anything, run `pytest tests/ -q` (146 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
216
250
 
217
251
  ---
218
252
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  # CoreCoder
4
4
 
5
- **编程 agent 里的 nanoGPT。1217 行纯 Python,读懂一个 coding agent 到底怎么运作,再 fork 出你自己的。**
5
+ **编程 agent 里的 nanoGPT。1161 行纯 Python,读懂一个 coding agent 到底怎么运作,再 fork 出你自己的。**
6
6
 
7
7
  *learn from it · fork it · ship something better*
8
8
 
@@ -12,7 +12,7 @@
12
12
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
13
13
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
14
14
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
15
- [![engine](https://img.shields.io/badge/engine-1217_LoC-blue)](article/)
15
+ [![engine](https://img.shields.io/badge/engine-1161_LoC-blue)](article/)
16
16
  [![源码导读](https://img.shields.io/badge/源码导读-8篇双语-orange)](article/)
17
17
 
18
18
  </div>
@@ -25,7 +25,7 @@
25
25
 
26
26
  | | CoreCoder | Claude Code | aider | nanoGPT |
27
27
  |---|---|---|---|---|
28
- | 代码量 | 引擎约 1217 行 / 整包 2107 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
28
+ | 代码量 | 引擎约 1161 行 / 整包 2384 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
29
29
  | 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
30
30
  | 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
31
31
  | 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
@@ -36,9 +36,9 @@ nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个
36
36
 
37
37
  我一直觉得 coding agent 被讲得太玄了。把 Claude Code、Cursor 这类工具扒到底,核心是一个 while 循环套着一个大模型,外加七八个让它能真正动手的工具。难的从来不是这个循环,而是循环跑进真实世界以后要兜的那些底。CoreCoder 就是把这个核心老老实实写出来的最小版本。
38
38
 
39
- 引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1217 行。连最外层的 CLI、配置、打包一起算,整个包 22 个文件、物理 2107 行、净 1697 行,每个文件都短到能一口气读完。
39
+ 引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1161 行。连最外层的 CLI、配置、打包一起算,整个包 24 个文件、物理 2384 行、净 1931 行,每个文件都短到能一口气读完。
40
40
 
41
- 它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你。任何要动你磁盘、要跑命令的调用,都会先停下来等你点头,119 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
41
+ 它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你。任何要动你磁盘、要跑命令的调用,都会先停下来等你点头,146 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
42
42
 
43
43
  代码来自一次公开拆解。公开的源码分析里,Claude Code 这类生产级 agent 暴露出不少关键架构,我挑出最核心的一层,用尽量少的代码诚实地复写了一遍。所以读 CoreCoder,约等于读一份基于公开源码分析的「可运行注释版」:讲的是这类 agent 的核心思路,而它本身只是最小复写,就摆在你机器上,随你拆、随你改。
44
44
 
@@ -85,27 +85,29 @@ corecoder -p "给 parse_config() 加错误处理" # 一次性模式,干完
85
85
 
86
86
  ```
87
87
  corecoder/
88
- ├── agent.py agent 主循环 + 并行工具执行 180 行 ← 从这里开始读
89
- ├── llm.py 流式客户端 + 重试 + 成本统计 336 行
88
+ ├── agent.py agent 主循环 + 并行工具执行 213 行 ← 从这里开始读
89
+ ├── llm.py 流式客户端 + 重试 + 成本统计 267 行
90
90
  ├── context.py 三层上下文压缩 210 行
91
91
  ├── session.py 会话存盘 / 续聊 + 路径穿越防护 97 行
92
92
  ├── permissions.py 改动类工具的用户授权 48 行
93
- ├── prompt.py 系统提示词 33 行
94
- ├── cli.py REPL + 斜杠命令 + 一次性模式 317 行
95
- ├── config.py 环境变量配置 57 行
93
+ ├── hooks.py 工具调用前后的用户 shell 钩子 85 行
94
+ ├── mcp.py MCP stdio 客户端,接外部工具 208 行
95
+ ├── prompt.py 系统提示词 41 行
96
+ ├── cli.py REPL + 斜杠命令 + 一次性模式 346 行
97
+ ├── config.py 环境变量配置 55 行
96
98
  └── tools/
97
- ├── bash.py shell + 危险命令闸 + cd 追踪 127 行
98
- ├── edit.py 唯一匹配搜索替换 + diff 92 行
99
- ├── grep.py 内容搜索 79 行
100
- ├── glob_tool.py 文件名匹配 47 行
101
- ├── read.py 文件读取 53 行
102
- ├── write.py 文件写入 38 行
103
- ├── todo.py agent 自维护的任务清单 79 行
104
- ├── agent.py 子 agent 派生 63 行
99
+ ├── bash.py shell + 危险命令闸 + cd 追踪 131 行
100
+ ├── edit.py 唯一匹配搜索替换 + diff 96 行
101
+ ├── grep.py 内容搜索 93 行
102
+ ├── glob_tool.py 文件名匹配 52 行
103
+ ├── read.py 文件读取 56 行
104
+ ├── write.py 文件写入 43 行
105
+ ├── todo.py agent 自维护的任务清单 79 行
106
+ ├── agent.py 子 agent 派生 64 行
105
107
  └── base.py 工具基类 27 行
106
108
  ```
107
109
 
108
- 八个工具:`bash`、`read_file`、`write_file`、`edit_file`、`glob`、`grep`、`todo_write`(agent 自己维护的任务清单)、`agent`(派子 agent)。其余都是包在引擎核心外面的 CLI 外壳、配置和打包。
110
+ 八个工具:`bash`、`read_file`、`write_file`、`edit_file`、`glob`、`grep`、`todo_write`(agent 自己维护的任务清单)、`agent`(派子 agent)。其余都是包在引擎核心外面的 CLI 外壳、配置和打包。存在 `~/.corecoder/mcp.json` 时,里面的 MCP 服务器会以 `mcp__*` 工具的身份并进这八件里,下面有专门一节讲。
109
111
 
110
112
  ## 一个 while 循环就是 agent 的本体
111
113
 
@@ -155,9 +157,9 @@ def chat(self, user_input):
155
157
 
156
158
  读懂之后,最自然的下一步就是 fork。起手不用伤筋动骨:
157
159
 
158
- - **换个你常用的模型。** 就是上面那两个环境变量,`llm.py`(336 行)是所有 provider 适配的入口。
160
+ - **换个你常用的模型。** 就是上面那两个环境变量,`llm.py`(267 行)是所有 provider 适配的入口。
159
161
  - **加一件你自己的工具。** 照 `tools/base.py`(27 行)的工具基类写个新文件,跑测试、抓网页、调 LSP 都行,第二篇文章末尾手把手带你写第一个。
160
- - **改系统提示词。** `prompt.py` 才 33 行,改一句就能看到 agent 的脾气变了,是门槛最低的「改一处就有反馈」。
162
+ - **改系统提示词。** `prompt.py` 才 41 行,改一句就能看到 agent 的脾气变了,是门槛最低的「改一处就有反馈」。
161
163
  - **直接当库 import。** 顶层导出了 `Agent`、`LLM`、`Config`,能嵌进你自己的程序:
162
164
 
163
165
  ```python
@@ -172,7 +174,7 @@ print(Agent(llm=llm).chat("找出项目里所有 TODO 注释并列出来"))
172
174
  - **bash 的危险命令拦截只是正则黑名单。** 防手滑,不是安全沙箱。要面对不可信输入,就得上 seccomp 或容器隔离。这条最硬,要一路走到系统调用和隔离那一层。
173
175
  - **重试只做了指数退避。** 没有 fallback 模型,也没有美元硬预算。顺着 `llm.py` 往下,加一条 fallback 模型链和超预算自动停的闸,改动基本就集中在这一个文件。
174
176
  - **子 agent 只有最朴素的同步执行。** 做成异步或流式执行器,正好补上第五篇点名的、相对生产级 agent 流式执行的那段差距。
175
- - **不做 MCP,不做 RAG。** 接上 MCP 让它用上外部工具生态,或给大仓加检索式的代码定位,都是从「最小核心」往「你自己的更强 agent」扩的真实方向。
177
+ - **不做 RAG,MCP 客户端也只讲工具这一小片。** 给大仓加检索式代码定位还空着,`mcp.py` 也特意没实现 resources 和 prompts。随便挑一个,都是从最小核心往你自己的更强 agent 扩的真实方向。
176
178
 
177
179
  README 只给方向,每条的代码细节第七篇接着讲。挑一个动手,就是把它做得更好的开始。
178
180
 
@@ -186,6 +188,7 @@ README 只给方向,每条的代码细节第七篇接着讲。挑一个动手
186
188
  /tokens 查看 token 用量和费用估算
187
189
  /diff 查看本次会话改过的文件
188
190
  /undo 撤销最近一次文件改动
191
+ /plan 开关计划模式(只读摸底,再交出待批准的计划)
189
192
  /save /sessions 保存 / 列出会话
190
193
  quit / exit 退出(Ctrl+C 取消当前回合)
191
194
  ```
@@ -200,6 +203,37 @@ quit / exit 退出(Ctrl+C 取消当前回合)
200
203
  - 一次性模式(`-p`)没人可问,改动类调用当场被拒,拒绝理由作为普通工具结果回给模型:循环绝不会卡在等一个永远不会来的输入上。要全部预授权就加 `--yes`(脚本、CI 场景)。
201
204
  - 判断本身是 `permissions.py` 里的纯逻辑,终端只是塞进来一个提问回调。不用 TTY 也能单测授权逻辑,或者直接搬进你自己的嵌入场景。
202
205
 
206
+ ## 计划模式
207
+
208
+ REPL 里 `/plan` 开关计划模式。开着的时候,提示符变成 `(plan)`,一切改动类调用(写文件、编辑、bash、MCP 工具、子 agent)当场被拒:拒绝理由作为普通工具结果回给模型,让它用只读工具继续摸底,给出一份编号计划。计划看着没问题,敲 `approve`(或再敲一次 `/plan`)就把执行权交回去,agent 接着干活。机制上就是 `Agent` 身上的一个开关,加授权闸前面多出来的一支拒绝分支,授权闸本身一行没动;不落盘计划文件,会话之间也不记任何东西。
209
+
210
+ ## 钩子
211
+
212
+ 在 `~/.corecoder` 下放一个 `hooks.json`,就能让你自己的 shell 命令在每次工具调用前后跑起来,思路和 Claude Code 的 hooks 一致:
213
+
214
+ ```json
215
+ {
216
+ "PreToolUse": [{"matcher": "bash", "command": "cat >> ~/.corecoder/audit.jsonl"}],
217
+ "PostToolUse": [{"matcher": "*", "command": "cat >> ~/.corecoder/trace.jsonl"}]
218
+ }
219
+ ```
220
+
221
+ 每个钩子从 stdin 拿到这次调用的 JSON(`tool_name`、`tool_input`,post 钩子还带 `tool_response`)。matcher 是精确的工具名,留空或写 `*` 表示对所有工具生效。pre 钩子可以否决这次调用:退出码 2,它的 stderr 会作为理由回给模型,让它换条路走。post 钩子只观察,永远拦不住。钩子报错或超过十秒会被跳过并记一条警告:钩子是来帮忙的,没权力弄死循环。整个机制就是 `hooks.py` 一个文件,REPL 启动横幅会显示加载了几条。
222
+
223
+ ## MCP 服务器
224
+
225
+ 在 `~/.corecoder` 下放一个 `mcp.json`,任何 MCP 服务器的工具就能通过 stdio 接进 agent,配置形状和 Claude Code 的一样:
226
+
227
+ ```json
228
+ {
229
+ "mcpServers": {
230
+ "filesystem": {"command": "npx", "args": ["-y", "@modelcontextprotocol/server-filesystem", "/tmp"]}
231
+ }
232
+ }
233
+ ```
234
+
235
+ 每个配好的服务器在启动时拉起一个子进程,握手、列出工具;每件工具都注册成 `mcp__<服务器>__<工具>`,钩子匹配和授权闸对它和内建工具一视同仁。MCP 工具不在只读名单里,模型要调,得先问过你。握手给十五秒,一次调用给六十秒;服务器挂了或者迟迟不应,那一次调用就以普通工具结果的形式报错,循环照常往下走。客户端只实现协议里工具那一小片(initialize、tools/list、tools/call),别的一概不碰,所以整块实现收在 `mcp.py` 一个文件里,两百行出头。没有 `mcp.json` 就没有 MCP,一切照旧。
236
+
203
237
  ## 相关项目
204
238
 
205
239
  如果你读 CoreCoder 读得还顺,下面几个我做的 agent / LLM 系统方向的工具也许用得上:
@@ -212,7 +246,7 @@ quit / exit 退出(Ctrl+C 取消当前回合)
212
246
 
213
247
  ## 贡献 / License
214
248
 
215
- 动手之前先跑一遍 `pytest tests/ -q`(119 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
249
+ 动手之前先跑一遍 `pytest tests/ -q`(146 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
216
250
 
217
251
  ---
218
252
 
@@ -1,6 +1,6 @@
1
1
  """CoreCoder - Minimal AI coding agent inspired by Claude Code's architecture."""
2
2
 
3
- __version__ = "0.5.0"
3
+ __version__ = "0.6.0"
4
4
 
5
5
  from corecoder.agent import Agent
6
6
  from corecoder.config import Config
@@ -14,7 +14,8 @@ import inspect
14
14
 
15
15
  from .context import ContextManager
16
16
  from .llm import LLM
17
- from .prompt import system_prompt
17
+ from .permissions import Permission
18
+ from .prompt import PLAN_MODE_PROMPT, system_prompt
18
19
  from .tools import ALL_TOOLS
19
20
  from .tools.agent import AgentTool
20
21
  from .tools.base import Tool
@@ -29,15 +30,18 @@ class Agent:
29
30
  max_context_tokens: int = 128_000,
30
31
  max_rounds: int = 50,
31
32
  permission=None,
33
+ hooks=None,
32
34
  ):
33
35
  self.llm = llm
34
36
  self.tools = tools if tools is not None else ALL_TOOLS
35
37
  self.permission = permission
38
+ self.hooks = hooks
36
39
  self._tool_by_name = {t.name: t for t in self.tools}
37
40
  self.messages: list[dict] = []
38
41
  self.context = ContextManager(max_tokens=max_context_tokens)
39
42
  self.max_rounds = max_rounds
40
43
  self._system = system_prompt(self.tools)
44
+ self.plan_mode = False # toggled by /plan; while on, mutating tools are refused
41
45
 
42
46
  # wire up sub-agent capability
43
47
  for t in self.tools:
@@ -48,6 +52,10 @@ class Agent:
48
52
 
49
53
  def _full_messages(self) -> list[dict]:
50
54
  system = self._system
55
+ # re-injected every round, like the task list below, so a toggle made
56
+ # between turns takes effect on the very next request
57
+ if self.plan_mode:
58
+ system += "\n\n" + PLAN_MODE_PROMPT
51
59
  # the task list is re-injected every round, so the model always sees the
52
60
  # current state rather than a stale copy buried in old tool results
53
61
  if self._todo is not None:
@@ -85,7 +93,10 @@ class Agent:
85
93
  tc = resp.tool_calls[0]
86
94
  if on_tool:
87
95
  on_tool(tc.name, tc.arguments)
88
- result = self._permit(tc) or self._exec_tool(tc)
96
+ result = self._pre_hooks(tc) or self._permit(tc)
97
+ if result is None:
98
+ result = self._exec_tool(tc)
99
+ self._post_hooks(tc, result)
89
100
  self.messages.append({
90
101
  "role": "tool",
91
102
  "tool_call_id": tc.id,
@@ -111,9 +122,29 @@ class Agent:
111
122
 
112
123
  return "(reached maximum tool-call rounds)"
113
124
 
125
+ def _pre_hooks(self, tc) -> str | None:
126
+ """PreToolUse hooks, fired before consent. A string return blocks the
127
+ call and becomes the tool result the model sees; None lets it through."""
128
+ if self.hooks is None:
129
+ return None
130
+ return self.hooks.run_pre(tc.name, tc.arguments)
131
+
132
+ def _post_hooks(self, tc, result: str):
133
+ """PostToolUse hooks observe a finished call; they can never block."""
134
+ if self.hooks is not None:
135
+ self.hooks.run_post(tc.name, tc.arguments, result)
136
+
114
137
  def _permit(self, tc) -> str | None:
115
138
  """Consent check for one call. None means go ahead; a string is the
116
139
  refusal, returned as the tool result instead of executing."""
140
+ # plan mode outranks consent, even --yes: while it's on nothing mutates
141
+ if self.plan_mode and tc.name not in Permission.READ_ONLY:
142
+ return (
143
+ "Plan mode is on, so this call was refused: plan mode is "
144
+ "read-only. Do not retry it. Keep investigating with the "
145
+ "read-only tools, then present the plan and stop. The user can "
146
+ 'approve it by typing "approve", or exit plan mode with /plan.'
147
+ )
117
148
  if self.permission is None:
118
149
  return None
119
150
  return self.permission.check(tc.name, tc.arguments)
@@ -146,9 +177,9 @@ class Agent:
146
177
  if on_tool:
147
178
  on_tool(tc.name, tc.arguments)
148
179
 
149
- # consent is settled up front on this thread: prompting from pool
150
- # workers would interleave several prompts on one terminal
151
- results = [self._permit(tc) for tc in tool_calls]
180
+ # hooks and consent are settled up front on this thread: prompting
181
+ # from pool workers would interleave several prompts on one terminal
182
+ results = [self._pre_hooks(tc) or self._permit(tc) for tc in tool_calls]
152
183
  with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool:
153
184
  futures = {
154
185
  i: pool.submit(self._exec_tool, tc)
@@ -157,6 +188,8 @@ class Agent:
157
188
  }
158
189
  for i, future in futures.items():
159
190
  results[i] = future.result()
191
+ for i in futures:
192
+ self._post_hooks(tool_calls[i], results[i])
160
193
  return results
161
194
 
162
195
  def _answer_pending_tool_calls(self, tool_calls):