corecoder 0.4.1__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. {corecoder-0.4.1 → corecoder-0.5.0}/PKG-INFO +20 -10
  2. {corecoder-0.4.1 → corecoder-0.5.0}/README.md +19 -9
  3. {corecoder-0.4.1 → corecoder-0.5.0}/README_CN.md +19 -9
  4. {corecoder-0.4.1 → corecoder-0.5.0}/article/00-index.md +1 -1
  5. {corecoder-0.4.1 → corecoder-0.5.0}/article/00-index_EN.md +1 -1
  6. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/__init__.py +1 -1
  7. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/agent.py +21 -3
  8. corecoder-0.5.0/corecoder/checkpoints.py +38 -0
  9. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/cli.py +40 -1
  10. corecoder-0.5.0/corecoder/permissions.py +48 -0
  11. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/agent.py +2 -0
  12. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/edit.py +2 -0
  13. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/write.py +2 -0
  14. {corecoder-0.4.1 → corecoder-0.5.0}/pyproject.toml +1 -1
  15. corecoder-0.5.0/tests/test_checkpoints.py +54 -0
  16. {corecoder-0.4.1 → corecoder-0.5.0}/tests/test_core.py +37 -3
  17. corecoder-0.5.0/tests/test_permissions.py +181 -0
  18. {corecoder-0.4.1 → corecoder-0.5.0}/.github/workflows/ci.yml +0 -0
  19. {corecoder-0.4.1 → corecoder-0.5.0}/.github/workflows/publish.yml +0 -0
  20. {corecoder-0.4.1 → corecoder-0.5.0}/.gitignore +0 -0
  21. {corecoder-0.4.1 → corecoder-0.5.0}/LICENSE +0 -0
  22. {corecoder-0.4.1 → corecoder-0.5.0}/article/01-the-loop.md +0 -0
  23. {corecoder-0.4.1 → corecoder-0.5.0}/article/01-the-loop_EN.md +0 -0
  24. {corecoder-0.4.1 → corecoder-0.5.0}/article/02-tools.md +0 -0
  25. {corecoder-0.4.1 → corecoder-0.5.0}/article/02-tools_EN.md +0 -0
  26. {corecoder-0.4.1 → corecoder-0.5.0}/article/03-llm-and-cost.md +0 -0
  27. {corecoder-0.4.1 → corecoder-0.5.0}/article/03-llm-and-cost_EN.md +0 -0
  28. {corecoder-0.4.1 → corecoder-0.5.0}/article/04-context.md +0 -0
  29. {corecoder-0.4.1 → corecoder-0.5.0}/article/04-context_EN.md +0 -0
  30. {corecoder-0.4.1 → corecoder-0.5.0}/article/05-parallel-and-subagents.md +0 -0
  31. {corecoder-0.4.1 → corecoder-0.5.0}/article/05-parallel-and-subagents_EN.md +0 -0
  32. {corecoder-0.4.1 → corecoder-0.5.0}/article/06-session-and-cli.md +0 -0
  33. {corecoder-0.4.1 → corecoder-0.5.0}/article/06-session-and-cli_EN.md +0 -0
  34. {corecoder-0.4.1 → corecoder-0.5.0}/article/07-build-your-own.md +0 -0
  35. {corecoder-0.4.1 → corecoder-0.5.0}/article/07-build-your-own_EN.md +0 -0
  36. {corecoder-0.4.1 → corecoder-0.5.0}/assets/demo.png +0 -0
  37. {corecoder-0.4.1 → corecoder-0.5.0}/assets/demo_en.png +0 -0
  38. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/__main__.py +0 -0
  39. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/config.py +0 -0
  40. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/context.py +0 -0
  41. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/demo.py +0 -0
  42. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/llm.py +0 -0
  43. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/prompt.py +0 -0
  44. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/session.py +0 -0
  45. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/__init__.py +0 -0
  46. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/base.py +0 -0
  47. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/bash.py +0 -0
  48. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/glob_tool.py +0 -0
  49. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/grep.py +0 -0
  50. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/read.py +0 -0
  51. {corecoder-0.4.1 → corecoder-0.5.0}/corecoder/tools/todo.py +0 -0
  52. {corecoder-0.4.1 → corecoder-0.5.0}/tests/__init__.py +0 -0
  53. {corecoder-0.4.1 → corecoder-0.5.0}/tests/test_demo.py +0 -0
  54. {corecoder-0.4.1 → corecoder-0.5.0}/tests/test_litellm.py +0 -0
  55. {corecoder-0.4.1 → corecoder-0.5.0}/tests/test_session.py +0 -0
  56. {corecoder-0.4.1 → corecoder-0.5.0}/tests/test_tools.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: corecoder
3
- Version: 0.4.1
3
+ Version: 0.5.0
4
4
  Summary: Minimal AI coding agent (~1,000 lines of Python) inspired by Claude Code. Works with any LLM. (formerly NanoCoder)
5
5
  Project-URL: Homepage, https://github.com/he-yufeng/CoreCoder
6
6
  Project-URL: Repository, https://github.com/he-yufeng/CoreCoder
@@ -37,7 +37,7 @@ Description-Content-Type: text/markdown
37
37
 
38
38
  # CoreCoder
39
39
 
40
- **The nanoGPT of coding agents. 1,081 lines of pure Python — understand how a coding agent actually works, then fork your own.**
40
+ **The nanoGPT of coding agents. 1,217 lines of pure Python — understand how a coding agent actually works, then fork your own.**
41
41
 
42
42
  *learn from it · fork it · ship something better*
43
43
 
@@ -47,7 +47,7 @@ Description-Content-Type: text/markdown
47
47
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
48
48
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
49
49
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
50
- [![engine](https://img.shields.io/badge/engine-1081_LoC-blue)](article/00-index_EN.md)
50
+ [![engine](https://img.shields.io/badge/engine-1217_LoC-blue)](article/00-index_EN.md)
51
51
  [![essays](https://img.shields.io/badge/source--reading-8_bilingual-orange)](article/00-index_EN.md)
52
52
 
53
53
  </div>
@@ -60,7 +60,7 @@ Description-Content-Type: text/markdown
60
60
 
61
61
  | | CoreCoder | Claude Code | aider | nanoGPT |
62
62
  |---|---|---|---|---|
63
- | Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
63
+ | Lines of code | ~1,217 engine / 2,107 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
64
64
  | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
65
65
  | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
66
66
  | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
@@ -71,9 +71,9 @@ The nanoGPT column is there as a reference point: minimal, readable, but it teac
71
71
 
72
72
  I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
73
73
 
74
- The engine (loop, model interface, context, tools, sessions) is 1,081 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 18 files: 1,714 physical lines, 1,385 net, every one short enough to read in a single sitting.
74
+ The engine (loop, model interface, context, tools, sessions) is 1,217 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 22 files: 2,107 physical lines, 1,697 net, every one short enough to read in a single sitting.
75
75
 
76
- And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
76
+ And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first. 119 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
77
77
 
78
78
  The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
79
79
 
@@ -120,12 +120,13 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
120
120
 
121
121
  ```
122
122
  corecoder/
123
- ├── agent.py agent loop + parallel tool exec 162 lines ← start here
123
+ ├── agent.py agent loop + parallel tool exec 180 lines ← start here
124
124
  ├── llm.py streaming client + retry + cost 336 lines
125
125
  ├── context.py three-tier context compaction 210 lines
126
126
  ├── session.py save / resume + path-traversal guard 97 lines
127
+ ├── permissions.py consent for mutating tools 48 lines
127
128
  ├── prompt.py system prompt 33 lines
128
- ├── cli.py REPL + slash commands + one-shot 270 lines
129
+ ├── cli.py REPL + slash commands + one-shot 317 lines
129
130
  ├── config.py env-var config 57 lines
130
131
  └── tools/
131
132
  ├── bash.py shell + dangerous-command gate + cd 127 lines
@@ -135,7 +136,7 @@ corecoder/
135
136
  ├── read.py file read 53 lines
136
137
  ├── write.py file write 38 lines
137
138
  ├── todo.py agent-maintained task checklist 79 lines
138
- ├── agent.py sub-agent spawning 58 lines
139
+ ├── agent.py sub-agent spawning 63 lines
139
140
  └── base.py tool base class 27 lines
140
141
  ```
141
142
 
@@ -219,12 +220,21 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
219
220
  /compact compact the context by hand
220
221
  /tokens token usage and cost estimate
221
222
  /diff files changed this session
223
+ /undo revert the most recent file change
222
224
  /save /sessions save / list sessions
223
225
  quit / exit exit (Ctrl+C cancels the current round)
224
226
  ```
225
227
 
226
228
  Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
227
229
 
230
+ ## Permissions
231
+
232
+ Read-only tools (`read_file`, `glob`, `grep`, `todo_write`) run the moment the model asks. The mutating ones (`edit_file`, `write_file`, `bash`, and spawning a sub-agent) stop for consent first, and the REPL banner shows which mode you're in:
233
+
234
+ - In the REPL you get one prompt per call: allow once, always allow this tool, or deny. "Always" is remembered per tool for the rest of the session, and a sub-agent inherits the same layer, so consent follows the work wherever it happens.
235
+ - In one-shot mode (`-p`) there is nobody to ask, so a mutating call is refused on the spot and the refusal goes back to the model as an ordinary tool result: the loop never hangs on input that can't arrive. Pass `--yes` to approve everything up front (scripts, CI).
236
+ - The decision itself is pure logic in `permissions.py`, with the terminal only supplying the prompt callback. You can unit-test consent without a TTY, or reuse the layer in your own embedding.
237
+
228
238
  ## Related Projects
229
239
 
230
240
  If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
@@ -237,7 +247,7 @@ If working through CoreCoder was useful, here are a few other tools I've built a
237
247
 
238
248
  ## Contributing / License
239
249
 
240
- Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
250
+ Before you send anything, run `pytest tests/ -q` (119 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
241
251
 
242
252
  ---
243
253
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  # CoreCoder
4
4
 
5
- **The nanoGPT of coding agents. 1,081 lines of pure Python — understand how a coding agent actually works, then fork your own.**
5
+ **The nanoGPT of coding agents. 1,217 lines of pure Python — understand how a coding agent actually works, then fork your own.**
6
6
 
7
7
  *learn from it · fork it · ship something better*
8
8
 
@@ -12,7 +12,7 @@
12
12
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
13
13
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
14
14
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
15
- [![engine](https://img.shields.io/badge/engine-1081_LoC-blue)](article/00-index_EN.md)
15
+ [![engine](https://img.shields.io/badge/engine-1217_LoC-blue)](article/00-index_EN.md)
16
16
  [![essays](https://img.shields.io/badge/source--reading-8_bilingual-orange)](article/00-index_EN.md)
17
17
 
18
18
  </div>
@@ -25,7 +25,7 @@
25
25
 
26
26
  | | CoreCoder | Claude Code | aider | nanoGPT |
27
27
  |---|---|---|---|---|
28
- | Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
28
+ | Lines of code | ~1,217 engine / 2,107 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
29
29
  | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
30
30
  | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
31
31
  | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
@@ -36,9 +36,9 @@ The nanoGPT column is there as a reference point: minimal, readable, but it teac
36
36
 
37
37
  I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
38
38
 
39
- The engine (loop, model interface, context, tools, sessions) is 1,081 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 18 files: 1,714 physical lines, 1,385 net, every one short enough to read in a single sitting.
39
+ The engine (loop, model interface, context, tools, sessions) is 1,217 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 22 files: 2,107 physical lines, 1,697 net, every one short enough to read in a single sitting.
40
40
 
41
- And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
41
+ And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first. 119 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
42
42
 
43
43
  The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
44
44
 
@@ -85,12 +85,13 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
85
85
 
86
86
  ```
87
87
  corecoder/
88
- ├── agent.py agent loop + parallel tool exec 162 lines ← start here
88
+ ├── agent.py agent loop + parallel tool exec 180 lines ← start here
89
89
  ├── llm.py streaming client + retry + cost 336 lines
90
90
  ├── context.py three-tier context compaction 210 lines
91
91
  ├── session.py save / resume + path-traversal guard 97 lines
92
+ ├── permissions.py consent for mutating tools 48 lines
92
93
  ├── prompt.py system prompt 33 lines
93
- ├── cli.py REPL + slash commands + one-shot 270 lines
94
+ ├── cli.py REPL + slash commands + one-shot 317 lines
94
95
  ├── config.py env-var config 57 lines
95
96
  └── tools/
96
97
  ├── bash.py shell + dangerous-command gate + cd 127 lines
@@ -100,7 +101,7 @@ corecoder/
100
101
  ├── read.py file read 53 lines
101
102
  ├── write.py file write 38 lines
102
103
  ├── todo.py agent-maintained task checklist 79 lines
103
- ├── agent.py sub-agent spawning 58 lines
104
+ ├── agent.py sub-agent spawning 63 lines
104
105
  └── base.py tool base class 27 lines
105
106
  ```
106
107
 
@@ -184,12 +185,21 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
184
185
  /compact compact the context by hand
185
186
  /tokens token usage and cost estimate
186
187
  /diff files changed this session
188
+ /undo revert the most recent file change
187
189
  /save /sessions save / list sessions
188
190
  quit / exit exit (Ctrl+C cancels the current round)
189
191
  ```
190
192
 
191
193
  Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
192
194
 
195
+ ## Permissions
196
+
197
+ Read-only tools (`read_file`, `glob`, `grep`, `todo_write`) run the moment the model asks. The mutating ones (`edit_file`, `write_file`, `bash`, and spawning a sub-agent) stop for consent first, and the REPL banner shows which mode you're in:
198
+
199
+ - In the REPL you get one prompt per call: allow once, always allow this tool, or deny. "Always" is remembered per tool for the rest of the session, and a sub-agent inherits the same layer, so consent follows the work wherever it happens.
200
+ - In one-shot mode (`-p`) there is nobody to ask, so a mutating call is refused on the spot and the refusal goes back to the model as an ordinary tool result: the loop never hangs on input that can't arrive. Pass `--yes` to approve everything up front (scripts, CI).
201
+ - The decision itself is pure logic in `permissions.py`, with the terminal only supplying the prompt callback. You can unit-test consent without a TTY, or reuse the layer in your own embedding.
202
+
193
203
  ## Related Projects
194
204
 
195
205
  If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
@@ -202,7 +212,7 @@ If working through CoreCoder was useful, here are a few other tools I've built a
202
212
 
203
213
  ## Contributing / License
204
214
 
205
- Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
215
+ Before you send anything, run `pytest tests/ -q` (119 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
206
216
 
207
217
  ---
208
218
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  # CoreCoder
4
4
 
5
- **编程 agent 里的 nanoGPT。1081 行纯 Python,读懂一个 coding agent 到底怎么运作,再 fork 出你自己的。**
5
+ **编程 agent 里的 nanoGPT。1217 行纯 Python,读懂一个 coding agent 到底怎么运作,再 fork 出你自己的。**
6
6
 
7
7
  *learn from it · fork it · ship something better*
8
8
 
@@ -12,7 +12,7 @@
12
12
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
13
13
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
14
14
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
15
- [![engine](https://img.shields.io/badge/engine-1081_LoC-blue)](article/)
15
+ [![engine](https://img.shields.io/badge/engine-1217_LoC-blue)](article/)
16
16
  [![源码导读](https://img.shields.io/badge/源码导读-8篇双语-orange)](article/)
17
17
 
18
18
  </div>
@@ -25,7 +25,7 @@
25
25
 
26
26
  | | CoreCoder | Claude Code | aider | nanoGPT |
27
27
  |---|---|---|---|---|
28
- | 代码量 | 引擎约 1081 行 / 整包 1714 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
28
+ | 代码量 | 引擎约 1217 行 / 整包 2107 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
29
29
  | 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
30
30
  | 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
31
31
  | 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
@@ -36,9 +36,9 @@ nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个
36
36
 
37
37
  我一直觉得 coding agent 被讲得太玄了。把 Claude Code、Cursor 这类工具扒到底,核心是一个 while 循环套着一个大模型,外加七八个让它能真正动手的工具。难的从来不是这个循环,而是循环跑进真实世界以后要兜的那些底。CoreCoder 就是把这个核心老老实实写出来的最小版本。
38
38
 
39
- 引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1081 行。连最外层的 CLI、配置、打包一起算,整个包 18 个文件、物理 1714 行、净 1385 行,每个文件都短到能一口气读完。
39
+ 引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1217 行。连最外层的 CLI、配置、打包一起算,整个包 22 个文件、物理 2107 行、净 1697 行,每个文件都短到能一口气读完。
40
40
 
41
- 它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,103 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
41
+ 它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你。任何要动你磁盘、要跑命令的调用,都会先停下来等你点头,119 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
42
42
 
43
43
  代码来自一次公开拆解。公开的源码分析里,Claude Code 这类生产级 agent 暴露出不少关键架构,我挑出最核心的一层,用尽量少的代码诚实地复写了一遍。所以读 CoreCoder,约等于读一份基于公开源码分析的「可运行注释版」:讲的是这类 agent 的核心思路,而它本身只是最小复写,就摆在你机器上,随你拆、随你改。
44
44
 
@@ -85,12 +85,13 @@ corecoder -p "给 parse_config() 加错误处理" # 一次性模式,干完
85
85
 
86
86
  ```
87
87
  corecoder/
88
- ├── agent.py agent 主循环 + 并行工具执行 162 行 ← 从这里开始读
88
+ ├── agent.py agent 主循环 + 并行工具执行 180 行 ← 从这里开始读
89
89
  ├── llm.py 流式客户端 + 重试 + 成本统计 336 行
90
90
  ├── context.py 三层上下文压缩 210 行
91
91
  ├── session.py 会话存盘 / 续聊 + 路径穿越防护 97 行
92
+ ├── permissions.py 改动类工具的用户授权 48 行
92
93
  ├── prompt.py 系统提示词 33 行
93
- ├── cli.py REPL + 斜杠命令 + 一次性模式 270 行
94
+ ├── cli.py REPL + 斜杠命令 + 一次性模式 317 行
94
95
  ├── config.py 环境变量配置 57 行
95
96
  └── tools/
96
97
  ├── bash.py shell + 危险命令闸 + cd 追踪 127 行
@@ -100,7 +101,7 @@ corecoder/
100
101
  ├── read.py 文件读取 53 行
101
102
  ├── write.py 文件写入 38 行
102
103
  ├── todo.py agent 自维护的任务清单 79 行
103
- ├── agent.py 子 agent 派生 58 行
104
+ ├── agent.py 子 agent 派生 63 行
104
105
  └── base.py 工具基类 27 行
105
106
  ```
106
107
 
@@ -184,12 +185,21 @@ README 只给方向,每条的代码细节第七篇接着讲。挑一个动手
184
185
  /compact 手动压缩上下文
185
186
  /tokens 查看 token 用量和费用估算
186
187
  /diff 查看本次会话改过的文件
188
+ /undo 撤销最近一次文件改动
187
189
  /save /sessions 保存 / 列出会话
188
190
  quit / exit 退出(Ctrl+C 取消当前回合)
189
191
  ```
190
192
 
191
193
  会话 ID 会先清洗成安全字符再拿去当文件名,存档统统落在 `~/.corecoder/sessions` 里,恶意会话名穿越不出去。
192
194
 
195
+ ## 权限
196
+
197
+ 只读工具(`read_file`、`glob`、`grep`、`todo_write`)模型一调就跑。会动手的那些(`edit_file`、`write_file`、`bash`,以及派生子 agent)先停下来等你点头,REPL 启动横幅里能看到当前是哪种模式:
198
+
199
+ - REPL 里每次调用问一次:允许这一次、本工具本次会话都允许、或者拒绝。「都允许」按工具记到会话结束;子 agent 继承同一层授权,活走到哪,许可跟到哪。
200
+ - 一次性模式(`-p`)没人可问,改动类调用当场被拒,拒绝理由作为普通工具结果回给模型:循环绝不会卡在等一个永远不会来的输入上。要全部预授权就加 `--yes`(脚本、CI 场景)。
201
+ - 判断本身是 `permissions.py` 里的纯逻辑,终端只是塞进来一个提问回调。不用 TTY 也能单测授权逻辑,或者直接搬进你自己的嵌入场景。
202
+
193
203
  ## 相关项目
194
204
 
195
205
  如果你读 CoreCoder 读得还顺,下面几个我做的 agent / LLM 系统方向的工具也许用得上:
@@ -202,7 +212,7 @@ quit / exit 退出(Ctrl+C 取消当前回合)
202
212
 
203
213
  ## 贡献 / License
204
214
 
205
- 动手之前先跑一遍 `pytest tests/ -q`(103 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
215
+ 动手之前先跑一遍 `pytest tests/ -q`(119 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
206
216
 
207
217
  ---
208
218
 
@@ -8,7 +8,7 @@
8
8
 
9
9
  因为 Claude Code 本体太大了。它的代码量在几十万行的量级,光一个负责跑 shell 命令的 BashTool 就上千行,真要逐行读完,多数人会在第三个文件就关掉编辑器。可 agent 的核心其实没那么复杂,复杂的是工程化之后那些为真实世界兜底的边角:模型半路被打断怎么办、上下文塞满了怎么办、十个工具调用同时回来怎么办、provider 偶尔抽风返回 500 又怎么办。这些东西摊在几十万行里,你很难一眼看到骨架。
10
10
 
11
- CoreCoder 做的事,是把这套骨架压到一千行出头的纯 Python。准确说,agent 引擎(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1081 行;连最外层的 CLI 终端、配置和打包都算上,整个包去掉空行和注释 1385 行,物理代码 1714 行。每个文件都短到能一口气读完,跑起来又确实是个能改代码、能跑命令、能自己开子任务、能压缩上下文、能算钱的 agent。它不是玩具。它是把生产级 agent 的每个设计决策,拣出最核心的那版,用最少的代码诚实地写一遍。
11
+ CoreCoder 做的事,是把这套骨架压到一千行出头的纯 Python。准确说,agent 引擎(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1201 行;连最外层的 CLI 终端、配置和打包都算上,整个包去掉空行和注释 1614 行,物理代码 2008 行。每个文件都短到能一口气读完,跑起来又确实是个能改代码、能跑命令、能自己开子任务、能压缩上下文、能算钱的 agent。它不是玩具。它是把生产级 agent 的每个设计决策,拣出最核心的那版,用最少的代码诚实地写一遍。
12
12
 
13
13
  读它,约等于读一份 Claude Code 的「可运行注释版」。区别在于,CoreCoder 的每一行你都能在自己机器上断点、改、跑、看它出什么效果。本系列里我引用的每一个行数、每一段代码,都是从仓库里现读现核的,不是凭印象。这点我后面会反复较真,因为编码 agent 这个领域,太多文章在凭感觉编数字。
14
14
 
@@ -8,7 +8,7 @@ First, use CoreCoder, an open source project whose core runs to just over a thou
8
8
 
9
9
  Because Claude Code itself is too big. Its codebase runs into the hundreds of thousands of lines, and the BashTool alone, the part that runs shell commands, is over a thousand. Read it line by line and most people close the editor by the third file. Yet the core of an agent isn't really that complicated. What's complicated is everything engineered around it to survive the real world: what to do when the model gets interrupted mid-stream, when the context window fills up, when ten tool calls come back at once, when a provider hiccups and returns a 500. Spread across hundreds of thousands of lines, that skeleton is hard to see.
10
10
 
11
- What CoreCoder does is squeeze the skeleton down to just over a thousand lines of pure Python. To be precise, the agent engine (the loop, model interface, context, tools, session) is 1081 lines once you drop blank lines and comments; add the outermost CLI terminal, config, and packaging and the whole package is 1385 lines without blanks and comments, 1714 physical. Every file is short enough to read in one sitting, and what runs is genuinely an agent: it edits code, runs commands, spins up its own subtasks, compresses context, tracks cost. It is not a toy. It takes every design decision a production agent makes, picks the most essential version of each, and writes it out honestly in the least code it can.
11
+ What CoreCoder does is squeeze the skeleton down to just over a thousand lines of pure Python. To be precise, the agent engine (the loop, model interface, context, tools, session) is 1201 lines once you drop blank lines and comments; add the outermost CLI terminal, config, and packaging and the whole package is 1614 lines without blanks and comments, 2008 physical. Every file is short enough to read in one sitting, and what runs is genuinely an agent: it edits code, runs commands, spins up its own subtasks, compresses context, tracks cost. It is not a toy. It takes every design decision a production agent makes, picks the most essential version of each, and writes it out honestly in the least code it can.
12
12
 
13
13
  Reading it is close to reading a runnable, annotated edition of Claude Code. The difference is that every line of CoreCoder is something you can breakpoint, change, run, and watch on your own machine. Every line count and every snippet I quote in this series is read straight out of the repository, not recalled from memory. I'll keep being pedantic about that, because in the coding-agent space far too many write-ups just make the numbers up.
14
14
 
@@ -1,6 +1,6 @@
1
1
  """CoreCoder - Minimal AI coding agent inspired by Claude Code's architecture."""
2
2
 
3
- __version__ = "0.4.1"
3
+ __version__ = "0.5.0"
4
4
 
5
5
  from corecoder.agent import Agent
6
6
  from corecoder.config import Config
@@ -28,9 +28,11 @@ class Agent:
28
28
  tools: list[Tool] | None = None,
29
29
  max_context_tokens: int = 128_000,
30
30
  max_rounds: int = 50,
31
+ permission=None,
31
32
  ):
32
33
  self.llm = llm
33
34
  self.tools = tools if tools is not None else ALL_TOOLS
35
+ self.permission = permission
34
36
  self._tool_by_name = {t.name: t for t in self.tools}
35
37
  self.messages: list[dict] = []
36
38
  self.context = ContextManager(max_tokens=max_context_tokens)
@@ -83,7 +85,7 @@ class Agent:
83
85
  tc = resp.tool_calls[0]
84
86
  if on_tool:
85
87
  on_tool(tc.name, tc.arguments)
86
- result = self._exec_tool(tc)
88
+ result = self._permit(tc) or self._exec_tool(tc)
87
89
  self.messages.append({
88
90
  "role": "tool",
89
91
  "tool_call_id": tc.id,
@@ -109,6 +111,13 @@ class Agent:
109
111
 
110
112
  return "(reached maximum tool-call rounds)"
111
113
 
114
+ def _permit(self, tc) -> str | None:
115
+ """Consent check for one call. None means go ahead; a string is the
116
+ refusal, returned as the tool result instead of executing."""
117
+ if self.permission is None:
118
+ return None
119
+ return self.permission.check(tc.name, tc.arguments)
120
+
112
121
  def _exec_tool(self, tc) -> str:
113
122
  """Execute a single tool call, returning the result string."""
114
123
  tool = self._tool_by_name.get(tc.name)
@@ -137,9 +146,18 @@ class Agent:
137
146
  if on_tool:
138
147
  on_tool(tc.name, tc.arguments)
139
148
 
149
+ # consent is settled up front on this thread: prompting from pool
150
+ # workers would interleave several prompts on one terminal
151
+ results = [self._permit(tc) for tc in tool_calls]
140
152
  with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool:
141
- futures = [pool.submit(self._exec_tool, tc) for tc in tool_calls]
142
- return [f.result() for f in futures]
153
+ futures = {
154
+ i: pool.submit(self._exec_tool, tc)
155
+ for i, tc in enumerate(tool_calls)
156
+ if results[i] is None
157
+ }
158
+ for i, future in futures.items():
159
+ results[i] = future.result()
160
+ return results
143
161
 
144
162
  def _answer_pending_tool_calls(self, tool_calls):
145
163
  """Backfill a tool reply for every call that didn't get one.
@@ -0,0 +1,38 @@
1
+ """Session-scoped undo for file mutations.
2
+
3
+ edit_file and write_file record a checkpoint before touching a file; /undo
4
+ pops the latest one and restores the previous bytes (or removes the file if
5
+ it did not exist). In-memory only: undo history dies with the process, and
6
+ bash side effects are not tracked, only the two file-writing tools.
7
+ """
8
+
9
+ from pathlib import Path
10
+
11
+ # (path, prior bytes or None if the file did not exist)
12
+ _stack: list[tuple[str, bytes | None]] = []
13
+
14
+
15
+ def record(path: Path) -> None:
16
+ """Capture the pre-mutation state of path. Call right before writing."""
17
+ _stack.append((str(path), path.read_bytes() if path.exists() else None))
18
+
19
+
20
+ def undo() -> str:
21
+ """Restore the most recent checkpoint."""
22
+ if not _stack:
23
+ return "Nothing to undo."
24
+ path_str, prior = _stack.pop()
25
+ p = Path(path_str)
26
+ if prior is None:
27
+ p.unlink(missing_ok=True)
28
+ return f"Removed {path_str} (created this session)."
29
+ p.write_bytes(prior)
30
+ return f"Restored {path_str}."
31
+
32
+
33
+ def pending() -> int:
34
+ return len(_stack)
35
+
36
+
37
+ def clear() -> None:
38
+ _stack.clear()
@@ -15,6 +15,7 @@ from . import __version__
15
15
  from .agent import Agent
16
16
  from .config import Config
17
17
  from .llm import LLM, LiteLLM
18
+ from .permissions import Permission
18
19
  from .session import list_sessions, load_session, save_session
19
20
 
20
21
  console = Console()
@@ -29,6 +30,7 @@ def _parse_args():
29
30
  p.add_argument("--base-url", help="API base URL (default: $OPENAI_BASE_URL)")
30
31
  p.add_argument("--api-key", help="API key (default: $OPENAI_API_KEY)")
31
32
  p.add_argument("-p", "--prompt", help="One-shot prompt (non-interactive mode)")
33
+ p.add_argument("--yes", action="store_true", help="Auto-approve every tool call (for scripts and CI)")
32
34
  p.add_argument("--demo", action="store_true", help="Run the offline scripted demo (no API key needed)")
33
35
  p.add_argument("-r", "--resume", metavar="ID", help="Resume a saved session")
34
36
  p.add_argument("-v", "--version", action="version", version=f"%(prog)s {__version__}")
@@ -76,7 +78,14 @@ def main():
76
78
  temperature=config.temperature,
77
79
  max_tokens=config.max_tokens,
78
80
  )
79
- agent = Agent(llm=llm, max_context_tokens=config.max_context_tokens)
81
+ # consent layer: ask in the REPL, refuse in one-shot mode, --yes skips it
82
+ if args.yes:
83
+ permission = Permission(allow_all=True)
84
+ elif args.prompt:
85
+ permission = Permission()
86
+ else:
87
+ permission = Permission(ask=_ask_permission)
88
+ agent = Agent(llm=llm, max_context_tokens=config.max_context_tokens, permission=permission)
80
89
 
81
90
  # resume saved session
82
91
  if args.resume:
@@ -101,8 +110,27 @@ def main():
101
110
  _repl(agent, config)
102
111
 
103
112
 
113
+ def _ask_permission(tool_name: str, arguments: dict) -> str:
114
+ """REPL consent prompt. Anything but a clear yes counts as a no."""
115
+ console.print(f"\n[bold yellow]permission requested:[/] [cyan]{tool_name}[/cyan]({_brief(arguments)})")
116
+ try:
117
+ answer = pt_prompt(" [y] allow once [a] always allow this tool [n] deny: ").strip().lower()
118
+ except (EOFError, KeyboardInterrupt):
119
+ console.print("[dim]denied[/dim]")
120
+ return "deny"
121
+ if answer in ("y", "yes"):
122
+ return "once"
123
+ if answer in ("a", "always"):
124
+ return "always"
125
+ return "deny"
126
+
127
+
104
128
  def _run_once(agent: Agent, prompt: str):
105
129
  """Non-interactive: run one prompt and exit."""
130
+ perm = getattr(agent, "permission", None)
131
+ if perm is not None and perm.ask is None and not perm.allow_all:
132
+ console.print("[dim]one-shot mode: mutating tools are refused unless you pass --yes[/dim]")
133
+
106
134
  def on_token(tok):
107
135
  print(tok, end="", flush=True)
108
136
 
@@ -123,10 +151,13 @@ def _run_once(agent: Agent, prompt: str):
123
151
 
124
152
  def _repl(agent: Agent, config: Config):
125
153
  """Interactive read-eval-print loop."""
154
+ perm = getattr(agent, "permission", None)
155
+ mode = "auto-approve every tool call (--yes)" if (perm and perm.allow_all) else "ask before mutating tools"
126
156
  console.print(Panel(
127
157
  f"[bold]CoreCoder[/bold] v{__version__}\n"
128
158
  f"Model: [cyan]{config.model}[/cyan]"
129
159
  + (f" Base: [dim]{config.base_url}[/dim]" if config.base_url else "")
160
+ + f"\nPermissions: [cyan]{mode}[/cyan]"
130
161
  + "\nType [bold]/help[/bold] for commands, [bold]Ctrl+C[/bold] to cancel, [bold]quit[/bold] to exit.",
131
162
  border_style="blue",
132
163
  ))
@@ -213,6 +244,13 @@ def _repl(agent: Agent, config: Config):
213
244
  for f in sorted(_changed_files):
214
245
  console.print(f" [cyan]{f}[/cyan]")
215
246
  continue
247
+ if user_input == "/undo":
248
+ from .checkpoints import pending, undo
249
+ console.print(undo())
250
+ left = pending()
251
+ if left:
252
+ console.print(f"[dim]{left} more checkpoint(s) on the stack.[/dim]")
253
+ continue
216
254
  if user_input == "/sessions":
217
255
  sessions = list_sessions()
218
256
  if not sessions:
@@ -261,6 +299,7 @@ def _show_help():
261
299
  " /tokens Show token usage\n"
262
300
  " /compact Compress conversation context\n"
263
301
  " /diff Show files modified this session\n"
302
+ " /undo Revert the most recent file change\n"
264
303
  " /save Save session to disk\n"
265
304
  " /sessions List saved sessions\n"
266
305
  " quit Exit CoreCoder\n"
@@ -0,0 +1,48 @@
1
+ """User consent for tool calls, distilled from Claude Code's permissions.
2
+
3
+ The tools split in two. Read-only ones (read_file, glob, grep, todo_write)
4
+ run the moment the model asks; the mutating ones (edit_file, write_file,
5
+ bash, and spawning a sub-agent) stop for a yes first. "Always allow" is
6
+ remembered per tool for the rest of the session: per tool rather than per
7
+ command, because one approved bash prefix says nothing about the next
8
+ command anyway.
9
+
10
+ When there is nobody to ask (one-shot -p mode, or a library embedding with
11
+ no callback), a mutating call is refused instead of blocking on input that
12
+ can never arrive. The refusal travels back as an ordinary tool result, so
13
+ the loop survives and the model can route around it.
14
+ """
15
+
16
+
17
+ class Permission:
18
+ """Session-scoped consent state. Pure: no I/O, the CLI hands in `ask`."""
19
+
20
+ READ_ONLY = frozenset({"read_file", "glob", "grep", "todo_write"})
21
+
22
+ def __init__(self, ask=None, allow_all: bool = False):
23
+ # ask(tool_name, arguments) -> "once" | "always" | "deny"
24
+ self.ask = ask
25
+ self.allow_all = allow_all
26
+ self._always: set[str] = set()
27
+
28
+ def check(self, tool_name: str, arguments: dict) -> str | None:
29
+ """Decide one call. None lets it through; a string is the refusal
30
+ the model receives as its tool result."""
31
+ if tool_name in self.READ_ONLY or self.allow_all or tool_name in self._always:
32
+ return None
33
+ if self.ask is None:
34
+ return (
35
+ f"Permission denied: {tool_name} mutates state and this session is "
36
+ "non-interactive, so nobody can approve it. Rerun with --yes, or "
37
+ "tell the user the exact step so they can run it themselves."
38
+ )
39
+ verdict = self.ask(tool_name, arguments)
40
+ if verdict == "always":
41
+ self._always.add(tool_name)
42
+ return None
43
+ if verdict == "once":
44
+ return None
45
+ return (
46
+ f"Permission denied: the user refused this {tool_name} call. "
47
+ "Do not retry it unchanged; ask what they would prefer instead."
48
+ )
@@ -48,6 +48,8 @@ class AgentTool(Tool):
48
48
  tools=[t for t in parent.tools if t.name != "agent"], # no recursive agents
49
49
  max_context_tokens=parent.context.max_tokens,
50
50
  max_rounds=20,
51
+ # the sub-agent answers to the same consent layer as the parent
52
+ permission=parent.permission,
51
53
  )
52
54
 
53
55
  # a sub-agent failure comes back as text, never propagates into the parent
@@ -10,6 +10,7 @@ import difflib
10
10
  from pathlib import Path
11
11
  from typing import ClassVar
12
12
 
13
+ from ..checkpoints import record as _record_checkpoint
13
14
  from .base import Tool
14
15
 
15
16
  # track files changed this session for /diff
@@ -67,6 +68,7 @@ class EditFileTool(Tool):
67
68
  )
68
69
 
69
70
  new_content = content.replace(old_string, new_string, 1)
71
+ _record_checkpoint(p)
70
72
  p.write_text(new_content, encoding="utf-8")
71
73
  _changed_files.add(str(p))
72
74
 
@@ -3,6 +3,7 @@
3
3
  from pathlib import Path
4
4
  from typing import ClassVar
5
5
 
6
+ from ..checkpoints import record as _record_checkpoint
6
7
  from .base import Tool
7
8
  from .edit import _changed_files
8
9
 
@@ -32,6 +33,7 @@ class WriteFileTool(Tool):
32
33
  try:
33
34
  p = Path(file_path).expanduser().resolve()
34
35
  p.parent.mkdir(parents=True, exist_ok=True)
36
+ _record_checkpoint(p)
35
37
  p.write_text(content, encoding="utf-8")
36
38
  _changed_files.add(str(p))
37
39
  n_lines = content.count("\n") + (1 if content and not content.endswith("\n") else 0)
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "corecoder"
7
- version = "0.4.1"
7
+ version = "0.5.0"
8
8
  description = "Minimal AI coding agent (~1,000 lines of Python) inspired by Claude Code. Works with any LLM. (formerly NanoCoder)"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -0,0 +1,54 @@
1
+ """Undo checkpoints for edit_file / write_file."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from corecoder import checkpoints
6
+ from corecoder.tools.edit import EditFileTool
7
+ from corecoder.tools.write import WriteFileTool
8
+
9
+
10
+ def setup_function():
11
+ checkpoints.clear()
12
+
13
+
14
+ def test_edit_then_undo_restores_previous_bytes(tmp_path):
15
+ f = tmp_path / "a.py"
16
+ f.write_text("v1\n", encoding="utf-8")
17
+ assert EditFileTool().execute(str(f), "v1", "v2").startswith("Edited")
18
+ assert f.read_text() == "v2\n"
19
+
20
+ assert checkpoints.undo() == f"Restored {f}."
21
+ assert f.read_text() == "v1\n"
22
+
23
+
24
+ def test_undo_removes_file_created_by_write(tmp_path):
25
+ f = tmp_path / "new.py"
26
+ assert not f.exists()
27
+ WriteFileTool().execute(str(f), "print(1)\n")
28
+ assert f.exists()
29
+
30
+ assert checkpoints.undo() == f"Removed {f} (created this session)."
31
+ assert not f.exists()
32
+
33
+
34
+ def test_undo_pops_one_mutation_at_a_time(tmp_path):
35
+ f = tmp_path / "a.py"
36
+ f.write_text("v1\n", encoding="utf-8")
37
+ EditFileTool().execute(str(f), "v1", "v2")
38
+ EditFileTool().execute(str(f), "v2", "v3")
39
+
40
+ assert checkpoints.pending() == 2
41
+ checkpoints.undo()
42
+ assert f.read_text() == "v2\n"
43
+ checkpoints.undo()
44
+ assert f.read_text() == "v1\n"
45
+
46
+
47
+ def test_failed_edit_leaves_no_checkpoint(tmp_path):
48
+ f = tmp_path / "a.py"
49
+ f.write_text("v1\n", encoding="utf-8")
50
+ # old_string absent -> error before any write
51
+ result = EditFileTool().execute(str(f), "missing", "v2")
52
+ assert result.startswith("Error:")
53
+ assert checkpoints.pending() == 0
54
+ assert checkpoints.undo() == "Nothing to undo."
@@ -1,6 +1,6 @@
1
1
  """Tests for core modules: config, context, session, imports."""
2
2
 
3
- import tomllib
3
+ import re
4
4
  from pathlib import Path
5
5
  from typing import ClassVar
6
6
 
@@ -12,8 +12,42 @@ from corecoder.tools import get_tool
12
12
 
13
13
 
14
14
  def test_version():
15
- pyproject = tomllib.loads(Path("pyproject.toml").read_text())
16
- assert __version__ == pyproject["project"]["version"]
15
+ # regex instead of tomllib: the latter only exists on 3.11+ and CI runs 3.10
16
+ m = re.search(r'(?m)^version = "([^"]+)"', Path("pyproject.toml").read_text())
17
+ assert m is not None
18
+ assert __version__ == m.group(1)
19
+
20
+
21
+ def test_readme_line_counts_are_current():
22
+ # The LoC numbers are the brand of this repo. If the engine or the
23
+ # package grows, the README has to move with it, and this test is the
24
+ # alarm: update the badge and the prose in README.md and README_CN.md.
25
+ root = Path(__file__).resolve().parent.parent
26
+ engine_files = [
27
+ root / "corecoder" / name
28
+ for name in ("agent.py", "llm.py", "context.py", "session.py")
29
+ ]
30
+ engine_files += sorted((root / "corecoder" / "tools").glob("*.py"))
31
+ package_files = sorted((root / "corecoder").rglob("*.py"))
32
+
33
+ def net_lines(path: Path) -> int:
34
+ return sum(
35
+ 1
36
+ for line in path.read_text(encoding="utf-8").splitlines()
37
+ if line.strip() and not line.strip().startswith("#")
38
+ )
39
+
40
+ engine = sum(net_lines(f) for f in engine_files)
41
+ physical = sum(
42
+ len(f.read_text(encoding="utf-8").splitlines()) for f in package_files
43
+ )
44
+ package_net = sum(net_lines(f) for f in package_files)
45
+
46
+ readme = (root / "README.md").read_text(encoding="utf-8")
47
+ assert f"engine-{engine}_LoC" in readme
48
+ assert f"{len(package_files)} files" in readme
49
+ assert f"{physical:,} physical lines" in readme
50
+ assert f"{package_net:,} net" in readme
17
51
 
18
52
 
19
53
  def test_public_api_exports():
@@ -0,0 +1,181 @@
1
+ """Consent gating for mutating tools: the Permission layer and its wiring."""
2
+
3
+ from corecoder import Agent
4
+ from corecoder.llm import LLMResponse, ScriptedLLM, ToolCall
5
+ from corecoder.permissions import Permission
6
+ from corecoder.tools import get_tool
7
+ from corecoder.tools.agent import AgentTool
8
+ from corecoder.tools.write import WriteFileTool
9
+
10
+
11
+ def _write_call(call_id, path):
12
+ return ToolCall(id=call_id, name="write_file",
13
+ arguments={"file_path": str(path), "content": "x\n"})
14
+
15
+
16
+ def _two_writes_then_text(tmp_path):
17
+ return [
18
+ LLMResponse(tool_calls=[_write_call("c1", tmp_path / "a.txt")]),
19
+ LLMResponse(tool_calls=[_write_call("c2", tmp_path / "b.txt")]),
20
+ LLMResponse(content="done"),
21
+ ]
22
+
23
+
24
+ def test_read_only_tools_run_without_consent(tmp_path):
25
+ f = tmp_path / "note.txt"
26
+ f.write_text("hello", encoding="utf-8")
27
+ agent = Agent(
28
+ llm=ScriptedLLM([
29
+ LLMResponse(tool_calls=[ToolCall(id="c1", name="read_file", arguments={"file_path": str(f)})]),
30
+ LLMResponse(content="read it"),
31
+ ]),
32
+ tools=[get_tool("read_file")],
33
+ permission=Permission(), # nobody to ask, and it doesn't matter
34
+ )
35
+
36
+ assert agent.chat("go") == "read it"
37
+ assert "hello" in agent.messages[2]["content"]
38
+
39
+
40
+ def test_allow_once_asks_again_for_the_next_call(tmp_path):
41
+ asked = []
42
+ agent = Agent(
43
+ llm=ScriptedLLM(_two_writes_then_text(tmp_path)),
44
+ tools=[WriteFileTool()],
45
+ permission=Permission(ask=lambda name, args: asked.append(name) or "once"),
46
+ )
47
+
48
+ assert agent.chat("go") == "done"
49
+ assert asked == ["write_file", "write_file"] # every call asks
50
+ assert (tmp_path / "a.txt").exists() and (tmp_path / "b.txt").exists()
51
+
52
+
53
+ def test_always_allow_is_remembered_for_the_session(tmp_path):
54
+ asked = []
55
+ agent = Agent(
56
+ llm=ScriptedLLM(_two_writes_then_text(tmp_path)),
57
+ tools=[WriteFileTool()],
58
+ permission=Permission(ask=lambda name, args: asked.append(name) or "always"),
59
+ )
60
+
61
+ assert agent.chat("go") == "done"
62
+ assert asked == ["write_file"] # the second call went straight through
63
+ assert (tmp_path / "a.txt").exists() and (tmp_path / "b.txt").exists()
64
+
65
+
66
+ def test_deny_skips_execution_and_reports_back(tmp_path):
67
+ agent = Agent(
68
+ llm=ScriptedLLM([
69
+ LLMResponse(tool_calls=[_write_call("c1", tmp_path / "a.txt")]),
70
+ LLMResponse(content="understood"),
71
+ ]),
72
+ tools=[WriteFileTool()],
73
+ permission=Permission(ask=lambda name, args: "deny"),
74
+ )
75
+
76
+ assert agent.chat("go") == "understood" # the loop survived the refusal
77
+ assert not (tmp_path / "a.txt").exists()
78
+ refusal = agent.messages[2]
79
+ assert refusal["role"] == "tool" and refusal["tool_call_id"] == "c1"
80
+ assert "Permission denied" in refusal["content"]
81
+
82
+
83
+ def test_no_callback_means_auto_deny(tmp_path):
84
+ # one-shot -p mode wires Permission() with no ask: refuse, never hang
85
+ agent = Agent(
86
+ llm=ScriptedLLM([
87
+ LLMResponse(tool_calls=[_write_call("c1", tmp_path / "a.txt")]),
88
+ LLMResponse(content="ok"),
89
+ ]),
90
+ tools=[WriteFileTool()],
91
+ permission=Permission(),
92
+ )
93
+
94
+ assert agent.chat("go") == "ok"
95
+ assert not (tmp_path / "a.txt").exists()
96
+ assert "non-interactive" in agent.messages[2]["content"]
97
+
98
+
99
+ def test_allow_all_approves_without_any_callback(tmp_path):
100
+ # what --yes wires in
101
+ agent = Agent(
102
+ llm=ScriptedLLM(_two_writes_then_text(tmp_path)),
103
+ tools=[WriteFileTool()],
104
+ permission=Permission(allow_all=True),
105
+ )
106
+
107
+ assert agent.chat("go") == "done"
108
+ assert (tmp_path / "a.txt").exists() and (tmp_path / "b.txt").exists()
109
+
110
+
111
+ def test_parallel_calls_each_get_their_own_decision(tmp_path):
112
+ marker = tmp_path / "touched"
113
+ calls = [
114
+ ToolCall(id="c1", name="bash", arguments={"command": f"touch {marker}"}),
115
+ _write_call("c2", tmp_path / "ok.txt"),
116
+ ]
117
+ agent = Agent(
118
+ llm=ScriptedLLM([LLMResponse(tool_calls=calls), LLMResponse(content="done")]),
119
+ tools=[get_tool("bash"), WriteFileTool()],
120
+ permission=Permission(ask=lambda name, args: "deny" if name == "bash" else "once"),
121
+ )
122
+
123
+ assert agent.chat("go") == "done"
124
+ assert not marker.exists() # the denied bash never ran
125
+ assert (tmp_path / "ok.txt").exists() # the allowed write did
126
+ results = {m["tool_call_id"]: m["content"] for m in agent.messages if m.get("role") == "tool"}
127
+ assert "Permission denied" in results["c1"]
128
+ assert results["c2"].startswith("Wrote")
129
+
130
+
131
+ def test_sub_agent_inherits_the_permission_layer(tmp_path):
132
+ seen = []
133
+
134
+ def ask(name, args):
135
+ seen.append(name)
136
+ return "once" if name == "agent" else "deny"
137
+
138
+ target = tmp_path / "sub.txt"
139
+ agent = Agent(
140
+ llm=ScriptedLLM([
141
+ LLMResponse(tool_calls=[ToolCall(id="c1", name="agent", arguments={"task": "write the file"})]),
142
+ LLMResponse(tool_calls=[_write_call("c2", target)]), # the sub-agent's move
143
+ LLMResponse(content="could not write"), # the sub-agent's reply
144
+ LLMResponse(content="parent done"),
145
+ ]),
146
+ tools=[AgentTool(), WriteFileTool()],
147
+ permission=Permission(ask=ask),
148
+ )
149
+
150
+ assert agent.chat("go") == "parent done"
151
+ assert seen == ["agent", "write_file"] # consent followed into the sub-agent
152
+ assert not target.exists()
153
+
154
+
155
+ # --- the CLI side of the layer ---
156
+
157
+ def test_yes_flag_parses(monkeypatch):
158
+ from corecoder.cli import _parse_args
159
+ monkeypatch.setattr("sys.argv", ["corecoder", "--yes"])
160
+ assert _parse_args().yes
161
+
162
+
163
+ def test_ask_prompt_maps_answers(monkeypatch):
164
+ from corecoder import cli
165
+ answers = iter(["y", "a", "n", "garbage"])
166
+ monkeypatch.setattr(cli, "pt_prompt", lambda *a, **k: next(answers))
167
+
168
+ assert cli._ask_permission("bash", {"command": "ls"}) == "once"
169
+ assert cli._ask_permission("bash", {}) == "always"
170
+ assert cli._ask_permission("bash", {}) == "deny"
171
+ assert cli._ask_permission("bash", {}) == "deny" # junk input is a no
172
+
173
+
174
+ def test_ask_prompt_eof_denies(monkeypatch):
175
+ from corecoder import cli
176
+
177
+ def eof(*a, **k):
178
+ raise EOFError
179
+
180
+ monkeypatch.setattr(cli, "pt_prompt", eof)
181
+ assert cli._ask_permission("bash", {}) == "deny"
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes