corecoder 0.4.0__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {corecoder-0.4.0 → corecoder-0.4.1}/PKG-INFO +32 -20
- {corecoder-0.4.0 → corecoder-0.4.1}/README.md +30 -18
- {corecoder-0.4.0 → corecoder-0.4.1}/README_CN.md +30 -18
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/__init__.py +3 -3
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/agent.py +17 -5
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/cli.py +19 -11
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/context.py +9 -9
- corecoder-0.4.1/corecoder/demo.py +84 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/llm.py +26 -1
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/__init__.py +5 -3
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/agent.py +5 -2
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/bash.py +6 -2
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/edit.py +4 -2
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/glob_tool.py +7 -2
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/grep.py +19 -5
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/read.py +5 -2
- corecoder-0.4.1/corecoder/tools/todo.py +79 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/write.py +5 -2
- {corecoder-0.4.0 → corecoder-0.4.1}/pyproject.toml +1 -1
- {corecoder-0.4.0 → corecoder-0.4.1}/tests/test_core.py +50 -6
- corecoder-0.4.1/tests/test_demo.py +39 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/tests/test_litellm.py +1 -2
- {corecoder-0.4.0 → corecoder-0.4.1}/tests/test_session.py +1 -1
- {corecoder-0.4.0 → corecoder-0.4.1}/tests/test_tools.py +106 -3
- {corecoder-0.4.0 → corecoder-0.4.1}/.github/workflows/ci.yml +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/.github/workflows/publish.yml +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/.gitignore +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/LICENSE +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/00-index.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/00-index_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/01-the-loop.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/01-the-loop_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/02-tools.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/02-tools_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/03-llm-and-cost.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/03-llm-and-cost_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/04-context.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/04-context_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/05-parallel-and-subagents.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/05-parallel-and-subagents_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/06-session-and-cli.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/06-session-and-cli_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/07-build-your-own.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/article/07-build-your-own_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/assets/demo.png +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/assets/demo_en.png +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/__main__.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/config.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/prompt.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/session.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/base.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.1}/tests/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: corecoder
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: Minimal AI coding agent (~1,000 lines of Python) inspired by Claude Code. Works with any LLM. (formerly NanoCoder)
|
|
5
5
|
Project-URL: Homepage, https://github.com/he-yufeng/CoreCoder
|
|
6
6
|
Project-URL: Repository, https://github.com/he-yufeng/CoreCoder
|
|
@@ -52,17 +52,28 @@ Description-Content-Type: text/markdown
|
|
|
52
52
|
|
|
53
53
|
</div>
|
|
54
54
|
|
|
55
|
-
- **Readable end to end.** Read the whole engine in an afternoon
|
|
56
|
-
- **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
|
|
55
|
+
- **Readable end to end.** Read the whole engine in an afternoon, with no magic hidden anywhere you can't follow it.
|
|
56
|
+
- **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
|
|
57
57
|
- **The gaps are the point.** It deliberately keeps only the minimal core; what's missing isn't half-finished, it's where you branch off and make it your own.
|
|
58
58
|
|
|
59
|
+
## How it compares
|
|
60
|
+
|
|
61
|
+
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
62
|
+
|---|---|---|---|---|
|
|
63
|
+
| Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
64
|
+
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
65
|
+
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
66
|
+
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
67
|
+
|
|
68
|
+
The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
|
|
69
|
+
|
|
59
70
|
## What this is
|
|
60
71
|
|
|
61
72
|
I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
|
|
62
73
|
|
|
63
74
|
The engine (loop, model interface, context, tools, sessions) is 1,081 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 18 files: 1,714 physical lines, 1,385 net, every one short enough to read in a single sitting.
|
|
64
75
|
|
|
65
|
-
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask.
|
|
76
|
+
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
|
|
66
77
|
|
|
67
78
|
The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
|
|
68
79
|
|
|
@@ -93,6 +104,7 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
|
|
|
93
104
|
|---|---|
|
|
94
105
|
| OpenAI (default `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
|
|
95
106
|
| DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
|
|
107
|
+
| OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
|
|
96
108
|
| Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
|
|
97
109
|
|
|
98
110
|
Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
|
|
@@ -108,7 +120,7 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
|
|
|
108
120
|
|
|
109
121
|
```
|
|
110
122
|
corecoder/
|
|
111
|
-
├── agent.py agent loop + parallel tool exec
|
|
123
|
+
├── agent.py agent loop + parallel tool exec 162 lines ← start here
|
|
112
124
|
├── llm.py streaming client + retry + cost 336 lines
|
|
113
125
|
├── context.py three-tier context compaction 210 lines
|
|
114
126
|
├── session.py save / resume + path-traversal guard 97 lines
|
|
@@ -122,11 +134,12 @@ corecoder/
|
|
|
122
134
|
├── glob_tool.py filename matching 47 lines
|
|
123
135
|
├── read.py file read 53 lines
|
|
124
136
|
├── write.py file write 38 lines
|
|
137
|
+
├── todo.py agent-maintained task checklist 79 lines
|
|
125
138
|
├── agent.py sub-agent spawning 58 lines
|
|
126
139
|
└── base.py tool base class 27 lines
|
|
127
140
|
```
|
|
128
141
|
|
|
129
|
-
|
|
142
|
+
Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
|
|
130
143
|
|
|
131
144
|
## A `while` loop is the whole agent
|
|
132
145
|
|
|
@@ -147,7 +160,7 @@ def chat(self, user_input):
|
|
|
147
160
|
return "(hit the round limit)"
|
|
148
161
|
```
|
|
149
162
|
|
|
150
|
-
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess
|
|
163
|
+
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the project, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
|
|
151
164
|
|
|
152
165
|
Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
|
|
153
166
|
|
|
@@ -197,17 +210,6 @@ Going deeper, the directions are out in the open too. None of the following is i
|
|
|
197
210
|
|
|
198
211
|
The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
|
|
199
212
|
|
|
200
|
-
## How it compares
|
|
201
|
-
|
|
202
|
-
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
203
|
-
|---|---|---|---|---|
|
|
204
|
-
| Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
205
|
-
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
206
|
-
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
207
|
-
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
208
|
-
|
|
209
|
-
The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
|
|
210
|
-
|
|
211
213
|
## Commands
|
|
212
214
|
|
|
213
215
|
Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
|
|
@@ -223,9 +225,19 @@ quit / exit exit (Ctrl+C cancels the current round)
|
|
|
223
225
|
|
|
224
226
|
Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
|
|
225
227
|
|
|
228
|
+
## Related Projects
|
|
229
|
+
|
|
230
|
+
If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
|
|
231
|
+
|
|
232
|
+
- **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — dropped into an unfamiliar codebase? It gives you a guided wiki and a where-to-start reading path, a self-hostable DeepWiki alternative.
|
|
233
|
+
- **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — stop sifting job boards by hand: it ranks postings against your resume and runs mock interviews.
|
|
234
|
+
- **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — catch the risky clauses before you sign: it reads contracts and flags the dangerous bits.
|
|
235
|
+
- **[GitSense](https://github.com/he-yufeng/GitSense)** — want to contribute to open source? It finds issues worth your time and gauges whether your PR will get merged.
|
|
236
|
+
- **[CodeABC](https://github.com/he-yufeng/CodeABC)** — understand any codebase even if you don't code, built for non-programmers.
|
|
237
|
+
|
|
226
238
|
## Contributing / License
|
|
227
239
|
|
|
228
|
-
Before you send anything, run `pytest tests/ -q` (
|
|
240
|
+
Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
|
|
229
241
|
|
|
230
242
|
---
|
|
231
243
|
|
|
@@ -17,17 +17,28 @@
|
|
|
17
17
|
|
|
18
18
|
</div>
|
|
19
19
|
|
|
20
|
-
- **Readable end to end.** Read the whole engine in an afternoon
|
|
21
|
-
- **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
|
|
20
|
+
- **Readable end to end.** Read the whole engine in an afternoon, with no magic hidden anywhere you can't follow it.
|
|
21
|
+
- **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
|
|
22
22
|
- **The gaps are the point.** It deliberately keeps only the minimal core; what's missing isn't half-finished, it's where you branch off and make it your own.
|
|
23
23
|
|
|
24
|
+
## How it compares
|
|
25
|
+
|
|
26
|
+
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
27
|
+
|---|---|---|---|---|
|
|
28
|
+
| Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
29
|
+
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
30
|
+
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
31
|
+
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
32
|
+
|
|
33
|
+
The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
|
|
34
|
+
|
|
24
35
|
## What this is
|
|
25
36
|
|
|
26
37
|
I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
|
|
27
38
|
|
|
28
39
|
The engine (loop, model interface, context, tools, sessions) is 1,081 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 18 files: 1,714 physical lines, 1,385 net, every one short enough to read in a single sitting.
|
|
29
40
|
|
|
30
|
-
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask.
|
|
41
|
+
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
|
|
31
42
|
|
|
32
43
|
The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
|
|
33
44
|
|
|
@@ -58,6 +69,7 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
|
|
|
58
69
|
|---|---|
|
|
59
70
|
| OpenAI (default `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
|
|
60
71
|
| DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
|
|
72
|
+
| OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
|
|
61
73
|
| Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
|
|
62
74
|
|
|
63
75
|
Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
|
|
@@ -73,7 +85,7 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
|
|
|
73
85
|
|
|
74
86
|
```
|
|
75
87
|
corecoder/
|
|
76
|
-
├── agent.py agent loop + parallel tool exec
|
|
88
|
+
├── agent.py agent loop + parallel tool exec 162 lines ← start here
|
|
77
89
|
├── llm.py streaming client + retry + cost 336 lines
|
|
78
90
|
├── context.py three-tier context compaction 210 lines
|
|
79
91
|
├── session.py save / resume + path-traversal guard 97 lines
|
|
@@ -87,11 +99,12 @@ corecoder/
|
|
|
87
99
|
├── glob_tool.py filename matching 47 lines
|
|
88
100
|
├── read.py file read 53 lines
|
|
89
101
|
├── write.py file write 38 lines
|
|
102
|
+
├── todo.py agent-maintained task checklist 79 lines
|
|
90
103
|
├── agent.py sub-agent spawning 58 lines
|
|
91
104
|
└── base.py tool base class 27 lines
|
|
92
105
|
```
|
|
93
106
|
|
|
94
|
-
|
|
107
|
+
Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
|
|
95
108
|
|
|
96
109
|
## A `while` loop is the whole agent
|
|
97
110
|
|
|
@@ -112,7 +125,7 @@ def chat(self, user_input):
|
|
|
112
125
|
return "(hit the round limit)"
|
|
113
126
|
```
|
|
114
127
|
|
|
115
|
-
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess
|
|
128
|
+
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the project, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
|
|
116
129
|
|
|
117
130
|
Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
|
|
118
131
|
|
|
@@ -162,17 +175,6 @@ Going deeper, the directions are out in the open too. None of the following is i
|
|
|
162
175
|
|
|
163
176
|
The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
|
|
164
177
|
|
|
165
|
-
## How it compares
|
|
166
|
-
|
|
167
|
-
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
168
|
-
|---|---|---|---|---|
|
|
169
|
-
| Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
170
|
-
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
171
|
-
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
172
|
-
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
173
|
-
|
|
174
|
-
The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
|
|
175
|
-
|
|
176
178
|
## Commands
|
|
177
179
|
|
|
178
180
|
Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
|
|
@@ -188,9 +190,19 @@ quit / exit exit (Ctrl+C cancels the current round)
|
|
|
188
190
|
|
|
189
191
|
Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
|
|
190
192
|
|
|
193
|
+
## Related Projects
|
|
194
|
+
|
|
195
|
+
If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
|
|
196
|
+
|
|
197
|
+
- **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — dropped into an unfamiliar codebase? It gives you a guided wiki and a where-to-start reading path, a self-hostable DeepWiki alternative.
|
|
198
|
+
- **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — stop sifting job boards by hand: it ranks postings against your resume and runs mock interviews.
|
|
199
|
+
- **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — catch the risky clauses before you sign: it reads contracts and flags the dangerous bits.
|
|
200
|
+
- **[GitSense](https://github.com/he-yufeng/GitSense)** — want to contribute to open source? It finds issues worth your time and gauges whether your PR will get merged.
|
|
201
|
+
- **[CodeABC](https://github.com/he-yufeng/CodeABC)** — understand any codebase even if you don't code, built for non-programmers.
|
|
202
|
+
|
|
191
203
|
## Contributing / License
|
|
192
204
|
|
|
193
|
-
Before you send anything, run `pytest tests/ -q` (
|
|
205
|
+
Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
|
|
194
206
|
|
|
195
207
|
---
|
|
196
208
|
|
|
@@ -17,17 +17,28 @@
|
|
|
17
17
|
|
|
18
18
|
</div>
|
|
19
19
|
|
|
20
|
-
- **读得完。**
|
|
21
|
-
- **改得动。**
|
|
20
|
+
- **读得完。** 一个下午读完整个引擎,没有一处藏着你看不懂的魔法。
|
|
21
|
+
- **改得动。** 每一行都能在你自己机器上下断点、改了再跑。它真能干活,所以这份参考是活的,不是示意图。
|
|
22
22
|
- **留白即起点。** 刻意只留最小核心,没做的那些不是半成品,是留给你 fork 出更好东西的地方。
|
|
23
23
|
|
|
24
|
+
## 和谁比
|
|
25
|
+
|
|
26
|
+
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
27
|
+
|---|---|---|---|---|
|
|
28
|
+
| 代码量 | 引擎约 1081 行 / 整包 1714 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
|
|
29
|
+
| 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
|
|
30
|
+
| 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
|
|
31
|
+
| 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
|
|
32
|
+
|
|
33
|
+
nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个 GPT。CoreCoder 想干的是同一件事,只是把对象换成一个能真正改代码的 agent。和 Claude Code、aider 摆在一起,不是要跟它们抢用户,CoreCoder 是借它们来学、来起步的那块地基,根本不在一个赛道。
|
|
34
|
+
|
|
24
35
|
## 这是什么
|
|
25
36
|
|
|
26
37
|
我一直觉得 coding agent 被讲得太玄了。把 Claude Code、Cursor 这类工具扒到底,核心是一个 while 循环套着一个大模型,外加七八个让它能真正动手的工具。难的从来不是这个循环,而是循环跑进真实世界以后要兜的那些底。CoreCoder 就是把这个核心老老实实写出来的最小版本。
|
|
27
38
|
|
|
28
39
|
引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1081 行。连最外层的 CLI、配置、打包一起算,整个包 18 个文件、物理 1714 行、净 1385 行,每个文件都短到能一口气读完。
|
|
29
40
|
|
|
30
|
-
它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,
|
|
41
|
+
它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,103 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
|
|
31
42
|
|
|
32
43
|
代码来自一次公开拆解。公开的源码分析里,Claude Code 这类生产级 agent 暴露出不少关键架构,我挑出最核心的一层,用尽量少的代码诚实地复写了一遍。所以读 CoreCoder,约等于读一份基于公开源码分析的「可运行注释版」:讲的是这类 agent 的核心思路,而它本身只是最小复写,就摆在你机器上,随你拆、随你改。
|
|
33
44
|
|
|
@@ -58,6 +69,7 @@ pip install -e .
|
|
|
58
69
|
|---|---|
|
|
59
70
|
| OpenAI(默认 `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
|
|
60
71
|
| DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
|
|
72
|
+
| OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
|
|
61
73
|
| 本地 Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
|
|
62
74
|
|
|
63
75
|
Kimi、Qwen 这些同样是改这两个变量;连 OpenAI 兼容接口都不给的 provider,装上可选的 LiteLLM 后端(`pip install "corecoder[litellm]"`)能路由一百多家。第三篇文章把这块讲得更细。key 可以直接 `export`,也可以在项目根目录扔个 `.env`,启动时自动加载。然后:
|
|
@@ -73,7 +85,7 @@ corecoder -p "给 parse_config() 加错误处理" # 一次性模式,干完
|
|
|
73
85
|
|
|
74
86
|
```
|
|
75
87
|
corecoder/
|
|
76
|
-
├── agent.py agent 主循环 + 并行工具执行
|
|
88
|
+
├── agent.py agent 主循环 + 并行工具执行 162 行 ← 从这里开始读
|
|
77
89
|
├── llm.py 流式客户端 + 重试 + 成本统计 336 行
|
|
78
90
|
├── context.py 三层上下文压缩 210 行
|
|
79
91
|
├── session.py 会话存盘 / 续聊 + 路径穿越防护 97 行
|
|
@@ -87,11 +99,12 @@ corecoder/
|
|
|
87
99
|
├── glob_tool.py 文件名匹配 47 行
|
|
88
100
|
├── read.py 文件读取 53 行
|
|
89
101
|
├── write.py 文件写入 38 行
|
|
102
|
+
├── todo.py agent 自维护的任务清单 79 行
|
|
90
103
|
├── agent.py 子 agent 派生 58 行
|
|
91
104
|
└── base.py 工具基类 27 行
|
|
92
105
|
```
|
|
93
106
|
|
|
94
|
-
|
|
107
|
+
八个工具:`bash`、`read_file`、`write_file`、`edit_file`、`glob`、`grep`、`todo_write`(agent 自己维护的任务清单)、`agent`(派子 agent)。其余都是包在引擎核心外面的 CLI 外壳、配置和打包。
|
|
95
108
|
|
|
96
109
|
## 一个 while 循环就是 agent 的本体
|
|
97
110
|
|
|
@@ -112,7 +125,7 @@ def chat(self, user_input):
|
|
|
112
125
|
return "(已达轮次上限)"
|
|
113
126
|
```
|
|
114
127
|
|
|
115
|
-
就这么点。这个循环的核心骨架就二十来行,把并行执行和被 Ctrl+C 打断后的回填都算上,也才四十多行。CoreCoder 一千多行里剩下的,几乎全在收拾它真跑起来之后冒出来的岔子。`llm.py`
|
|
128
|
+
就这么点。这个循环的核心骨架就二十来行,把并行执行和被 Ctrl+C 打断后的回填都算上,也才四十多行。CoreCoder 一千多行里剩下的,几乎全在收拾它真跑起来之后冒出来的岔子。`llm.py` 最后成了全项目最大的文件,不是因为调模型有多难,而是流式返回里一个工具调用的参数会被切成好几段先后送到、得按顺序拼回去,provider 偶尔吐半截 JSON 或把 usage 填成 null,限流(429)、超时、连接中断和 5xx 都得退避重试,其余 4xx 该直接抛就别硬试。这些不起眼的脏活,而不是那个循环,才是一个 agent 从能演示走到能交付真正吃工程功夫的地方;第三篇文章顺着它拆到每一行。
|
|
116
129
|
|
|
117
130
|
有三个决定值得单独看,因为它们是「先读懂别人怎么做」之后才做得出的取舍,也是你 fork 自己 agent 时可以直接抄走的判断。
|
|
118
131
|
|
|
@@ -162,17 +175,6 @@ print(Agent(llm=llm).chat("找出项目里所有 TODO 注释并列出来"))
|
|
|
162
175
|
|
|
163
176
|
README 只给方向,每条的代码细节第七篇接着讲。挑一个动手,就是把它做得更好的开始。
|
|
164
177
|
|
|
165
|
-
## 和谁比
|
|
166
|
-
|
|
167
|
-
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
168
|
-
|---|---|---|---|---|
|
|
169
|
-
| 代码量 | 引擎约 1081 行 / 整包 1714 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
|
|
170
|
-
| 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
|
|
171
|
-
| 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
|
|
172
|
-
| 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
|
|
173
|
-
|
|
174
|
-
nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个 GPT。CoreCoder 想干的是同一件事,只是把对象换成一个能真正改代码的 agent。和 Claude Code、aider 摆在一起,不是要跟它们抢用户,CoreCoder 是借它们来学、来起步的那块地基,根本不在一个赛道。
|
|
175
|
-
|
|
176
178
|
## 命令
|
|
177
179
|
|
|
178
180
|
进了 REPL,`/help` 列全部,常用的这几个:
|
|
@@ -188,9 +190,19 @@ quit / exit 退出(Ctrl+C 取消当前回合)
|
|
|
188
190
|
|
|
189
191
|
会话 ID 会先清洗成安全字符再拿去当文件名,存档统统落在 `~/.corecoder/sessions` 里,恶意会话名穿越不出去。
|
|
190
192
|
|
|
193
|
+
## 相关项目
|
|
194
|
+
|
|
195
|
+
如果你读 CoreCoder 读得还顺,下面几个我做的 agent / LLM 系统方向的工具也许用得上:
|
|
196
|
+
|
|
197
|
+
- **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — 被丢进一个陌生代码库?它给你一份带「从哪读起」路径的 wiki,一个可自托管的 DeepWiki 替代。
|
|
198
|
+
- **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — 别再手动刷招聘网站:它按你的简历给岗位排序,还能跑模拟面试。
|
|
199
|
+
- **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — 签字前先把有风险的条款挑出来:它读合同、标出危险点。
|
|
200
|
+
- **[GitSense](https://github.com/he-yufeng/GitSense)** — 想给开源做贡献?它帮你找到值得做的 issue,还能估你的 PR 多大概率被合。
|
|
201
|
+
- **[CodeABC](https://github.com/he-yufeng/CodeABC)** — 不会写代码也能看懂一个项目,专给小白做的。
|
|
202
|
+
|
|
191
203
|
## 贡献 / License
|
|
192
204
|
|
|
193
|
-
动手之前先跑一遍 `pytest tests/ -q`(
|
|
205
|
+
动手之前先跑一遍 `pytest tests/ -q`(103 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
|
|
194
206
|
|
|
195
207
|
---
|
|
196
208
|
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
"""CoreCoder - Minimal AI coding agent inspired by Claude Code's architecture."""
|
|
2
2
|
|
|
3
|
-
__version__ = "0.4.
|
|
3
|
+
__version__ = "0.4.1"
|
|
4
4
|
|
|
5
5
|
from corecoder.agent import Agent
|
|
6
|
-
from corecoder.llm import LLM
|
|
7
6
|
from corecoder.config import Config
|
|
7
|
+
from corecoder.llm import LLM
|
|
8
8
|
from corecoder.tools import ALL_TOOLS
|
|
9
9
|
|
|
10
|
-
__all__ = ["
|
|
10
|
+
__all__ = ["ALL_TOOLS", "LLM", "Agent", "Config", "__version__"]
|
|
@@ -11,12 +11,14 @@ which means it's done working and ready to report back.
|
|
|
11
11
|
|
|
12
12
|
import concurrent.futures
|
|
13
13
|
import inspect
|
|
14
|
+
|
|
15
|
+
from .context import ContextManager
|
|
14
16
|
from .llm import LLM
|
|
17
|
+
from .prompt import system_prompt
|
|
15
18
|
from .tools import ALL_TOOLS
|
|
16
|
-
from .tools.base import Tool
|
|
17
19
|
from .tools.agent import AgentTool
|
|
18
|
-
from .
|
|
19
|
-
from .
|
|
20
|
+
from .tools.base import Tool
|
|
21
|
+
from .tools.todo import TodoWriteTool
|
|
20
22
|
|
|
21
23
|
|
|
22
24
|
class Agent:
|
|
@@ -40,8 +42,17 @@ class Agent:
|
|
|
40
42
|
if isinstance(t, AgentTool):
|
|
41
43
|
t._parent_agent = self
|
|
42
44
|
|
|
45
|
+
self._todo = next((t for t in self.tools if isinstance(t, TodoWriteTool)), None)
|
|
46
|
+
|
|
43
47
|
def _full_messages(self) -> list[dict]:
|
|
44
|
-
|
|
48
|
+
system = self._system
|
|
49
|
+
# the task list is re-injected every round, so the model always sees the
|
|
50
|
+
# current state rather than a stale copy buried in old tool results
|
|
51
|
+
if self._todo is not None:
|
|
52
|
+
rendered = self._todo.render()
|
|
53
|
+
if rendered:
|
|
54
|
+
system += "\n\n# Current task list\n" + rendered
|
|
55
|
+
return [{"role": "system", "content": system}] + self.messages
|
|
45
56
|
|
|
46
57
|
def _tool_schemas(self) -> list[dict]:
|
|
47
58
|
return [t.schema() for t in self.tools]
|
|
@@ -109,9 +120,10 @@ class Agent:
|
|
|
109
120
|
inspect.signature(tool.execute).bind(**tc.arguments)
|
|
110
121
|
except TypeError as e:
|
|
111
122
|
return f"Error: bad arguments for {tc.name}: {e}"
|
|
123
|
+
# a tool that blows up gets reported back as text, never kills the loop
|
|
112
124
|
try:
|
|
113
125
|
return tool.execute(**tc.arguments)
|
|
114
|
-
except Exception as e:
|
|
126
|
+
except Exception as e: # noqa: BLE001
|
|
115
127
|
return f"Error executing {tc.name}: {e}"
|
|
116
128
|
|
|
117
129
|
def _exec_tools_parallel(self, tool_calls, on_tool=None) -> list[str]:
|
|
@@ -1,21 +1,21 @@
|
|
|
1
1
|
"""Interactive REPL - the user-facing terminal interface."""
|
|
2
2
|
|
|
3
|
-
import sys
|
|
4
|
-
import os
|
|
5
3
|
import argparse
|
|
4
|
+
import os
|
|
5
|
+
import sys
|
|
6
6
|
|
|
7
|
-
from rich.console import Console
|
|
8
|
-
from rich.markdown import Markdown
|
|
9
|
-
from rich.panel import Panel
|
|
10
7
|
from prompt_toolkit import prompt as pt_prompt
|
|
11
8
|
from prompt_toolkit.history import FileHistory
|
|
12
9
|
from prompt_toolkit.key_binding import KeyBindings
|
|
10
|
+
from rich.console import Console
|
|
11
|
+
from rich.markdown import Markdown
|
|
12
|
+
from rich.panel import Panel
|
|
13
13
|
|
|
14
|
+
from . import __version__
|
|
14
15
|
from .agent import Agent
|
|
15
|
-
from .llm import LLM, LiteLLM
|
|
16
16
|
from .config import Config
|
|
17
|
-
from .
|
|
18
|
-
from . import
|
|
17
|
+
from .llm import LLM, LiteLLM
|
|
18
|
+
from .session import list_sessions, load_session, save_session
|
|
19
19
|
|
|
20
20
|
console = Console()
|
|
21
21
|
|
|
@@ -29,6 +29,7 @@ def _parse_args():
|
|
|
29
29
|
p.add_argument("--base-url", help="API base URL (default: $OPENAI_BASE_URL)")
|
|
30
30
|
p.add_argument("--api-key", help="API key (default: $OPENAI_API_KEY)")
|
|
31
31
|
p.add_argument("-p", "--prompt", help="One-shot prompt (non-interactive mode)")
|
|
32
|
+
p.add_argument("--demo", action="store_true", help="Run the offline scripted demo (no API key needed)")
|
|
32
33
|
p.add_argument("-r", "--resume", metavar="ID", help="Resume a saved session")
|
|
33
34
|
p.add_argument("-v", "--version", action="version", version=f"%(prog)s {__version__}")
|
|
34
35
|
return p.parse_args()
|
|
@@ -36,6 +37,11 @@ def _parse_args():
|
|
|
36
37
|
|
|
37
38
|
def main():
|
|
38
39
|
args = _parse_args()
|
|
40
|
+
|
|
41
|
+
if args.demo:
|
|
42
|
+
from .demo import run_demo
|
|
43
|
+
raise SystemExit(run_demo())
|
|
44
|
+
|
|
39
45
|
config = Config.from_env()
|
|
40
46
|
|
|
41
47
|
# CLI args override env vars
|
|
@@ -108,7 +114,8 @@ def _run_once(agent: Agent, prompt: str):
|
|
|
108
114
|
except KeyboardInterrupt:
|
|
109
115
|
console.print("\n[yellow]Interrupted.[/yellow]")
|
|
110
116
|
sys.exit(130)
|
|
111
|
-
except Exception as e:
|
|
117
|
+
except Exception as e: # noqa: BLE001
|
|
118
|
+
# one-shot mode: print whatever went wrong and exit non-zero
|
|
112
119
|
console.print(f"\n[red]Error: {e}[/red]")
|
|
113
120
|
sys.exit(1)
|
|
114
121
|
print()
|
|
@@ -223,7 +230,7 @@ def _repl(agent: Agent, config: Config):
|
|
|
223
230
|
# call the agent
|
|
224
231
|
streamed: list[str] = []
|
|
225
232
|
|
|
226
|
-
def on_token(tok):
|
|
233
|
+
def on_token(tok, streamed=streamed):
|
|
227
234
|
streamed.append(tok)
|
|
228
235
|
print(tok, end="", flush=True)
|
|
229
236
|
|
|
@@ -239,7 +246,8 @@ def _repl(agent: Agent, config: Config):
|
|
|
239
246
|
console.print(Markdown(response))
|
|
240
247
|
except KeyboardInterrupt:
|
|
241
248
|
console.print("\n[yellow]Interrupted.[/yellow]")
|
|
242
|
-
except Exception as e:
|
|
249
|
+
except Exception as e: # noqa: BLE001
|
|
250
|
+
# keep the REPL alive no matter what chat() throws
|
|
243
251
|
console.print(f"\n[red]Error: {e}[/red]")
|
|
244
252
|
|
|
245
253
|
|
|
@@ -13,6 +13,7 @@ CoreCoder implements the same idea in 3 layers:
|
|
|
13
13
|
"""
|
|
14
14
|
|
|
15
15
|
from __future__ import annotations
|
|
16
|
+
|
|
16
17
|
from typing import TYPE_CHECKING
|
|
17
18
|
|
|
18
19
|
if TYPE_CHECKING:
|
|
@@ -48,16 +49,14 @@ class ContextManager:
|
|
|
48
49
|
compressed = False
|
|
49
50
|
|
|
50
51
|
# Layer 1: snip verbose tool outputs
|
|
51
|
-
if current > self._snip_at:
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
current = estimate_tokens(messages)
|
|
52
|
+
if current > self._snip_at and self._snip_tool_outputs(messages):
|
|
53
|
+
compressed = True
|
|
54
|
+
current = estimate_tokens(messages)
|
|
55
55
|
|
|
56
56
|
# Layer 2: LLM-powered summarization of old turns
|
|
57
|
-
if current > self._summarize_at and len(messages) > 10:
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
current = estimate_tokens(messages)
|
|
57
|
+
if current > self._summarize_at and len(messages) > 10 and self._summarize_old(messages, llm, keep_recent=8):
|
|
58
|
+
compressed = True
|
|
59
|
+
current = estimate_tokens(messages)
|
|
61
60
|
|
|
62
61
|
# Layer 3: hard collapse - last resort
|
|
63
62
|
if current > self._collapse_at and len(messages) > 4:
|
|
@@ -169,7 +168,8 @@ class ContextManager:
|
|
|
169
168
|
],
|
|
170
169
|
)
|
|
171
170
|
return resp.content
|
|
172
|
-
except Exception:
|
|
171
|
+
except Exception: # noqa: BLE001, S110
|
|
172
|
+
# summarization is best-effort; fall back to extraction below
|
|
173
173
|
pass
|
|
174
174
|
|
|
175
175
|
# fallback: extract key lines
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Offline scripted demo: watch the full agent loop with no API key.
|
|
2
|
+
|
|
3
|
+
Runs one scripted task through the real Agent loop with a ScriptedLLM, so the
|
|
4
|
+
thought -> tool call -> observation cycle renders exactly as it would against
|
|
5
|
+
a live model. Useful for demos, screenshots, and as a smoke test of the loop.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import tempfile
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from rich.console import Console
|
|
12
|
+
from rich.markdown import Markdown
|
|
13
|
+
from rich.panel import Panel
|
|
14
|
+
|
|
15
|
+
from .agent import Agent
|
|
16
|
+
from .llm import LLMResponse, ScriptedLLM, ToolCall
|
|
17
|
+
|
|
18
|
+
console = Console()
|
|
19
|
+
|
|
20
|
+
_TASK = (
|
|
21
|
+
"Write a Python function `fib(n)` in fib.py returning the nth Fibonacci "
|
|
22
|
+
"number, add a pytest file, then run the tests."
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _script(workdir: Path) -> list[LLMResponse]:
|
|
27
|
+
fib_py = (
|
|
28
|
+
"def fib(n):\n"
|
|
29
|
+
" if n < 2:\n"
|
|
30
|
+
" return n\n"
|
|
31
|
+
" a, b = 0, 1\n"
|
|
32
|
+
" for _ in range(2, n + 1):\n"
|
|
33
|
+
" a, b = b, a + b\n"
|
|
34
|
+
" return b\n"
|
|
35
|
+
)
|
|
36
|
+
test_py = (
|
|
37
|
+
"from fib import fib\n"
|
|
38
|
+
"\n"
|
|
39
|
+
"def test_fib():\n"
|
|
40
|
+
" assert fib(0) == 0\n"
|
|
41
|
+
" assert fib(1) == 1\n"
|
|
42
|
+
" assert fib(10) == 55\n"
|
|
43
|
+
)
|
|
44
|
+
return [
|
|
45
|
+
LLMResponse(
|
|
46
|
+
content="I'll write fib.py with an iterative implementation.",
|
|
47
|
+
tool_calls=[ToolCall(id="c1", name="write_file", arguments={"file_path": str(workdir / "fib.py"), "content": fib_py})],
|
|
48
|
+
),
|
|
49
|
+
LLMResponse(
|
|
50
|
+
content="Now the test file covering the base cases and fib(10).",
|
|
51
|
+
tool_calls=[ToolCall(id="c2", name="write_file", arguments={"file_path": str(workdir / "test_fib.py"), "content": test_py})],
|
|
52
|
+
),
|
|
53
|
+
LLMResponse(
|
|
54
|
+
content="Running the tests.",
|
|
55
|
+
tool_calls=[ToolCall(id="c3", name="bash", arguments={"command": f"cd {workdir} && python -m pytest test_fib.py -q"})],
|
|
56
|
+
),
|
|
57
|
+
LLMResponse(
|
|
58
|
+
content="All three assertions pass. `fib` is iterative, and the test covers the base cases plus fib(10) == 55."
|
|
59
|
+
),
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _summarize(args: dict) -> str:
|
|
64
|
+
parts = []
|
|
65
|
+
for key, value in args.items():
|
|
66
|
+
text = str(value)
|
|
67
|
+
if len(text) > 60:
|
|
68
|
+
text = text[:57] + "..."
|
|
69
|
+
parts.append(f"{key}={text}")
|
|
70
|
+
return " ".join(parts)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def run_demo() -> int:
|
|
74
|
+
workdir = Path(tempfile.mkdtemp(prefix="corecoder-demo-"))
|
|
75
|
+
agent = Agent(llm=ScriptedLLM(_script(workdir)))
|
|
76
|
+
|
|
77
|
+
console.print(Panel.fit(f"[bold]{_TASK}[/]", title="corecoder demo (offline)"))
|
|
78
|
+
result = agent.chat(
|
|
79
|
+
_TASK,
|
|
80
|
+
on_tool=lambda name, args: console.print(f"[cyan]tool:[/] {name} {_summarize(args)}"),
|
|
81
|
+
)
|
|
82
|
+
console.print(Panel.fit(Markdown(result), title="final"))
|
|
83
|
+
console.print(f"[dim]workspace kept at {workdir}[/]")
|
|
84
|
+
return 0
|