qai-cli 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qai_cli-0.1.0/.gitignore +13 -0
- qai_cli-0.1.0/LICENSE +21 -0
- qai_cli-0.1.0/PKG-INFO +304 -0
- qai_cli-0.1.0/README.md +262 -0
- qai_cli-0.1.0/pyproject.toml +90 -0
- qai_cli-0.1.0/src/qai/__init__.py +6 -0
- qai_cli-0.1.0/src/qai/__main__.py +6 -0
- qai_cli-0.1.0/src/qai/agent.py +162 -0
- qai_cli-0.1.0/src/qai/backend.py +239 -0
- qai_cli-0.1.0/src/qai/cli.py +365 -0
- qai_cli-0.1.0/src/qai/config.py +160 -0
- qai_cli-0.1.0/src/qai/context.py +139 -0
- qai_cli-0.1.0/src/qai/context_manager.py +368 -0
- qai_cli-0.1.0/src/qai/py.typed +0 -0
- qai_cli-0.1.0/src/qai/tools/__init__.py +8 -0
- qai_cli-0.1.0/src/qai/tools/base.py +119 -0
- qai_cli-0.1.0/src/qai/tools/fs.py +308 -0
- qai_cli-0.1.0/src/qai/ui.py +157 -0
- qai_cli-0.1.0/src/qai/workspace.py +118 -0
- qai_cli-0.1.0/tests/test_agent.py +112 -0
- qai_cli-0.1.0/tests/test_config.py +52 -0
- qai_cli-0.1.0/tests/test_context_manager.py +263 -0
- qai_cli-0.1.0/tests/test_integration.py +248 -0
- qai_cli-0.1.0/tests/test_tools.py +226 -0
- qai_cli-0.1.0/tests/test_workspace.py +60 -0
qai_cli-0.1.0/.gitignore
ADDED
qai_cli-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Nodoubvt
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
qai_cli-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: qai-cli
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Offline terminal coding agent. Runs Qwen2.5-Coder 1.5B locally via llama.cpp - no API key, no cloud, no telemetry.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Nodoubvt/local-coding-agent
|
|
6
|
+
Project-URL: Repository, https://github.com/Nodoubvt/local-coding-agent
|
|
7
|
+
Project-URL: Issues, https://github.com/Nodoubvt/local-coding-agent/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/Nodoubvt/local-coding-agent/blob/main/CHANGELOG.md
|
|
9
|
+
Author: Nodoubvt
|
|
10
|
+
License: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: agent,cli,coding-agent,coding-assistant,gguf,llama-cpp,llama.cpp,llm,local-llm,offline,privacy,qwen
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Environment :: Console
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Software Development :: Code Generators
|
|
24
|
+
Classifier: Topic :: Utilities
|
|
25
|
+
Classifier: Typing :: Typed
|
|
26
|
+
Requires-Python: >=3.10
|
|
27
|
+
Requires-Dist: huggingface-hub>=0.23
|
|
28
|
+
Requires-Dist: pathspec>=0.12
|
|
29
|
+
Requires-Dist: platformdirs>=4.2
|
|
30
|
+
Requires-Dist: prompt-toolkit>=3.0.43
|
|
31
|
+
Requires-Dist: rich>=13.7
|
|
32
|
+
Requires-Dist: typer>=0.12
|
|
33
|
+
Provides-Extra: dev
|
|
34
|
+
Requires-Dist: build>=1.2; extra == 'dev'
|
|
35
|
+
Requires-Dist: mypy>=1.11; extra == 'dev'
|
|
36
|
+
Requires-Dist: pytest>=8.2; extra == 'dev'
|
|
37
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
38
|
+
Requires-Dist: twine>=5.1; extra == 'dev'
|
|
39
|
+
Provides-Extra: local
|
|
40
|
+
Requires-Dist: llama-cpp-python>=0.3.2; extra == 'local'
|
|
41
|
+
Description-Content-Type: text/markdown
|
|
42
|
+
|
|
43
|
+
<div align="center">
|
|
44
|
+
|
|
45
|
+

|
|
46
|
+

|
|
47
|
+

|
|
48
|
+

|
|
49
|
+

|
|
50
|
+

|
|
51
|
+

|
|
52
|
+
|
|
53
|
+
</div>
|
|
54
|
+
|
|
55
|
+
# local-coding-agent
|
|
56
|
+
|
|
57
|
+
**An offline AI coding assistant that runs entirely on your own machine.**
|
|
58
|
+
`qai` is a terminal coding agent powered by **Qwen2.5-Coder 1.5B** running on
|
|
59
|
+
[llama.cpp](https://github.com/ggml-org/llama.cpp) — no cloud, no API keys, no
|
|
60
|
+
telemetry, no data leaving your laptop.
|
|
61
|
+
|
|
62
|
+
It is deliberately small: about **1,700 lines** of readable Python you can finish
|
|
63
|
+
in one sitting, built to be extended rather than depended on.
|
|
64
|
+
|
|
65
|
+
> **CLI command:** `qai` · **PyPI package:** `qai-cli` · **Repository:**
|
|
66
|
+
> `local-coding-agent` · **License:** MIT
|
|
67
|
+
|
|
68
|
+
```
|
|
69
|
+
qai> add a docstring to every function in utils.py
|
|
70
|
+
tool read_file(path=utils.py)
|
|
71
|
+
ok read_file
|
|
72
|
+
tool edit_file(path=utils.py, old_string=..., new_string=...)
|
|
73
|
+
Approve edit utils.py
|
|
74
|
+
▸ 12 lines changed
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## Why this exists
|
|
80
|
+
|
|
81
|
+
Most "AI coding agent" tools assume a 70B model, a GPU, and a subscription.
|
|
82
|
+
This one assumes nothing:
|
|
83
|
+
|
|
84
|
+
- **Runs a 1.5B model on a laptop CPU.** No CUDA, no 24 GB of VRAM, no
|
|
85
|
+
electricity bill. It is a *small* model, and the tooling around it is built to
|
|
86
|
+
make a small model useful.
|
|
87
|
+
- **~1.1 GB download.** Q4_K_M GGUF weights, cached after the first run.
|
|
88
|
+
- **1,700 lines, fully readable.** The whole agent — sandbox, tools, approval
|
|
89
|
+
gate, context optimizer — is short enough to read, audit and fork.
|
|
90
|
+
- **MIT, no strings.** The safety model is code you can inspect, not a policy
|
|
91
|
+
page.
|
|
92
|
+
|
|
93
|
+
## Features
|
|
94
|
+
|
|
95
|
+
- **Local inference** — GGUF weights from Hugging Face, run through
|
|
96
|
+
`llama-cpp-python`. CPU-only works out of the box; GPU offload is on by default
|
|
97
|
+
when a backend exists.
|
|
98
|
+
- **Agentic file tools** — `read_file`, `edit_file`, `write_file`, `list_dir`,
|
|
99
|
+
`grep`, all sandboxed to the workspace root.
|
|
100
|
+
- **Approval gate** — every write shows a unified diff and waits for you. `--yes`
|
|
101
|
+
skips it, `--read-only` disables writes entirely.
|
|
102
|
+
- **32k context with a real optimizer** — tool results and file dumps are trimmed
|
|
103
|
+
head-and-tail, then older turns are auto-compacted into a summary before the
|
|
104
|
+
buffer fills. Tool calls are never orphaned from their results.
|
|
105
|
+
- **Context injection** — `@path` mentions, `--attach` globs, and an automatic
|
|
106
|
+
workspace manifest so a 1.5B model knows what files exist.
|
|
107
|
+
- **Two modes** — one-shot `qai "..."` for scripting, or a REPL with `@file`
|
|
108
|
+
completion, history and slash commands.
|
|
109
|
+
|
|
110
|
+
## Requirements
|
|
111
|
+
|
|
112
|
+
- Python 3.10 or newer
|
|
113
|
+
- ~2 GB free disk for weights, ~2 GB RAM at inference
|
|
114
|
+
- A C++ toolchain, *or* a prebuilt `llama-cpp-python` wheel (below)
|
|
115
|
+
|
|
116
|
+
## Install
|
|
117
|
+
|
|
118
|
+
### Fastest — no install, no virtualenv
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
uvx qai-cli # run it instantly
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### Global install, still isolated
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
uv tool install qai-cli # or: pipx install qai-cli
|
|
128
|
+
qai "what does main() do?"
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
> **Local inference needs `llama-cpp-python`**, a compiled C++ extension. If you
|
|
132
|
+
> have no C++ toolchain, use the prebuilt CPU wheel:
|
|
133
|
+
>
|
|
134
|
+
> ```bash
|
|
135
|
+
> pip install llama-cpp-python --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu
|
|
136
|
+
> ```
|
|
137
|
+
>
|
|
138
|
+
> Then run `uvx --with llama-cpp-python qai-cli` to try it in one shot.
|
|
139
|
+
|
|
140
|
+
### From source
|
|
141
|
+
|
|
142
|
+
```bash
|
|
143
|
+
git clone https://github.com/Nodoubvt/local-coding-agent
|
|
144
|
+
cd local-coding-agent
|
|
145
|
+
|
|
146
|
+
python -m venv .venv
|
|
147
|
+
source .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
148
|
+
pip install -e ".[local]" # [local] adds llama-cpp-python
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
Inspect resolved settings, or fetch weights ahead of time:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
qai --print-config
|
|
155
|
+
python -c "from qai.config import load_settings; from qai.backend import ensure_model_file; print(ensure_model_file(load_settings()))"
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
## Usage
|
|
159
|
+
|
|
160
|
+
```bash
|
|
161
|
+
qai # interactive session
|
|
162
|
+
qai "why does main() hang here?" # one-shot
|
|
163
|
+
qai "refactor this" -a src/api.py # attach a file
|
|
164
|
+
qai "find every TODO" --no-tools # pure chat, no filesystem access
|
|
165
|
+
|
|
166
|
+
qai --model qwen2.5-coder-1.5b-instruct-q8_0.gguf # higher-quality quant
|
|
167
|
+
qai --gpu-layers 0 --threads 8 # force CPU, pin thread count
|
|
168
|
+
qai --read-only # refuse every write
|
|
169
|
+
qai --compact-at 0.5 # summarize earlier, at 50% of budget
|
|
170
|
+
qai --no-compact # disable the compactor
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
### REPL commands
|
|
174
|
+
|
|
175
|
+
| command | effect |
|
|
176
|
+
| --- | --- |
|
|
177
|
+
| `/help` | show commands |
|
|
178
|
+
| `/new` | clear conversation history |
|
|
179
|
+
| `/context [path]` | re-send the manifest, or attach specific paths |
|
|
180
|
+
| `/stats` | context usage, budget, compaction count |
|
|
181
|
+
| `/tools` | list available tools |
|
|
182
|
+
| `/model` | show active model and context size |
|
|
183
|
+
| `/read-only on\|off` | toggle write access |
|
|
184
|
+
| `/exit` | quit (ctrl-d also works) |
|
|
185
|
+
|
|
186
|
+
## Configuration
|
|
187
|
+
|
|
188
|
+
Settings merge in this order: **defaults → `config.json` → env vars → CLI flags**.
|
|
189
|
+
|
|
190
|
+
```bash
|
|
191
|
+
python -c "from qai.config import Settings, save_settings; save_settings(Settings(n_ctx=32768))"
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
| setting | default | meaning |
|
|
195
|
+
| --- | --- | --- |
|
|
196
|
+
| `n_ctx` | `32768` | context window |
|
|
197
|
+
| `max_tokens` | `4096` | reply ceiling, also reserved from the budget |
|
|
198
|
+
| `compact_threshold` | `0.70` | auto-compact above this context ratio |
|
|
199
|
+
| `max_tool_result_chars` | `8000` | per-result trim limit |
|
|
200
|
+
| `keep_recent_blocks` | `6` | turns kept verbatim through compaction |
|
|
201
|
+
| `auto_compact` | `true` | enable the compactor |
|
|
202
|
+
| `n_gpu_layers` | `-1` | `-1` offload all, `0` CPU only |
|
|
203
|
+
| `temperature` | `0.2` | sampling temperature |
|
|
204
|
+
| `auto_approve` | `false` | skip write approval |
|
|
205
|
+
|
|
206
|
+
Env equivalents: `QAI_N_CTX`, `QAI_TEMPERATURE`, `QAI_AUTO_COMPACT`,
|
|
207
|
+
`QAI_MODEL_DIR`, `QAI_CONFIG_DIR`, and the rest follow `QAI_` + the setting name.
|
|
208
|
+
|
|
209
|
+
## How the context optimizer works
|
|
210
|
+
|
|
211
|
+
A 1.5B model with a 32k window still hits the wall fast, because a single
|
|
212
|
+
`read_file` of a large file can eat a quarter of it. Three layers keep a long
|
|
213
|
+
session degrading instead of failing:
|
|
214
|
+
|
|
215
|
+
1. **Budget.** `budget = n_ctx - max_tokens - tool_schema_reserve` — 27,472
|
|
216
|
+
tokens at defaults. Everything is measured with a deliberately pessimistic
|
|
217
|
+
chars-per-token estimate, so the agent under-fills rather than overflows.
|
|
218
|
+
2. **Trim.** Oversized tool results and `<context>` blocks are cut head-and-tail,
|
|
219
|
+
so the end of an error message survives. Assistant messages carrying
|
|
220
|
+
`tool_calls` are grouped with their results into a single atomic block, so
|
|
221
|
+
trimming can never leave an orphan that breaks the chat template.
|
|
222
|
+
3. **Compact.** Past `compact_threshold`, older turns are summarized — by the
|
|
223
|
+
model when it is available, otherwise by a deterministic extractive fallback —
|
|
224
|
+
into a single `<summary>` block, with the last `keep_recent_blocks` turns kept
|
|
225
|
+
verbatim.
|
|
226
|
+
|
|
227
|
+
Compaction runs between tool rounds, never mid-flight, so the model always sees a
|
|
228
|
+
consistent transcript, and the compacted transcript is carried into the next turn
|
|
229
|
+
so a long REPL session stays bounded instead of being re-trimmed from scratch
|
|
230
|
+
every time. `/stats` shows live usage.
|
|
231
|
+
|
|
232
|
+
## Safety
|
|
233
|
+
|
|
234
|
+
- All tool paths resolve through `Workspace`, which refuses anything escaping the
|
|
235
|
+
workspace root after symlink resolution.
|
|
236
|
+
- `.git`, `node_modules`, `.venv` and model weights are hidden from listing,
|
|
237
|
+
search and context injection by default; add your own with `--ignore`.
|
|
238
|
+
- Writes require explicit approval with a diff shown first.
|
|
239
|
+
- `--read-only` disables every mutating tool at the registry level, not just in
|
|
240
|
+
the prompt.
|
|
241
|
+
|
|
242
|
+
## Development
|
|
243
|
+
|
|
244
|
+
```bash
|
|
245
|
+
pip install -e ".[dev]"
|
|
246
|
+
pytest # 70 tests, no model or network required
|
|
247
|
+
ruff check .
|
|
248
|
+
mypy src
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
Tests run against a scripted fake backend, so nothing downloads 1 GB. Coverage
|
|
252
|
+
includes sandbox escapes, the approval gate, tool-call pairing across trims, and
|
|
253
|
+
a simulated 32k session that reads a 20k-line file fifteen times.
|
|
254
|
+
|
|
255
|
+
## Releasing
|
|
256
|
+
|
|
257
|
+
Push a `v*` tag. The workflow verifies the tag matches the `pyproject.toml`
|
|
258
|
+
version, builds the sdist and wheel, checks metadata, smoke-tests the wheel, and
|
|
259
|
+
publishes to PyPI via OIDC trusted publishing — no API token is stored in the
|
|
260
|
+
repository.
|
|
261
|
+
|
|
262
|
+
```bash
|
|
263
|
+
# 1. bump version in pyproject.toml + CHANGELOG.md, commit
|
|
264
|
+
# 2. tag and push
|
|
265
|
+
git tag v0.1.0 && git push origin v0.1.0
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
## License
|
|
269
|
+
|
|
270
|
+
MIT — see [LICENSE](LICENSE).
|
|
271
|
+
|
|
272
|
+
---
|
|
273
|
+
|
|
274
|
+
## Beyond the terminal
|
|
275
|
+
|
|
276
|
+
<p align="center">
|
|
277
|
+
<img src="docs/images/vibe.png" alt="Agentic Studio in Vibe Mode: a multi-pane IDE with a VS-style code editor, a live tool-call log showing grep_files and execute_shell, an agent conversation panel, and L1-L4 context cache gauges along the bottom" width="100%">
|
|
278
|
+
</p>
|
|
279
|
+
|
|
280
|
+
<p align="center"><em>Vibe Mode — the agent snapshots the workspace, then runs
|
|
281
|
+
unattended with the sandbox still hard-caged. Note the live tool log and the
|
|
282
|
+
L1–L4 context gauges along the bottom edge.</em></p>
|
|
283
|
+
|
|
284
|
+
<p align="center">
|
|
285
|
+
<img src="docs/images/ss-ctx1.png" alt="The Context window in Agentic Studio: a model picker, a 125k token slider with 16k through 1M presets, and below it the Token Context Heap heatmap colouring each cluster as free space, pinned, reading, writing, bloated or defragmented" width="100%">
|
|
286
|
+
</p>
|
|
287
|
+
|
|
288
|
+
<p align="center"><em>The Context Engine — a per-model window from 16k to 1M,
|
|
289
|
+
with the Token Context Heap showing exactly which clusters are pinned, bloating
|
|
290
|
+
or still free.</em></p>
|
|
291
|
+
|
|
292
|
+
`qai` is the terminal. If you want the graphical version of the same idea —
|
|
293
|
+
**5 coordinated agents, an inspectable context engine, embedded PHP/MariaDB/WebGPU
|
|
294
|
+
emulators and one-click rollback** — that is
|
|
295
|
+
**[Devhead Agentic Studio](https://dev-head.com/products/astudio.html)**, a
|
|
296
|
+
separate commercial product built by the same author.
|
|
297
|
+
|
|
298
|
+
The two are deliberately different tools: `qai` is free, MIT, and 1,700 lines of
|
|
299
|
+
Python you can audit today; Agentic Studio is a Windows desktop IDE. The CLI
|
|
300
|
+
stays free and MIT-licensed either way — nothing here is a trial, and nothing
|
|
301
|
+
here phones home.
|
|
302
|
+
|
|
303
|
+
*(Screenshots above are from the author's own product page and remain the
|
|
304
|
+
property of their respective owner.)*
|
qai_cli-0.1.0/README.md
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
|
|
3
|
+

|
|
4
|
+

|
|
5
|
+

|
|
6
|
+

|
|
7
|
+

|
|
8
|
+

|
|
9
|
+

|
|
10
|
+
|
|
11
|
+
</div>
|
|
12
|
+
|
|
13
|
+
# local-coding-agent
|
|
14
|
+
|
|
15
|
+
**An offline AI coding assistant that runs entirely on your own machine.**
|
|
16
|
+
`qai` is a terminal coding agent powered by **Qwen2.5-Coder 1.5B** running on
|
|
17
|
+
[llama.cpp](https://github.com/ggml-org/llama.cpp) — no cloud, no API keys, no
|
|
18
|
+
telemetry, no data leaving your laptop.
|
|
19
|
+
|
|
20
|
+
It is deliberately small: about **1,700 lines** of readable Python you can finish
|
|
21
|
+
in one sitting, built to be extended rather than depended on.
|
|
22
|
+
|
|
23
|
+
> **CLI command:** `qai` · **PyPI package:** `qai-cli` · **Repository:**
|
|
24
|
+
> `local-coding-agent` · **License:** MIT
|
|
25
|
+
|
|
26
|
+
```
|
|
27
|
+
qai> add a docstring to every function in utils.py
|
|
28
|
+
tool read_file(path=utils.py)
|
|
29
|
+
ok read_file
|
|
30
|
+
tool edit_file(path=utils.py, old_string=..., new_string=...)
|
|
31
|
+
Approve edit utils.py
|
|
32
|
+
▸ 12 lines changed
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## Why this exists
|
|
38
|
+
|
|
39
|
+
Most "AI coding agent" tools assume a 70B model, a GPU, and a subscription.
|
|
40
|
+
This one assumes nothing:
|
|
41
|
+
|
|
42
|
+
- **Runs a 1.5B model on a laptop CPU.** No CUDA, no 24 GB of VRAM, no
|
|
43
|
+
electricity bill. It is a *small* model, and the tooling around it is built to
|
|
44
|
+
make a small model useful.
|
|
45
|
+
- **~1.1 GB download.** Q4_K_M GGUF weights, cached after the first run.
|
|
46
|
+
- **1,700 lines, fully readable.** The whole agent — sandbox, tools, approval
|
|
47
|
+
gate, context optimizer — is short enough to read, audit and fork.
|
|
48
|
+
- **MIT, no strings.** The safety model is code you can inspect, not a policy
|
|
49
|
+
page.
|
|
50
|
+
|
|
51
|
+
## Features
|
|
52
|
+
|
|
53
|
+
- **Local inference** — GGUF weights from Hugging Face, run through
|
|
54
|
+
`llama-cpp-python`. CPU-only works out of the box; GPU offload is on by default
|
|
55
|
+
when a backend exists.
|
|
56
|
+
- **Agentic file tools** — `read_file`, `edit_file`, `write_file`, `list_dir`,
|
|
57
|
+
`grep`, all sandboxed to the workspace root.
|
|
58
|
+
- **Approval gate** — every write shows a unified diff and waits for you. `--yes`
|
|
59
|
+
skips it, `--read-only` disables writes entirely.
|
|
60
|
+
- **32k context with a real optimizer** — tool results and file dumps are trimmed
|
|
61
|
+
head-and-tail, then older turns are auto-compacted into a summary before the
|
|
62
|
+
buffer fills. Tool calls are never orphaned from their results.
|
|
63
|
+
- **Context injection** — `@path` mentions, `--attach` globs, and an automatic
|
|
64
|
+
workspace manifest so a 1.5B model knows what files exist.
|
|
65
|
+
- **Two modes** — one-shot `qai "..."` for scripting, or a REPL with `@file`
|
|
66
|
+
completion, history and slash commands.
|
|
67
|
+
|
|
68
|
+
## Requirements
|
|
69
|
+
|
|
70
|
+
- Python 3.10 or newer
|
|
71
|
+
- ~2 GB free disk for weights, ~2 GB RAM at inference
|
|
72
|
+
- A C++ toolchain, *or* a prebuilt `llama-cpp-python` wheel (below)
|
|
73
|
+
|
|
74
|
+
## Install
|
|
75
|
+
|
|
76
|
+
### Fastest — no install, no virtualenv
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
uvx qai-cli # run it instantly
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
### Global install, still isolated
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
uv tool install qai-cli # or: pipx install qai-cli
|
|
86
|
+
qai "what does main() do?"
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
> **Local inference needs `llama-cpp-python`**, a compiled C++ extension. If you
|
|
90
|
+
> have no C++ toolchain, use the prebuilt CPU wheel:
|
|
91
|
+
>
|
|
92
|
+
> ```bash
|
|
93
|
+
> pip install llama-cpp-python --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu
|
|
94
|
+
> ```
|
|
95
|
+
>
|
|
96
|
+
> Then run `uvx --with llama-cpp-python qai-cli` to try it in one shot.
|
|
97
|
+
|
|
98
|
+
### From source
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
git clone https://github.com/Nodoubvt/local-coding-agent
|
|
102
|
+
cd local-coding-agent
|
|
103
|
+
|
|
104
|
+
python -m venv .venv
|
|
105
|
+
source .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
106
|
+
pip install -e ".[local]" # [local] adds llama-cpp-python
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Inspect resolved settings, or fetch weights ahead of time:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
qai --print-config
|
|
113
|
+
python -c "from qai.config import load_settings; from qai.backend import ensure_model_file; print(ensure_model_file(load_settings()))"
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
## Usage
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
qai # interactive session
|
|
120
|
+
qai "why does main() hang here?" # one-shot
|
|
121
|
+
qai "refactor this" -a src/api.py # attach a file
|
|
122
|
+
qai "find every TODO" --no-tools # pure chat, no filesystem access
|
|
123
|
+
|
|
124
|
+
qai --model qwen2.5-coder-1.5b-instruct-q8_0.gguf # higher-quality quant
|
|
125
|
+
qai --gpu-layers 0 --threads 8 # force CPU, pin thread count
|
|
126
|
+
qai --read-only # refuse every write
|
|
127
|
+
qai --compact-at 0.5 # summarize earlier, at 50% of budget
|
|
128
|
+
qai --no-compact # disable the compactor
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
### REPL commands
|
|
132
|
+
|
|
133
|
+
| command | effect |
|
|
134
|
+
| --- | --- |
|
|
135
|
+
| `/help` | show commands |
|
|
136
|
+
| `/new` | clear conversation history |
|
|
137
|
+
| `/context [path]` | re-send the manifest, or attach specific paths |
|
|
138
|
+
| `/stats` | context usage, budget, compaction count |
|
|
139
|
+
| `/tools` | list available tools |
|
|
140
|
+
| `/model` | show active model and context size |
|
|
141
|
+
| `/read-only on\|off` | toggle write access |
|
|
142
|
+
| `/exit` | quit (ctrl-d also works) |
|
|
143
|
+
|
|
144
|
+
## Configuration
|
|
145
|
+
|
|
146
|
+
Settings merge in this order: **defaults → `config.json` → env vars → CLI flags**.
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
python -c "from qai.config import Settings, save_settings; save_settings(Settings(n_ctx=32768))"
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
| setting | default | meaning |
|
|
153
|
+
| --- | --- | --- |
|
|
154
|
+
| `n_ctx` | `32768` | context window |
|
|
155
|
+
| `max_tokens` | `4096` | reply ceiling, also reserved from the budget |
|
|
156
|
+
| `compact_threshold` | `0.70` | auto-compact above this context ratio |
|
|
157
|
+
| `max_tool_result_chars` | `8000` | per-result trim limit |
|
|
158
|
+
| `keep_recent_blocks` | `6` | turns kept verbatim through compaction |
|
|
159
|
+
| `auto_compact` | `true` | enable the compactor |
|
|
160
|
+
| `n_gpu_layers` | `-1` | `-1` offload all, `0` CPU only |
|
|
161
|
+
| `temperature` | `0.2` | sampling temperature |
|
|
162
|
+
| `auto_approve` | `false` | skip write approval |
|
|
163
|
+
|
|
164
|
+
Env equivalents: `QAI_N_CTX`, `QAI_TEMPERATURE`, `QAI_AUTO_COMPACT`,
|
|
165
|
+
`QAI_MODEL_DIR`, `QAI_CONFIG_DIR`, and the rest follow `QAI_` + the setting name.
|
|
166
|
+
|
|
167
|
+
## How the context optimizer works
|
|
168
|
+
|
|
169
|
+
A 1.5B model with a 32k window still hits the wall fast, because a single
|
|
170
|
+
`read_file` of a large file can eat a quarter of it. Three layers keep a long
|
|
171
|
+
session degrading instead of failing:
|
|
172
|
+
|
|
173
|
+
1. **Budget.** `budget = n_ctx - max_tokens - tool_schema_reserve` — 27,472
|
|
174
|
+
tokens at defaults. Everything is measured with a deliberately pessimistic
|
|
175
|
+
chars-per-token estimate, so the agent under-fills rather than overflows.
|
|
176
|
+
2. **Trim.** Oversized tool results and `<context>` blocks are cut head-and-tail,
|
|
177
|
+
so the end of an error message survives. Assistant messages carrying
|
|
178
|
+
`tool_calls` are grouped with their results into a single atomic block, so
|
|
179
|
+
trimming can never leave an orphan that breaks the chat template.
|
|
180
|
+
3. **Compact.** Past `compact_threshold`, older turns are summarized — by the
|
|
181
|
+
model when it is available, otherwise by a deterministic extractive fallback —
|
|
182
|
+
into a single `<summary>` block, with the last `keep_recent_blocks` turns kept
|
|
183
|
+
verbatim.
|
|
184
|
+
|
|
185
|
+
Compaction runs between tool rounds, never mid-flight, so the model always sees a
|
|
186
|
+
consistent transcript, and the compacted transcript is carried into the next turn
|
|
187
|
+
so a long REPL session stays bounded instead of being re-trimmed from scratch
|
|
188
|
+
every time. `/stats` shows live usage.
|
|
189
|
+
|
|
190
|
+
## Safety
|
|
191
|
+
|
|
192
|
+
- All tool paths resolve through `Workspace`, which refuses anything escaping the
|
|
193
|
+
workspace root after symlink resolution.
|
|
194
|
+
- `.git`, `node_modules`, `.venv` and model weights are hidden from listing,
|
|
195
|
+
search and context injection by default; add your own with `--ignore`.
|
|
196
|
+
- Writes require explicit approval with a diff shown first.
|
|
197
|
+
- `--read-only` disables every mutating tool at the registry level, not just in
|
|
198
|
+
the prompt.
|
|
199
|
+
|
|
200
|
+
## Development
|
|
201
|
+
|
|
202
|
+
```bash
|
|
203
|
+
pip install -e ".[dev]"
|
|
204
|
+
pytest # 70 tests, no model or network required
|
|
205
|
+
ruff check .
|
|
206
|
+
mypy src
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
Tests run against a scripted fake backend, so nothing downloads 1 GB. Coverage
|
|
210
|
+
includes sandbox escapes, the approval gate, tool-call pairing across trims, and
|
|
211
|
+
a simulated 32k session that reads a 20k-line file fifteen times.
|
|
212
|
+
|
|
213
|
+
## Releasing
|
|
214
|
+
|
|
215
|
+
Push a `v*` tag. The workflow verifies the tag matches the `pyproject.toml`
|
|
216
|
+
version, builds the sdist and wheel, checks metadata, smoke-tests the wheel, and
|
|
217
|
+
publishes to PyPI via OIDC trusted publishing — no API token is stored in the
|
|
218
|
+
repository.
|
|
219
|
+
|
|
220
|
+
```bash
|
|
221
|
+
# 1. bump version in pyproject.toml + CHANGELOG.md, commit
|
|
222
|
+
# 2. tag and push
|
|
223
|
+
git tag v0.1.0 && git push origin v0.1.0
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
## License
|
|
227
|
+
|
|
228
|
+
MIT — see [LICENSE](LICENSE).
|
|
229
|
+
|
|
230
|
+
---
|
|
231
|
+
|
|
232
|
+
## Beyond the terminal
|
|
233
|
+
|
|
234
|
+
<p align="center">
|
|
235
|
+
<img src="docs/images/vibe.png" alt="Agentic Studio in Vibe Mode: a multi-pane IDE with a VS-style code editor, a live tool-call log showing grep_files and execute_shell, an agent conversation panel, and L1-L4 context cache gauges along the bottom" width="100%">
|
|
236
|
+
</p>
|
|
237
|
+
|
|
238
|
+
<p align="center"><em>Vibe Mode — the agent snapshots the workspace, then runs
|
|
239
|
+
unattended with the sandbox still hard-caged. Note the live tool log and the
|
|
240
|
+
L1–L4 context gauges along the bottom edge.</em></p>
|
|
241
|
+
|
|
242
|
+
<p align="center">
|
|
243
|
+
<img src="docs/images/ss-ctx1.png" alt="The Context window in Agentic Studio: a model picker, a 125k token slider with 16k through 1M presets, and below it the Token Context Heap heatmap colouring each cluster as free space, pinned, reading, writing, bloated or defragmented" width="100%">
|
|
244
|
+
</p>
|
|
245
|
+
|
|
246
|
+
<p align="center"><em>The Context Engine — a per-model window from 16k to 1M,
|
|
247
|
+
with the Token Context Heap showing exactly which clusters are pinned, bloating
|
|
248
|
+
or still free.</em></p>
|
|
249
|
+
|
|
250
|
+
`qai` is the terminal. If you want the graphical version of the same idea —
|
|
251
|
+
**5 coordinated agents, an inspectable context engine, embedded PHP/MariaDB/WebGPU
|
|
252
|
+
emulators and one-click rollback** — that is
|
|
253
|
+
**[Devhead Agentic Studio](https://dev-head.com/products/astudio.html)**, a
|
|
254
|
+
separate commercial product built by the same author.
|
|
255
|
+
|
|
256
|
+
The two are deliberately different tools: `qai` is free, MIT, and 1,700 lines of
|
|
257
|
+
Python you can audit today; Agentic Studio is a Windows desktop IDE. The CLI
|
|
258
|
+
stays free and MIT-licensed either way — nothing here is a trial, and nothing
|
|
259
|
+
here phones home.
|
|
260
|
+
|
|
261
|
+
*(Screenshots above are from the author's own product page and remain the
|
|
262
|
+
property of their respective owner.)*
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "qai-cli"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Offline terminal coding agent. Runs Qwen2.5-Coder 1.5B locally via llama.cpp - no API key, no cloud, no telemetry."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
authors = [{ name = "Nodoubvt" }]
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
keywords = [
|
|
14
|
+
"llm",
|
|
15
|
+
"cli",
|
|
16
|
+
"agent",
|
|
17
|
+
"coding-agent",
|
|
18
|
+
"llama.cpp",
|
|
19
|
+
"llama-cpp",
|
|
20
|
+
"gguf",
|
|
21
|
+
"qwen",
|
|
22
|
+
"local-llm",
|
|
23
|
+
"offline",
|
|
24
|
+
"privacy",
|
|
25
|
+
"coding-assistant",
|
|
26
|
+
]
|
|
27
|
+
classifiers = [
|
|
28
|
+
"Development Status :: 3 - Alpha",
|
|
29
|
+
"Environment :: Console",
|
|
30
|
+
"Intended Audience :: Developers",
|
|
31
|
+
"License :: OSI Approved :: MIT License",
|
|
32
|
+
"Operating System :: OS Independent",
|
|
33
|
+
"Programming Language :: Python :: 3",
|
|
34
|
+
"Programming Language :: Python :: 3.10",
|
|
35
|
+
"Programming Language :: Python :: 3.11",
|
|
36
|
+
"Programming Language :: Python :: 3.12",
|
|
37
|
+
"Programming Language :: Python :: 3.13",
|
|
38
|
+
"Topic :: Software Development :: Code Generators",
|
|
39
|
+
"Topic :: Utilities",
|
|
40
|
+
"Typing :: Typed",
|
|
41
|
+
]
|
|
42
|
+
dependencies = [
|
|
43
|
+
"typer>=0.12",
|
|
44
|
+
"rich>=13.7",
|
|
45
|
+
"prompt-toolkit>=3.0.43",
|
|
46
|
+
"platformdirs>=4.2",
|
|
47
|
+
"huggingface-hub>=0.23",
|
|
48
|
+
"pathspec>=0.12",
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
[project.optional-dependencies]
|
|
52
|
+
local = ["llama-cpp-python>=0.3.2"]
|
|
53
|
+
dev = [
|
|
54
|
+
"pytest>=8.2",
|
|
55
|
+
"ruff>=0.5",
|
|
56
|
+
"mypy>=1.11",
|
|
57
|
+
"build>=1.2",
|
|
58
|
+
"twine>=5.1",
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
[project.scripts]
|
|
62
|
+
qai = "qai.cli:run"
|
|
63
|
+
|
|
64
|
+
[project.urls]
|
|
65
|
+
Homepage = "https://github.com/Nodoubvt/local-coding-agent"
|
|
66
|
+
Repository = "https://github.com/Nodoubvt/local-coding-agent"
|
|
67
|
+
Issues = "https://github.com/Nodoubvt/local-coding-agent/issues"
|
|
68
|
+
Changelog = "https://github.com/Nodoubvt/local-coding-agent/blob/main/CHANGELOG.md"
|
|
69
|
+
|
|
70
|
+
[tool.hatch.build.targets.wheel]
|
|
71
|
+
packages = ["src/qai"]
|
|
72
|
+
|
|
73
|
+
[tool.hatch.build.targets.sdist]
|
|
74
|
+
include = ["src/qai", "tests", "README.md", "LICENSE", "pyproject.toml"]
|
|
75
|
+
|
|
76
|
+
[tool.pytest.ini_options]
|
|
77
|
+
testpaths = ["tests"]
|
|
78
|
+
addopts = "-q"
|
|
79
|
+
|
|
80
|
+
[tool.ruff]
|
|
81
|
+
line-length = 100
|
|
82
|
+
target-version = "py310"
|
|
83
|
+
|
|
84
|
+
[tool.ruff.lint]
|
|
85
|
+
select = ["E", "F", "I", "UP", "B", "SIM"]
|
|
86
|
+
|
|
87
|
+
[tool.mypy]
|
|
88
|
+
python_version = "3.10"
|
|
89
|
+
warn_unused_ignores = true
|
|
90
|
+
ignore_missing_imports = true
|