ai-code-engineer 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ai_code_engineer-0.1.0/LICENSE +21 -0
- ai_code_engineer-0.1.0/PKG-INFO +7 -0
- ai_code_engineer-0.1.0/README.md +301 -0
- ai_code_engineer-0.1.0/pyproject.toml +23 -0
- ai_code_engineer-0.1.0/setup.cfg +4 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/__init__.py +2 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/catalog.py +143 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/chat.py +181 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/cli.py +384 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/config.py +405 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/engine.py +1282 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/errors.py +27 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/git_integration.py +443 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/gui.py +2646 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/host.py +81 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/ignore.py +269 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/intent.py +222 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/labels.py +871 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/memory.py +91 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/modes.py +156 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/overrides.py +540 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/planbook.py +192 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/providers.py +404 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/redaction.py +54 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/repair.py +564 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/report.py +352 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/runner.py +854 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/setup.py +386 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/symbols.py +1286 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/verification.py +218 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/__init__.py +1 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/__main__.py +45 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/contract.py +36 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/controller.py +3556 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/fake.py +1141 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/launch.py +108 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/server.py +349 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/app.css +780 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/app.js +2118 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/boot.js +19 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/index.html +89 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/tokens.css +173 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer/workspace.py +385 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/PKG-INFO +7 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/SOURCES.txt +72 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/dependency_links.txt +1 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/entry_points.txt +2 -0
- ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/top_level.txt +1 -0
- ai_code_engineer-0.1.0/tests/test_agent.py +1949 -0
- ai_code_engineer-0.1.0/tests/test_chat.py +216 -0
- ai_code_engineer-0.1.0/tests/test_cli.py +595 -0
- ai_code_engineer-0.1.0/tests/test_connection.py +925 -0
- ai_code_engineer-0.1.0/tests/test_controller.py +3947 -0
- ai_code_engineer-0.1.0/tests/test_doubles.py +150 -0
- ai_code_engineer-0.1.0/tests/test_git.py +911 -0
- ai_code_engineer-0.1.0/tests/test_gui.py +1258 -0
- ai_code_engineer-0.1.0/tests/test_host.py +189 -0
- ai_code_engineer-0.1.0/tests/test_ignore.py +267 -0
- ai_code_engineer-0.1.0/tests/test_intent.py +188 -0
- ai_code_engineer-0.1.0/tests/test_labels.py +586 -0
- ai_code_engineer-0.1.0/tests/test_memory.py +95 -0
- ai_code_engineer-0.1.0/tests/test_modes.py +153 -0
- ai_code_engineer-0.1.0/tests/test_overrides.py +620 -0
- ai_code_engineer-0.1.0/tests/test_planbook.py +214 -0
- ai_code_engineer-0.1.0/tests/test_redaction.py +95 -0
- ai_code_engineer-0.1.0/tests/test_repair.py +721 -0
- ai_code_engineer-0.1.0/tests/test_report.py +264 -0
- ai_code_engineer-0.1.0/tests/test_runner.py +831 -0
- ai_code_engineer-0.1.0/tests/test_setup.py +443 -0
- ai_code_engineer-0.1.0/tests/test_symbols.py +1204 -0
- ai_code_engineer-0.1.0/tests/test_transport.py +217 -0
- ai_code_engineer-0.1.0/tests/test_verification.py +514 -0
- ai_code_engineer-0.1.0/tests/test_webapp.py +1670 -0
- ai_code_engineer-0.1.0/tests/test_workspace.py +451 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Mohamed Saad (mohamedsaad208)
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
|
|
3
|
+
<img src="assets/banner.svg" alt="AI Code Engineer Banner" width="100%"/>
|
|
4
|
+
|
|
5
|
+
# AI Code Engineer
|
|
6
|
+
### Give your coding workflow an autonomous, privacy-first software engineer.
|
|
7
|
+
|
|
8
|
+
AI Code Engineer is an open-source autonomous agent framework for real-world software engineering: AST repository indexing, cryptographic diff proposals, zero-trust workspace security, automated JUnit test loops, and multi-provider LLM support (Ollama local, OpenRouter, OpenAI, Groq, DeepSeek).
|
|
9
|
+
|
|
10
|
+
[](LICENSE)
|
|
11
|
+
[](https://www.python.org/)
|
|
12
|
+
[]()
|
|
13
|
+
[]()
|
|
14
|
+
[](https://github.com/mohamedsaad208/AI-Agent/actions/workflows/ci.yml)
|
|
15
|
+
|
|
16
|
+
[Quick Start](#-quick-start) •
|
|
17
|
+
[Why AI Code Engineer](#-why-ai-code-engineer) •
|
|
18
|
+
[Platform Guide](#-platform-guide) •
|
|
19
|
+
[Try These First](#-try-these-first) •
|
|
20
|
+
[Architecture](#-architecture) •
|
|
21
|
+
[Contributing](#-contributing)
|
|
22
|
+
|
|
23
|
+
<br/>
|
|
24
|
+
|
|
25
|
+
<img src="assets/demo-complex.gif" alt="AI Code Engineer Multi-File Refactor & Self-Healing Loop" width="100%"/>
|
|
26
|
+
|
|
27
|
+
</div>
|
|
28
|
+
|
|
29
|
+
---
|
|
30
|
+
|
|
31
|
+
# 🚀 Why AI Code Engineer
|
|
32
|
+
|
|
33
|
+
| Feature | Why it matters |
|
|
34
|
+
| :--- | :--- |
|
|
35
|
+
| 🔒 **Zero-Trust Security** | Path-traversal guards, symlink blocking, and automatic secret redaction (AWS, GitHub tokens, Bearer keys, `.env`) prevent leakage to logs or LLMs. |
|
|
36
|
+
| 🛡️ **Reviewed diffs** | Every change is a proposal with a `SHA-256` hash, and applying it is a decision you make (or hand over per folder, explicitly). Rollback is one step, as long as nothing else edited those files afterwards. |
|
|
37
|
+
| 🌿 **Git Checkpoints & Restore** | Applying an approved proposal inside a git repo automatically creates a non-destructive checkpoint commit (`--no-verify`, skips hooks). If subsequent edits block local rollback, targeted single-file git restore offers a safe escalation path back to the pre-task commit. Never pushes, pulls, or rewrites history. |
|
|
38
|
+
| 🌐 **Native Bilingual & RTL** | First-class Arabic and English dual-engine. Dynamic Right-to-Left (RTL) support in the WebApp, automatic language detection (`is_arabic`), and fully localized system notices, diagnostic reports, and `--arabic` CLI flags. |
|
|
39
|
+
| ⚡ **100% Offline & Local** | Full first-class support for **Ollama** (`qwen2.5-coder`, `deepseek-coder`, `llama3`). Code stays on your hardware. |
|
|
40
|
+
| ☁️ **Multi-Provider Cloud** | Seamlessly switch between **OpenRouter**, **OpenAI**, **Groq**, **DeepSeek**, or custom OpenAI-compatible endpoints. |
|
|
41
|
+
| 🔌 **Endpoints are configuration** | No model server's address is written in the code — the provider table carries one default per row, and a URL resolves as *what you typed → your environment (`OLLAMA_HOST`, `GROQ_BASE_URL`, …) → the table's own row*. Point Ollama at another port without editing a file, and a profile that forgets its endpoint gets **its own** provider's address, never a local default. |
|
|
42
|
+
| 🗺️ **AST Symbol Indexing** | In-memory symbol extractor (Python, Java/Kotlin, TypeScript/JS, Go, Rust) provides classes, methods, and types without burning context window tokens. |
|
|
43
|
+
| 🧪 **Self-Healing Test Loop** | Auto-detects `pytest`, `unittest`, `Maven`, `Gradle`, `npm`, `cargo`, `go test`. Parses JUnit XML output and feeds failures back to the agent for autonomous repair (up to 3 rounds). |
|
|
44
|
+
| ⏱️ **Zero-Drop Task Queueing** | Messages typed while a task or build is in progress are safely enqueued without race conditions, running automatically in FIFO order when the active job finishes. |
|
|
45
|
+
| 📑 **Session Audit & Export** | Transcripts, diffs, and proof tallies are exportable to structured JSON or clean, readable Markdown reports (`agent export-session`) for documentation and audits. |
|
|
46
|
+
| 🖥️ **Desktop WebApp & CLI** | Beautiful local WebApp with real-time streaming, diff previews, task queuing, and an interactive terminal menu. |
|
|
47
|
+
|
|
48
|
+
---
|
|
49
|
+
|
|
50
|
+
# 🧭 Three modes, and the limits that go with them
|
|
51
|
+
|
|
52
|
+
**Chat** answers in prose and reads no files until you hand it a folder. **Change** produces a
|
|
53
|
+
proposal you review, and only `Apply` writes. **Auto-Apply** is a switch you turn on *per folder*:
|
|
54
|
+
the agent then writes what it proposes without a click, runs that folder's own command afterwards,
|
|
55
|
+
and keeps the rollback. It still asks first if the proposal would empty or delete an existing file, or
|
|
56
|
+
if a previous task left that folder half-written. The window says so on the card and in the Activity
|
|
57
|
+
log when a write happened without a click — that is the one thing about this tool that is easiest to
|
|
58
|
+
forget and hardest to undo.
|
|
59
|
+
|
|
60
|
+
| What you should know before you rely on it | |
|
|
61
|
+
| :--- | :--- |
|
|
62
|
+
| **Run and Check syntax execute your project's own code** | They invoke `mvn`, `gradle`, `npm`, `pytest`, `cargo`, `go` in that folder with *your* permissions. A build script is code, and code from a repository you did not write gets run here. The command list is an allowlist and the environment is stripped of credentials — that is a **limit**, not a sandbox. The row below is the sandbox. |
|
|
63
|
+
| **The sandbox is a tick on the Run button** | `Run in Docker` runs the project's command against a **copy** of the folder, in a container with no network, no capabilities, a read-only root and an image pinned by `sha256` digest — so the build's reports are still read afterwards, and your tree is never mounted into it. Without Docker the run is refused rather than quietly done on the host. Verified here as an argv, not as a build: no container has run on the machine this was written on. |
|
|
64
|
+
| **Read-only is a fact about the folder, not about a window** | Setting it in either window, or sealing a folder with `agent read-only --repo PATH`, is written down where every surface reads it back: a terminal that never opened a window cannot `apply` or `rollback` into it, and one window saving its own preferences cannot unseal what the other was told. Lifting it is an explicit act, and the refusal names the line that does it. What it never does is stop reading, searching, mapping, a static check, or a command you asked for by name. |
|
|
65
|
+
| **The settings you change from inside the program are signed, not encrypted** | `Settings → Overrides` and `agent overrides --set TARGET KEY VALUE` write `.agent-overrides.json` in this tool's own folder — created at first run, `.gitignore`d, never written into your project. The program signs every row, so a row you typed into the JSON by hand (or one moved, changed, added or deleted there) is **refused and named** rather than quietly obeyed, and the run falls back to your profile. That is tamper evidence, not a password: your own account can rewrite the file, and the standard library has no cipher. It never redirects an address a profile or a field already states, never swaps the provider, and never holds a key — a credential-shaped value is refused on the way in. |
|
|
66
|
+
| **Git checkpoints back every applied proposal** | Inside a git repository, applying a proposal commits the approved files with `--no-verify` (skipping hooks) and names the session. If files are edited after review, local rollback is blocked to prevent clobbering your later edits, and the tool offers **Targeted Git Restore** to put only that task's files back to its pre-task commit. Strictly no network commands (`push`/`pull`) and no rewritten history. |
|
|
67
|
+
| **The repository map is context, not a compiler** | Symbols are parsed with `ast` for Python and bounded scanners for Java, Kotlin, Go, Rust and TypeScript. It tells the model what files declare; it does not type-check, resolve imports or prove the code works. Only running the project's command does that, and a run that never ran is reported as `unverified`, not as a pass. |
|
|
68
|
+
| **Small local models write small diffs** | The reference setup is a CPU-only `qwen2.5-coder` on Ollama. Larger models produce better proposals; none of them produce a diff you should apply without reading. |
|
|
69
|
+
| **A cloud endpoint means your code leaves the device** | Cloud rows are refused until you approve, the approval is asked per task, cleartext to a remote host is refused outright, and API keys live in memory only — never in a config file, a log line or an error message. |
|
|
70
|
+
|
|
71
|
+
Running `python agent.py doctor` prints what this machine can actually reach — the local model list,
|
|
72
|
+
whether Docker is installed, and which key variables are set (their values are never printed).
|
|
73
|
+
|
|
74
|
+
---
|
|
75
|
+
|
|
76
|
+
# ⚡ Quick Start
|
|
77
|
+
|
|
78
|
+
## 1. Clone & Verify
|
|
79
|
+
```bash
|
|
80
|
+
git clone https://github.com/mohamedsaad208/AI-Agent.git
|
|
81
|
+
cd AI-Agent
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
# First run, in order: what this machine can run, who answers, which models exist, what the folder
|
|
86
|
+
# grants, the offline proof, the five promises, and what `Send` may become. Exit 1 if a row blocks.
|
|
87
|
+
python agent.py setup # add --repo PATH to check a folder, --arabic, or --yes
|
|
88
|
+
|
|
89
|
+
# Self-diagnostics: python version, Ollama reachability, Docker, key variables (never their values)
|
|
90
|
+
python agent.py doctor
|
|
91
|
+
|
|
92
|
+
# Deterministic offline demo — no model, no network, and nothing from your repository is executed
|
|
93
|
+
python agent.py demo
|
|
94
|
+
|
|
95
|
+
# The suite. It is stdlib-only and needs no install step: `tests/*` add `src/` to sys.path itself.
|
|
96
|
+
python -m unittest discover -s tests
|
|
97
|
+
|
|
98
|
+
# Windows: the same command, with the interpreter this project is counted against.
|
|
99
|
+
# 3.11 is named on purpose — newer interpreters tally subtests differently, so the total moves.
|
|
100
|
+
run-tests.cmd
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
The suite is the gate CI runs (`.github/workflows/ci.yml`: 3.11 on Windows and Linux, `compileall`,
|
|
104
|
+
`node --check` on the two UI scripts, a wheel build checked for the files the window needs). There is
|
|
105
|
+
no pytest, no `pip install -e .` step and no network access in it.
|
|
106
|
+
|
|
107
|
+
To open the local web window without an engine behind it — the same UI, scripted data, useful for
|
|
108
|
+
reading the interface before trusting a folder to it:
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
python -m ai_code_engineer.webapp --fake --no-browser --port 8765
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
---
|
|
115
|
+
|
|
116
|
+
# 💻 Platform Guide
|
|
117
|
+
|
|
118
|
+
| Platform | GUI WebApp Launcher | Interactive CLI Launcher | Direct Terminal Command |
|
|
119
|
+
| :--- | :--- | :--- | :--- |
|
|
120
|
+
| **🪟 Windows** | Double-click `Run-Agent.bat` | Double-click `Run-Agent-CLI.bat` | `python desktop.pyw` |
|
|
121
|
+
| **🐧 Linux** | `./run-agent.sh` | `./run-agent-cli.sh` | `python3 desktop.pyw` |
|
|
122
|
+
| **🍎 macOS** | `./run-agent.sh` | `./run-agent-cli.sh` | `python3 desktop.pyw` |
|
|
123
|
+
|
|
124
|
+
<details>
|
|
125
|
+
<summary><strong>🐧 Linux & 🍎 macOS First-Time Setup</strong></summary>
|
|
126
|
+
|
|
127
|
+
Make the shell launchers executable:
|
|
128
|
+
```bash
|
|
129
|
+
chmod +x run-agent.sh run-agent-cli.sh
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
Launch the GUI:
|
|
133
|
+
```bash
|
|
134
|
+
./run-agent.sh
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
Launch the interactive CLI:
|
|
138
|
+
```bash
|
|
139
|
+
./run-agent-cli.sh
|
|
140
|
+
```
|
|
141
|
+
</details>
|
|
142
|
+
|
|
143
|
+
<details>
|
|
144
|
+
<summary><strong>⚙️ Advanced Headless CLI Usage</strong></summary>
|
|
145
|
+
|
|
146
|
+
```powershell
|
|
147
|
+
# Index repository symbols
|
|
148
|
+
python agent.py map --repo examples/demo_repo
|
|
149
|
+
|
|
150
|
+
# Declare a folder Read-only for every surface — window and terminal alike — and lift it again
|
|
151
|
+
python agent.py read-only --repo examples/demo_repo
|
|
152
|
+
python agent.py read-only --repo examples/demo_repo --off
|
|
153
|
+
|
|
154
|
+
# Plan and propose code changes with local Ollama
|
|
155
|
+
python agent.py plan "Fix add in calculator.py so it adds two numbers" --repo examples/demo_repo --config profiles/local.toml
|
|
156
|
+
|
|
157
|
+
# Review proposed diff
|
|
158
|
+
python agent.py review "<session_id>"
|
|
159
|
+
|
|
160
|
+
# Apply approved proposal (cryptographically verified)
|
|
161
|
+
python agent.py apply "<session_id>" --approve "<sha256_hash>"
|
|
162
|
+
|
|
163
|
+
# Execute automated test suite
|
|
164
|
+
python agent.py verify "<session_id>"
|
|
165
|
+
|
|
166
|
+
# Check status and outcome of any session
|
|
167
|
+
python agent.py status "<session_id>"
|
|
168
|
+
|
|
169
|
+
# Export session transcript and diff report to Markdown or JSON
|
|
170
|
+
python agent.py export-session "<session_id>" --format markdown --out session-report.md
|
|
171
|
+
|
|
172
|
+
# Rollback if needed
|
|
173
|
+
python agent.py rollback "<session_id>" --approve "<sha256_hash>"
|
|
174
|
+
```
|
|
175
|
+
</details>
|
|
176
|
+
|
|
177
|
+
---
|
|
178
|
+
|
|
179
|
+
# 🖥️ Interactive Desktop WebApp
|
|
180
|
+
|
|
181
|
+
<div align="center">
|
|
182
|
+
<img src="assets/demo-webapp.gif" alt="AI Code Engineer Desktop WebApp Interface" width="100%"/>
|
|
183
|
+
</div>
|
|
184
|
+
|
|
185
|
+
The application features a sleek, local WebApp interface served on `127.0.0.1` with:
|
|
186
|
+
- 📂 **Multi-Project Workspace:** Manage isolated branches, project memory notes, and saved tasks.
|
|
187
|
+
- ⚡ **Review-First Diff Inspector:** Side-by-side **Diff / Now / Was** inspector with one-click rollbacks.
|
|
188
|
+
- 🤖 **Universal Model Selector:** Switch on the fly between local models (Ollama/DeepSeek) and cloud APIs (Groq, OpenAI, OpenRouter). A thinking model's deliberation arrives as its own collapsible row — capped, redacted, and never folded into the JSON the loop acts on — and an answer or a proposal streams in as it is written instead of appearing all at once after a minute of dots.
|
|
189
|
+
- 🧪 **Evidence-Based Checks:** Native test suite runner with JUnit XML proofs and self-healing fix rounds.
|
|
190
|
+
- 🐳 **A sandbox you can tick:** the same Checks card runs the project's command inside a pinned image —
|
|
191
|
+
no network, no capabilities, the project as a copy — and says so in the run line and in the exported
|
|
192
|
+
report, so a green from a container never reads as a green from your machine.
|
|
193
|
+
- 🧭 **First-Run Card:** On a machine that has never granted a folder, the window opens with the same
|
|
194
|
+
audit `agent setup` prints — what it can run, what answers, the five promises, the three positions —
|
|
195
|
+
built from rows that ask nothing over the network until you press **Run the checks**.
|
|
196
|
+
- ✏️ **Overrides you can see:** Settings → Overrides lists every configuration row the program is
|
|
197
|
+
running on — whose profile, whose field, whose signed file — and refuses to pretend a row it cannot
|
|
198
|
+
verify is in force. Both windows and the terminal read the same one file.
|
|
199
|
+
- 🌐 **Full Bilingual Arabic & RTL Support:** Dynamic Right-to-Left (RTL) layout when interacting in Arabic, with comprehensive Arabic localization across system notices, error diagnostics, step cards, and review audits.
|
|
200
|
+
|
|
201
|
+
Both windows are the same product: the web window and the `--tk` fallback share the engine, the
|
|
202
|
+
sentences and the decisions, and each keeps its proposed files in a viewer of its own — a sheet over the
|
|
203
|
+
chat there, a second window you can move beside the conversation on the desktop.
|
|
204
|
+
|
|
205
|
+
---
|
|
206
|
+
|
|
207
|
+
# 🎯 Try These First
|
|
208
|
+
|
|
209
|
+
- **Fix a Bug with Automated Verification:**
|
|
210
|
+
*"Read the failing tests in `tests/test_auth.py` and inspect `src/auth.py`. Fix the token expiration validation without breaking backwards compatibility, then run tests."*
|
|
211
|
+
|
|
212
|
+
- **Implement a Multi-Phase Plan:**
|
|
213
|
+
Attach a `plan.md` file using the **+ Plan** button in the WebApp:
|
|
214
|
+
*"Implement Phase 1 from the attached plan only. Create missing DTO classes and verify syntax."*
|
|
215
|
+
|
|
216
|
+
- **Refactor with AST Context:**
|
|
217
|
+
*"Inspect the repository structure and refactor `UserService` to extract email notifications into an independent `NotificationService` interface."*
|
|
218
|
+
|
|
219
|
+
- **Autonomous Self-Healing Loop:**
|
|
220
|
+
Click **Run & Fix** on the Checks card to allow the agent to run the test suite, read compiler errors, and rewrite code until all tests turn green.
|
|
221
|
+
|
|
222
|
+
---
|
|
223
|
+
|
|
224
|
+
# 🏗️ Architecture
|
|
225
|
+
|
|
226
|
+
```
|
|
227
|
+
+---------------------------------------+
|
|
228
|
+
| Desktop WebApp / Native UI |
|
|
229
|
+
+---------------------------------------+
|
|
230
|
+
|
|
|
231
|
+
v
|
|
232
|
+
+---------------------------------------------------------------------------------+
|
|
233
|
+
| Agent Orchestration Engine |
|
|
234
|
+
| - Task Planner & Queue Manager - Step-by-Step Ledger (Planbook) |
|
|
235
|
+
| - AST Symbol Indexer & Repo Map - Persistent Project Memory |
|
|
236
|
+
+---------------------------------------------------------------------------------+
|
|
237
|
+
| |
|
|
238
|
+
v v
|
|
239
|
+
+--------------------------+ +-------------------------------+
|
|
240
|
+
| Model Providers | | Workspace & Security |
|
|
241
|
+
| - Ollama (Local) | | - Path Traversal Guard |
|
|
242
|
+
| - OpenRouter (Cloud) | | - Zero-Trust Secret Redactor |
|
|
243
|
+
| - OpenAI / Groq / Custom| | - SHA-256 Hash-Locked Diffs |
|
|
244
|
+
+--------------------------+ +-------------------------------+
|
|
245
|
+
|
|
|
246
|
+
v
|
|
247
|
+
+-------------------------------+
|
|
248
|
+
| Verification & Checks |
|
|
249
|
+
| - Toolchain Detectors |
|
|
250
|
+
| - JUnit XML Evidence Parser |
|
|
251
|
+
| - Autonomous Repair Loop |
|
|
252
|
+
| - Optional Docker Sandbox |
|
|
253
|
+
+-------------------------------+
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
---
|
|
257
|
+
|
|
258
|
+
# 📁 Where everything lives
|
|
259
|
+
|
|
260
|
+
| path | what it is |
|
|
261
|
+
| :--- | :--- |
|
|
262
|
+
| `src/ai_code_engineer/` | **the product.** `engine.py` runs the loop, `config.py` is the provider table, `providers.py` and `catalog.py` speak to a model, `runner.py` runs *your* project's command — here, or inside the one container shape the tool knows how to seal — `labels.py` holds every sentence in both languages, `intent.py` holds the three write positions and every refusal they speak, `modes.py` keeps a folder's position where the other window and the terminal both read it, `host.py` is the seam the two windows share, `redaction.py` keeps credentials out of what gets stored. |
|
|
263
|
+
| `src/ai_code_engineer/webapp/` | the local web window: `server.py` (loopback-only, per-launch token, Host/Origin/CSP), `controller.py` (the state the UI reads), `static/`. |
|
|
264
|
+
| `src/ai_code_engineer/gui.py` | the Tk window. Same engine, same sentences, different screen. |
|
|
265
|
+
| `tests/` | **the gate.** 1534 offline tests, stdlib `unittest`, no network. `doubles.py` and `helpers.py` are the shared fixtures. |
|
|
266
|
+
| `agent.py` · `desktop.pyw` · `launcher.py` | entry points: CLI, the desktop window, the interactive menu. |
|
|
267
|
+
| `profiles/` | TOML model presets. They name the *variable* holding a key and never a key. |
|
|
268
|
+
| `docs/` | plans, implementation status, code reviews — **local working notes, gitignored.** They quote this machine's paths, ports and counts, so they are kept on the device that measured them instead of shipped as product files. The rules they describe live in the modules' own docstrings, and the test suite is the gate: nothing in `src/`, `tests/` or CI reads a `docs/` file. |
|
|
269
|
+
| `tools/` | development aids for this repository — contrast checks, the dogfood ledger, wheel inspection, demo generators. |
|
|
270
|
+
| `sandbox/` | probes that produced a number someone quoted, kept so the number can be re-measured. |
|
|
271
|
+
| `examples/` | folders the agent is pointed at to try it out. |
|
|
272
|
+
| `archive/` | see `archive/README.md` — including why two experiment folders were **not** moved into it. |
|
|
273
|
+
| `.agent-chats/`, `.agent-runs/`, `.agent-projects.json` | what the tool writes next to itself: conversations, run records, the granted-folder registry. Ignored by git, and the history in them is addressed by absolute path. |
|
|
274
|
+
|
|
275
|
+
---
|
|
276
|
+
|
|
277
|
+
# 🤝 Contributing
|
|
278
|
+
|
|
279
|
+
We welcome contributions from the global open-source community!
|
|
280
|
+
|
|
281
|
+
1. **Fork** the repository.
|
|
282
|
+
2. **Create your feature branch:**
|
|
283
|
+
```bash
|
|
284
|
+
git checkout -b feature/amazing-feature
|
|
285
|
+
```
|
|
286
|
+
3. **Ensure all tests pass:**
|
|
287
|
+
```bash
|
|
288
|
+
python -m unittest discover -s tests -v
|
|
289
|
+
```
|
|
290
|
+
4. **Commit your changes:**
|
|
291
|
+
```bash
|
|
292
|
+
git commit -m "feat: add amazing new feature"
|
|
293
|
+
```
|
|
294
|
+
5. **Push to your fork and submit a Pull Request (PR).**
|
|
295
|
+
|
|
296
|
+
> ⚠️ **Branch Protection Note:** Direct pushes to `main` are restricted. All contributions must go through Pull Requests and pass automated verification.
|
|
297
|
+
|
|
298
|
+
---
|
|
299
|
+
|
|
300
|
+
# 📄 License
|
|
301
|
+
This project is open-source software licensed under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "ai-code-engineer"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "A local-first, review-first developer agent for Ollama and OpenRouter"
|
|
9
|
+
requires-python = ">=3.11"
|
|
10
|
+
dependencies = []
|
|
11
|
+
|
|
12
|
+
[project.scripts]
|
|
13
|
+
agent = "ai_code_engineer.cli:main"
|
|
14
|
+
|
|
15
|
+
[tool.setuptools.packages.find]
|
|
16
|
+
where = ["src"]
|
|
17
|
+
|
|
18
|
+
# webapp/server.py resolves `STATIC = Path(__file__).resolve().parent / "static"` at runtime,
|
|
19
|
+
# so the UI has to travel inside the package. Without this the built wheel contains only .py
|
|
20
|
+
# files and an installed app starts with a blank window. `static/*` also covers any file later
|
|
21
|
+
# added to that folder, which is the only place under src/ holding non-.py resources today.
|
|
22
|
+
[tool.setuptools.package-data]
|
|
23
|
+
"ai_code_engineer.webapp" = ["static/*"]
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""Read-only model discovery. Catalog requests do not submit source code or generate tokens.
|
|
2
|
+
|
|
3
|
+
Discovery and generation must read the *same* endpoint. They used to disagree — the Ollama list was
|
|
4
|
+
a literal loopback URL while generation used ``settings.endpoint`` — so a relocated service listed
|
|
5
|
+
zero models and the window called that "no models found" instead of "wrong address".
|
|
6
|
+
"""
|
|
7
|
+
from decimal import Decimal, InvalidOperation
|
|
8
|
+
import os
|
|
9
|
+
|
|
10
|
+
from .config import Kind, OLLAMA, OPENROUTER, check_endpoint
|
|
11
|
+
from .errors import ProviderError
|
|
12
|
+
from .providers import request_json
|
|
13
|
+
|
|
14
|
+
LIVE = "live"
|
|
15
|
+
BUILT_IN = "built-in"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def ollama_models(endpoint: str = "") -> list[dict]:
|
|
19
|
+
base = check_endpoint(OLLAMA, endpoint)
|
|
20
|
+
data = request_json(base + "/api/tags", timeout=10)
|
|
21
|
+
entries = data.get("models")
|
|
22
|
+
if not isinstance(entries, list):
|
|
23
|
+
raise ProviderError("Ollama returned an invalid model list.")
|
|
24
|
+
found = {}
|
|
25
|
+
for item in entries:
|
|
26
|
+
if not isinstance(item, dict):
|
|
27
|
+
continue
|
|
28
|
+
name = item.get("name") or item.get("model")
|
|
29
|
+
if not isinstance(name, str) or not name:
|
|
30
|
+
continue
|
|
31
|
+
cloud = bool("cloud" in name.casefold() or item.get("remote_host") or item.get("remote_model"))
|
|
32
|
+
size = item.get("size")
|
|
33
|
+
size_label = f"{size / 1_000_000_000:.2f} GB" if isinstance(size, (int, float)) and size > 0 else ""
|
|
34
|
+
found[name] = {"id": name, "name": name, "cloud": cloud,
|
|
35
|
+
"description": ("Ollama cloud model — internet and Ollama account access required. "
|
|
36
|
+
"Billing depends on your Ollama plan." if cloud else "Runs locally on your device.")
|
|
37
|
+
+ (" | " + size_label if size_label else "")}
|
|
38
|
+
return sorted(found.values(), key=lambda item: item["id"].casefold())
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def price(value) -> Decimal | None:
|
|
42
|
+
try:
|
|
43
|
+
result = Decimal(str(value))
|
|
44
|
+
return result if result.is_finite() and result >= 0 else None
|
|
45
|
+
except (InvalidOperation, ValueError, TypeError):
|
|
46
|
+
return None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def openrouter_models(api_key: str | None = None, endpoint: str = "") -> list[dict]:
|
|
50
|
+
base = check_endpoint(OPENROUTER, endpoint)
|
|
51
|
+
data = request_json(base + "/models", key=api_key or os.environ.get("OPENROUTER_API_KEY"),
|
|
52
|
+
timeout=20, max_bytes=16_000_000)
|
|
53
|
+
entries = data.get("data")
|
|
54
|
+
if not isinstance(entries, list):
|
|
55
|
+
raise ProviderError("OpenRouter returned an invalid model catalog.")
|
|
56
|
+
found = {}
|
|
57
|
+
for item in entries:
|
|
58
|
+
if not isinstance(item, dict):
|
|
59
|
+
continue
|
|
60
|
+
name = item.get("id")
|
|
61
|
+
if not isinstance(name, str) or not name:
|
|
62
|
+
continue
|
|
63
|
+
architecture = item.get("architecture") or {}
|
|
64
|
+
if "text" not in architecture.get("output_modalities", []):
|
|
65
|
+
continue
|
|
66
|
+
pricing = item.get("pricing") or {}
|
|
67
|
+
prompt, completion = price(pricing.get("prompt")), price(pricing.get("completion"))
|
|
68
|
+
request_cost = price(pricing.get("request", "0"))
|
|
69
|
+
free = (name == "openrouter/free" or name.endswith(":free")) and all(
|
|
70
|
+
amount == 0 for amount in (prompt, completion, request_cost))
|
|
71
|
+
# A free suffix with missing/contradictory pricing is not advertised as free.
|
|
72
|
+
if (name.endswith(":free") or name == "openrouter/free") and not free:
|
|
73
|
+
continue
|
|
74
|
+
label = lambda cost: "unknown" if cost is None else f"${cost * 1_000_000:,.4f}"
|
|
75
|
+
pricing_text = "Free inference; provider limits apply." if free else (
|
|
76
|
+
f"Input {label(prompt)} / 1M tokens | Output {label(completion)} / 1M tokens")
|
|
77
|
+
if request_cost:
|
|
78
|
+
pricing_text += f" | ${request_cost} / request"
|
|
79
|
+
context = item.get("context_length")
|
|
80
|
+
context_text = f" | Context: {context:,}" if isinstance(context, int) else ""
|
|
81
|
+
params = item.get("supported_parameters") or []
|
|
82
|
+
json_support = "response_format" in params or "structured_outputs" in params
|
|
83
|
+
found[name] = {"id": name, "name": item.get("name", name), "free": free, "cloud": True,
|
|
84
|
+
"description": pricing_text + context_text +
|
|
85
|
+
(" | JSON output listed" if json_support else " | JSON support not listed")}
|
|
86
|
+
return sorted(found.values(), key=lambda item: item["id"].casefold())
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def openai_models(base: str, api_key: str | None = None, *, cloud: bool = True) -> list[dict]:
|
|
90
|
+
"""``GET {base}/models`` — the OpenAI-shaped list every compatible server answers.
|
|
91
|
+
|
|
92
|
+
The payload is a list of identifiers and nothing else that is useful here: pricing is not in
|
|
93
|
+
it, so the description says what the request can prove (where it runs) and nothing more.
|
|
94
|
+
"""
|
|
95
|
+
data = request_json(base + "/models", key=api_key, timeout=20, max_bytes=8_000_000)
|
|
96
|
+
entries = data.get("data")
|
|
97
|
+
if not isinstance(entries, list):
|
|
98
|
+
entries = data.get("models")
|
|
99
|
+
if not isinstance(entries, list):
|
|
100
|
+
raise ProviderError("This provider returned an invalid model list.")
|
|
101
|
+
found = {}
|
|
102
|
+
for item in entries:
|
|
103
|
+
name = item.get("id") or item.get("name") if isinstance(item, dict) else item
|
|
104
|
+
if not isinstance(name, str) or not name.strip():
|
|
105
|
+
continue
|
|
106
|
+
found[name] = {"id": name, "name": name, "cloud": cloud,
|
|
107
|
+
"free": False,
|
|
108
|
+
"description": "Listed by this provider's /models endpoint. Pricing is not "
|
|
109
|
+
"reported there — check the service."}
|
|
110
|
+
return sorted(found.values(), key=lambda item: item["id"].casefold())
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def built_in(kind: Kind) -> list[dict]:
|
|
114
|
+
"""The names shipped with the row, used when the live request fails.
|
|
115
|
+
|
|
116
|
+
They are a starting point, not a claim that the service still lists them: the sentence that
|
|
117
|
+
presents them says so, because a stale id costs one refused request while a false promise of
|
|
118
|
+
"available" costs a task.
|
|
119
|
+
"""
|
|
120
|
+
return [{"id": name, "name": name, "cloud": kind.cloud, "free": False,
|
|
121
|
+
"description": "Built-in name for this provider — not confirmed by a live request."}
|
|
122
|
+
for name in kind.verified]
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def models_for(kind: Kind, endpoint: str = "",
|
|
126
|
+
api_key: str | None = None) -> tuple[list[dict], str]:
|
|
127
|
+
"""Discover what a provider row has, and say *where* the answer came from.
|
|
128
|
+
|
|
129
|
+
Returns ``(entries, source)`` with ``source`` one of ``LIVE`` or ``BUILT_IN``. Only rows that
|
|
130
|
+
carry verified names fall back; a local server with nothing on it correctly reports zero models
|
|
131
|
+
rather than a list of guesses.
|
|
132
|
+
"""
|
|
133
|
+
base = check_endpoint(kind, endpoint)
|
|
134
|
+
try:
|
|
135
|
+
if kind.shape == "ollama":
|
|
136
|
+
return ollama_models(base), LIVE
|
|
137
|
+
if kind.key == OPENROUTER.key:
|
|
138
|
+
return openrouter_models(api_key), LIVE
|
|
139
|
+
return openai_models(base, api_key, cloud=kind.cloud), LIVE
|
|
140
|
+
except ProviderError:
|
|
141
|
+
if kind.verified:
|
|
142
|
+
return built_in(kind), BUILT_IN
|
|
143
|
+
raise
|