golden-agent 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,15 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *$py.class
4
+ *.egg-info/
5
+ dist/
6
+ build/
7
+ .eggs/
8
+ *.egg
9
+ .env
10
+ .venv/
11
+ venv/
12
+ .pytest_cache/
13
+ .ruff_cache/
14
+ .mypy_cache/
15
+ .vscode/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Yashneil Gajjala
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,158 @@
1
+ Metadata-Version: 2.5
2
+ Name: golden-agent
3
+ Version: 0.1.0
4
+ Summary: Your local AI agent. Runs on basically any hardware — auto-setup, GPU-accelerated, zero cloud, zero API keys.
5
+ Author-email: Yashneil Gajjala <yashneil75@gmail.com>
6
+ License-Expression: MIT
7
+ License-File: LICENSE
8
+ Keywords: agent,cli,coding-agent,gguf,llama-cpp,llm,local,local-llm,offline,tools,vulkan
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Environment :: Console
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Operating System :: MacOS
14
+ Classifier: Operating System :: Microsoft :: Windows
15
+ Classifier: Operating System :: POSIX :: Linux
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Topic :: Software Development
22
+ Requires-Python: >=3.10
23
+ Requires-Dist: ddgs>=9.0.0
24
+ Requires-Dist: httpx>=0.27.0
25
+ Requires-Dist: markrender>=1.0.7
26
+ Requires-Dist: prompt-toolkit>=3.0.0
27
+ Requires-Dist: rich>=13.0.0
28
+ Provides-Extra: dev
29
+ Requires-Dist: pytest>=7.0; extra == 'dev'
30
+ Requires-Dist: ruff>=0.4.0; extra == 'dev'
31
+ Description-Content-Type: text/markdown
32
+
33
+ # Golden Agent
34
+
35
+ **Your local AI agent. No cloud. No API keys. No excuses.**
36
+
37
+ A terminal coding agent that runs entirely on your machine — auto-installs its own
38
+ GPU-accelerated inference stack, auto-downloads models, and gets to work on basically
39
+ any post 2000 hardware.
40
+ ---
41
+
42
+ ## Why Golden Agent?
43
+
44
+ Every AI coding assistant wants your code uploaded to someone else's server.
45
+ Golden Agent flips that: the model, the tools, and your data never leave your machine.
46
+
47
+ - **100% local** — inference happens on-device via llama.cpp. Nothing is sent anywhere (except when *you* ask it to search or fetch the web).
48
+ - **Zero-config setup** — one `pip install`, one command. Golden Agent detects your GPU and downloads the matching prebuilt llama.cpp backend automatically (Vulkan on Windows/Linux, Metal on macOS, CUDA where available, CPU otherwise), then pulls the model weights from Hugging Face with resumable, retry-hardened downloads.
49
+ - **Runs on basically any hardware** — Vulkan means one code path for NVIDIA, AMD, and Intel GPUs. Full offload, flash attention, and Q5_0 KV-cache quantization squeeze maximum performance out of modest VRAM. Small models start at ~1.5 GB — a laptop iGPU can run it.
50
+ - **A real agent, not a chatbot** — reads and edits files, runs shell commands, searches the web, fetches pages, and loops on results until the job is done. Writes outside your project ask first.
51
+ - **Models for everyone** — three sizes from 1.5 GB to 8.2 GB, switchable with a numbered picker on startup.
52
+
53
+ ## Quickstart
54
+
55
+ ```bash
56
+ pip install golden-agent
57
+ golden-agent setup
58
+ golden-agent
59
+ ```
60
+
61
+ That's it. First launch:
62
+ 1. Golden Agent sets up the llama.cpp wheel for your platform.
63
+ 1. Pick a model
64
+ 3. The model downloads on first run (~1.5–9 GB, resumable if interrupted).
65
+ 4. You get a prompt: `❯` — start asking it to do things.
66
+
67
+ ```text
68
+ ❯ fix the failing test in tests/test_agent.py
69
+ ❯ search for how uv handles lockfiles and summarize
70
+ ❯ write a script that renames all photos by EXIF date
71
+ ```
72
+
73
+ ## Models
74
+
75
+ | Key | Model | Size | Context | Good for |
76
+ |---|---|---|---|---|
77
+ | `tiny` | [LFM2.5-2.6B](https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF) | ~1.5 GB | 128K | Low-end hardware, laptops, iGPUs |
78
+ | `lite` *(default)* | [Ornith-1.5-9B](https://huggingface.co/AtomicChat/Ornith-1.5-9B-GGUF) | ~5.2 GB | 153K | Best balance of speed and capability |
79
+ | `pro` | [Qwen3.8-27B](https://huggingface.co/sdkyuan/qwen3.8-27b-qat-q2_0-gguf) | ~8.2 GB | 153K | Heavy reasoning when you can afford it |
80
+
81
+ ## Built-in tools
82
+
83
+ Golden Agent ships with six tools the model uses autonomously:
84
+
85
+ | Tool | What it does |
86
+ |---|---|
87
+ | `read_file` | Reads files with automatic paging for large ones |
88
+ | `write_file` | Creates/overwrites files |
89
+ | `edit_file` | Exact-match string replacement (with `replace_all`) |
90
+ | `bash` | Runs real shell commands — actual bash even on Windows |
91
+ | `web_search` | DuckDuckGo search |
92
+ | `web_fetch` | Fetches a URL and extracts readable text |
93
+
94
+ **Guardrails:** file writes outside the directory you launched from trigger an
95
+ interactive permission prompt (`once` / `always this session` / `deny`). Tool output
96
+ is capped so runaway commands can't flood the context.
97
+
98
+ ## What you'll see
99
+
100
+ - **Live streaming markdown** as the model writes.
101
+ - **Visible reasoning** — `<think>` blocks stream inline, dimmed, so you can watch it think.
102
+ - **Tool activity panels** — file diffs and command output render with syntax highlighting; reads and searches show as quiet status lines.
103
+ - **Automatic context compaction** — past 140K tokens, older turns are summarized by the model itself so long sessions keep going.
104
+
105
+ ## Slash commands
106
+
107
+ ```
108
+ /tools list available tools
109
+ /clear wipe conversation history
110
+ /help show help
111
+ /exit quit
112
+ ```
113
+
114
+ ## Requirements
115
+
116
+ - Python 3.10+
117
+ - A GPU (NVIDIA, AMD, Intel) — or it still works on CPU, just slower.
118
+ - Internet once, for the initial setup + model download. After that: fully offline.
119
+
120
+ Golden Agent downloads a prebuilt llama.cpp binary for your detected backend, so no
121
+ compilation is required. If you want to build a custom backend yourself, the
122
+ [Vulkan SDK](https://vulkan.lunarg.com/sdk/home) is only needed for that manual path.
123
+
124
+ ## Under the hood
125
+
126
+ - **Inference:** the prebuilt `llama-server` with full GPU offload (`-ngl 99`), flash attention, and Q5_0-quantized KV cache.
127
+ - **Downloads:** direct-from-HuggingFace GGUF fetching with HTTP range resume, byte-exact validation, and exponential-backoff retries.
128
+ - **Tool calling:** hand-rolled parsers for both Qwen-style XML calls and LFM2 pythonic call syntax — robust to streaming chunk boundaries, unclosed tags, and parallel calls.
129
+ - **Model cache:** `%LOCALAPPDATA%\golden-agent\models` (Windows), `~/Library/Caches/golden-agent/models` (macOS), `$XDG_CACHE_HOME/golden-agent/models` (Linux).
130
+
131
+ ### Debugging
132
+
133
+ Set `GOLDEN_AGENT_RAW_LOG` to a file path to capture every raw model response — invaluable
134
+ when a tool call goes sideways:
135
+
136
+ ```bash
137
+ GOLDEN_AGENT_RAW_LOG=./raw.log golden-agent
138
+ ```
139
+
140
+ ## Development
141
+
142
+ ```bash
143
+ git clone <repo> && cd golden-agent
144
+ pip install -e .[dev]
145
+ pytest
146
+ ruff check src tests
147
+ ```
148
+
149
+ Run without installing:
150
+
151
+ ```bash
152
+ pip install -e .
153
+ python -m golden_agent
154
+ ```
155
+
156
+ ## License
157
+
158
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,126 @@
1
+ # Golden Agent
2
+
3
+ **Your local AI agent. No cloud. No API keys. No excuses.**
4
+
5
+ A terminal coding agent that runs entirely on your machine — auto-installs its own
6
+ GPU-accelerated inference stack, auto-downloads models, and gets to work on basically
7
+ any post 2000 hardware.
8
+ ---
9
+
10
+ ## Why Golden Agent?
11
+
12
+ Every AI coding assistant wants your code uploaded to someone else's server.
13
+ Golden Agent flips that: the model, the tools, and your data never leave your machine.
14
+
15
+ - **100% local** — inference happens on-device via llama.cpp. Nothing is sent anywhere (except when *you* ask it to search or fetch the web).
16
+ - **Zero-config setup** — one `pip install`, one command. Golden Agent detects your GPU and downloads the matching prebuilt llama.cpp backend automatically (Vulkan on Windows/Linux, Metal on macOS, CUDA where available, CPU otherwise), then pulls the model weights from Hugging Face with resumable, retry-hardened downloads.
17
+ - **Runs on basically any hardware** — Vulkan means one code path for NVIDIA, AMD, and Intel GPUs. Full offload, flash attention, and Q5_0 KV-cache quantization squeeze maximum performance out of modest VRAM. Small models start at ~1.5 GB — a laptop iGPU can run it.
18
+ - **A real agent, not a chatbot** — reads and edits files, runs shell commands, searches the web, fetches pages, and loops on results until the job is done. Writes outside your project ask first.
19
+ - **Models for everyone** — three sizes from 1.5 GB to 8.2 GB, switchable with a numbered picker on startup.
20
+
21
+ ## Quickstart
22
+
23
+ ```bash
24
+ pip install golden-agent
25
+ golden-agent setup
26
+ golden-agent
27
+ ```
28
+
29
+ That's it. First launch:
30
+ 1. Golden Agent sets up the llama.cpp wheel for your platform.
31
+ 1. Pick a model
32
+ 3. The model downloads on first run (~1.5–9 GB, resumable if interrupted).
33
+ 4. You get a prompt: `❯` — start asking it to do things.
34
+
35
+ ```text
36
+ ❯ fix the failing test in tests/test_agent.py
37
+ ❯ search for how uv handles lockfiles and summarize
38
+ ❯ write a script that renames all photos by EXIF date
39
+ ```
40
+
41
+ ## Models
42
+
43
+ | Key | Model | Size | Context | Good for |
44
+ |---|---|---|---|---|
45
+ | `tiny` | [LFM2.5-2.6B](https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF) | ~1.5 GB | 128K | Low-end hardware, laptops, iGPUs |
46
+ | `lite` *(default)* | [Ornith-1.5-9B](https://huggingface.co/AtomicChat/Ornith-1.5-9B-GGUF) | ~5.2 GB | 153K | Best balance of speed and capability |
47
+ | `pro` | [Qwen3.8-27B](https://huggingface.co/sdkyuan/qwen3.8-27b-qat-q2_0-gguf) | ~8.2 GB | 153K | Heavy reasoning when you can afford it |
48
+
49
+ ## Built-in tools
50
+
51
+ Golden Agent ships with six tools the model uses autonomously:
52
+
53
+ | Tool | What it does |
54
+ |---|---|
55
+ | `read_file` | Reads files with automatic paging for large ones |
56
+ | `write_file` | Creates/overwrites files |
57
+ | `edit_file` | Exact-match string replacement (with `replace_all`) |
58
+ | `bash` | Runs real shell commands — actual bash even on Windows |
59
+ | `web_search` | DuckDuckGo search |
60
+ | `web_fetch` | Fetches a URL and extracts readable text |
61
+
62
+ **Guardrails:** file writes outside the directory you launched from trigger an
63
+ interactive permission prompt (`once` / `always this session` / `deny`). Tool output
64
+ is capped so runaway commands can't flood the context.
65
+
66
+ ## What you'll see
67
+
68
+ - **Live streaming markdown** as the model writes.
69
+ - **Visible reasoning** — `<think>` blocks stream inline, dimmed, so you can watch it think.
70
+ - **Tool activity panels** — file diffs and command output render with syntax highlighting; reads and searches show as quiet status lines.
71
+ - **Automatic context compaction** — past 140K tokens, older turns are summarized by the model itself so long sessions keep going.
72
+
73
+ ## Slash commands
74
+
75
+ ```
76
+ /tools list available tools
77
+ /clear wipe conversation history
78
+ /help show help
79
+ /exit quit
80
+ ```
81
+
82
+ ## Requirements
83
+
84
+ - Python 3.10+
85
+ - A GPU (NVIDIA, AMD, Intel) — or it still works on CPU, just slower.
86
+ - Internet once, for the initial setup + model download. After that: fully offline.
87
+
88
+ Golden Agent downloads a prebuilt llama.cpp binary for your detected backend, so no
89
+ compilation is required. If you want to build a custom backend yourself, the
90
+ [Vulkan SDK](https://vulkan.lunarg.com/sdk/home) is only needed for that manual path.
91
+
92
+ ## Under the hood
93
+
94
+ - **Inference:** the prebuilt `llama-server` with full GPU offload (`-ngl 99`), flash attention, and Q5_0-quantized KV cache.
95
+ - **Downloads:** direct-from-HuggingFace GGUF fetching with HTTP range resume, byte-exact validation, and exponential-backoff retries.
96
+ - **Tool calling:** hand-rolled parsers for both Qwen-style XML calls and LFM2 pythonic call syntax — robust to streaming chunk boundaries, unclosed tags, and parallel calls.
97
+ - **Model cache:** `%LOCALAPPDATA%\golden-agent\models` (Windows), `~/Library/Caches/golden-agent/models` (macOS), `$XDG_CACHE_HOME/golden-agent/models` (Linux).
98
+
99
+ ### Debugging
100
+
101
+ Set `GOLDEN_AGENT_RAW_LOG` to a file path to capture every raw model response — invaluable
102
+ when a tool call goes sideways:
103
+
104
+ ```bash
105
+ GOLDEN_AGENT_RAW_LOG=./raw.log golden-agent
106
+ ```
107
+
108
+ ## Development
109
+
110
+ ```bash
111
+ git clone <repo> && cd golden-agent
112
+ pip install -e .[dev]
113
+ pytest
114
+ ruff check src tests
115
+ ```
116
+
117
+ Run without installing:
118
+
119
+ ```bash
120
+ pip install -e .
121
+ python -m golden_agent
122
+ ```
123
+
124
+ ## License
125
+
126
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,74 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "golden-agent"
7
+ version = "0.1.0"
8
+ description = "Your local AI agent. Runs on basically any hardware — auto-setup, GPU-accelerated, zero cloud, zero API keys."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ requires-python = ">=3.10"
12
+ authors = [{ name = "Yashneil Gajjala", email = "yashneil75@gmail.com" }]
13
+ keywords = [
14
+ "agent",
15
+ "cli",
16
+ "coding-agent",
17
+ "gguf",
18
+ "llama-cpp",
19
+ "llm",
20
+ "local",
21
+ "local-llm",
22
+ "offline",
23
+ "tools",
24
+ "vulkan",
25
+ ]
26
+ classifiers = [
27
+ "Development Status :: 3 - Alpha",
28
+ "Environment :: Console",
29
+ "Intended Audience :: Developers",
30
+ "License :: OSI Approved :: MIT License",
31
+ "Operating System :: Microsoft :: Windows",
32
+ "Operating System :: POSIX :: Linux",
33
+ "Operating System :: MacOS",
34
+ "Programming Language :: Python :: 3",
35
+ "Programming Language :: Python :: 3.10",
36
+ "Programming Language :: Python :: 3.11",
37
+ "Programming Language :: Python :: 3.12",
38
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
39
+ "Topic :: Software Development",
40
+ ]
41
+ dependencies = [
42
+ "httpx>=0.27.0",
43
+ "rich>=13.0.0",
44
+ "markrender>=1.0.7",
45
+ "ddgs>=9.0.0",
46
+ "prompt_toolkit>=3.0.0",
47
+ ]
48
+
49
+ [project.optional-dependencies]
50
+ dev = [
51
+ "pytest>=7.0",
52
+ "ruff>=0.4.0",
53
+ ]
54
+
55
+ [project.scripts]
56
+ golden-agent = "golden_agent.cli:repl"
57
+
58
+ [tool.hatch.build.targets.wheel]
59
+ packages = ["src/golden_agent"]
60
+
61
+ [tool.ruff]
62
+ target-version = "py310"
63
+ line-length = 100
64
+
65
+ [tool.ruff.lint]
66
+ select = ["E", "F", "I", "N", "UP", "B"]
67
+
68
+ [tool.ruff.lint.per-file-ignores]
69
+ "src/golden_agent/config.py" = ["E501"]
70
+ "src/golden_agent/cli.py" = ["E501"]
71
+ "src/golden_agent/system_prompt.py" = ["E501"]
72
+
73
+ [tool.pytest.ini_options]
74
+ testpaths = ["tests"]
@@ -0,0 +1,3 @@
1
+ from golden_agent.cli import repl
2
+
3
+ repl()