contextzip 0.3.2__tar.gz → 0.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {contextzip-0.3.2 → contextzip-0.3.4}/PKG-INFO +76 -3
- {contextzip-0.3.2 → contextzip-0.3.4}/README.md +75 -2
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/__init__.py +1 -1
- contextzip-0.3.4/contextzip/ai/__init__.py +1 -0
- contextzip-0.3.4/contextzip/ai/gemini.py +282 -0
- contextzip-0.3.4/contextzip/ai/heuristic.py +333 -0
- contextzip-0.3.4/contextzip/ai/selector.py +158 -0
- contextzip-0.3.4/contextzip/brief.py +280 -0
- contextzip-0.3.4/contextzip/claude_artifacts.py +150 -0
- contextzip-0.3.4/contextzip/claude_export.py +73 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/cli.py +238 -32
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/cli_display.py +55 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/cli_onboard.py +59 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/clipboard.py +90 -14
- contextzip-0.3.4/contextzip/code_changes.py +257 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/config.py +46 -1
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/git.py +142 -11
- contextzip-0.3.4/contextzip/markers.py +61 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/packager.py +9 -11
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/base.py +1 -1
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/PKG-INFO +76 -3
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/SOURCES.txt +9 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/top_level.txt +1 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/pyproject.toml +1 -2
- {contextzip-0.3.2 → contextzip-0.3.4}/LICENSE +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/api.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/cli_ai.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/detector.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/error_parser.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/filters.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/__init__.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/errors/__init__.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/errors/node.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/errors/python.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/go.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/node.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/python.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/ruby.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/rust.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/watcher.py +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/dependency_links.txt +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/entry_points.txt +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/requires.txt +0 -0
- {contextzip-0.3.2 → contextzip-0.3.4}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: contextzip
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: Intelligently package your codebase for AI tools
|
|
5
5
|
Author-email: Deepesh <akadeepesh@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -52,6 +52,7 @@ contextzip eliminates that entirely. Run it from your project root — it detect
|
|
|
52
52
|
- **Git-aware packaging** — use `--git-changes` to package only modified, staged, and untracked files; perfect for incremental debugging and PR review sessions
|
|
53
53
|
- **AI-powered file selection** — describe your task in plain English with `--prompt` and Gemini selects the minimum relevant files automatically, no manual hunting required
|
|
54
54
|
- **Terminal error watcher** — wrap any dev server with `contextzip watch` to auto-detect errors and package a ready-to-upload debug context in one keypress
|
|
55
|
+
- **End-of-day / handoff prompts** — `contextzip eod` and `contextzip handoff` turn a Claude/ChatGPT conversation export plus today's code changes into a paste-ready prompt, copied straight to your clipboard
|
|
55
56
|
- **Persistent workspace** — all generated ZIPs land in `.contextzip/` at your project root, discoverable, reusable, and git-ignored automatically
|
|
56
57
|
- **Warns before it's a problem** — flags large (≥ 1 MB) and binary files that AI tools can't read, before you waste an upload
|
|
57
58
|
- **Handles edge cases** — dangling symlinks, unreadable files, and paths outside the project tree are caught and reported, never silently dropped
|
|
@@ -117,7 +118,7 @@ contextzip [OPTIONS]
|
|
|
117
118
|
| `--no-clipboard` | Skip the clipboard / folder-open step |
|
|
118
119
|
| `--no-gitignore` | Ignore the project's `.gitignore` |
|
|
119
120
|
|
|
120
|
-
**Subcommands:** `exclude`, `include`, `watch`, `config` — run `contextzip --help` for full details.
|
|
121
|
+
**Subcommands:** `exclude`, `include`, `watch`, `config`, `eod`, `handoff` — run `contextzip --help` for full details.
|
|
121
122
|
|
|
122
123
|
---
|
|
123
124
|
|
|
@@ -148,6 +149,40 @@ contextzip --output ~/Desktop/project-context.zip
|
|
|
148
149
|
|
|
149
150
|
---
|
|
150
151
|
|
|
152
|
+
## Python API
|
|
153
|
+
|
|
154
|
+
contextzip is also usable as a library. All CLI capabilities are available as plain Python functions — no Click, no Rich output, no `SystemExit`.
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
from contextzip import get_git_changes, get_files, create_zip
|
|
158
|
+
|
|
159
|
+
# Get changed files and use them directly
|
|
160
|
+
collection = get_git_changes()
|
|
161
|
+
for f in collection.files: # plain pathlib.Path objects
|
|
162
|
+
upload(f) # no zip required
|
|
163
|
+
|
|
164
|
+
# Or zip them and upload the archive
|
|
165
|
+
pkg = create_zip(collection, output="/tmp/changes.zip")
|
|
166
|
+
with open(pkg.zip_path, "rb") as f:
|
|
167
|
+
upload_to_s3(f)
|
|
168
|
+
|
|
169
|
+
# Full project scan with filters
|
|
170
|
+
collection = get_files(include=["src/"], exclude=["tests/"])
|
|
171
|
+
pkg = create_zip(collection, output="/tmp/upload.zip")
|
|
172
|
+
print(f"{pkg.file_count} files, {pkg.compressed_bytes} bytes")
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
| Function | Description |
|
|
176
|
+
|---|---|
|
|
177
|
+
| `get_git_changes(path?)` | Modified, added, and untracked files from git |
|
|
178
|
+
| `get_files(path?, include?, exclude?)` | All project files after exclusion rules |
|
|
179
|
+
| `create_zip(collection, output?)` | Write a `FileCollection` to a ZIP archive |
|
|
180
|
+
| `detect_ecosystem(path?)` | Detect framework and confidence level |
|
|
181
|
+
|
|
182
|
+
All functions default `path` to `Path.cwd()`. Errors raise typed exceptions (`NotARepositoryError`, `GitNotFoundError`, `NoFilesError`, etc.) rather than exiting.
|
|
183
|
+
|
|
184
|
+
---
|
|
185
|
+
|
|
151
186
|
## AI-powered file selection
|
|
152
187
|
|
|
153
188
|
The `--prompt` flag lets you describe a task in plain English. contextzip scans your project, builds a lightweight file map, and asks Gemini to return the minimum set of files needed for that task — typically 2–5, never more than 10. The result is a tightly scoped ZIP with only what you'd actually open to make the change.
|
|
@@ -209,11 +244,49 @@ Press **D** and contextzip immediately writes `.contextzip/debug-context.zip`. Y
|
|
|
209
244
|
|
|
210
245
|
---
|
|
211
246
|
|
|
247
|
+
## End-of-day reports and chat handoffs
|
|
248
|
+
|
|
249
|
+
If you work through a problem in a Claude or ChatGPT conversation and need to either (a) summarize what you did for an end-of-day report, or (b) continue the same work in a fresh chat after hitting a usage limit, `eod` and `handoff` build the prompt for you — contextzip does no summarizing itself; that's left to whichever AI tool you paste the result into.
|
|
250
|
+
|
|
251
|
+
```bash
|
|
252
|
+
contextzip eod
|
|
253
|
+
contextzip handoff
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
**Setup:** export your conversation (any markdown export works) and drop the `.md` file into `exports/` at your project root. Both commands pick the most recently modified file there automatically.
|
|
257
|
+
|
|
258
|
+
**What gets built:**
|
|
259
|
+
|
|
260
|
+
- The conversation itself — pasted directly into the prompt if it's short, or referenced as an attachment if it's long enough that inlining it would burn through the next chat's context budget
|
|
261
|
+
- Whatever code changed, resolved per file in priority order:
|
|
262
|
+
1. **Not pushed** — diffed against your upstream branch, plus the complete current file
|
|
263
|
+
2. **Diverged from Claude** — diffed against the version Claude last produced (if you've set up a session key — see below), plus the complete current codebase file
|
|
264
|
+
3. **Since last run** — diffed against a per-branch checkpoint that `eod`/`handoff` remember automatically, plus the complete current file
|
|
265
|
+
- New, untracked files are included as full content (there's nothing to diff them against)
|
|
266
|
+
- `eod` ends the prompt with a flat instruction to produce a work-log table; `handoff` frames it as continuing the project in a new chat
|
|
267
|
+
|
|
268
|
+
The result is copied straight to your clipboard, and also saved to `.contextzip/` if you want to look it over first.
|
|
269
|
+
|
|
270
|
+
**Optional — diffing against Claude's own version (case 2):** if you give contextzip your Claude.ai session key, `eod`/`handoff` will fetch the files Claude actually produced and compare them against your codebase, catching drift even after everything's pushed and in sync.
|
|
271
|
+
|
|
272
|
+
```bash
|
|
273
|
+
contextzip config --set-session-key
|
|
274
|
+
```
|
|
275
|
+
|
|
276
|
+
This is best-effort by design — it depends on an undocumented Claude.ai endpoint, so a missing key, an expired cookie, or the endpoint changing shape just skips this check with a warning rather than failing the whole command.
|
|
277
|
+
|
|
278
|
+
```bash
|
|
279
|
+
contextzip eod --dry-run # preview without advancing the checkpoint
|
|
280
|
+
contextzip handoff --no-fetch # skip the Claude-artifact fetch for this run
|
|
281
|
+
```
|
|
282
|
+
|
|
283
|
+
---
|
|
284
|
+
|
|
212
285
|
## What gets excluded
|
|
213
286
|
|
|
214
287
|
contextzip stacks exclusion rules based on your detected stack, on top of your `.gitignore`.
|
|
215
288
|
|
|
216
|
-
**Always excluded:** `.git/`, `.env` files, logs, caches, editor config (`.vscode/`, `.idea/`), OS files (`.DS_Store`, `Thumbs.db`),
|
|
289
|
+
**Always excluded:** `.git/`, `.env` files, logs, caches, editor config (`.vscode/`, `.idea/`), OS files (`.DS_Store`, `Thumbs.db`), common binary formats, and contextzip's own `.contextzip/` and `exports/` working folders.
|
|
217
290
|
|
|
218
291
|
**By framework:**
|
|
219
292
|
|
|
@@ -23,6 +23,7 @@ contextzip eliminates that entirely. Run it from your project root — it detect
|
|
|
23
23
|
- **Git-aware packaging** — use `--git-changes` to package only modified, staged, and untracked files; perfect for incremental debugging and PR review sessions
|
|
24
24
|
- **AI-powered file selection** — describe your task in plain English with `--prompt` and Gemini selects the minimum relevant files automatically, no manual hunting required
|
|
25
25
|
- **Terminal error watcher** — wrap any dev server with `contextzip watch` to auto-detect errors and package a ready-to-upload debug context in one keypress
|
|
26
|
+
- **End-of-day / handoff prompts** — `contextzip eod` and `contextzip handoff` turn a Claude/ChatGPT conversation export plus today's code changes into a paste-ready prompt, copied straight to your clipboard
|
|
26
27
|
- **Persistent workspace** — all generated ZIPs land in `.contextzip/` at your project root, discoverable, reusable, and git-ignored automatically
|
|
27
28
|
- **Warns before it's a problem** — flags large (≥ 1 MB) and binary files that AI tools can't read, before you waste an upload
|
|
28
29
|
- **Handles edge cases** — dangling symlinks, unreadable files, and paths outside the project tree are caught and reported, never silently dropped
|
|
@@ -88,7 +89,7 @@ contextzip [OPTIONS]
|
|
|
88
89
|
| `--no-clipboard` | Skip the clipboard / folder-open step |
|
|
89
90
|
| `--no-gitignore` | Ignore the project's `.gitignore` |
|
|
90
91
|
|
|
91
|
-
**Subcommands:** `exclude`, `include`, `watch`, `config` — run `contextzip --help` for full details.
|
|
92
|
+
**Subcommands:** `exclude`, `include`, `watch`, `config`, `eod`, `handoff` — run `contextzip --help` for full details.
|
|
92
93
|
|
|
93
94
|
---
|
|
94
95
|
|
|
@@ -119,6 +120,40 @@ contextzip --output ~/Desktop/project-context.zip
|
|
|
119
120
|
|
|
120
121
|
---
|
|
121
122
|
|
|
123
|
+
## Python API
|
|
124
|
+
|
|
125
|
+
contextzip is also usable as a library. All CLI capabilities are available as plain Python functions — no Click, no Rich output, no `SystemExit`.
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
from contextzip import get_git_changes, get_files, create_zip
|
|
129
|
+
|
|
130
|
+
# Get changed files and use them directly
|
|
131
|
+
collection = get_git_changes()
|
|
132
|
+
for f in collection.files: # plain pathlib.Path objects
|
|
133
|
+
upload(f) # no zip required
|
|
134
|
+
|
|
135
|
+
# Or zip them and upload the archive
|
|
136
|
+
pkg = create_zip(collection, output="/tmp/changes.zip")
|
|
137
|
+
with open(pkg.zip_path, "rb") as f:
|
|
138
|
+
upload_to_s3(f)
|
|
139
|
+
|
|
140
|
+
# Full project scan with filters
|
|
141
|
+
collection = get_files(include=["src/"], exclude=["tests/"])
|
|
142
|
+
pkg = create_zip(collection, output="/tmp/upload.zip")
|
|
143
|
+
print(f"{pkg.file_count} files, {pkg.compressed_bytes} bytes")
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
| Function | Description |
|
|
147
|
+
|---|---|
|
|
148
|
+
| `get_git_changes(path?)` | Modified, added, and untracked files from git |
|
|
149
|
+
| `get_files(path?, include?, exclude?)` | All project files after exclusion rules |
|
|
150
|
+
| `create_zip(collection, output?)` | Write a `FileCollection` to a ZIP archive |
|
|
151
|
+
| `detect_ecosystem(path?)` | Detect framework and confidence level |
|
|
152
|
+
|
|
153
|
+
All functions default `path` to `Path.cwd()`. Errors raise typed exceptions (`NotARepositoryError`, `GitNotFoundError`, `NoFilesError`, etc.) rather than exiting.
|
|
154
|
+
|
|
155
|
+
---
|
|
156
|
+
|
|
122
157
|
## AI-powered file selection
|
|
123
158
|
|
|
124
159
|
The `--prompt` flag lets you describe a task in plain English. contextzip scans your project, builds a lightweight file map, and asks Gemini to return the minimum set of files needed for that task — typically 2–5, never more than 10. The result is a tightly scoped ZIP with only what you'd actually open to make the change.
|
|
@@ -180,11 +215,49 @@ Press **D** and contextzip immediately writes `.contextzip/debug-context.zip`. Y
|
|
|
180
215
|
|
|
181
216
|
---
|
|
182
217
|
|
|
218
|
+
## End-of-day reports and chat handoffs
|
|
219
|
+
|
|
220
|
+
If you work through a problem in a Claude or ChatGPT conversation and need to either (a) summarize what you did for an end-of-day report, or (b) continue the same work in a fresh chat after hitting a usage limit, `eod` and `handoff` build the prompt for you — contextzip does no summarizing itself; that's left to whichever AI tool you paste the result into.
|
|
221
|
+
|
|
222
|
+
```bash
|
|
223
|
+
contextzip eod
|
|
224
|
+
contextzip handoff
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
**Setup:** export your conversation (any markdown export works) and drop the `.md` file into `exports/` at your project root. Both commands pick the most recently modified file there automatically.
|
|
228
|
+
|
|
229
|
+
**What gets built:**
|
|
230
|
+
|
|
231
|
+
- The conversation itself — pasted directly into the prompt if it's short, or referenced as an attachment if it's long enough that inlining it would burn through the next chat's context budget
|
|
232
|
+
- Whatever code changed, resolved per file in priority order:
|
|
233
|
+
1. **Not pushed** — diffed against your upstream branch, plus the complete current file
|
|
234
|
+
2. **Diverged from Claude** — diffed against the version Claude last produced (if you've set up a session key — see below), plus the complete current codebase file
|
|
235
|
+
3. **Since last run** — diffed against a per-branch checkpoint that `eod`/`handoff` remember automatically, plus the complete current file
|
|
236
|
+
- New, untracked files are included as full content (there's nothing to diff them against)
|
|
237
|
+
- `eod` ends the prompt with a flat instruction to produce a work-log table; `handoff` frames it as continuing the project in a new chat
|
|
238
|
+
|
|
239
|
+
The result is copied straight to your clipboard, and also saved to `.contextzip/` if you want to look it over first.
|
|
240
|
+
|
|
241
|
+
**Optional — diffing against Claude's own version (case 2):** if you give contextzip your Claude.ai session key, `eod`/`handoff` will fetch the files Claude actually produced and compare them against your codebase, catching drift even after everything's pushed and in sync.
|
|
242
|
+
|
|
243
|
+
```bash
|
|
244
|
+
contextzip config --set-session-key
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
This is best-effort by design — it depends on an undocumented Claude.ai endpoint, so a missing key, an expired cookie, or the endpoint changing shape just skips this check with a warning rather than failing the whole command.
|
|
248
|
+
|
|
249
|
+
```bash
|
|
250
|
+
contextzip eod --dry-run # preview without advancing the checkpoint
|
|
251
|
+
contextzip handoff --no-fetch # skip the Claude-artifact fetch for this run
|
|
252
|
+
```
|
|
253
|
+
|
|
254
|
+
---
|
|
255
|
+
|
|
183
256
|
## What gets excluded
|
|
184
257
|
|
|
185
258
|
contextzip stacks exclusion rules based on your detected stack, on top of your `.gitignore`.
|
|
186
259
|
|
|
187
|
-
**Always excluded:** `.git/`, `.env` files, logs, caches, editor config (`.vscode/`, `.idea/`), OS files (`.DS_Store`, `Thumbs.db`),
|
|
260
|
+
**Always excluded:** `.git/`, `.env` files, logs, caches, editor config (`.vscode/`, `.idea/`), OS files (`.DS_Store`, `Thumbs.db`), common binary formats, and contextzip's own `.contextzip/` and `exports/` working folders.
|
|
188
261
|
|
|
189
262
|
**By framework:**
|
|
190
263
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""contextzip.ai — AI-powered file selection for --prompt mode."""
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ai/gemini.py — Thin Gemini API client for contextzip's prompt-aware mode.
|
|
3
|
+
|
|
4
|
+
Makes a single POST to the Gemini generateContent endpoint and returns
|
|
5
|
+
a ranked list of file paths relevant to the user's task description.
|
|
6
|
+
|
|
7
|
+
Design principles:
|
|
8
|
+
- Raw httpx calls only — no Google SDK dependency
|
|
9
|
+
- Strict JSON output from the model — no markdown, no prose
|
|
10
|
+
- Validates every returned path against the real file tree
|
|
11
|
+
- Single responsibility: call API, parse response, validate paths
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
|
|
18
|
+
try:
|
|
19
|
+
import httpx
|
|
20
|
+
|
|
21
|
+
_HTTPX_AVAILABLE = True
|
|
22
|
+
except ImportError:
|
|
23
|
+
_HTTPX_AVAILABLE = False
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# ---------------------------------------------------------------------------
|
|
27
|
+
# Constants
|
|
28
|
+
# ---------------------------------------------------------------------------
|
|
29
|
+
|
|
30
|
+
_API_URL = (
|
|
31
|
+
"https://generativelanguage.googleapis.com/v1beta/models"
|
|
32
|
+
"/{model}:generateContent?key={key}"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
DEFAULT_MODEL = "gemini-2.5-flash-lite"
|
|
36
|
+
|
|
37
|
+
# Hard cap: never return more than this many files regardless of model output.
|
|
38
|
+
# Minimum context is the goal — the prompt enforces this, but we double-guard.
|
|
39
|
+
_MAX_FILES = 12
|
|
40
|
+
|
|
41
|
+
_TIMEOUT_SECONDS = 30
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
# ---------------------------------------------------------------------------
|
|
45
|
+
# Exceptions
|
|
46
|
+
# ---------------------------------------------------------------------------
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class GeminiError(Exception):
|
|
50
|
+
"""Raised when the Gemini API call fails for any reason."""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class GeminiUnavailable(GeminiError):
|
|
54
|
+
"""Raised when httpx is not installed."""
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class GeminiRateLimitError(GeminiError):
|
|
58
|
+
"""Raised specifically on HTTP 429 — allows callers to trigger fallback."""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
# ---------------------------------------------------------------------------
|
|
62
|
+
# Public API
|
|
63
|
+
# ---------------------------------------------------------------------------
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def select_files(
|
|
67
|
+
*,
|
|
68
|
+
api_key: str,
|
|
69
|
+
prompt: str,
|
|
70
|
+
file_tree: list[tuple[str, int]], # [(rel_path, size_bytes), ...]
|
|
71
|
+
ecosystem: str,
|
|
72
|
+
model: str = DEFAULT_MODEL,
|
|
73
|
+
) -> list[str]:
|
|
74
|
+
"""
|
|
75
|
+
Ask Gemini which files are relevant to *prompt* and return their paths.
|
|
76
|
+
|
|
77
|
+
Parameters
|
|
78
|
+
----------
|
|
79
|
+
api_key:
|
|
80
|
+
Gemini API key from Google AI Studio.
|
|
81
|
+
prompt:
|
|
82
|
+
The user's natural-language task description.
|
|
83
|
+
file_tree:
|
|
84
|
+
All candidate files as (relative_posix_path, size_bytes) tuples.
|
|
85
|
+
These should already have contextzip's standard exclusions applied —
|
|
86
|
+
no node_modules, no build artifacts, no .env files.
|
|
87
|
+
ecosystem:
|
|
88
|
+
Human-readable detected framework string, e.g. "Next.js + TypeScript".
|
|
89
|
+
Gives the model important context for relevance scoring.
|
|
90
|
+
model:
|
|
91
|
+
Gemini model identifier. Defaults to gemini-2.5-flash-lite.
|
|
92
|
+
|
|
93
|
+
Returns
|
|
94
|
+
-------
|
|
95
|
+
list[str]
|
|
96
|
+
Relative POSIX paths of the selected files, ordered by relevance
|
|
97
|
+
(most relevant first). Always a strict subset of the input file_tree
|
|
98
|
+
paths — hallucinated or non-existent paths are silently dropped.
|
|
99
|
+
|
|
100
|
+
Raises
|
|
101
|
+
------
|
|
102
|
+
GeminiUnavailable
|
|
103
|
+
If httpx is not installed.
|
|
104
|
+
GeminiError
|
|
105
|
+
On any API or parsing failure.
|
|
106
|
+
"""
|
|
107
|
+
if not _HTTPX_AVAILABLE:
|
|
108
|
+
raise GeminiUnavailable(
|
|
109
|
+
"httpx is required for AI-powered selection. "
|
|
110
|
+
"Install it with: pip install httpx"
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
system_prompt = _build_system_prompt()
|
|
114
|
+
user_message = _build_user_message(prompt, file_tree, ecosystem)
|
|
115
|
+
|
|
116
|
+
url = _API_URL.format(model=model, key=api_key)
|
|
117
|
+
|
|
118
|
+
payload = {
|
|
119
|
+
"contents": [
|
|
120
|
+
{
|
|
121
|
+
"role": "user",
|
|
122
|
+
"parts": [{"text": f"{system_prompt}\n\n{user_message}"}],
|
|
123
|
+
}
|
|
124
|
+
],
|
|
125
|
+
"generationConfig": {
|
|
126
|
+
"temperature": 0.0, # deterministic — this is a ranking task
|
|
127
|
+
"maxOutputTokens": 512, # a list of paths needs very few tokens
|
|
128
|
+
"responseMimeType": "application/json",
|
|
129
|
+
},
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
try:
|
|
133
|
+
response = httpx.post(url, json=payload, timeout=_TIMEOUT_SECONDS)
|
|
134
|
+
except httpx.TimeoutException:
|
|
135
|
+
raise GeminiError("Request timed out after 30 seconds.")
|
|
136
|
+
except httpx.RequestError as exc:
|
|
137
|
+
raise GeminiError(f"Network error: {exc}")
|
|
138
|
+
|
|
139
|
+
if response.status_code == 400:
|
|
140
|
+
raise GeminiError("Invalid request — check your API key format.")
|
|
141
|
+
if response.status_code == 401 or response.status_code == 403:
|
|
142
|
+
raise GeminiError(
|
|
143
|
+
"API key rejected. Run [cyan]contextzip config --reset-key[/] to update it."
|
|
144
|
+
)
|
|
145
|
+
if response.status_code == 429:
|
|
146
|
+
raise GeminiRateLimitError(
|
|
147
|
+
"Rate limit reached on the free tier (15 req/min). "
|
|
148
|
+
"Wait a moment and try again, or your new key may still be activating — "
|
|
149
|
+
"Google can take up to 60 seconds after key creation."
|
|
150
|
+
)
|
|
151
|
+
if response.status_code != 200:
|
|
152
|
+
raise GeminiError(
|
|
153
|
+
f"Gemini API returned HTTP {response.status_code}: {response.text[:200]}"
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
return _parse_response(response.json(), file_tree)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# ---------------------------------------------------------------------------
|
|
160
|
+
# Prompt construction
|
|
161
|
+
# ---------------------------------------------------------------------------
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _build_system_prompt() -> str:
|
|
165
|
+
return """\
|
|
166
|
+
You are a precise file relevance assistant for software projects.
|
|
167
|
+
|
|
168
|
+
Your only job: given a developer's task description and a project file tree,
|
|
169
|
+
return the MINIMUM set of files a developer would need to open to complete
|
|
170
|
+
that task. Think like a senior engineer doing a surgical code change —
|
|
171
|
+
open only what you must touch or read to understand the change.
|
|
172
|
+
|
|
173
|
+
Rules you must follow:
|
|
174
|
+
- Return ONLY a JSON array of file path strings. No explanation, no markdown,
|
|
175
|
+
no extra keys. Example: ["src/auth.ts", "components/Toast.tsx"]
|
|
176
|
+
- Be ruthless about exclusion. If a file is not directly relevant to the
|
|
177
|
+
stated task, leave it out. Err heavily on the side of fewer files.
|
|
178
|
+
- Prefer files that will be MODIFIED over files that are merely referenced.
|
|
179
|
+
- Config files, test files, and documentation should only appear if the
|
|
180
|
+
task explicitly concerns them.
|
|
181
|
+
- Never return more than 10 files. For most tasks 2–5 files is correct.
|
|
182
|
+
- Order by relevance: most directly relevant file first.\
|
|
183
|
+
"""
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _build_user_message(
|
|
187
|
+
prompt: str,
|
|
188
|
+
file_tree: list[tuple[str, int]],
|
|
189
|
+
ecosystem: str,
|
|
190
|
+
) -> str:
|
|
191
|
+
tree_lines = "\n".join(f"{path} ({_human_size(size)})" for path, size in file_tree)
|
|
192
|
+
return f"""\
|
|
193
|
+
Framework: {ecosystem}
|
|
194
|
+
|
|
195
|
+
Task: {prompt}
|
|
196
|
+
|
|
197
|
+
Project files (exclusions already applied):
|
|
198
|
+
{tree_lines}
|
|
199
|
+
|
|
200
|
+
Return only a JSON array of the most relevant file paths for this task.\
|
|
201
|
+
"""
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
# ---------------------------------------------------------------------------
|
|
205
|
+
# Response parsing and validation
|
|
206
|
+
# ---------------------------------------------------------------------------
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _parse_response(
|
|
210
|
+
data: dict,
|
|
211
|
+
file_tree: list[tuple[str, int]],
|
|
212
|
+
) -> list[str]:
|
|
213
|
+
"""
|
|
214
|
+
Extract and validate the file list from the Gemini API response.
|
|
215
|
+
|
|
216
|
+
- Parses the JSON array from the model's text output
|
|
217
|
+
- Drops any path the model hallucinated (not in the real file tree)
|
|
218
|
+
- Enforces the _MAX_FILES hard cap
|
|
219
|
+
- Warns (via exception) if the model returned mostly invalid paths
|
|
220
|
+
"""
|
|
221
|
+
# Navigate the Gemini response structure
|
|
222
|
+
try:
|
|
223
|
+
text = data["candidates"][0]["content"]["parts"][0]["text"]
|
|
224
|
+
except (KeyError, IndexError) as exc:
|
|
225
|
+
raise GeminiError(f"Unexpected API response structure: {exc}\n{data}")
|
|
226
|
+
|
|
227
|
+
# Strip any accidental markdown fences the model might add
|
|
228
|
+
text = text.strip().strip("`").strip()
|
|
229
|
+
if text.startswith("json"):
|
|
230
|
+
text = text[4:].strip()
|
|
231
|
+
|
|
232
|
+
try:
|
|
233
|
+
raw_paths: list = json.loads(text)
|
|
234
|
+
except json.JSONDecodeError as exc:
|
|
235
|
+
raise GeminiError(
|
|
236
|
+
f"Model returned non-JSON output: {exc}\nRaw output: {text[:300]}"
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
if not isinstance(raw_paths, list):
|
|
240
|
+
raise GeminiError(
|
|
241
|
+
f"Expected a JSON array, got {type(raw_paths).__name__}: {text[:200]}"
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
# Build a set of valid paths for O(1) lookup
|
|
245
|
+
valid_paths: set[str] = {path for path, _ in file_tree}
|
|
246
|
+
|
|
247
|
+
validated: list[str] = []
|
|
248
|
+
hallucinated = 0
|
|
249
|
+
|
|
250
|
+
for item in raw_paths:
|
|
251
|
+
if not isinstance(item, str):
|
|
252
|
+
continue
|
|
253
|
+
# Normalise separators (model may return backslashes on Windows prompts)
|
|
254
|
+
normalised = item.replace("\\", "/").strip()
|
|
255
|
+
if normalised in valid_paths:
|
|
256
|
+
validated.append(normalised)
|
|
257
|
+
else:
|
|
258
|
+
hallucinated += 1
|
|
259
|
+
|
|
260
|
+
# Warn if the model was mostly making things up
|
|
261
|
+
total_returned = len(raw_paths)
|
|
262
|
+
if total_returned > 0 and hallucinated / total_returned > 0.5:
|
|
263
|
+
raise GeminiError(
|
|
264
|
+
f"Model returned {hallucinated}/{total_returned} non-existent paths. "
|
|
265
|
+
"This may indicate a model or prompt issue. Try a more specific prompt."
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
# Enforce hard cap
|
|
269
|
+
return validated[:_MAX_FILES]
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
# ---------------------------------------------------------------------------
|
|
273
|
+
# Helpers
|
|
274
|
+
# ---------------------------------------------------------------------------
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _human_size(n: int) -> str:
|
|
278
|
+
for unit in ("B", "KB", "MB"):
|
|
279
|
+
if n < 1024:
|
|
280
|
+
return f"{n:.0f} {unit}"
|
|
281
|
+
n /= 1024
|
|
282
|
+
return f"{n:.1f} GB"
|