contextzip 0.3.2__tar.gz → 0.3.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. {contextzip-0.3.2 → contextzip-0.3.4}/PKG-INFO +76 -3
  2. {contextzip-0.3.2 → contextzip-0.3.4}/README.md +75 -2
  3. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/__init__.py +1 -1
  4. contextzip-0.3.4/contextzip/ai/__init__.py +1 -0
  5. contextzip-0.3.4/contextzip/ai/gemini.py +282 -0
  6. contextzip-0.3.4/contextzip/ai/heuristic.py +333 -0
  7. contextzip-0.3.4/contextzip/ai/selector.py +158 -0
  8. contextzip-0.3.4/contextzip/brief.py +280 -0
  9. contextzip-0.3.4/contextzip/claude_artifacts.py +150 -0
  10. contextzip-0.3.4/contextzip/claude_export.py +73 -0
  11. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/cli.py +238 -32
  12. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/cli_display.py +55 -0
  13. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/cli_onboard.py +59 -0
  14. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/clipboard.py +90 -14
  15. contextzip-0.3.4/contextzip/code_changes.py +257 -0
  16. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/config.py +46 -1
  17. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/git.py +142 -11
  18. contextzip-0.3.4/contextzip/markers.py +61 -0
  19. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/packager.py +9 -11
  20. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/base.py +1 -1
  21. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/PKG-INFO +76 -3
  22. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/SOURCES.txt +9 -0
  23. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/top_level.txt +1 -0
  24. {contextzip-0.3.2 → contextzip-0.3.4}/pyproject.toml +1 -2
  25. {contextzip-0.3.2 → contextzip-0.3.4}/LICENSE +0 -0
  26. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/api.py +0 -0
  27. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/cli_ai.py +0 -0
  28. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/detector.py +0 -0
  29. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/error_parser.py +0 -0
  30. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/filters.py +0 -0
  31. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/__init__.py +0 -0
  32. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/errors/__init__.py +0 -0
  33. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/errors/node.py +0 -0
  34. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/errors/python.py +0 -0
  35. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/go.py +0 -0
  36. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/node.py +0 -0
  37. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/python.py +0 -0
  38. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/ruby.py +0 -0
  39. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/rules/rust.py +0 -0
  40. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip/watcher.py +0 -0
  41. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/dependency_links.txt +0 -0
  42. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/entry_points.txt +0 -0
  43. {contextzip-0.3.2 → contextzip-0.3.4}/contextzip.egg-info/requires.txt +0 -0
  44. {contextzip-0.3.2 → contextzip-0.3.4}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: contextzip
3
- Version: 0.3.2
3
+ Version: 0.3.4
4
4
  Summary: Intelligently package your codebase for AI tools
5
5
  Author-email: Deepesh <akadeepesh@gmail.com>
6
6
  License-Expression: MIT
@@ -52,6 +52,7 @@ contextzip eliminates that entirely. Run it from your project root — it detect
52
52
  - **Git-aware packaging** — use `--git-changes` to package only modified, staged, and untracked files; perfect for incremental debugging and PR review sessions
53
53
  - **AI-powered file selection** — describe your task in plain English with `--prompt` and Gemini selects the minimum relevant files automatically, no manual hunting required
54
54
  - **Terminal error watcher** — wrap any dev server with `contextzip watch` to auto-detect errors and package a ready-to-upload debug context in one keypress
55
+ - **End-of-day / handoff prompts** — `contextzip eod` and `contextzip handoff` turn a Claude/ChatGPT conversation export plus today's code changes into a paste-ready prompt, copied straight to your clipboard
55
56
  - **Persistent workspace** — all generated ZIPs land in `.contextzip/` at your project root, discoverable, reusable, and git-ignored automatically
56
57
  - **Warns before it's a problem** — flags large (≥ 1 MB) and binary files that AI tools can't read, before you waste an upload
57
58
  - **Handles edge cases** — dangling symlinks, unreadable files, and paths outside the project tree are caught and reported, never silently dropped
@@ -117,7 +118,7 @@ contextzip [OPTIONS]
117
118
  | `--no-clipboard` | Skip the clipboard / folder-open step |
118
119
  | `--no-gitignore` | Ignore the project's `.gitignore` |
119
120
 
120
- **Subcommands:** `exclude`, `include`, `watch`, `config` — run `contextzip --help` for full details.
121
+ **Subcommands:** `exclude`, `include`, `watch`, `config`, `eod`, `handoff` — run `contextzip --help` for full details.
121
122
 
122
123
  ---
123
124
 
@@ -148,6 +149,40 @@ contextzip --output ~/Desktop/project-context.zip
148
149
 
149
150
  ---
150
151
 
152
+ ## Python API
153
+
154
+ contextzip is also usable as a library. All CLI capabilities are available as plain Python functions — no Click, no Rich output, no `SystemExit`.
155
+
156
+ ```python
157
+ from contextzip import get_git_changes, get_files, create_zip
158
+
159
+ # Get changed files and use them directly
160
+ collection = get_git_changes()
161
+ for f in collection.files: # plain pathlib.Path objects
162
+ upload(f) # no zip required
163
+
164
+ # Or zip them and upload the archive
165
+ pkg = create_zip(collection, output="/tmp/changes.zip")
166
+ with open(pkg.zip_path, "rb") as f:
167
+ upload_to_s3(f)
168
+
169
+ # Full project scan with filters
170
+ collection = get_files(include=["src/"], exclude=["tests/"])
171
+ pkg = create_zip(collection, output="/tmp/upload.zip")
172
+ print(f"{pkg.file_count} files, {pkg.compressed_bytes} bytes")
173
+ ```
174
+
175
+ | Function | Description |
176
+ |---|---|
177
+ | `get_git_changes(path?)` | Modified, added, and untracked files from git |
178
+ | `get_files(path?, include?, exclude?)` | All project files after exclusion rules |
179
+ | `create_zip(collection, output?)` | Write a `FileCollection` to a ZIP archive |
180
+ | `detect_ecosystem(path?)` | Detect framework and confidence level |
181
+
182
+ All functions default `path` to `Path.cwd()`. Errors raise typed exceptions (`NotARepositoryError`, `GitNotFoundError`, `NoFilesError`, etc.) rather than exiting.
183
+
184
+ ---
185
+
151
186
  ## AI-powered file selection
152
187
 
153
188
  The `--prompt` flag lets you describe a task in plain English. contextzip scans your project, builds a lightweight file map, and asks Gemini to return the minimum set of files needed for that task — typically 2–5, never more than 10. The result is a tightly scoped ZIP with only what you'd actually open to make the change.
@@ -209,11 +244,49 @@ Press **D** and contextzip immediately writes `.contextzip/debug-context.zip`. Y
209
244
 
210
245
  ---
211
246
 
247
+ ## End-of-day reports and chat handoffs
248
+
249
+ If you work through a problem in a Claude or ChatGPT conversation and need to either (a) summarize what you did for an end-of-day report, or (b) continue the same work in a fresh chat after hitting a usage limit, `eod` and `handoff` build the prompt for you — contextzip does no summarizing itself; that's left to whichever AI tool you paste the result into.
250
+
251
+ ```bash
252
+ contextzip eod
253
+ contextzip handoff
254
+ ```
255
+
256
+ **Setup:** export your conversation (any markdown export works) and drop the `.md` file into `exports/` at your project root. Both commands pick the most recently modified file there automatically.
257
+
258
+ **What gets built:**
259
+
260
+ - The conversation itself — pasted directly into the prompt if it's short, or referenced as an attachment if it's long enough that inlining it would burn through the next chat's context budget
261
+ - Whatever code changed, resolved per file in priority order:
262
+ 1. **Not pushed** — diffed against your upstream branch, plus the complete current file
263
+ 2. **Diverged from Claude** — diffed against the version Claude last produced (if you've set up a session key — see below), plus the complete current codebase file
264
+ 3. **Since last run** — diffed against a per-branch checkpoint that `eod`/`handoff` remember automatically, plus the complete current file
265
+ - New, untracked files are included as full content (there's nothing to diff them against)
266
+ - `eod` ends the prompt with a flat instruction to produce a work-log table; `handoff` frames it as continuing the project in a new chat
267
+
268
+ The result is copied straight to your clipboard, and also saved to `.contextzip/` if you want to look it over first.
269
+
270
+ **Optional — diffing against Claude's own version (case 2):** if you give contextzip your Claude.ai session key, `eod`/`handoff` will fetch the files Claude actually produced and compare them against your codebase, catching drift even after everything's pushed and in sync.
271
+
272
+ ```bash
273
+ contextzip config --set-session-key
274
+ ```
275
+
276
+ This is best-effort by design — it depends on an undocumented Claude.ai endpoint, so a missing key, an expired cookie, or the endpoint changing shape just skips this check with a warning rather than failing the whole command.
277
+
278
+ ```bash
279
+ contextzip eod --dry-run # preview without advancing the checkpoint
280
+ contextzip handoff --no-fetch # skip the Claude-artifact fetch for this run
281
+ ```
282
+
283
+ ---
284
+
212
285
  ## What gets excluded
213
286
 
214
287
  contextzip stacks exclusion rules based on your detected stack, on top of your `.gitignore`.
215
288
 
216
- **Always excluded:** `.git/`, `.env` files, logs, caches, editor config (`.vscode/`, `.idea/`), OS files (`.DS_Store`, `Thumbs.db`), and common binary formats.
289
+ **Always excluded:** `.git/`, `.env` files, logs, caches, editor config (`.vscode/`, `.idea/`), OS files (`.DS_Store`, `Thumbs.db`), common binary formats, and contextzip's own `.contextzip/` and `exports/` working folders.
217
290
 
218
291
  **By framework:**
219
292
 
@@ -23,6 +23,7 @@ contextzip eliminates that entirely. Run it from your project root — it detect
23
23
  - **Git-aware packaging** — use `--git-changes` to package only modified, staged, and untracked files; perfect for incremental debugging and PR review sessions
24
24
  - **AI-powered file selection** — describe your task in plain English with `--prompt` and Gemini selects the minimum relevant files automatically, no manual hunting required
25
25
  - **Terminal error watcher** — wrap any dev server with `contextzip watch` to auto-detect errors and package a ready-to-upload debug context in one keypress
26
+ - **End-of-day / handoff prompts** — `contextzip eod` and `contextzip handoff` turn a Claude/ChatGPT conversation export plus today's code changes into a paste-ready prompt, copied straight to your clipboard
26
27
  - **Persistent workspace** — all generated ZIPs land in `.contextzip/` at your project root, discoverable, reusable, and git-ignored automatically
27
28
  - **Warns before it's a problem** — flags large (≥ 1 MB) and binary files that AI tools can't read, before you waste an upload
28
29
  - **Handles edge cases** — dangling symlinks, unreadable files, and paths outside the project tree are caught and reported, never silently dropped
@@ -88,7 +89,7 @@ contextzip [OPTIONS]
88
89
  | `--no-clipboard` | Skip the clipboard / folder-open step |
89
90
  | `--no-gitignore` | Ignore the project's `.gitignore` |
90
91
 
91
- **Subcommands:** `exclude`, `include`, `watch`, `config` — run `contextzip --help` for full details.
92
+ **Subcommands:** `exclude`, `include`, `watch`, `config`, `eod`, `handoff` — run `contextzip --help` for full details.
92
93
 
93
94
  ---
94
95
 
@@ -119,6 +120,40 @@ contextzip --output ~/Desktop/project-context.zip
119
120
 
120
121
  ---
121
122
 
123
+ ## Python API
124
+
125
+ contextzip is also usable as a library. All CLI capabilities are available as plain Python functions — no Click, no Rich output, no `SystemExit`.
126
+
127
+ ```python
128
+ from contextzip import get_git_changes, get_files, create_zip
129
+
130
+ # Get changed files and use them directly
131
+ collection = get_git_changes()
132
+ for f in collection.files: # plain pathlib.Path objects
133
+ upload(f) # no zip required
134
+
135
+ # Or zip them and upload the archive
136
+ pkg = create_zip(collection, output="/tmp/changes.zip")
137
+ with open(pkg.zip_path, "rb") as f:
138
+ upload_to_s3(f)
139
+
140
+ # Full project scan with filters
141
+ collection = get_files(include=["src/"], exclude=["tests/"])
142
+ pkg = create_zip(collection, output="/tmp/upload.zip")
143
+ print(f"{pkg.file_count} files, {pkg.compressed_bytes} bytes")
144
+ ```
145
+
146
+ | Function | Description |
147
+ |---|---|
148
+ | `get_git_changes(path?)` | Modified, added, and untracked files from git |
149
+ | `get_files(path?, include?, exclude?)` | All project files after exclusion rules |
150
+ | `create_zip(collection, output?)` | Write a `FileCollection` to a ZIP archive |
151
+ | `detect_ecosystem(path?)` | Detect framework and confidence level |
152
+
153
+ All functions default `path` to `Path.cwd()`. Errors raise typed exceptions (`NotARepositoryError`, `GitNotFoundError`, `NoFilesError`, etc.) rather than exiting.
154
+
155
+ ---
156
+
122
157
  ## AI-powered file selection
123
158
 
124
159
  The `--prompt` flag lets you describe a task in plain English. contextzip scans your project, builds a lightweight file map, and asks Gemini to return the minimum set of files needed for that task — typically 2–5, never more than 10. The result is a tightly scoped ZIP with only what you'd actually open to make the change.
@@ -180,11 +215,49 @@ Press **D** and contextzip immediately writes `.contextzip/debug-context.zip`. Y
180
215
 
181
216
  ---
182
217
 
218
+ ## End-of-day reports and chat handoffs
219
+
220
+ If you work through a problem in a Claude or ChatGPT conversation and need to either (a) summarize what you did for an end-of-day report, or (b) continue the same work in a fresh chat after hitting a usage limit, `eod` and `handoff` build the prompt for you — contextzip does no summarizing itself; that's left to whichever AI tool you paste the result into.
221
+
222
+ ```bash
223
+ contextzip eod
224
+ contextzip handoff
225
+ ```
226
+
227
+ **Setup:** export your conversation (any markdown export works) and drop the `.md` file into `exports/` at your project root. Both commands pick the most recently modified file there automatically.
228
+
229
+ **What gets built:**
230
+
231
+ - The conversation itself — pasted directly into the prompt if it's short, or referenced as an attachment if it's long enough that inlining it would burn through the next chat's context budget
232
+ - Whatever code changed, resolved per file in priority order:
233
+ 1. **Not pushed** — diffed against your upstream branch, plus the complete current file
234
+ 2. **Diverged from Claude** — diffed against the version Claude last produced (if you've set up a session key — see below), plus the complete current codebase file
235
+ 3. **Since last run** — diffed against a per-branch checkpoint that `eod`/`handoff` remember automatically, plus the complete current file
236
+ - New, untracked files are included as full content (there's nothing to diff them against)
237
+ - `eod` ends the prompt with a flat instruction to produce a work-log table; `handoff` frames it as continuing the project in a new chat
238
+
239
+ The result is copied straight to your clipboard, and also saved to `.contextzip/` if you want to look it over first.
240
+
241
+ **Optional — diffing against Claude's own version (case 2):** if you give contextzip your Claude.ai session key, `eod`/`handoff` will fetch the files Claude actually produced and compare them against your codebase, catching drift even after everything's pushed and in sync.
242
+
243
+ ```bash
244
+ contextzip config --set-session-key
245
+ ```
246
+
247
+ This is best-effort by design — it depends on an undocumented Claude.ai endpoint, so a missing key, an expired cookie, or the endpoint changing shape just skips this check with a warning rather than failing the whole command.
248
+
249
+ ```bash
250
+ contextzip eod --dry-run # preview without advancing the checkpoint
251
+ contextzip handoff --no-fetch # skip the Claude-artifact fetch for this run
252
+ ```
253
+
254
+ ---
255
+
183
256
  ## What gets excluded
184
257
 
185
258
  contextzip stacks exclusion rules based on your detected stack, on top of your `.gitignore`.
186
259
 
187
- **Always excluded:** `.git/`, `.env` files, logs, caches, editor config (`.vscode/`, `.idea/`), OS files (`.DS_Store`, `Thumbs.db`), and common binary formats.
260
+ **Always excluded:** `.git/`, `.env` files, logs, caches, editor config (`.vscode/`, `.idea/`), OS files (`.DS_Store`, `Thumbs.db`), common binary formats, and contextzip's own `.contextzip/` and `exports/` working folders.
188
261
 
189
262
  **By framework:**
190
263
 
@@ -1,6 +1,6 @@
1
1
  """contextzip — intelligent codebase packager for AI tools."""
2
2
 
3
- __version__ = "0.3.2"
3
+ __version__ = "0.3.4"
4
4
 
5
5
  from contextzip.api import (
6
6
  FileCollection,
@@ -0,0 +1 @@
1
+ """contextzip.ai — AI-powered file selection for --prompt mode."""
@@ -0,0 +1,282 @@
1
+ """
2
+ ai/gemini.py — Thin Gemini API client for contextzip's prompt-aware mode.
3
+
4
+ Makes a single POST to the Gemini generateContent endpoint and returns
5
+ a ranked list of file paths relevant to the user's task description.
6
+
7
+ Design principles:
8
+ - Raw httpx calls only — no Google SDK dependency
9
+ - Strict JSON output from the model — no markdown, no prose
10
+ - Validates every returned path against the real file tree
11
+ - Single responsibility: call API, parse response, validate paths
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import json
17
+
18
+ try:
19
+ import httpx
20
+
21
+ _HTTPX_AVAILABLE = True
22
+ except ImportError:
23
+ _HTTPX_AVAILABLE = False
24
+
25
+
26
+ # ---------------------------------------------------------------------------
27
+ # Constants
28
+ # ---------------------------------------------------------------------------
29
+
30
+ _API_URL = (
31
+ "https://generativelanguage.googleapis.com/v1beta/models"
32
+ "/{model}:generateContent?key={key}"
33
+ )
34
+
35
+ DEFAULT_MODEL = "gemini-2.5-flash-lite"
36
+
37
+ # Hard cap: never return more than this many files regardless of model output.
38
+ # Minimum context is the goal — the prompt enforces this, but we double-guard.
39
+ _MAX_FILES = 12
40
+
41
+ _TIMEOUT_SECONDS = 30
42
+
43
+
44
+ # ---------------------------------------------------------------------------
45
+ # Exceptions
46
+ # ---------------------------------------------------------------------------
47
+
48
+
49
+ class GeminiError(Exception):
50
+ """Raised when the Gemini API call fails for any reason."""
51
+
52
+
53
+ class GeminiUnavailable(GeminiError):
54
+ """Raised when httpx is not installed."""
55
+
56
+
57
+ class GeminiRateLimitError(GeminiError):
58
+ """Raised specifically on HTTP 429 — allows callers to trigger fallback."""
59
+
60
+
61
+ # ---------------------------------------------------------------------------
62
+ # Public API
63
+ # ---------------------------------------------------------------------------
64
+
65
+
66
+ def select_files(
67
+ *,
68
+ api_key: str,
69
+ prompt: str,
70
+ file_tree: list[tuple[str, int]], # [(rel_path, size_bytes), ...]
71
+ ecosystem: str,
72
+ model: str = DEFAULT_MODEL,
73
+ ) -> list[str]:
74
+ """
75
+ Ask Gemini which files are relevant to *prompt* and return their paths.
76
+
77
+ Parameters
78
+ ----------
79
+ api_key:
80
+ Gemini API key from Google AI Studio.
81
+ prompt:
82
+ The user's natural-language task description.
83
+ file_tree:
84
+ All candidate files as (relative_posix_path, size_bytes) tuples.
85
+ These should already have contextzip's standard exclusions applied —
86
+ no node_modules, no build artifacts, no .env files.
87
+ ecosystem:
88
+ Human-readable detected framework string, e.g. "Next.js + TypeScript".
89
+ Gives the model important context for relevance scoring.
90
+ model:
91
+ Gemini model identifier. Defaults to gemini-2.5-flash-lite.
92
+
93
+ Returns
94
+ -------
95
+ list[str]
96
+ Relative POSIX paths of the selected files, ordered by relevance
97
+ (most relevant first). Always a strict subset of the input file_tree
98
+ paths — hallucinated or non-existent paths are silently dropped.
99
+
100
+ Raises
101
+ ------
102
+ GeminiUnavailable
103
+ If httpx is not installed.
104
+ GeminiError
105
+ On any API or parsing failure.
106
+ """
107
+ if not _HTTPX_AVAILABLE:
108
+ raise GeminiUnavailable(
109
+ "httpx is required for AI-powered selection. "
110
+ "Install it with: pip install httpx"
111
+ )
112
+
113
+ system_prompt = _build_system_prompt()
114
+ user_message = _build_user_message(prompt, file_tree, ecosystem)
115
+
116
+ url = _API_URL.format(model=model, key=api_key)
117
+
118
+ payload = {
119
+ "contents": [
120
+ {
121
+ "role": "user",
122
+ "parts": [{"text": f"{system_prompt}\n\n{user_message}"}],
123
+ }
124
+ ],
125
+ "generationConfig": {
126
+ "temperature": 0.0, # deterministic — this is a ranking task
127
+ "maxOutputTokens": 512, # a list of paths needs very few tokens
128
+ "responseMimeType": "application/json",
129
+ },
130
+ }
131
+
132
+ try:
133
+ response = httpx.post(url, json=payload, timeout=_TIMEOUT_SECONDS)
134
+ except httpx.TimeoutException:
135
+ raise GeminiError("Request timed out after 30 seconds.")
136
+ except httpx.RequestError as exc:
137
+ raise GeminiError(f"Network error: {exc}")
138
+
139
+ if response.status_code == 400:
140
+ raise GeminiError("Invalid request — check your API key format.")
141
+ if response.status_code == 401 or response.status_code == 403:
142
+ raise GeminiError(
143
+ "API key rejected. Run [cyan]contextzip config --reset-key[/] to update it."
144
+ )
145
+ if response.status_code == 429:
146
+ raise GeminiRateLimitError(
147
+ "Rate limit reached on the free tier (15 req/min). "
148
+ "Wait a moment and try again, or your new key may still be activating — "
149
+ "Google can take up to 60 seconds after key creation."
150
+ )
151
+ if response.status_code != 200:
152
+ raise GeminiError(
153
+ f"Gemini API returned HTTP {response.status_code}: {response.text[:200]}"
154
+ )
155
+
156
+ return _parse_response(response.json(), file_tree)
157
+
158
+
159
+ # ---------------------------------------------------------------------------
160
+ # Prompt construction
161
+ # ---------------------------------------------------------------------------
162
+
163
+
164
+ def _build_system_prompt() -> str:
165
+ return """\
166
+ You are a precise file relevance assistant for software projects.
167
+
168
+ Your only job: given a developer's task description and a project file tree,
169
+ return the MINIMUM set of files a developer would need to open to complete
170
+ that task. Think like a senior engineer doing a surgical code change —
171
+ open only what you must touch or read to understand the change.
172
+
173
+ Rules you must follow:
174
+ - Return ONLY a JSON array of file path strings. No explanation, no markdown,
175
+ no extra keys. Example: ["src/auth.ts", "components/Toast.tsx"]
176
+ - Be ruthless about exclusion. If a file is not directly relevant to the
177
+ stated task, leave it out. Err heavily on the side of fewer files.
178
+ - Prefer files that will be MODIFIED over files that are merely referenced.
179
+ - Config files, test files, and documentation should only appear if the
180
+ task explicitly concerns them.
181
+ - Never return more than 10 files. For most tasks 2–5 files is correct.
182
+ - Order by relevance: most directly relevant file first.\
183
+ """
184
+
185
+
186
+ def _build_user_message(
187
+ prompt: str,
188
+ file_tree: list[tuple[str, int]],
189
+ ecosystem: str,
190
+ ) -> str:
191
+ tree_lines = "\n".join(f"{path} ({_human_size(size)})" for path, size in file_tree)
192
+ return f"""\
193
+ Framework: {ecosystem}
194
+
195
+ Task: {prompt}
196
+
197
+ Project files (exclusions already applied):
198
+ {tree_lines}
199
+
200
+ Return only a JSON array of the most relevant file paths for this task.\
201
+ """
202
+
203
+
204
+ # ---------------------------------------------------------------------------
205
+ # Response parsing and validation
206
+ # ---------------------------------------------------------------------------
207
+
208
+
209
+ def _parse_response(
210
+ data: dict,
211
+ file_tree: list[tuple[str, int]],
212
+ ) -> list[str]:
213
+ """
214
+ Extract and validate the file list from the Gemini API response.
215
+
216
+ - Parses the JSON array from the model's text output
217
+ - Drops any path the model hallucinated (not in the real file tree)
218
+ - Enforces the _MAX_FILES hard cap
219
+ - Warns (via exception) if the model returned mostly invalid paths
220
+ """
221
+ # Navigate the Gemini response structure
222
+ try:
223
+ text = data["candidates"][0]["content"]["parts"][0]["text"]
224
+ except (KeyError, IndexError) as exc:
225
+ raise GeminiError(f"Unexpected API response structure: {exc}\n{data}")
226
+
227
+ # Strip any accidental markdown fences the model might add
228
+ text = text.strip().strip("`").strip()
229
+ if text.startswith("json"):
230
+ text = text[4:].strip()
231
+
232
+ try:
233
+ raw_paths: list = json.loads(text)
234
+ except json.JSONDecodeError as exc:
235
+ raise GeminiError(
236
+ f"Model returned non-JSON output: {exc}\nRaw output: {text[:300]}"
237
+ )
238
+
239
+ if not isinstance(raw_paths, list):
240
+ raise GeminiError(
241
+ f"Expected a JSON array, got {type(raw_paths).__name__}: {text[:200]}"
242
+ )
243
+
244
+ # Build a set of valid paths for O(1) lookup
245
+ valid_paths: set[str] = {path for path, _ in file_tree}
246
+
247
+ validated: list[str] = []
248
+ hallucinated = 0
249
+
250
+ for item in raw_paths:
251
+ if not isinstance(item, str):
252
+ continue
253
+ # Normalise separators (model may return backslashes on Windows prompts)
254
+ normalised = item.replace("\\", "/").strip()
255
+ if normalised in valid_paths:
256
+ validated.append(normalised)
257
+ else:
258
+ hallucinated += 1
259
+
260
+ # Warn if the model was mostly making things up
261
+ total_returned = len(raw_paths)
262
+ if total_returned > 0 and hallucinated / total_returned > 0.5:
263
+ raise GeminiError(
264
+ f"Model returned {hallucinated}/{total_returned} non-existent paths. "
265
+ "This may indicate a model or prompt issue. Try a more specific prompt."
266
+ )
267
+
268
+ # Enforce hard cap
269
+ return validated[:_MAX_FILES]
270
+
271
+
272
+ # ---------------------------------------------------------------------------
273
+ # Helpers
274
+ # ---------------------------------------------------------------------------
275
+
276
+
277
+ def _human_size(n: int) -> str:
278
+ for unit in ("B", "KB", "MB"):
279
+ if n < 1024:
280
+ return f"{n:.0f} {unit}"
281
+ n /= 1024
282
+ return f"{n:.1f} GB"