contextzip 0.3.2__tar.gz → 0.3.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. {contextzip-0.3.2 → contextzip-0.3.3}/PKG-INFO +35 -1
  2. {contextzip-0.3.2 → contextzip-0.3.3}/README.md +34 -0
  3. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/__init__.py +1 -1
  4. contextzip-0.3.3/contextzip/ai/__init__.py +1 -0
  5. contextzip-0.3.3/contextzip/ai/gemini.py +282 -0
  6. contextzip-0.3.3/contextzip/ai/heuristic.py +333 -0
  7. contextzip-0.3.3/contextzip/ai/selector.py +158 -0
  8. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip.egg-info/PKG-INFO +35 -1
  9. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip.egg-info/SOURCES.txt +4 -0
  10. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip.egg-info/top_level.txt +1 -0
  11. {contextzip-0.3.2 → contextzip-0.3.3}/pyproject.toml +1 -2
  12. {contextzip-0.3.2 → contextzip-0.3.3}/LICENSE +0 -0
  13. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/api.py +0 -0
  14. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/cli.py +0 -0
  15. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/cli_ai.py +0 -0
  16. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/cli_display.py +0 -0
  17. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/cli_onboard.py +0 -0
  18. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/clipboard.py +0 -0
  19. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/config.py +0 -0
  20. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/detector.py +0 -0
  21. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/error_parser.py +0 -0
  22. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/filters.py +0 -0
  23. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/git.py +0 -0
  24. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/packager.py +0 -0
  25. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/__init__.py +0 -0
  26. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/base.py +0 -0
  27. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/errors/__init__.py +0 -0
  28. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/errors/node.py +0 -0
  29. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/errors/python.py +0 -0
  30. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/go.py +0 -0
  31. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/node.py +0 -0
  32. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/python.py +0 -0
  33. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/ruby.py +0 -0
  34. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/rules/rust.py +0 -0
  35. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip/watcher.py +0 -0
  36. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip.egg-info/dependency_links.txt +0 -0
  37. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip.egg-info/entry_points.txt +0 -0
  38. {contextzip-0.3.2 → contextzip-0.3.3}/contextzip.egg-info/requires.txt +0 -0
  39. {contextzip-0.3.2 → contextzip-0.3.3}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: contextzip
3
- Version: 0.3.2
3
+ Version: 0.3.3
4
4
  Summary: Intelligently package your codebase for AI tools
5
5
  Author-email: Deepesh <akadeepesh@gmail.com>
6
6
  License-Expression: MIT
@@ -148,6 +148,40 @@ contextzip --output ~/Desktop/project-context.zip
148
148
 
149
149
  ---
150
150
 
151
+ ## Python API
152
+
153
+ contextzip is also usable as a library. All CLI capabilities are available as plain Python functions — no Click, no Rich output, no `SystemExit`.
154
+
155
+ ```python
156
+ from contextzip import get_git_changes, get_files, create_zip
157
+
158
+ # Get changed files and use them directly
159
+ collection = get_git_changes()
160
+ for f in collection.files: # plain pathlib.Path objects
161
+ upload(f) # no zip required
162
+
163
+ # Or zip them and upload the archive
164
+ pkg = create_zip(collection, output="/tmp/changes.zip")
165
+ with open(pkg.zip_path, "rb") as f:
166
+ upload_to_s3(f)
167
+
168
+ # Full project scan with filters
169
+ collection = get_files(include=["src/"], exclude=["tests/"])
170
+ pkg = create_zip(collection, output="/tmp/upload.zip")
171
+ print(f"{pkg.file_count} files, {pkg.compressed_bytes} bytes")
172
+ ```
173
+
174
+ | Function | Description |
175
+ |---|---|
176
+ | `get_git_changes(path?)` | Modified, added, and untracked files from git |
177
+ | `get_files(path?, include?, exclude?)` | All project files after exclusion rules |
178
+ | `create_zip(collection, output?)` | Write a `FileCollection` to a ZIP archive |
179
+ | `detect_ecosystem(path?)` | Detect framework and confidence level |
180
+
181
+ All functions default `path` to `Path.cwd()`. Errors raise typed exceptions (`NotARepositoryError`, `GitNotFoundError`, `NoFilesError`, etc.) rather than exiting.
182
+
183
+ ---
184
+
151
185
  ## AI-powered file selection
152
186
 
153
187
  The `--prompt` flag lets you describe a task in plain English. contextzip scans your project, builds a lightweight file map, and asks Gemini to return the minimum set of files needed for that task — typically 2–5, never more than 10. The result is a tightly scoped ZIP with only what you'd actually open to make the change.
@@ -119,6 +119,40 @@ contextzip --output ~/Desktop/project-context.zip
119
119
 
120
120
  ---
121
121
 
122
+ ## Python API
123
+
124
+ contextzip is also usable as a library. All CLI capabilities are available as plain Python functions — no Click, no Rich output, no `SystemExit`.
125
+
126
+ ```python
127
+ from contextzip import get_git_changes, get_files, create_zip
128
+
129
+ # Get changed files and use them directly
130
+ collection = get_git_changes()
131
+ for f in collection.files: # plain pathlib.Path objects
132
+ upload(f) # no zip required
133
+
134
+ # Or zip them and upload the archive
135
+ pkg = create_zip(collection, output="/tmp/changes.zip")
136
+ with open(pkg.zip_path, "rb") as f:
137
+ upload_to_s3(f)
138
+
139
+ # Full project scan with filters
140
+ collection = get_files(include=["src/"], exclude=["tests/"])
141
+ pkg = create_zip(collection, output="/tmp/upload.zip")
142
+ print(f"{pkg.file_count} files, {pkg.compressed_bytes} bytes")
143
+ ```
144
+
145
+ | Function | Description |
146
+ |---|---|
147
+ | `get_git_changes(path?)` | Modified, added, and untracked files from git |
148
+ | `get_files(path?, include?, exclude?)` | All project files after exclusion rules |
149
+ | `create_zip(collection, output?)` | Write a `FileCollection` to a ZIP archive |
150
+ | `detect_ecosystem(path?)` | Detect framework and confidence level |
151
+
152
+ All functions default `path` to `Path.cwd()`. Errors raise typed exceptions (`NotARepositoryError`, `GitNotFoundError`, `NoFilesError`, etc.) rather than exiting.
153
+
154
+ ---
155
+
122
156
  ## AI-powered file selection
123
157
 
124
158
  The `--prompt` flag lets you describe a task in plain English. contextzip scans your project, builds a lightweight file map, and asks Gemini to return the minimum set of files needed for that task — typically 2–5, never more than 10. The result is a tightly scoped ZIP with only what you'd actually open to make the change.
@@ -1,6 +1,6 @@
1
1
  """contextzip — intelligent codebase packager for AI tools."""
2
2
 
3
- __version__ = "0.3.2"
3
+ __version__ = "0.3.3"
4
4
 
5
5
  from contextzip.api import (
6
6
  FileCollection,
@@ -0,0 +1 @@
1
+ """contextzip.ai — AI-powered file selection for --prompt mode."""
@@ -0,0 +1,282 @@
1
+ """
2
+ ai/gemini.py — Thin Gemini API client for contextzip's prompt-aware mode.
3
+
4
+ Makes a single POST to the Gemini generateContent endpoint and returns
5
+ a ranked list of file paths relevant to the user's task description.
6
+
7
+ Design principles:
8
+ - Raw httpx calls only — no Google SDK dependency
9
+ - Strict JSON output from the model — no markdown, no prose
10
+ - Validates every returned path against the real file tree
11
+ - Single responsibility: call API, parse response, validate paths
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import json
17
+
18
+ try:
19
+ import httpx
20
+
21
+ _HTTPX_AVAILABLE = True
22
+ except ImportError:
23
+ _HTTPX_AVAILABLE = False
24
+
25
+
26
+ # ---------------------------------------------------------------------------
27
+ # Constants
28
+ # ---------------------------------------------------------------------------
29
+
30
+ _API_URL = (
31
+ "https://generativelanguage.googleapis.com/v1beta/models"
32
+ "/{model}:generateContent?key={key}"
33
+ )
34
+
35
+ DEFAULT_MODEL = "gemini-2.5-flash-lite"
36
+
37
+ # Hard cap: never return more than this many files regardless of model output.
38
+ # Minimum context is the goal — the prompt enforces this, but we double-guard.
39
+ _MAX_FILES = 12
40
+
41
+ _TIMEOUT_SECONDS = 30
42
+
43
+
44
+ # ---------------------------------------------------------------------------
45
+ # Exceptions
46
+ # ---------------------------------------------------------------------------
47
+
48
+
49
+ class GeminiError(Exception):
50
+ """Raised when the Gemini API call fails for any reason."""
51
+
52
+
53
+ class GeminiUnavailable(GeminiError):
54
+ """Raised when httpx is not installed."""
55
+
56
+
57
+ class GeminiRateLimitError(GeminiError):
58
+ """Raised specifically on HTTP 429 — allows callers to trigger fallback."""
59
+
60
+
61
+ # ---------------------------------------------------------------------------
62
+ # Public API
63
+ # ---------------------------------------------------------------------------
64
+
65
+
66
+ def select_files(
67
+ *,
68
+ api_key: str,
69
+ prompt: str,
70
+ file_tree: list[tuple[str, int]], # [(rel_path, size_bytes), ...]
71
+ ecosystem: str,
72
+ model: str = DEFAULT_MODEL,
73
+ ) -> list[str]:
74
+ """
75
+ Ask Gemini which files are relevant to *prompt* and return their paths.
76
+
77
+ Parameters
78
+ ----------
79
+ api_key:
80
+ Gemini API key from Google AI Studio.
81
+ prompt:
82
+ The user's natural-language task description.
83
+ file_tree:
84
+ All candidate files as (relative_posix_path, size_bytes) tuples.
85
+ These should already have contextzip's standard exclusions applied —
86
+ no node_modules, no build artifacts, no .env files.
87
+ ecosystem:
88
+ Human-readable detected framework string, e.g. "Next.js + TypeScript".
89
+ Gives the model important context for relevance scoring.
90
+ model:
91
+ Gemini model identifier. Defaults to gemini-2.5-flash-lite.
92
+
93
+ Returns
94
+ -------
95
+ list[str]
96
+ Relative POSIX paths of the selected files, ordered by relevance
97
+ (most relevant first). Always a strict subset of the input file_tree
98
+ paths — hallucinated or non-existent paths are silently dropped.
99
+
100
+ Raises
101
+ ------
102
+ GeminiUnavailable
103
+ If httpx is not installed.
104
+ GeminiError
105
+ On any API or parsing failure.
106
+ """
107
+ if not _HTTPX_AVAILABLE:
108
+ raise GeminiUnavailable(
109
+ "httpx is required for AI-powered selection. "
110
+ "Install it with: pip install httpx"
111
+ )
112
+
113
+ system_prompt = _build_system_prompt()
114
+ user_message = _build_user_message(prompt, file_tree, ecosystem)
115
+
116
+ url = _API_URL.format(model=model, key=api_key)
117
+
118
+ payload = {
119
+ "contents": [
120
+ {
121
+ "role": "user",
122
+ "parts": [{"text": f"{system_prompt}\n\n{user_message}"}],
123
+ }
124
+ ],
125
+ "generationConfig": {
126
+ "temperature": 0.0, # deterministic — this is a ranking task
127
+ "maxOutputTokens": 512, # a list of paths needs very few tokens
128
+ "responseMimeType": "application/json",
129
+ },
130
+ }
131
+
132
+ try:
133
+ response = httpx.post(url, json=payload, timeout=_TIMEOUT_SECONDS)
134
+ except httpx.TimeoutException:
135
+ raise GeminiError("Request timed out after 30 seconds.")
136
+ except httpx.RequestError as exc:
137
+ raise GeminiError(f"Network error: {exc}")
138
+
139
+ if response.status_code == 400:
140
+ raise GeminiError("Invalid request — check your API key format.")
141
+ if response.status_code == 401 or response.status_code == 403:
142
+ raise GeminiError(
143
+ "API key rejected. Run [cyan]contextzip config --reset-key[/] to update it."
144
+ )
145
+ if response.status_code == 429:
146
+ raise GeminiRateLimitError(
147
+ "Rate limit reached on the free tier (15 req/min). "
148
+ "Wait a moment and try again, or your new key may still be activating — "
149
+ "Google can take up to 60 seconds after key creation."
150
+ )
151
+ if response.status_code != 200:
152
+ raise GeminiError(
153
+ f"Gemini API returned HTTP {response.status_code}: {response.text[:200]}"
154
+ )
155
+
156
+ return _parse_response(response.json(), file_tree)
157
+
158
+
159
+ # ---------------------------------------------------------------------------
160
+ # Prompt construction
161
+ # ---------------------------------------------------------------------------
162
+
163
+
164
+ def _build_system_prompt() -> str:
165
+ return """\
166
+ You are a precise file relevance assistant for software projects.
167
+
168
+ Your only job: given a developer's task description and a project file tree,
169
+ return the MINIMUM set of files a developer would need to open to complete
170
+ that task. Think like a senior engineer doing a surgical code change —
171
+ open only what you must touch or read to understand the change.
172
+
173
+ Rules you must follow:
174
+ - Return ONLY a JSON array of file path strings. No explanation, no markdown,
175
+ no extra keys. Example: ["src/auth.ts", "components/Toast.tsx"]
176
+ - Be ruthless about exclusion. If a file is not directly relevant to the
177
+ stated task, leave it out. Err heavily on the side of fewer files.
178
+ - Prefer files that will be MODIFIED over files that are merely referenced.
179
+ - Config files, test files, and documentation should only appear if the
180
+ task explicitly concerns them.
181
+ - Never return more than 10 files. For most tasks 2–5 files is correct.
182
+ - Order by relevance: most directly relevant file first.\
183
+ """
184
+
185
+
186
+ def _build_user_message(
187
+ prompt: str,
188
+ file_tree: list[tuple[str, int]],
189
+ ecosystem: str,
190
+ ) -> str:
191
+ tree_lines = "\n".join(f"{path} ({_human_size(size)})" for path, size in file_tree)
192
+ return f"""\
193
+ Framework: {ecosystem}
194
+
195
+ Task: {prompt}
196
+
197
+ Project files (exclusions already applied):
198
+ {tree_lines}
199
+
200
+ Return only a JSON array of the most relevant file paths for this task.\
201
+ """
202
+
203
+
204
+ # ---------------------------------------------------------------------------
205
+ # Response parsing and validation
206
+ # ---------------------------------------------------------------------------
207
+
208
+
209
+ def _parse_response(
210
+ data: dict,
211
+ file_tree: list[tuple[str, int]],
212
+ ) -> list[str]:
213
+ """
214
+ Extract and validate the file list from the Gemini API response.
215
+
216
+ - Parses the JSON array from the model's text output
217
+ - Drops any path the model hallucinated (not in the real file tree)
218
+ - Enforces the _MAX_FILES hard cap
219
+ - Warns (via exception) if the model returned mostly invalid paths
220
+ """
221
+ # Navigate the Gemini response structure
222
+ try:
223
+ text = data["candidates"][0]["content"]["parts"][0]["text"]
224
+ except (KeyError, IndexError) as exc:
225
+ raise GeminiError(f"Unexpected API response structure: {exc}\n{data}")
226
+
227
+ # Strip any accidental markdown fences the model might add
228
+ text = text.strip().strip("`").strip()
229
+ if text.startswith("json"):
230
+ text = text[4:].strip()
231
+
232
+ try:
233
+ raw_paths: list = json.loads(text)
234
+ except json.JSONDecodeError as exc:
235
+ raise GeminiError(
236
+ f"Model returned non-JSON output: {exc}\nRaw output: {text[:300]}"
237
+ )
238
+
239
+ if not isinstance(raw_paths, list):
240
+ raise GeminiError(
241
+ f"Expected a JSON array, got {type(raw_paths).__name__}: {text[:200]}"
242
+ )
243
+
244
+ # Build a set of valid paths for O(1) lookup
245
+ valid_paths: set[str] = {path for path, _ in file_tree}
246
+
247
+ validated: list[str] = []
248
+ hallucinated = 0
249
+
250
+ for item in raw_paths:
251
+ if not isinstance(item, str):
252
+ continue
253
+ # Normalise separators (model may return backslashes on Windows prompts)
254
+ normalised = item.replace("\\", "/").strip()
255
+ if normalised in valid_paths:
256
+ validated.append(normalised)
257
+ else:
258
+ hallucinated += 1
259
+
260
+ # Warn if the model was mostly making things up
261
+ total_returned = len(raw_paths)
262
+ if total_returned > 0 and hallucinated / total_returned > 0.5:
263
+ raise GeminiError(
264
+ f"Model returned {hallucinated}/{total_returned} non-existent paths. "
265
+ "This may indicate a model or prompt issue. Try a more specific prompt."
266
+ )
267
+
268
+ # Enforce hard cap
269
+ return validated[:_MAX_FILES]
270
+
271
+
272
+ # ---------------------------------------------------------------------------
273
+ # Helpers
274
+ # ---------------------------------------------------------------------------
275
+
276
+
277
+ def _human_size(n: int) -> str:
278
+ for unit in ("B", "KB", "MB"):
279
+ if n < 1024:
280
+ return f"{n:.0f} {unit}"
281
+ n /= 1024
282
+ return f"{n:.1f} GB"
@@ -0,0 +1,333 @@
1
+ """
2
+ ai/heuristic.py — Keyword-based file relevance scorer.
3
+
4
+ Used as a fallback when the Gemini API is unavailable (rate limit, network
5
+ error, no key). Scores each file purely from the prompt text and file path —
6
+ no API calls, no dependencies beyond stdlib.
7
+
8
+ This is intentionally simple and transparent. It will never be as accurate
9
+ as Gemini, but it is always available and produces reasonable results for
10
+ common cases like "update the login page" or "fix the toast component".
11
+
12
+ Scoring model
13
+ ─────────────
14
+ Each file receives a score based on:
15
+
16
+ 1. Token overlap — how many meaningful words from the prompt appear in
17
+ the file's path components (stem + parent dirs).
18
+ Longer matches score higher.
19
+
20
+ 2. Directory bias — source-like directories (src/, app/, lib/, components/,
21
+ utils/, routes/, pages/, server/, api/) score higher.
22
+ Test and documentation directories score lower.
23
+
24
+ 3. Extension bias — code files score higher than config, lock, or data files.
25
+
26
+ Files with a score of zero are excluded entirely. The top-N results are
27
+ returned, where N is capped at MAX_FILES.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import re
33
+ from pathlib import Path
34
+
35
+ # Never return more files than this regardless of score
36
+ MAX_FILES = 8
37
+
38
+ # Words that carry no signal — filtered out before scoring
39
+ _STOPWORDS = frozenset(
40
+ {
41
+ "i",
42
+ "a",
43
+ "an",
44
+ "the",
45
+ "to",
46
+ "in",
47
+ "on",
48
+ "at",
49
+ "of",
50
+ "for",
51
+ "and",
52
+ "or",
53
+ "but",
54
+ "is",
55
+ "it",
56
+ "be",
57
+ "do",
58
+ "my",
59
+ "this",
60
+ "that",
61
+ "with",
62
+ "from",
63
+ "want",
64
+ "need",
65
+ "make",
66
+ "update",
67
+ "change",
68
+ "fix",
69
+ "add",
70
+ "get",
71
+ "use",
72
+ "set",
73
+ "new",
74
+ "old",
75
+ "file",
76
+ "code",
77
+ "function",
78
+ "method",
79
+ "class",
80
+ "variable",
81
+ }
82
+ )
83
+
84
+ # Directory names that suggest source code worth including
85
+ _SOURCE_DIRS = frozenset(
86
+ {
87
+ "src",
88
+ "app",
89
+ "lib",
90
+ "libs",
91
+ "components",
92
+ "component",
93
+ "utils",
94
+ "util",
95
+ "helpers",
96
+ "helper",
97
+ "routes",
98
+ "route",
99
+ "pages",
100
+ "page",
101
+ "server",
102
+ "api",
103
+ "services",
104
+ "service",
105
+ "hooks",
106
+ "hook",
107
+ "store",
108
+ "stores",
109
+ "context",
110
+ "contexts",
111
+ "middleware",
112
+ "handlers",
113
+ "handler",
114
+ "controllers",
115
+ "controller",
116
+ "views",
117
+ "view",
118
+ "models",
119
+ "model",
120
+ "core",
121
+ "common",
122
+ "shared",
123
+ }
124
+ )
125
+
126
+ # Directory names that suggest low relevance for most coding tasks
127
+ _NOISE_DIRS = frozenset(
128
+ {
129
+ "test",
130
+ "tests",
131
+ "__tests__",
132
+ "spec",
133
+ "specs",
134
+ "docs",
135
+ "doc",
136
+ "documentation",
137
+ "examples",
138
+ "example",
139
+ "fixtures",
140
+ "mocks",
141
+ "mock",
142
+ "stubs",
143
+ "stub",
144
+ "scripts",
145
+ "bin",
146
+ "dist",
147
+ "build",
148
+ "out",
149
+ "coverage",
150
+ }
151
+ )
152
+
153
+ # Extensions that suggest runnable/editable source code
154
+ _CODE_EXTENSIONS = frozenset(
155
+ {
156
+ ".py",
157
+ ".ts",
158
+ ".tsx",
159
+ ".js",
160
+ ".jsx",
161
+ ".go",
162
+ ".rs",
163
+ ".rb",
164
+ ".java",
165
+ ".kt",
166
+ ".swift",
167
+ ".cs",
168
+ ".cpp",
169
+ ".c",
170
+ ".h",
171
+ ".vue",
172
+ ".svelte",
173
+ ".astro",
174
+ }
175
+ )
176
+
177
+ # Extensions that are config-like (lower relevance unless prompt mentions them)
178
+ _CONFIG_EXTENSIONS = frozenset(
179
+ {
180
+ ".json",
181
+ ".toml",
182
+ ".yaml",
183
+ ".yml",
184
+ ".ini",
185
+ ".cfg",
186
+ ".env",
187
+ ".md",
188
+ ".txt",
189
+ ".lock",
190
+ ".sum",
191
+ }
192
+ )
193
+
194
+
195
+ # ---------------------------------------------------------------------------
196
+ # Public API
197
+ # ---------------------------------------------------------------------------
198
+
199
+
200
+ def select_files(
201
+ *,
202
+ prompt: str,
203
+ file_tree: list[tuple[str, int]],
204
+ ) -> list[str]:
205
+ """
206
+ Score and rank *file_tree* entries by relevance to *prompt*.
207
+
208
+ Parameters
209
+ ----------
210
+ prompt:
211
+ The user's natural-language task description.
212
+ file_tree:
213
+ Candidate files as (relative_posix_path, size_bytes) tuples.
214
+ Standard exclusions must already have been applied by the caller.
215
+
216
+ Returns
217
+ -------
218
+ list[str]
219
+ Relative POSIX paths of the top-scoring files, best first.
220
+ Files scoring zero are excluded. Result is capped at MAX_FILES.
221
+ """
222
+ tokens = _tokenize(prompt)
223
+ if not tokens:
224
+ # Degenerate prompt — return nothing rather than random files
225
+ return []
226
+
227
+ scored: list[tuple[float, str]] = []
228
+
229
+ for rel_path, size_bytes in file_tree:
230
+ score = _score(rel_path, tokens, size_bytes)
231
+ if score > 0:
232
+ scored.append((score, rel_path))
233
+
234
+ # Sort descending by score, then alphabetically for determinism on ties
235
+ scored.sort(key=lambda x: (-x[0], x[1]))
236
+
237
+ return [path for _, path in scored[:MAX_FILES]]
238
+
239
+
240
+ # ---------------------------------------------------------------------------
241
+ # Scoring
242
+ # ---------------------------------------------------------------------------
243
+
244
+
245
+ def _score(rel_path: str, tokens: set[str], size_bytes: int) -> float:
246
+ """Compute a relevance score for a single file path."""
247
+ path = Path(rel_path)
248
+ stem = path.stem.lower()
249
+ suffix = path.suffix.lower()
250
+ parts = [p.lower() for p in path.parts]
251
+ dirs = parts[:-1] # everything except the filename itself
252
+
253
+ score = 0.0
254
+
255
+ # ── 1. Token overlap ─────────────────────────────────────────────────────
256
+ # Split the stem by common separators (camelCase, kebab-case, snake_case)
257
+ stem_words = set(_split_identifier(stem))
258
+ dir_words: set[str] = set()
259
+ for d in dirs:
260
+ dir_words.update(_split_identifier(d))
261
+
262
+ # Direct hits in filename stem score highest
263
+ stem_hits = tokens & stem_words
264
+ score += len(stem_hits) * 3.0
265
+
266
+ # Partial substring matches in stem (e.g. "toast" in "useToast")
267
+ for token in tokens:
268
+ if len(token) >= 4 and token in stem and token not in stem_hits:
269
+ score += 1.0
270
+
271
+ # Hits in parent directory names score lower
272
+ dir_hits = tokens & dir_words
273
+ score += len(dir_hits) * 1.5
274
+
275
+ # No token overlap at all → zero score, file is irrelevant
276
+ if score == 0.0:
277
+ return 0.0
278
+
279
+ # ── 2. Directory bias ────────────────────────────────────────────────────
280
+ dir_set = set(dirs)
281
+ if dir_set & _SOURCE_DIRS:
282
+ score *= 1.4
283
+ if dir_set & _NOISE_DIRS:
284
+ score *= 0.4
285
+
286
+ # ── 3. Extension bias ────────────────────────────────────────────────────
287
+ if suffix in _CODE_EXTENSIONS:
288
+ score *= 1.2
289
+ elif suffix in _CONFIG_EXTENSIONS:
290
+ score *= 0.7
291
+
292
+ # ── 4. Penalise very large files ─────────────────────────────────────────
293
+ # Files over 100 KB are less likely to be what you want to change
294
+ if size_bytes > 100_000:
295
+ score *= 0.6
296
+
297
+ return score
298
+
299
+
300
+ # ---------------------------------------------------------------------------
301
+ # Text helpers
302
+ # ---------------------------------------------------------------------------
303
+
304
+
305
+ def _tokenize(text: str) -> set[str]:
306
+ """
307
+ Extract meaningful lowercase tokens from *text*.
308
+
309
+ - Splits on whitespace and punctuation
310
+ - Removes stopwords
311
+ - Keeps only tokens of 3+ characters
312
+ """
313
+ raw = re.findall(r"[a-zA-Z]+", text.lower())
314
+ return {w for w in raw if len(w) >= 3 and w not in _STOPWORDS}
315
+
316
+
317
+ def _split_identifier(name: str) -> list[str]:
318
+ """
319
+ Split a file/directory name into component words.
320
+
321
+ Handles: kebab-case, snake_case, camelCase, PascalCase.
322
+
323
+ Examples:
324
+ "LoginPage" → ["login", "page"]
325
+ "use-toast" → ["use", "toast"]
326
+ "auth_utils" → ["auth", "utils"]
327
+ "apiClient" → ["api", "client"]
328
+ """
329
+ # Insert space before uppercase letters following lowercase (camelCase)
330
+ spaced = re.sub(r"([a-z])([A-Z])", r"\1 \2", name)
331
+ # Split on non-alphanumeric separators
332
+ parts = re.split(r"[^a-zA-Z0-9]+", spaced)
333
+ return [p.lower() for p in parts if p]
@@ -0,0 +1,158 @@
1
+ """
2
+ ai/selector.py — Orchestrates AI-powered file selection for --prompt mode.
3
+
4
+ Responsibilities:
5
+ 1. Build a lean project map from the already-filtered file list
6
+ 2. Call the Gemini client (with heuristic fallback on rate limit)
7
+ 3. Return the selected subset as Path objects ready for packaging
8
+ 4. Generate the prompt.txt content to include in the ZIP
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from pathlib import Path
14
+
15
+ from contextzip.ai.gemini import select_files, GeminiRateLimitError
16
+ from contextzip.filters import ResolveResult
17
+
18
+ # Sentinels — tell the CLI which selection path was taken
19
+ USED_GEMINI = "gemini"
20
+ USED_HEURISTIC = "heuristic"
21
+
22
+
23
+ # ---------------------------------------------------------------------------
24
+ # Public API
25
+ # ---------------------------------------------------------------------------
26
+
27
+
28
+ def ai_select(
29
+ *,
30
+ resolved: ResolveResult,
31
+ project_dir: Path,
32
+ prompt: str,
33
+ ecosystem: str,
34
+ api_key: str,
35
+ ) -> tuple[list[Path], str, str]:
36
+ """
37
+ Select the minimum relevant files for *prompt* from *resolved.included*.
38
+
39
+ Returns
40
+ -------
41
+ (selected_paths, prompt_txt, method)
42
+ selected_paths : list[Path] of absolute paths chosen
43
+ prompt_txt : string content ready to write as prompt.txt in the ZIP
44
+ method : USED_GEMINI or USED_HEURISTIC
45
+ """
46
+ from contextzip.ai import heuristic as _heuristic
47
+
48
+ file_tree = _build_file_tree(resolved.included, project_dir)
49
+ method = USED_GEMINI
50
+
51
+ try:
52
+ selected_rel = select_files(
53
+ api_key=api_key,
54
+ prompt=prompt,
55
+ file_tree=file_tree,
56
+ ecosystem=ecosystem,
57
+ )
58
+ except GeminiRateLimitError:
59
+ # Confirmed HTTP 429 only — fall back to keyword heuristic
60
+ selected_rel = _heuristic.select_files(
61
+ prompt=prompt,
62
+ file_tree=file_tree,
63
+ )
64
+ method = USED_HEURISTIC
65
+ # All other GeminiErrors (bad key, network, unexpected status) propagate
66
+ # up to _run_ai_selection in cli.py which displays the error and exits.
67
+
68
+ # Map relative path strings back to absolute Path objects
69
+ rel_to_abs: dict[str, Path] = {
70
+ p.relative_to(project_dir).as_posix(): p for p in resolved.included
71
+ }
72
+ selected_paths = [rel_to_abs[rel] for rel in selected_rel if rel in rel_to_abs]
73
+
74
+ prompt_txt = _build_prompt_txt(prompt, selected_rel, ecosystem, method)
75
+
76
+ return selected_paths, prompt_txt, method
77
+
78
+
79
+ def build_prompt_only_txt(prompt: str, ecosystem: str) -> str:
80
+ """Build a minimal prompt.txt when no AI selection was performed."""
81
+ return _build_prompt_txt(prompt, [], ecosystem, USED_GEMINI)
82
+
83
+
84
+ # ---------------------------------------------------------------------------
85
+ # Internal helpers
86
+ # ---------------------------------------------------------------------------
87
+
88
+
89
+ def _build_file_tree(
90
+ included: list[Path],
91
+ project_dir: Path,
92
+ ) -> list[tuple[str, int]]:
93
+ """
94
+ Convert the included file list into (relative_posix_path, size_bytes) tuples.
95
+
96
+ Binary files and files that can't be stat'd are silently skipped —
97
+ the model can't reason about them anyway.
98
+ """
99
+ tree: list[tuple[str, int]] = []
100
+
101
+ for abs_path in included:
102
+ try:
103
+ rel = abs_path.relative_to(project_dir).as_posix()
104
+ size = abs_path.stat().st_size
105
+ except (ValueError, OSError):
106
+ continue
107
+
108
+ if _is_binary(abs_path):
109
+ continue
110
+
111
+ tree.append((rel, size))
112
+
113
+ return tree
114
+
115
+
116
+ def _build_prompt_txt(
117
+ prompt: str,
118
+ selected_rel: list[str],
119
+ ecosystem: str,
120
+ method: str,
121
+ ) -> str:
122
+ """
123
+ Build the prompt.txt to include inside the ZIP.
124
+
125
+ Any AI tool that receives the ZIP immediately sees the task description,
126
+ the framework, and exactly which files were selected and why.
127
+ """
128
+
129
+ selector_label = (
130
+ "contextzip AI (Gemini)"
131
+ if method == USED_GEMINI
132
+ else "contextzip (keyword heuristic — Gemini was rate limited)"
133
+ )
134
+
135
+ lines: list[str] = [
136
+ f"Task: {prompt}",
137
+ "",
138
+ f"Framework: {ecosystem}",
139
+ "",
140
+ ]
141
+
142
+ if selected_rel:
143
+ lines.append(f"Files selected by {selector_label}:")
144
+ for rel in selected_rel:
145
+ lines.append(f" - {rel}")
146
+ else:
147
+ lines.append("(No files selected)")
148
+
149
+ return "\n".join(lines) + "\n"
150
+
151
+
152
+ def _is_binary(path: Path, peek: int = 512) -> bool:
153
+ """Return True if the file appears to be binary (null bytes in first peek bytes)."""
154
+ try:
155
+ with path.open("rb") as fh:
156
+ return b"\x00" in fh.read(peek)
157
+ except OSError:
158
+ return False
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: contextzip
3
- Version: 0.3.2
3
+ Version: 0.3.3
4
4
  Summary: Intelligently package your codebase for AI tools
5
5
  Author-email: Deepesh <akadeepesh@gmail.com>
6
6
  License-Expression: MIT
@@ -148,6 +148,40 @@ contextzip --output ~/Desktop/project-context.zip
148
148
 
149
149
  ---
150
150
 
151
+ ## Python API
152
+
153
+ contextzip is also usable as a library. All CLI capabilities are available as plain Python functions — no Click, no Rich output, no `SystemExit`.
154
+
155
+ ```python
156
+ from contextzip import get_git_changes, get_files, create_zip
157
+
158
+ # Get changed files and use them directly
159
+ collection = get_git_changes()
160
+ for f in collection.files: # plain pathlib.Path objects
161
+ upload(f) # no zip required
162
+
163
+ # Or zip them and upload the archive
164
+ pkg = create_zip(collection, output="/tmp/changes.zip")
165
+ with open(pkg.zip_path, "rb") as f:
166
+ upload_to_s3(f)
167
+
168
+ # Full project scan with filters
169
+ collection = get_files(include=["src/"], exclude=["tests/"])
170
+ pkg = create_zip(collection, output="/tmp/upload.zip")
171
+ print(f"{pkg.file_count} files, {pkg.compressed_bytes} bytes")
172
+ ```
173
+
174
+ | Function | Description |
175
+ |---|---|
176
+ | `get_git_changes(path?)` | Modified, added, and untracked files from git |
177
+ | `get_files(path?, include?, exclude?)` | All project files after exclusion rules |
178
+ | `create_zip(collection, output?)` | Write a `FileCollection` to a ZIP archive |
179
+ | `detect_ecosystem(path?)` | Detect framework and confidence level |
180
+
181
+ All functions default `path` to `Path.cwd()`. Errors raise typed exceptions (`NotARepositoryError`, `GitNotFoundError`, `NoFilesError`, etc.) rather than exiting.
182
+
183
+ ---
184
+
151
185
  ## AI-powered file selection
152
186
 
153
187
  The `--prompt` flag lets you describe a task in plain English. contextzip scans your project, builds a lightweight file map, and asks Gemini to return the minimum set of files needed for that task — typically 2–5, never more than 10. The result is a tightly scoped ZIP with only what you'd actually open to make the change.
@@ -21,6 +21,10 @@ contextzip.egg-info/dependency_links.txt
21
21
  contextzip.egg-info/entry_points.txt
22
22
  contextzip.egg-info/requires.txt
23
23
  contextzip.egg-info/top_level.txt
24
+ contextzip/ai/__init__.py
25
+ contextzip/ai/gemini.py
26
+ contextzip/ai/heuristic.py
27
+ contextzip/ai/selector.py
24
28
  contextzip/rules/__init__.py
25
29
  contextzip/rules/base.py
26
30
  contextzip/rules/go.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "contextzip"
7
- version = "0.3.2"
7
+ version = "0.3.3"
8
8
  description = "Intelligently package your codebase for AI tools"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -43,4 +43,3 @@ contextzip = "contextzip.cli:main"
43
43
 
44
44
  [tool.setuptools.packages.find]
45
45
  where = ["."]
46
- include = ["contextzip", "contextzip.rules", "contextzip.rules.errors"]
File without changes
File without changes
File without changes
File without changes
File without changes