gemini-web 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,16 @@
1
+ Metadata-Version: 2.4
2
+ Name: gemini-web
3
+ Version: 1.0.0
4
+ Summary: Gemini 網頁版自動化 — 文字對話、圖片生成、Google GenAI API 相容
5
+ Requires-Python: >=3.11
6
+ Requires-Dist: fastapi>=0.115
7
+ Requires-Dist: httpx>=0.27
8
+ Requires-Dist: numpy>=1.26
9
+ Requires-Dist: pillow>=10.0
10
+ Requires-Dist: playwright>=1.49
11
+ Requires-Dist: python-dotenv>=1.0
12
+ Requires-Dist: uvicorn[standard]>=0.30
13
+ Provides-Extra: dev
14
+ Requires-Dist: httpx>=0.27; extra == 'dev'
15
+ Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
16
+ Requires-Dist: pytest>=8.0; extra == 'dev'
@@ -0,0 +1,22 @@
1
+ src/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ src/browser.py,sha256=hCA_mqdf4cL254v_ZdEiDe-UnPSxBKOXOIc4kefnuAc,5752
3
+ src/cli.py,sha256=_UmjG6CitZARPVwh3l-04Murh55XAAkutDiqEAGEyhY,8762
4
+ src/config.py,sha256=Hxem6_tjzcySkKnufDLwqXrzno2l52qWOtBrkkjxZio,1887
5
+ src/gemini.py,sha256=67Cjkxwr36tY3bRlvEemIKisgwxIU-1I0oSjSzfbbas,16181
6
+ src/main.py,sha256=5ByZVT9pGa6Ca0xNgcjY52AvJ_iQ5fTW_2L9u0LhXt8,8908
7
+ src/queue.py,sha256=sjiPWKrq2PiSHfteVd4t_bOM7o_c3cbbJD9DjYs7YgM,2345
8
+ src/selectors.py,sha256=jiDv5RgYoX1GLD-JskvKyhjojcRpYiwaBdITBjnA1f4,1203
9
+ src/watermark.py,sha256=0y3mEcqR997jcPayGC-qV_J2EfFeiI6ZkJcqqhCmhzE,3719
10
+ src/assets/bg_48.png,sha256=SvyZr-DvEI1nrMRb9NxdqGfdt5O-vInJJDuxIc5_D1c,1677
11
+ src/assets/bg_96.png,sha256=PibyIzoSpYKaysF02N8fPbQOB_7wTs3Q4DVzIVQHeRE,8165
12
+ src/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
13
+ src/commands/chat.md,sha256=0NuHjRV8Fk1uCc8GWh05G8QnrpqqXv9tvhN700wRlto,748
14
+ src/commands/chat.toml,sha256=TKA3VC_15-xbnz0MmsLQORoAQZUnPLe6jUW7-YooUpw,512
15
+ src/commands/gemini-web.md,sha256=7V0OwobhLNhJNSn_the3R4B7pbT1goPWbRg24j0gXws,1708
16
+ src/commands/gemini-web.toml,sha256=OwRQrBUspg-zkBsn-EGKE_9wefb0Zq_aEbKkg54dQpA,1256
17
+ src/commands/generate.md,sha256=U8WQ8XauJ1WTq7lsuyOHCSR-u71j6f-ULgSnYlx8t30,1045
18
+ src/commands/generate.toml,sha256=qGFtFGb23OoPiVJc7xWvC7FCuS_mqhEr-4I6m3_dN9o,726
19
+ gemini_web-1.0.0.dist-info/METADATA,sha256=bYhoAkSP8B6ls-MassdMDWf7iGgeBAffV0wJmnuDVMA,546
20
+ gemini_web-1.0.0.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
21
+ gemini_web-1.0.0.dist-info/entry_points.txt,sha256=iZSUMvou4z35vyx0S0mzR6Ek-2qcz5PLDo-J0Jmwiqc,72
22
+ gemini_web-1.0.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.29.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ gemini-image = src.cli:main
3
+ gemini-web = src.cli:main
src/__init__.py ADDED
File without changes
src/assets/bg_48.png ADDED
Binary file
src/assets/bg_96.png ADDED
Binary file
src/browser.py ADDED
@@ -0,0 +1,171 @@
1
+ """Playwright 瀏覽器管理 — 啟動、stealth、session 持久化"""
2
+ import asyncio
3
+ import logging
4
+ from pathlib import Path
5
+
6
+ from playwright.async_api import async_playwright, BrowserContext, Page
7
+
8
+ from .config import settings
9
+
10
+ logger = logging.getLogger(__name__)
11
+
12
+ # Stealth 注入腳本 — 參考 Project Golem BrowserLauncher
13
+ _STEALTH_SCRIPT = """
14
+ () => {
15
+ // 隱藏 webdriver 標記
16
+ Object.defineProperty(navigator, 'webdriver', { get: () => false });
17
+
18
+ // 偽裝 languages
19
+ Object.defineProperty(navigator, 'languages', {
20
+ get: () => LANGUAGES_PLACEHOLDER,
21
+ });
22
+
23
+ // 偽裝 platform
24
+ Object.defineProperty(navigator, 'platform', {
25
+ get: () => 'Linux x86_64',
26
+ });
27
+
28
+ // 偽裝 plugins(空陣列會被偵測)
29
+ Object.defineProperty(navigator, 'plugins', {
30
+ get: () => [1, 2, 3, 4, 5],
31
+ });
32
+
33
+ // 偽裝 WebGL vendor/renderer
34
+ const getParameter = WebGLRenderingContext.prototype.getParameter;
35
+ WebGLRenderingContext.prototype.getParameter = function(param) {
36
+ if (param === 37445) return 'Intel Inc.';
37
+ if (param === 37446) return 'Intel Iris OpenGL Engine';
38
+ return getParameter.call(this, param);
39
+ };
40
+ }
41
+ """
42
+
43
+
44
+ class BrowserManager:
45
+ """管理單一 Playwright 瀏覽器實例"""
46
+
47
+ def __init__(self, headless: bool | None = None) -> None:
48
+ self._playwright = None
49
+ self._context: BrowserContext | None = None
50
+ self._page: Page | None = None
51
+ self._heartbeat_task: asyncio.Task | None = None
52
+ self._headless_override = headless # None = 用 settings
53
+
54
+ @property
55
+ def page(self) -> Page | None:
56
+ return self._page
57
+
58
+ async def start(self) -> None:
59
+ """啟動瀏覽器,導航到 Gemini"""
60
+ profile_path = str(Path(settings.profile_dir).resolve())
61
+ Path(profile_path).mkdir(parents=True, exist_ok=True)
62
+
63
+ languages = settings.stealth_language.split(",")
64
+ stealth_js = _STEALTH_SCRIPT.replace(
65
+ "LANGUAGES_PLACEHOLDER", str(languages)
66
+ )
67
+
68
+ self._playwright = await async_playwright().start()
69
+ self._context = await self._playwright.chromium.launch_persistent_context(
70
+ profile_path,
71
+ headless=self._headless_override if self._headless_override is not None else settings.headless,
72
+ locale=languages[0] if languages else "zh-TW",
73
+ timezone_id=settings.stealth_timezone,
74
+ user_agent=(
75
+ "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
76
+ "(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
77
+ ),
78
+ viewport={"width": 1440, "height": 900},
79
+ args=[
80
+ "--disable-blink-features=AutomationControlled",
81
+ "--no-sandbox",
82
+ ],
83
+ )
84
+
85
+ # 注入 stealth 腳本
86
+ await self._context.add_init_script(stealth_js)
87
+
88
+ # 取得或建立頁面
89
+ pages = self._context.pages
90
+ self._page = pages[0] if pages else await self._context.new_page()
91
+
92
+ # 擋掉不必要的資源
93
+ await self._page.route(
94
+ "**/*",
95
+ lambda route: (
96
+ route.abort()
97
+ if route.request.resource_type in ("font", "stylesheet")
98
+ and "gemini" not in route.request.url
99
+ else route.continue_()
100
+ ),
101
+ )
102
+
103
+ # 導航到 Gemini
104
+ await self._page.goto(settings.gemini_url, wait_until="domcontentloaded")
105
+ logger.info("瀏覽器已啟動,導航至 %s", settings.gemini_url)
106
+
107
+ # 啟動心跳
108
+ self._heartbeat_task = asyncio.create_task(self._heartbeat_loop())
109
+
110
+ async def stop(self) -> None:
111
+ """關閉瀏覽器"""
112
+ if self._heartbeat_task:
113
+ self._heartbeat_task.cancel()
114
+ try:
115
+ await self._heartbeat_task
116
+ except asyncio.CancelledError:
117
+ pass
118
+
119
+ if self._context:
120
+ await self._context.close()
121
+ if self._playwright:
122
+ await self._playwright.stop()
123
+ logger.info("瀏覽器已關閉")
124
+
125
+ async def is_alive(self) -> bool:
126
+ """檢查瀏覽器頁面是否還活著"""
127
+ if not self._page:
128
+ return False
129
+ try:
130
+ await self._page.evaluate("() => document.title")
131
+ return True
132
+ except Exception:
133
+ return False
134
+
135
+ async def is_logged_in(self, wait: bool = False) -> bool:
136
+ """檢查是否已登入 Google(偵測輸入框是否存在)
137
+
138
+ Args:
139
+ wait: 是否等待頁面載入完成(首次檢查用)
140
+ """
141
+ if not self._page:
142
+ return False
143
+ try:
144
+ from .selectors import SELECTORS
145
+ if wait:
146
+ # Gemini 是 Angular SPA,需要等 JS 渲染完成
147
+ el = await self._page.wait_for_selector(
148
+ SELECTORS["input"], state="visible", timeout=15_000
149
+ )
150
+ return el is not None
151
+ else:
152
+ el = await self._page.query_selector(SELECTORS["input"])
153
+ return el is not None
154
+ except Exception:
155
+ return False
156
+
157
+ async def _heartbeat_loop(self) -> None:
158
+ """定時心跳檢查"""
159
+ while True:
160
+ await asyncio.sleep(settings.heartbeat_interval)
161
+ alive = await self.is_alive()
162
+ if not alive:
163
+ logger.warning("心跳檢查失敗:瀏覽器頁面無回應")
164
+ else:
165
+ logged_in = await self.is_logged_in()
166
+ if not logged_in:
167
+ logger.warning("心跳檢查:Google 登入狀態可能已過期")
168
+
169
+
170
+ # 全域單例
171
+ browser_manager = BrowserManager()
src/cli.py ADDED
@@ -0,0 +1,260 @@
1
+ """Gemini Image CLI — 命令列生圖工具"""
2
+ import argparse
3
+ import asyncio
4
+ import base64
5
+ import logging
6
+ import shutil
7
+ import subprocess
8
+ import sys
9
+ from pathlib import Path
10
+
11
+
12
+
13
+ def _get_commands_dir() -> Path:
14
+ """取得 package 內建 commands 目錄路徑"""
15
+ return Path(__file__).parent / "commands"
16
+
17
+
18
+ def _install_commands():
19
+ """偵測 Claude Code / Gemini CLI 並安裝對應 commands"""
20
+ commands_src = _get_commands_dir()
21
+ if not commands_src.exists():
22
+ print("警告:找不到內建 commands 目錄,跳過 AI Agent 整合")
23
+ return
24
+
25
+ installed = []
26
+
27
+ # Claude Code: 複製 .md 到 ~/.claude/commands/gemini-web/
28
+ claude_dir = Path.home() / ".claude"
29
+ if claude_dir.exists():
30
+ dest = claude_dir / "commands" / "gemini-web"
31
+ dest.mkdir(parents=True, exist_ok=True)
32
+ for f in commands_src.glob("*.md"):
33
+ shutil.copy2(f, dest / f.name)
34
+ installed.append(f"Claude Code → {dest}")
35
+
36
+ # Gemini CLI: 複製 .toml 到 ~/.gemini/commands/gemini-web/
37
+ gemini_dir = Path.home() / ".gemini"
38
+ if gemini_dir.exists():
39
+ dest = gemini_dir / "commands" / "gemini-web"
40
+ dest.mkdir(parents=True, exist_ok=True)
41
+ for f in commands_src.glob("*.toml"):
42
+ shutil.copy2(f, dest / f.name)
43
+ installed.append(f"Gemini CLI → {dest}")
44
+
45
+ if installed:
46
+ print("\nAI Agent commands 已安裝:")
47
+ for line in installed:
48
+ print(f" {line}")
49
+ print("可用指令:/gemini-web, /generate")
50
+ else:
51
+ print("\n未偵測到 Claude Code 或 Gemini CLI,跳過 commands 安裝")
52
+ print("如需手動安裝,請參考:https://github.com/yazelin/gemini-web#ai-agent-整合")
53
+
54
+
55
+ def _setup_logging(verbose: bool = False):
56
+ logging.basicConfig(
57
+ level=logging.DEBUG if verbose else logging.INFO,
58
+ format="%(asctime)s [%(name)s] %(levelname)s: %(message)s",
59
+ )
60
+
61
+
62
+ async def _do_login():
63
+ """開啟瀏覽器讓用戶手動登入 Google"""
64
+ from .browser import BrowserManager
65
+ bm = BrowserManager(headless=False) # login 一律 headed
66
+ await bm.start()
67
+
68
+ print("\n瀏覽器已開啟,請登入 Google 帳號。")
69
+ print("登入完成後按 Enter 關閉瀏覽器...")
70
+ await asyncio.get_event_loop().run_in_executor(None, input)
71
+
72
+ await bm.stop()
73
+ print("登入狀態已儲存。之後可以用 headless 模式生圖。")
74
+
75
+
76
+ async def _do_chat(prompt: str, verbose: bool):
77
+ """文字對話"""
78
+ from .browser import BrowserManager
79
+ from .gemini import chat, new_chat
80
+ from .config import settings
81
+
82
+ bm = BrowserManager(headless=True)
83
+ await bm.start()
84
+
85
+ try:
86
+ page = bm.page
87
+ if not page:
88
+ print("錯誤:瀏覽器未啟動", file=sys.stderr)
89
+ sys.exit(1)
90
+
91
+ logged_in = await bm.is_logged_in(wait=True)
92
+ if not logged_in:
93
+ print("錯誤:尚未登入 Google,請先執行 `gemini-web login`", file=sys.stderr)
94
+ sys.exit(1)
95
+
96
+ print(f"提問中... prompt: {prompt[:60]}{'...' if len(prompt) > 60 else ''}")
97
+
98
+ result = await chat(page, prompt, timeout=settings.default_timeout)
99
+ await new_chat(page)
100
+
101
+ if not result.get("success"):
102
+ error = result.get("error", "unknown")
103
+ message = result.get("message", "")
104
+ print(f"失敗 [{error}]:{message}", file=sys.stderr)
105
+ sys.exit(1)
106
+
107
+ print(result.get("text", ""))
108
+ print(f"\n({result.get('elapsed_seconds', 0)}秒)", file=sys.stderr)
109
+
110
+ finally:
111
+ await bm.stop()
112
+
113
+
114
+ async def _do_generate(prompt: str, output: str, no_watermark: bool, verbose: bool):
115
+ """生成圖片"""
116
+ from .browser import BrowserManager
117
+ from .gemini import generate_image, new_chat
118
+ from .config import settings
119
+
120
+ bm = BrowserManager(headless=True) # generate 一律 headless
121
+ await bm.start()
122
+
123
+ try:
124
+ page = bm.page
125
+ if not page:
126
+ print("錯誤:瀏覽器未啟動", file=sys.stderr)
127
+ sys.exit(1)
128
+
129
+ # 檢查登入狀態(等待頁面完全載入)
130
+ logged_in = await bm.is_logged_in(wait=True)
131
+ if not logged_in:
132
+ print("錯誤:尚未登入 Google,請先執行 `gemini-web login`", file=sys.stderr)
133
+ sys.exit(1)
134
+
135
+ print(f"生成中... prompt: {prompt[:60]}{'...' if len(prompt) > 60 else ''}")
136
+
137
+ result = await generate_image(page, prompt, timeout=settings.default_timeout)
138
+ await new_chat(page)
139
+
140
+ if not result.get("success"):
141
+ error = result.get("error", "unknown")
142
+ message = result.get("message", "")
143
+ print(f"失敗 [{error}]:{message}", file=sys.stderr)
144
+ sys.exit(1)
145
+
146
+ images = result.get("images", [])
147
+ if not images:
148
+ print("失敗:無圖片資料", file=sys.stderr)
149
+ sys.exit(1)
150
+
151
+ # 儲存圖片
152
+ output_path = Path(output)
153
+ for i, img_data in enumerate(images):
154
+ if "," in img_data:
155
+ header, b64 = img_data.split(",", 1)
156
+ else:
157
+ b64 = img_data
158
+
159
+ raw = base64.b64decode(b64)
160
+
161
+ if len(images) == 1:
162
+ save_path = output_path
163
+ else:
164
+ stem = output_path.stem
165
+ ext = output_path.suffix
166
+ save_path = output_path.parent / f"{stem}_{i}{ext}"
167
+
168
+ save_path.parent.mkdir(parents=True, exist_ok=True)
169
+ save_path.write_bytes(raw)
170
+
171
+ # 去水印
172
+ if no_watermark:
173
+ from .watermark import remove_watermark
174
+ remove_watermark(str(save_path))
175
+ print(f"已存檔(去水印):{save_path}")
176
+ else:
177
+ print(f"已存檔:{save_path}")
178
+
179
+ print(f"完成({result.get('elapsed_seconds', 0)}秒)")
180
+
181
+ finally:
182
+ await bm.stop()
183
+
184
+
185
+ def main():
186
+ parser = argparse.ArgumentParser(
187
+ prog="gemini-web",
188
+ description="Gemini Image — AI 圖片生成 CLI 工具",
189
+ )
190
+ sub = parser.add_subparsers(dest="command", help="可用指令")
191
+
192
+ # install
193
+ sub.add_parser("install", help="安裝 Chromium 瀏覽器(首次使用前必須執行)")
194
+
195
+ # login
196
+ login_parser = sub.add_parser("login", help="開啟瀏覽器登入 Google")
197
+
198
+ # chat
199
+ chat_parser = sub.add_parser("chat", help="文字對話")
200
+ chat_parser.add_argument("prompt", help="提問內容")
201
+ chat_parser.add_argument("-v", "--verbose", action="store_true", help="顯示詳細 log")
202
+
203
+ # generate
204
+ gen_parser = sub.add_parser("generate", help="生成圖片")
205
+ gen_parser.add_argument("prompt", help="圖片描述(建議英文)")
206
+ gen_parser.add_argument("-o", "--output", default="output.png", help="輸出檔案路徑(預設 output.png)")
207
+ gen_parser.add_argument("--no-watermark", action="store_true", help="自動移除可見水印")
208
+ gen_parser.add_argument("-v", "--verbose", action="store_true", help="顯示詳細 log")
209
+
210
+ # health
211
+ health_parser = sub.add_parser("health", help="檢查服務狀態(API 模式)")
212
+
213
+ # serve
214
+ serve_parser = sub.add_parser("serve", help="啟動 API 服務")
215
+ serve_parser.add_argument("--host", default="0.0.0.0")
216
+ serve_parser.add_argument("--port", type=int, default=8070)
217
+
218
+ args = parser.parse_args()
219
+
220
+ if args.command == "install":
221
+ # 1. 安裝 Chromium
222
+ subprocess.run([sys.executable, "-m", "playwright", "install", "chromium"])
223
+
224
+ # 2. 安裝 AI Agent commands
225
+ _install_commands()
226
+
227
+ print("\n下一步:執行 `gemini-web login` 登入 Google")
228
+
229
+ elif args.command == "login":
230
+ asyncio.run(_do_login())
231
+
232
+ elif args.command == "chat":
233
+ _setup_logging(getattr(args, "verbose", False))
234
+ asyncio.run(_do_chat(args.prompt, getattr(args, "verbose", False)))
235
+
236
+ elif args.command == "generate":
237
+ _setup_logging(getattr(args, "verbose", False))
238
+ asyncio.run(_do_generate(args.prompt, args.output, args.no_watermark, getattr(args, "verbose", False)))
239
+
240
+ elif args.command == "serve":
241
+ import uvicorn
242
+ uvicorn.run("src.main:app", host=args.host, port=args.port)
243
+
244
+ elif args.command == "health":
245
+ import httpx
246
+ from .config import settings
247
+ try:
248
+ resp = httpx.get(f"http://localhost:{settings.port}/api/health", timeout=5)
249
+ import json
250
+ print(json.dumps(resp.json(), indent=2, ensure_ascii=False))
251
+ except Exception as e:
252
+ print(f"服務未啟動或無法連線:{e}", file=sys.stderr)
253
+ sys.exit(1)
254
+
255
+ else:
256
+ parser.print_help()
257
+
258
+
259
+ if __name__ == "__main__":
260
+ main()
File without changes
src/commands/chat.md ADDED
@@ -0,0 +1,38 @@
1
+ ---
2
+ description: Chat with Gemini via gemini-web (text input, text output)
3
+ argument-hint: <prompt>
4
+ ---
5
+
6
+ You are a command handler for gemini-web chat. Send the user's prompt to Gemini and return the text response.
7
+
8
+ ## Usage
9
+
10
+ ```bash
11
+ gemini-web chat "<prompt>"
12
+ ```
13
+
14
+ Or via HTTP API:
15
+
16
+ ```bash
17
+ curl -X POST http://localhost:8070/api/chat \
18
+ -H "Content-Type: application/json" \
19
+ -d '{"prompt": "<prompt>"}'
20
+ ```
21
+
22
+ ## Rules
23
+
24
+ 1. Pass the user's prompt directly to the command
25
+ 2. The response is plain text from Gemini
26
+ 3. Typical response time: 5-30 seconds
27
+
28
+ ## Examples
29
+
30
+ ```
31
+ /chat What is quantum computing?
32
+ → gemini-web chat "What is quantum computing?"
33
+
34
+ /chat 解釋量子力學
35
+ → gemini-web chat "解釋量子力學"
36
+ ```
37
+
38
+ User input: $ARGUMENTS
src/commands/chat.toml ADDED
@@ -0,0 +1,18 @@
1
+ description = "使用 Gemini 文字對話(文字輸入,文字回覆)"
2
+
3
+ prompt = """
4
+ Send the user's prompt to Gemini using gemini-web chat command and return the text response.
5
+
6
+ Run this shell command:
7
+ gemini-web chat "<prompt>"
8
+
9
+ Or via HTTP API:
10
+ curl -X POST http://localhost:8070/api/chat -H "Content-Type: application/json" -d '{"prompt": "<prompt>"}'
11
+
12
+ Rules:
13
+ - Pass the user's prompt directly
14
+ - The response is plain text from Gemini
15
+ - Typical response time: 5-30 seconds
16
+
17
+ User input: {{args}}
18
+ """
@@ -0,0 +1,37 @@
1
+ ---
2
+ description: Generate and manipulate images with Gemini using natural language prompts
3
+ argument-hint: <natural language request>
4
+ ---
5
+
6
+ You are a command handler for gemini-web. Analyze the user's natural language request and generate an image.
7
+
8
+ ## Steps
9
+
10
+ 1. **Understand the user's intent** from their natural language input
11
+ 2. **Expand into a detailed English prompt** describing: subject, style, composition, colors, mood
12
+ 3. **If Chinese text is needed in the image**, use quotes: `with text "歡迎光臨"`
13
+ 4. **Run the command**:
14
+
15
+ ```bash
16
+ gemini-web generate "<detailed_english_prompt>" -o <output_path> --no-watermark
17
+ ```
18
+
19
+ 5. **Inform the user** the image has been generated (takes 30-120 seconds)
20
+
21
+ ## Examples
22
+
23
+ | User says | You send |
24
+ |-----------|----------|
25
+ | 畫一隻貓 | `gemini-web generate "A cute fluffy orange tabby cat sitting on a windowsill, warm afternoon sunlight, soft watercolor style, gentle expression" -o cat.png --no-watermark` |
26
+ | 做開幕海報 | `gemini-web generate "A modern grand opening poster with text '盛大開幕', red and gold color scheme, confetti, professional design" -o poster.png --no-watermark` |
27
+ | 畫公司 logo | `gemini-web generate "A minimalist corporate logo design, clean lines, modern typography, professional business style" -o logo.png --no-watermark` |
28
+
29
+ ## Important
30
+
31
+ - **Always use `--no-watermark`**
32
+ - **Always expand the prompt** — never forward user's raw text directly
33
+ - **Use English prompts** for best results
34
+ - If the command is not found, tell the user to run: `uv tool install gemini-web && gemini-web install`
35
+ - If login is needed, tell the user to run: `gemini-web login` (requires manual browser login)
36
+
37
+ User request: $ARGUMENTS
@@ -0,0 +1,25 @@
1
+ description = "使用 Gemini 生成圖片(自然語言輸入,自動擴寫 prompt)"
2
+
3
+ prompt = """
4
+ Analyze the user's natural language request and generate an image using gemini-web CLI.
5
+
6
+ Steps:
7
+ 1. Understand the user's intent from their input
8
+ 2. Expand into a detailed English prompt describing: subject, style, composition, colors, mood
9
+ 3. If Chinese text is needed in the image, use quotes: with text "歡迎光臨"
10
+ 4. Run the shell command:
11
+ gemini-web generate "<detailed_english_prompt>" -o <output_path> --no-watermark
12
+ 5. Inform the user the image has been generated (takes 30-120 seconds)
13
+
14
+ Examples:
15
+ - User says "畫一隻貓" → run: gemini-web generate "A cute fluffy orange tabby cat sitting on a windowsill, warm afternoon sunlight, soft watercolor style" -o cat.png --no-watermark
16
+ - User says "做開幕海報" → run: gemini-web generate "A modern grand opening poster with text '盛大開幕', red and gold, confetti, professional design" -o poster.png --no-watermark
17
+
18
+ Important:
19
+ - Always use --no-watermark
20
+ - Always expand the prompt, never forward user's raw text directly
21
+ - Use English prompts for best results
22
+ - If command not found, tell user to run: uv tool install gemini-web && gemini-web install
23
+
24
+ User request: {{args}}
25
+ """
@@ -0,0 +1,36 @@
1
+ ---
2
+ description: Generate an image with gemini-web using a detailed prompt
3
+ argument-hint: <prompt> [-o output.png] [--no-watermark]
4
+ ---
5
+
6
+ You are a command parser for the gemini-web generate command.
7
+
8
+ Parse the user input and run:
9
+
10
+ ```bash
11
+ gemini-web generate "<prompt>" -o <output_path> --no-watermark
12
+ ```
13
+
14
+ ## Options
15
+
16
+ - `-o`, `--output`: Output file path (default: `output.png`)
17
+ - `--no-watermark`: Remove Gemini watermark (always recommended)
18
+
19
+ ## Rules
20
+
21
+ 1. Extract the prompt text (everything before options)
22
+ 2. If no `-o` is specified, use `output.png`
23
+ 3. Always add `--no-watermark` unless user explicitly says not to
24
+ 4. Run the command via Bash
25
+
26
+ ## Examples
27
+
28
+ ```
29
+ /generate A cute cat sitting on a windowsill, warm sunlight, watercolor style
30
+ → gemini-web generate "A cute cat sitting on a windowsill, warm sunlight, watercolor style" -o output.png --no-watermark
31
+
32
+ /generate A poster with text "歡迎光臨" -o poster.png
33
+ → gemini-web generate "A poster with text '歡迎光臨'" -o poster.png --no-watermark
34
+ ```
35
+
36
+ User input: $ARGUMENTS
@@ -0,0 +1,22 @@
1
+ description = "使用 gemini-web 生成圖片(直接指定 prompt)"
2
+
3
+ prompt = """
4
+ Run the gemini-web generate command with the user's prompt.
5
+
6
+ Parse the input and execute:
7
+ gemini-web generate "<prompt>" -o <output_path> --no-watermark
8
+
9
+ Rules:
10
+ - Extract the prompt text (everything before options like -o)
11
+ - If no -o is specified, use output.png
12
+ - Always add --no-watermark unless user explicitly says not to
13
+
14
+ Examples:
15
+ - Input: A cute cat, watercolor style
16
+ Run: gemini-web generate "A cute cat, watercolor style" -o output.png --no-watermark
17
+
18
+ - Input: A poster with text "歡迎光臨" -o poster.png
19
+ Run: gemini-web generate "A poster with text '歡迎光臨'" -o poster.png --no-watermark
20
+
21
+ User input: {{args}}
22
+ """