sqlbuddy-nwi 0.3.2__tar.gz → 0.3.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. sqlbuddy_nwi-0.3.3/PKG-INFO +193 -0
  2. sqlbuddy_nwi-0.3.3/README.md +164 -0
  3. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/_version.py +1 -1
  4. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/db_query.py +154 -30
  5. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/pyproject.toml +1 -1
  6. sqlbuddy_nwi-0.3.3/sqlbuddy_nwi.egg-info/PKG-INFO +193 -0
  7. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/sqlbuddy_nwi.egg-info/SOURCES.txt +4 -1
  8. sqlbuddy_nwi-0.3.3/tests/test_get_info.py +98 -0
  9. sqlbuddy_nwi-0.3.3/tests/test_parse_create_table.py +106 -0
  10. sqlbuddy_nwi-0.3.3/tests/test_sanitize_search.py +51 -0
  11. sqlbuddy_nwi-0.3.2/PKG-INFO +0 -775
  12. sqlbuddy_nwi-0.3.2/README.md +0 -746
  13. sqlbuddy_nwi-0.3.2/sqlbuddy_nwi.egg-info/PKG-INFO +0 -775
  14. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/LICENSE +0 -0
  15. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/cli.py +0 -0
  16. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/connections/__init__.py +0 -0
  17. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/connections/config_util.py +0 -0
  18. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/connections/manager.py +0 -0
  19. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/connections/readonly.py +0 -0
  20. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/connections/runtime.py +0 -0
  21. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/connections/wrappers.py +0 -0
  22. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/db_security.py +0 -0
  23. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/init.py +0 -0
  24. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/server.py +0 -0
  25. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/setup.cfg +0 -0
  26. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/sqlbuddy_nwi.egg-info/dependency_links.txt +0 -0
  27. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/sqlbuddy_nwi.egg-info/entry_points.txt +0 -0
  28. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/sqlbuddy_nwi.egg-info/requires.txt +0 -0
  29. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/sqlbuddy_nwi.egg-info/top_level.txt +0 -0
  30. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/test_vendor.py +0 -0
  31. {sqlbuddy_nwi-0.3.2 → sqlbuddy_nwi-0.3.3}/tunnel.py +0 -0
@@ -0,0 +1,193 @@
1
+ Metadata-Version: 2.4
2
+ Name: sqlbuddy-nwi
3
+ Version: 0.3.3
4
+ Summary: MCP server for read-only database access (MySQL, ClickHouse)
5
+ Author: ying.yuxiang
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/yyx462/sql-buddy
8
+ Project-URL: Repository, https://github.com/yyx462/sql-buddy
9
+ Project-URL: Issues, https://github.com/yyx462/sql-buddy/issues
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Topic :: Database
19
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
20
+ Requires-Python: >=3.11
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Requires-Dist: mcp<2,>=1.0.0
24
+ Requires-Dist: pymysql>=1.1.0
25
+ Requires-Dist: clickhouse-driver>=0.2.6
26
+ Requires-Dist: pandas>=2.0.0
27
+ Requires-Dist: platformdirs>=4.0.0
28
+ Dynamic: license-file
29
+
30
+ # SQL Buddy
31
+
32
+ A read-only **MCP server** that lets your AI agent (Claude Code, and any
33
+ MCP-compatible client) safely query your **MySQL** and **ClickHouse** databases.
34
+
35
+ Your agent asks in plain language; SQL Buddy runs the SQL, enforces **read-only**
36
+ (no `INSERT`/`UPDATE`/`DROP` ever reaches your data), adds safety `LIMIT`s, and
37
+ caches results. You stay in control — the agent can look, never touch.
38
+
39
+ ```
40
+ Your AI agent ──MCP──► SQL Buddy (runs locally) ──► your databases
41
+ • opens its own SSH tunnel through a bastion
42
+ • read-only guard · auto LIMIT · schema lookup
43
+ ```
44
+
45
+ SQL Buddy runs **locally** (a small Python program, not a cloud service). It
46
+ **opens the SSH tunnel itself** from a bastion + key + port mapping in `db.ini`
47
+ — no separate `ssh -N` terminal, and your credentials never leave your machine.
48
+
49
+ ---
50
+
51
+ ## Prerequisites
52
+
53
+ - **Python 3.11+** — `python3 --version`
54
+ - **[uv](https://docs.astral.sh/uv/)** (recommended) or plain `pip`.
55
+ - **SSH access** to a bastion/jump-host that can reach your databases: its
56
+ `user@host` + ssh `port` (often a high port like `22222`, **not** 22), a
57
+ **private key file**, and each DB's `host:port` **as seen from the bastion**.
58
+ - **Read-only DB credentials** — a user that can `SELECT` but not write.
59
+ - An **MCP-compatible client** (e.g. Claude Code, **Trae**).
60
+ - **Windows only (Auto tunnels):** [OpenSSH Client](https://learn.microsoft.com/windows-server/administration/openssh/openssh_install_firstuse).
61
+ (Skip if you run the tunnel yourself in another terminal.)
62
+
63
+ > Deep notes on tunnels, Windows/Trae, and every `db.ini` field live in
64
+ > [`docs/setup.md`](docs/setup.md) and [`docs/windows-trae.md`](docs/windows-trae.md).
65
+
66
+ ---
67
+
68
+ ## Quickstart
69
+
70
+ ```bash
71
+ git clone <this-repo-url> sql-buddy && cd sql-buddy
72
+ uv run server.py # installs deps into an isolated .venv, then serves (Ctrl+C to stop)
73
+ uv run sql-buddy init # interactive wizard → writes connections/config/db.ini
74
+ uv run sql-buddy doctor --connect # green-checks the venv, db.ini, tunnels, MCP client config
75
+ uv run sql-buddy mcp add # registers sql-buddy with Claude Code (writes .mcp.json)
76
+ ```
77
+
78
+ Restart your client, then ask your agent *"list the tables in the `ck`
79
+ connection"* — if it answers, you're live. 🎉
80
+
81
+ > **Bare `sql-buddy` (no command) runs the MCP server** — that's what your AI
82
+ > client invokes, so configs hold no fragile absolute `.venv/bin/python` path.
83
+ > Other commands: `version`, `doctor [--connect]`, `mcp print|add`, `update-check`, `init`.
84
+ > On **Windows**, see [`docs/windows-trae.md`](docs/windows-trae.md) and use
85
+ > `sql-buddy mcp add --client trae`.
86
+
87
+ ---
88
+
89
+ ## Configure connections (`db.ini`)
90
+
91
+ Run the wizard (easiest) or copy the template:
92
+
93
+ ```bash
94
+ uv run sql-buddy init
95
+ # or by hand:
96
+ cp connections/config/db.ini.example connections/config/db.ini
97
+ $EDITOR connections/config/db.ini
98
+ ```
99
+
100
+ Each database is one **section**; the `[bracketed]` name becomes the
101
+ **connection name** your agent uses (e.g. `ck`, `mysql`, `prod-read`).
102
+
103
+ ```ini
104
+ # db.ini — NEVER commit this file (gitignored). It holds live credentials.
105
+
106
+ [ssh:prod]
107
+ host = bastion.example.com
108
+ port = 22
109
+ user = your_ssh_user
110
+ key = ~/.ssh/id_ed25519 # path to your private key (~ expanded)
111
+
112
+ [ck]
113
+ type = clickhouse
114
+ tunnel = ssh:prod # self-managed tunnel through the bastion above
115
+ remote_host = 10.0.0.21 # DB host AS SEEN FROM the bastion
116
+ remote_port = 9000 # DB port AS SEEN FROM the bastion
117
+ local_port = 10021 # localhost port the driver connects to
118
+ user = your_readonly_user
119
+ password = your_password
120
+ database = default
121
+
122
+ [local-db] # direct (non-tunneled) connection also works
123
+ type = mysql
124
+ host = localhost
125
+ port = 3306
126
+ user = your_readonly_user
127
+ password = your_password
128
+ database = myapp
129
+ ```
130
+
131
+ `remote_host`/`remote_port` are the DB **as seen from the bastion**, not your
132
+ laptop — getting this wrong is the #1 setup failure. For the full field
133
+ reference, SSH troubleshooting, and External (`ssh -N`) mode, see
134
+ [`docs/setup.md`](docs/setup.md).
135
+
136
+ 🔒 Double-check `git status` does **not** list `db.ini` before any commit.
137
+
138
+ ---
139
+
140
+ ## Connect your AI agent
141
+
142
+ ```bash
143
+ uv run sql-buddy mcp add # Claude Code project → ./.mcp.json
144
+ uv run sql-buddy mcp add --scope user # Claude Code user → ~/.claude.json
145
+ uv run sql-buddy mcp add --client trae # Trae project → ./.trae/mcp.json
146
+ uv run sql-buddy mcp add --client trae --scope user # Trae user-global
147
+ ```
148
+
149
+ Restart your client. Preview the exact block first with `uv run sql-buddy mcp print`
150
+ (paths filled in):
151
+
152
+ ```json
153
+ {
154
+ "mcpServers": {
155
+ "sql-buddy": {
156
+ "command": "uv",
157
+ "args": ["--directory", "/absolute/path/to/sql-buddy", "run", "server.py"]
158
+ }
159
+ }
160
+ }
161
+ ```
162
+
163
+ > No `uv` on the machine that runs your MCP client? `sql-buddy mcp print --python`
164
+ > sets `command` to the running interpreter (`sys.executable`).
165
+ > For other MCP clients (Cursor, Continue, WorkBuddy, …) and Windows/Trae
166
+ > specifics, see [`docs/setup.md`](docs/setup.md) and [`docs/windows-trae.md`](docs/windows-trae.md).
167
+
168
+ ---
169
+
170
+ ## Security model (short)
171
+
172
+ - **Read-only enforced in code**, before any SQL reaches the network. Write
173
+ keywords (`INSERT`, `UPDATE`, `DELETE`, `DROP`, `ALTER`, `TRUNCATE`, …) are
174
+ rejected by a guard layer wrapping every connection.
175
+ - **Credentials stay local** — in `db.ini` on your machine, never sent anywhere.
176
+ - **SSH tunnels bind to localhost only** (`127.0.0.1`); key-based auth only.
177
+ - **`LIMIT` is always applied** to `SELECT`/`WITH` (default 10, hard cap 2000).
178
+
179
+ Full security posture and an error/troubleshooting matrix:
180
+ [`docs/reference.md`](docs/reference.md).
181
+
182
+ ---
183
+
184
+ ## Documentation
185
+
186
+ | Topic | Where |
187
+ |-------|-------|
188
+ | Full setup, SSH tunnels, `db.ini` field reference, other MCP clients, agent-install rules | [`docs/setup.md`](docs/setup.md) |
189
+ | Windows & Trae IDE notes | [`docs/windows-trae.md`](docs/windows-trae.md) |
190
+ | MCP tools, configuration vars, troubleshooting, security reference | [`docs/reference.md`](docs/reference.md) |
191
+ | Domain language (Connection vs database, read-only invariant) | [`UBIQUITOUS_LANGUAGE.md`](UBIQUITOUS_LANGUAGE.md) |
192
+ | Contributing & the review rubric | [`CONTRIBUTING.md`](CONTRIBUTING.md) · [`AGENTS.md`](AGENTS.md) |
193
+ | Architecture decisions (ADRs) | [`docs/adr/`](docs/adr/) |
@@ -0,0 +1,164 @@
1
+ # SQL Buddy
2
+
3
+ A read-only **MCP server** that lets your AI agent (Claude Code, and any
4
+ MCP-compatible client) safely query your **MySQL** and **ClickHouse** databases.
5
+
6
+ Your agent asks in plain language; SQL Buddy runs the SQL, enforces **read-only**
7
+ (no `INSERT`/`UPDATE`/`DROP` ever reaches your data), adds safety `LIMIT`s, and
8
+ caches results. You stay in control — the agent can look, never touch.
9
+
10
+ ```
11
+ Your AI agent ──MCP──► SQL Buddy (runs locally) ──► your databases
12
+ • opens its own SSH tunnel through a bastion
13
+ • read-only guard · auto LIMIT · schema lookup
14
+ ```
15
+
16
+ SQL Buddy runs **locally** (a small Python program, not a cloud service). It
17
+ **opens the SSH tunnel itself** from a bastion + key + port mapping in `db.ini`
18
+ — no separate `ssh -N` terminal, and your credentials never leave your machine.
19
+
20
+ ---
21
+
22
+ ## Prerequisites
23
+
24
+ - **Python 3.11+** — `python3 --version`
25
+ - **[uv](https://docs.astral.sh/uv/)** (recommended) or plain `pip`.
26
+ - **SSH access** to a bastion/jump-host that can reach your databases: its
27
+ `user@host` + ssh `port` (often a high port like `22222`, **not** 22), a
28
+ **private key file**, and each DB's `host:port` **as seen from the bastion**.
29
+ - **Read-only DB credentials** — a user that can `SELECT` but not write.
30
+ - An **MCP-compatible client** (e.g. Claude Code, **Trae**).
31
+ - **Windows only (Auto tunnels):** [OpenSSH Client](https://learn.microsoft.com/windows-server/administration/openssh/openssh_install_firstuse).
32
+ (Skip if you run the tunnel yourself in another terminal.)
33
+
34
+ > Deep notes on tunnels, Windows/Trae, and every `db.ini` field live in
35
+ > [`docs/setup.md`](docs/setup.md) and [`docs/windows-trae.md`](docs/windows-trae.md).
36
+
37
+ ---
38
+
39
+ ## Quickstart
40
+
41
+ ```bash
42
+ git clone <this-repo-url> sql-buddy && cd sql-buddy
43
+ uv run server.py # installs deps into an isolated .venv, then serves (Ctrl+C to stop)
44
+ uv run sql-buddy init # interactive wizard → writes connections/config/db.ini
45
+ uv run sql-buddy doctor --connect # green-checks the venv, db.ini, tunnels, MCP client config
46
+ uv run sql-buddy mcp add # registers sql-buddy with Claude Code (writes .mcp.json)
47
+ ```
48
+
49
+ Restart your client, then ask your agent *"list the tables in the `ck`
50
+ connection"* — if it answers, you're live. 🎉
51
+
52
+ > **Bare `sql-buddy` (no command) runs the MCP server** — that's what your AI
53
+ > client invokes, so configs hold no fragile absolute `.venv/bin/python` path.
54
+ > Other commands: `version`, `doctor [--connect]`, `mcp print|add`, `update-check`, `init`.
55
+ > On **Windows**, see [`docs/windows-trae.md`](docs/windows-trae.md) and use
56
+ > `sql-buddy mcp add --client trae`.
57
+
58
+ ---
59
+
60
+ ## Configure connections (`db.ini`)
61
+
62
+ Run the wizard (easiest) or copy the template:
63
+
64
+ ```bash
65
+ uv run sql-buddy init
66
+ # or by hand:
67
+ cp connections/config/db.ini.example connections/config/db.ini
68
+ $EDITOR connections/config/db.ini
69
+ ```
70
+
71
+ Each database is one **section**; the `[bracketed]` name becomes the
72
+ **connection name** your agent uses (e.g. `ck`, `mysql`, `prod-read`).
73
+
74
+ ```ini
75
+ # db.ini — NEVER commit this file (gitignored). It holds live credentials.
76
+
77
+ [ssh:prod]
78
+ host = bastion.example.com
79
+ port = 22
80
+ user = your_ssh_user
81
+ key = ~/.ssh/id_ed25519 # path to your private key (~ expanded)
82
+
83
+ [ck]
84
+ type = clickhouse
85
+ tunnel = ssh:prod # self-managed tunnel through the bastion above
86
+ remote_host = 10.0.0.21 # DB host AS SEEN FROM the bastion
87
+ remote_port = 9000 # DB port AS SEEN FROM the bastion
88
+ local_port = 10021 # localhost port the driver connects to
89
+ user = your_readonly_user
90
+ password = your_password
91
+ database = default
92
+
93
+ [local-db] # direct (non-tunneled) connection also works
94
+ type = mysql
95
+ host = localhost
96
+ port = 3306
97
+ user = your_readonly_user
98
+ password = your_password
99
+ database = myapp
100
+ ```
101
+
102
+ `remote_host`/`remote_port` are the DB **as seen from the bastion**, not your
103
+ laptop — getting this wrong is the #1 setup failure. For the full field
104
+ reference, SSH troubleshooting, and External (`ssh -N`) mode, see
105
+ [`docs/setup.md`](docs/setup.md).
106
+
107
+ 🔒 Double-check `git status` does **not** list `db.ini` before any commit.
108
+
109
+ ---
110
+
111
+ ## Connect your AI agent
112
+
113
+ ```bash
114
+ uv run sql-buddy mcp add # Claude Code project → ./.mcp.json
115
+ uv run sql-buddy mcp add --scope user # Claude Code user → ~/.claude.json
116
+ uv run sql-buddy mcp add --client trae # Trae project → ./.trae/mcp.json
117
+ uv run sql-buddy mcp add --client trae --scope user # Trae user-global
118
+ ```
119
+
120
+ Restart your client. Preview the exact block first with `uv run sql-buddy mcp print`
121
+ (paths filled in):
122
+
123
+ ```json
124
+ {
125
+ "mcpServers": {
126
+ "sql-buddy": {
127
+ "command": "uv",
128
+ "args": ["--directory", "/absolute/path/to/sql-buddy", "run", "server.py"]
129
+ }
130
+ }
131
+ }
132
+ ```
133
+
134
+ > No `uv` on the machine that runs your MCP client? `sql-buddy mcp print --python`
135
+ > sets `command` to the running interpreter (`sys.executable`).
136
+ > For other MCP clients (Cursor, Continue, WorkBuddy, …) and Windows/Trae
137
+ > specifics, see [`docs/setup.md`](docs/setup.md) and [`docs/windows-trae.md`](docs/windows-trae.md).
138
+
139
+ ---
140
+
141
+ ## Security model (short)
142
+
143
+ - **Read-only enforced in code**, before any SQL reaches the network. Write
144
+ keywords (`INSERT`, `UPDATE`, `DELETE`, `DROP`, `ALTER`, `TRUNCATE`, …) are
145
+ rejected by a guard layer wrapping every connection.
146
+ - **Credentials stay local** — in `db.ini` on your machine, never sent anywhere.
147
+ - **SSH tunnels bind to localhost only** (`127.0.0.1`); key-based auth only.
148
+ - **`LIMIT` is always applied** to `SELECT`/`WITH` (default 10, hard cap 2000).
149
+
150
+ Full security posture and an error/troubleshooting matrix:
151
+ [`docs/reference.md`](docs/reference.md).
152
+
153
+ ---
154
+
155
+ ## Documentation
156
+
157
+ | Topic | Where |
158
+ |-------|-------|
159
+ | Full setup, SSH tunnels, `db.ini` field reference, other MCP clients, agent-install rules | [`docs/setup.md`](docs/setup.md) |
160
+ | Windows & Trae IDE notes | [`docs/windows-trae.md`](docs/windows-trae.md) |
161
+ | MCP tools, configuration vars, troubleshooting, security reference | [`docs/reference.md`](docs/reference.md) |
162
+ | Domain language (Connection vs database, read-only invariant) | [`UBIQUITOUS_LANGUAGE.md`](UBIQUITOUS_LANGUAGE.md) |
163
+ | Contributing & the review rubric | [`CONTRIBUTING.md`](CONTRIBUTING.md) · [`AGENTS.md`](AGENTS.md) |
164
+ | Architecture decisions (ADRs) | [`docs/adr/`](docs/adr/) |
@@ -7,4 +7,4 @@ still report a version. `cli.get_version()` prefers installed metadata when
7
7
  available and falls back to this.
8
8
  """
9
9
 
10
- __version__ = "0.3.2"
10
+ __version__ = "0.3.3"
@@ -164,11 +164,40 @@ def _result_to_json(client, sql: str) -> str:
164
164
 
165
165
  # --- Schema ---
166
166
 
167
+ # MergeTree clause keywords that terminate a DDL value expression. All of these
168
+ # follow the column list in ClickHouse DDL (PARTITION BY / PRIMARY KEY /
169
+ # ORDER BY / SAMPLE BY / SETTINGS / TTL come after `ENGINE = ...`).
170
+ _DDL_CLAUSE_STOP = (
171
+ r'ENGINE\b|PARTITION\s+BY\b|PRIMARY\s+KEY\b|ORDER\s+BY\b'
172
+ r'|SAMPLE\s+BY\b|SETTINGS\b|TTL\b'
173
+ )
174
+
175
+
176
+ def _extract_ddl_clause(schema: str, clause: str) -> str:
177
+ """Extract the expression following a CREATE TABLE clause keyword.
178
+
179
+ `clause` is a regex fragment such as ``PRIMARY\\s+KEY`` or ``ORDER\\s+BY``.
180
+ Captures from just after the keyword up to the next sibling clause keyword
181
+ or end of string. Works for both newline-separated DDL (real
182
+ ``SHOW CREATE TABLE`` output) and single-line DDL (tests), unlike the old
183
+ ``[^\n]+?`` regexes that required a trailing newline terminator and so
184
+ silently returned '' for one-line schemas.
185
+ """
186
+ m = re.search(
187
+ rf'{clause}\s+(.+?)(?=\s+(?:{_DDL_CLAUSE_STOP})|\s*$)',
188
+ schema, re.IGNORECASE | re.DOTALL,
189
+ )
190
+ if not m:
191
+ return ''
192
+ return m.group(1).strip().rstrip(',').strip()
193
+
194
+
167
195
  def _parse_create_table(schema: str) -> dict:
168
196
  schema_clean = schema.replace('\\n', '\n')
169
197
  result = {
170
198
  'table_name': '', 'engine': '', 'partition': '',
171
- 'order_by': '', 'columns': [], 'projections': []
199
+ 'primary_key': '', 'order_by': '', 'sampling_key': '',
200
+ 'columns': [], 'projections': []
172
201
  }
173
202
 
174
203
  m = re.search(r'CREATE\s+TABLE\s+([\w.`]+)', schema_clean, re.IGNORECASE)
@@ -179,19 +208,19 @@ def _parse_create_table(schema: str) -> dict:
179
208
  if m:
180
209
  result['engine'] = m.group(1)
181
210
 
182
- m = re.search(
183
- r'PARTITION\s+BY\s+([^\n]+?)(?=\n\s*(?:ORDER|PRIMARY|KEY|ENGINE|SETTINGS)|\n\))',
184
- schema_clean, re.IGNORECASE
185
- )
186
- if m:
187
- result['partition'] = m.group(1).strip()
188
-
189
- m = re.search(
190
- r'ORDER\s+BY\s+([^\n]+?)(?=\n\s*(?:PRIMARY|KEY|SETTINGS)|\n\))',
191
- schema_clean, re.IGNORECASE
192
- )
193
- if m:
194
- result['order_by'] = m.group(1).strip()
211
+ # PARTITION BY / PRIMARY KEY / ORDER BY / SAMPLE BY live AFTER the column
212
+ # list in ClickHouse DDL. Restrict the clause search to the tail starting at
213
+ # ENGINE so a MySQL `PRIMARY KEY (...)` declared *inside* the column list is
214
+ # never mistaken for a MergeTree sparse-index key. PRIMARY KEY and ORDER BY
215
+ # are distinct in CH: PRIMARY KEY sets sparse-index granularity (a prefix of
216
+ # ORDER BY), ORDER BY sets the on-disk sort — filtering on an ORDER BY column
217
+ # that is NOT in PRIMARY KEY is a full granule scan.
218
+ engine_match = re.search(r'ENGINE\s*=', schema_clean, re.IGNORECASE)
219
+ clause_tail = schema_clean[engine_match.start():] if engine_match else schema_clean
220
+ result['partition'] = _extract_ddl_clause(clause_tail, r'PARTITION\s+BY')
221
+ result['primary_key'] = _extract_ddl_clause(clause_tail, r'PRIMARY\s+KEY')
222
+ result['order_by'] = _extract_ddl_clause(clause_tail, r'ORDER\s+BY')
223
+ result['sampling_key'] = _extract_ddl_clause(clause_tail, r'SAMPLE\s+BY')
195
224
 
196
225
  for match in re.finditer(
197
226
  r'`(\w+)`\s+(\w+(?:\([^)]*\))?)\s*(?:DEFAULT\s+[^\s,]+)?\s*(?:COMMENT\s+\'[^\']*\')?',
@@ -212,18 +241,36 @@ def _parse_create_table(schema: str) -> dict:
212
241
  return result
213
242
 
214
243
 
244
+ # Column names that imply a time column regardless of declared type.
245
+ _TIME_COL_NAMES = {'create_time', 'insert_time', 'update_time'}
246
+ # Type families that mark a column as time-related.
247
+ _TIME_TYPE_RE = re.compile(
248
+ r'^(DateTime64|DateTime|Date32|Date|Timestamp|TIMESTAMP)', re.IGNORECASE
249
+ )
250
+
251
+
215
252
  def _extract_time_columns(schema: str) -> list:
253
+ """Return time-related column names from a CREATE TABLE statement.
254
+
255
+ A column is time-related if its name is create_time/insert_time/update_time
256
+ or its declared type is a Date/DateTime/Timestamp family type. Names are
257
+ returned clean — never a bare backtick or a `` `col` `` token. The old
258
+ heuristic grabbed the whitespace token preceding a keyword and could yield
259
+ ``"`"`` (a lone backtick, from the opening quote of `` `insert_time` ``) or
260
+ ``"`insert_time`"`` (backtick-wrapped); this scans real column definitions
261
+ instead and strips surrounding quotes/punctuation by construction.
262
+ """
216
263
  time_cols = []
217
- schema_lower = schema.lower()
218
- for kw in ['create_time', 'insert_time', 'update_time', 'date', 'datetime', 'timestamp']:
219
- if kw in schema_lower:
220
- idx = schema_lower.find(kw)
221
- segment = schema[max(0, idx - 30):idx]
222
- parts = segment.replace(',', ' ').split()
223
- if parts:
224
- col = parts[-1].strip()
225
- if col and col not in time_cols:
226
- time_cols.append(col)
264
+ seen = set()
265
+ # A column definition: an optional backtick-quoted identifier then a type.
266
+ col_pattern = re.compile(r'`?(\w+)`?\s+(\w+(?:\([^)]*\))?)')
267
+ for m in col_pattern.finditer(schema):
268
+ col, ctype = m.group(1), m.group(2)
269
+ if col in seen:
270
+ continue
271
+ if col.lower() in _TIME_COL_NAMES or _TIME_TYPE_RE.match(ctype):
272
+ seen.add(col)
273
+ time_cols.append(col)
227
274
  return time_cols
228
275
 
229
276
 
@@ -273,6 +320,79 @@ def table_schema_query(tbl_name: str, connection: str = 'ck', force_refresh: boo
273
320
 
274
321
  # --- Table Info ---
275
322
 
323
+ def _resolve_connection_type(connection: str) -> str:
324
+ """Resolve a connection name to its DBMS type ('clickhouse' | 'mysql').
325
+
326
+ Replaces the brittle ``hasattr(client, 'query_dataframe')`` dispatch: BOTH
327
+ Locked wrappers define ``query_dataframe`` (MySQL natively, ClickHouse via
328
+ the adapter in connections/wrappers.py), so that attribute check *always*
329
+ picked the MySQL branch — even for ClickHouse, which then fabricated a
330
+ fake PRIMARY index from ClickHouse's stub ``SHOW INDEX``. This reads the
331
+ section's declared ``type`` (with the section-name prefix fallback), the
332
+ same source of truth ``find_table_query`` uses. Connection names use
333
+ underscores; db.ini sections use dashes, so normalize the way
334
+ DBManager.__getattr__ does.
335
+ """
336
+ cfg = DBManager.load_config(CONFIG_FILE)
337
+ section = connection.replace('_', '-')
338
+ return connection_type(cfg, section)
339
+
340
+
341
+ def _split_qualified_table(tbl_name: str) -> tuple:
342
+ """Split a (possibly db-qualified) table name into (database, table).
343
+
344
+ ``asia.elevenst_kr_seo_result`` -> ``('asia', 'elevenst_kr_seo_result')``.
345
+ Backticks are stripped. Unqualified names return ``('', name)``.
346
+ """
347
+ tbl = tbl_name.strip().strip('`')
348
+ if '.' in tbl:
349
+ db_part, _, tbl_part = tbl.partition('.')
350
+ return db_part.strip('`'), tbl_part.strip('`')
351
+ return '', tbl
352
+
353
+
354
+ def _clickhouse_parts(client, tbl_name: str) -> list:
355
+ """Active parts for a ClickHouse table, read from ``system.parts``.
356
+
357
+ Returns up to 20 rows of ``{partition, name, rows, bytes_on_disk}`` — the
358
+ real scan-cost signals (partition count, part count, row distribution) that
359
+ ``SHOW PARTITIONS FROM <table>`` could never provide (it is invalid
360
+ ClickHouse syntax: ``Code: 62. DB::Exception ... PARTITIONS``). Uses
361
+ parameter binding for the identifiers so a mistyped/malicious name can never
362
+ inject SQL, and falls back to ``currentDatabase()`` when the name is
363
+ unqualified. Returns ``[]`` on error (never raises) so ``get_info`` still
364
+ reports ``row_count``.
365
+ """
366
+ db_part, tbl_part = _split_qualified_table(tbl_name)
367
+ if db_part:
368
+ sql = (
369
+ "SELECT partition, name, rows, bytes_on_disk "
370
+ "FROM system.parts "
371
+ "WHERE database = %(db)s AND table = %(tbl)s AND active = 1 "
372
+ "ORDER BY partition, name "
373
+ "LIMIT 20"
374
+ )
375
+ params = {'db': db_part, 'tbl': tbl_part}
376
+ else:
377
+ sql = (
378
+ "SELECT partition, name, rows, bytes_on_disk "
379
+ "FROM system.parts "
380
+ "WHERE database = currentDatabase() AND table = %(tbl)s AND active = 1 "
381
+ "ORDER BY partition, name "
382
+ "LIMIT 20"
383
+ )
384
+ params = {'tbl': tbl_part}
385
+ try:
386
+ result = client.execute(sql, params, with_column_types=True)
387
+ except Exception:
388
+ return []
389
+ if not result or not result[0]:
390
+ return []
391
+ data, columns = result
392
+ col_names = [c[0] for c in columns]
393
+ return [_sanitize_row(dict(zip(col_names, row))) for row in data]
394
+
395
+
276
396
  def table_info_query(tbl_name: str, connection: str = 'ck', force_refresh: bool = False) -> dict:
277
397
  cache_file = _get_cache_file_path('info', connection)
278
398
  cache = _load_json_safe(cache_file)
@@ -285,9 +405,11 @@ def table_info_query(tbl_name: str, connection: str = 'ck', force_refresh: bool
285
405
  try:
286
406
  db = DBManager()
287
407
  client = getattr(db, connection)
408
+ ctype = _resolve_connection_type(connection)
288
409
  info = {'table_name': tbl_name, 'cached_at': datetime.now().isoformat()}
289
410
 
290
- if hasattr(client, 'query_dataframe'):
411
+ if ctype == 'mysql':
412
+ # MySQL: SHOW INDEX returns real BTREE indexes with real cardinality.
291
413
  try:
292
414
  result = client.query_dataframe(f"SELECT COUNT(*) as cnt FROM {tbl_name}")
293
415
  info['row_count'] = int(result['cnt'].iloc[0]) if not result.empty else 0
@@ -300,16 +422,18 @@ def table_info_query(tbl_name: str, connection: str = 'ck', force_refresh: bool
300
422
  except Exception:
301
423
  info['indexes'] = []
302
424
  else:
425
+ # ClickHouse has no MySQL-style PRIMARY index: PRIMARY KEY only sets
426
+ # sparse-index granularity (no uniqueness, no lookup), and SHOW INDEX
427
+ # returns a stub that fabricates a fake PRIMARY key with cardinality
428
+ # 0 — which misleads agents into wrong join-cost estimates. So we do
429
+ # NOT synthesize `indexes`; we report real scan-cost signals from
430
+ # system.parts (partition / parts / rows per part) instead.
303
431
  try:
304
432
  result = client.execute(f"SELECT count() as cnt FROM {tbl_name}")
305
433
  info['row_count'] = result[0][0] if result else 0
306
434
  except Exception:
307
435
  info['row_count'] = None
308
- try:
309
- part_result = client.execute(f"SHOW PARTITIONS FROM {tbl_name}")
310
- info['partitions'] = str(part_result[:5])
311
- except Exception:
312
- info['partitions'] = None
436
+ info['parts'] = _clickhouse_parts(client, tbl_name)
313
437
 
314
438
  cache[tbl_name] = info
315
439
  _save_json_safe(cache_file, cache)
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sqlbuddy-nwi"
3
- version = "0.3.2"
3
+ version = "0.3.3"
4
4
  description = "MCP server for read-only database access (MySQL, ClickHouse)"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"