querychat 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {querychat-0.3.0 → querychat-0.4.0}/LICENSE.md +1 -1
- querychat-0.4.0/PKG-INFO +104 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/LICENSE +1 -1
- querychat-0.4.0/pkg-py/README.md +50 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/_datasource.py +199 -42
- querychat-0.4.0/pkg-py/src/querychat/_df_compat.py +74 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/_querychat.py +287 -32
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/_querychat_module.py +27 -17
- querychat-0.4.0/pkg-py/src/querychat/_system_prompt.py +81 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/_utils.py +91 -5
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/data/__init__.py +15 -9
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/prompts/prompt.md +68 -23
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/tools.py +56 -30
- querychat-0.4.0/pkg-py/src/querychat/types/__init__.py +19 -0
- {querychat-0.3.0 → querychat-0.4.0}/pyproject.toml +10 -4
- querychat-0.3.0/PKG-INFO +0 -71
- querychat-0.3.0/pkg-py/README.md +0 -20
- querychat-0.3.0/pkg-py/src/querychat/types/__init__.py +0 -9
- {querychat-0.3.0 → querychat-0.4.0}/.gitignore +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/__init__.py +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/_deprecated.py +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/_icons.py +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/data/tips.csv.gz +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/data/titanic.csv.gz +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/express/__init__.py +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/prompts/tool-query.md +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/prompts/tool-reset-dashboard.md +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/prompts/tool-update-dashboard.md +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/static/css/styles.css +0 -0
- {querychat-0.3.0 → querychat-0.4.0}/pkg-py/src/querychat/static/js/querychat.js +0 -0
querychat-0.4.0/PKG-INFO
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: querychat
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Chat with your data using natural language
|
|
5
|
+
Project-URL: Homepage, https://github.com/posit-dev/querychat
|
|
6
|
+
Project-URL: Repository, https://github.com/posit-dev/querychat
|
|
7
|
+
Project-URL: Issues, https://github.com/posit-dev/querychat/issues
|
|
8
|
+
Project-URL: Source, https://github.com/posit-dev/querychat/tree/main/pkg-py
|
|
9
|
+
Author-email: Joe Cheng <joe@posit.co>, Garrick Aden-Buie <garrick@posit.co>, Carson Sievert <carson@posit.co>, Dan Chen <daniel.chen@posit.co>, Barret Schloerke <barret@posit.co>
|
|
10
|
+
Maintainer-email: Garrick Aden-Buie <garrick@posit.co>
|
|
11
|
+
License: # MIT License
|
|
12
|
+
|
|
13
|
+
Copyright (c) 2026 Posit Software, PBC
|
|
14
|
+
|
|
15
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
16
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
17
|
+
in the Software without restriction, including without limitation the rights
|
|
18
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
19
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
20
|
+
furnished to do so, subject to the following conditions:
|
|
21
|
+
|
|
22
|
+
The above copyright notice and this permission notice shall be included in all
|
|
23
|
+
copies or substantial portions of the Software.
|
|
24
|
+
|
|
25
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
26
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
27
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
28
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
29
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
30
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
31
|
+
SOFTWARE.
|
|
32
|
+
License-File: LICENSE.md
|
|
33
|
+
Classifier: Programming Language :: Python
|
|
34
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
38
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
39
|
+
Requires-Python: >=3.10
|
|
40
|
+
Requires-Dist: chatlas>=0.13.2
|
|
41
|
+
Requires-Dist: chevron
|
|
42
|
+
Requires-Dist: duckdb
|
|
43
|
+
Requires-Dist: great-tables>=0.16.0
|
|
44
|
+
Requires-Dist: htmltools
|
|
45
|
+
Requires-Dist: narwhals
|
|
46
|
+
Requires-Dist: shiny>=1.5.1
|
|
47
|
+
Requires-Dist: shinychat>=0.2.8
|
|
48
|
+
Requires-Dist: sqlalchemy>=2.0.0
|
|
49
|
+
Provides-Extra: pandas
|
|
50
|
+
Requires-Dist: pandas; extra == 'pandas'
|
|
51
|
+
Provides-Extra: polars
|
|
52
|
+
Requires-Dist: polars; extra == 'polars'
|
|
53
|
+
Description-Content-Type: text/markdown
|
|
54
|
+
|
|
55
|
+
# querychat <a href="https://posit-dev.github.io/querychat/py/"><img src="https://posit-dev.github.io/querychat/images/querychat.png" align="right" height="138" alt="querychat website" /></a>
|
|
56
|
+
|
|
57
|
+
<p>
|
|
58
|
+
<!-- badges start -->
|
|
59
|
+
<a href="https://pypi.org/project/querychat/"><img alt="PyPI" src="https://img.shields.io/pypi/v/querychat?logo=python&logoColor=white&color=orange"></a>
|
|
60
|
+
<a href="https://choosealicense.com/licenses/mit/"><img src="https://img.shields.io/badge/License-MIT-blue.svg" alt="MIT License"></a>
|
|
61
|
+
<a href="https://pypi.org/project/querychat"><img src="https://img.shields.io/pypi/pyversions/querychat.svg" alt="versions"></a>
|
|
62
|
+
<a href="https://github.com/posit-dev/querychat"><img src="https://github.com/posit-dev/querychat/actions/workflows/test.yml/badge.svg?branch=main" alt="Python Tests"></a>
|
|
63
|
+
<!-- badges end -->
|
|
64
|
+
</p>
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
QueryChat facilitates safe and reliable natural language exploration of tabular data, powered by SQL and large language models (LLMs). For analysts, it offers an intuitive web application where they can quickly ask questions of their data and receive verifiable data-driven answers. For software developers, QueryChat provides a comprehensive Python API to access core functionality -- including chat UI, generated SQL statements, resulting data, and more. This capability enables the seamless integration of natural language querying into bespoke data applications.
|
|
68
|
+
|
|
69
|
+
## Installation
|
|
70
|
+
|
|
71
|
+
Install the latest stable release [from PyPI](https://pypi.org/project/querychat/):
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
pip install querychat
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
## Quick start
|
|
78
|
+
|
|
79
|
+
The main entry point is the [`QueryChat` class](https://posit-dev.github.io/querychat/py/reference/QueryChat.html). It requires a [data source](https://posit-dev.github.io/querychat/py/data-sources.html) (e.g., pandas, polars, etc) and a name for the data.
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
from querychat import QueryChat
|
|
83
|
+
from querychat.data import titanic
|
|
84
|
+
|
|
85
|
+
qc = QueryChat(titanic(), "titanic")
|
|
86
|
+
app = qc.app()
|
|
87
|
+
# app.run()
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
<p align="center">
|
|
91
|
+
<img src="docs/images/quickstart.png" alt="QueryChat interface showing natural language queries" width="85%">
|
|
92
|
+
</p>
|
|
93
|
+
|
|
94
|
+
## Custom apps
|
|
95
|
+
|
|
96
|
+
Build your own custom web apps with natural language querying capabilities, such as [this one](https://github.com/posit-conf-2025/llm/blob/main/_solutions/25_querychat/25_querychat_02-end-app.R) which provides a bespoke interface for exploring Airbnb listings:
|
|
97
|
+
|
|
98
|
+
<p align="center">
|
|
99
|
+
<img src="docs/images/airbnb.png" alt="A custom app for exploring Airbnb listings, powered by QueryChat." width="85%">
|
|
100
|
+
</p>
|
|
101
|
+
|
|
102
|
+
## Learn more
|
|
103
|
+
|
|
104
|
+
See the [website](https://posit-dev.github.io/querychat/py) to learn more.
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# querychat <a href="https://posit-dev.github.io/querychat/py/"><img src="https://posit-dev.github.io/querychat/images/querychat.png" align="right" height="138" alt="querychat website" /></a>
|
|
2
|
+
|
|
3
|
+
<p>
|
|
4
|
+
<!-- badges start -->
|
|
5
|
+
<a href="https://pypi.org/project/querychat/"><img alt="PyPI" src="https://img.shields.io/pypi/v/querychat?logo=python&logoColor=white&color=orange"></a>
|
|
6
|
+
<a href="https://choosealicense.com/licenses/mit/"><img src="https://img.shields.io/badge/License-MIT-blue.svg" alt="MIT License"></a>
|
|
7
|
+
<a href="https://pypi.org/project/querychat"><img src="https://img.shields.io/pypi/pyversions/querychat.svg" alt="versions"></a>
|
|
8
|
+
<a href="https://github.com/posit-dev/querychat"><img src="https://github.com/posit-dev/querychat/actions/workflows/test.yml/badge.svg?branch=main" alt="Python Tests"></a>
|
|
9
|
+
<!-- badges end -->
|
|
10
|
+
</p>
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
QueryChat facilitates safe and reliable natural language exploration of tabular data, powered by SQL and large language models (LLMs). For analysts, it offers an intuitive web application where they can quickly ask questions of their data and receive verifiable data-driven answers. For software developers, QueryChat provides a comprehensive Python API to access core functionality -- including chat UI, generated SQL statements, resulting data, and more. This capability enables the seamless integration of natural language querying into bespoke data applications.
|
|
14
|
+
|
|
15
|
+
## Installation
|
|
16
|
+
|
|
17
|
+
Install the latest stable release [from PyPI](https://pypi.org/project/querychat/):
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install querychat
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## Quick start
|
|
24
|
+
|
|
25
|
+
The main entry point is the [`QueryChat` class](https://posit-dev.github.io/querychat/py/reference/QueryChat.html). It requires a [data source](https://posit-dev.github.io/querychat/py/data-sources.html) (e.g., pandas, polars, etc) and a name for the data.
|
|
26
|
+
|
|
27
|
+
```python
|
|
28
|
+
from querychat import QueryChat
|
|
29
|
+
from querychat.data import titanic
|
|
30
|
+
|
|
31
|
+
qc = QueryChat(titanic(), "titanic")
|
|
32
|
+
app = qc.app()
|
|
33
|
+
# app.run()
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
<p align="center">
|
|
37
|
+
<img src="docs/images/quickstart.png" alt="QueryChat interface showing natural language queries" width="85%">
|
|
38
|
+
</p>
|
|
39
|
+
|
|
40
|
+
## Custom apps
|
|
41
|
+
|
|
42
|
+
Build your own custom web apps with natural language querying capabilities, such as [this one](https://github.com/posit-conf-2025/llm/blob/main/_solutions/25_querychat/25_querychat_02-end-app.R) which provides a bespoke interface for exploring Airbnb listings:
|
|
43
|
+
|
|
44
|
+
<p align="center">
|
|
45
|
+
<img src="docs/images/airbnb.png" alt="A custom app for exploring Airbnb listings, powered by QueryChat." width="85%">
|
|
46
|
+
</p>
|
|
47
|
+
|
|
48
|
+
## Learn more
|
|
49
|
+
|
|
50
|
+
See the [website](https://posit-dev.github.io/querychat/py) to learn more.
|
|
@@ -5,15 +5,20 @@ from typing import TYPE_CHECKING
|
|
|
5
5
|
|
|
6
6
|
import duckdb
|
|
7
7
|
import narwhals.stable.v1 as nw
|
|
8
|
-
import pandas as pd
|
|
9
8
|
from sqlalchemy import inspect, text
|
|
10
9
|
from sqlalchemy.sql import sqltypes
|
|
11
10
|
|
|
11
|
+
from ._df_compat import duckdb_result_to_nw, read_sql
|
|
12
|
+
from ._utils import check_query
|
|
13
|
+
|
|
12
14
|
if TYPE_CHECKING:
|
|
13
|
-
from narwhals.stable.v1.typing import IntoFrame
|
|
14
15
|
from sqlalchemy.engine import Connection, Engine
|
|
15
16
|
|
|
16
17
|
|
|
18
|
+
class MissingColumnsError(ValueError):
|
|
19
|
+
"""Raised when a query result is missing required columns."""
|
|
20
|
+
|
|
21
|
+
|
|
17
22
|
class DataSource(ABC):
|
|
18
23
|
"""
|
|
19
24
|
An abstract class defining the interface for data sources used by QueryChat.
|
|
@@ -53,7 +58,7 @@ class DataSource(ABC):
|
|
|
53
58
|
...
|
|
54
59
|
|
|
55
60
|
@abstractmethod
|
|
56
|
-
def execute_query(self, query: str) ->
|
|
61
|
+
def execute_query(self, query: str) -> nw.DataFrame:
|
|
57
62
|
"""
|
|
58
63
|
Execute SQL query and return results as DataFrame.
|
|
59
64
|
|
|
@@ -65,20 +70,48 @@ class DataSource(ABC):
|
|
|
65
70
|
Returns
|
|
66
71
|
-------
|
|
67
72
|
:
|
|
68
|
-
Query results as a
|
|
73
|
+
Query results as a narwhals DataFrame
|
|
69
74
|
|
|
70
75
|
"""
|
|
71
76
|
...
|
|
72
77
|
|
|
73
78
|
@abstractmethod
|
|
74
|
-
def
|
|
79
|
+
def test_query(
|
|
80
|
+
self, query: str, *, require_all_columns: bool = False
|
|
81
|
+
) -> nw.DataFrame:
|
|
82
|
+
"""
|
|
83
|
+
Test SQL query by fetching only one row.
|
|
84
|
+
|
|
85
|
+
Parameters
|
|
86
|
+
----------
|
|
87
|
+
query
|
|
88
|
+
SQL query to test
|
|
89
|
+
require_all_columns
|
|
90
|
+
If True, validates that result includes all original table columns.
|
|
91
|
+
Additional computed columns are allowed.
|
|
92
|
+
|
|
93
|
+
Returns
|
|
94
|
+
-------
|
|
95
|
+
:
|
|
96
|
+
Query results as a narwhals DataFrame with at most one row
|
|
97
|
+
|
|
98
|
+
Raises
|
|
99
|
+
------
|
|
100
|
+
MissingColumnsError
|
|
101
|
+
If require_all_columns is True and result is missing required columns
|
|
102
|
+
|
|
103
|
+
"""
|
|
104
|
+
...
|
|
105
|
+
|
|
106
|
+
@abstractmethod
|
|
107
|
+
def get_data(self) -> nw.DataFrame:
|
|
75
108
|
"""
|
|
76
109
|
Return the unfiltered data as a DataFrame.
|
|
77
110
|
|
|
78
111
|
Returns
|
|
79
112
|
-------
|
|
80
113
|
:
|
|
81
|
-
The complete dataset as a
|
|
114
|
+
The complete dataset as a narwhals DataFrame
|
|
82
115
|
|
|
83
116
|
"""
|
|
84
117
|
...
|
|
@@ -99,27 +132,44 @@ class DataSource(ABC):
|
|
|
99
132
|
|
|
100
133
|
|
|
101
134
|
class DataFrameSource(DataSource):
|
|
102
|
-
"""A DataSource implementation that wraps a
|
|
135
|
+
"""A DataSource implementation that wraps a DataFrame using DuckDB."""
|
|
103
136
|
|
|
104
|
-
_df: nw.DataFrame
|
|
137
|
+
_df: nw.DataFrame
|
|
105
138
|
|
|
106
|
-
def __init__(self, df:
|
|
139
|
+
def __init__(self, df: nw.DataFrame, table_name: str):
|
|
107
140
|
"""
|
|
108
|
-
Initialize with a
|
|
141
|
+
Initialize with a DataFrame.
|
|
109
142
|
|
|
110
143
|
Parameters
|
|
111
144
|
----------
|
|
112
145
|
df
|
|
113
|
-
The DataFrame to wrap
|
|
146
|
+
The DataFrame to wrap (pandas, polars, or any narwhals-compatible frame)
|
|
114
147
|
table_name
|
|
115
148
|
Name of the table in SQL queries
|
|
116
149
|
|
|
117
150
|
"""
|
|
118
|
-
self.
|
|
119
|
-
self._df = nw.from_native(df)
|
|
151
|
+
self._df = nw.from_native(df) if not isinstance(df, nw.DataFrame) else df
|
|
120
152
|
self.table_name = table_name
|
|
121
|
-
|
|
122
|
-
self._conn
|
|
153
|
+
|
|
154
|
+
self._conn = duckdb.connect(database=":memory:")
|
|
155
|
+
self._conn.register(table_name, self._df.to_native())
|
|
156
|
+
self._conn.execute("""
|
|
157
|
+
-- extensions: lock down supply chain + auto behaviors
|
|
158
|
+
SET allow_community_extensions = false;
|
|
159
|
+
SET allow_unsigned_extensions = false;
|
|
160
|
+
SET autoinstall_known_extensions = false;
|
|
161
|
+
SET autoload_known_extensions = false;
|
|
162
|
+
|
|
163
|
+
-- external I/O: block file/database/network access from SQL
|
|
164
|
+
SET enable_external_access = false;
|
|
165
|
+
SET disabled_filesystems = 'LocalFileSystem';
|
|
166
|
+
|
|
167
|
+
-- freeze configuration so user SQL can't relax anything
|
|
168
|
+
SET lock_configuration = true;
|
|
169
|
+
""")
|
|
170
|
+
|
|
171
|
+
# Store original column names for validation
|
|
172
|
+
self._colnames = list(self._df.columns)
|
|
123
173
|
|
|
124
174
|
def get_db_type(self) -> str:
|
|
125
175
|
"""
|
|
@@ -151,16 +201,8 @@ class DataFrameSource(DataSource):
|
|
|
151
201
|
"""
|
|
152
202
|
schema = [f"Table: {self.table_name}", "Columns:"]
|
|
153
203
|
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
self._df.head(10).collect()
|
|
157
|
-
if isinstance(self._df, nw.LazyFrame)
|
|
158
|
-
else self._df
|
|
159
|
-
)
|
|
160
|
-
|
|
161
|
-
for column in ndf.columns:
|
|
162
|
-
# Map pandas dtypes to SQL-like types
|
|
163
|
-
dtype = ndf[column].dtype
|
|
204
|
+
for column in self._df.columns:
|
|
205
|
+
dtype = self._df[column].dtype
|
|
164
206
|
if dtype.is_integer():
|
|
165
207
|
sql_type = "INTEGER"
|
|
166
208
|
elif dtype.is_float():
|
|
@@ -176,17 +218,14 @@ class DataFrameSource(DataSource):
|
|
|
176
218
|
|
|
177
219
|
column_info = [f"- {column} ({sql_type})"]
|
|
178
220
|
|
|
179
|
-
# For TEXT columns, check if they're categorical
|
|
180
221
|
if sql_type == "TEXT":
|
|
181
|
-
unique_values =
|
|
222
|
+
unique_values = self._df[column].drop_nulls().unique()
|
|
182
223
|
if unique_values.len() <= categorical_threshold:
|
|
183
224
|
categories = unique_values.to_list()
|
|
184
225
|
categories_str = ", ".join([f"'{c}'" for c in categories])
|
|
185
226
|
column_info.append(f" Categorical values: {categories_str}")
|
|
186
|
-
|
|
187
|
-
# For numeric columns, include range
|
|
188
227
|
elif sql_type in ["INTEGER", "FLOAT", "DATE", "TIME"]:
|
|
189
|
-
rng =
|
|
228
|
+
rng = self._df[column].min(), self._df[column].max()
|
|
190
229
|
if rng[0] is None and rng[1] is None:
|
|
191
230
|
column_info.append(" Range: NULL to NULL")
|
|
192
231
|
else:
|
|
@@ -196,10 +235,12 @@ class DataFrameSource(DataSource):
|
|
|
196
235
|
|
|
197
236
|
return "\n".join(schema)
|
|
198
237
|
|
|
199
|
-
def execute_query(self, query: str) ->
|
|
238
|
+
def execute_query(self, query: str) -> nw.DataFrame:
|
|
200
239
|
"""
|
|
201
240
|
Execute query using DuckDB.
|
|
202
241
|
|
|
242
|
+
Uses polars if available, otherwise falls back to pandas.
|
|
243
|
+
|
|
203
244
|
Parameters
|
|
204
245
|
----------
|
|
205
246
|
query
|
|
@@ -208,23 +249,73 @@ class DataFrameSource(DataSource):
|
|
|
208
249
|
Returns
|
|
209
250
|
-------
|
|
210
251
|
:
|
|
211
|
-
Query results as
|
|
252
|
+
Query results as narwhals DataFrame
|
|
253
|
+
|
|
254
|
+
Raises
|
|
255
|
+
------
|
|
256
|
+
UnsafeQueryError
|
|
257
|
+
If the query starts with a disallowed SQL operation
|
|
212
258
|
|
|
213
259
|
"""
|
|
214
|
-
|
|
260
|
+
check_query(query)
|
|
261
|
+
return duckdb_result_to_nw(self._conn.execute(query))
|
|
215
262
|
|
|
216
|
-
def
|
|
263
|
+
def test_query(
|
|
264
|
+
self, query: str, *, require_all_columns: bool = False
|
|
265
|
+
) -> nw.DataFrame:
|
|
266
|
+
"""
|
|
267
|
+
Test query by fetching only one row.
|
|
268
|
+
|
|
269
|
+
Parameters
|
|
270
|
+
----------
|
|
271
|
+
query
|
|
272
|
+
SQL query to test
|
|
273
|
+
require_all_columns
|
|
274
|
+
If True, validates that result includes all original table columns
|
|
275
|
+
|
|
276
|
+
Returns
|
|
277
|
+
-------
|
|
278
|
+
:
|
|
279
|
+
Query results with at most one row
|
|
280
|
+
|
|
281
|
+
Raises
|
|
282
|
+
------
|
|
283
|
+
UnsafeQueryError
|
|
284
|
+
If the query starts with a disallowed SQL operation
|
|
285
|
+
MissingColumnsError
|
|
286
|
+
If require_all_columns is True and result is missing required columns
|
|
287
|
+
|
|
288
|
+
"""
|
|
289
|
+
check_query(query)
|
|
290
|
+
result = duckdb_result_to_nw(self._conn.execute(f"{query} LIMIT 1"))
|
|
291
|
+
|
|
292
|
+
if require_all_columns:
|
|
293
|
+
result_columns = set(result.columns)
|
|
294
|
+
original_columns_set = set(self._colnames)
|
|
295
|
+
missing_columns = original_columns_set - result_columns
|
|
296
|
+
|
|
297
|
+
if missing_columns:
|
|
298
|
+
missing_list = ", ".join(f"'{col}'" for col in sorted(missing_columns))
|
|
299
|
+
original_list = ", ".join(f"'{col}'" for col in self._colnames)
|
|
300
|
+
raise MissingColumnsError(
|
|
301
|
+
f"Query result missing required columns: {missing_list}. "
|
|
302
|
+
f"The query must return all original table columns. "
|
|
303
|
+
f"Original columns: {original_list}"
|
|
304
|
+
)
|
|
305
|
+
|
|
306
|
+
return result
|
|
307
|
+
|
|
308
|
+
def get_data(self) -> nw.DataFrame:
|
|
217
309
|
"""
|
|
218
310
|
Return the unfiltered data as a DataFrame.
|
|
219
311
|
|
|
220
312
|
Returns
|
|
221
313
|
-------
|
|
222
314
|
:
|
|
223
|
-
The complete dataset as a
|
|
315
|
+
The complete dataset as a narwhals DataFrame
|
|
224
316
|
|
|
225
317
|
"""
|
|
226
|
-
|
|
227
|
-
return self._df.lazy().collect().to_pandas()
|
|
318
|
+
return self._df
|
|
228
319
|
|
|
229
320
|
def cleanup(self) -> None:
|
|
230
321
|
"""
|
|
@@ -268,6 +359,10 @@ class SQLAlchemySource(DataSource):
|
|
|
268
359
|
if not inspector.has_table(table_name):
|
|
269
360
|
raise ValueError(f"Table '{table_name}' not found in database")
|
|
270
361
|
|
|
362
|
+
# Store original column names for validation
|
|
363
|
+
columns_info = inspector.get_columns(table_name)
|
|
364
|
+
self._colnames = [col["name"] for col in columns_info]
|
|
365
|
+
|
|
271
366
|
def get_db_type(self) -> str:
|
|
272
367
|
"""
|
|
273
368
|
Get the database type.
|
|
@@ -412,10 +507,12 @@ class SQLAlchemySource(DataSource):
|
|
|
412
507
|
|
|
413
508
|
return "\n".join(schema)
|
|
414
509
|
|
|
415
|
-
def execute_query(self, query: str) ->
|
|
510
|
+
def execute_query(self, query: str) -> nw.DataFrame:
|
|
416
511
|
"""
|
|
417
512
|
Execute SQL query and return results as DataFrame.
|
|
418
513
|
|
|
514
|
+
Uses polars if available, otherwise falls back to pandas.
|
|
515
|
+
|
|
419
516
|
Parameters
|
|
420
517
|
----------
|
|
421
518
|
query
|
|
@@ -424,20 +521,80 @@ class SQLAlchemySource(DataSource):
|
|
|
424
521
|
Returns
|
|
425
522
|
-------
|
|
426
523
|
:
|
|
427
|
-
Query results as
|
|
524
|
+
Query results as narwhals DataFrame
|
|
525
|
+
|
|
526
|
+
Raises
|
|
527
|
+
------
|
|
528
|
+
UnsafeQueryError
|
|
529
|
+
If the query starts with a disallowed SQL operation
|
|
530
|
+
|
|
531
|
+
"""
|
|
532
|
+
check_query(query)
|
|
533
|
+
with self._get_connection() as conn:
|
|
534
|
+
return read_sql(text(query), conn)
|
|
535
|
+
|
|
536
|
+
def test_query(
|
|
537
|
+
self, query: str, *, require_all_columns: bool = False
|
|
538
|
+
) -> nw.DataFrame:
|
|
539
|
+
"""
|
|
540
|
+
Test query by fetching only one row.
|
|
541
|
+
|
|
542
|
+
Parameters
|
|
543
|
+
----------
|
|
544
|
+
query
|
|
545
|
+
SQL query to test
|
|
546
|
+
require_all_columns
|
|
547
|
+
If True, validates that result includes all original table columns
|
|
548
|
+
|
|
549
|
+
Returns
|
|
550
|
+
-------
|
|
551
|
+
:
|
|
552
|
+
Query results with at most one row
|
|
553
|
+
|
|
554
|
+
Raises
|
|
555
|
+
------
|
|
556
|
+
UnsafeQueryError
|
|
557
|
+
If the query starts with a disallowed SQL operation
|
|
558
|
+
MissingColumnsError
|
|
559
|
+
If require_all_columns is True and result is missing required columns
|
|
428
560
|
|
|
429
561
|
"""
|
|
562
|
+
check_query(query)
|
|
430
563
|
with self._get_connection() as conn:
|
|
431
|
-
|
|
564
|
+
# Use read_sql with limit to get at most one row
|
|
565
|
+
limit_query = f"SELECT * FROM ({query}) AS subquery LIMIT 1"
|
|
566
|
+
try:
|
|
567
|
+
result = read_sql(text(limit_query), conn)
|
|
568
|
+
except Exception:
|
|
569
|
+
# If LIMIT syntax doesn't work, fall back to regular read and take first row
|
|
570
|
+
result = read_sql(text(query), conn).head(1)
|
|
571
|
+
|
|
572
|
+
if require_all_columns:
|
|
573
|
+
result_columns = set(result.columns)
|
|
574
|
+
original_columns_set = set(self._colnames)
|
|
575
|
+
missing_columns = original_columns_set - result_columns
|
|
576
|
+
|
|
577
|
+
if missing_columns:
|
|
578
|
+
missing_list = ", ".join(
|
|
579
|
+
f"'{col}'" for col in sorted(missing_columns)
|
|
580
|
+
)
|
|
581
|
+
original_list = ", ".join(f"'{col}'" for col in self._colnames)
|
|
582
|
+
raise MissingColumnsError(
|
|
583
|
+
f"Query result missing required columns: {missing_list}. "
|
|
584
|
+
f"The query must return all original table columns. "
|
|
585
|
+
f"Original columns: {original_list}"
|
|
586
|
+
)
|
|
587
|
+
|
|
588
|
+
return result
|
|
432
589
|
|
|
433
|
-
def get_data(self) ->
|
|
590
|
+
def get_data(self) -> nw.DataFrame:
|
|
434
591
|
"""
|
|
435
592
|
Return the unfiltered data as a DataFrame.
|
|
436
593
|
|
|
437
594
|
Returns
|
|
438
595
|
-------
|
|
439
596
|
:
|
|
440
|
-
The complete dataset as a
|
|
597
|
+
The complete dataset as a narwhals DataFrame
|
|
441
598
|
|
|
442
599
|
"""
|
|
443
600
|
return self.execute_query(f"SELECT * FROM {self.table_name}")
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""
|
|
2
|
+
DataFrame compatibility: try polars first, fall back to pandas.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
from typing import TYPE_CHECKING
|
|
8
|
+
|
|
9
|
+
import narwhals.stable.v1 as nw
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
import duckdb
|
|
13
|
+
from sqlalchemy.engine import Connection
|
|
14
|
+
from sqlalchemy.sql.elements import TextClause
|
|
15
|
+
|
|
16
|
+
_INSTALL_MSG = "Install one with: pip install polars OR pip install pandas"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def read_sql(query: TextClause, conn: Connection) -> nw.DataFrame:
|
|
20
|
+
try:
|
|
21
|
+
import polars as pl # noqa: PLC0415 # pyright: ignore[reportMissingImports]
|
|
22
|
+
|
|
23
|
+
return nw.from_native(pl.read_database(query, connection=conn))
|
|
24
|
+
except Exception: # noqa: S110
|
|
25
|
+
# Catches ImportError for polars, and other errors (e.g., missing pyarrow)
|
|
26
|
+
# Intentional fallback to pandas - no logging needed
|
|
27
|
+
pass
|
|
28
|
+
|
|
29
|
+
try:
|
|
30
|
+
import pandas as pd # noqa: PLC0415 # pyright: ignore[reportMissingImports]
|
|
31
|
+
|
|
32
|
+
return nw.from_native(pd.read_sql_query(query, conn))
|
|
33
|
+
except ImportError:
|
|
34
|
+
pass
|
|
35
|
+
|
|
36
|
+
raise ImportError(f"SQLAlchemySource requires 'polars' or 'pandas'. {_INSTALL_MSG}")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def duckdb_result_to_nw(
|
|
40
|
+
result: duckdb.DuckDBPyRelation | duckdb.DuckDBPyConnection,
|
|
41
|
+
) -> nw.DataFrame:
|
|
42
|
+
try:
|
|
43
|
+
return nw.from_native(result.pl())
|
|
44
|
+
except Exception: # noqa: S110
|
|
45
|
+
# Catches ImportError for polars, and other errors (e.g., missing pyarrow)
|
|
46
|
+
# Intentional fallback to pandas - no logging needed
|
|
47
|
+
pass
|
|
48
|
+
|
|
49
|
+
try:
|
|
50
|
+
return nw.from_native(result.df())
|
|
51
|
+
except ImportError:
|
|
52
|
+
pass
|
|
53
|
+
|
|
54
|
+
raise ImportError(f"DataFrameSource requires 'polars' or 'pandas'. {_INSTALL_MSG}")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def read_csv(path: str) -> nw.DataFrame:
|
|
58
|
+
try:
|
|
59
|
+
import polars as pl # noqa: PLC0415 # pyright: ignore[reportMissingImports]
|
|
60
|
+
|
|
61
|
+
return nw.from_native(pl.read_csv(path))
|
|
62
|
+
except Exception: # noqa: S110
|
|
63
|
+
# Catches ImportError for polars, and other errors (e.g., missing pyarrow)
|
|
64
|
+
# Intentional fallback to pandas - no logging needed
|
|
65
|
+
pass
|
|
66
|
+
|
|
67
|
+
try:
|
|
68
|
+
import pandas as pd # noqa: PLC0415 # pyright: ignore[reportMissingImports]
|
|
69
|
+
|
|
70
|
+
return nw.from_native(pd.read_csv(path, compression="gzip"))
|
|
71
|
+
except ImportError:
|
|
72
|
+
pass
|
|
73
|
+
|
|
74
|
+
raise ImportError(f"Loading data requires 'polars' or 'pandas'. {_INSTALL_MSG}")
|