pac-pincers 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pac_pincers-0.1.0/PKG-INFO +11 -0
- pac_pincers-0.1.0/pyproject.toml +23 -0
- pac_pincers-0.1.0/src/pac_pincers/__init__.py +1 -0
- pac_pincers-0.1.0/src/pac_pincers/__main__.py +9 -0
- pac_pincers-0.1.0/src/pac_pincers/_transforms.py +100 -0
- pac_pincers-0.1.0/src/pac_pincers/server.py +497 -0
- pac_pincers-0.1.0/src/pac_pincers/store.py +110 -0
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pac-pincers
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Local stdio MCP server for CSV/Excel data manipulation via Polars virtual files
|
|
5
|
+
Requires-Python: >=3.11
|
|
6
|
+
Requires-Dist: fastexcel>=0.12.0
|
|
7
|
+
Requires-Dist: mcp[cli]>=1.3.0
|
|
8
|
+
Requires-Dist: openpyxl>=3.1.0
|
|
9
|
+
Requires-Dist: polars>=1.0.0
|
|
10
|
+
Requires-Dist: uuid7>=0.1.0
|
|
11
|
+
Requires-Dist: xlsxwriter>=3.2.0
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pac-pincers"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Local stdio MCP server for CSV/Excel data manipulation via Polars virtual files"
|
|
9
|
+
requires-python = ">=3.11"
|
|
10
|
+
dependencies = [
|
|
11
|
+
"mcp[cli]>=1.3.0",
|
|
12
|
+
"polars>=1.0.0",
|
|
13
|
+
"uuid7>=0.1.0",
|
|
14
|
+
"openpyxl>=3.1.0",
|
|
15
|
+
"xlsxwriter>=3.2.0",
|
|
16
|
+
"fastexcel>=0.12.0",
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
[project.scripts]
|
|
20
|
+
pac-pincers = "pac_pincers.__main__:main"
|
|
21
|
+
|
|
22
|
+
[tool.hatch.build.targets.wheel]
|
|
23
|
+
packages = ["src/pac_pincers"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""pac-pincers: Local stdio MCP server for CSV/Excel manipulation via Polars virtual files."""
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import polars as pl
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
_AGG_FN_MAP: dict[str, Any] = {
|
|
9
|
+
"sum": lambda c: c.sum(),
|
|
10
|
+
"mean": lambda c: c.mean(),
|
|
11
|
+
"min": lambda c: c.min(),
|
|
12
|
+
"max": lambda c: c.max(),
|
|
13
|
+
"count": lambda c: c.count(),
|
|
14
|
+
"std": lambda c: c.std(),
|
|
15
|
+
"var": lambda c: c.var(),
|
|
16
|
+
"first": lambda c: c.first(),
|
|
17
|
+
"last": lambda c: c.last(),
|
|
18
|
+
"median": lambda c: c.median(),
|
|
19
|
+
"n_unique": lambda c: c.n_unique(),
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def aggregate(
|
|
24
|
+
df: pl.DataFrame,
|
|
25
|
+
group_by: list[str] | None,
|
|
26
|
+
aggregations: list[dict],
|
|
27
|
+
) -> pl.DataFrame:
|
|
28
|
+
agg_exprs = []
|
|
29
|
+
for agg in aggregations:
|
|
30
|
+
col = pl.col(agg["column"])
|
|
31
|
+
fn_name = agg["function"]
|
|
32
|
+
if fn_name not in _AGG_FN_MAP:
|
|
33
|
+
raise ValueError(
|
|
34
|
+
f"Unknown aggregation function {fn_name!r}. "
|
|
35
|
+
f"Valid: {sorted(_AGG_FN_MAP)}"
|
|
36
|
+
)
|
|
37
|
+
expr = _AGG_FN_MAP[fn_name](col)
|
|
38
|
+
if alias := agg.get("alias"):
|
|
39
|
+
expr = expr.alias(alias)
|
|
40
|
+
agg_exprs.append(expr)
|
|
41
|
+
|
|
42
|
+
if group_by:
|
|
43
|
+
return df.group_by(group_by).agg(agg_exprs)
|
|
44
|
+
return df.select(agg_exprs)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def filter_rows(
|
|
48
|
+
df: pl.DataFrame,
|
|
49
|
+
conditions: list[dict],
|
|
50
|
+
logic: str = "and",
|
|
51
|
+
) -> pl.DataFrame:
|
|
52
|
+
if not conditions:
|
|
53
|
+
return df
|
|
54
|
+
|
|
55
|
+
exprs: list[pl.Expr] = []
|
|
56
|
+
for cond in conditions:
|
|
57
|
+
col = pl.col(cond["column"])
|
|
58
|
+
op = cond["operator"]
|
|
59
|
+
val: Any = cond.get("value")
|
|
60
|
+
|
|
61
|
+
match op:
|
|
62
|
+
case "=": expr = col == val
|
|
63
|
+
case "!=": expr = col != val
|
|
64
|
+
case ">": expr = col > val
|
|
65
|
+
case ">=": expr = col >= val
|
|
66
|
+
case "<": expr = col < val
|
|
67
|
+
case "<=": expr = col <= val
|
|
68
|
+
case "contains": expr = col.str.contains(val)
|
|
69
|
+
case "starts_with": expr = col.str.starts_with(val)
|
|
70
|
+
case "ends_with": expr = col.str.ends_with(val)
|
|
71
|
+
case "in": expr = col.is_in(val)
|
|
72
|
+
case "not_in": expr = ~col.is_in(val)
|
|
73
|
+
case "is_null": expr = col.is_null()
|
|
74
|
+
case "is_not_null": expr = col.is_not_null()
|
|
75
|
+
case _: raise ValueError(f"Unknown filter operator {op!r}")
|
|
76
|
+
exprs.append(expr)
|
|
77
|
+
|
|
78
|
+
combined = exprs[0]
|
|
79
|
+
for e in exprs[1:]:
|
|
80
|
+
combined = (combined & e) if logic == "and" else (combined | e)
|
|
81
|
+
return df.filter(combined)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def sort_rows(df: pl.DataFrame, by: list[dict]) -> pl.DataFrame:
|
|
85
|
+
columns = [b["column"] for b in by]
|
|
86
|
+
descending = [b.get("descending", False) for b in by]
|
|
87
|
+
return df.sort(columns, descending=descending)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def join_frames(
|
|
91
|
+
left: pl.DataFrame,
|
|
92
|
+
right: pl.DataFrame,
|
|
93
|
+
left_on: list[str],
|
|
94
|
+
right_on: list[str],
|
|
95
|
+
how: str,
|
|
96
|
+
) -> pl.DataFrame:
|
|
97
|
+
valid = {"inner", "left", "right", "full", "semi", "anti", "cross"}
|
|
98
|
+
if how not in valid:
|
|
99
|
+
raise ValueError(f"Unknown join type {how!r}. Valid: {sorted(valid)}")
|
|
100
|
+
return left.join(right, left_on=left_on, right_on=right_on, how=how) # type: ignore[arg-type]
|
|
@@ -0,0 +1,497 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import shutil
|
|
4
|
+
from datetime import date, datetime
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any, Optional
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
from mcp.server.fastmcp import FastMCP
|
|
10
|
+
|
|
11
|
+
from .store import VirtualFileStore
|
|
12
|
+
from . import _transforms as T
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
mcp = FastMCP(
|
|
16
|
+
"pac-pincers",
|
|
17
|
+
instructions=(
|
|
18
|
+
"Local stdio MCP for in-memory CSV/Excel manipulation. "
|
|
19
|
+
"Files are loaded as 'virtual files' (Polars DataFrames) identified by unique IDs. "
|
|
20
|
+
"Transform tools delete the source virtual file and replace it with the result "
|
|
21
|
+
"unless keep_original=True."
|
|
22
|
+
),
|
|
23
|
+
)
|
|
24
|
+
_store = VirtualFileStore()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# ── helpers ──────────────────────────────────────────────────────────────────
|
|
28
|
+
|
|
29
|
+
def _read_file(path: str) -> pl.DataFrame:
|
|
30
|
+
p = Path(path)
|
|
31
|
+
if not p.exists():
|
|
32
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
33
|
+
ext = p.suffix.lower()
|
|
34
|
+
if ext == ".csv":
|
|
35
|
+
return pl.read_csv(p)
|
|
36
|
+
if ext in (".xlsx", ".xls", ".xlsm"):
|
|
37
|
+
return pl.read_excel(p)
|
|
38
|
+
raise ValueError(f"Unsupported format {ext!r} — use .csv or .xlsx/.xls/.xlsm")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _write_file(df: pl.DataFrame, path: str) -> None:
|
|
42
|
+
p = Path(path)
|
|
43
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
44
|
+
ext = p.suffix.lower()
|
|
45
|
+
if ext == ".csv":
|
|
46
|
+
df.write_csv(p)
|
|
47
|
+
elif ext in (".xlsx", ".xls", ".xlsm"):
|
|
48
|
+
df.write_excel(p)
|
|
49
|
+
else:
|
|
50
|
+
raise ValueError(f"Unsupported format {ext!r} — use .csv or .xlsx")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _safe(v: Any) -> Any:
|
|
54
|
+
if isinstance(v, (datetime, date)):
|
|
55
|
+
return v.isoformat()
|
|
56
|
+
return v
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _paginate(df: pl.DataFrame, offset: int, limit: int) -> dict:
|
|
60
|
+
sliced = df.slice(offset, limit)
|
|
61
|
+
return {
|
|
62
|
+
"total_rows": df.height,
|
|
63
|
+
"offset": offset,
|
|
64
|
+
"limit": limit,
|
|
65
|
+
"rows_returned": sliced.height,
|
|
66
|
+
"columns": list(df.columns),
|
|
67
|
+
"schema": {col: str(dtype) for col, dtype in df.schema.items()},
|
|
68
|
+
"data": [{k: _safe(v) for k, v in row.items()} for row in sliced.to_dicts()],
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _vf_meta(vf_id: str, df: pl.DataFrame) -> dict:
|
|
73
|
+
return {
|
|
74
|
+
"id": vf_id,
|
|
75
|
+
"rows": df.height,
|
|
76
|
+
"columns": list(df.columns),
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
# ── virtual file operations ───────────────────────────────────────────────────
|
|
81
|
+
|
|
82
|
+
@mcp.tool()
|
|
83
|
+
def list_virtual_files() -> list[dict]:
|
|
84
|
+
"""List all virtual files currently held in memory.
|
|
85
|
+
|
|
86
|
+
A virtual file is a named Polars DataFrame kept in server memory.
|
|
87
|
+
Each has a unique ID with format `{name}__{uuid7hex}`.
|
|
88
|
+
Returns metadata only (no row data). Use `get` to retrieve rows.
|
|
89
|
+
|
|
90
|
+
Returns: list of { id, source, rows, columns, schema }
|
|
91
|
+
"""
|
|
92
|
+
return _store.list_all()
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@mcp.tool()
|
|
96
|
+
def load(path: str, id: Optional[str] = None) -> dict:
|
|
97
|
+
"""Load a CSV or Excel file from disk into a new virtual file in memory.
|
|
98
|
+
|
|
99
|
+
Reads the file at `path` into a Polars DataFrame and registers it as a virtual
|
|
100
|
+
file. Supports .csv, .xlsx, .xls, .xlsm. The returned `id` is the handle for
|
|
101
|
+
all subsequent operations on this data. The disk file is not modified.
|
|
102
|
+
|
|
103
|
+
Args:
|
|
104
|
+
path: Absolute or relative path to the source file.
|
|
105
|
+
id: Optional custom ID (must be unique). Auto-generated from filename if omitted.
|
|
106
|
+
|
|
107
|
+
Returns: { id, source, rows, columns, schema }
|
|
108
|
+
"""
|
|
109
|
+
df = _read_file(path)
|
|
110
|
+
vf = _store.add(df, Path(path).stem, id=id, source=str(path))
|
|
111
|
+
return {
|
|
112
|
+
"id": vf.id,
|
|
113
|
+
"source": vf.source,
|
|
114
|
+
"rows": df.height,
|
|
115
|
+
"columns": list(df.columns),
|
|
116
|
+
"schema": {col: str(dtype) for col, dtype in df.schema.items()},
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@mcp.tool()
|
|
121
|
+
def dump(id: str, path: str) -> dict:
|
|
122
|
+
"""Write a virtual file from memory to a disk file.
|
|
123
|
+
|
|
124
|
+
Serialises the virtual file identified by `id` to `path`. Format is inferred from
|
|
125
|
+
the extension: .csv → CSV, .xlsx/.xls/.xlsm → Excel. Parent directories are
|
|
126
|
+
created if missing. The virtual file remains in memory unchanged.
|
|
127
|
+
|
|
128
|
+
Args:
|
|
129
|
+
id: ID of the virtual file to write.
|
|
130
|
+
path: Destination path (extension determines format).
|
|
131
|
+
|
|
132
|
+
Returns: { id, path, rows }
|
|
133
|
+
"""
|
|
134
|
+
vf = _store.get(id)
|
|
135
|
+
_write_file(vf.df, path)
|
|
136
|
+
return {"id": id, "path": str(path), "rows": vf.df.height}
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
@mcp.tool()
|
|
140
|
+
def delete(id: str) -> dict:
|
|
141
|
+
"""Remove a virtual file from memory.
|
|
142
|
+
|
|
143
|
+
Frees the in-memory DataFrame. Does NOT touch any disk file.
|
|
144
|
+
The ID becomes invalid immediately. Use `copy` first if you need to keep the data.
|
|
145
|
+
|
|
146
|
+
Args:
|
|
147
|
+
id: ID of the virtual file to remove.
|
|
148
|
+
|
|
149
|
+
Returns: { deleted_id }
|
|
150
|
+
"""
|
|
151
|
+
_store.delete(id)
|
|
152
|
+
return {"deleted_id": id}
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
@mcp.tool()
|
|
156
|
+
def rename(id: str, new_name: str, new_id: Optional[str] = None) -> dict:
|
|
157
|
+
"""Rename a virtual file — assigns it a new ID, old ID is invalidated.
|
|
158
|
+
|
|
159
|
+
Re-registers the virtual file under a new ID derived from `new_name`
|
|
160
|
+
(format: `new_name__{uuid7hex}`). Supply `new_id` to set the full ID explicitly.
|
|
161
|
+
The DataFrame content is unchanged. The old ID must not be used after this call.
|
|
162
|
+
|
|
163
|
+
Args:
|
|
164
|
+
id: Current ID of the virtual file.
|
|
165
|
+
new_name: Base name for the new auto-generated ID (ignored if new_id is given).
|
|
166
|
+
new_id: Optional fully custom new ID (must be unique).
|
|
167
|
+
|
|
168
|
+
Returns: { old_id, new_id }
|
|
169
|
+
"""
|
|
170
|
+
old_id = id
|
|
171
|
+
vf = _store.rename(id, new_name, new_id=new_id)
|
|
172
|
+
return {"old_id": old_id, "new_id": vf.id}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@mcp.tool()
|
|
176
|
+
def copy(id: str, new_id: Optional[str] = None) -> dict:
|
|
177
|
+
"""Duplicate a virtual file, creating a new independent virtual file.
|
|
178
|
+
|
|
179
|
+
Deep-copies the DataFrame and registers it under a new ID. The original virtual
|
|
180
|
+
file is unchanged.
|
|
181
|
+
|
|
182
|
+
Args:
|
|
183
|
+
id: ID of the virtual file to copy.
|
|
184
|
+
new_id: Optional custom ID for the copy (must be unique).
|
|
185
|
+
|
|
186
|
+
Returns: { source_id, new_id }
|
|
187
|
+
"""
|
|
188
|
+
new_vf = _store.copy(id, new_id=new_id)
|
|
189
|
+
return {"source_id": id, "new_id": new_vf.id}
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
@mcp.tool()
|
|
193
|
+
def get(id: str, offset: int = 0, limit: int = 100) -> dict:
|
|
194
|
+
"""Read paginated rows from a virtual file.
|
|
195
|
+
|
|
196
|
+
Returns at most `limit` rows starting at `offset`. Always includes `total_rows`
|
|
197
|
+
so you can determine whether more pages exist. Advance with offset += limit.
|
|
198
|
+
|
|
199
|
+
Args:
|
|
200
|
+
id: ID of the virtual file.
|
|
201
|
+
offset: First row index to return (0-based, default 0).
|
|
202
|
+
limit: Max rows to return (default 100).
|
|
203
|
+
|
|
204
|
+
Returns: { id, total_rows, offset, limit, rows_returned, columns, schema, data }
|
|
205
|
+
"""
|
|
206
|
+
vf = _store.get(id)
|
|
207
|
+
result = _paginate(vf.df, offset, limit)
|
|
208
|
+
result["id"] = id
|
|
209
|
+
return result
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
# ── local (disk) file operations ─────────────────────────────────────────────
|
|
213
|
+
|
|
214
|
+
@mcp.tool()
|
|
215
|
+
def delete_local(path: str) -> dict:
|
|
216
|
+
"""Delete a file from disk. Does not affect any virtual file in memory.
|
|
217
|
+
|
|
218
|
+
Args:
|
|
219
|
+
path: Path to the file to delete.
|
|
220
|
+
|
|
221
|
+
Returns: { deleted_path }
|
|
222
|
+
"""
|
|
223
|
+
p = Path(path)
|
|
224
|
+
if not p.exists():
|
|
225
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
226
|
+
p.unlink()
|
|
227
|
+
return {"deleted_path": str(path)}
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
@mcp.tool()
|
|
231
|
+
def rename_local(path: str, new_path: str) -> dict:
|
|
232
|
+
"""Rename (move) a file on disk. Does not affect any virtual file in memory.
|
|
233
|
+
|
|
234
|
+
Args:
|
|
235
|
+
path: Current file path.
|
|
236
|
+
new_path: New file path. Parent directory must already exist.
|
|
237
|
+
|
|
238
|
+
Returns: { old_path, new_path }
|
|
239
|
+
"""
|
|
240
|
+
p = Path(path)
|
|
241
|
+
if not p.exists():
|
|
242
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
243
|
+
p.rename(new_path)
|
|
244
|
+
return {"old_path": str(path), "new_path": str(new_path)}
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
@mcp.tool()
|
|
248
|
+
def copy_local(path: str, dest: str) -> dict:
|
|
249
|
+
"""Copy a file on disk to a new location. Does not affect any virtual file in memory.
|
|
250
|
+
|
|
251
|
+
Args:
|
|
252
|
+
path: Source file path.
|
|
253
|
+
dest: Destination file path.
|
|
254
|
+
|
|
255
|
+
Returns: { source, dest }
|
|
256
|
+
"""
|
|
257
|
+
p = Path(path)
|
|
258
|
+
if not p.exists():
|
|
259
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
260
|
+
shutil.copy2(str(p), str(dest))
|
|
261
|
+
return {"source": str(path), "dest": str(dest)}
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
@mcp.tool()
|
|
265
|
+
def get_local(path: str, offset: int = 0, limit: int = 100) -> dict:
|
|
266
|
+
"""Read paginated rows directly from a disk file without creating a virtual file.
|
|
267
|
+
|
|
268
|
+
Reads a CSV or Excel file, returns a page of rows, then discards the data.
|
|
269
|
+
Use this to inspect a file before deciding whether to load it into memory.
|
|
270
|
+
|
|
271
|
+
Args:
|
|
272
|
+
path: Path to the CSV or Excel file.
|
|
273
|
+
offset: First row index to return (0-based, default 0).
|
|
274
|
+
limit: Max rows to return (default 100).
|
|
275
|
+
|
|
276
|
+
Returns: { path, total_rows, offset, limit, rows_returned, columns, schema, data }
|
|
277
|
+
"""
|
|
278
|
+
df = _read_file(path)
|
|
279
|
+
result = _paginate(df, offset, limit)
|
|
280
|
+
result["path"] = str(path)
|
|
281
|
+
return result
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
# ── transform operations ──────────────────────────────────────────────────────
|
|
285
|
+
#
|
|
286
|
+
# All transforms produce a NEW virtual file and return its ID.
|
|
287
|
+
#
|
|
288
|
+
# keep_original=False (default):
|
|
289
|
+
# The source virtual file is DELETED; the result takes a new auto-generated ID
|
|
290
|
+
# (or the ID given via new_id). Use this when you want to replace the source.
|
|
291
|
+
#
|
|
292
|
+
# keep_original=True:
|
|
293
|
+
# The source virtual file is KEPT unchanged; the result is a separate new entry.
|
|
294
|
+
# Supply new_id to control the result's ID; otherwise it is auto-generated.
|
|
295
|
+
#
|
|
296
|
+
# new_id must be unique across all virtual files or the call returns an error.
|
|
297
|
+
# All transforms return { result_id, source_id, rows, columns }.
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
@mcp.tool()
|
|
301
|
+
def aggregate(
|
|
302
|
+
id: str,
|
|
303
|
+
aggregations: list[dict],
|
|
304
|
+
group_by: Optional[list[str]] = None,
|
|
305
|
+
new_id: Optional[str] = None,
|
|
306
|
+
keep_original: bool = False,
|
|
307
|
+
) -> dict:
|
|
308
|
+
"""Group and aggregate a virtual file's rows.
|
|
309
|
+
|
|
310
|
+
Each entry in `aggregations`:
|
|
311
|
+
{ "column": str, "function": str, "alias": str (optional) }
|
|
312
|
+
|
|
313
|
+
Supported functions: sum, mean, min, max, count, std, var, first, last, median, n_unique.
|
|
314
|
+
|
|
315
|
+
`group_by`: list of column names to group by. Omit for a global (single-row) aggregate.
|
|
316
|
+
|
|
317
|
+
Args:
|
|
318
|
+
id: Source virtual file ID.
|
|
319
|
+
aggregations: List of aggregation specs (see above).
|
|
320
|
+
group_by: Columns to group by (optional).
|
|
321
|
+
new_id: Custom ID for the result virtual file (auto-generated if omitted).
|
|
322
|
+
keep_original: Keep source virtual file (default False — source is deleted).
|
|
323
|
+
|
|
324
|
+
Returns: { result_id, source_id, rows, columns }
|
|
325
|
+
"""
|
|
326
|
+
vf = _store.get(id)
|
|
327
|
+
result_df = T.aggregate(vf.df, group_by, aggregations)
|
|
328
|
+
new_vf = _store.apply_transform(id, result_df, new_id=new_id, keep_original=keep_original)
|
|
329
|
+
return _vf_meta(new_vf.id, result_df) | {"source_id": id}
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
@mcp.tool()
|
|
333
|
+
def filter_rows(
|
|
334
|
+
id: str,
|
|
335
|
+
conditions: list[dict],
|
|
336
|
+
logic: str = "and",
|
|
337
|
+
new_id: Optional[str] = None,
|
|
338
|
+
keep_original: bool = False,
|
|
339
|
+
) -> dict:
|
|
340
|
+
"""Filter rows of a virtual file by one or more conditions.
|
|
341
|
+
|
|
342
|
+
Each condition:
|
|
343
|
+
{ "column": str, "operator": str, "value": any }
|
|
344
|
+
Omit "value" for is_null / is_not_null operators.
|
|
345
|
+
|
|
346
|
+
Operators: =, !=, >, >=, <, <=, contains, starts_with, ends_with,
|
|
347
|
+
in, not_in, is_null, is_not_null.
|
|
348
|
+
`in` / `not_in` require value to be a list.
|
|
349
|
+
|
|
350
|
+
`logic`: "and" | "or" — how multiple conditions are combined (default "and").
|
|
351
|
+
|
|
352
|
+
Args:
|
|
353
|
+
id: Source virtual file ID.
|
|
354
|
+
conditions: List of filter conditions (see above).
|
|
355
|
+
logic: Combine conditions with "and" or "or" (default "and").
|
|
356
|
+
new_id: Custom ID for the result virtual file (auto-generated if omitted).
|
|
357
|
+
keep_original: Keep source virtual file (default False — source is deleted).
|
|
358
|
+
|
|
359
|
+
Returns: { result_id, source_id, rows, columns }
|
|
360
|
+
"""
|
|
361
|
+
vf = _store.get(id)
|
|
362
|
+
result_df = T.filter_rows(vf.df, conditions, logic)
|
|
363
|
+
new_vf = _store.apply_transform(id, result_df, new_id=new_id, keep_original=keep_original)
|
|
364
|
+
return _vf_meta(new_vf.id, result_df) | {"source_id": id}
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
@mcp.tool()
|
|
368
|
+
def select_columns(
|
|
369
|
+
id: str,
|
|
370
|
+
columns: list[str],
|
|
371
|
+
new_id: Optional[str] = None,
|
|
372
|
+
keep_original: bool = False,
|
|
373
|
+
) -> dict:
|
|
374
|
+
"""Retain only the specified columns, dropping all others.
|
|
375
|
+
|
|
376
|
+
Args:
|
|
377
|
+
id: Source virtual file ID.
|
|
378
|
+
columns: Ordered list of column names to keep.
|
|
379
|
+
new_id: Custom ID for the result virtual file (auto-generated if omitted).
|
|
380
|
+
keep_original: Keep source virtual file (default False — source is deleted).
|
|
381
|
+
|
|
382
|
+
Returns: { result_id, source_id, rows, columns }
|
|
383
|
+
"""
|
|
384
|
+
vf = _store.get(id)
|
|
385
|
+
result_df = vf.df.select(columns)
|
|
386
|
+
new_vf = _store.apply_transform(id, result_df, new_id=new_id, keep_original=keep_original)
|
|
387
|
+
return _vf_meta(new_vf.id, result_df) | {"source_id": id}
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
@mcp.tool()
|
|
391
|
+
def rename_columns(
|
|
392
|
+
id: str,
|
|
393
|
+
mapping: dict[str, str],
|
|
394
|
+
new_id: Optional[str] = None,
|
|
395
|
+
keep_original: bool = False,
|
|
396
|
+
) -> dict:
|
|
397
|
+
"""Rename one or more columns. Columns not listed in mapping are unchanged.
|
|
398
|
+
|
|
399
|
+
Args:
|
|
400
|
+
id: Source virtual file ID.
|
|
401
|
+
mapping: { "old_name": "new_name", ... }
|
|
402
|
+
new_id: Custom ID for the result virtual file (auto-generated if omitted).
|
|
403
|
+
keep_original: Keep source virtual file (default False — source is deleted).
|
|
404
|
+
|
|
405
|
+
Returns: { result_id, source_id, rows, columns }
|
|
406
|
+
"""
|
|
407
|
+
vf = _store.get(id)
|
|
408
|
+
result_df = vf.df.rename(mapping)
|
|
409
|
+
new_vf = _store.apply_transform(id, result_df, new_id=new_id, keep_original=keep_original)
|
|
410
|
+
return _vf_meta(new_vf.id, result_df) | {"source_id": id}
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
@mcp.tool()
|
|
414
|
+
def sort(
|
|
415
|
+
id: str,
|
|
416
|
+
by: list[dict],
|
|
417
|
+
new_id: Optional[str] = None,
|
|
418
|
+
keep_original: bool = False,
|
|
419
|
+
) -> dict:
|
|
420
|
+
"""Sort rows by one or more columns.
|
|
421
|
+
|
|
422
|
+
Each entry in `by`:
|
|
423
|
+
{ "column": str, "descending": bool (default false) }
|
|
424
|
+
Rows are sorted by the first key, ties broken by subsequent keys.
|
|
425
|
+
|
|
426
|
+
Args:
|
|
427
|
+
id: Source virtual file ID.
|
|
428
|
+
by: List of sort keys (see above).
|
|
429
|
+
new_id: Custom ID for the result virtual file (auto-generated if omitted).
|
|
430
|
+
keep_original: Keep source virtual file (default False — source is deleted).
|
|
431
|
+
|
|
432
|
+
Returns: { result_id, source_id, rows, columns }
|
|
433
|
+
"""
|
|
434
|
+
vf = _store.get(id)
|
|
435
|
+
result_df = T.sort_rows(vf.df, by)
|
|
436
|
+
new_vf = _store.apply_transform(id, result_df, new_id=new_id, keep_original=keep_original)
|
|
437
|
+
return _vf_meta(new_vf.id, result_df) | {"source_id": id}
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
@mcp.tool()
|
|
441
|
+
def join(
|
|
442
|
+
id: str,
|
|
443
|
+
right_id: str,
|
|
444
|
+
left_on: list[str],
|
|
445
|
+
right_on: list[str],
|
|
446
|
+
how: str = "inner",
|
|
447
|
+
new_id: Optional[str] = None,
|
|
448
|
+
keep_original: bool = False,
|
|
449
|
+
) -> dict:
|
|
450
|
+
"""Join two virtual files into a new virtual file.
|
|
451
|
+
|
|
452
|
+
`id` is the left table; `right_id` is the right table. Both must already be
|
|
453
|
+
loaded as virtual files. The right virtual file is never modified or deleted.
|
|
454
|
+
keep_original applies to the left table only.
|
|
455
|
+
|
|
456
|
+
`left_on` / `right_on`: parallel lists of join-key column names.
|
|
457
|
+
`how`: inner | left | right | full | semi | anti | cross (default inner).
|
|
458
|
+
|
|
459
|
+
Args:
|
|
460
|
+
id: Left virtual file ID.
|
|
461
|
+
right_id: Right virtual file ID.
|
|
462
|
+
left_on: Key column(s) from the left table.
|
|
463
|
+
right_on: Key column(s) from the right table (same length as left_on).
|
|
464
|
+
how: Join type (default "inner").
|
|
465
|
+
new_id: Custom ID for the result virtual file (auto-generated if omitted).
|
|
466
|
+
keep_original: Keep the left virtual file (default False — left source is deleted).
|
|
467
|
+
|
|
468
|
+
Returns: { result_id, source_id, right_id, rows, columns }
|
|
469
|
+
"""
|
|
470
|
+
left_vf = _store.get(id)
|
|
471
|
+
right_vf = _store.get(right_id)
|
|
472
|
+
result_df = T.join_frames(left_vf.df, right_vf.df, left_on, right_on, how)
|
|
473
|
+
new_vf = _store.apply_transform(id, result_df, new_id=new_id, keep_original=keep_original)
|
|
474
|
+
return _vf_meta(new_vf.id, result_df) | {"source_id": id, "right_id": right_id}
|
|
475
|
+
|
|
476
|
+
|
|
477
|
+
@mcp.tool()
|
|
478
|
+
def head(
|
|
479
|
+
id: str,
|
|
480
|
+
n: int = 10,
|
|
481
|
+
new_id: Optional[str] = None,
|
|
482
|
+
keep_original: bool = False,
|
|
483
|
+
) -> dict:
|
|
484
|
+
"""Keep only the first N rows of a virtual file.
|
|
485
|
+
|
|
486
|
+
Args:
|
|
487
|
+
id: Source virtual file ID.
|
|
488
|
+
n: Number of rows to keep (default 10).
|
|
489
|
+
new_id: Custom ID for the result virtual file (auto-generated if omitted).
|
|
490
|
+
keep_original: Keep source virtual file (default False — source is deleted).
|
|
491
|
+
|
|
492
|
+
Returns: { result_id, source_id, rows, columns }
|
|
493
|
+
"""
|
|
494
|
+
vf = _store.get(id)
|
|
495
|
+
result_df = vf.df.head(n)
|
|
496
|
+
new_vf = _store.apply_transform(id, result_df, new_id=new_id, keep_original=keep_original)
|
|
497
|
+
return _vf_meta(new_vf.id, result_df) | {"source_id": id}
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
import polars as pl
|
|
8
|
+
from uuid_extensions import uuid7 as _uuid7
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _new_uuid() -> str:
|
|
12
|
+
return _uuid7().hex
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class VirtualFile:
|
|
17
|
+
id: str
|
|
18
|
+
df: pl.DataFrame
|
|
19
|
+
source: Optional[str] = None # original file path, if any
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class VirtualFileStore:
|
|
23
|
+
def __init__(self) -> None:
|
|
24
|
+
self._files: dict[str, VirtualFile] = {}
|
|
25
|
+
|
|
26
|
+
def _gen_id(self, name: str) -> str:
|
|
27
|
+
stem = Path(name).stem if name else "virtual"
|
|
28
|
+
stem = "".join(c if c.isalnum() or c == "_" else "_" for c in stem)
|
|
29
|
+
while True:
|
|
30
|
+
candidate = f"{stem}__{_new_uuid()}"
|
|
31
|
+
if candidate not in self._files:
|
|
32
|
+
return candidate
|
|
33
|
+
|
|
34
|
+
def add(
|
|
35
|
+
self,
|
|
36
|
+
df: pl.DataFrame,
|
|
37
|
+
name: str,
|
|
38
|
+
*,
|
|
39
|
+
id: Optional[str] = None,
|
|
40
|
+
source: Optional[str] = None,
|
|
41
|
+
) -> VirtualFile:
|
|
42
|
+
if id is None:
|
|
43
|
+
id = self._gen_id(name)
|
|
44
|
+
elif id in self._files:
|
|
45
|
+
raise ValueError(f"Virtual file id already exists: {id!r}")
|
|
46
|
+
vf = VirtualFile(id=id, df=df, source=source)
|
|
47
|
+
self._files[id] = vf
|
|
48
|
+
return vf
|
|
49
|
+
|
|
50
|
+
def get(self, id: str) -> VirtualFile:
|
|
51
|
+
try:
|
|
52
|
+
return self._files[id]
|
|
53
|
+
except KeyError:
|
|
54
|
+
raise KeyError(f"Virtual file not found: {id!r}")
|
|
55
|
+
|
|
56
|
+
def delete(self, id: str) -> None:
|
|
57
|
+
self.get(id)
|
|
58
|
+
del self._files[id]
|
|
59
|
+
|
|
60
|
+
def rename(self, id: str, new_name: str, *, new_id: Optional[str] = None) -> VirtualFile:
|
|
61
|
+
"""Re-register the virtual file under a new ID derived from new_name (or custom new_id). Old ID is removed."""
|
|
62
|
+
vf = self.get(id)
|
|
63
|
+
if new_id is None:
|
|
64
|
+
new_id = self._gen_id(new_name)
|
|
65
|
+
elif new_id in self._files:
|
|
66
|
+
raise ValueError(f"Virtual file id already exists: {new_id!r}")
|
|
67
|
+
new_vf = VirtualFile(id=new_id, df=vf.df, source=vf.source)
|
|
68
|
+
self._files[new_id] = new_vf
|
|
69
|
+
del self._files[id]
|
|
70
|
+
return new_vf
|
|
71
|
+
|
|
72
|
+
def copy(self, id: str, *, new_id: Optional[str] = None) -> VirtualFile:
|
|
73
|
+
"""Duplicate a virtual file; original is kept."""
|
|
74
|
+
vf = self.get(id)
|
|
75
|
+
stem = id.split("__")[0]
|
|
76
|
+
if new_id is None:
|
|
77
|
+
new_id = self._gen_id(stem)
|
|
78
|
+
elif new_id in self._files:
|
|
79
|
+
raise ValueError(f"Virtual file id already exists: {new_id!r}")
|
|
80
|
+
new_vf = VirtualFile(id=new_id, df=vf.df.clone(), source=vf.source)
|
|
81
|
+
self._files[new_id] = new_vf
|
|
82
|
+
return new_vf
|
|
83
|
+
|
|
84
|
+
def apply_transform(
|
|
85
|
+
self,
|
|
86
|
+
source_id: str,
|
|
87
|
+
result_df: pl.DataFrame,
|
|
88
|
+
*,
|
|
89
|
+
new_id: Optional[str] = None,
|
|
90
|
+
keep_original: bool = False,
|
|
91
|
+
) -> VirtualFile:
|
|
92
|
+
"""Store result_df as a new virtual file. Removes source_id unless keep_original is True."""
|
|
93
|
+
source = self.get(source_id)
|
|
94
|
+
stem = source_id.split("__")[0]
|
|
95
|
+
new_vf = self.add(result_df, stem, id=new_id, source=source.source)
|
|
96
|
+
if not keep_original:
|
|
97
|
+
del self._files[source_id]
|
|
98
|
+
return new_vf
|
|
99
|
+
|
|
100
|
+
def list_all(self) -> list[dict]:
|
|
101
|
+
return [
|
|
102
|
+
{
|
|
103
|
+
"id": vf.id,
|
|
104
|
+
"source": vf.source,
|
|
105
|
+
"rows": vf.df.height,
|
|
106
|
+
"columns": list(vf.df.columns),
|
|
107
|
+
"schema": {col: str(dtype) for col, dtype in vf.df.schema.items()},
|
|
108
|
+
}
|
|
109
|
+
for vf in self._files.values()
|
|
110
|
+
]
|