duduexcel 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- duduexcel/__init__.py +4 -0
- duduexcel/__main__.py +8 -0
- duduexcel/advanced.py +447 -0
- duduexcel/analytics.py +590 -0
- duduexcel/excel_ops.py +684 -0
- duduexcel/recalc.py +303 -0
- duduexcel/safety.py +163 -0
- duduexcel/server.py +570 -0
- duduexcel/styling.py +259 -0
- duduexcel-0.2.0.dist-info/METADATA +227 -0
- duduexcel-0.2.0.dist-info/RECORD +15 -0
- duduexcel-0.2.0.dist-info/WHEEL +5 -0
- duduexcel-0.2.0.dist-info/entry_points.txt +2 -0
- duduexcel-0.2.0.dist-info/licenses/LICENSE +21 -0
- duduexcel-0.2.0.dist-info/top_level.txt +1 -0
duduexcel/__init__.py
ADDED
duduexcel/__main__.py
ADDED
duduexcel/advanced.py
ADDED
|
@@ -0,0 +1,447 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""
|
|
4
|
+
M6:透视表、条件格式、多表关联。
|
|
5
|
+
|
|
6
|
+
补齐此前 README 中列为"待做"的能力,让 duduExcel 覆盖日常表格处理的主要场景。
|
|
7
|
+
|
|
8
|
+
设计原则(延续前序阶段的调研结论):
|
|
9
|
+
1. **服务端算完只回传结果**:多表关联/比较只返回差异摘要,不把两张表都搬进上下文。
|
|
10
|
+
2. **诚实标注实现边界**:openpyxl 原生不支持创建真正的 PivotTable 对象
|
|
11
|
+
(只能保留已有的),因此这里用"分组聚合 + 写回新表"的方式实现等价效果,
|
|
12
|
+
并在返回中明确说明它生成的是**静态汇总表**而非可交互透视表——
|
|
13
|
+
不把等价物伪装成真透视表(学 knorq 诚实列 Known Limitations 的态度)。
|
|
14
|
+
3. 条件格式用 openpyxl 原生规则(真实生效,非视觉近似)。
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from openpyxl import Workbook, load_workbook
|
|
22
|
+
from openpyxl.formatting.rule import CellIsRule, ColorScaleRule, DataBarRule
|
|
23
|
+
from openpyxl.styles import Alignment, Font, PatternFill
|
|
24
|
+
from openpyxl.utils import get_column_letter
|
|
25
|
+
|
|
26
|
+
from duduexcel.analytics import _load_frame, _py, _r
|
|
27
|
+
|
|
28
|
+
# 条件格式类型
|
|
29
|
+
COND_TYPES = {
|
|
30
|
+
"data_bar": "数据条(长度表示大小)",
|
|
31
|
+
"color_scale": "色阶(颜色深浅表示大小)",
|
|
32
|
+
"greater_than": "大于阈值时高亮",
|
|
33
|
+
"less_than": "小于阈值时高亮",
|
|
34
|
+
"equal": "等于指定值时高亮",
|
|
35
|
+
"between": "介于两值之间时高亮",
|
|
36
|
+
"contains_text": "包含指定文本时高亮",
|
|
37
|
+
"duplicate": "重复值高亮",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class AdvancedError(Exception):
|
|
42
|
+
"""高级操作参数或执行错误。"""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# ---------------------------------------------------------------------------
|
|
46
|
+
# 透视表(静态汇总表,等价物)
|
|
47
|
+
# ---------------------------------------------------------------------------
|
|
48
|
+
def create_pivot(
|
|
49
|
+
file_path: Path,
|
|
50
|
+
source_sheet: str | None,
|
|
51
|
+
rows: list[str],
|
|
52
|
+
values: list[str],
|
|
53
|
+
agg_func: str = "sum",
|
|
54
|
+
columns: list[str] | None = None,
|
|
55
|
+
filters: list[dict] | None = None,
|
|
56
|
+
target_sheet: str = "透视表",
|
|
57
|
+
) -> dict:
|
|
58
|
+
"""生成透视汇总表(写入新工作表)。
|
|
59
|
+
|
|
60
|
+
重要说明:openpyxl 无法创建真正可交互的 PivotTable 对象,
|
|
61
|
+
本工具用"分组聚合 + 写回"实现等价的**静态汇总表**。
|
|
62
|
+
若需要可交互透视表,请在 Excel 中基于结果表插入。
|
|
63
|
+
|
|
64
|
+
参数:
|
|
65
|
+
- rows:行分组字段(可多列)
|
|
66
|
+
- values:要聚合的数值列
|
|
67
|
+
- agg_func:sum/mean/count/min/max
|
|
68
|
+
- columns:列分组字段(可选,做交叉表)
|
|
69
|
+
- target_sheet:结果写入的工作表名(默认"透视表")
|
|
70
|
+
"""
|
|
71
|
+
df, sheet_name, has_pd = _load_frame(file_path, source_sheet)
|
|
72
|
+
if not has_pd:
|
|
73
|
+
raise AdvancedError("create_pivot 需要 pandas,请安装:pip install pandas")
|
|
74
|
+
|
|
75
|
+
missing = [c for c in (rows + values + (columns or [])) if c not in df.columns]
|
|
76
|
+
if missing:
|
|
77
|
+
raise AdvancedError(
|
|
78
|
+
f"列不存在:{missing}。可用列:{', '.join(map(str, df.columns))}"
|
|
79
|
+
)
|
|
80
|
+
if agg_func not in ("sum", "mean", "count", "min", "max"):
|
|
81
|
+
raise AdvancedError(f"不支持的聚合函数 '{agg_func}'。可用:sum/mean/count/min/max")
|
|
82
|
+
|
|
83
|
+
# 过滤
|
|
84
|
+
if filters:
|
|
85
|
+
from duduexcel.analytics import _apply_filters
|
|
86
|
+
|
|
87
|
+
df = _apply_filters(df, filters)
|
|
88
|
+
|
|
89
|
+
# 交叉表:columns 做列维度
|
|
90
|
+
if columns:
|
|
91
|
+
pivot = df.pivot_table(
|
|
92
|
+
index=rows, columns=columns, values=values, aggfunc=agg_func, fill_value=0
|
|
93
|
+
)
|
|
94
|
+
else:
|
|
95
|
+
pivot = df.groupby(rows, dropna=False)[values].agg(agg_func)
|
|
96
|
+
|
|
97
|
+
wb = load_workbook(filename=str(file_path))
|
|
98
|
+
try:
|
|
99
|
+
if target_sheet in wb.sheetnames:
|
|
100
|
+
del wb[target_sheet]
|
|
101
|
+
ws = wb.create_sheet(target_sheet)
|
|
102
|
+
|
|
103
|
+
# 写入表头
|
|
104
|
+
if columns:
|
|
105
|
+
# 多级列:先写列分组,再写值列名
|
|
106
|
+
ws.append(list(rows) + [f"{v}" for _, v in pivot.columns])
|
|
107
|
+
for idx, row_vals in enumerate(pivot.index, start=2):
|
|
108
|
+
key = row_vals if isinstance(row_vals, tuple) else (row_vals,)
|
|
109
|
+
ws.append(list(key) + [_r(v) for v in pivot.iloc[idx - 2].tolist()])
|
|
110
|
+
else:
|
|
111
|
+
ws.append(list(rows) + list(values))
|
|
112
|
+
for idx, row_vals in enumerate(pivot.index, start=2):
|
|
113
|
+
key = row_vals if isinstance(row_vals, tuple) else (row_vals,)
|
|
114
|
+
ws.append(list(key) + [_r(v) for _, v in pivot.iloc[idx - 2].items()])
|
|
115
|
+
|
|
116
|
+
# 表头加粗
|
|
117
|
+
for c in range(1, ws.max_column + 1):
|
|
118
|
+
cell = ws.cell(row=1, column=c)
|
|
119
|
+
cell.font = Font(bold=True)
|
|
120
|
+
cell.alignment = Alignment(horizontal="center")
|
|
121
|
+
|
|
122
|
+
wb.save(str(file_path))
|
|
123
|
+
|
|
124
|
+
return {
|
|
125
|
+
"file": str(file_path),
|
|
126
|
+
"source_sheet": sheet_name,
|
|
127
|
+
"target_sheet": target_sheet,
|
|
128
|
+
"rows": rows,
|
|
129
|
+
"columns": columns,
|
|
130
|
+
"values": values,
|
|
131
|
+
"agg_func": agg_func,
|
|
132
|
+
"result_rows": int(len(pivot)),
|
|
133
|
+
"result_columns": int(pivot.shape[1] if hasattr(pivot, "shape") else len(values)),
|
|
134
|
+
"is_interactive_pivot": False,
|
|
135
|
+
"note": (
|
|
136
|
+
"生成的是**静态汇总表**(openpyxl 无法创建真正可交互的 PivotTable)。"
|
|
137
|
+
"需要可交互透视表时,请在 Excel 中基于本表插入。"
|
|
138
|
+
),
|
|
139
|
+
}
|
|
140
|
+
finally:
|
|
141
|
+
wb.close()
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
# ---------------------------------------------------------------------------
|
|
145
|
+
# 条件格式(openpyxl 原生规则,真实生效)
|
|
146
|
+
# ---------------------------------------------------------------------------
|
|
147
|
+
def add_conditional_format(
|
|
148
|
+
file_path: Path,
|
|
149
|
+
sheet: str | None,
|
|
150
|
+
cell_range: str,
|
|
151
|
+
cond_type: str,
|
|
152
|
+
value: float | str | None = None,
|
|
153
|
+
value2: float | str | None = None,
|
|
154
|
+
color: str = "FFC7CE",
|
|
155
|
+
) -> dict:
|
|
156
|
+
"""给区域添加条件格式。
|
|
157
|
+
|
|
158
|
+
cond_type 可选(见 COND_TYPES):
|
|
159
|
+
- data_bar / color_scale:无需 value
|
|
160
|
+
- greater_than / less_than / equal:需要 value
|
|
161
|
+
- between:需要 value 与 value2
|
|
162
|
+
- contains_text:需要 value(文本)
|
|
163
|
+
- duplicate:无需 value(高亮重复值)
|
|
164
|
+
"""
|
|
165
|
+
ct = (cond_type or "").lower()
|
|
166
|
+
if ct not in COND_TYPES:
|
|
167
|
+
raise AdvancedError(
|
|
168
|
+
f"不支持的条件格式类型 '{cond_type}'。可用:{', '.join(COND_TYPES)}"
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
wb = load_workbook(filename=str(file_path))
|
|
172
|
+
try:
|
|
173
|
+
ws = wb[sheet] if sheet and sheet in wb.sheetnames else wb.active
|
|
174
|
+
sheet_name = ws.title
|
|
175
|
+
|
|
176
|
+
fill = PatternFill(start_color=color, end_color=color, fill_type="solid")
|
|
177
|
+
|
|
178
|
+
if ct == "data_bar":
|
|
179
|
+
rule = DataBarRule(start_type="min", end_type="max", color="638EC6")
|
|
180
|
+
elif ct == "color_scale":
|
|
181
|
+
rule = ColorScaleRule(
|
|
182
|
+
start_type="min", start_color="F8696B",
|
|
183
|
+
mid_type="percentile", mid_value=50, mid_color="FFEB84",
|
|
184
|
+
end_type="max", end_color="63BE7B",
|
|
185
|
+
)
|
|
186
|
+
elif ct == "greater_than":
|
|
187
|
+
if value is None:
|
|
188
|
+
raise AdvancedError("greater_than 需要 value 参数")
|
|
189
|
+
rule = CellIsRule(operator="greaterThan", formula=[str(value)], fill=fill)
|
|
190
|
+
elif ct == "less_than":
|
|
191
|
+
if value is None:
|
|
192
|
+
raise AdvancedError("less_than 需要 value 参数")
|
|
193
|
+
rule = CellIsRule(operator="lessThan", formula=[str(value)], fill=fill)
|
|
194
|
+
elif ct == "equal":
|
|
195
|
+
if value is None:
|
|
196
|
+
raise AdvancedError("equal 需要 value 参数")
|
|
197
|
+
rule = CellIsRule(operator="equal", formula=[str(value)], fill=fill)
|
|
198
|
+
elif ct == "between":
|
|
199
|
+
if value is None or value2 is None:
|
|
200
|
+
raise AdvancedError("between 需要 value 与 value2 两个参数")
|
|
201
|
+
rule = CellIsRule(operator="between", formula=[str(value), str(value2)], fill=fill)
|
|
202
|
+
elif ct == "contains_text":
|
|
203
|
+
if value is None:
|
|
204
|
+
raise AdvancedError("contains_text 需要 value 参数")
|
|
205
|
+
from openpyxl.formatting.rule import FormulaRule
|
|
206
|
+
|
|
207
|
+
first = cell_range.split(":")[0]
|
|
208
|
+
col = "".join(ch for ch in first if ch.isalpha())
|
|
209
|
+
row = "".join(ch for ch in first if ch.isdigit())
|
|
210
|
+
rule = FormulaRule(
|
|
211
|
+
formula=[f'ISNUMBER(SEARCH("{value}",{col}{row}))'], fill=fill
|
|
212
|
+
)
|
|
213
|
+
else: # duplicate
|
|
214
|
+
from openpyxl.formatting.rule import FormulaRule
|
|
215
|
+
|
|
216
|
+
first = cell_range.split(":")[0]
|
|
217
|
+
col = "".join(ch for ch in first if ch.isalpha())
|
|
218
|
+
row = "".join(ch for ch in first if ch.isdigit())
|
|
219
|
+
last = cell_range.split(":")[-1] if ":" in cell_range else first
|
|
220
|
+
last_row = "".join(ch for ch in last if ch.isdigit())
|
|
221
|
+
rule = FormulaRule(
|
|
222
|
+
formula=[
|
|
223
|
+
f"COUNTIF(${col}${row}:${col}${last_row},{col}{row})>1"
|
|
224
|
+
],
|
|
225
|
+
fill=fill,
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
ws.conditional_formatting.add(cell_range, rule)
|
|
229
|
+
wb.save(str(file_path))
|
|
230
|
+
return {
|
|
231
|
+
"file": str(file_path),
|
|
232
|
+
"sheet": sheet_name,
|
|
233
|
+
"range": cell_range,
|
|
234
|
+
"cond_type": ct,
|
|
235
|
+
"description": COND_TYPES[ct],
|
|
236
|
+
"value": value,
|
|
237
|
+
"value2": value2,
|
|
238
|
+
}
|
|
239
|
+
finally:
|
|
240
|
+
wb.close()
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
# ---------------------------------------------------------------------------
|
|
244
|
+
# 条件格式读取
|
|
245
|
+
# ---------------------------------------------------------------------------
|
|
246
|
+
def list_conditional_formats(file_path: Path, sheet: str | None = None) -> dict:
|
|
247
|
+
"""读取工作表中已有的条件格式规则。
|
|
248
|
+
|
|
249
|
+
说明:调研时竞品 knorq 的 Known Limitations 写着"不支持条件格式",
|
|
250
|
+
官方文档也常称 openpyxl 读取条件格式受限;但**实测是可以完整读回的**
|
|
251
|
+
(范围 / 类型 / 运算符 / 阈值 / 填充色 / 优先级都能拿到),因此这里补上读取能力,
|
|
252
|
+
形成"写入 + 读取"的闭环。
|
|
253
|
+
|
|
254
|
+
用途:
|
|
255
|
+
- 接手一张别人的表时,先看清它埋了哪些规则(哪些格子会自动变红/变色)
|
|
256
|
+
- 修改前确认不破坏既有规则
|
|
257
|
+
"""
|
|
258
|
+
from openpyxl import load_workbook
|
|
259
|
+
|
|
260
|
+
wb = load_workbook(filename=str(file_path))
|
|
261
|
+
try:
|
|
262
|
+
ws = wb[sheet] if sheet and sheet in wb.sheetnames else wb.active
|
|
263
|
+
sheet_name = ws.title
|
|
264
|
+
|
|
265
|
+
rules: list[dict[str, Any]] = []
|
|
266
|
+
try:
|
|
267
|
+
for rng in ws.conditional_formatting:
|
|
268
|
+
for rule in rng.rules:
|
|
269
|
+
item: dict[str, Any] = {
|
|
270
|
+
"range": str(rng.sqref),
|
|
271
|
+
"type": rule.type,
|
|
272
|
+
"operator": getattr(rule, "operator", None),
|
|
273
|
+
"formula": list(getattr(rule, "formula", None) or []),
|
|
274
|
+
"priority": getattr(rule, "priority", None),
|
|
275
|
+
}
|
|
276
|
+
# 填充色(高亮类规则才有 dxf)
|
|
277
|
+
dxf = getattr(rule, "dxf", None)
|
|
278
|
+
if dxf is not None:
|
|
279
|
+
fill = getattr(dxf, "fill", None)
|
|
280
|
+
if fill is not None:
|
|
281
|
+
color = getattr(getattr(fill, "bgColor", None), "rgb", None)
|
|
282
|
+
if color:
|
|
283
|
+
item["fill_color"] = color
|
|
284
|
+
|
|
285
|
+
# 关键:构造完必须加入结果列表(此前漏掉这行导致永远返回 0 条)
|
|
286
|
+
rules.append(item)
|
|
287
|
+
|
|
288
|
+
# 按优先级排序,便于阅读
|
|
289
|
+
rules.sort(key=lambda r: (r.get("priority") is None, r.get("priority") or 0))
|
|
290
|
+
except Exception as e:
|
|
291
|
+
return {
|
|
292
|
+
"file": str(file_path),
|
|
293
|
+
"sheet": sheet_name,
|
|
294
|
+
"count": 0,
|
|
295
|
+
"rules": [],
|
|
296
|
+
"note": f"读取条件格式失败:{e}",
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
return {
|
|
300
|
+
"file": str(file_path),
|
|
301
|
+
"sheet": sheet_name,
|
|
302
|
+
"count": len(rules),
|
|
303
|
+
"rules": rules,
|
|
304
|
+
"note": (
|
|
305
|
+
f"该工作表共有 {len(rules)} 条条件格式规则。修改这些区域前请先确认不会破坏既有规则。"
|
|
306
|
+
if rules else "该工作表没有条件格式规则"
|
|
307
|
+
),
|
|
308
|
+
}
|
|
309
|
+
finally:
|
|
310
|
+
wb.close()
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
# ---------------------------------------------------------------------------
|
|
314
|
+
# 多表关联:比较 / 连接
|
|
315
|
+
# ---------------------------------------------------------------------------
|
|
316
|
+
def compare_sheets(
|
|
317
|
+
file_path: Path,
|
|
318
|
+
sheet1: str,
|
|
319
|
+
sheet2: str,
|
|
320
|
+
key_column: str,
|
|
321
|
+
compare_columns: list[str] | None = None,
|
|
322
|
+
max_diff: int = 50,
|
|
323
|
+
) -> dict:
|
|
324
|
+
"""按关键列比对两个工作表的差异(只回传差异摘要,不回传整表)。
|
|
325
|
+
|
|
326
|
+
用途:版本对比、变更检测、对账(学 jwadow 的 compare_sheets)。
|
|
327
|
+
|
|
328
|
+
返回:仅在表1/仅在表2/值有差异 三类统计 + 最多 max_diff 条差异明细。
|
|
329
|
+
"""
|
|
330
|
+
df1, s1, has_pd = _load_frame(file_path, sheet1)
|
|
331
|
+
df2, s2, _ = _load_frame(file_path, sheet2)
|
|
332
|
+
if not has_pd:
|
|
333
|
+
raise AdvancedError("compare_sheets 需要 pandas,请安装:pip install pandas")
|
|
334
|
+
if key_column not in df1.columns or key_column not in df2.columns:
|
|
335
|
+
raise AdvancedError(
|
|
336
|
+
f"关键列 '{key_column}' 必须在两个表中都存在。"
|
|
337
|
+
f"{s1} 的列:{', '.join(map(str, df1.columns))};"
|
|
338
|
+
f"{s2} 的列:{', '.join(map(str, df2.columns))}"
|
|
339
|
+
)
|
|
340
|
+
|
|
341
|
+
compare_columns = compare_columns or [
|
|
342
|
+
c for c in df1.columns if c in df2.columns and c != key_column
|
|
343
|
+
]
|
|
344
|
+
|
|
345
|
+
m1 = df1.set_index(key_column)
|
|
346
|
+
m2 = df2.set_index(key_column)
|
|
347
|
+
keys1, keys2 = set(m1.index), set(m2.index)
|
|
348
|
+
|
|
349
|
+
only1 = sorted(keys1 - keys2, key=str)
|
|
350
|
+
only2 = sorted(keys2 - keys1, key=str)
|
|
351
|
+
common = sorted(keys1 & keys2, key=str)
|
|
352
|
+
|
|
353
|
+
diffs = []
|
|
354
|
+
for k in common:
|
|
355
|
+
for col in compare_columns:
|
|
356
|
+
v1 = _py(m1.at[k, col]) if col in m1.columns else None
|
|
357
|
+
v2 = _py(m2.at[k, col]) if col in m2.columns else None
|
|
358
|
+
if v1 != v2:
|
|
359
|
+
diffs.append({"key": _py(k), "column": col, "left": v1, "right": v2})
|
|
360
|
+
|
|
361
|
+
truncated = len(diffs) > max_diff
|
|
362
|
+
return {
|
|
363
|
+
"file": str(file_path),
|
|
364
|
+
"left_sheet": s1,
|
|
365
|
+
"right_sheet": s2,
|
|
366
|
+
"key_column": key_column,
|
|
367
|
+
"compared_columns": compare_columns,
|
|
368
|
+
"left_rows": len(df1),
|
|
369
|
+
"right_rows": len(df2),
|
|
370
|
+
"only_in_left": len(only1),
|
|
371
|
+
"only_in_right": len(only2),
|
|
372
|
+
"value_differences": len(diffs),
|
|
373
|
+
"only_in_left_sample": [_py(k) for k in only1[:max_diff]],
|
|
374
|
+
"only_in_right_sample": [_py(k) for k in only2[:max_diff]],
|
|
375
|
+
"differences": diffs[:max_diff],
|
|
376
|
+
"differences_truncated": truncated,
|
|
377
|
+
"note": (
|
|
378
|
+
f"差异明细仅展示前 {max_diff} 条(共 {len(diffs)} 条)。"
|
|
379
|
+
"请以 value_differences 为准判断差异规模。"
|
|
380
|
+
if truncated
|
|
381
|
+
else None
|
|
382
|
+
),
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def join_sheets(
|
|
387
|
+
file_path: Path,
|
|
388
|
+
left_sheet: str,
|
|
389
|
+
right_sheet: str,
|
|
390
|
+
on: str,
|
|
391
|
+
how: str = "left",
|
|
392
|
+
columns: list[str] | None = None,
|
|
393
|
+
limit: int = 20,
|
|
394
|
+
) -> dict:
|
|
395
|
+
"""关联两个工作表(类似 SQL JOIN),只回传前 limit 行结果。
|
|
396
|
+
|
|
397
|
+
参数:
|
|
398
|
+
- on:关联键列名(两表都要有)
|
|
399
|
+
- how:left/right/inner/outer(默认 left)
|
|
400
|
+
- columns:只返回这些列(省 token),省略返回全部
|
|
401
|
+
- limit:返回行数上限(默认 20)
|
|
402
|
+
"""
|
|
403
|
+
df1, s1, has_pd = _load_frame(file_path, left_sheet)
|
|
404
|
+
df2, s2, _ = _load_frame(file_path, right_sheet)
|
|
405
|
+
if not has_pd:
|
|
406
|
+
raise AdvancedError("join_sheets 需要 pandas,请安装:pip install pandas")
|
|
407
|
+
if on not in df1.columns or on not in df2.columns:
|
|
408
|
+
raise AdvancedError(
|
|
409
|
+
f"关联键 '{on}' 必须在两个表中都存在。"
|
|
410
|
+
f"{s1} 的列:{', '.join(map(str, df1.columns))};"
|
|
411
|
+
f"{s2} 的列:{', '.join(map(str, df2.columns))}"
|
|
412
|
+
)
|
|
413
|
+
if how not in ("left", "right", "inner", "outer"):
|
|
414
|
+
raise AdvancedError(f"不支持的关联方式 '{how}'。可用:left/right/inner/outer")
|
|
415
|
+
|
|
416
|
+
merged = df1.merge(df2, on=on, how=how, suffixes=("_left", "_right"))
|
|
417
|
+
total = len(merged)
|
|
418
|
+
|
|
419
|
+
keep = [c for c in (columns or []) if c in merged.columns]
|
|
420
|
+
view = merged[keep] if keep else merged
|
|
421
|
+
rows = [
|
|
422
|
+
{str(k): _py(v) for k, v in rec.items()}
|
|
423
|
+
for _, rec in view.head(limit).iterrows()
|
|
424
|
+
]
|
|
425
|
+
|
|
426
|
+
from duduexcel.analytics import _meta, _to_tsv
|
|
427
|
+
|
|
428
|
+
result = {
|
|
429
|
+
"file": str(file_path),
|
|
430
|
+
"left_sheet": s1,
|
|
431
|
+
"right_sheet": s2,
|
|
432
|
+
"on": on,
|
|
433
|
+
"how": how,
|
|
434
|
+
"total_rows": total,
|
|
435
|
+
"returned": len(rows),
|
|
436
|
+
"truncated": total > len(rows),
|
|
437
|
+
"columns": list(view.columns),
|
|
438
|
+
"rows": rows,
|
|
439
|
+
"tsv": _to_tsv(rows),
|
|
440
|
+
"note": (
|
|
441
|
+
f"仅返回前 {len(rows)} 行(共 {total} 行)。如需更多请用 read_range 读取结果表。"
|
|
442
|
+
if total > len(rows)
|
|
443
|
+
else None
|
|
444
|
+
),
|
|
445
|
+
}
|
|
446
|
+
result["_meta"] = _meta(total, len(view.columns), result)
|
|
447
|
+
return result
|