gongwen-skill 2.2.0 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +6 -6
- package/dsh/index.js +2 -2
- package/engine/core/document/_generator_helpers.py +61 -0
- package/engine/core/document/generator.py +63 -36
- package/engine/core/document/markdown_converter.py +10 -0
- package/engine/core/document/models.py +4 -0
- package/engine/core/document/modifier.py +89 -22
- package/engine/core/document/parser.py +95 -22
- package/engine/core/document/parser_format.py +5 -2
- package/engine/core/rules/checker.py +134 -5
- package/engine/core/rules/fixer.py +62 -0
- package/gongwen/__init__.py +1 -1
- package/gongwen/_legacy.py +241 -39
- package/gongwen/cli/doctor_cmds.py +157 -4
- package/gongwen/md2docx_render.py +87 -9
- package/package.json +12 -2
- package/prompts/usage-prompts.md +1 -1
- package/pyproject.toml +1 -1
- package/rules/official/_common.yaml +112 -2
|
@@ -320,27 +320,63 @@ def _assign_paragraph_roles(paragraphs: list[Paragraph]) -> None:
|
|
|
320
320
|
para.role = 'cc'
|
|
321
321
|
continue
|
|
322
322
|
|
|
323
|
-
# 落款:日期前的短文本(< 20
|
|
323
|
+
# 落款:日期前的短文本(< 20 字,或含机关关键词)——支持多行落款单位
|
|
324
|
+
# V2.3 修复:原实现只识别日期前一段;长单位名跨两行(如"XX民族传统体育运动会"换行"XX办公室")
|
|
325
|
+
# 时只标最后一行,第一行被当 body 应用正文缩进。现从日期前一段向上连续扫描落款单位行。
|
|
326
|
+
# 收集附件说明内容("附件:xxx"),用于把"附件标题"与"落款单位"区分开:
|
|
327
|
+
# 附件标题(heading 且带分页,或文本与附件说明一致)不是落款,扫描到此为止;
|
|
328
|
+
# 被 B-0 启发式误判为标题的居中落款单位行(如"XX民族传统体育运动会")则照常标为 signature。
|
|
329
|
+
_att_note_items: list[str] = []
|
|
330
|
+
for _p in paragraphs:
|
|
331
|
+
_pt = (_p.text or '').strip()
|
|
332
|
+
if _pt.startswith(('附件:', '附:')):
|
|
333
|
+
_rest = re.sub(r'^附[::]', '', _pt)
|
|
334
|
+
for _seg in re.split(r'[、,]', _rest):
|
|
335
|
+
_seg = re.sub(r'^d+[.、]', '', _seg).strip()
|
|
336
|
+
if _seg:
|
|
337
|
+
_att_note_items.append(_seg)
|
|
338
|
+
|
|
324
339
|
date_indices = [i for i, p in non_empty if p.role == 'date']
|
|
325
340
|
if date_indices:
|
|
326
341
|
date_idx = date_indices[-1]
|
|
327
|
-
for i in
|
|
328
|
-
|
|
329
|
-
|
|
342
|
+
date_pos = next((i for i, (idx, p) in enumerate(non_empty) if idx == date_idx), -1)
|
|
343
|
+
if date_pos > 0:
|
|
344
|
+
for i in range(date_pos - 1, -1, -1):
|
|
345
|
+
idx, para = non_empty[i]
|
|
330
346
|
text = para.text.strip()
|
|
347
|
+
# 已识别其他角色(如抄送/印发/attachment)→ 落款区到此为止
|
|
348
|
+
if para.role:
|
|
349
|
+
break
|
|
350
|
+
# 句末标点结尾 = 正文收尾,不是落款单位
|
|
351
|
+
if not text or text[-1] in '。?!;,、':
|
|
352
|
+
break
|
|
353
|
+
# 附件说明/附件标题("附件:"开头)不是落款
|
|
354
|
+
if text.startswith(('附件', '附:')):
|
|
355
|
+
break
|
|
331
356
|
is_sig = False
|
|
332
357
|
if len(text) < 20:
|
|
333
358
|
is_sig = True
|
|
334
359
|
elif any(kw in text for kw in ['人民政府', '委员会', '办公厅', '办公室', '管理局', '局', '部']):
|
|
335
360
|
if len(text) < 40:
|
|
336
361
|
is_sig = True
|
|
337
|
-
if is_sig:
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
362
|
+
if not is_sig:
|
|
363
|
+
break
|
|
364
|
+
# 续行是标题段:仅当是"附件标题"(带分页 或 文本与附件说明一致)才停止,
|
|
365
|
+
# 否则(如 B-0 误判的居中落款单位)继续标为 signature 并清除标题标记
|
|
366
|
+
if i < date_pos - 1 and para.is_heading:
|
|
367
|
+
_pb = bool(getattr(para, "page_break", False))
|
|
368
|
+
_matches_att = any(
|
|
369
|
+
_seg and (_seg == text or _seg.endswith(text))
|
|
370
|
+
for _seg in _att_note_items
|
|
371
|
+
)
|
|
372
|
+
if _pb or _matches_att:
|
|
373
|
+
break
|
|
374
|
+
para.role = 'signature'
|
|
375
|
+
# P4 修复:居中署名段(如"陈龙")可能被 B-0 启发式误判为标题,
|
|
376
|
+
# 落款区识别后清除标题标记,避免 optimize 套用标题格式破坏署名段
|
|
377
|
+
if para.is_heading:
|
|
378
|
+
para.is_heading = False
|
|
379
|
+
para.heading_level = None
|
|
344
380
|
|
|
345
381
|
# 主送机关:标题后第一段,以冒号结尾
|
|
346
382
|
if len(non_empty) >= 2:
|
|
@@ -352,10 +388,15 @@ def _assign_paragraph_roles(paragraphs: list[Paragraph]) -> None:
|
|
|
352
388
|
elif any(kw in text for kw in _RECIPIENT_KEYWORDS) and text.endswith((':', ':')):
|
|
353
389
|
second_para.role = 'recipient'
|
|
354
390
|
|
|
355
|
-
#
|
|
391
|
+
# 附件(区分附件说明 vs 附件页标题)
|
|
392
|
+
# 附件标题(# 附件:,heading level 0,如"附件:报名表")应标为 title,
|
|
393
|
+
# 而非 attachment——否则 CHK-C027 会误把标题当附件说明查"左空二字"缩进。
|
|
356
394
|
for idx, para in non_empty:
|
|
357
395
|
if not para.role and _ATTACHMENT_RE.match(para.text.strip()):
|
|
358
|
-
para.
|
|
396
|
+
if para.is_heading and para.heading_level == 0:
|
|
397
|
+
para.role = 'title'
|
|
398
|
+
else:
|
|
399
|
+
para.role = 'attachment'
|
|
359
400
|
|
|
360
401
|
# AI 声明段(末尾批注,如"(内容由GongWen-skill-AI生成,仅供参考)")
|
|
361
402
|
# 标记为 annotation 角色,避免 check 将其误判为正文并报格式违规
|
|
@@ -506,7 +547,10 @@ def parse_docx(file_path: Path | str) -> DocumentModel:
|
|
|
506
547
|
from lxml import etree
|
|
507
548
|
from engine.core.document.ooxml_parser import OOXMLParser # I7: 集成 OOXMLParser
|
|
508
549
|
para_count = 0
|
|
509
|
-
|
|
550
|
+
# {table_element: last_para_index_before_it}
|
|
551
|
+
# 注意:必须用 lxml 元素对象本身作 key,不能用 id(child)——body 迭代产生的
|
|
552
|
+
# 临时代理对象 id 会被复用(多表格时键冲突/错位,锚点全部落回 -1)。
|
|
553
|
+
table_position_map = {}
|
|
510
554
|
ooxml = OOXMLParser()
|
|
511
555
|
para_index_map = ooxml.get_paragraph_index_map(str(file_path)) # I7: 段落索引映射
|
|
512
556
|
for child in doc.element.body:
|
|
@@ -514,14 +558,14 @@ def parse_docx(file_path: Path | str) -> DocumentModel:
|
|
|
514
558
|
if tag == 'p':
|
|
515
559
|
para_count += 1
|
|
516
560
|
elif tag == 'tbl':
|
|
517
|
-
table_position_map[
|
|
561
|
+
table_position_map[child] = para_count - 1 # 紧跟在哪个段落之后
|
|
518
562
|
|
|
519
563
|
tables = []
|
|
520
564
|
for idx, table in enumerate(doc.tables):
|
|
521
565
|
parsed_table = _parse_table(table, idx)
|
|
522
|
-
# 计算 insert_after_index
|
|
566
|
+
# 计算 insert_after_index(用元素对象查表,与上面 body 遍历的 key 一致)
|
|
523
567
|
tbl_elem = table._tbl
|
|
524
|
-
parsed_table.insert_after_index = table_position_map.get(
|
|
568
|
+
parsed_table.insert_after_index = table_position_map.get(tbl_elem, -1)
|
|
525
569
|
tables.append(parsed_table)
|
|
526
570
|
|
|
527
571
|
model = DocumentModel(
|
|
@@ -771,6 +815,9 @@ def _parse_paragraph(para, index: int) -> Paragraph:
|
|
|
771
815
|
# 公文大标题(方正小标宋简体)优先级更高
|
|
772
816
|
heading_level = 0
|
|
773
817
|
|
|
818
|
+
# 段前分页(--- 附件分页标记):读取 w:pageBreakBefore
|
|
819
|
+
page_break = bool(getattr(para.paragraph_format, "page_break_before", False))
|
|
820
|
+
|
|
774
821
|
return Paragraph(
|
|
775
822
|
text=para.text,
|
|
776
823
|
index=index,
|
|
@@ -779,6 +826,7 @@ def _parse_paragraph(para, index: int) -> Paragraph:
|
|
|
779
826
|
heading_level=heading_level,
|
|
780
827
|
format=para_format,
|
|
781
828
|
runs=runs,
|
|
829
|
+
page_break=page_break,
|
|
782
830
|
)
|
|
783
831
|
|
|
784
832
|
|
|
@@ -819,16 +867,27 @@ _PRINT_RE = re.compile(r'^印发机关|^印发日期')
|
|
|
819
867
|
|
|
820
868
|
def _parse_table(table, index: int) -> Table:
|
|
821
869
|
"""Parse a table with cell-level paragraph preservation."""
|
|
870
|
+
from docx.oxml.ns import qn as _qn
|
|
822
871
|
cells = []
|
|
823
872
|
# P1-5 修复:合并单元格时 python-docx 对同一 XML 元素返回多个 cell 对象,
|
|
824
|
-
#
|
|
825
|
-
|
|
873
|
+
# 用元素对象本身去重,避免重复添加破坏写回。
|
|
874
|
+
# V2.3 修复:此前用 id(cell._element) 去重——lxml 代理对象 id 会被 GC 复用,
|
|
875
|
+
# 非合并表格也会误丢大量单元格(实测 88 个误丢 51 个),改用元素对象做 key
|
|
876
|
+
# (与 parse_docx 的 table_position_map[child] 同法,元素对象哈希基于节点身份)。
|
|
877
|
+
seen_cells: set = set()
|
|
826
878
|
for row_idx, row in enumerate(table.rows):
|
|
827
879
|
for col_idx, cell in enumerate(row.cells):
|
|
828
|
-
|
|
829
|
-
if
|
|
880
|
+
cell_elem = cell._element
|
|
881
|
+
if cell_elem in seen_cells:
|
|
830
882
|
continue
|
|
831
|
-
seen_cells.add(
|
|
883
|
+
seen_cells.add(cell_elem)
|
|
884
|
+
# V2.3:表头单元格底色(w:shd fill)——供表格样式检测/修复
|
|
885
|
+
cell_fill = None
|
|
886
|
+
_tcPr = cell._tc.tcPr
|
|
887
|
+
if _tcPr is not None:
|
|
888
|
+
_shd = _tcPr.find(_qn('w:shd'))
|
|
889
|
+
if _shd is not None:
|
|
890
|
+
cell_fill = _shd.get(_qn('w:fill')) or None
|
|
832
891
|
# Parse cell paragraphs
|
|
833
892
|
cell_paras = []
|
|
834
893
|
for p_idx, para in enumerate(cell.paragraphs):
|
|
@@ -839,13 +898,27 @@ def _parse_table(table, index: int) -> Table:
|
|
|
839
898
|
row=row_idx,
|
|
840
899
|
col=col_idx,
|
|
841
900
|
paragraphs=cell_paras,
|
|
901
|
+
fill=cell_fill,
|
|
842
902
|
))
|
|
843
903
|
|
|
904
|
+
# V2.3:单元格边距(w:tblCellMar,twips)——供表格样式检测/修复
|
|
905
|
+
cell_margin = None
|
|
906
|
+
_tblPr = table._tbl.tblPr
|
|
907
|
+
if _tblPr is not None:
|
|
908
|
+
_cm = _tblPr.find(_qn('w:tblCellMar'))
|
|
909
|
+
if _cm is not None:
|
|
910
|
+
cell_margin = {}
|
|
911
|
+
for _edge in ('top', 'left', 'bottom', 'right'):
|
|
912
|
+
_el = _cm.find(_qn('w:' + _edge))
|
|
913
|
+
if _el is not None:
|
|
914
|
+
cell_margin[_edge] = int(_el.get(_qn('w:w'), 0) or 0)
|
|
915
|
+
|
|
844
916
|
return Table(
|
|
845
917
|
index=index,
|
|
846
918
|
rows=len(table.rows),
|
|
847
919
|
cols=len(table.columns) if table.rows else 0,
|
|
848
920
|
cells=cells,
|
|
921
|
+
cell_margin=cell_margin,
|
|
849
922
|
)
|
|
850
923
|
|
|
851
924
|
# 已迁移至 parser_format.py
|
|
@@ -121,7 +121,10 @@ def parse_run(run, index: int) -> Run:
|
|
|
121
121
|
color_rgb = str(font.color.rgb)
|
|
122
122
|
|
|
123
123
|
# 通过 XML 检测删除线(python-docx 1.2.0 无 font.strikethrough)
|
|
124
|
-
#
|
|
124
|
+
# V2.3 修复:w:val 布尔取值须处理 OOXML 全套写法(true/1/on 为真;
|
|
125
|
+
# false/0/off 为假)。原实现只认 'false',导致 <w:strike w:val="0"/>
|
|
126
|
+
# (无删除线)被误判为真删除线,optimize 的 clean_path_b_markers 会
|
|
127
|
+
# 借此删掉标题/正文 run,进而丢失字体(全部回退为仿宋_GB2312)。
|
|
125
128
|
_strikethrough = False
|
|
126
129
|
ns = '{http://schemas.openxmlformats.org/wordprocessingml/2006/main}'
|
|
127
130
|
rPr = run._element.find(f'{ns}rPr')
|
|
@@ -129,7 +132,7 @@ def parse_run(run, index: int) -> Run:
|
|
|
129
132
|
strike_elem = rPr.find(f'{ns}strike')
|
|
130
133
|
if strike_elem is not None:
|
|
131
134
|
val = strike_elem.get(f'{ns}val', 'true')
|
|
132
|
-
_strikethrough = val.lower()
|
|
135
|
+
_strikethrough = val.strip().lower() not in ('false', '0', 'off')
|
|
133
136
|
|
|
134
137
|
return Run(
|
|
135
138
|
text=run.text,
|
|
@@ -102,6 +102,11 @@ def check_document(model: DocumentModel, rules: dict[str, Any]) -> list[CheckIss
|
|
|
102
102
|
# 此前无分发分支,规则定义但从不执行
|
|
103
103
|
issues.extend(_check_header_field(model, rule_id, severity, name, field_path,
|
|
104
104
|
expected, message))
|
|
105
|
+
elif field_path.startswith("table."):
|
|
106
|
+
# V2.3:table.* 表格样式检查(表头字体/字号/加粗/对齐/底色、表体字体/字号、
|
|
107
|
+
# 单元格边距)——此前 _common.yaml 的 table 配置块为死配置,无任何 CHK 规则
|
|
108
|
+
issues.extend(_check_table_style(model, rule_id, severity, name, field_path,
|
|
109
|
+
expected, message))
|
|
105
110
|
else:
|
|
106
111
|
# P1-6 修复:删除 generic else 中的硬编码索引逻辑(model.paragraphs[0]/[1]
|
|
107
112
|
# 不一定是标题/正文,检查结果会指向错误段落),未识别的 field 直接 skip + warning
|
|
@@ -348,7 +353,11 @@ def _check_body(model, rule_id, severity, name, field_path, expected, message) -
|
|
|
348
353
|
issues = []
|
|
349
354
|
# 顶格左对齐的段落(主送机关/称呼段、署名、日期、AI 声明批注)不属于正文,
|
|
350
355
|
# 不应套用正文的缩进/对齐/字体检查
|
|
351
|
-
|
|
356
|
+
# V2.3 修复:附件说明(attachment)有自己的格式规范(CHK-C027 左空二字),
|
|
357
|
+
# 不属于正文——若纳入正文行距检查,会报出 FIX-C015(target=body,仅 role=='body')
|
|
358
|
+
# 修不到的问题,形成"检查报错但优化不动"的不一致。
|
|
359
|
+
_EXCLUDE_ROLES = {'signature', 'date', 'annotation', 'notes', 'recipient',
|
|
360
|
+
'salutation', 'attachment'}
|
|
352
361
|
body_paras = [p for p in model.paragraphs
|
|
353
362
|
if not p.is_heading and p.text.strip() and p.role not in _EXCLUDE_ROLES]
|
|
354
363
|
if not body_paras:
|
|
@@ -971,16 +980,24 @@ def _check_common_issues(model: DocumentModel) -> list[CheckIssue]:
|
|
|
971
980
|
))
|
|
972
981
|
|
|
973
982
|
# Extra blank lines (empty paragraphs)
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
983
|
+
# V2.3 修复:连续 2 个空行是规范允许的(blank_line_rules.body_to_signature=2,
|
|
984
|
+
# 附件说明/正文与落款之间空 2 行);仅当连续空行 >= 3 时,第 3 个起才算多余。
|
|
985
|
+
# 原实现把任意连续 2 空行都报"多余空行",与规范冲突(误报)。
|
|
986
|
+
if not text.strip() and para.index > 1:
|
|
987
|
+
# 连续空行数:从本段向前数(含本段)
|
|
988
|
+
run_len = 0
|
|
989
|
+
j = para.index
|
|
990
|
+
while j >= 0 and j < len(model.paragraphs) and not model.paragraphs[j].text.strip():
|
|
991
|
+
run_len += 1
|
|
992
|
+
j -= 1
|
|
993
|
+
if run_len >= 3:
|
|
977
994
|
issues.append(CheckIssue(
|
|
978
995
|
rule_id="CHK-HEUR-002", check_type="format", severity="P2",
|
|
979
996
|
name="多余空行",
|
|
980
997
|
location=f"paragraph:{para.index}",
|
|
981
998
|
original_text="(空行)",
|
|
982
999
|
suggested_fix="移除多余空行",
|
|
983
|
-
reason="
|
|
1000
|
+
reason="连续出现多个空行(规范允许落款前 2 空行)",
|
|
984
1001
|
))
|
|
985
1002
|
|
|
986
1003
|
# --- 页码检查(GB/T 9704: 公文应标注页码)---
|
|
@@ -1001,3 +1018,115 @@ def _check_common_issues(model: DocumentModel) -> list[CheckIssue]:
|
|
|
1001
1018
|
))
|
|
1002
1019
|
|
|
1003
1020
|
return issues
|
|
1021
|
+
|
|
1022
|
+
|
|
1023
|
+
# ---------------------------------------------------------------------------
|
|
1024
|
+
# Table style checks (V2.3)
|
|
1025
|
+
# ---------------------------------------------------------------------------
|
|
1026
|
+
|
|
1027
|
+
def _check_table_style(model, rule_id, severity, name, field_path, expected, message) -> list:
|
|
1028
|
+
"""检查表格具体样式(field: table.header.* / table.body.* / table.cell_margin.*)。
|
|
1029
|
+
|
|
1030
|
+
- header.font / header.size / header.bold / header.align:取自表头行(row==0)首段首 run
|
|
1031
|
+
- header.fill:取自 TableCell.fill(parser 解析 w:shd)
|
|
1032
|
+
- body.font / body.size:取自数据行(row>0)非空单元格首段首 run
|
|
1033
|
+
- cell_margin.left/right/top/bottom:取自 Table.cell_margin(parser 解析 w:tblCellMar)
|
|
1034
|
+
"""
|
|
1035
|
+
issues: list = []
|
|
1036
|
+
if not model.tables:
|
|
1037
|
+
return issues
|
|
1038
|
+
sub = field_path.split(".", 1)[1] if "." in field_path else ""
|
|
1039
|
+
|
|
1040
|
+
def _style_run(cell):
|
|
1041
|
+
"""取单元格用于样式检查的 run:优先第一个非空文本 run,其次第一个有样式 run。"""
|
|
1042
|
+
if not cell.paragraphs:
|
|
1043
|
+
return None, None
|
|
1044
|
+
for para in cell.paragraphs:
|
|
1045
|
+
for run in para.runs:
|
|
1046
|
+
if run.text.strip():
|
|
1047
|
+
return para, run
|
|
1048
|
+
for para in cell.paragraphs:
|
|
1049
|
+
if para.runs:
|
|
1050
|
+
return para, para.runs[0]
|
|
1051
|
+
return None, None
|
|
1052
|
+
|
|
1053
|
+
for ti, table in enumerate(model.tables):
|
|
1054
|
+
loc = f"table:{ti}"
|
|
1055
|
+
|
|
1056
|
+
if sub in ("header.font", "header.size", "header.bold", "header.align"):
|
|
1057
|
+
hdr_cells = [c for c in table.cells if c.row == 0]
|
|
1058
|
+
if not hdr_cells:
|
|
1059
|
+
continue
|
|
1060
|
+
for c in hdr_cells:
|
|
1061
|
+
para, run = _style_run(c)
|
|
1062
|
+
if para is None or run is None:
|
|
1063
|
+
continue
|
|
1064
|
+
if sub == "header.font":
|
|
1065
|
+
got = run.format.font_name
|
|
1066
|
+
if got != expected:
|
|
1067
|
+
issues.append(CheckIssue(rule_id, "format", severity, name, loc,
|
|
1068
|
+
str(got), str(expected), message))
|
|
1069
|
+
break
|
|
1070
|
+
elif sub == "header.size":
|
|
1071
|
+
got = run.format.font_size_pt
|
|
1072
|
+
exp_pt = float(str(expected).replace("pt", ""))
|
|
1073
|
+
if got is None or abs(got - exp_pt) > 0.5:
|
|
1074
|
+
issues.append(CheckIssue(rule_id, "format", severity, name, loc,
|
|
1075
|
+
str(got), str(expected), message))
|
|
1076
|
+
break
|
|
1077
|
+
elif sub == "header.bold":
|
|
1078
|
+
got = run.format.bold
|
|
1079
|
+
if bool(got) != bool(expected):
|
|
1080
|
+
issues.append(CheckIssue(rule_id, "format", severity, name, loc,
|
|
1081
|
+
str(got), str(expected), message))
|
|
1082
|
+
break
|
|
1083
|
+
elif sub == "header.align":
|
|
1084
|
+
got = para.format.alignment if para.format else None
|
|
1085
|
+
if got != expected:
|
|
1086
|
+
issues.append(CheckIssue(rule_id, "format", severity, name, loc,
|
|
1087
|
+
str(got), str(expected), message))
|
|
1088
|
+
break
|
|
1089
|
+
|
|
1090
|
+
elif sub == "header.fill":
|
|
1091
|
+
hdr_cells = [c for c in table.cells if c.row == 0]
|
|
1092
|
+
if not hdr_cells:
|
|
1093
|
+
continue
|
|
1094
|
+
for c in hdr_cells:
|
|
1095
|
+
got = getattr(c, "fill", None)
|
|
1096
|
+
if (got or "").strip().lower() != str(expected).strip().lower():
|
|
1097
|
+
issues.append(CheckIssue(rule_id, "format", severity, name, loc,
|
|
1098
|
+
str(got), str(expected), message))
|
|
1099
|
+
break
|
|
1100
|
+
|
|
1101
|
+
elif sub in ("body.font", "body.size"):
|
|
1102
|
+
body_cells = [c for c in table.cells if c.row > 0 and (c.text or "").strip()]
|
|
1103
|
+
if not body_cells:
|
|
1104
|
+
continue
|
|
1105
|
+
for c in body_cells:
|
|
1106
|
+
_para, run = _style_run(c)
|
|
1107
|
+
if run is None:
|
|
1108
|
+
continue
|
|
1109
|
+
if sub == "body.font":
|
|
1110
|
+
got = run.format.font_name
|
|
1111
|
+
if got != expected:
|
|
1112
|
+
issues.append(CheckIssue(rule_id, "format", severity, name, loc,
|
|
1113
|
+
str(got), str(expected), message))
|
|
1114
|
+
break
|
|
1115
|
+
else: # body.size
|
|
1116
|
+
got = run.format.font_size_pt
|
|
1117
|
+
exp_pt = float(str(expected).replace("pt", ""))
|
|
1118
|
+
if got is None or abs(got - exp_pt) > 0.5:
|
|
1119
|
+
issues.append(CheckIssue(rule_id, "format", severity, name, loc,
|
|
1120
|
+
str(got), str(expected), message))
|
|
1121
|
+
break
|
|
1122
|
+
|
|
1123
|
+
elif sub.startswith("cell_margin"):
|
|
1124
|
+
margin = getattr(table, "cell_margin", None) or {}
|
|
1125
|
+
edge = sub.split(".")[-1]
|
|
1126
|
+
got = margin.get(edge)
|
|
1127
|
+
exp = int(expected)
|
|
1128
|
+
if got is None or got != exp:
|
|
1129
|
+
issues.append(CheckIssue(rule_id, "format", severity, name, loc,
|
|
1130
|
+
str(got), str(expected), message))
|
|
1131
|
+
|
|
1132
|
+
return issues
|
|
@@ -18,6 +18,7 @@ from engine.core.document.models import DocumentModel
|
|
|
18
18
|
from engine.core.document.modifier import (
|
|
19
19
|
modify_font, modify_size, modify_alignment, modify_line_spacing,
|
|
20
20
|
modify_first_line_indent, modify_margins, modify_bold,
|
|
21
|
+
modify_paper_size, # V2.3:纸张尺寸修复(FIX-C054)
|
|
21
22
|
remove_extra_spaces, remove_extra_blank_lines,
|
|
22
23
|
normalize_punctuation, normalize_heading_content,
|
|
23
24
|
convert_markdown, fix_bold_range,
|
|
@@ -49,6 +50,12 @@ _ACTION_MAP = {
|
|
|
49
50
|
),
|
|
50
51
|
"set_margins": lambda model, target, value, _rules: modify_margins(model, value),
|
|
51
52
|
"set_page_margins": lambda model, target, value, _rules: modify_margins(model, value),
|
|
53
|
+
"set_paper_size": lambda model, target, value, _rules: modify_paper_size(
|
|
54
|
+
model,
|
|
55
|
+
width_mm=value.get("width_mm") if isinstance(value, dict) else None,
|
|
56
|
+
height_mm=value.get("height_mm") if isinstance(value, dict) else None,
|
|
57
|
+
),
|
|
58
|
+
"set_table_style": lambda model, target, value, _rules: _apply_table_style(model, value),
|
|
52
59
|
"remove_extra_spaces": lambda model, target, value, _rules: remove_extra_spaces(model),
|
|
53
60
|
"remove_extra_blank_lines": lambda model, target, value, _rules: remove_extra_blank_lines(
|
|
54
61
|
model,
|
|
@@ -273,3 +280,58 @@ def _apply_fix_paragraph_type(model: DocumentModel, target: str, value: Any) ->
|
|
|
273
280
|
modify_size(model, target, _parse_pt_value(value["size"]))
|
|
274
281
|
if value.get("first_line_indent") is not None:
|
|
275
282
|
modify_first_line_indent(model, target, _parse_indent_value(value["first_line_indent"]))
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _apply_table_style(model: DocumentModel, value: Any) -> None:
|
|
286
|
+
"""应用表格具体样式(V2.3,供 FIX-C055 set_table_style)。
|
|
287
|
+
|
|
288
|
+
value 格式(dict):
|
|
289
|
+
{
|
|
290
|
+
"header": {"font": "黑体", "size": "12pt", "bold": true,
|
|
291
|
+
"align": "center", "fill": "D9E2F3"},
|
|
292
|
+
"body": {"font": "仿宋_GB2312", "size": "12pt", "align": "left"},
|
|
293
|
+
"cell_margin": {"top": 0, "left": 80, "bottom": 0, "right": 80},
|
|
294
|
+
}
|
|
295
|
+
对 model 中所有表格的表头行(row==0)/数据行应用样式,并设置 cell.fill 与
|
|
296
|
+
table.cell_margin;随后由 generator(_generator_helpers)写出到 docx。
|
|
297
|
+
数字单元格按智能对齐右对齐(与 _smart_align_cell 一致)。
|
|
298
|
+
"""
|
|
299
|
+
import re as _re
|
|
300
|
+
if not isinstance(value, dict) or not model.tables:
|
|
301
|
+
return
|
|
302
|
+
header = value.get("header") or {}
|
|
303
|
+
body = value.get("body") or {}
|
|
304
|
+
margin = value.get("cell_margin")
|
|
305
|
+
_num_re = _re.compile(r'^[\d.,%‰]+$')
|
|
306
|
+
|
|
307
|
+
for table in model.tables:
|
|
308
|
+
if isinstance(margin, dict) and margin:
|
|
309
|
+
base = dict(table.cell_margin or {})
|
|
310
|
+
base.update(margin)
|
|
311
|
+
table.cell_margin = base
|
|
312
|
+
|
|
313
|
+
for cell in table.cells:
|
|
314
|
+
is_header = cell.row == 0
|
|
315
|
+
style = header if is_header else body
|
|
316
|
+
# 单元格段落 run 样式
|
|
317
|
+
for p in cell.paragraphs:
|
|
318
|
+
for run in p.runs:
|
|
319
|
+
if style.get("font"):
|
|
320
|
+
run.format.font_name = str(style["font"])
|
|
321
|
+
if style.get("size") is not None:
|
|
322
|
+
run.format.font_size_pt = float(str(style["size"]).replace("pt", ""))
|
|
323
|
+
if style.get("bold") is not None and is_header:
|
|
324
|
+
run.format.bold = bool(style["bold"])
|
|
325
|
+
# 对齐:表头居中;数据行数字右对齐、其余按 body.align
|
|
326
|
+
if is_header:
|
|
327
|
+
if header.get("align") and p.format is not None:
|
|
328
|
+
p.format.alignment = str(header["align"])
|
|
329
|
+
elif body.get("align") and p.format is not None:
|
|
330
|
+
txt = (cell.text or "").strip()
|
|
331
|
+
if _num_re.match(txt):
|
|
332
|
+
p.format.alignment = "right"
|
|
333
|
+
else:
|
|
334
|
+
p.format.alignment = str(body["align"])
|
|
335
|
+
# 表头底色
|
|
336
|
+
if is_header and header.get("fill"):
|
|
337
|
+
cell.fill = str(header["fill"])
|
package/gongwen/__init__.py
CHANGED