gongwen-skill 2.3.0 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -4,6 +4,18 @@
4
4
  Licensed under the MIT License. See the LICENSE file for details.
5
5
  -->
6
6
 
7
+ ## v2.4.0 (2026-09-01)
8
+
9
+ ### Added
10
+ - **表格样式检测与修复全链路**:新增 8 条表格样式检查规则(CHK-C051~C058:表头字体/字号/加粗/对齐/底色、表体字体/字号、单元格左右边距)与 set_table_style 修复动作(FIX-C055),对齐 rules/official/_common.yaml table 配置块——此前该配置为死配置,无 CHK 规则也从不被检查
11
+ - **模型新增表格样式字段**:TableCell.fill(表头底色 w:shd)、Table.cell_margin(单元格边距 w:tblCellMar),parser 解析、generator 原位更新/新建两路径写出
12
+ - **md2docx 表格样式应用**:渲染时读取 table 配置块,表头写入 D9E2F3 底色 + 单元格边距 80twips + 数字列右对齐(与 optimize 智能对齐一致)
13
+
14
+ ### Fixed
15
+ - **表格单元格解析丢失修复**:_parse_table 原用 id(cell._element) 去重,lxml 代理对象 id 被 GC 复用导致非合并表格也误丢约 58% 单元格(实测 88 个误丢 51 个),改用 lxml 元素对象做 key——修复后表格内容/样式修复可完整覆盖全部单元格
16
+ - **署名与开头样式修复**:主送机关延迟插入(标题后、防误判标题)、recipients 顺序/重复去重、导语段误判排除、正文内联落款识别(避免与 --signer/--date 重复落款、补齐署名前 2 空行)
17
+ - **生成文档作者元数据统一为 Jose AI**
18
+
7
19
  ## v2.3.0 (2026-08-31)
8
20
 
9
21
  ### Added
package/README.md CHANGED
@@ -337,7 +337,7 @@ DSH 采用 **Cordis 模块化微内核架构**:技能体系基于本地文件
337
337
  git clone https://github.com/linhut/gongwen-skill.git
338
338
  cd gongwen-skill
339
339
  pip install -r requirements.txt # 或 pip install gongwen-skill(已上 PyPI)
340
- python -m gongwen --version # 检验:gongwen-skill v2.3.0
340
+ python -m gongwen --version # 检验:gongwen-skill v2.4.0
341
341
  ```
342
342
 
343
343
  ### 方式一:作为 DSH Skill 注册(基于本地文件系统)
@@ -393,7 +393,7 @@ pnpm add -w gongwen-skill
393
393
  "dependencies": {
394
394
  "@deepseek-ai/dsh-base": "...",
395
395
  "@deepseek-ai/dsh-web-app": "...",
396
- "gongwen-skill": "^2.3.0"
396
+ "gongwen-skill": "^2.4.0"
397
397
  },
398
398
  "dsh": {
399
399
  "profile": {
@@ -448,9 +448,9 @@ dsh --profile web
448
448
  | CLI 独立可执行(`python -m gongwen <命令>`) | ✅ |
449
449
  | PyPI 上架(`pip install gongwen-skill`) | ✅ |
450
450
  | 零外部运行时依赖(仅 python-docx/pydantic/pyyaml) | ✅ |
451
- | DSH 配置化排版参数(页边距/行距/字体/默认模板版本) | ✅ v2.3.0+ |
451
+ | DSH 配置化排版参数(页边距/行距/字体/默认模板版本) | ✅ v2.4.0+ |
452
452
 
453
- ### DSH 插件配置化(v2.3.0+)
453
+ ### DSH 插件配置化(v2.4.0+)
454
454
 
455
455
  DSH 插件支持通过配置文件管理排版参数,Agent 调用时自动注入,纯 CLI 用户不受影响。
456
456
 
@@ -574,7 +574,7 @@ pip install -r requirements.txt
574
574
  用户:帮我优化这份会议通知的第二章节措辞
575
575
 
576
576
  Agent:📋 合规自检报告
577
- Skill 版本: v2.3.0(多渠道自检已确认最新)
577
+ Skill 版本: v2.4.0(多渠道自检已确认最新)
578
578
  路径判定: B(内容优化)
579
579
  依据: 用户指定了已有文档,且要求"优化措辞"
580
580
  命令调用: 1. python -m gongwen optimize-content 会议通知.docx --changes changes.json --apply --paragraphs "5-8"
package/dsh/index.js CHANGED
@@ -1,4 +1,4 @@
1
- // 公文全流程处理工具 - DSH plugin bridge (gongwen-skill, v2.3.0+)
1
+ // 公文全流程处理工具 - DSH plugin bridge (gongwen-skill, v2.4.0+)
2
2
  // (c) 2026 Jose AI (https://www.linhut.cn)
3
3
  // https://github.com/linhut/gongwen-skill
4
4
  // Licensed under the MIT License. See the LICENSE file for details.
@@ -58,12 +58,54 @@ def _smart_align_cell(cell_text: str, is_header: bool, col_idx: int, total_cols:
58
58
  return 'left'
59
59
 
60
60
 
61
+ def _write_cell_shd(cell, fill: str | None) -> None:
62
+ """写入单元格底色(w:shd fill)。"""
63
+ if not fill:
64
+ return
65
+ tcPr = cell._tc.get_or_add_tcPr()
66
+ shd = tcPr.find(qn('w:shd'))
67
+ if shd is None:
68
+ shd = tcPr.makeelement(qn('w:shd'), {})
69
+ tcPr.append(shd)
70
+ shd.set(qn('w:val'), 'clear')
71
+ shd.set(qn('w:color'), 'auto')
72
+ shd.set(qn('w:fill'), str(fill))
73
+
74
+
75
+ def _write_table_cell_mar(table, cell_margin: dict | None) -> None:
76
+ """写入表格单元格边距(w:tblCellMar,twips)。"""
77
+ if not cell_margin:
78
+ return
79
+ tblPr = table._tbl.tblPr
80
+ if tblPr is None:
81
+ tblPr = table._tbl._add_tblPr()
82
+ cell_mar = tblPr.find(qn('w:tblCellMar'))
83
+ if cell_mar is None:
84
+ cell_mar = tblPr.makeelement(qn('w:tblCellMar'), {})
85
+ tblPr.append(cell_mar)
86
+ for edge in ('top', 'left', 'bottom', 'right'):
87
+ val = cell_margin.get(edge)
88
+ if val is None:
89
+ continue
90
+ el = cell_mar.find(qn('w:' + edge))
91
+ if el is None:
92
+ el = cell_mar.makeelement(qn('w:' + edge), {})
93
+ cell_mar.append(el)
94
+ el.set(qn('w:w'), str(int(val)))
95
+ el.set(qn('w:type'), 'dxa')
96
+
97
+
61
98
  def _update_table_content(table, table_model: TableModel):
62
99
  """更新已有表格的单元格内容(带智能对齐)。"""
63
100
  total_cols = len(table.columns) if hasattr(table, 'columns') else 0
101
+ # V2.3:写出单元格边距(供表格样式修复)
102
+ _write_table_cell_mar(table, getattr(table_model, 'cell_margin', None))
64
103
  for cell_model in table_model.cells:
65
104
  try:
66
105
  cell = table.cell(cell_model.row, cell_model.col)
106
+ # V2.3:写出表头底色(供表格样式修复)
107
+ if cell_model.fill:
108
+ _write_cell_shd(cell, cell_model.fill)
67
109
  # 更新单元格中的段落内容
68
110
  if cell_model.paragraphs:
69
111
  for p_idx, para_model in enumerate(cell_model.paragraphs):
@@ -144,11 +186,17 @@ def _add_table(doc: Document, table_model: TableModel):
144
186
  borders.append(border)
145
187
  tblPr.append(borders)
146
188
 
189
+ # V2.3:写出单元格边距(供表格样式修复)
190
+ _write_table_cell_mar(table, getattr(table_model, 'cell_margin', None))
191
+
147
192
  # 智能对齐:表头行居中加粗,数据行按内容类型对齐
148
193
  total_cols = max(1, table_model.cols)
149
194
  for cell_model in table_model.cells:
150
195
  try:
151
196
  cell = table.cell(cell_model.row, cell_model.col)
197
+ # V2.3:写出表头底色
198
+ if getattr(cell_model, 'fill', None):
199
+ _write_cell_shd(cell, cell_model.fill)
152
200
  # 清除默认段落
153
201
  for para in cell.paragraphs:
154
202
  for run in list(para.runs):
@@ -316,6 +364,15 @@ def _update_pPr(p_element, para_model: Paragraph):
316
364
  pPr = OxmlElement('w:pPr')
317
365
  p_element.insert(0, pPr)
318
366
 
367
+ # 段前分页(--- 附件分页):model 有 page_break 时写入/移除 w:pageBreakBefore
368
+ if getattr(para_model, 'page_break', False):
369
+ if pPr.find(qn('w:pageBreakBefore')) is None:
370
+ pPr.append(OxmlElement('w:pageBreakBefore'))
371
+ else:
372
+ pb = pPr.find(qn('w:pageBreakBefore'))
373
+ if pb is not None:
374
+ pPr.remove(pb)
375
+
319
376
  # 对齐方式:仅当 model 有值时替换
320
377
  if fmt.alignment:
321
378
  jc = pPr.find(qn('w:jc'))
@@ -404,6 +461,10 @@ def _apply_paragraph_format(para, para_model: Paragraph):
404
461
  pf = para.paragraph_format
405
462
  fmt = para_model.format
406
463
 
464
+ # 段前分页(--- 附件分页标记)
465
+ if getattr(para_model, "page_break", False):
466
+ pf.page_break_before = True
467
+
407
468
  # Alignment
408
469
  if fmt.alignment:
409
470
  alignment_map = {
@@ -377,47 +377,74 @@ def _apply_page_setup(doc: Document, model: DocumentModel):
377
377
 
378
378
  def _replace_paragraphs(doc: Document, model: DocumentModel):
379
379
  """
380
- 替换文档中的段落内容,同时保留表格和图片在原始位置。
380
+ 按模型布局重建 body 中的段落与表格序列。
381
381
 
382
- 策略(P0-1 加固):
383
- 1. 交错遍历 body 直接子元素:<w:p> 按序消耗 model.paragraphs 内容,
384
- <w:tbl> 及表格内段落保持不动(表格不消耗 model 段落索引)
385
- 2. model 比原文多的段落,追加到 body 末尾
386
- 3. 原文比 model 多的 <w:p>,从 body 中移除
387
- 4. 图片:内联图片(<w:drawing> 在 <w:p> 内)通过段落内容替换间接保留,
388
- 浮动图片(锚定)不受影响。前提是源文档保留策略生效。
382
+ 旧实现假定"源 body 段落数 == 模型段落数且一一对应":段落只能原位替换、
383
+ 多出的段落追加到末尾、表格停留在源位置。当优化规则在文档中间插入段落
384
+ (如补齐空行、插入落款/附件说明)时,多出的段落被追加到末尾而表格仍在
385
+ 源位置,导致"落款后附表"被插到落款之前等布局错乱。
386
+
387
+ 此版本以模型为唯一布局依据:
388
+ - 段落严格按 model.paragraphs 顺序放置(源段落元素尽量复用,保留内联图片);
389
+ - 表格按其 insert_after_index 锚点放在对应段落之后(锚点 -1 置于文档最前);
390
+ - 未建模的源元素(sectPr、未建模表格等)保留在末尾(保留策略)。
389
391
 
390
392
  注意:段落索引必须与模型中的 index 字段严格对齐。
391
- 索引错位会导致内联图片跟随错误的段落移位。
392
393
  """
393
394
  body = doc.element.body
394
395
  p_tag = qn('w:p')
396
+ tbl_tag = qn('w:tbl')
395
397
  model_paras = model.paragraphs
396
- para_idx = 0
397
-
398
- # 交错遍历:跳过 <w:tbl>(不消耗 model 索引),仅对 <w:p> 按序替换/移除
399
- for child in list(body):
400
- if child.tag == p_tag:
401
- if para_idx < len(model_paras):
402
- _replace_paragraph_content(doc, child, model_paras[para_idx])
403
- para_idx += 1
404
- else:
405
- # 原文段落多于 model → 移除多余段落
406
- try:
407
- body.remove(child)
408
- except Exception:
409
- pass # 已被移除则跳过
410
-
411
- # model 比原文多的段落,追加到 body 末尾
412
- while para_idx < len(model_paras):
413
- new_para = doc.add_paragraph()
414
- _apply_paragraph_format(new_para, model_paras[para_idx])
415
- _add_runs_to_paragraph(new_para, model_paras[para_idx])
416
- para_idx += 1
417
-
418
- logger.debug(f"Replaced {min(len(model_paras), len([c for c in body if c.tag == p_tag]))} paragraphs, "
419
- f"added {max(0, len(model_paras) - len([c for c in body if c.tag == p_tag]))}, "
420
- f"removed {max(0, len([c for c in body if c.tag == p_tag]) - len(model_paras))}")
398
+
399
+ # 1. 收集源 body 子元素(保持原顺序)
400
+ src_p = [c for c in body if c.tag == p_tag]
401
+ src_tbl = [c for c in body if c.tag == tbl_tag]
402
+ src_other = [c for c in body if c.tag not in (p_tag, tbl_tag)]
403
+
404
+ # 2. 表格锚点 → 模型表格索引(保持模型顺序)
405
+ anchor_map: dict[int, list[int]] = {}
406
+ for ti, t in enumerate(model.tables):
407
+ anchor_map.setdefault(t.insert_after_index, []).append(ti)
408
+ used_tbl: set[int] = set()
409
+
410
+ # 3. 构建目标序列
411
+ seq: list = []
412
+ # 文档最前的表格(锚点 -1);模型表格超出源文档时跳过,由 _update_tables 新建
413
+ for ti in anchor_map.get(-1, []):
414
+ if ti < len(src_tbl):
415
+ seq.append(src_tbl[ti])
416
+ used_tbl.add(ti)
417
+
418
+ for i, para_model in enumerate(model_paras):
419
+ # 段落元素:优先复用源元素(保留内联图片),超出则新建
420
+ if i < len(src_p):
421
+ p_elem = src_p[i]
422
+ else:
423
+ p_elem = OxmlElement('w:p')
424
+ _replace_paragraph_content(doc, p_elem, para_model)
425
+ seq.append(p_elem)
426
+ # 锚定在此段落之后的表格(模型表格超出源文档时跳过,由 _update_tables/_add_table 新建放置)
427
+ for ti in anchor_map.get(i, []):
428
+ if ti < len(src_tbl):
429
+ seq.append(src_tbl[ti])
430
+ used_tbl.add(ti)
431
+
432
+ # 4. 未在模型锚定的源表格保留在末尾(保留策略)
433
+ for ti, t in enumerate(src_tbl):
434
+ if ti not in used_tbl:
435
+ seq.append(t)
436
+
437
+ # 5. 其它未建模元素(sectPr 等)保持在末尾
438
+ seq.extend(src_other)
439
+
440
+ # 6. 重建 body
441
+ for c in list(body):
442
+ body.remove(c)
443
+ for c in seq:
444
+ body.append(c)
445
+
446
+ logger.debug(f"Replaced {len(model_paras)} paragraphs, "
447
+ f"{len(seq)} body elements in model order")
421
448
 
422
449
 
423
450
  def _replace_paragraph_content(doc: Document, p_element, para_model: Paragraph):
@@ -635,8 +662,8 @@ def _update_metadata(doc: Document, model: DocumentModel):
635
662
  props = doc.core_properties
636
663
  if meta.title:
637
664
  props.title = meta.title
638
- if meta.author:
639
- props.author = meta.author
665
+ # 统一文档作者为 "Jose AI"(用户要求:生成文档的作者都写为 Jose AI)
666
+ props.author = "Jose AI"
640
667
  if meta.subject:
641
668
  props.subject = meta.subject
642
669
  if meta.category:
@@ -39,6 +39,9 @@ _MD_TABLE_SEP_RE = re.compile(r'^\|[\s\-:|]+\|$')
39
39
  # markdown 水平分隔线:--- *** ___
40
40
  _MD_HR_RE = re.compile(r'^[-*_]{3,}$')
41
41
 
42
+ # 纯连字符分隔线 --- 用作附件分页标记(SKILL.md 约定)
43
+ _MD_PAGEBREAK_RE = re.compile(r'^-{3,}$')
44
+
42
45
  # HTML 标签
43
46
  _HTML_TAG_RE = re.compile(r'<[^>]+>')
44
47
 
@@ -184,6 +187,13 @@ def convert_markdown(model: DocumentModel) -> int:
184
187
 
185
188
  # 水平分隔线 --- *** ___
186
189
  if _MD_HR_RE.match(text):
190
+ # 纯 --- 作为附件分页标记(SKILL.md:落款后、附件标题前保留 --- 分页)
191
+ if _MD_PAGEBREAK_RE.match(text):
192
+ para.text = ''
193
+ para.runs = []
194
+ para.page_break = True
195
+ changes += 1
196
+ continue
187
197
  to_remove.append(i)
188
198
  continue
189
199
 
@@ -98,6 +98,8 @@ class TableCell(BaseModel):
98
98
  col: int
99
99
  text: str
100
100
  paragraphs: list[Paragraph] = Field(default_factory=list)
101
+ # V2.3:表头单元格底色(w:shd fill,如 D9E2F3)——供表格样式检测/修复
102
+ fill: Optional[str] = None
101
103
 
102
104
 
103
105
  class Table(BaseModel):
@@ -107,6 +109,8 @@ class Table(BaseModel):
107
109
  cols: int
108
110
  cells: list[TableCell] = Field(default_factory=list)
109
111
  insert_after_index: int = -1 # 表格紧跟在哪个段落索引之后(-1 表示文档开头)
112
+ # V2.3:表格单元格边距(twips,{top,left,bottom,right})——供表格样式检测/修复
113
+ cell_margin: Optional[dict] = None
110
114
 
111
115
 
112
116
  class HeaderFooter(BaseModel):
@@ -101,6 +101,10 @@ def _select_paragraphs(model: DocumentModel, target: str) -> list[Paragraph]:
101
101
  if re.match(r'^\d{4}年\d{1,2}月\d{1,2}日$', last) or re.match(r'^\d{4}[.\-/]\d{1,2}[.\-/]\d{1,2}$', last):
102
102
  return non_empty[-1:]
103
103
  return []
104
+ elif target == "attachment":
105
+ # V2.3:附件说明(role='attachment')选择器——此前缺失,导致
106
+ # target=attachment 的 FIX 规则走 Unknown target 警告而静默失效
107
+ return [p for p in model.paragraphs if p.role == 'attachment']
104
108
  elif target in ('salutation', 'introduction', 'transition', 'meeting_date', 'numbered_body'):
105
109
  # N2: 段落类型 target —— 使用 detect_paragraph_type 内容匹配
106
110
  return [p for p in model.paragraphs if detect_paragraph_type(p.text, p.role) == target]
@@ -229,6 +233,19 @@ def modify_margins(model: DocumentModel, margins: dict[str, str | float]) -> Non
229
233
  setattr(ps, attr, parsed)
230
234
 
231
235
 
236
+ def modify_paper_size(model: DocumentModel, width_mm: float | None = None,
237
+ height_mm: float | None = None) -> None:
238
+ """修改纸张尺寸(毫米)。V2.3 新增:供 FIX-C054 把非 A4 纸张归一为 A4。
239
+
240
+ 仅当提供了对应的毫米值才修改,未提供的维度保持不变。
241
+ """
242
+ ps = model.page_setup
243
+ if width_mm is not None and 50 <= width_mm <= 1000:
244
+ ps.paper_width_mm = float(width_mm)
245
+ if height_mm is not None and 50 <= height_mm <= 1000:
246
+ ps.paper_height_mm = float(height_mm)
247
+
248
+
232
249
  def clean_path_b_markers(model: DocumentModel) -> int:
233
250
  """清理路径 B 遗留的修改说明段落和删除线标记。
234
251
 
@@ -407,6 +424,11 @@ def should_bold_first_sentence(text: str | None, role: str | None = None) -> boo
407
424
  称呼/导语/过渡/署名/会议日期段 → False(不加粗);
408
425
  编号正文/普通正文 → True(首句加粗)。
409
426
  """
427
+ # V2.3 修复:联系人段("(联系人:XXX,联系电话:XXX)")是落款区备注,
428
+ # 不应首句加粗——否则加粗仿宋段会被启发式误判为三级标题(CHK-C037 误报)。
429
+ raw = (text or "").strip()
430
+ if raw.startswith(("(联系人", "(联系人", "联系人:")):
431
+ return False
410
432
  para_type = detect_paragraph_type(text, role)
411
433
  return PARAGRAPH_TYPE_RULES.get(para_type, True)
412
434
 
@@ -719,14 +741,47 @@ def _insert_blank_lines(model: DocumentModel, rules: dict | None = None) -> int:
719
741
  def _blank_para() -> Paragraph:
720
742
  return Paragraph(index=0, text="", role="body", runs=[], format=ParagraphFormat())
721
743
 
722
- # 1. 公文大标题前/后空行
744
+ def _insert_blank_at(pos: int) -> None:
745
+ """在 pos 处插入一个空行,并同步后移受影响表格的锚点。
746
+
747
+ 表格锚点 insert_after_index 指向"表格所跟随的段落索引";
748
+ 在其位置(含)之后插入段落会使后续段落索引整体 +1,
749
+ 因此所有锚点 >= pos 的表格都需同步 +1,否则附表会被插到错误位置
750
+ (如"落款后附表"被插到落款之前)。
751
+ """
752
+ nonlocal inserted
753
+ for _t in model.tables:
754
+ if getattr(_t, 'insert_after_index', -1) >= pos:
755
+ _t.insert_after_index += 1
756
+ model.paragraphs.insert(pos, _blank_para())
757
+ inserted += 1
758
+
759
+ def _count_blanks_before(idx: int) -> int:
760
+ """统计 idx 位置之前连续空行数(不含 idx 本身)。"""
761
+ n = 0
762
+ j = idx - 1
763
+ while j >= 0 and not model.paragraphs[j].text.strip():
764
+ n += 1
765
+ j -= 1
766
+ return n
767
+
768
+ def _count_blanks_after(idx: int) -> int:
769
+ """统计 idx 位置之后连续空行数(不含 idx 本身)。"""
770
+ n = 0
771
+ j = idx + 1
772
+ while j < len(model.paragraphs) and not model.paragraphs[j].text.strip():
773
+ n += 1
774
+ j += 1
775
+ return n
776
+
777
+ # 1. 公文大标题前/后空行(V2.3 幂等:只补差额,不叠加已有空行)
723
778
  title_indices = [i for i, p in enumerate(model.paragraphs)
724
779
  if p.is_heading and p.heading_level == 0]
725
780
  if title_indices:
726
781
  before = int(bl.get('doc_title_before', 0) or 0)
727
782
  after = int(bl.get('doc_title_after', 0) or 0)
728
783
  first = title_indices[0]
729
- # 标题前:往前找首个非空段落,在其后插入空行(避免文档开头堆空行)
784
+ # 标题前:往前找首个非空段落,在其后补足空行(避免文档开头堆空行)
730
785
  if before > 0:
731
786
  anchor = -1
732
787
  for j in range(first - 1, -1, -1):
@@ -734,37 +789,45 @@ def _insert_blank_lines(model: DocumentModel, rules: dict | None = None) -> int:
734
789
  anchor = j
735
790
  break
736
791
  if anchor >= 0:
737
- for _ in range(before):
738
- model.paragraphs.insert(anchor + 1, _blank_para())
739
- inserted += 1
740
- anchor += 1
741
- # 标题后:在标题段后插入空行
792
+ existing = first - anchor - 1 # anchor 与标题之间的已有空行数
793
+ shortfall = before - existing
794
+ if shortfall > 0:
795
+ for _ in range(shortfall):
796
+ anchor += 1
797
+ _insert_blank_at(anchor)
798
+ # 标题后:在标题段后补足空行
742
799
  if after > 0:
743
- for _ in range(after):
744
- model.paragraphs.insert(first + 1, _blank_para())
745
- inserted += 1
746
- first += 1
747
-
748
- # 2. 正文末尾与落款前空 N 行(body_to_signature)
800
+ existing = _count_blanks_after(first)
801
+ shortfall = after - existing
802
+ if shortfall > 0:
803
+ for _ in range(shortfall):
804
+ first += 1
805
+ _insert_blank_at(first)
806
+
807
+ # 2. 正文末尾与落款前空 N 行(body_to_signature,幂等)
749
808
  sig_gap = int(bl.get('body_to_signature', 0) or 0)
750
809
  if sig_gap > 0:
751
810
  sig_idx = next((i for i, p in enumerate(model.paragraphs)
752
811
  if p.role in ('signature', 'date') and p.text.strip()), None)
753
812
  if sig_idx is not None:
754
- for _ in range(sig_gap):
755
- model.paragraphs.insert(sig_idx, _blank_para())
756
- inserted += 1
813
+ existing = _count_blanks_before(sig_idx)
814
+ shortfall = sig_gap - existing
815
+ if shortfall > 0:
816
+ for _ in range(shortfall):
817
+ _insert_blank_at(sig_idx)
757
818
 
758
- # 3. 附件标题与正文间空 N 行(attachment_gap)
819
+ # 3. 附件标题与正文间空 N 行(attachment_gap,幂等)
759
820
  att_gap = int(bl.get('attachment_gap', 0) or 0)
760
821
  if att_gap > 0:
761
822
  att_idx = next((i for i, p in enumerate(model.paragraphs)
762
823
  if p.role == 'attachment' and p.text.strip()), None)
763
824
  if att_idx is not None:
764
- for _ in range(att_gap):
765
- model.paragraphs.insert(att_idx + 1, _blank_para())
766
- inserted += 1
767
- att_idx += 1
825
+ existing = _count_blanks_after(att_idx)
826
+ shortfall = att_gap - existing
827
+ if shortfall > 0:
828
+ for _ in range(shortfall):
829
+ att_idx += 1
830
+ _insert_blank_at(att_idx)
768
831
 
769
832
  if inserted:
770
833
  for i, p in enumerate(model.paragraphs):
@@ -1116,7 +1179,11 @@ def bold_first_sentence_of_body(model: DocumentModel) -> int:
1116
1179
  from copy import deepcopy
1117
1180
 
1118
1181
  changes = 0
1119
- exclude_roles = {'signature', 'date', 'title', 'recipient', 'annotation'}
1182
+ # V2.3 修复:附件说明/抄送等非正文段加入排除——否则"附件:xxx"会被首句加粗,
1183
+ # 加粗的仿宋短文本被 parser 启发式误判为三级标题(CHK-C037 误报),
1184
+ # 且破坏附件说明"左空二字、不加粗"的规范排版。
1185
+ exclude_roles = {'signature', 'date', 'title', 'recipient', 'annotation',
1186
+ 'attachment', 'cc'}
1120
1187
  for para in model.paragraphs:
1121
1188
  if para.is_heading:
1122
1189
  continue