gongwen-skill 2.4.0 → 2.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -4,6 +4,17 @@
4
4
  Licensed under the MIT License. See the LICENSE file for details.
5
5
  -->
6
6
 
7
+ ## v2.5.0 (2026-09-01)
8
+
9
+ ### Added
10
+ - **样式学习引擎修复**:style-learn 多 section 文档只学主文档(不再把横向附件页当正文页);页边距 mm→cm 换算修复(26.9mm 不再写成 26.9cm);角色判定增强(识别红头版头/发文字号/整段日期/落款机关名);解析 styles.xml 样式链(basedOn + docDefaults),run 未指定 eastAsia 时继承段落样式字体,修复落款误学 Times New Roman
11
+ - **模板规则断链打通**:manager 新增 _sync_style_expected,将配置段权威样式值(doc_title/body/signature/date/table/page_setup 等)同步到 CHK expected 与 FIX value(标量覆盖 + 复合 dict 保守合并),使 style-learn 模板 / user_rules / config-overrides 覆盖配置段后 check/optimize 期望值自动跟随
12
+
13
+ ### Fixed
14
+ - **style-learn 学错横向附件页**:多 section 文档(正文纵向 + 附件横向)此前学到 297×210mm 横向纸张与附件页边距
15
+ - **样式继承未解析**:run 无显式 eastAsia 时直接回退 ascii=Times New Roman,现按段落样式链(basedOn → docDefaults)解析正确中文字体
16
+ - **模板样式不生效**:style-learn 生成的模板仅覆盖配置段,但 CHK expected/FIX value 硬编码在规则 YAML,导致 -t 模板时检查/修复仍用默认值——现已打通
17
+
7
18
  ## v2.4.0 (2026-09-01)
8
19
 
9
20
  ### Added
package/README.md CHANGED
@@ -337,7 +337,7 @@ DSH 采用 **Cordis 模块化微内核架构**:技能体系基于本地文件
337
337
  git clone https://github.com/linhut/gongwen-skill.git
338
338
  cd gongwen-skill
339
339
  pip install -r requirements.txt # 或 pip install gongwen-skill(已上 PyPI)
340
- python -m gongwen --version # 检验:gongwen-skill v2.4.0
340
+ python -m gongwen --version # 检验:gongwen-skill v2.5.0
341
341
  ```
342
342
 
343
343
  ### 方式一:作为 DSH Skill 注册(基于本地文件系统)
@@ -393,7 +393,7 @@ pnpm add -w gongwen-skill
393
393
  "dependencies": {
394
394
  "@deepseek-ai/dsh-base": "...",
395
395
  "@deepseek-ai/dsh-web-app": "...",
396
- "gongwen-skill": "^2.4.0"
396
+ "gongwen-skill": "^2.5.0"
397
397
  },
398
398
  "dsh": {
399
399
  "profile": {
@@ -448,9 +448,9 @@ dsh --profile web
448
448
  | CLI 独立可执行(`python -m gongwen <命令>`) | ✅ |
449
449
  | PyPI 上架(`pip install gongwen-skill`) | ✅ |
450
450
  | 零外部运行时依赖(仅 python-docx/pydantic/pyyaml) | ✅ |
451
- | DSH 配置化排版参数(页边距/行距/字体/默认模板版本) | ✅ v2.4.0+ |
451
+ | DSH 配置化排版参数(页边距/行距/字体/默认模板版本) | ✅ v2.5.0+ |
452
452
 
453
- ### DSH 插件配置化(v2.4.0+)
453
+ ### DSH 插件配置化(v2.5.0+)
454
454
 
455
455
  DSH 插件支持通过配置文件管理排版参数,Agent 调用时自动注入,纯 CLI 用户不受影响。
456
456
 
@@ -574,7 +574,7 @@ pip install -r requirements.txt
574
574
  用户:帮我优化这份会议通知的第二章节措辞
575
575
 
576
576
  Agent:📋 合规自检报告
577
- Skill 版本: v2.4.0(多渠道自检已确认最新)
577
+ Skill 版本: v2.5.0(多渠道自检已确认最新)
578
578
  路径判定: B(内容优化)
579
579
  依据: 用户指定了已有文档,且要求"优化措辞"
580
580
  命令调用: 1. python -m gongwen optimize-content 会议通知.docx --changes changes.json --apply --paragraphs "5-8"
package/dsh/index.js CHANGED
@@ -1,4 +1,4 @@
1
- // 公文全流程处理工具 - DSH plugin bridge (gongwen-skill, v2.4.0+)
1
+ // 公文全流程处理工具 - DSH plugin bridge (gongwen-skill, v2.5.0+)
2
2
  // (c) 2026 Jose AI (https://www.linhut.cn)
3
3
  // https://github.com/linhut/gongwen-skill
4
4
  // Licensed under the MIT License. See the LICENSE file for details.
@@ -80,6 +80,8 @@ def load_rules_merged(doc_type: str = "") -> dict[str, Any]:
80
80
  # Merge fix_rules and check_rules as distinct lists, not overwritten
81
81
  merged.setdefault("fix_rules", [])
82
82
  merged.setdefault("check_rules", [])
83
+ # 配置段 → 规则期望值同步:使模板/用户规则覆盖配置段后 check/optimize 自动跟随
84
+ _sync_style_expected(merged)
83
85
  return merged
84
86
 
85
87
 
@@ -117,6 +119,93 @@ def _deep_merge(base: dict, overlay: dict) -> None:
117
119
  base[key] = copy.deepcopy(val)
118
120
 
119
121
 
122
+ # 字段前缀 → 配置段键(配置段是样式的权威定义)
123
+ _FIELD_CFG_SECTION = {
124
+ 'title': 'doc_title', 'doc_title': 'doc_title', 'heading_0': 'doc_title',
125
+ 'heading_1': 'heading_1', 'heading_2': 'heading_2',
126
+ 'heading_3': 'heading_3', 'heading_4': 'heading_4',
127
+ 'body': 'body', 'signature': 'signature', 'date': 'date',
128
+ 'meeting_date': 'meeting_date', 'salutation': 'salutation',
129
+ 'introduction': 'introduction', 'transition': 'transition',
130
+ }
131
+ # 可直接同名映射的样式子键
132
+ _STYLE_KEYS = {'font', 'size', 'line_spacing', 'align', 'bold',
133
+ 'first_line_indent', 'fill'}
134
+
135
+
136
+ def _resolve_style_config(rules: dict[str, Any], field: str):
137
+ """按 field 路径取配置段的权威样式值;无法解析返回 None。
138
+
139
+ 样式配置段(doc_title/body/signature/date/heading_*/table/page_setup)
140
+ 是样式的权威定义;check_rules.expected 与 fix_rules.value 默认是它的快照副本。
141
+ 模板/用户规则覆盖配置段后,此函数负责把覆盖值回填到检查/修复规则。
142
+ """
143
+ if not field or '.' not in field:
144
+ return None
145
+ prefix, sub = field.split('.', 1)
146
+
147
+ # table.* / page_setup.* 直接按路径取(table.header.font → table.header.font)
148
+ if prefix in ('table', 'page_setup'):
149
+ node = rules
150
+ for part in field.split('.'):
151
+ if not isinstance(node, dict) or part not in node:
152
+ return None
153
+ node = node[part]
154
+ if isinstance(node, (dict, list)):
155
+ return None
156
+ return node
157
+
158
+ section = _FIELD_CFG_SECTION.get(prefix)
159
+ if section is None or sub not in _STYLE_KEYS:
160
+ return None
161
+ cfg = rules.get(section)
162
+ if not isinstance(cfg, dict) or sub not in cfg:
163
+ return None
164
+ val = cfg[sub]
165
+ if isinstance(val, (dict, list)):
166
+ return None
167
+ return val
168
+
169
+
170
+ def _sync_style_expected(rules: dict[str, Any]) -> None:
171
+ """把配置段的权威样式值同步到 CHK expected 与 FIX value(就地修改)。
172
+
173
+ 使 style-learn 模板 / user_rules / config-overrides 覆盖配置段后,
174
+ check 与 optimize 的期望值自动跟随,而默认配置下幂等(配置段值==快照值)。
175
+ """
176
+ # 1) 同步 check_rules.expected
177
+ for rule in rules.get('check_rules', []):
178
+ resolved = _resolve_style_config(rules, rule.get('field', ''))
179
+ if resolved is not None:
180
+ rule['expected'] = resolved
181
+
182
+ # 2) 同步 fix_rules.value(经 ref_check 定位 CHK → field → 配置值)
183
+ # 保守策略:标量 value 可直接覆盖;复合 dict value(一次修复多属性)仅合并
184
+ # 格式一致的键(alignment/bold/fill),避免破坏 FIX-C013/C041 等复合修复。
185
+ chk_by_id = {r.get('id'): r for r in rules.get('check_rules', []) if r.get('id')}
186
+ for rule in rules.get('fix_rules', []):
187
+ ref = rule.get('ref_check')
188
+ if not ref or ref not in chk_by_id:
189
+ continue
190
+ resolved = _resolve_style_config(rules, chk_by_id[ref].get('field', ''))
191
+ if resolved is None or rule.get('value') is None:
192
+ continue
193
+ value = rule['value']
194
+ # 样式配置子键(如 body.align → align)→ FIX value dict 中的键名映射
195
+ _FIX_DICT_KEY = {
196
+ 'align': 'alignment',
197
+ 'bold': 'bold',
198
+ 'fill': 'fill',
199
+ }
200
+ if isinstance(value, dict):
201
+ sub_field = chk_by_id[ref].get('field', '').split('.', 1)[-1]
202
+ dk = _FIX_DICT_KEY.get(sub_field)
203
+ if dk and dk in value and not isinstance(resolved, (dict, list)):
204
+ value[dk] = resolved
205
+ else:
206
+ rule['value'] = resolved
207
+
208
+
120
209
  def _dedup_extend(base_list: list, new_items: list, dedup_key) -> None:
121
210
  """Extend base_list with new_items, replacing duplicates by dedup_key."""
122
211
  # P2-25 修复:dedup_key 返回 None 时给出 warning,避免静默追加重复项
@@ -327,6 +416,8 @@ def apply_config_overrides(rules: dict[str, Any], overrides: dict[str, Any]) ->
327
416
  if not overrides or not isinstance(overrides, dict):
328
417
  return rules
329
418
  _deep_merge(rules, copy.deepcopy(overrides))
419
+ # 配置段 → 规则期望值同步:config-overrides 覆盖后 check/optimize 自动跟随
420
+ _sync_style_expected(rules)
330
421
  return rules
331
422
 
332
423
 
@@ -110,7 +110,9 @@ def _extract_run_style(rPr) -> Dict[str, Any]:
110
110
  v = rFonts.get(f'{{{W}}}{attr}')
111
111
  if v:
112
112
  style[f'font_{attr}'] = v
113
- style['font'] = (rFonts.get(f'{{{W}}}eastAsia') or rFonts.get(f'{{{W}}}ascii') or '')
113
+ # font 仅取 eastAsia(中文字体);ascii 仅用于西文 run,
114
+ # 避免中文 run 无 eastAsia 时误学成 Times New Roman 等默认拉丁字体
115
+ style['font'] = (rFonts.get(f'{{{W}}}eastAsia') or '')
114
116
 
115
117
  sz = rPr.find(f'{{{W}}}sz')
116
118
  if sz is not None:
@@ -181,6 +183,64 @@ def _extract_para_style(pPr) -> Dict[str, Any]:
181
183
  return style
182
184
 
183
185
 
186
+ def _load_style_fonts(docx_path: str | Path) -> Dict[str, Dict[str, Any]]:
187
+ """解析 styles.xml,构建 styleId → 解析后的中文字体/字号映射。
188
+
189
+ 沿 basedOn 样式链向上解析(含环保护),最终回退到 docDefaults,
190
+ 供 run 未显式指定 eastAsia 字体时继承段落样式的中文字体。
191
+ """
192
+ fonts: Dict[str, Dict[str, Any]] = {}
193
+ with zipfile.ZipFile(docx_path) as z:
194
+ if 'word/styles.xml' not in z.namelist():
195
+ return fonts
196
+ styles = etree.fromstring(z.read('word/styles.xml'))
197
+
198
+ raw: Dict[str, Dict[str, Any]] = {}
199
+ for st in styles.findall(f'{{{W}}}style'):
200
+ sid = st.get(f'{{{W}}}styleId')
201
+ if not sid:
202
+ continue
203
+ rPr = st.find(f'{{{W}}}rPr')
204
+ rf = rPr.find(f'{{{W}}}rFonts') if rPr is not None else None
205
+ sz = rPr.find(f'{{{W}}}sz') if rPr is not None else None
206
+ bo = st.find(f'{{{W}}}basedOn')
207
+ sz_val = sz.get(f'{{{W}}}val') if sz is not None else None
208
+ raw[sid] = {
209
+ 'eastAsia': rf.get(f'{{{W}}}eastAsia') if rf is not None else None,
210
+ 'size_pt': int(sz_val) / 2.0 if sz_val else None,
211
+ 'basedOn': bo.get(f'{{{W}}}val') if bo is not None else None,
212
+ }
213
+
214
+ # docDefaults(文档默认字体)
215
+ defaults: Dict[str, Any] = {'eastAsia': None, 'size_pt': None}
216
+ dd = styles.find(f'{{{W}}}docDefaults')
217
+ if dd is not None:
218
+ rf = dd.find(f'{{{W}}}rPrDefault/{{{W}}}rPr/{{{W}}}rFonts')
219
+ if rf is not None:
220
+ defaults['eastAsia'] = rf.get(f'{{{W}}}eastAsia')
221
+ sz = dd.find(f'{{{W}}}rPrDefault/{{{W}}}rPr/{{{W}}}sz')
222
+ sz_val = sz.get(f'{{{W}}}val') if sz is not None else None
223
+ if sz_val:
224
+ defaults['size_pt'] = int(sz_val) / 2.0
225
+
226
+ def resolve(sid: str, seen: frozenset) -> Dict[str, Any]:
227
+ if sid in fonts:
228
+ return fonts[sid]
229
+ if sid in seen or sid not in raw:
230
+ return dict(defaults)
231
+ cur = raw[sid]
232
+ parent = resolve(cur.get('basedOn') or '', seen | {sid}) if cur.get('basedOn') else dict(defaults)
233
+ return {
234
+ 'eastAsia': cur.get('eastAsia') or parent.get('eastAsia'),
235
+ 'size_pt': cur.get('size_pt') if cur.get('size_pt') is not None else parent.get('size_pt'),
236
+ }
237
+
238
+ for sid in raw:
239
+ if sid not in fonts:
240
+ fonts[sid] = resolve(sid, frozenset())
241
+ return fonts
242
+
243
+
184
244
  # ---------------------------------------------------------------------------
185
245
  # 主学习流程
186
246
  # ---------------------------------------------------------------------------
@@ -199,8 +259,30 @@ def learn_style_profile(docx_path: str | Path) -> StyleProfile:
199
259
 
200
260
  profile = StyleProfile()
201
261
 
262
+ # 加载 styles.xml 样式链字体映射(供 run 未显式指定 eastAsia 时继承段落样式字体)
263
+ style_fonts = _load_style_fonts(docx_path)
264
+
202
265
  # ---- 页面设置 ----
203
- sectPr = body.find(f'{{{W}}}sectPr')
266
+ def _find_main_sect_pr(body_el) -> Optional[etree._Element]:
267
+ """定位主文档(第一节)的 sectPr。
268
+
269
+ 多 section 文档(如正文纵向 + 附件横向)第一节的 sectPr 内嵌于
270
+ 首个分节符段落 pPr;单 section 文档即 body 末尾 sectPr。
271
+ 避免用 body.find() 只能取到最后一节(如横向附件页)而学错纸张/边距。
272
+ """
273
+ for child in body_el:
274
+ tag = etree.QName(child.tag).localname if child.tag else ''
275
+ if tag == 'p':
276
+ pPr = child.find(f'{{{W}}}pPr')
277
+ if pPr is not None:
278
+ s = pPr.find(f'{{{W}}}sectPr')
279
+ if s is not None:
280
+ return s
281
+ elif tag == 'sectPr':
282
+ return child
283
+ return body_el.find(f'{{{W}}}sectPr')
284
+
285
+ sectPr = _find_main_sect_pr(body)
204
286
  if sectPr is not None:
205
287
  pgSz = sectPr.find(f'{{{W}}}pgSz')
206
288
  if pgSz is not None:
@@ -232,12 +314,33 @@ def learn_style_profile(docx_path: str | Path) -> StyleProfile:
232
314
  if not texts:
233
315
  continue
234
316
  pPr = p.find(f'{{{W}}}pPr')
317
+ # 段落样式 id(用于 run 无显式 eastAsia 时的样式继承解析)
318
+ p_style_id = None
319
+ if pPr is not None:
320
+ _ps = pPr.find(f'{{{W}}}pStyle')
321
+ if _ps is not None:
322
+ p_style_id = _ps.get(f'{{{W}}}val')
323
+ para_style_font = style_fonts.get(p_style_id, {}) if p_style_id else {}
324
+
235
325
  runs = p.findall(f'{{{W}}}r')
236
326
  run_styles = []
237
327
  for r in runs:
328
+ # 跳过无文本或纯空白 run(如落款/署名前导空格产生的空 run),
329
+ # 避免把 Word 默认 Latin 字体(Times New Roman)误学成中文样式
330
+ r_text = ''.join(t.text or '' for t in r.iter(f'{{{W}}}t')).strip()
331
+ if not r_text:
332
+ continue
238
333
  rPr = r.find(f'{{{W}}}rPr')
239
334
  if rPr is not None:
240
335
  rs = _extract_run_style(rPr)
336
+ # 样式继承:run 未显式指定 eastAsia/字号时,从段落样式链回退
337
+ # (font 可能被 _extract_run_style 置为空串 '',需用直接赋值覆盖)
338
+ if rs.get('font_eastAsia') is None and para_style_font.get('eastAsia'):
339
+ if not rs.get('font'):
340
+ rs['font'] = para_style_font['eastAsia']
341
+ rs['font_eastAsia'] = para_style_font['eastAsia']
342
+ if rs.get('size_pt') is None and para_style_font.get('size_pt'):
343
+ rs['size_pt'] = para_style_font['size_pt']
241
344
  if rs:
242
345
  run_styles.append(rs)
243
346
 
@@ -251,7 +354,14 @@ def learn_style_profile(docx_path: str | Path) -> StyleProfile:
251
354
  size = first_run.get('size_pt', 0)
252
355
  _font = first_run.get('font', '') # noqa: F841
253
356
 
254
- if idx == 0 and align == 'center':
357
+ # 红头版头:超大字号(>=36pt)居中,跳过学习(不计入 doc_title/body)
358
+ if size >= 36 and align == 'center':
359
+ role = 'letterhead'
360
+ # 发文字号:居中、含〔〕或【】、字小、文本短(如 民筹办函〔2026〕 号),跳过学习
361
+ elif align == 'center' and ('〔' in texts or '【' in texts) \
362
+ and size <= 18 and len(texts) < 30:
363
+ role = 'letterhead'
364
+ elif idx == 0 and align == 'center':
255
365
  role = 'doc_title'
256
366
  elif size >= 20 and align == 'center':
257
367
  role = 'doc_title'
@@ -261,10 +371,15 @@ def learn_style_profile(docx_path: str | Path) -> StyleProfile:
261
371
  role = 'heading_2'
262
372
  elif re.match(r'^\d+[\.、]', texts):
263
373
  role = 'heading_3'
264
- elif align == 'right' and ('年' in texts and '月' in texts and '日' in texts):
374
+ # 成文日期:整段为 YYYY年M月D日(与对齐无关,优先于落款判定)
375
+ elif re.fullmatch(r'[\s\u3000]*\d{4}年\d{1,2}月\d{1,2}日[\s\u3000]*', texts):
265
376
  role = 'date'
266
377
  elif align == 'right':
267
378
  role = 'signature'
379
+ # 落款机关名:右对齐或居中且以机关特征词结尾
380
+ elif align in ('right', 'center') and re.search(
381
+ r'(办公室|委员会|运动会|管理局|人民政府|厅|局)$', texts.strip()):
382
+ role = 'signature'
268
383
  elif texts.endswith((':', ':')):
269
384
  role = 'recipient' # 与 parser 一致:冒号结尾为主送机关
270
385
 
@@ -341,10 +456,11 @@ def build_user_rule_yaml(profile: StyleProfile, template_name: str) -> str:
341
456
  'paper_width_mm': profile.page.get('width_mm'),
342
457
  'paper_height_mm': profile.page.get('height_mm'),
343
458
  'margins': {
344
- 'top': f"{profile.margins.get('top', 2.8)}cm",
345
- 'bottom': f"{profile.margins.get('bottom', 2.8)}cm",
346
- 'left': f"{profile.margins.get('left', 2.7)}cm",
347
- 'right': f"{profile.margins.get('right', 2.7)}cm",
459
+ # margins 值存的是 mm,此处换算为 cm(与 _common.yaml 单位一致)
460
+ 'top': f"{profile.margins.get('top', 28.0) / 10.0:.1f}cm",
461
+ 'bottom': f"{profile.margins.get('bottom', 28.0) / 10.0:.1f}cm",
462
+ 'left': f"{profile.margins.get('left', 27.0) / 10.0:.1f}cm",
463
+ 'right': f"{profile.margins.get('right', 27.0) / 10.0:.1f}cm",
348
464
  },
349
465
  }
350
466
 
@@ -7,7 +7,7 @@
7
7
  #
8
8
  # 公文全流程处理工具 - gongwen-skill Python package
9
9
 
10
- __version__ = "2.4.0"
10
+ __version__ = "2.5.0"
11
11
 
12
12
  # Re-export everything from the legacy module for backward compatibility
13
13
  # This allows: from gongwen import main, cmd_check, etc.
@@ -45,7 +45,7 @@ from gongwen.cli.helpers import (
45
45
  parse_config_overrides as _parse_config_overrides,
46
46
  load_rules_with_overrides as _load_rules_with_overrides,
47
47
  )
48
- __version__ = "2.4.0"
48
+ __version__ = "2.5.0"
49
49
  # 版本号应与 gongwen/__init__.py 保持一致,每次发版同步更新
50
50
  """
51
51
  中文公文全流程处理工具 —— 基于 GB/T 9704《党政机关公文格式》国家标准。
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "gongwen-skill",
3
- "version": "2.4.0",
3
+ "version": "2.5.0",
4
4
  "description": "公文全流程处理工具 - GB/T 9704 格式检查/修复/内容优化/模板生成/版式注入",
5
5
  "type": "module",
6
6
  "main": "dsh/index.js",
@@ -201,7 +201,7 @@ python -m gongwen optimize 文件.docx -o 优化版.docx
201
201
  **Agent 响应**:
202
202
  ```
203
203
  📋 合规自检报告
204
- Skill 版本: v2.4.0
204
+ Skill 版本: v2.5.0
205
205
  路径判定: B(内容优化)
206
206
  依据: 用户提供纯文本+优化要求,无已有文档
207
207
  命令调用: 将原文保存为临时文件后执行 optimize-content --apply --paragraphs "1-3"
package/pyproject.toml CHANGED
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "gongwen-skill"
7
- version = "2.4.0"
7
+ version = "2.5.0"
8
8
  description = "公文全流程处理工具 - GB/T 9704 格式检查/修复/内容优化/模板生成/版式注入"
9
9
  readme = "README.md"
10
10
  license = {text = "MIT"}