gongwen-skill 2.0.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dsh/skills/gongwen-skill/SKILL.md +3 -4
- package/.dsh/skills/gongwen-skill.md +3 -4
- package/CHANGELOG.md +41 -0
- package/README.md +7 -6
- package/SKILL.md +3 -4
- package/cordis.patch.yml +4 -0
- package/dsh/client.js +3 -1
- package/dsh/index.js +4 -2
- package/engine/__init__.py +1 -0
- package/engine/auto_optimizer.py +5 -0
- package/engine/chat_review.py +5 -0
- package/engine/config.py +1 -0
- package/engine/core/__init__.py +1 -0
- package/engine/core/document/__init__.py +1 -0
- package/engine/core/document/_generator_helpers.py +5 -0
- package/engine/core/document/ai_structure_analyzer.py +1 -0
- package/engine/core/document/annotator.py +5 -0
- package/engine/core/document/editor.py +6 -0
- package/engine/core/document/font_utils.py +1 -0
- package/engine/core/document/generator.py +1 -0
- package/engine/core/document/markdown_converter.py +61 -2
- package/engine/core/document/models.py +1 -0
- package/engine/core/document/modifier.py +20 -5
- package/engine/core/document/ooxml_parser.py +5 -0
- package/engine/core/document/ooxml_workflow.py +5 -0
- package/engine/core/document/parser.py +30 -5
- package/engine/core/document/parser_format.py +6 -0
- package/engine/core/document/reviewer_comments.py +5 -0
- package/engine/core/document/structure_analyzer.py +1 -0
- package/engine/core/document/tracked_annotator.py +5 -0
- package/engine/core/document/tracked_changes.py +5 -0
- package/engine/core/document/tracked_common.py +5 -0
- package/engine/core/rules/__init__.py +1 -0
- package/engine/core/rules/checker.py +230 -13
- package/engine/core/rules/engine.py +1 -0
- package/engine/core/rules/fixer.py +1 -0
- package/engine/core/rules/loader.py +1 -0
- package/engine/core/rules/manager.py +1 -0
- package/engine/docx_to_image.py +5 -0
- package/engine/fact_check.py +5 -0
- package/engine/focus_checker.py +5 -0
- package/engine/handoff.py +1 -0
- package/engine/inject.py +27 -9
- package/engine/live_edit.py +5 -0
- package/engine/optimizer.py +1 -1
- package/engine/review_generator.py +5 -0
- package/engine/structure_checker.py +5 -0
- package/engine/style_profile.py +5 -0
- package/engine/table_sign_generator.py +5 -0
- package/engine/table_sign_template.py +6 -0
- package/engine/template_builder.py +1 -0
- package/engine/utils/__init__.py +6 -0
- package/engine/utils/errors.py +5 -0
- package/engine/utils/logger.py +1 -0
- package/engine/utils/parse.py +5 -0
- package/engine/utils/zip_utils.py +5 -0
- package/gongwen/__init__.py +2 -1
- package/gongwen/__main__.py +1 -0
- package/gongwen/_bootstrap.py +6 -0
- package/gongwen/_legacy.py +32 -13
- package/gongwen/cli/__init__.py +1 -0
- package/gongwen/cli/content_cmds.py +6 -0
- package/gongwen/cli/doctor_cmds.py +6 -0
- package/gongwen/cli/font_cmds.py +7 -0
- package/gongwen/cli/helpers.py +7 -0
- package/gongwen/cli/misc_cmds.py +6 -0
- package/gongwen/cli/review_cmds.py +6 -0
- package/gongwen/cli/style_helpers.py +6 -0
- package/gongwen/cli/update_cmds.py +6 -0
- package/gongwen/md2docx_render.py +509 -0
- package/package.json +1 -1
- package/prompts/style-prompts.md +5 -4
- package/prompts/usage-prompts.md +7 -1
- package/pyproject.toml +1 -1
- package/rules/official/_common.yaml +7 -3
- package/rules/official/announcement.yaml +4 -0
- package/rules/official/bill.yaml +13 -0
- package/rules/official/bulletin.yaml +13 -0
- package/rules/official/command.yaml +4 -0
- package/rules/official/communique.yaml +21 -0
- package/rules/official/decision.yaml +4 -0
- package/rules/official/instruction.yaml +13 -0
- package/rules/official/letter.yaml +4 -0
- package/rules/official/meeting.yaml +13 -0
- package/rules/official/minutes.yaml +13 -0
- package/rules/official/news.yaml +50 -0
- package/rules/official/notice.yaml +27 -1
- package/rules/official/notice_public.yaml +4 -0
- package/rules/official/opinion.yaml +13 -0
- package/rules/official/regulation.yaml +13 -0
- package/rules/official/reply.yaml +13 -0
- package/rules/official/report.yaml +13 -0
- package/rules/official/request.yaml +4 -0
- package/rules/official/resolution.yaml +13 -0
- package/rules/official/speech.yaml +4 -0
- package/rules/official/summary.yaml +13 -0
- package/rules/official/table_sign.yaml +4 -0
- package/rules/official/technical_proposal.yaml +4 -0
- package/rules/official/work_plan.yaml +13 -0
|
@@ -14,10 +14,9 @@ metadata:
|
|
|
14
14
|
---
|
|
15
15
|
|
|
16
16
|
<!--
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
Licensed under the MIT License. See the LICENSE file for details.
|
|
17
|
+
(c) 2026 Jose AI (https://www.linhut.cn)
|
|
18
|
+
https://github.com/linhut/gongwen-skill
|
|
19
|
+
Licensed under the MIT License. See the LICENSE file for details.
|
|
21
20
|
-->
|
|
22
21
|
|
|
23
22
|
# 公文文档格式化 Skill(GB/T 9704)
|
|
@@ -14,10 +14,9 @@ metadata:
|
|
|
14
14
|
---
|
|
15
15
|
|
|
16
16
|
<!--
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
Licensed under the MIT License. See the LICENSE file for details.
|
|
17
|
+
(c) 2026 Jose AI (https://www.linhut.cn)
|
|
18
|
+
https://github.com/linhut/gongwen-skill
|
|
19
|
+
Licensed under the MIT License. See the LICENSE file for details.
|
|
21
20
|
-->
|
|
22
21
|
|
|
23
22
|
# 公文文档格式化 Skill(GB/T 9704)
|
package/CHANGELOG.md
CHANGED
|
@@ -1,3 +1,44 @@
|
|
|
1
|
+
<!--
|
|
2
|
+
(c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
https://github.com/linhut/gongwen-skill
|
|
4
|
+
Licensed under the MIT License. See the LICENSE file for details.
|
|
5
|
+
-->
|
|
6
|
+
|
|
7
|
+
## v2.2.0 (2026-08-28)
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
- **内容要素检查规则真正生效**:修复 37 条 `content.*` 规则(通知事项/会议要素/请示理由等)此前"定义但从不执行"的问题,新增 `_check_content_field` 按字段名映射要素关键词做宽松检查,消除每次 check 的"未支持检查字段"告警
|
|
11
|
+
- **版头检查规则生效**:修复 `header.*` 规则(CHK-CM002 令号检查、CHK-R003 主送机关检查)此前无分发分支的问题,新增 `_check_header_field`
|
|
12
|
+
- **成文日期"右空四字"**:md2docx 渲染日期段新增右缩进 4em(GB/T 9704 规范),消除 CHK-C017 长期存在的日期右空四字检查项
|
|
13
|
+
|
|
14
|
+
### Changed
|
|
15
|
+
- **signature 段落选择器修正**:`_select_paragraphs` 的 `signature` target 不再包含 `date` 段落——修复 FIX-C013(署名居中)与 FIX-C013b(18pt)误把成文日期强制居中/放大字号的问题,日期段独立按"右对齐 + 右空四字 + 16pt"处理
|
|
16
|
+
- **页码重复注入幂等化**:`_inject_even_page_footer_direct` 重写为幂等版,多次注入仅保留一个 even footerReference 与完整关系,杜绝 Word 打开报"文档损坏"(BUG-1)
|
|
17
|
+
- **加粗+链接/代码组合不再整段加粗**:markdown_converter 单片段按自身 bold 标志处理,重建片段继承原 run 字体字号(BUG-2)
|
|
18
|
+
- **标题启发式误判修复**:黑体/楷体/仿宋加粗正文句不再误判为标题(新增 `_SENTENCE_END` 句末标点保护,Level 1/2/3 三级判定)
|
|
19
|
+
- **CHK-C030 整段加粗误报修复**:忽略纯标点 run、单句段落不报,多句整段加粗仍正确检出
|
|
20
|
+
|
|
21
|
+
### Fixed
|
|
22
|
+
- `is_title` vs `is_heading` 属性引用错误:`_check_ending`/`_check_content_field` 此前用不存在的 `is_title` 属性,导致标题段被误纳入正文统计
|
|
23
|
+
- 一是/二是 领句段改用仿宋_GB2312(原误用楷体触发标题误判/正文字体误报)
|
|
24
|
+
- 令号正则 `\d` 被 JS 转义破坏(`d`),导致含令号命令误报 CHK-CM002
|
|
25
|
+
|
|
26
|
+
## v2.1.0 (2026-08-20)
|
|
27
|
+
|
|
28
|
+
### Added
|
|
29
|
+
- **格式规则完善**:为 24 种公文类型新增 30+ 条 check/fix 规则,提升规则覆盖率
|
|
30
|
+
- `news.yaml`:从 0 条规则扩展至 5 条(导语段/稿源信息/标题风格/标题字数/听取通报段检查)
|
|
31
|
+
- `notice.yaml`:新增标题格式和通知事项检查
|
|
32
|
+
- `communique.yaml`:新增公报签署和标题格式检查
|
|
33
|
+
- 12 个原只有 2 条规则的类型各新增 1 条检查规则(bill/bulletin/instruction/meeting/minutes/opinion/regulation/reply/report/resolution/summary/work_plan)
|
|
34
|
+
- **版本发布清单**:RELEASE.md 扩展版本号检查点从 4 处代码→9 处(含文档/提示词/DSH 桥接)
|
|
35
|
+
- **pre-commit hook**:`.githooks/pre-commit` 自动检查 4 处代码版本号一致性,阻止版本漂移的提交
|
|
36
|
+
- **动态 PyPI 徽章**:README.md 的 PyPI 徽章改为动态版本号(`https://img.shields.io/pypi/v/gongwen-skill`),自动显示最新版本
|
|
37
|
+
|
|
38
|
+
### Changed
|
|
39
|
+
- RELEASE.md 全面更新:修复 12 处残留 1.12.x 引用,更新 npm 发布状态,新增 pre-commit hook 说明
|
|
40
|
+
- 规则文件统一添加 `template_name` 和 `document_type` 字段(news.yaml 补齐)
|
|
41
|
+
|
|
1
42
|
## v2.0.0 (2026-08-20)
|
|
2
43
|
|
|
3
44
|
### Major
|
package/README.md
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
<!--
|
|
2
2
|
(c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
https://github.com/linhut/gongwen-skill
|
|
3
4
|
Licensed under the MIT License. See the LICENSE file for details.
|
|
4
5
|
-->
|
|
5
6
|
|
|
@@ -12,7 +13,7 @@ Licensed under the MIT License. See the LICENSE file for details.
|
|
|
12
13
|
> 中文公文全流程处理工具——基于 **GB/T 9704《党政机关公文格式》** 国家标准,支持 **格式检查与修复、内容优化(Word 原生修订+批注/差异对比版)、模板生成、Markdown 转公文、版头版记页码注入、事实核验、风格增强** 等完整能力。原生支持 **DeepSeek Harness (DSH)** 技能系统,打包为可被 AI Agent 直接调用的 Skill,完全自包含,克隆即用。
|
|
13
14
|
|
|
14
15
|
[](https://github.com/linhut/gongwen-skill/actions)
|
|
15
|
-
[](https://pypi.org/project/gongwen-skill/)
|
|
16
17
|
[](./LICENSE)
|
|
17
18
|

|
|
18
19
|

|
|
@@ -336,7 +337,7 @@ DSH 采用 **Cordis 模块化微内核架构**:技能体系基于本地文件
|
|
|
336
337
|
git clone https://github.com/linhut/gongwen-skill.git
|
|
337
338
|
cd gongwen-skill
|
|
338
339
|
pip install -r requirements.txt # 或 pip install gongwen-skill(已上 PyPI)
|
|
339
|
-
python -m gongwen --version # 检验:gongwen-skill v2.
|
|
340
|
+
python -m gongwen --version # 检验:gongwen-skill v2.2.0
|
|
340
341
|
```
|
|
341
342
|
|
|
342
343
|
### 方式一:作为 DSH Skill 注册(基于本地文件系统)
|
|
@@ -392,7 +393,7 @@ pnpm add -w gongwen-skill
|
|
|
392
393
|
"dependencies": {
|
|
393
394
|
"@deepseek-ai/dsh-base": "...",
|
|
394
395
|
"@deepseek-ai/dsh-web-app": "...",
|
|
395
|
-
"gongwen-skill": "^2.
|
|
396
|
+
"gongwen-skill": "^2.2.0"
|
|
396
397
|
},
|
|
397
398
|
"dsh": {
|
|
398
399
|
"profile": {
|
|
@@ -447,9 +448,9 @@ dsh --profile web
|
|
|
447
448
|
| CLI 独立可执行(`python -m gongwen <命令>`) | ✅ |
|
|
448
449
|
| PyPI 上架(`pip install gongwen-skill`) | ✅ |
|
|
449
450
|
| 零外部运行时依赖(仅 python-docx/pydantic/pyyaml) | ✅ |
|
|
450
|
-
| DSH 配置化排版参数(页边距/行距/字体/默认模板版本) | ✅ v2.
|
|
451
|
+
| DSH 配置化排版参数(页边距/行距/字体/默认模板版本) | ✅ v2.2.0+ |
|
|
451
452
|
|
|
452
|
-
### DSH 插件配置化(v2.
|
|
453
|
+
### DSH 插件配置化(v2.2.0+)
|
|
453
454
|
|
|
454
455
|
DSH 插件支持通过配置文件管理排版参数,Agent 调用时自动注入,纯 CLI 用户不受影响。
|
|
455
456
|
|
|
@@ -573,7 +574,7 @@ pip install -r requirements.txt
|
|
|
573
574
|
用户:帮我优化这份会议通知的第二章节措辞
|
|
574
575
|
|
|
575
576
|
Agent:📋 合规自检报告
|
|
576
|
-
Skill 版本: v2.
|
|
577
|
+
Skill 版本: v2.2.0(多渠道自检已确认最新)
|
|
577
578
|
路径判定: B(内容优化)
|
|
578
579
|
依据: 用户指定了已有文档,且要求"优化措辞"
|
|
579
580
|
命令调用: 1. python -m gongwen optimize-content 会议通知.docx --changes changes.json --apply --paragraphs "5-8"
|
package/SKILL.md
CHANGED
|
@@ -14,10 +14,9 @@ metadata:
|
|
|
14
14
|
---
|
|
15
15
|
|
|
16
16
|
<!--
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
Licensed under the MIT License. See the LICENSE file for details.
|
|
17
|
+
(c) 2026 Jose AI (https://www.linhut.cn)
|
|
18
|
+
https://github.com/linhut/gongwen-skill
|
|
19
|
+
Licensed under the MIT License. See the LICENSE file for details.
|
|
21
20
|
-->
|
|
22
21
|
|
|
23
22
|
# 公文文档格式化 Skill(GB/T 9704)
|
package/cordis.patch.yml
CHANGED
package/dsh/client.js
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
// 公文全流程处理工具 DSH client — 可视化配置面板(gongwen-skill,不做皮肤,仅设置面板)
|
|
2
|
-
// (c) 2026 Jose AI (https://www.linhut.cn)
|
|
2
|
+
// (c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
// https://github.com/linhut/gongwen-skill
|
|
4
|
+
// Licensed under the MIT License. See the LICENSE file for details.
|
|
3
5
|
//
|
|
4
6
|
// 在 DSH Web GUI 系统设置 → 插件配置 中渲染公文排版参数编辑界面。
|
|
5
7
|
// 通过 /plugins/gongwen-skill/api/config (GET/POST) 与宿主端通信。
|
package/dsh/index.js
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
// 公文全流程处理工具 - DSH plugin bridge (gongwen-skill, v2.
|
|
2
|
-
// (c) 2026 Jose AI (https://www.linhut.cn)
|
|
1
|
+
// 公文全流程处理工具 - DSH plugin bridge (gongwen-skill, v2.2.0+)
|
|
2
|
+
// (c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
// https://github.com/linhut/gongwen-skill
|
|
4
|
+
// Licensed under the MIT License. See the LICENSE file for details.
|
|
3
5
|
//
|
|
4
6
|
// 分层架构:
|
|
5
7
|
// - Python CLI:纯工具层,通过 --config-overrides 接收规则覆盖 JSON
|
package/engine/__init__.py
CHANGED
package/engine/auto_optimizer.py
CHANGED
package/engine/chat_review.py
CHANGED
package/engine/config.py
CHANGED
package/engine/core/__init__.py
CHANGED
|
@@ -1,4 +1,9 @@
|
|
|
1
1
|
# -*- coding: utf-8 -*-
|
|
2
|
+
#
|
|
3
|
+
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
4
|
+
# https://github.com/linhut/gongwen-skill
|
|
5
|
+
# Licensed under the MIT License. See the LICENSE file for details.
|
|
6
|
+
#
|
|
2
7
|
"""
|
|
3
8
|
Generator helper functions: table and page number field creation.
|
|
4
9
|
Extracted from generator.py (tier-2 split).
|
|
@@ -1,3 +1,9 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
#
|
|
3
|
+
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
4
|
+
# https://github.com/linhut/gongwen-skill
|
|
5
|
+
# Licensed under the MIT License. See the LICENSE file for details.
|
|
6
|
+
#
|
|
1
7
|
"""
|
|
2
8
|
Content revision engine for official document optimization.
|
|
3
9
|
Produces comparison documents showing original vs revised content
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# This file is part of the Official Document AI Assistant.
|
|
2
2
|
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
# https://github.com/linhut/gongwen-skill
|
|
3
4
|
# Licensed under the MIT License. See the LICENSE file for details.
|
|
4
5
|
"""
|
|
5
6
|
Font utilities: Handle Chinese font settings correctly for Word documents.
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# This file is part of the Official Document AI Assistant.
|
|
2
2
|
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
# https://github.com/linhut/gongwen-skill
|
|
3
4
|
# Licensed under the MIT License. See the LICENSE file for details.
|
|
4
5
|
"""
|
|
5
6
|
Document generator: converts DocumentModel back into a .docx file.
|
|
@@ -1,4 +1,9 @@
|
|
|
1
1
|
# -*- coding: utf-8 -*-
|
|
2
|
+
#
|
|
3
|
+
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
4
|
+
# https://github.com/linhut/gongwen-skill
|
|
5
|
+
# Licensed under the MIT License. See the LICENSE file for details.
|
|
6
|
+
#
|
|
2
7
|
"""
|
|
3
8
|
Markdown → DocumentModel 转换器。
|
|
4
9
|
从 modifier.py 提取(阶梯2 拆分)。
|
|
@@ -46,6 +51,32 @@ _MD_CODE_BLOCK_RE = re.compile(r'^`{3,}')
|
|
|
46
51
|
# 行内代码 `code`
|
|
47
52
|
_MD_INLINE_CODE_RE = re.compile(r'`([^`]+)`')
|
|
48
53
|
|
|
54
|
+
# 加粗标记(**text** 或 __text__)
|
|
55
|
+
_MD_BOLD_SEG_RE = re.compile(r'(\*\*[^*]+\*\*|__[^_]+__)')
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _reconstruct_bold_segments(raw: str, cleaned: str):
|
|
59
|
+
"""把含 **..** 的原始文本映射为 [(文本, 是否加粗), ...] 片段。
|
|
60
|
+
|
|
61
|
+
片段拼接结果与 cleaned 一致时才返回片段列表;否则(含链接/代码等被清理、
|
|
62
|
+
或文本被折叠)回退为整段(避免丢失文本)。
|
|
63
|
+
"""
|
|
64
|
+
if not raw or ('**' not in raw and '__' not in raw):
|
|
65
|
+
return [(cleaned or "", False)]
|
|
66
|
+
segments = []
|
|
67
|
+
for part in _MD_BOLD_SEG_RE.split(raw):
|
|
68
|
+
if not part:
|
|
69
|
+
continue
|
|
70
|
+
if part.startswith('**') and part.endswith('**') and len(part) > 4:
|
|
71
|
+
segments.append((part[2:-2], True))
|
|
72
|
+
elif part.startswith('__') and part.endswith('__') and len(part) > 4:
|
|
73
|
+
segments.append((part[2:-2], True))
|
|
74
|
+
else:
|
|
75
|
+
segments.append((part, False))
|
|
76
|
+
if ''.join(s[0] for s in segments).strip() == (cleaned or "").strip():
|
|
77
|
+
return segments
|
|
78
|
+
return [(cleaned or "", False)]
|
|
79
|
+
|
|
49
80
|
|
|
50
81
|
def convert_markdown(model: DocumentModel) -> int:
|
|
51
82
|
"""
|
|
@@ -196,6 +227,7 @@ def convert_markdown(model: DocumentModel) -> int:
|
|
|
196
227
|
# --- 识别加粗标记 **text** ---
|
|
197
228
|
|
|
198
229
|
has_bold = False
|
|
230
|
+
raw_before_bold = text # 保存剥离 ** 前的原始文本,用于加粗段重构
|
|
199
231
|
if _MD_BOLD_RE.search(text) or _MD_BOLD_UNDER_RE.search(text):
|
|
200
232
|
has_bold = True
|
|
201
233
|
text = _MD_BOLD_RE.sub(r'\1', text)
|
|
@@ -252,8 +284,35 @@ def convert_markdown(model: DocumentModel) -> int:
|
|
|
252
284
|
r.text = ""
|
|
253
285
|
|
|
254
286
|
if has_bold and not para.is_heading:
|
|
255
|
-
|
|
256
|
-
|
|
287
|
+
# 仅加粗 **..** 标记的片段,避免整段误加粗(CHK-C030 修复)
|
|
288
|
+
_segments = _reconstruct_bold_segments(raw_before_bold, para.text)
|
|
289
|
+
if len(_segments) > 1:
|
|
290
|
+
# 重建片段时继承原 run 的字体/字号,避免丢失字体(I20 修复)
|
|
291
|
+
_base_format = None
|
|
292
|
+
for _r in para.runs:
|
|
293
|
+
if _r.text and (_r.format.font_name or _r.format.font_size_pt):
|
|
294
|
+
_base_format = _r.format
|
|
295
|
+
break
|
|
296
|
+
_new_runs = []
|
|
297
|
+
for _t, _b in _segments:
|
|
298
|
+
if not _t:
|
|
299
|
+
continue
|
|
300
|
+
_nr = Run(index=len(_new_runs), text=_t,
|
|
301
|
+
format=RunFormat(
|
|
302
|
+
font_name=_base_format.font_name if _base_format else None,
|
|
303
|
+
font_size_pt=_base_format.font_size_pt if _base_format else 16.0,
|
|
304
|
+
))
|
|
305
|
+
if _b:
|
|
306
|
+
_nr.format.bold = True
|
|
307
|
+
_new_runs.append(_nr)
|
|
308
|
+
para.runs = _new_runs
|
|
309
|
+
else:
|
|
310
|
+
# 单片段:按其自身 bold 标志处理——纯粗体段(**全段**)整段加粗;
|
|
311
|
+
# raw/cleaned 不匹配回退的单片段 bold=False 保持不加粗(I20 修复:
|
|
312
|
+
# 此前把回退也整段加粗,导致"加粗+链接/代码"段落整段变粗)
|
|
313
|
+
if _segments and _segments[0][1]:
|
|
314
|
+
for r in para.runs:
|
|
315
|
+
r.format.bold = True
|
|
257
316
|
|
|
258
317
|
if is_list and list_indent_pt > 0 and not para.is_heading:
|
|
259
318
|
para.format.left_indent_pt = list_indent_pt
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# This file is part of the Official Document AI Assistant.
|
|
2
2
|
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
# https://github.com/linhut/gongwen-skill
|
|
3
4
|
# Licensed under the MIT License. See the LICENSE file for details.
|
|
4
5
|
"""
|
|
5
6
|
Pydantic data models for the intermediate document representation.
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# This file is part of the Official Document AI Assistant.
|
|
2
2
|
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
# https://github.com/linhut/gongwen-skill
|
|
3
4
|
# Licensed under the MIT License. See the LICENSE file for details.
|
|
4
5
|
"""
|
|
5
6
|
Document modifier: the single source of truth for all DocumentModel mutations.
|
|
@@ -73,19 +74,33 @@ def _select_paragraphs(model: DocumentModel, target: str) -> list[Paragraph]:
|
|
|
73
74
|
return [p for p in model.paragraphs
|
|
74
75
|
if not p.is_heading and p.text.strip() and id(p) not in sig_set]
|
|
75
76
|
elif target == "signature":
|
|
76
|
-
#
|
|
77
|
-
|
|
77
|
+
# P2-24 修复:signature target 只匹配署名段(role='signature'),不再包含 date——
|
|
78
|
+
# 此前匹配 role in ('signature','date') 会把落款日期段一并选中,
|
|
79
|
+
# 导致 FIX-C013(署名居中)把日期改成 center、FIX-C013b(18pt)把日期改成 18pt,
|
|
80
|
+
# 违反 GB/T 9704 成文日期"右空四字、三号仿宋(16pt)"的规范。
|
|
81
|
+
# 日期段由 target="date" 分支独立处理(右对齐 + right_indent 保留)。
|
|
82
|
+
role_sig = [p for p in model.paragraphs if p.role == 'signature']
|
|
78
83
|
if role_sig:
|
|
79
84
|
return role_sig
|
|
80
|
-
|
|
81
|
-
|
|
85
|
+
# 仅当无署名 role 时,才允许位置回退(末两段:署名+日期);
|
|
86
|
+
# 回退时仍只修署名语义的段落,不影响日期段对齐
|
|
87
|
+
non_empty = [p for p in model.paragraphs if p.text.strip() and p.role != 'date']
|
|
88
|
+
if len(non_empty) >= 2:
|
|
89
|
+
last = non_empty[-1].text.strip()
|
|
90
|
+
if re.match(r'^\d{4}年\d{1,2}月\d{1,2}日$', last) or re.match(r'^\d{4}[.\-/]\d{1,2}[.\-/]\d{1,2}$', last):
|
|
91
|
+
return non_empty[-1:]
|
|
92
|
+
return []
|
|
82
93
|
elif target == "date":
|
|
83
94
|
# 同 signature 的处理逻辑
|
|
84
95
|
role_date = [p for p in model.paragraphs if p.role == 'date']
|
|
85
96
|
if role_date:
|
|
86
97
|
return role_date
|
|
87
98
|
non_empty = [p for p in model.paragraphs if p.text.strip()]
|
|
88
|
-
|
|
99
|
+
if non_empty:
|
|
100
|
+
last = non_empty[-1].text.strip()
|
|
101
|
+
if re.match(r'^\d{4}年\d{1,2}月\d{1,2}日$', last) or re.match(r'^\d{4}[.\-/]\d{1,2}[.\-/]\d{1,2}$', last):
|
|
102
|
+
return non_empty[-1:]
|
|
103
|
+
return []
|
|
89
104
|
elif target in ('salutation', 'introduction', 'transition', 'meeting_date', 'numbered_body'):
|
|
90
105
|
# N2: 段落类型 target —— 使用 detect_paragraph_type 内容匹配
|
|
91
106
|
return [p for p in model.paragraphs if detect_paragraph_type(p.text, p.role) == target]
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# This file is part of the Official Document AI Assistant.
|
|
2
2
|
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
3
|
+
# https://github.com/linhut/gongwen-skill
|
|
3
4
|
# Licensed under the MIT License. See the LICENSE file for details.
|
|
4
5
|
"""
|
|
5
6
|
Document parser: converts a .docx file into DocumentModel.
|
|
@@ -50,6 +51,10 @@ _H1_WESTERN_PATTERN = re.compile(r'^[IVXLCDM]+\.\s+')
|
|
|
50
51
|
_H2_WESTERN_PATTERN = re.compile(r'^[A-Z]\.\s+')
|
|
51
52
|
_H4_WESTERN_PATTERN = re.compile(r'^[a-z]\.\s+')
|
|
52
53
|
|
|
54
|
+
# 句末标点(P2-16 修复:标题不以这些标点结尾,正文强调句/引用句通常会以它们结尾,
|
|
55
|
+
# 用于排除"黑体/楷体/仿宋加粗的正文短句"被误判为标题)
|
|
56
|
+
_SENTENCE_END = ('。', '!', '?', ':', ';', ',', '.', '!', '?', ':', ';', ',')
|
|
57
|
+
|
|
53
58
|
|
|
54
59
|
def _detect_heading_heuristic(
|
|
55
60
|
text: str, runs: list[Run], para_format: ParagraphFormat
|
|
@@ -102,7 +107,13 @@ def _detect_heading_heuristic(
|
|
|
102
107
|
|
|
103
108
|
# --- Level 1: 一级标题(黑体)---
|
|
104
109
|
if has_font_signal and ("黑体" in font_lower or font_lower in ("simhei",)):
|
|
105
|
-
|
|
110
|
+
# P2-16 修复:排除以句末标点结尾的正文强调句(如"重要提示:…")。
|
|
111
|
+
# 一级标题通常较短且不以句号/冒号等结尾。
|
|
112
|
+
if alignment == "center":
|
|
113
|
+
return True, 1
|
|
114
|
+
if len(text_stripped) < 30 and not text_stripped.endswith(_SENTENCE_END):
|
|
115
|
+
return True, 1
|
|
116
|
+
if is_bold and not text_stripped.endswith(_SENTENCE_END):
|
|
106
117
|
return True, 1
|
|
107
118
|
|
|
108
119
|
# 格式信号:"一、" + 加粗或黑体
|
|
@@ -114,9 +125,15 @@ def _detect_heading_heuristic(
|
|
|
114
125
|
if _H1_WESTERN_PATTERN.match(text_stripped) and len(text_stripped) < 50:
|
|
115
126
|
return True, 1
|
|
116
127
|
|
|
117
|
-
# --- Level 2:
|
|
118
|
-
|
|
119
|
-
|
|
128
|
+
# --- Level 2: 二级标题(楷体,无需加粗)---
|
|
129
|
+
# GB/T 9704 二级标题为楷体,不加粗。若段落字体为楷体(含 楷体_GB2312),
|
|
130
|
+
# 即使内容为 "2.1" 等数字编号,也优先识别为二级标题而非三级标题,
|
|
131
|
+
# 避免 converter 的 "###"→二级标题与 check 的 "2.1"→三级标题启发式冲突。
|
|
132
|
+
# P2-16 修复:增加"不以句末标点结尾"约束,排除正文楷体引用句
|
|
133
|
+
# (如"会议指出,…。")被误判为二级标题。
|
|
134
|
+
if has_font_signal and ("楷体" in font_lower or font_lower in ("kaiti", "楷体_gb2312")):
|
|
135
|
+
if len(text_stripped) < 40 and not text_stripped.endswith(_SENTENCE_END):
|
|
136
|
+
return True, 2
|
|
120
137
|
|
|
121
138
|
# "(一)" 格式(无论字体如何,此模式足够唯一)
|
|
122
139
|
if _H2_PATTERN.match(text_stripped) and len(text_stripped) < 50:
|
|
@@ -127,7 +144,8 @@ def _detect_heading_heuristic(
|
|
|
127
144
|
|
|
128
145
|
# --- Level 3: 三级标题(仿宋加粗 或 "1." + 加粗)---
|
|
129
146
|
if has_font_signal and ("仿宋" in font_lower or font_lower in ("fangsong", "仿宋_gb2312")) and is_bold:
|
|
130
|
-
|
|
147
|
+
# P2-16 修复:排除以句末标点结尾的仿宋加粗正文短句被误判为三级标题
|
|
148
|
+
if len(text_stripped) < 50 and not text_stripped.endswith(_SENTENCE_END):
|
|
131
149
|
return True, 3
|
|
132
150
|
|
|
133
151
|
if _H3_PATTERN.match(text_stripped) and is_bold and len(text_stripped) < 60:
|
|
@@ -339,6 +357,13 @@ def _assign_paragraph_roles(paragraphs: list[Paragraph]) -> None:
|
|
|
339
357
|
if not para.role and _ATTACHMENT_RE.match(para.text.strip()):
|
|
340
358
|
para.role = 'attachment'
|
|
341
359
|
|
|
360
|
+
# AI 声明段(末尾批注,如"(内容由GongWen-skill-AI生成,仅供参考)")
|
|
361
|
+
# 标记为 annotation 角色,避免 check 将其误判为正文并报格式违规
|
|
362
|
+
_AI_DECL_MARKERS = ("由GongWen-skill-AI生成", "由AI生成", "仅供参考")
|
|
363
|
+
for idx, para in non_empty:
|
|
364
|
+
if not para.role and any(m in para.text for m in _AI_DECL_MARKERS):
|
|
365
|
+
para.role = 'annotation'
|
|
366
|
+
|
|
342
367
|
# 其余非空段落默认为 body
|
|
343
368
|
for idx, para in non_empty:
|
|
344
369
|
if not para.role:
|
|
@@ -1,4 +1,9 @@
|
|
|
1
1
|
# -*- coding: utf-8 -*-
|
|
2
|
+
#
|
|
3
|
+
# (c) 2026 Jose AI (https://www.linhut.cn)
|
|
4
|
+
# https://github.com/linhut/gongwen-skill
|
|
5
|
+
# Licensed under the MIT License. See the LICENSE file for details.
|
|
6
|
+
#
|
|
2
7
|
"""
|
|
3
8
|
tracked 修订公共工具(P2-10 修复:消除 tracked_changes / tracked_annotator 重复代码)。
|
|
4
9
|
|