@topmindspace/tms-skills 2.0.0 → 2.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/CHANGELOG.md +139 -123
  2. package/README.md +126 -103
  3. package/bin/tms-skills.js +160 -160
  4. package/package.json +48 -48
  5. package/top-ppt-html/README.md +254 -253
  6. package/top-ppt-html/SKILL.md +111 -111
  7. package/top-ppt-html/assets/examples/2026-09-09-architecture-graphite-dark.html +3873 -3926
  8. package/top-ppt-html/assets/examples/2026-09-09-architecture-graphite-dark.model.json +167 -167
  9. package/top-ppt-html/assets/examples/2026-09-09-architecture-spectrum.html +3873 -3926
  10. package/top-ppt-html/assets/examples/2026-09-09-architecture-spectrum.model.json +167 -167
  11. package/top-ppt-html/assets/examples/2026-09-09-presentation-apple-mono.html +4272 -4325
  12. package/top-ppt-html/assets/examples/2026-09-09-presentation-apple-mono.model.json +320 -320
  13. package/top-ppt-html/assets/examples/2026-09-09-presentation-brand-red.html +4272 -4325
  14. package/top-ppt-html/assets/examples/2026-09-09-presentation-brand-red.model.json +320 -320
  15. package/top-ppt-html/assets/examples/2026-09-09-presentation-business-blue.html +4272 -4325
  16. package/top-ppt-html/assets/examples/2026-09-09-presentation-business-blue.model.json +320 -320
  17. package/top-ppt-html/assets/examples/2026-09-09-research-deep-teal.html +5474 -5527
  18. package/top-ppt-html/assets/examples/2026-09-09-research-deep-teal.model.json +913 -913
  19. package/top-ppt-html/assets/examples/2026-09-09-research-indigo-violet.html +5474 -5527
  20. package/top-ppt-html/assets/examples/2026-09-09-research-indigo-violet.model.json +913 -913
  21. package/top-ppt-html/assets/examples/2026-09-09-research-mckinsey.html +5474 -5527
  22. package/top-ppt-html/assets/examples/2026-09-09-research-mckinsey.model.json +913 -913
  23. package/top-ppt-html/assets/examples/2026-09-09-research-warm-sand.html +5474 -5527
  24. package/top-ppt-html/assets/examples/2026-09-09-research-warm-sand.model.json +913 -913
  25. package/top-ppt-html/assets/pptx-export.js +1944 -1944
  26. package/top-ppt-html/assets/style-gallery.html +559 -589
  27. package/top-ppt-html/assets/templates/architecture.html +3675 -3728
  28. package/top-ppt-html/assets/templates/engine.css +787 -840
  29. package/top-ppt-html/assets/templates/presentation.html +3685 -3738
  30. package/top-ppt-html/assets/templates/research.html +3964 -4017
  31. package/top-ppt-html/assets/templates/ui.js +520 -520
  32. package/top-ppt-html/assets/theme-overview-architecture.png +0 -0
  33. package/top-ppt-html/assets/theme-overview-presentation.png +0 -0
  34. package/top-ppt-html/assets/theme-overview-research.png +0 -0
  35. package/top-ppt-html/assets/theme-overview.png +0 -0
  36. package/top-ppt-html/evals/prompts.csv +15 -15
  37. package/top-ppt-html/evals/rubric.schema.json +25 -25
  38. package/top-ppt-html/evals/run_evals.py +220 -220
  39. package/top-ppt-html/evals/trace.example.json +16 -16
  40. package/top-ppt-html/package-lock.json +175 -0
  41. package/top-ppt-html/package.json +30 -35
  42. package/top-ppt-html/references/charts-basic.md +624 -624
  43. package/top-ppt-html/references/charts-discipline.md +108 -108
  44. package/top-ppt-html/references/charts-extended.md +482 -482
  45. package/top-ppt-html/references/charts.md +28 -28
  46. package/top-ppt-html/references/components-atoms.md +624 -624
  47. package/top-ppt-html/references/components.md +30 -30
  48. package/top-ppt-html/references/content-rules.md +490 -490
  49. package/top-ppt-html/references/design-system-engine.md +235 -235
  50. package/top-ppt-html/references/design-system.md +471 -478
  51. package/top-ppt-html/references/failure-modes.md +214 -214
  52. package/top-ppt-html/references/high-fidelity.md +127 -127
  53. package/top-ppt-html/references/icons.md +397 -397
  54. package/top-ppt-html/references/industry-benchmark.md +105 -105
  55. package/top-ppt-html/references/infographics-stats.md +308 -308
  56. package/top-ppt-html/references/infographics-structure.md +226 -226
  57. package/top-ppt-html/references/infographics.md +43 -43
  58. package/top-ppt-html/references/layout-grammar.md +315 -315
  59. package/top-ppt-html/references/layouts-architecture.md +108 -108
  60. package/top-ppt-html/references/layouts-combo.md +600 -600
  61. package/top-ppt-html/references/layouts-research.md +160 -160
  62. package/top-ppt-html/references/modes.md +254 -254
  63. package/top-ppt-html/references/outline-design.md +275 -275
  64. package/top-ppt-html/references/playbook.md +266 -266
  65. package/top-ppt-html/references/pptx-export.md +209 -209
  66. package/top-ppt-html/references/reform-plan.md +252 -252
  67. package/top-ppt-html/references/styles.md +336 -370
  68. package/top-ppt-html/references/tech-design.md +138 -138
  69. package/top-ppt-html/scripts/audit_css.py +109 -109
  70. package/top-ppt-html/scripts/audit_docs.py +176 -176
  71. package/top-ppt-html/scripts/audit_skill.py +220 -220
  72. package/top-ppt-html/scripts/audit_styles.py +293 -351
  73. package/top-ppt-html/scripts/build_examples.py +2276 -2276
  74. package/top-ppt-html/scripts/build_pptx.js +2380 -2380
  75. package/top-ppt-html/scripts/capture_theme_overview.js +78 -78
  76. package/top-ppt-html/scripts/checks_html.py +127 -127
  77. package/top-ppt-html/scripts/cross_verify.py +294 -294
  78. package/top-ppt-html/scripts/env_probe.py +158 -158
  79. package/top-ppt-html/scripts/extract_model.py +210 -210
  80. package/top-ppt-html/scripts/extract_snippet.py +374 -374
  81. package/top-ppt-html/scripts/gen_channel_a.js +214 -214
  82. package/top-ppt-html/scripts/layout-constants.json +3309 -3377
  83. package/top-ppt-html/scripts/layout_slots.json +830 -830
  84. package/top-ppt-html/scripts/lib_layout_regions.js +412 -412
  85. package/top-ppt-html/scripts/measure_height.py +178 -178
  86. package/top-ppt-html/scripts/model-schema.json +547 -547
  87. package/top-ppt-html/scripts/negative_tests.py +307 -307
  88. package/top-ppt-html/scripts/package_skill.py +291 -291
  89. package/top-ppt-html/scripts/prepare_images.py +341 -341
  90. package/top-ppt-html/scripts/probe_image_export.py +188 -188
  91. package/top-ppt-html/scripts/quality_gate.py +301 -301
  92. package/top-ppt-html/scripts/regression.py +307 -308
  93. package/top-ppt-html/scripts/render_compare.py +275 -275
  94. package/top-ppt-html/scripts/render_from_model.py +698 -698
  95. package/top-ppt-html/scripts/scaffold_report.py +1054 -1054
  96. package/top-ppt-html/scripts/section-file-map.json +104 -104
  97. package/top-ppt-html/scripts/sync_runtime.py +662 -662
  98. package/top-ppt-html/scripts/validate_pptx.py +1510 -1510
  99. package/top-ppt-html/scripts/validate_report.py +1456 -1456
@@ -1,210 +1,210 @@
1
- #!/usr/bin/env python3
2
- # -*- coding: utf-8 -*-
3
- """
4
- TopPPT HTML· 从 HTML 报告抽取 PPTX 内容模型
5
- 用法:
6
- python extract_model.py <报告.html> [输出.json]
7
-
8
- 从报告的 window.REPORT_MODEL(JSON)抽取内容模型,供 build_pptx.js --model= 使用。
9
- 保证 PPTX 与页面同源——模型即唯一事实源,不再两处维护。
10
- 校验双端同源:页型字段与必填约束读 scripts/model-schema.json(与浏览器端 validateModel
11
- 消费同一份 schema,经 sync_runtime.py 注入 assets/pptx-export.js,杜绝漂移)。
12
- 含模型-正文一致性抽查(标题/agenda 条数/页数粗对齐),不一致打印警告(硬门禁在 validate_report.py)。
13
- """
14
- import sys
15
- import re
16
- import json
17
- from pathlib import Path
18
-
19
- # Windows GBK 控制台兜底:强制 UTF-8 输出
20
- try:
21
- sys.stdout.reconfigure(encoding='utf-8', errors='replace')
22
- sys.stderr.reconfigure(encoding='utf-8', errors='replace')
23
- except Exception:
24
- pass
25
-
26
- SCHEMA_PATH = Path(__file__).resolve().parent / 'model-schema.json'
27
-
28
-
29
- def load_schema():
30
- """DSL schema 单源(model-schema.json);与浏览器端 validateModel 同一份定义。"""
31
- try:
32
- return json.loads(SCHEMA_PATH.read_text(encoding='utf-8'))
33
- except (OSError, json.JSONDecodeError) as e:
34
- print(f"警告: 无法读取 schema 单源 {SCHEMA_PATH}: {e}")
35
- return None
36
-
37
-
38
- def schema_get(obj, path):
39
- cur = obj
40
- for k in path.split('.'):
41
- if cur is None or not isinstance(cur, dict):
42
- return None
43
- cur = cur.get(k)
44
- return cur
45
-
46
-
47
- def schema_field_ok(sec, spec):
48
- """spec = 'a.b:array' / 'a.b:str' / 'a.b'(真值检查)/ 'anyof:a|b:c'(任一满足)。"""
49
- if spec.startswith('anyof:'):
50
- return any(schema_field_ok(sec, alt) for alt in spec[len('anyof:'):].split('|') if alt)
51
- m = re.match(r'^(.*?)(?::(array|str))?$', spec)
52
- path, kind = m.group(1), m.group(2)
53
- v = schema_get(sec, path)
54
- if kind == 'array':
55
- return isinstance(v, list) and len(v) > 0
56
- if kind == 'str':
57
- return isinstance(v, str) and bool(v.strip())
58
- return bool(v)
59
-
60
-
61
- def validate_against_schema(model, schema):
62
- """按 schema 单源校验模型;返回 (缺失列表, 警告列表)。与浏览器端 validateModel 同语义。"""
63
- missing, warnings = [], []
64
- if not schema:
65
- return missing, warnings
66
- ms = schema.get('model') or {}
67
- for spec in ms.get('required', []):
68
- if not schema_field_ok(model, spec):
69
- missing.append(f"{spec}(顶层必填)")
70
- mode = model.get('mode') or 'presentation'
71
- ag = model.get('agenda') or []
72
- ag_min = (ms.get('agendaMin') or {}).get(mode, 0)
73
- if ag_min > 0 and len(ag) < ag_min:
74
- missing.append(f"agenda(大纲 ≥{ag_min} 条)")
75
- secs = model.get('sections') or []
76
- if not isinstance(secs, list) or len(secs) < ms.get('sectionsMin', 1):
77
- missing.append("sections(章节页 ≥1)")
78
- page_types = schema.get('pageTypes') or {}
79
- for i, sec in enumerate(secs if isinstance(secs, list) else []):
80
- t = (sec.get('type') if isinstance(sec, dict) else None) or 'points'
81
- defn = page_types.get(t)
82
- if not defn:
83
- missing.append(f"sections[{i}].type={t!r}(未知页型)")
84
- continue
85
- if not (isinstance(sec, dict) and sec.get('title')):
86
- missing.append(f"sections[{i}].title(第 {i + 1} 章标题)")
87
- for spec in defn.get('required', []):
88
- if not schema_field_ok(sec, spec):
89
- missing.append(f"sections[{i}].{spec}({defn.get('label', t)}必填)")
90
- if defn.get('modes') and mode not in defn['modes']:
91
- warnings.append(f"sections[{i}] 页型 {t!r} 适用于 {'/'.join(defn['modes'])},当前 mode={mode!r}")
92
- if schema.get('modes') and model.get('mode') and model['mode'] not in schema['modes']:
93
- warnings.append(f"mode={model['mode']!r} 未知,按 presentation 处理")
94
- if secs and len(secs) < ms.get('sectionsRecommended', 3):
95
- warnings.append(f"章节页仅 {len(secs)} 页,正式报告建议 ≥3")
96
- if len(ag) > ms.get('agendaComfortMax', 16):
97
- warnings.append(f"agenda {len(ag)} 条超出单页舒适上限,建议拆分")
98
- return missing, warnings
99
-
100
-
101
- def main():
102
- if len(sys.argv) < 2:
103
- print(__doc__)
104
- return 2
105
- path = Path(sys.argv[1])
106
- if not path.exists():
107
- print(f"文件不存在: {path}")
108
- return 2
109
- txt = path.read_text(encoding='utf-8')
110
-
111
- m = re.search(r'window\.REPORT_MODEL\s*=\s*(\{[\s\S]*?\})\s*;', txt)
112
- if not m:
113
- print("未找到 window.REPORT_MODEL —— 报告未内嵌内容模型。")
114
- print("请按 references/pptx-export.md 在报告 <script> 中补齐模型后再抽取。")
115
- return 1
116
- raw = m.group(1)
117
- try:
118
- model = json.loads(raw)
119
- except json.JSONDecodeError as e:
120
- print(f"REPORT_MODEL 不是合法 JSON: {e}")
121
- print("注意:模型必须是严格 JSON(双引号、无尾逗号、无注释)。")
122
- return 1
123
-
124
- # 模型字符串字段净化:禁止把 HTML 标签当纯文本写入(cite 只允许 [n])。
125
- # 泄漏标签会在 PPTX/预览里原样露出 <a class="cite"…>,属交付硬缺陷。
126
- stripped = {'count': 0, 'samples': []}
127
-
128
- def _scrub_assign(obj, path=''):
129
- if isinstance(obj, dict):
130
- for k in list(obj.keys()):
131
- v = obj[k]
132
- p = f'{path}.{k}' if path else k
133
- if isinstance(v, str) and re.search(r'</?[a-zA-Z][^>]*>', v):
134
- obj[k] = re.sub(r'\s+', ' ', re.sub(r'</?[a-zA-Z][^>]*>', '', v)).strip()
135
- stripped['count'] += 1
136
- if len(stripped['samples']) < 5:
137
- stripped['samples'].append(p)
138
- else:
139
- _scrub_assign(v, p)
140
- elif isinstance(obj, list):
141
- for i, v in enumerate(obj):
142
- p = f'{path}[{i}]'
143
- if isinstance(v, str) and re.search(r'</?[a-zA-Z][^>]*>', v):
144
- obj[i] = re.sub(r'\s+', ' ', re.sub(r'</?[a-zA-Z][^>]*>', '', v)).strip()
145
- stripped['count'] += 1
146
- if len(stripped['samples']) < 5:
147
- stripped['samples'].append(p)
148
- else:
149
- _scrub_assign(v, p)
150
-
151
- _scrub_assign(model)
152
- if stripped['count']:
153
- print(f"[净化] 剥离 {stripped['count']} 处模型字段中的 HTML 标签: {stripped['samples']}")
154
- print(" 注意:引用在模型里写 [n] 纯文本;HTML 正文才用 <a class=\"cite\">。禁止把标签写进模型。")
155
-
156
- # 风格/模式/主题兜底:模型未写时取页面 data-style / data-mode / data-theme
157
- if not model.get('style'):
158
- sm = re.search(r'<html[^>]*data-style="([^"]+)"', txt)
159
- if sm:
160
- model['style'] = sm.group(1)
161
- if not model.get('mode'):
162
- mm = re.search(r'<html[^>]*data-mode="([^"]+)"', txt)
163
- if mm:
164
- model['mode'] = mm.group(1)
165
- if not model.get('theme'):
166
- tm = re.search(r'<html[^>]*data-theme="([^"]+)"', txt)
167
- if tm:
168
- model['theme'] = tm.group(1)
169
-
170
- # schema 单源校验(与浏览器端 validateModel 同一份 model-schema.json)
171
- schema = load_schema()
172
- missing, schema_warn = validate_against_schema(model, schema)
173
- if missing:
174
- print(f"模型缺字段 {len(missing)} 项(导出的 PPTX 相应页面会为空或跳过):")
175
- for msg in missing:
176
- print(f" - {msg}")
177
- for msg in schema_warn:
178
- print(f"[schema 警告] {msg}")
179
-
180
- # 模型-正文一致性抽查(标题 / agenda 条数 / 页数粗对齐)
181
- issues = []
182
- plain = re.sub(r'<[^>]+>', ' ', txt)
183
- secs = model.get('sections') or []
184
- miss_titles = [str(s.get('title'))[:14] for s in secs
185
- if s.get('title') and str(s.get('title')) not in plain]
186
- if miss_titles:
187
- issues.append(f"{len(miss_titles)} 个章节标题未在正文出现: {miss_titles[:3]}")
188
- n_ag = len(model.get('agenda') or [])
189
- n_html_ag = txt.count('class="agenda__a"')
190
- if n_ag and n_html_ag and n_ag != n_html_ag:
191
- issues.append(f"agenda 条数不一致: model={n_ag} 正文={n_html_ag}")
192
- n_bands = len(re.findall(r'<section class="band', txt))
193
- if secs and not (len(secs) + 3 <= n_bands <= len(secs) + 7):
194
- issues.append(f"页数粗不匹配: 正文 {n_bands} 页 vs 模型 {len(secs)}+3~7")
195
- for msg in issues:
196
- print(f"[一致性警告] {msg}")
197
-
198
- out = Path(sys.argv[2]) if len(sys.argv) > 2 else path.with_suffix('.model.json')
199
- out.write_text(json.dumps(model, ensure_ascii=False, indent=2), encoding='utf-8')
200
- n_sec = len(model.get('sections', []))
201
- print(f"已抽取: {out} (agenda {len(model.get('agenda', []))} 条 · sections {n_sec} 页 · style={model.get('style', 'business-blue')} · theme={model.get('theme', 'light')})")
202
- if missing:
203
- print("注意:模型不完整,正式交付前应回 AI 对话补全(页面预览模态含可复制提示词)。")
204
- print("下一步:")
205
- print(f' NODE_PATH=<pptxgenjs 所在 node_modules> <node> scripts/build_pptx.js "报告.pptx" --model="{out}"')
206
- return 0
207
-
208
-
209
- if __name__ == '__main__':
210
- sys.exit(main())
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """
4
+ TopPPT HTML· 从 HTML 报告抽取 PPTX 内容模型
5
+ 用法:
6
+ python extract_model.py <报告.html> [输出.json]
7
+
8
+ 从报告的 window.REPORT_MODEL(JSON)抽取内容模型,供 build_pptx.js --model= 使用。
9
+ 保证 PPTX 与页面同源——模型即唯一事实源,不再两处维护。
10
+ 校验双端同源:页型字段与必填约束读 scripts/model-schema.json(与浏览器端 validateModel
11
+ 消费同一份 schema,经 sync_runtime.py 注入 assets/pptx-export.js,杜绝漂移)。
12
+ 含模型-正文一致性抽查(标题/agenda 条数/页数粗对齐),不一致打印警告(硬门禁在 validate_report.py)。
13
+ """
14
+ import sys
15
+ import re
16
+ import json
17
+ from pathlib import Path
18
+
19
+ # Windows GBK 控制台兜底:强制 UTF-8 输出
20
+ try:
21
+ sys.stdout.reconfigure(encoding='utf-8', errors='replace')
22
+ sys.stderr.reconfigure(encoding='utf-8', errors='replace')
23
+ except Exception:
24
+ pass
25
+
26
+ SCHEMA_PATH = Path(__file__).resolve().parent / 'model-schema.json'
27
+
28
+
29
+ def load_schema():
30
+ """DSL schema 单源(model-schema.json);与浏览器端 validateModel 同一份定义。"""
31
+ try:
32
+ return json.loads(SCHEMA_PATH.read_text(encoding='utf-8'))
33
+ except (OSError, json.JSONDecodeError) as e:
34
+ print(f"警告: 无法读取 schema 单源 {SCHEMA_PATH}: {e}")
35
+ return None
36
+
37
+
38
+ def schema_get(obj, path):
39
+ cur = obj
40
+ for k in path.split('.'):
41
+ if cur is None or not isinstance(cur, dict):
42
+ return None
43
+ cur = cur.get(k)
44
+ return cur
45
+
46
+
47
+ def schema_field_ok(sec, spec):
48
+ """spec = 'a.b:array' / 'a.b:str' / 'a.b'(真值检查)/ 'anyof:a|b:c'(任一满足)。"""
49
+ if spec.startswith('anyof:'):
50
+ return any(schema_field_ok(sec, alt) for alt in spec[len('anyof:'):].split('|') if alt)
51
+ m = re.match(r'^(.*?)(?::(array|str))?$', spec)
52
+ path, kind = m.group(1), m.group(2)
53
+ v = schema_get(sec, path)
54
+ if kind == 'array':
55
+ return isinstance(v, list) and len(v) > 0
56
+ if kind == 'str':
57
+ return isinstance(v, str) and bool(v.strip())
58
+ return bool(v)
59
+
60
+
61
+ def validate_against_schema(model, schema):
62
+ """按 schema 单源校验模型;返回 (缺失列表, 警告列表)。与浏览器端 validateModel 同语义。"""
63
+ missing, warnings = [], []
64
+ if not schema:
65
+ return missing, warnings
66
+ ms = schema.get('model') or {}
67
+ for spec in ms.get('required', []):
68
+ if not schema_field_ok(model, spec):
69
+ missing.append(f"{spec}(顶层必填)")
70
+ mode = model.get('mode') or 'presentation'
71
+ ag = model.get('agenda') or []
72
+ ag_min = (ms.get('agendaMin') or {}).get(mode, 0)
73
+ if ag_min > 0 and len(ag) < ag_min:
74
+ missing.append(f"agenda(大纲 ≥{ag_min} 条)")
75
+ secs = model.get('sections') or []
76
+ if not isinstance(secs, list) or len(secs) < ms.get('sectionsMin', 1):
77
+ missing.append("sections(章节页 ≥1)")
78
+ page_types = schema.get('pageTypes') or {}
79
+ for i, sec in enumerate(secs if isinstance(secs, list) else []):
80
+ t = (sec.get('type') if isinstance(sec, dict) else None) or 'points'
81
+ defn = page_types.get(t)
82
+ if not defn:
83
+ missing.append(f"sections[{i}].type={t!r}(未知页型)")
84
+ continue
85
+ if not (isinstance(sec, dict) and sec.get('title')):
86
+ missing.append(f"sections[{i}].title(第 {i + 1} 章标题)")
87
+ for spec in defn.get('required', []):
88
+ if not schema_field_ok(sec, spec):
89
+ missing.append(f"sections[{i}].{spec}({defn.get('label', t)}必填)")
90
+ if defn.get('modes') and mode not in defn['modes']:
91
+ warnings.append(f"sections[{i}] 页型 {t!r} 适用于 {'/'.join(defn['modes'])},当前 mode={mode!r}")
92
+ if schema.get('modes') and model.get('mode') and model['mode'] not in schema['modes']:
93
+ warnings.append(f"mode={model['mode']!r} 未知,按 presentation 处理")
94
+ if secs and len(secs) < ms.get('sectionsRecommended', 3):
95
+ warnings.append(f"章节页仅 {len(secs)} 页,正式报告建议 ≥3")
96
+ if len(ag) > ms.get('agendaComfortMax', 16):
97
+ warnings.append(f"agenda {len(ag)} 条超出单页舒适上限,建议拆分")
98
+ return missing, warnings
99
+
100
+
101
+ def main():
102
+ if len(sys.argv) < 2:
103
+ print(__doc__)
104
+ return 2
105
+ path = Path(sys.argv[1])
106
+ if not path.exists():
107
+ print(f"文件不存在: {path}")
108
+ return 2
109
+ txt = path.read_text(encoding='utf-8')
110
+
111
+ m = re.search(r'window\.REPORT_MODEL\s*=\s*(\{[\s\S]*?\})\s*;', txt)
112
+ if not m:
113
+ print("未找到 window.REPORT_MODEL —— 报告未内嵌内容模型。")
114
+ print("请按 references/pptx-export.md 在报告 <script> 中补齐模型后再抽取。")
115
+ return 1
116
+ raw = m.group(1)
117
+ try:
118
+ model = json.loads(raw)
119
+ except json.JSONDecodeError as e:
120
+ print(f"REPORT_MODEL 不是合法 JSON: {e}")
121
+ print("注意:模型必须是严格 JSON(双引号、无尾逗号、无注释)。")
122
+ return 1
123
+
124
+ # 模型字符串字段净化:禁止把 HTML 标签当纯文本写入(cite 只允许 [n])。
125
+ # 泄漏标签会在 PPTX/预览里原样露出 <a class="cite"…>,属交付硬缺陷。
126
+ stripped = {'count': 0, 'samples': []}
127
+
128
+ def _scrub_assign(obj, path=''):
129
+ if isinstance(obj, dict):
130
+ for k in list(obj.keys()):
131
+ v = obj[k]
132
+ p = f'{path}.{k}' if path else k
133
+ if isinstance(v, str) and re.search(r'</?[a-zA-Z][^>]*>', v):
134
+ obj[k] = re.sub(r'\s+', ' ', re.sub(r'</?[a-zA-Z][^>]*>', '', v)).strip()
135
+ stripped['count'] += 1
136
+ if len(stripped['samples']) < 5:
137
+ stripped['samples'].append(p)
138
+ else:
139
+ _scrub_assign(v, p)
140
+ elif isinstance(obj, list):
141
+ for i, v in enumerate(obj):
142
+ p = f'{path}[{i}]'
143
+ if isinstance(v, str) and re.search(r'</?[a-zA-Z][^>]*>', v):
144
+ obj[i] = re.sub(r'\s+', ' ', re.sub(r'</?[a-zA-Z][^>]*>', '', v)).strip()
145
+ stripped['count'] += 1
146
+ if len(stripped['samples']) < 5:
147
+ stripped['samples'].append(p)
148
+ else:
149
+ _scrub_assign(v, p)
150
+
151
+ _scrub_assign(model)
152
+ if stripped['count']:
153
+ print(f"[净化] 剥离 {stripped['count']} 处模型字段中的 HTML 标签: {stripped['samples']}")
154
+ print(" 注意:引用在模型里写 [n] 纯文本;HTML 正文才用 <a class=\"cite\">。禁止把标签写进模型。")
155
+
156
+ # 风格/模式/主题兜底:模型未写时取页面 data-style / data-mode / data-theme
157
+ if not model.get('style'):
158
+ sm = re.search(r'<html[^>]*data-style="([^"]+)"', txt)
159
+ if sm:
160
+ model['style'] = sm.group(1)
161
+ if not model.get('mode'):
162
+ mm = re.search(r'<html[^>]*data-mode="([^"]+)"', txt)
163
+ if mm:
164
+ model['mode'] = mm.group(1)
165
+ if not model.get('theme'):
166
+ tm = re.search(r'<html[^>]*data-theme="([^"]+)"', txt)
167
+ if tm:
168
+ model['theme'] = tm.group(1)
169
+
170
+ # schema 单源校验(与浏览器端 validateModel 同一份 model-schema.json)
171
+ schema = load_schema()
172
+ missing, schema_warn = validate_against_schema(model, schema)
173
+ if missing:
174
+ print(f"模型缺字段 {len(missing)} 项(导出的 PPTX 相应页面会为空或跳过):")
175
+ for msg in missing:
176
+ print(f" - {msg}")
177
+ for msg in schema_warn:
178
+ print(f"[schema 警告] {msg}")
179
+
180
+ # 模型-正文一致性抽查(标题 / agenda 条数 / 页数粗对齐)
181
+ issues = []
182
+ plain = re.sub(r'<[^>]+>', ' ', txt)
183
+ secs = model.get('sections') or []
184
+ miss_titles = [str(s.get('title'))[:14] for s in secs
185
+ if s.get('title') and str(s.get('title')) not in plain]
186
+ if miss_titles:
187
+ issues.append(f"{len(miss_titles)} 个章节标题未在正文出现: {miss_titles[:3]}")
188
+ n_ag = len(model.get('agenda') or [])
189
+ n_html_ag = txt.count('class="agenda__a"')
190
+ if n_ag and n_html_ag and n_ag != n_html_ag:
191
+ issues.append(f"agenda 条数不一致: model={n_ag} 正文={n_html_ag}")
192
+ n_bands = len(re.findall(r'<section class="band', txt))
193
+ if secs and not (len(secs) + 3 <= n_bands <= len(secs) + 7):
194
+ issues.append(f"页数粗不匹配: 正文 {n_bands} 页 vs 模型 {len(secs)}+3~7")
195
+ for msg in issues:
196
+ print(f"[一致性警告] {msg}")
197
+
198
+ out = Path(sys.argv[2]) if len(sys.argv) > 2 else path.with_suffix('.model.json')
199
+ out.write_text(json.dumps(model, ensure_ascii=False, indent=2), encoding='utf-8')
200
+ n_sec = len(model.get('sections', []))
201
+ print(f"已抽取: {out} (agenda {len(model.get('agenda', []))} 条 · sections {n_sec} 页 · style={model.get('style', 'business-blue')} · theme={model.get('theme', 'light')})")
202
+ if missing:
203
+ print("注意:模型不完整,正式交付前应回 AI 对话补全(页面预览模态含可复制提示词)。")
204
+ print("下一步:")
205
+ print(f' NODE_PATH=<pptxgenjs 所在 node_modules> <node> scripts/build_pptx.js "报告.pptx" --model="{out}"')
206
+ return 0
207
+
208
+
209
+ if __name__ == '__main__':
210
+ sys.exit(main())