@topmindspace/tms-skills 0.1.1 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +82 -0
- package/README.md +13 -3
- package/bin/tms-skills.js +7 -2
- package/package.json +1 -1
- package/top-ppt-html/README.md +69 -148
- package/top-ppt-html/SKILL.md +42 -23
- package/top-ppt-html/agents/openai.yaml +5 -0
- package/top-ppt-html/assets/theme-overview-architecture.png +0 -0
- package/top-ppt-html/assets/theme-overview-research.png +0 -0
- package/top-ppt-html/assets/theme-overview.png +0 -0
- package/top-ppt-html/evals/trigger-queries.json +181 -0
- package/top-ppt-html/package.json +3 -2
- package/top-ppt-html/references/chart-decision-tree.md +42 -0
- package/top-ppt-html/references/charts-discipline.md +3 -1
- package/top-ppt-html/references/charts.md +24 -15
- package/top-ppt-html/references/content-rules.md +54 -34
- package/top-ppt-html/references/default-surface.md +62 -0
- package/top-ppt-html/references/design-system.md +2 -2
- package/top-ppt-html/references/failure-modes.md +14 -1
- package/top-ppt-html/references/icons.md +56 -277
- package/top-ppt-html/references/illustration-layout.md +39 -0
- package/top-ppt-html/references/layout-grammar.md +16 -6
- package/top-ppt-html/references/modes.md +16 -11
- package/top-ppt-html/references/outline-design.md +24 -0
- package/top-ppt-html/references/page-type-matrix.md +35 -0
- package/top-ppt-html/references/playbook.md +42 -86
- package/top-ppt-html/references/pptx-export.md +20 -2
- package/top-ppt-html/references/presentation-craft.md +77 -0
- package/top-ppt-html/references/tech-design.md +11 -5
- package/top-ppt-html/scripts/audit_docs.py +1 -1
- package/top-ppt-html/scripts/audit_skill.py +45 -6
- package/top-ppt-html/scripts/build_examples.py +39 -2260
- package/top-ppt-html/scripts/build_pptx.js +6 -0
- package/top-ppt-html/scripts/check_triggers.py +153 -0
- package/top-ppt-html/scripts/checks_html.py +39 -0
- package/top-ppt-html/scripts/cross_verify.py +5 -1
- package/top-ppt-html/scripts/extract_snippet.py +2 -2
- package/top-ppt-html/scripts/layout-constants.json +111 -18
- package/top-ppt-html/scripts/lib_layout_regions.js +7 -9
- package/top-ppt-html/scripts/negative_tests.py +137 -1
- package/top-ppt-html/scripts/package_skill.py +29 -10
- package/top-ppt-html/scripts/quality_gate.py +11 -3
- package/top-ppt-html/scripts/recommend_layout.py +384 -0
- package/top-ppt-html/scripts/regression.py +1 -1
- package/top-ppt-html/scripts/smoke_pptx.sh +30 -0
- package/top-ppt-html/scripts/sync_runtime.py +22 -25
- package/top-ppt-html/scripts/validate_pptx.py +69 -0
- package/top-ppt-html/scripts/validate_report.py +336 -10
- package/top-ppt-html/assets/examples/2026-09-09-architecture-spectrum.html +0 -3926
- package/top-ppt-html/assets/examples/2026-09-09-architecture-spectrum.model.json +0 -168
- package/top-ppt-html/assets/examples/2026-09-09-presentation-apple-mono.html +0 -4325
- package/top-ppt-html/assets/examples/2026-09-09-presentation-apple-mono.model.json +0 -321
- package/top-ppt-html/assets/examples/2026-09-09-presentation-brand-red.html +0 -4325
- package/top-ppt-html/assets/examples/2026-09-09-presentation-brand-red.model.json +0 -321
- package/top-ppt-html/assets/examples/2026-09-09-research-deep-teal.html +0 -5527
- package/top-ppt-html/assets/examples/2026-09-09-research-deep-teal.model.json +0 -914
- package/top-ppt-html/assets/examples/2026-09-09-research-indigo-violet.html +0 -5527
- package/top-ppt-html/assets/examples/2026-09-09-research-indigo-violet.model.json +0 -914
- package/top-ppt-html/assets/examples/2026-09-09-research-warm-sand.html +0 -5527
- package/top-ppt-html/assets/examples/2026-09-09-research-warm-sand.model.json +0 -914
- package/top-ppt-html/references/design-system-engine.md +0 -235
- package/top-ppt-html/references/industry-benchmark.md +0 -105
- package/top-ppt-html/references/reform-plan.md +0 -252
- package/top-ppt-html/scripts/layout_slots.json +0 -830
|
@@ -3,10 +3,12 @@
|
|
|
3
3
|
"""
|
|
4
4
|
TopPPT HTML· HTML 报告质量校验(交付闭环)
|
|
5
5
|
用法:
|
|
6
|
-
python validate_report.py <报告.html> [--strict] [--json]
|
|
6
|
+
python validate_report.py <报告.html> [--strict] [--layout-qa] [--json]
|
|
7
7
|
|
|
8
8
|
输出每项 PASS/FAIL,结尾给汇总与结论;任一 FAIL 时退出码为 1(--strict 时 WARN 也计失败)。
|
|
9
9
|
--json 以 JSON 输出全部检查结果(供脚本/流水线读取)。
|
|
10
|
+
Mode A / presentation + --strict:自动启用 --layout-qa(V 契约/截断/溢出未拆页/半空卡/对齐);
|
|
11
|
+
research / architecture 仍须显式传 --layout-qa(不强制)。
|
|
10
12
|
生成流程:生成 → 跑本脚本 → 修复 FAIL → 再跑,直至全部 PASS 才交付。
|
|
11
13
|
|
|
12
14
|
阈值全部来自单源 scripts/layout-constants.json(checkBudgets / charts / styleAccents / aiFlavor),
|
|
@@ -218,7 +220,12 @@ CARRIERS = checks_html.CARRIERS # 单源 scripts/checks_html.py
|
|
|
218
220
|
|
|
219
221
|
|
|
220
222
|
def _check_chart_variety(txt, chk, mode):
|
|
221
|
-
"""图表与版式多样性(阈值单源 charts.variety;判定逻辑 checks_html)。
|
|
223
|
+
"""图表与版式多样性(阈值单源 charts.variety;判定逻辑 checks_html)。
|
|
224
|
+
|
|
225
|
+
多样性仍是特性:全部 data-chart(含 advanced)计入下限——用 advanced 不罚。
|
|
226
|
+
preferCoreFirst:核图多样性不足却靠 advanced 撑场时 WARN(勿凑下限)。
|
|
227
|
+
禁反模式由 layout-qa / sizeByComplexity / CHART_SKEW 另检。
|
|
228
|
+
"""
|
|
222
229
|
v = CHART_VARIETY
|
|
223
230
|
if not v:
|
|
224
231
|
return
|
|
@@ -230,8 +237,14 @@ def _check_chart_variety(txt, chk, mode):
|
|
|
230
237
|
n_chart_pages = len([1 for ts in per_page if ts])
|
|
231
238
|
|
|
232
239
|
floor = checks_html.variety_floor(mode, n_chart_pages, v)
|
|
240
|
+
# advanced 与核图一并计入 distinct(不因用 advanced 受罚)
|
|
233
241
|
chk("图表多样性(全篇不同 data-chart 类型数)", len(distinct) >= floor,
|
|
234
242
|
f"{len(distinct)} 种 / 下限 {floor}({n_chart_pages} 个图表页): {distinct}")
|
|
243
|
+
# preferCoreFirst WARN 仅 A/B:C 架构以结构/甘特等 advanced 为主属正常,不噪音
|
|
244
|
+
if mode in ('presentation', 'research'):
|
|
245
|
+
pref = checks_html.variety_core_preference(used, v)
|
|
246
|
+
if pref:
|
|
247
|
+
chk("图表多样性·核图优先(preferCoreFirst)", False, pref, level="WARN")
|
|
235
248
|
|
|
236
249
|
if v.get('noRepeatAdjacent'):
|
|
237
250
|
adj = checks_html.adjacent_same_type(per_page)
|
|
@@ -323,7 +336,7 @@ def _check_bands(txt, chk, mode="presentation"):
|
|
|
323
336
|
if est_h > SCREEN_BUDGET_PX * 1.08:
|
|
324
337
|
est_over.append(f"第{i}页≈{est_h:.0f}px")
|
|
325
338
|
chk(f"页高溢出估算(字×行高+组件 ≤ ~{SCREEN_BUDGET_PX}px/屏 · {mode})", not est_over,
|
|
326
|
-
("; ".join(est_over) + "
|
|
339
|
+
("; ".join(est_over) + "(按「重构承载→拆页/分章→换形态→有限 fontShrink」;禁静默截断;或长结构页加 band--flow)")
|
|
327
340
|
if est_over else "")
|
|
328
341
|
|
|
329
342
|
bad_grids = 0
|
|
@@ -888,6 +901,45 @@ def _check_v9_hard_gates(txt, chk, model):
|
|
|
888
901
|
not oversize, "; ".join(oversize[:4]) if oversize else "")
|
|
889
902
|
|
|
890
903
|
|
|
904
|
+
|
|
905
|
+
def _class_tokens(tag_or_html: str) -> set[str]:
|
|
906
|
+
"""Extract HTML class tokens from class="..." attributes."""
|
|
907
|
+
out: set[str] = set()
|
|
908
|
+
for m in re.finditer(r'class="([^"]*)"', tag_or_html):
|
|
909
|
+
out.update(m.group(1).split())
|
|
910
|
+
return out
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
def _mixed_grid_needs_align(band: str) -> bool:
|
|
914
|
+
"""True when a grid/g-* region pairs media with cards/lists (token-exact; no t-metric / data-chart false hits)."""
|
|
915
|
+
def has_media(classes: set[str], chunk: str) -> bool:
|
|
916
|
+
if classes & {'fig', 'media', 'chart'}:
|
|
917
|
+
return True
|
|
918
|
+
if any(c.startswith('media') or c.startswith('fig') for c in classes):
|
|
919
|
+
return True
|
|
920
|
+
return 'data-chart=' in chunk
|
|
921
|
+
|
|
922
|
+
def has_cards(classes: set[str]) -> bool:
|
|
923
|
+
if classes & {'card', 'ul', 'metric'}:
|
|
924
|
+
return True
|
|
925
|
+
return any(
|
|
926
|
+
c.startswith('ul--') or c.startswith('card') or c.startswith('metric__')
|
|
927
|
+
for c in classes)
|
|
928
|
+
|
|
929
|
+
# Find each grid opening and inspect following chunk
|
|
930
|
+
for m in re.finditer(r'<div class="([^"]*)"', band):
|
|
931
|
+
classes_on_grid = set(m.group(1).split())
|
|
932
|
+
if not (classes_on_grid & {'grid'} or any(re.fullmatch(r'g-\d+', c) or c.startswith('g-') for c in classes_on_grid)):
|
|
933
|
+
# allow g-hero / g-side / g-2 etc
|
|
934
|
+
if not any(c == 'grid' or c.startswith('g-') for c in classes_on_grid):
|
|
935
|
+
continue
|
|
936
|
+
chunk = band[m.start(): m.start() + 4500]
|
|
937
|
+
classes = _class_tokens(chunk)
|
|
938
|
+
if has_media(classes, chunk) and has_cards(classes):
|
|
939
|
+
return True
|
|
940
|
+
return False
|
|
941
|
+
|
|
942
|
+
|
|
891
943
|
def _check_layout_grammar(txt, chk, mode):
|
|
892
944
|
"""布局语法门禁:骨架类 / 单一重心 / 混排对齐 / 间距 token / 图标尺寸 / 标签防换行。"""
|
|
893
945
|
LS = LC.get('layoutSystem') or {}
|
|
@@ -920,7 +972,10 @@ def _check_layout_grammar(txt, chk, mode):
|
|
|
920
972
|
|
|
921
973
|
for i, b in enumerate(_bands(txt), 1):
|
|
922
974
|
head = b[:240]
|
|
923
|
-
|
|
975
|
+
# intentional whitespace:封面/章节幕/金句/收尾等休止页不参与 FILL 门禁
|
|
976
|
+
if any(f'id="{x}' in head for x in (
|
|
977
|
+
'refs', 'appendix', 'cover', 'agenda', 'quote',
|
|
978
|
+
'closing', 'section', 'chapter', 'next')):
|
|
924
979
|
continue
|
|
925
980
|
if 'band--flow' in head:
|
|
926
981
|
continue
|
|
@@ -956,11 +1011,10 @@ def _check_layout_grammar(txt, chk, mode):
|
|
|
956
1011
|
if n_big >= 2:
|
|
957
1012
|
multi_focus.append(f"第{i}页大件×{n_big}")
|
|
958
1013
|
|
|
959
|
-
# ③
|
|
960
|
-
mixed = (
|
|
961
|
-
re.search(r'class="[^"]*(?:card|ul|metric)', b))
|
|
1014
|
+
# ③ 混排对齐(grid 内同时有图/媒体与卡/列表;token 精确)
|
|
1015
|
+
mixed = _mixed_grid_needs_align(b)
|
|
962
1016
|
if mixed and mixed_need:
|
|
963
|
-
if not any(
|
|
1017
|
+
if not any(re.search(r'\b' + re.escape(tok) + r'\b', b) for tok in align_tokens):
|
|
964
1018
|
align_miss.append(f"第{i}页图卡混排缺 a-start/a-c")
|
|
965
1019
|
|
|
966
1020
|
# ④ 间距写死(margin/padding/gap 非 token / 非 clamp;≤8px 微调白名单)
|
|
@@ -1161,6 +1215,33 @@ def _check_media(txt, chk, model):
|
|
|
1161
1215
|
chk("模型配图占位 ↔ 正文 .media--ph 对应", False,
|
|
1162
1216
|
f"{n_ph} 页模型声明 image.placeholder 但正文无 .media--ph", level="WARN")
|
|
1163
1217
|
|
|
1218
|
+
# 配图页 caption / so-what(廉价 WARN);近邻主张/导语亦可
|
|
1219
|
+
bands = re.split(r'(?=<section\b)', txt)
|
|
1220
|
+
miss_cap = []
|
|
1221
|
+
near_empty = []
|
|
1222
|
+
for i, b in enumerate(bands):
|
|
1223
|
+
if not re.search(r'class="[^"]*\bmedia\b|<img\b|class="[^"]*media--', b):
|
|
1224
|
+
continue
|
|
1225
|
+
if len(b) < 80:
|
|
1226
|
+
continue
|
|
1227
|
+
has_cap = bool(re.search(
|
|
1228
|
+
r'class="[^"]*(?:fig__cap|media__cap|media__ph|caption|exhibit__src|footnote|so-what|lead)', b)
|
|
1229
|
+
or re.search(r'<figcaption\b|class="[^"]*\bsoWhat\b', b))
|
|
1230
|
+
if not has_cap:
|
|
1231
|
+
miss_cap.append(f'band[{i}]')
|
|
1232
|
+
# 近图过空:有 media 但正文文本极少且无要点/指标(WARN)
|
|
1233
|
+
textish = re.sub(r'<script[\s\S]*?</script>|<style[\s\S]*?</style>|<[^>]+>', ' ', b)
|
|
1234
|
+
textish = re.sub(r'\s+', ' ', textish).strip()
|
|
1235
|
+
has_points = bool(re.search(r'class="[^"]*(?:points|bullets|kpi|metrics|card)', b))
|
|
1236
|
+
if len(textish) < 40 and not has_points and not re.search(r'media--ph', b):
|
|
1237
|
+
near_empty.append(f'band[{i}]')
|
|
1238
|
+
chk("配图页含 caption/图注(IMAGE_CAPTION)", not miss_cap,
|
|
1239
|
+
f"{len(miss_cap)} 处配图区缺 .fig__cap/.media__cap/.footnote/so-what 等图注" if miss_cap else "",
|
|
1240
|
+
level="WARN")
|
|
1241
|
+
chk("配图页近邻过空(IMAGE_NEAR_EMPTY)", not near_empty,
|
|
1242
|
+
f"{len(near_empty)} 处配图页几乎无注解/要点(补 caption 或要点条)" if near_empty else "",
|
|
1243
|
+
level="WARN")
|
|
1244
|
+
|
|
1164
1245
|
|
|
1165
1246
|
# 失败检查项 → 失败模式 / 处置动作 / 精确取码命令。
|
|
1166
1247
|
# 目的:校验失败时直接给出「改什么、按什么顺序改、去哪取代码」,
|
|
@@ -1233,11 +1314,227 @@ def _print_fix_guide(results, strict, width):
|
|
|
1233
1314
|
print(f" 取码: python scripts/extract_snippet.py {cmd}")
|
|
1234
1315
|
else:
|
|
1235
1316
|
print(" 未匹配到已登记的失败模式,按通用顺序处置:")
|
|
1236
|
-
print(" ①
|
|
1237
|
-
"④
|
|
1317
|
+
print(" ① 重构承载(列表/卡/表/图)② 拆页/分章 ③ 换布局形态 "
|
|
1318
|
+
"④ 有限缩字号 ⑤ 禁止静默截断/砍 so-what")
|
|
1238
1319
|
print(" 完整失败模式库与错误解释纠正表: references/failure-modes.md")
|
|
1239
1320
|
|
|
1240
1321
|
|
|
1322
|
+
|
|
1323
|
+
def _check_layout_qa(txt, chk, mode, model):
|
|
1324
|
+
"""布局 QA(--layout-qa;presentation+--strict 自动开):
|
|
1325
|
+
V 契约 / 极偏环图 / 骨架连用 / 缺 data-skel / 简单全幅 /
|
|
1326
|
+
截断迹象 / 溢出未拆页 / 半空卡 / 列对齐节奏。
|
|
1327
|
+
"""
|
|
1328
|
+
bands = _bands(txt)
|
|
1329
|
+
skip_ids = {'cover', 'agenda', 'refs', 'appendix', 'next', 'quote'}
|
|
1330
|
+
content = []
|
|
1331
|
+
for i, b in enumerate(bands, 1):
|
|
1332
|
+
head = b[:240]
|
|
1333
|
+
if any(f'id="{sid}"' in head for sid in skip_ids):
|
|
1334
|
+
continue
|
|
1335
|
+
if 'band--deep' in head or 'band--accent' in head:
|
|
1336
|
+
# 金句/强调带允许无骨架
|
|
1337
|
+
if 'data-skel=' not in b and not re.search(r'class="[^"]*\bP\d+\b', b):
|
|
1338
|
+
continue
|
|
1339
|
+
content.append((i, b, head))
|
|
1340
|
+
|
|
1341
|
+
# ① 缺 data-skel:若全文已出现 data-skel(scaffold/render 管线),则内容页必须都有
|
|
1342
|
+
has_any_skel = 'data-skel="' in txt
|
|
1343
|
+
missing = []
|
|
1344
|
+
if has_any_skel:
|
|
1345
|
+
for i, b, head in content:
|
|
1346
|
+
if 'data-skel="' not in b and not re.search(r'class="[^"]*\bP\d+\b', b):
|
|
1347
|
+
missing.append(f"第{i}页")
|
|
1348
|
+
chk("LAYOUT_QA_MISSING_SKEL 内容页须带 data-skel(管线产物)",
|
|
1349
|
+
not missing,
|
|
1350
|
+
("缺骨架: " + "; ".join(missing[:5])) if missing else
|
|
1351
|
+
("(全文无 data-skel,跳过——手写示例豁免;scaffold/render 会写入)" if not has_any_skel else ""))
|
|
1352
|
+
|
|
1353
|
+
# ② 连续 ≥3 页同一 data-skel(或同一 layout 签名)
|
|
1354
|
+
skels = []
|
|
1355
|
+
for i, b, head in content:
|
|
1356
|
+
m = re.search(r'data-skel="(P\d+)"', b)
|
|
1357
|
+
if m:
|
|
1358
|
+
skels.append((i, m.group(1)))
|
|
1359
|
+
else:
|
|
1360
|
+
sig = _layout_sig(b)
|
|
1361
|
+
if sig:
|
|
1362
|
+
skels.append((i, f"sig:{sig}"))
|
|
1363
|
+
streak_hits = []
|
|
1364
|
+
run_v, run_s, run_n = None, 0, 0
|
|
1365
|
+
for i, v in skels:
|
|
1366
|
+
if v == run_v:
|
|
1367
|
+
run_n += 1
|
|
1368
|
+
else:
|
|
1369
|
+
if run_v and run_n >= 3:
|
|
1370
|
+
streak_hits.append(f"{run_v}×{run_n}(起第{run_s}页)")
|
|
1371
|
+
run_v, run_s, run_n = v, i, 1
|
|
1372
|
+
if run_v and run_n >= 3:
|
|
1373
|
+
streak_hits.append(f"{run_v}×{run_n}(起第{run_s}页)")
|
|
1374
|
+
chk("LAYOUT_QA_SKEL_STREAK 同一 data-skel/版式签名连续 <3 页",
|
|
1375
|
+
not streak_hits, "; ".join(streak_hits[:3]) if streak_hits else "")
|
|
1376
|
+
|
|
1377
|
+
# ③ 极偏仍用 donut/pie(模型侧;与 CHART_SKEW 互补,专打 layout-qa 关键字)
|
|
1378
|
+
skew_hits = []
|
|
1379
|
+
sections = (model or {}).get('sections') or []
|
|
1380
|
+
for si, sec in enumerate(sections, 1):
|
|
1381
|
+
if not isinstance(sec, dict):
|
|
1382
|
+
continue
|
|
1383
|
+
ch = sec.get('chart') if isinstance(sec.get('chart'), dict) else {}
|
|
1384
|
+
ctype = str(ch.get('type') or (sec.get('type') if sec.get('type') in
|
|
1385
|
+
('donut', 'pie', 'multidonut') else '') or '').lower()
|
|
1386
|
+
if sec.get('type') == 'donut':
|
|
1387
|
+
ctype = 'donut'
|
|
1388
|
+
if ctype not in ('donut', 'pie', 'multidonut'):
|
|
1389
|
+
continue
|
|
1390
|
+
vals = [float(v) for v in (ch.get('values') or []) if isinstance(v, (int, float))]
|
|
1391
|
+
vals = [v for v in vals if v >= 0]
|
|
1392
|
+
if len(vals) < 2:
|
|
1393
|
+
continue
|
|
1394
|
+
total = sum(vals) or 1.0
|
|
1395
|
+
pcts = [v / total * 100 for v in vals]
|
|
1396
|
+
mn, mx = min(pcts), max(pcts)
|
|
1397
|
+
ratio = (mx / mn) if mn > 0 else 999
|
|
1398
|
+
if mn < 5.0 or ratio > 20:
|
|
1399
|
+
skew_hits.append(f"sections[{si}] {ctype} 最小{mn:.1f}% 比{ratio:.0f}:1 → 改 V3/KPI")
|
|
1400
|
+
chk("LAYOUT_QA_SKEW_DONUT 极偏占比禁用 donut/pie(改 V3 KPI)",
|
|
1401
|
+
not skew_hits, "; ".join(skew_hits[:3]) if skew_hits else "")
|
|
1402
|
+
|
|
1403
|
+
# ④ 演示模式:简单图(≤2 类)却近全幅(svg height≥320 或 width:100% 无注解带)
|
|
1404
|
+
bleed_hits = []
|
|
1405
|
+
if mode == 'presentation':
|
|
1406
|
+
for i, b, head in content:
|
|
1407
|
+
for m in re.finditer(
|
|
1408
|
+
r'<svg\b[^>]*data-chart="([^"]+)"[^>]*style="([^"]*)"[^>]*>', b):
|
|
1409
|
+
ctype, style = m.group(1), m.group(2)
|
|
1410
|
+
svg_end = b.find('</svg>', m.end())
|
|
1411
|
+
block = b[m.start():svg_end if svg_end > 0 else m.end() + 600]
|
|
1412
|
+
n_lab = len(re.findall(r'<text\b', block))
|
|
1413
|
+
n_pts = max(n_lab, len(re.findall(r'<rect\b', block)),
|
|
1414
|
+
len(re.findall(r'<circle\b', block)))
|
|
1415
|
+
hm = re.search(r'height:\s*(\d+(?:\.\d+)?)px', style)
|
|
1416
|
+
h_px = float(hm.group(1)) if hm else 0
|
|
1417
|
+
simple = n_lab <= 2 or n_pts <= 3
|
|
1418
|
+
full = h_px >= 320 or ('width:100%' in style and h_px >= 240)
|
|
1419
|
+
has_anno = bool(re.search(r'class="[^"]*(?:sowhat|anno|points|callout)', b))
|
|
1420
|
+
if simple and full and not has_anno:
|
|
1421
|
+
bleed_hits.append(f"第{i}页 {ctype} 简单全幅无注解")
|
|
1422
|
+
chk("LAYOUT_QA_SIMPLE_FULLBLEED 演示页简单图禁全幅无注解(用 V1–V4)",
|
|
1423
|
+
not bleed_hits, "; ".join(bleed_hits[:3]) if bleed_hits else "")
|
|
1424
|
+
|
|
1425
|
+
# ⑤ V 契约:演示页 data-v / data-skel 与内容复杂度粗检
|
|
1426
|
+
v_hits = []
|
|
1427
|
+
if mode == 'presentation' and sections:
|
|
1428
|
+
for si, sec in enumerate(sections, 1):
|
|
1429
|
+
if not isinstance(sec, dict):
|
|
1430
|
+
continue
|
|
1431
|
+
pt = str(sec.get('type') or '')
|
|
1432
|
+
if pt in ('cover', 'agenda', 'closing', 'quote'):
|
|
1433
|
+
continue
|
|
1434
|
+
ch = sec.get('chart') if isinstance(sec.get('chart'), dict) else {}
|
|
1435
|
+
vals = ch.get('values') or []
|
|
1436
|
+
n = len(vals) if vals else len(sec.get('points') or sec.get('items') or [])
|
|
1437
|
+
skel = str(sec.get('layoutPreset') or '')
|
|
1438
|
+
# 极偏却仍 donut
|
|
1439
|
+
if pt == 'donut' or ch.get('type') in ('donut', 'pie'):
|
|
1440
|
+
if vals:
|
|
1441
|
+
nums = [float(v) for v in vals if isinstance(v, (int, float)) and v >= 0]
|
|
1442
|
+
if len(nums) >= 2:
|
|
1443
|
+
total = sum(nums) or 1
|
|
1444
|
+
pcts = [v / total * 100 for v in nums]
|
|
1445
|
+
if min(pcts) < 5:
|
|
1446
|
+
v_hits.append(f"sections[{si}] 极偏仍 {pt or ch.get('type')}(应 V3/kpi)")
|
|
1447
|
+
# 简单 1–2 点却声明 V1 全幅主视觉(layoutPreset P1 + 简单)
|
|
1448
|
+
if skel == 'P1' and n <= 2 and (pt in ('bar', 'donut') or ch.get('type')):
|
|
1449
|
+
v_hits.append(f"sections[{si}] P1+简单{n}点(应 V3/P3 或加注解)")
|
|
1450
|
+
chk("LAYOUT_QA_V_CONTRACT 演示 V1–V4 与复杂度匹配",
|
|
1451
|
+
not v_hits, "; ".join(v_hits[:3]) if v_hits else "")
|
|
1452
|
+
|
|
1453
|
+
# ⑥ 截断迹象:正文/列表项以省略号截断充数(antiTruncation)
|
|
1454
|
+
at = (LC.get('qualityGates') or {}).get('antiTruncation') or {}
|
|
1455
|
+
trunc_hits = []
|
|
1456
|
+
if at.get('forbidEllipsisTruncate', True):
|
|
1457
|
+
patterns = list(at.get('ellipsisPatterns') or ['…', '...', '……'])
|
|
1458
|
+
for i, b, head in content:
|
|
1459
|
+
plain_parts = re.findall(r'<(?:li|p)[^>]*>([\s\S]*?)</(?:li|p)>', b)
|
|
1460
|
+
plain_parts += re.findall(
|
|
1461
|
+
r'class="[^"]*(?:card__b|sowhat__v|point)[^"]*"[^>]*>([\s\S]*?)</',
|
|
1462
|
+
b)
|
|
1463
|
+
for raw in plain_parts:
|
|
1464
|
+
plain = _plain(raw).strip()
|
|
1465
|
+
if len(plain) < 8:
|
|
1466
|
+
continue
|
|
1467
|
+
for pat in patterns:
|
|
1468
|
+
if plain.endswith(pat) or plain.endswith(pat + '。'):
|
|
1469
|
+
# 排除「等…」短收口
|
|
1470
|
+
if re.search(r'等[…\.]{1,3}$', plain) and len(plain) <= 16:
|
|
1471
|
+
continue
|
|
1472
|
+
trunc_hits.append(f"第{i}页「{plain[:18]}」")
|
|
1473
|
+
break
|
|
1474
|
+
chk("LAYOUT_QA_TRUNCATION 禁静默截断(列表/卡/结论勿以省略号砍义)",
|
|
1475
|
+
not trunc_hits, "; ".join(trunc_hits[:4]) if trunc_hits else "")
|
|
1476
|
+
|
|
1477
|
+
# ⑦ 溢出却无拆页策略:页高估算超预算,且无 band--flow / 续页标记 / 多部分 title
|
|
1478
|
+
overflow_hits = []
|
|
1479
|
+
if at.get('overflowNoSplitFail', True):
|
|
1480
|
+
B = MODE_BUDGETS.get(mode, MODE_BUDGETS['presentation'])
|
|
1481
|
+
fs_px = int(B.get('bodyPx') or 17)
|
|
1482
|
+
wrap_px = int(B.get('wrap') or 1400) - WRAP_INSET_PX
|
|
1483
|
+
cpl = max(10, int(wrap_px / fs_px))
|
|
1484
|
+
for i, b, head in content:
|
|
1485
|
+
if 'band--flow' in head:
|
|
1486
|
+
continue
|
|
1487
|
+
# 已有拆页/续页信号则豁免
|
|
1488
|
+
if re.search(r'(续|续表|01[ab]|02[ab]|part\s*[12]|跟进|详见下页)', b, re.I):
|
|
1489
|
+
continue
|
|
1490
|
+
b_clean = re.sub(r'<script\b[\s\S]*?</script>', ' ', b)
|
|
1491
|
+
b_clean = re.sub(r'<style\b[\s\S]*?</style>', ' ', b_clean)
|
|
1492
|
+
plen = len(_plain(b_clean).strip())
|
|
1493
|
+
u = _band_heavy_units(b)
|
|
1494
|
+
est_h = plen / cpl * fs_px * LINE_FACTOR + u * UNIT_PX + SHEAD_PX
|
|
1495
|
+
if est_h > SCREEN_BUDGET_PX * 1.08:
|
|
1496
|
+
overflow_hits.append(f"第{i}页≈{est_h:.0f}px 无拆页/换形态信号")
|
|
1497
|
+
chk("LAYOUT_QA_OVERFLOW_NO_SPLIT 溢出须拆页/换形态(禁硬塞)",
|
|
1498
|
+
not overflow_hits, "; ".join(overflow_hits[:3]) if overflow_hits else "")
|
|
1499
|
+
|
|
1500
|
+
# ⑧ 半空卡 vs 塞爆:同页多卡时过半卡正文过短
|
|
1501
|
+
he = (LC.get('qualityGates') or {}).get('halfEmpty') or {}
|
|
1502
|
+
half_hits = []
|
|
1503
|
+
min_cards = int(he.get('minCards') or 3)
|
|
1504
|
+
short_n = int(he.get('shortPlainChars') or 12)
|
|
1505
|
+
max_ratio = float(he.get('maxShortRatio') or 0.5)
|
|
1506
|
+
for i, b, head in content:
|
|
1507
|
+
cards = re.findall(r'class="[^"]*\bcard\b[^"]*"[^>]*>([\s\S]*?)(?=<div class="[^"]*\bcard\b|</section>|$)', b)
|
|
1508
|
+
if len(cards) < min_cards:
|
|
1509
|
+
# also count .card blocks via simpler split
|
|
1510
|
+
cards = re.split(r'class="[^"]*\bcard\b', b)[1:]
|
|
1511
|
+
if len(cards) < min_cards:
|
|
1512
|
+
continue
|
|
1513
|
+
shorts = 0
|
|
1514
|
+
for c in cards:
|
|
1515
|
+
plen = len(_plain(c[:800]).strip())
|
|
1516
|
+
if plen < short_n:
|
|
1517
|
+
shorts += 1
|
|
1518
|
+
if shorts / max(len(cards), 1) > max_ratio and shorts >= 2:
|
|
1519
|
+
half_hits.append(f"第{i}页半空卡 {shorts}/{len(cards)}")
|
|
1520
|
+
chk("LAYOUT_QA_HALF_EMPTY 半空卡过多(与塞爆同样不合格)",
|
|
1521
|
+
not half_hits, "; ".join(half_hits[:3]) if half_hits else "",
|
|
1522
|
+
level="WARN")
|
|
1523
|
+
|
|
1524
|
+
# ⑨ 列对齐节奏:同页多个 grid 混排却完全无对齐 token(加强 LAYOUT_ALIGN)
|
|
1525
|
+
align_cfg = (LC.get('layoutSystem') or {}).get('align') or {}
|
|
1526
|
+
rhythm_hits = []
|
|
1527
|
+
if align_cfg.get('requireColumnRhythm', True):
|
|
1528
|
+
tokens = set(align_cfg.get('mixedGridClasses') or ['a-start', 'a-c', 'a-end'])
|
|
1529
|
+
for i, b, head in content:
|
|
1530
|
+
mixed = _mixed_grid_needs_align(b)
|
|
1531
|
+
has_align = any(re.search(r'\b' + re.escape(tok) + r'\b', b) for tok in tokens)
|
|
1532
|
+
if mixed and not has_align:
|
|
1533
|
+
rhythm_hits.append(f"第{i}页混排缺对齐类")
|
|
1534
|
+
chk("LAYOUT_QA_ALIGN_RHYTHM 混排列节奏/共享对齐类",
|
|
1535
|
+
not rhythm_hits, "; ".join(rhythm_hits[:3]) if rhythm_hits else "")
|
|
1536
|
+
|
|
1537
|
+
|
|
1241
1538
|
def main():
|
|
1242
1539
|
if len(sys.argv) < 2:
|
|
1243
1540
|
print(__doc__)
|
|
@@ -1245,6 +1542,9 @@ def main():
|
|
|
1245
1542
|
path = Path(sys.argv[1])
|
|
1246
1543
|
strict = '--strict' in sys.argv
|
|
1247
1544
|
as_json = '--json' in sys.argv
|
|
1545
|
+
layout_qa = '--layout-qa' in sys.argv
|
|
1546
|
+
# Mode A / presentation:--strict 隐含 --layout-qa(正式演示交付不漏 V 契约)
|
|
1547
|
+
# research/architecture 不自动开启,避免污染密排/架构路径
|
|
1248
1548
|
if not path.exists():
|
|
1249
1549
|
print(f"文件不存在: {path}")
|
|
1250
1550
|
return 2
|
|
@@ -1278,6 +1578,8 @@ def main():
|
|
|
1278
1578
|
f"读到 {mode!r}(未声明按 presentation 处理)" if mode not in MODE_BUDGETS else "",
|
|
1279
1579
|
level="WARN" if mode is None else "FAIL")
|
|
1280
1580
|
mode = mode if mode in MODE_BUDGETS else 'presentation'
|
|
1581
|
+
if strict and mode == 'presentation' and not layout_qa:
|
|
1582
|
+
layout_qa = True # presentation + --strict → 自动 layout-qa
|
|
1281
1583
|
|
|
1282
1584
|
# ── 宽屏与页面高度模型 ──
|
|
1283
1585
|
B0 = MODE_BUDGETS[mode]
|
|
@@ -1400,9 +1702,33 @@ def main():
|
|
|
1400
1702
|
chk("research 行动标题(章节主标题 ≥12 字,标题即结论)", not short,
|
|
1401
1703
|
f"过短: {short[:3]}" if short else "", level="WARN")
|
|
1402
1704
|
|
|
1705
|
+
if mode == 'presentation':
|
|
1706
|
+
# Mode A:主张/行动句标题(action title);纯话题标签 WARN(硬 FAIL 过脆)
|
|
1707
|
+
STRUCT_A = ('报告大纲', '大纲', '议程', 'Agenda', '参考资料', '下一步', '结论',
|
|
1708
|
+
'封面', '目录', '附录', '谢谢', 'Thank', 'Q&A', '问答')
|
|
1709
|
+
TOPIC_ONLY = (
|
|
1710
|
+
'现状分析', '市场格局', '风险与挑战', '背景介绍', '项目概述', '总结',
|
|
1711
|
+
'概览', '概述', '简介', '背景', '方案', '规划', '进展', '回顾',
|
|
1712
|
+
'分析', '对比', '数据', '附录', '下一步计划', '内容', '主题',
|
|
1713
|
+
)
|
|
1714
|
+
h1s_a = re.findall(r'<h2 class="t-h1 shead__title"[^>]*>(.*?)</h2>', txt)
|
|
1715
|
+
titles_a = [re.sub(r'<[^>]+>', '', h).strip() for h in h1s_a]
|
|
1716
|
+
topic_hits = []
|
|
1717
|
+
for h in titles_a:
|
|
1718
|
+
if not h or any(h.startswith(s) or h == s for s in STRUCT_A):
|
|
1719
|
+
continue
|
|
1720
|
+
# 纯话题:命中话题词表,或极短且无判断/数字/动词痕迹
|
|
1721
|
+
if h in TOPIC_ONLY or (len(h) <= 6 and not re.search(
|
|
1722
|
+
r'\d|是|应|须|将|已|要|可|能|达|超|降|升|破|卡|成|未|无|有', h)):
|
|
1723
|
+
topic_hits.append(h)
|
|
1724
|
+
chk("presentation 主张/行动标题(禁纯话题标签)", not topic_hits,
|
|
1725
|
+
f"话题式: {topic_hits[:4]}" if topic_hits else "", level="WARN")
|
|
1726
|
+
|
|
1403
1727
|
_check_content_quality(txt, chk, mode, model)
|
|
1404
1728
|
_check_v9_hard_gates(txt, chk, model)
|
|
1405
1729
|
_check_layout_grammar(txt, chk, mode)
|
|
1730
|
+
if layout_qa:
|
|
1731
|
+
_check_layout_qa(txt, chk, mode, model)
|
|
1406
1732
|
|
|
1407
1733
|
# ── 去AI味(词表来自单源) ──
|
|
1408
1734
|
body_plain = _plain(body_txt)
|