@ccoalm/ccl-skills 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/references/public-data-acquisition.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/references/public-disclosure-channels.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +20 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +12 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ci_checkout_ref_binding.sh +85 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_domain_scan_terms.sh +123 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/deliverable-doc-genre-skeletons.md +133 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/doc-charter-first.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/figure-and-table-craft.md +345 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/AGENTS.md +46 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/doc-lint-repo.py +203 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/doc-lint.py +246 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/figure-lint.py +1092 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/mutation_probe.sh +100 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/test_doc_lint_repo.py +420 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/test_figure_and_doc_lint.sh +375 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/control.md +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/empty-header.md +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/fake-header.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/fenced-noise.md +14 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/fig-dangling.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/fig-orphan-captioned.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/fig-orphan.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/fig.png +0 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/imbalance.md +41 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/no-unit.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/should-be-chart.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/tables-only-clean.md +35 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/unfilled.md +7 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/doc/wide-table.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/bad-viewbox.svg +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/blackmarker.svg +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/control.svg +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/crossings.svg +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/cvd-confusable.svg +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/decorative-line.svg +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/edge-no-arrow.svg +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/edge-vague.svg +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/figure-contract.json +21 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/figure-is-a-list.svg +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/flow-mixed.svg +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/low-contrast.svg +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/malformed.svg +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/no-aria.svg +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/no-group.svg +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/no-legend.svg +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/no-title.svg +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/no-viewbox.svg +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/offcontract-shape.svg +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/overflow.svg +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/transformed.svg +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/ungrouped-card.svg +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/tests/svg/unlabeled-edge.svg +13 -0
- package/dist/assets/release.json +254 -14
- package/package.json +1 -1
package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/scripts/doc-lint-repo.py
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Fail when a tracked Markdown doc in a repository carries an ERROR-class defect.
|
|
3
|
+
|
|
4
|
+
`doc-lint.py` (this script's sibling) decides every predicate for ONE document.
|
|
5
|
+
This is the repository-wide enumerator: it lists the tracked Markdown of a repo
|
|
6
|
+
you point it at, runs the linter over all of it in one pass, and turns the ERROR
|
|
7
|
+
tier into a landing gate. Nothing here judges a document; it decides only WHICH
|
|
8
|
+
files are in scope and WHICH tier blocks.
|
|
9
|
+
|
|
10
|
+
python3 doc-lint-repo.py /path/to/your/repo
|
|
11
|
+
|
|
12
|
+
It is written to run against ANY repository, not only the one that ships it. That
|
|
13
|
+
is the point: a repo-local scanner would have made this skill's own documents the
|
|
14
|
+
only ones ever checked. Both paths it depends on are derived, not assumed — the
|
|
15
|
+
linter is the sibling file next to this script, and the fixture exclusion is that
|
|
16
|
+
linter's own `tests/` directory, resolved against whatever repo you scanned.
|
|
17
|
+
|
|
18
|
+
WHY ERROR ONLY. Measured over the shipping repository's 481 tracked non-fixture
|
|
19
|
+
Markdown files: 0 ERROR, 75 WARN. Blocking on the WARN tier would fail that repo
|
|
20
|
+
on its own docs the day the gate lands, and the right conclusion is not a waiver
|
|
21
|
+
list — the WARN predicates are labelled `[工]` engineering proxies in
|
|
22
|
+
`../references/figure-and-table-craft.md` (§9b), and that file's own rule is that
|
|
23
|
+
a proxy which cannot separate a defect from a judgement call must not gate. So
|
|
24
|
+
the WARN count is printed for visibility and does not block. The ERROR tier is
|
|
25
|
+
the objective half: an empty or absent table header, a dangling figure reference,
|
|
26
|
+
an unreadable file.
|
|
27
|
+
|
|
28
|
+
WHY A FIXTURE DIRECTORY IS EXCLUDED, and why that is not convenience. The
|
|
29
|
+
linter's own corpus under its `tests/` directory contains documents built to
|
|
30
|
+
violate each predicate; measured, it reports 2 ERROR by construction. Scanning it
|
|
31
|
+
would make this gate permanently red for the exact inputs that prove the linter
|
|
32
|
+
works. The exclusion is ONE derived prefix and applies only when that directory
|
|
33
|
+
actually sits inside the scanned repo — in a consuming repo the skill is
|
|
34
|
+
installed elsewhere, so nothing is excluded and nothing needs to be. It is
|
|
35
|
+
asserted in both directions: a real doc must not be dropped by it, and a fixture
|
|
36
|
+
must not be scanned. A wider exclusion is the failure mode to guard against — it
|
|
37
|
+
hides real documents while still printing a pass.
|
|
38
|
+
|
|
39
|
+
Scope is TRACKED files only: an untracked scratch file is not part of the
|
|
40
|
+
delivered repo, and including it would make the verdict depend on the working
|
|
41
|
+
tree rather than on what lands.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
from __future__ import annotations
|
|
45
|
+
|
|
46
|
+
import os
|
|
47
|
+
import re
|
|
48
|
+
import subprocess
|
|
49
|
+
import sys
|
|
50
|
+
from pathlib import Path
|
|
51
|
+
|
|
52
|
+
HERE = Path(__file__).resolve().parent
|
|
53
|
+
# The linter that owns every predicate — the sibling file, NOT a path guessed
|
|
54
|
+
# inside the scanned repo. Deriving it is what lets this run against a repo that
|
|
55
|
+
# has never heard of this skill.
|
|
56
|
+
LINTER = HERE / "doc-lint.py"
|
|
57
|
+
# Documents built to violate the predicates so the linter can prove it detects
|
|
58
|
+
# them. Derived from the linter's own location, and applied only if that location
|
|
59
|
+
# is inside the repo being scanned.
|
|
60
|
+
FIXTURE_DIR = HERE / "tests"
|
|
61
|
+
|
|
62
|
+
TOTAL_RE = re.compile(r"^合计: (\d+) ERROR, (\d+) WARN", re.M)
|
|
63
|
+
ERROR_ROW_RE = re.compile(r"^ ERROR +([A-Z][A-Z0-9-]+)", re.M)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def fixture_prefix(root: Path) -> str | None:
|
|
67
|
+
"""The fixture directory as a repo-relative prefix, or None if outside it."""
|
|
68
|
+
try:
|
|
69
|
+
rel = FIXTURE_DIR.resolve().relative_to(root)
|
|
70
|
+
except ValueError:
|
|
71
|
+
return None
|
|
72
|
+
return f"{rel.as_posix()}/"
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def tracked_markdown(root: Path) -> list[str]:
|
|
76
|
+
"""Tracked Markdown paths, fixtures removed. Fails closed on a git error.
|
|
77
|
+
|
|
78
|
+
Enumerate EVERYTHING and filter on a case-folded suffix rather than asking
|
|
79
|
+
git for `*.md`. That pathspec is case-sensitive on a case-sensitive
|
|
80
|
+
filesystem, so a tracked `NOTES.MD` is silently outside a gate that reports
|
|
81
|
+
repository-wide Markdown coverage — a hole in exactly the property this
|
|
82
|
+
script exists to provide.
|
|
83
|
+
"""
|
|
84
|
+
proc = subprocess.run(
|
|
85
|
+
["git", "-C", str(root), "ls-files", "-z"],
|
|
86
|
+
capture_output=True,
|
|
87
|
+
text=True,
|
|
88
|
+
)
|
|
89
|
+
if proc.returncode != 0:
|
|
90
|
+
raise RuntimeError(f"git ls-files exited {proc.returncode}: {proc.stderr.strip()}")
|
|
91
|
+
paths = [p for p in proc.stdout.split("\0") if p]
|
|
92
|
+
md = [p for p in paths if Path(p).suffix.casefold() == ".md"]
|
|
93
|
+
prefix = fixture_prefix(root)
|
|
94
|
+
if prefix is None:
|
|
95
|
+
return md
|
|
96
|
+
return [p for p in md if not p.startswith(prefix)]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def main(argv: list[str]) -> int:
|
|
100
|
+
root = Path(argv[0] if argv else ".").resolve()
|
|
101
|
+
linter = LINTER
|
|
102
|
+
if not linter.is_file():
|
|
103
|
+
# Fail closed: a missing linter must not read as "nothing to report".
|
|
104
|
+
print(
|
|
105
|
+
f"doc_structure_check_failed: linter not found at {linter}; "
|
|
106
|
+
"the gate cannot certify a corpus it never scanned",
|
|
107
|
+
file=sys.stderr,
|
|
108
|
+
)
|
|
109
|
+
return 1
|
|
110
|
+
|
|
111
|
+
try:
|
|
112
|
+
files = tracked_markdown(root)
|
|
113
|
+
except RuntimeError as exc:
|
|
114
|
+
print(f"doc_structure_check_failed: {exc}", file=sys.stderr)
|
|
115
|
+
return 1
|
|
116
|
+
|
|
117
|
+
if not files:
|
|
118
|
+
# An empty scope is a scoping bug, not a clean corpus. A repository worth
|
|
119
|
+
# gating has tracked Markdown; zero means the enumeration broke or the
|
|
120
|
+
# path is not the repo you meant.
|
|
121
|
+
print(
|
|
122
|
+
"doc_structure_check_failed: no tracked Markdown in scope — the "
|
|
123
|
+
"enumeration is broken, not the corpus clean",
|
|
124
|
+
file=sys.stderr,
|
|
125
|
+
)
|
|
126
|
+
return 1
|
|
127
|
+
|
|
128
|
+
proc = subprocess.run(
|
|
129
|
+
[sys.executable, str(linter), *files],
|
|
130
|
+
capture_output=True,
|
|
131
|
+
text=True,
|
|
132
|
+
cwd=str(root),
|
|
133
|
+
)
|
|
134
|
+
# doc-lint's contract: 0 clean / 2 WARN-only / 1 has ERROR. Anything else is
|
|
135
|
+
# the linter itself failing, which this gate must not read as a clean corpus.
|
|
136
|
+
if proc.returncode not in (0, 1, 2):
|
|
137
|
+
print(
|
|
138
|
+
f"doc_structure_check_failed: linter exited {proc.returncode} "
|
|
139
|
+
f"(expected 0/1/2); its output cannot be trusted as a verdict\n"
|
|
140
|
+
f"{proc.stderr.strip()[:2000]}",
|
|
141
|
+
file=sys.stderr,
|
|
142
|
+
)
|
|
143
|
+
return 1
|
|
144
|
+
|
|
145
|
+
total = TOTAL_RE.search(proc.stdout)
|
|
146
|
+
if not total:
|
|
147
|
+
print(
|
|
148
|
+
"doc_structure_check_failed: linter produced no 合计 line, so the "
|
|
149
|
+
"count it reports cannot be read; treat as unscanned",
|
|
150
|
+
file=sys.stderr,
|
|
151
|
+
)
|
|
152
|
+
return 1
|
|
153
|
+
n_err, n_warn = int(total.group(1)), int(total.group(2))
|
|
154
|
+
|
|
155
|
+
# CROSS-CHECK the exit code against the counts. Accepting the code as merely
|
|
156
|
+
# "in range" and then trusting the totals alone leaves the contradictory case
|
|
157
|
+
# open: a linter that exits 1 — its own signal for "there are ERRORs" — while
|
|
158
|
+
# printing `合计: 0 ERROR` is read here as a clean corpus and returns 0. That
|
|
159
|
+
# is the shape a partial failure after summary emission takes, and it is the
|
|
160
|
+
# one combination where every other guard in this file is satisfied. Both
|
|
161
|
+
# review lanes found it independently; the crash and unparseable-output tests
|
|
162
|
+
# do not reach it, because this output is perfectly parseable and merely lying.
|
|
163
|
+
expected_rc = 1 if n_err else (2 if n_warn else 0)
|
|
164
|
+
if proc.returncode != expected_rc:
|
|
165
|
+
print(
|
|
166
|
+
f"doc_structure_check_failed: linter exit code {proc.returncode} "
|
|
167
|
+
f"contradicts its own summary ({n_err} ERROR, {n_warn} WARN, which "
|
|
168
|
+
f"its contract maps to {expected_rc}). One of the two is wrong, so "
|
|
169
|
+
f"neither can be trusted as a verdict; treat as unscanned.",
|
|
170
|
+
file=sys.stderr,
|
|
171
|
+
)
|
|
172
|
+
return 1
|
|
173
|
+
|
|
174
|
+
if n_err:
|
|
175
|
+
# Print the ERROR rows only. The WARN tier is visibility, and dumping 75
|
|
176
|
+
# advisory lines into a failure message buries the blocking ones.
|
|
177
|
+
for line in proc.stdout.splitlines():
|
|
178
|
+
if line.startswith(" ERROR") or (
|
|
179
|
+
line.strip() and not line.startswith(" ") and "ERROR" in line
|
|
180
|
+
):
|
|
181
|
+
print(line, file=sys.stderr)
|
|
182
|
+
codes = sorted(set(ERROR_ROW_RE.findall(proc.stdout)))
|
|
183
|
+
print(
|
|
184
|
+
f"doc_structure_check_failed: {n_err} ERROR-class structure defect(s) "
|
|
185
|
+
f"across {len(files)} tracked doc(s): {', '.join(codes)}. "
|
|
186
|
+
f"Fix the document; these are objective defects (a table with no real "
|
|
187
|
+
f"header, a figure reference with no such figure, an unreadable file), "
|
|
188
|
+
f"not style preferences. Run "
|
|
189
|
+
f"`python3 {linter} <file>` for the per-file detail.",
|
|
190
|
+
file=sys.stderr,
|
|
191
|
+
)
|
|
192
|
+
return 1
|
|
193
|
+
|
|
194
|
+
print(
|
|
195
|
+
f"doc_structure_check_ok: {len(files)} tracked doc(s), 0 ERROR, "
|
|
196
|
+
f"{n_warn} WARN (advisory, non-blocking — the WARN predicates are `[工]` "
|
|
197
|
+
f"proxies and do not gate)"
|
|
198
|
+
)
|
|
199
|
+
return 0
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
if __name__ == "__main__":
|
|
203
|
+
raise SystemExit(main(sys.argv[1:]))
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""doc-lint — 交付文档的表格与结构的起草期确定性检查(Markdown)。
|
|
3
|
+
|
|
4
|
+
判据来源:
|
|
5
|
+
[外] WCAG 2.2 SC 1.3.1 (Level A) — 通过视觉呈现传达的信息与关系必须可被程序确定。
|
|
6
|
+
对表格即:真表头,不能用加粗行冒充表头。
|
|
7
|
+
[外] ICD 203 §9 — tables 属于 visual information;且「图形式比文字更能传达
|
|
8
|
+
空间/时间关系时应当出图」。据此判「该出图却压成表」。
|
|
9
|
+
[外] ICD 203 §1/§3 — 承载判断的内容要能追到来源,并区分信息与假设。
|
|
10
|
+
[工] 其余为工程判断(列数过多、占位符未填、数值列无单位、图文引用一致)。
|
|
11
|
+
|
|
12
|
+
已删除的谓词(勿再加回):标题过长、单元格塞整段、段落内并列枚举、列表项多句。
|
|
13
|
+
它们都拿宽度或计数当「表达好不好」的代理,而这类阈值经查证无可靠来源;
|
|
14
|
+
全仓 392 份文档试跑时它们产生 318 条 ERROR,抽样中无一条是其声称的缺陷。
|
|
15
|
+
|
|
16
|
+
用法: doc-lint.py <file.md>... [--json]
|
|
17
|
+
"""
|
|
18
|
+
import sys, re, os, json, collections
|
|
19
|
+
|
|
20
|
+
CJK = re.compile(r'[\u4e00-\u9fff]')
|
|
21
|
+
|
|
22
|
+
def width(s):
|
|
23
|
+
"""近似显示宽度:汉字 2,其余 1。"""
|
|
24
|
+
return sum(2 if CJK.match(c) else 1 for c in s)
|
|
25
|
+
|
|
26
|
+
def parse_tables(lines):
|
|
27
|
+
"""返回 [(start_line, header_cells, sep_ok, rows)]"""
|
|
28
|
+
tables = []
|
|
29
|
+
i = 0
|
|
30
|
+
while i < len(lines):
|
|
31
|
+
if lines[i].lstrip().startswith('|') and i + 1 < len(lines):
|
|
32
|
+
sep = lines[i + 1].strip()
|
|
33
|
+
sep_ok = bool(re.match(r'^\|?[\s:\-\|]+\|[\s:\-\|]*$', sep)) and '-' in sep
|
|
34
|
+
if sep_ok:
|
|
35
|
+
header = [c.strip() for c in lines[i].strip().strip('|').split('|')]
|
|
36
|
+
rows = []
|
|
37
|
+
j = i + 2
|
|
38
|
+
while j < len(lines) and lines[j].lstrip().startswith('|'):
|
|
39
|
+
rows.append([c.strip() for c in lines[j].strip().strip('|').split('|')])
|
|
40
|
+
j += 1
|
|
41
|
+
tables.append((i + 1, header, sep_ok, rows))
|
|
42
|
+
i = j
|
|
43
|
+
continue
|
|
44
|
+
i += 1
|
|
45
|
+
return tables
|
|
46
|
+
|
|
47
|
+
PLACEHOLDER = {'-', '—', '–', 'N/A', 'n/a', 'TBD', 'TODO', '待定', '待补', '?', '待填'}
|
|
48
|
+
NUMRE = re.compile(r'^[¥$€]?\s*-?[\d,]+(\.\d+)?\s*[%‰]?$')
|
|
49
|
+
UNIT_HINT = re.compile(r'[%‰]|元|美元|万|亿|GB|TB|MB|ms|s\b|次|人|天|月|年|条|个|倍|Ki?B|/')
|
|
50
|
+
|
|
51
|
+
FENCE = re.compile(r'^\s*(```|~~~)')
|
|
52
|
+
|
|
53
|
+
def strip_fences(lines):
|
|
54
|
+
"""把围栏代码块内的行替换成空行(保留行号)。
|
|
55
|
+
|
|
56
|
+
代码块里的内容不是文档结构:ASCII 分隔线、缩进的示例标题、
|
|
57
|
+
示例表格都会被结构谓词误判。全仓试跑时 `── Report ──` 这类
|
|
58
|
+
分隔线被当成标题,就是漏了这一层。
|
|
59
|
+
"""
|
|
60
|
+
out = []
|
|
61
|
+
in_fence = False
|
|
62
|
+
for ln in lines:
|
|
63
|
+
if FENCE.match(ln):
|
|
64
|
+
in_fence = not in_fence
|
|
65
|
+
out.append('')
|
|
66
|
+
continue
|
|
67
|
+
out.append('' if in_fence else ln)
|
|
68
|
+
return out
|
|
69
|
+
|
|
70
|
+
def lint(path):
|
|
71
|
+
# 非 UTF-8 文本(图片等)不是本检查器的对象:跳过而不是崩——
|
|
72
|
+
# 调用方按目录通配传入时,目录里混着图片是常态。
|
|
73
|
+
# 与 figure-lint 对齐:坏文件报 READ 并继续整批,不能静默判干净——
|
|
74
|
+
# 早先返回空发现集 + skipped=true,等于让损坏的 .md「干净通过」。
|
|
75
|
+
try:
|
|
76
|
+
src = open(path, encoding='utf8').read()
|
|
77
|
+
except (UnicodeDecodeError, OSError) as e:
|
|
78
|
+
return ([{'level': 'ERROR', 'code': 'READ',
|
|
79
|
+
'msg': f'无法读取: {type(e).__name__}: {e}', 'line': None}],
|
|
80
|
+
{'tables': 0, 'figures': 0, 'lines': 0})
|
|
81
|
+
raw_lines = src.split('\n')
|
|
82
|
+
# mermaid 图是**图**不是代码:必须在剥离围栏前数,否则恒为 0,
|
|
83
|
+
# 既会制造 CARRIER-IMBALANCE 假报,又让图文引用检查漏检。
|
|
84
|
+
n_mermaid = sum(1 for ln in raw_lines if re.match(r'^\s*(```|~~~)\s*mermaid\b', ln))
|
|
85
|
+
lines = strip_fences(raw_lines)
|
|
86
|
+
src = '\n'.join(lines) # 后续按剥离围栏后的正文分析
|
|
87
|
+
F = []
|
|
88
|
+
def add(level, code, msg, line=None):
|
|
89
|
+
F.append({'level': level, 'code': code, 'msg': msg, 'line': line})
|
|
90
|
+
|
|
91
|
+
# ---- 表格 ----
|
|
92
|
+
tables = parse_tables(lines)
|
|
93
|
+
fake_header_rows = 0
|
|
94
|
+
for (ln, header, sep_ok, rows) in tables:
|
|
95
|
+
ncol = len(header)
|
|
96
|
+
|
|
97
|
+
# [外] WCAG 1.3.1:表头必须是真表头
|
|
98
|
+
if all((not h) or h in PLACEHOLDER for h in header):
|
|
99
|
+
add('ERROR', 'WCAG-131-TABLE', f'表格表头为空——表头必须可被程序确定,不能靠视觉暗示', ln)
|
|
100
|
+
|
|
101
|
+
# [工] 列数过多
|
|
102
|
+
if ncol > 8:
|
|
103
|
+
add('WARN', 'TABLE-WIDE', f'{ncol} 列——超出一屏可读范围,考虑拆表或转置', ln)
|
|
104
|
+
|
|
105
|
+
# [工] 占位符未填
|
|
106
|
+
cells = [c for row in rows for c in row]
|
|
107
|
+
if cells:
|
|
108
|
+
ph = sum(1 for c in cells if c in PLACEHOLDER or not c)
|
|
109
|
+
if ph / len(cells) > 0.25:
|
|
110
|
+
add('WARN', 'TABLE-UNFILLED',
|
|
111
|
+
f'{ph}/{len(cells)} 个单元格为空或占位符({ph/len(cells)*100:.0f}%)——表未填完', ln)
|
|
112
|
+
|
|
113
|
+
# [工] 数值列缺单位/口径
|
|
114
|
+
for c_i in range(ncol):
|
|
115
|
+
col = [row[c_i] for row in rows if c_i < len(row)]
|
|
116
|
+
nums = [c for c in col if NUMRE.match(c)]
|
|
117
|
+
if len(nums) >= 3 and len(nums) / max(len(col), 1) > 0.6:
|
|
118
|
+
head = header[c_i] if c_i < len(header) else ''
|
|
119
|
+
if not UNIT_HINT.search(head) and not any(UNIT_HINT.search(c) for c in nums):
|
|
120
|
+
add('WARN', 'TABLE-NO-UNIT',
|
|
121
|
+
f'数值列「{head or f"第{c_i+1}列"}」表头与单元格均无单位/口径', ln)
|
|
122
|
+
|
|
123
|
+
# 载体选择原则是 [外](ICD 203 §9:图形式更能传达空间/时间关系时应出图),
|
|
124
|
+
# 但「≥6 行、数值占比 >0.8、≤3 列」这三个识别阈值是 [工]——本检查器自定的
|
|
125
|
+
# 保守下界,无外部依据。两者档位不同,不能一起挂在 [外] 名下。
|
|
126
|
+
if len(rows) >= 6:
|
|
127
|
+
numcols = 0
|
|
128
|
+
for c_i in range(ncol):
|
|
129
|
+
col = [row[c_i] for row in rows if c_i < len(row)]
|
|
130
|
+
if col and sum(1 for c in col if NUMRE.match(c)) / len(col) > 0.8:
|
|
131
|
+
numcols += 1
|
|
132
|
+
if numcols == 1 and ncol <= 3:
|
|
133
|
+
add('WARN', 'ICD203-9-SHOULD-BE-CHART',
|
|
134
|
+
f'{len(rows)} 行 × {ncol} 列且仅一列数值——量级对比用图形式更能传达。'
|
|
135
|
+
f'[外] 载体选择原则来自 ICD 203 §9;[工] 触发阈值(≥6 行 / 数值占比 >0.8 / ≤3 列)'
|
|
136
|
+
f'为本检查器自定的保守下界,无外部依据', ln)
|
|
137
|
+
|
|
138
|
+
# ---- 文档级:表图配比 ----
|
|
139
|
+
n_fig = len(re.findall(r'!\[', src)) + n_mermaid + len(re.findall(r'<img', src))
|
|
140
|
+
n_tab = len(tables)
|
|
141
|
+
# [工] 计数阈值是工程启发式,不是 ICD 203 的内容。ICD 203 §9 的判据是
|
|
142
|
+
# 「图形式是否比文字更能传达」——那取决于内容,不取决于表的个数。
|
|
143
|
+
# 八张查询表 / 模式表 / 证据表本来就不需要图,所以这条只报 WARN 不阻断,
|
|
144
|
+
# 且只在存在「量级对比型」表格(已由 ICD203-9-SHOULD-BE-CHART 认定)时才提示。
|
|
145
|
+
chart_worthy = any(x['code'] == 'ICD203-9-SHOULD-BE-CHART' for x in F)
|
|
146
|
+
if n_tab >= 8 and n_fig == 0 and chart_worthy:
|
|
147
|
+
add('WARN', 'CARRIER-IMBALANCE',
|
|
148
|
+
f'{n_tab} 个表、0 张图,且其中有适合出图的量级对比表——'
|
|
149
|
+
f'信息可能整体压给了表格(计数阈值为工程启发式,非外部标准)')
|
|
150
|
+
elif n_tab >= 10 and n_fig and n_tab / n_fig > 8 and chart_worthy:
|
|
151
|
+
add('WARN', 'CARRIER-IMBALANCE',
|
|
152
|
+
f'{n_tab} 表 / {n_fig} 图,比值 {n_tab/n_fig:.1f}——偏表(工程启发式)')
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
# ---- 图文一致 ----
|
|
156
|
+
# 正文引用的图号 vs 实际图数;以及有图从不被正文引用(孤图)
|
|
157
|
+
# 采集「正文引用」时必须排除图注行本身,否则图注会把自己算成引用(假阴性)
|
|
158
|
+
CAPTION_LINE = re.compile(r'^\s*[*_]*图\s*\d{1,2}\s*[::]')
|
|
159
|
+
fig_refs = set()
|
|
160
|
+
for _ln in lines:
|
|
161
|
+
if CAPTION_LINE.match(_ln):
|
|
162
|
+
continue
|
|
163
|
+
for m in re.finditer(r'图\s*(\d{1,2})', _ln):
|
|
164
|
+
fig_refs.add(int(m.group(1)))
|
|
165
|
+
# 图注锚(![...] 的 alt、或「图 N:」形式的说明行)
|
|
166
|
+
caption_nums = set()
|
|
167
|
+
for m in re.finditer(r'^\s*\*?图\s*(\d{1,2})\s*[::]', src, re.M):
|
|
168
|
+
caption_nums.add(int(m.group(1)))
|
|
169
|
+
n_fig_local = n_fig
|
|
170
|
+
# 采用编号图注约定时按**实际编号**比对;早先按数量比(n > max(图数, 图注数))
|
|
171
|
+
# 两个方向都会错:引用图 2 而图注只有 1 和 3 不报,唯一图注是图 10 却误报。
|
|
172
|
+
if fig_refs and caption_nums and not n_fig_local:
|
|
173
|
+
# 有编号图注、却没有任何真实图实例:图注在描述不存在的图。
|
|
174
|
+
add('ERROR', 'FIG-REF-DANGLING',
|
|
175
|
+
f'文档有编号图注 {sorted(caption_nums)} 但没有任何图片或 mermaid 图——图注指向的图不存在')
|
|
176
|
+
elif fig_refs and caption_nums:
|
|
177
|
+
missing = sorted(fig_refs - caption_nums)
|
|
178
|
+
if missing:
|
|
179
|
+
add('ERROR', 'FIG-REF-DANGLING',
|
|
180
|
+
f'正文引用了「图 {missing}」但文档没有对应编号的图注'
|
|
181
|
+
f'(现有图注编号 {sorted(caption_nums)})——引用悬空')
|
|
182
|
+
elif fig_refs and not caption_nums and n_fig_local:
|
|
183
|
+
# 「引用悬空」的前提是文档确实在用图系统。一张图都没有的文档里出现「图 N」,
|
|
184
|
+
# 那是在**谈论**图(举例、引用规范条文),不是在引用本文档的图——
|
|
185
|
+
# 本检查器自己的判据文档就因此被误判过一次。
|
|
186
|
+
missing = sorted(n for n in fig_refs if n > n_fig_local)
|
|
187
|
+
if missing:
|
|
188
|
+
add('ERROR', 'FIG-REF-DANGLING',
|
|
189
|
+
f'正文引用了「图 {missing}」但文档只有 {n_fig_local} 张图且无编号图注——引用悬空')
|
|
190
|
+
# 只有当文档已采用编号图注约定,或图多到需要索引(>=3)时,才要求正文引用。
|
|
191
|
+
# 单张随文插图不强制编号——否则控制组会被误报。
|
|
192
|
+
if n_fig_local and not fig_refs and (caption_nums or n_fig_local >= 3):
|
|
193
|
+
add('WARN', 'FIG-ORPHAN',
|
|
194
|
+
f'{n_fig_local} 张图,正文一次也没引用「图 N」——图与正文各说各的,读者不知道该在哪一步看图')
|
|
195
|
+
if caption_nums and fig_refs:
|
|
196
|
+
never_ref = sorted(caption_nums - fig_refs)
|
|
197
|
+
if never_ref:
|
|
198
|
+
add('WARN', 'FIG-ORPHAN', f'图注存在但正文未引用:图 {never_ref}')
|
|
199
|
+
|
|
200
|
+
# ---- 加粗行冒充小节标题 ----
|
|
201
|
+
for idx, ln in enumerate(lines, 1):
|
|
202
|
+
s = ln.strip()
|
|
203
|
+
if re.match(r'^\*\*[^*]{2,40}\*\*[::]?$', s):
|
|
204
|
+
fake_header_rows += 1
|
|
205
|
+
if fake_header_rows >= 5:
|
|
206
|
+
add('WARN', 'WCAG-131-FAKE-HEADING',
|
|
207
|
+
f'{fake_header_rows} 处「独占一行的加粗短语」——若充当小节标题,结构无法被程序确定(WCAG 1.3.1),且目录不可用')
|
|
208
|
+
|
|
209
|
+
return F, {'tables': n_tab, 'figures': n_fig, 'lines': len(lines)}
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def main(argv):
|
|
213
|
+
as_json = '--json' in argv
|
|
214
|
+
files = [a for a in argv[1:] if not a.startswith('--')]
|
|
215
|
+
if not files:
|
|
216
|
+
print('用法: doc-lint.py <file.md>... [--json]', file=sys.stderr)
|
|
217
|
+
return 2
|
|
218
|
+
out = {}
|
|
219
|
+
n_err = n_warn = 0
|
|
220
|
+
for f in files:
|
|
221
|
+
F, st = lint(f)
|
|
222
|
+
out[f] = {'findings': F, 'stats': st}
|
|
223
|
+
n_err += sum(1 for x in F if x['level'] == 'ERROR')
|
|
224
|
+
n_warn += sum(1 for x in F if x['level'] == 'WARN')
|
|
225
|
+
if as_json:
|
|
226
|
+
print(json.dumps(out, ensure_ascii=False, indent=1))
|
|
227
|
+
else:
|
|
228
|
+
for f, r in out.items():
|
|
229
|
+
st = r['stats']
|
|
230
|
+
print(f"\n{os.path.basename(f)} [{st['lines']} 行 / {st['tables']} 表 / {st['figures']} 图]")
|
|
231
|
+
seen = collections.Counter()
|
|
232
|
+
for x in r['findings']:
|
|
233
|
+
seen[x['code']] += 1
|
|
234
|
+
if seen[x['code']] <= 3:
|
|
235
|
+
loc = f"L{x['line']}" if x['line'] else '--'
|
|
236
|
+
print(f" {x['level']:5} {x['code']:26} {loc:>6} {x['msg']}")
|
|
237
|
+
for c, n in seen.items():
|
|
238
|
+
if n > 3:
|
|
239
|
+
print(f" ... {c:26} 另有 {n-3} 处")
|
|
240
|
+
if not r['findings']:
|
|
241
|
+
print(' ✓ 无发现')
|
|
242
|
+
print(f'\n合计: {n_err} ERROR, {n_warn} WARN')
|
|
243
|
+
return 1 if n_err else (2 if n_warn else 0)
|
|
244
|
+
|
|
245
|
+
if __name__ == '__main__':
|
|
246
|
+
sys.exit(main(sys.argv))
|