dflow-sdd-ddd 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +824 -1
- package/CONTRIBUTING.md +16 -10
- package/README.en.md +156 -200
- package/README.md +89 -144
- package/TEMPLATE-COVERAGE.md +15 -8
- package/TEMPLATE-LANGUAGE-GLOSSARY.md +15 -1
- package/bin/dflow.js +36 -4
- package/docs/commands.en.md +110 -0
- package/docs/commands.md +101 -0
- package/docs/doctor-uncertainty.en.md +212 -0
- package/docs/doctor-uncertainty.md +212 -0
- package/docs/evaluating-dflow.en.md +29 -11
- package/docs/evaluating-dflow.md +8 -6
- package/docs/npm-publish-checklist.md +3 -1
- package/docs/release-versioning-policy.md +8 -2
- package/docs/upgrading.en.md +196 -0
- package/docs/upgrading.md +197 -0
- package/docs/using-with-claude-code.en.md +25 -10
- package/docs/using-with-claude-code.md +20 -7
- package/docs/using-with-codex.en.md +18 -6
- package/docs/using-with-codex.md +16 -5
- package/docs/using-with-github-copilot.en.md +25 -10
- package/docs/using-with-github-copilot.md +21 -8
- package/lib/doc-shapes.json +997 -0
- package/lib/doctor-checks.js +2654 -0
- package/lib/init.js +3583 -107
- package/lib/render-diagrams.js +1474 -0
- package/lib/render.js +865 -49
- package/package.json +2 -2
- package/templates/brownfield/references/drift-verification.md +4 -0
- package/templates/brownfield/references/finish-feature-flow.md +635 -88
- package/templates/brownfield/references/finish-feature-follow-up.md +60 -0
- package/templates/brownfield/references/finish-feature-minimal-host.md +406 -0
- package/templates/brownfield/references/finish-feature-post-hoc-hotfix.md +95 -0
- package/templates/brownfield/references/git-integration.md +160 -15
- package/templates/brownfield/references/init-project-flow.md +26 -4
- package/templates/brownfield/references/modify-existing-flow.md +412 -87
- package/templates/brownfield/references/modify-existing-follow-up.md +121 -0
- package/templates/brownfield/references/modify-existing-post-hoc-hotfix.md +82 -0
- package/templates/brownfield/references/new-feature-flow.md +61 -6
- package/templates/brownfield/references/new-phase-flow.md +57 -7
- package/templates/brownfield/references/pr-review-checklist.md +303 -10
- package/templates/brownfield/scaffolding/AI-AGENT-GUIDE.md +158 -34
- package/templates/brownfield/scaffolding/CLAUDE-md-snippet.md +1 -1
- package/templates/brownfield/scaffolding/Git-principles-gitflow.md +75 -6
- package/templates/brownfield/scaffolding/Git-principles-trunk.md +82 -8
- package/templates/brownfield/scaffolding/_conventions.md +50 -28
- package/templates/brownfield/scaffolding/_overview.md +1 -0
- package/templates/brownfield/templates/_index.md +151 -7
- package/templates/brownfield/templates/analysis.md +79 -0
- package/templates/brownfield/templates/behavior.md +1 -0
- package/templates/brownfield/templates/context-definition.md +1 -0
- package/templates/brownfield/templates/context-map.md +2 -1
- package/templates/brownfield/templates/glossary.md +1 -0
- package/templates/brownfield/templates/lightweight-spec.md +154 -11
- package/templates/brownfield/templates/models.md +1 -0
- package/templates/brownfield/templates/phase-spec.md +9 -1
- package/templates/brownfield/templates/rules.md +1 -0
- package/templates/brownfield/templates/tech-debt.md +1 -0
- package/templates/common/references/ddd-modeling-guide.md +33 -16
- package/templates/{greenfield → common}/references/dflow-feedback-flow.md +2 -1
- package/templates/common/references/flow-rationale-registry.md +130 -0
- package/templates/common/skill/SKILL.md +13 -11
- package/templates/greenfield/references/drift-verification.md +4 -0
- package/templates/greenfield/references/finish-feature-flow.md +625 -89
- package/templates/greenfield/references/finish-feature-follow-up.md +60 -0
- package/templates/greenfield/references/finish-feature-minimal-host.md +363 -0
- package/templates/greenfield/references/finish-feature-post-hoc-hotfix.md +95 -0
- package/templates/greenfield/references/git-integration.md +148 -15
- package/templates/greenfield/references/init-project-flow.md +28 -8
- package/templates/greenfield/references/modify-existing-flow.md +378 -85
- package/templates/greenfield/references/modify-existing-follow-up.md +103 -0
- package/templates/greenfield/references/modify-existing-post-hoc-hotfix.md +82 -0
- package/templates/greenfield/references/new-feature-flow.md +67 -4
- package/templates/greenfield/references/new-phase-flow.md +56 -7
- package/templates/greenfield/references/pr-review-checklist.md +287 -8
- package/templates/greenfield/scaffolding/AI-AGENT-GUIDE.md +153 -32
- package/templates/greenfield/scaffolding/CLAUDE-md-snippet.md +5 -2
- package/templates/greenfield/scaffolding/Git-principles-gitflow.md +74 -6
- package/templates/greenfield/scaffolding/Git-principles-trunk.md +88 -12
- package/templates/greenfield/scaffolding/_conventions.md +50 -28
- package/templates/greenfield/scaffolding/_overview.md +6 -2
- package/templates/greenfield/templates/_index.md +137 -7
- package/templates/greenfield/templates/aggregate-design.md +1 -0
- package/templates/greenfield/templates/analysis.md +79 -0
- package/templates/greenfield/templates/behavior.md +1 -0
- package/templates/greenfield/templates/context-definition.md +1 -0
- package/templates/greenfield/templates/context-map.md +2 -1
- package/templates/greenfield/templates/events.md +4 -1
- package/templates/greenfield/templates/glossary.md +1 -0
- package/templates/greenfield/templates/lightweight-spec.md +154 -11
- package/templates/greenfield/templates/models.md +1 -0
- package/templates/greenfield/templates/phase-spec.md +9 -1
- package/templates/greenfield/templates/rules.md +1 -0
- package/templates/greenfield/templates/tech-debt.md +1 -0
- package/templates/brownfield/references/dflow-feedback-flow.md +0 -251
|
@@ -0,0 +1,2654 @@
|
|
|
1
|
+
// PROPOSAL-058: pure, shippable drift-detection helpers for `dflow doctor`.
|
|
2
|
+
//
|
|
3
|
+
// The dev-only cross-ref resolver (scripts/check-cross-refs.mjs, PROPOSAL-055)
|
|
4
|
+
// is not part of the npm package, so the runtime checks reimplement the narrow
|
|
5
|
+
// subset doctor needs: fence-aware heading extraction, "<file> § Heading"
|
|
6
|
+
// reference extraction with soft-wrap joining, tolerant heading matching, and
|
|
7
|
+
// template-shape comparison. Everything here is I/O-free — callers read the
|
|
8
|
+
// files and pass contents — which keeps the checks unit-testable.
|
|
9
|
+
|
|
10
|
+
'use strict';
|
|
11
|
+
|
|
12
|
+
// Machine-readable context lines. Shared with lib/init.js inference so the
|
|
13
|
+
// doctor "machine format" checks can never drift from what inference actually
|
|
14
|
+
// parses: inferGitPolicy / inferAiCommitMarker / inferProseLanguage read
|
|
15
|
+
// _conventions.md; inferTechStackSummary / inferMigrationContext read the
|
|
16
|
+
// `| Tech stack |` / `| Migration / legacy context |` rows of the guide's
|
|
17
|
+
// "## Project Context" table (PROPOSAL-076 — no packaged _overview.md template
|
|
18
|
+
// ever carried those rows).
|
|
19
|
+
const GIT_POLICY_LINE_RE = /Selected Git policy:\s*`([^`]+)`/;
|
|
20
|
+
const AI_COMMIT_MARKER_LINE_RE = /AI commit marker:\s*`([^`]+)`/;
|
|
21
|
+
const PROSE_LANGUAGE_LINE_RE = /Project prose language:\s*`([^`]+)`/;
|
|
22
|
+
// The row values accept Markdown-escaped pipes (`\|`) so a cell like
|
|
23
|
+
// `Node \| Express` is captured whole; parseContextLine unescapes them
|
|
24
|
+
// (PROPOSAL-076 gate G1 — a bare `[^|\n]` capture silently truncated at the
|
|
25
|
+
// escaped pipe while doctor still called the row machine-readable).
|
|
26
|
+
const TECH_STACK_ROW_RE = /\|\s*Tech stack\s*\|\s*((?:\\\||[^|\n])+?)\s*\|/i;
|
|
27
|
+
const MIGRATION_CONTEXT_ROW_RE = /\|\s*Migration \/ legacy context\s*\|\s*((?:\\\||[^|\n])+?)\s*\|/i;
|
|
28
|
+
|
|
29
|
+
const GIT_POLICY_VALUES = new Set(['gitflow', 'trunk']);
|
|
30
|
+
const AI_COMMIT_MARKER_VALUES = new Set(['none', 'co-authored-by', 'prefix']);
|
|
31
|
+
|
|
32
|
+
// Trimmed capture of a machine-readable context line, or null when the line is
|
|
33
|
+
// absent or its value is whitespace-only. Inference and the doctor checks must
|
|
34
|
+
// both parse through this helper: a value doctor accepts has to be the exact
|
|
35
|
+
// value configure-agents consumes (e.g. a stray space inside the backticks
|
|
36
|
+
// would otherwise pass doctor's trimmed validation yet miss strict comparisons
|
|
37
|
+
// like buildSubstitutionMap's `gitflow` check downstream).
|
|
38
|
+
function parseContextLine(content, re) {
|
|
39
|
+
// ⚠ Read only text a reader can see. A commented-out
|
|
40
|
+
// `<!-- AI commit marker: `prefix` -->` was parsed as though it were the live
|
|
41
|
+
// setting, so doctor reported the policy satisfied by a line the user had
|
|
42
|
+
// deliberately switched off — silent pass (`p082-b3-k1`). Callers pass whole
|
|
43
|
+
// files and single rows alike; a row with no HTML block in it is unchanged.
|
|
44
|
+
//
|
|
45
|
+
// ⚠⚠ THE TYPE CHECK IS PART OF THAT FIX, NOT TIDINESS. This function used to
|
|
46
|
+
// reach `content.match(re)` directly, so a caller handing it `undefined` got a
|
|
47
|
+
// loud TypeError. Routing through the projection would have swallowed that:
|
|
48
|
+
// `classifiedVisible` calls `String(content)`, turning `undefined` into the
|
|
49
|
+
// four-character text "undefined", which matches nothing and returns null —
|
|
50
|
+
// and null here means "the policy line is absent", a finding about the user's
|
|
51
|
+
// file caused by a bug in ours. Trading a crash for a silent wrong answer is
|
|
52
|
+
// the exact direction this module exists to avoid, so the crash is kept.
|
|
53
|
+
if (typeof content !== 'string') {
|
|
54
|
+
throw new TypeError(`parseContextLine expects a string, got ${typeof content}`);
|
|
55
|
+
}
|
|
56
|
+
const match = visibleTextLines(content).join('\n').match(re);
|
|
57
|
+
if (!match) return null;
|
|
58
|
+
// Unescape Markdown-escaped pipes captured by the table-row patterns; the
|
|
59
|
+
// backtick context lines never contain `\|`, so this is a no-op for them.
|
|
60
|
+
const value = match[1].replace(/\\\|/g, '|').trim();
|
|
61
|
+
return value === '' ? null : value;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
// CommonMark's line-ending definition, and the reason it is a named constant:
|
|
65
|
+
// every splitter in this file must use the SAME one. `/\r?\n/` handled LF and
|
|
66
|
+
// CRLF but not a lone CR, while the sibling checks in `lib/init.js` match with
|
|
67
|
+
// the `m` flag — and JS's `m` treats a lone `\r` as a line terminator. So a
|
|
68
|
+
// CR-only file was one giant line to this module and a normal file to them, and
|
|
69
|
+
// the two reported different things about it (`p082-b3-g1` finding 3).
|
|
70
|
+
const LINE_SPLIT_RE = /\r\n|\r|\n/;
|
|
71
|
+
|
|
72
|
+
// A thematic break: three or more `-`, `_` or `*`, optionally separated by
|
|
73
|
+
// spaces or tabs.
|
|
74
|
+
//
|
|
75
|
+
// ⚠ `-` is a thematic break at 3+ and a setext underline at ANY length, so the
|
|
76
|
+
// two roles overlap and callers must not treat this as "is an underline".
|
|
77
|
+
// `=` is NOT here on purpose: a lone `=+` run is not a thematic break at all —
|
|
78
|
+
// at a block start it is ordinary paragraph text, which is exactly the
|
|
79
|
+
// asymmetry the old combined `(=+|-+)` pattern erased.
|
|
80
|
+
//
|
|
81
|
+
// ⚠ WHICH of the two roles wins is not decided here, and that is the shape of
|
|
82
|
+
// the whole rewrite. `classifyLines` decides it from block context, in one
|
|
83
|
+
// place: when a document-level paragraph is open the setext reading wins
|
|
84
|
+
// (CommonMark 4.3); otherwise this one does. Two predicates each answering that
|
|
85
|
+
// question from a line's own shape is what produced this module's defect class.
|
|
86
|
+
//
|
|
87
|
+
// An ATX heading's opening sequence is 1-6 hashes followed by a space or tab.
|
|
88
|
+
// `[ \t]` and NOT `\s`: CommonMark accepts only a space or tab there, while JS
|
|
89
|
+
// `\s` also matches U+00A0, form feed and vertical tab. A line sitting in that
|
|
90
|
+
// gap was neither a paragraph nor a heading, so a following `---` could not
|
|
91
|
+
// close the section and it ran on — silent pass (`p082-b3-g3`, `p082-b3-g4`).
|
|
92
|
+
// U+00A0 is the realistic input: it arrives by copy-paste from a web page or a
|
|
93
|
+
// word processor.
|
|
94
|
+
//
|
|
95
|
+
// ⚠ There is no literal U+00A0 anywhere in this file. The paragraph above used
|
|
96
|
+
// to demonstrate the defect with a real one, which is indistinguishable from a
|
|
97
|
+
// space in a diff and which an editor may silently normalize away.
|
|
98
|
+
// `test/upgrade-drift.mjs` builds these characters with `String.fromCharCode`
|
|
99
|
+
// for precisely that reason, and this file had been ignoring its own advice.
|
|
100
|
+
// `test/upgrade-drift.mjs` now asserts the absence rather than trusting this
|
|
101
|
+
// sentence.
|
|
102
|
+
//
|
|
103
|
+
// ⚠ There is exactly ONE ATX expression in this module now, and that is a
|
|
104
|
+
// deliberate outcome of the debt-12 rewrite. There used to be FOUR — this
|
|
105
|
+
// constant plus a full-line copy inside `headingAt`, `extractHeadings` and
|
|
106
|
+
// `extractH2Headings` — and three separate review rounds found them disagreeing
|
|
107
|
+
// with each other. A shared constant that only two of the four copies read is
|
|
108
|
+
// not a single source of truth.
|
|
109
|
+
const ATX_HEADING_RE = /^ {0,3}(#{1,6})([ \t].*)?$/;
|
|
110
|
+
|
|
111
|
+
// The `{ level, text }` of an ATX heading line, or null.
|
|
112
|
+
//
|
|
113
|
+
// ⚠ The closing sequence is stripped in a SECOND step rather than by an
|
|
114
|
+
// optional group in the same expression, and that is a correctness fix, not a
|
|
115
|
+
// style preference. The old one-expression form
|
|
116
|
+
// `(?:[ \t]+(.*?))?(?:[ \t]+#+)?[ \t]*$` gave `## #` the text `#`: the lazy
|
|
117
|
+
// capture backtracks into the closing sequence, so the group meant to absorb it
|
|
118
|
+
// never gets the chance. CommonMark says that heading's content is EMPTY. Two
|
|
119
|
+
// steps cannot express the ambiguity, so they simply get it right — while still
|
|
120
|
+
// keeping `## Git Policy#` as the text `Git Policy#`, because the closing
|
|
121
|
+
// sequence must be PRECEDED by whitespace (`p082-b3-g2` finding 2 paid for
|
|
122
|
+
// that, and `test/upgrade-drift.mjs` pins both).
|
|
123
|
+
//
|
|
124
|
+
// An EMPTY heading (`#`, `## `, `## #`) is a real heading and really does end a
|
|
125
|
+
// section, so it is reported as one. `headingResolves` must go on refusing to
|
|
126
|
+
// resolve a reference against it — `''` prefix-matches every reference, which
|
|
127
|
+
// silently satisfied every cross-ref in the guide (`p082-b3-g5` finding 1).
|
|
128
|
+
function parseAtxHeading(line) {
|
|
129
|
+
const m = line.match(ATX_HEADING_RE);
|
|
130
|
+
if (!m) return null;
|
|
131
|
+
const rest = (m[2] || '').replace(/[ \t]+#+[ \t]*$/, '');
|
|
132
|
+
// Through `stripSpaceTab`, not a second copy of the same expression: this used
|
|
133
|
+
// to spell the rule inline while the setext branch used `.trim()`, and the two
|
|
134
|
+
// heading kinds then disagreed about a trailing U+00A0.
|
|
135
|
+
return { level: m[1].length, text: stripSpaceTab(rest), setext: false };
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function isThematicBreak(line) {
|
|
139
|
+
return /^ {0,3}(-[ \t]*){3,}$/.test(line)
|
|
140
|
+
|| /^ {0,3}(\*[ \t]*){3,}$/.test(line)
|
|
141
|
+
|| /^ {0,3}(_[ \t]*){3,}$/.test(line);
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// A fenced-code marker: three or more backticks, or three or more tildes.
|
|
145
|
+
const FENCE_MARKER = '(```+|~~~+)';
|
|
146
|
+
const FENCE_OPEN_RE = new RegExp(`^ {0,3}${FENCE_MARKER}(.*)$`);
|
|
147
|
+
const FENCE_CLOSE_RE = new RegExp(`^ {0,3}${FENCE_MARKER}[ \\t]*$`);
|
|
148
|
+
|
|
149
|
+
// Blank out fenced code blocks (``` / ~~~) line-by-line so headings and § refs
|
|
150
|
+
// inside examples are never extracted. Line positions are preserved. CommonMark
|
|
151
|
+
// close rules (PROPOSAL-076 gates G4/G6): a block closes only on the same fence
|
|
152
|
+
// character repeated at least the opening length, indented at most three
|
|
153
|
+
// spaces, with nothing but whitespace after — so a three-backtick line inside a
|
|
154
|
+
// four-backtick example, or an info-string line like ```js inside an open
|
|
155
|
+
// fence, is content and does not end the block early.
|
|
156
|
+
//
|
|
157
|
+
// This is the module's ONLY line splitter; everything downstream works on its
|
|
158
|
+
// output, which is what makes `LINE_SPLIT_RE` a single point of control.
|
|
159
|
+
//
|
|
160
|
+
// ⚠ The fence marker is spelled ONCE, in `FENCE_MARKER`. The open and close
|
|
161
|
+
// tests are genuinely different rules — a close must have nothing after it,
|
|
162
|
+
// while an open takes an info string — but they must agree about what a fence
|
|
163
|
+
// LOOKS like, and two literals cannot be relied on to. Same reason as
|
|
164
|
+
// `ATX_HEADING_RE` and `BULLET_MARKER`.
|
|
165
|
+
function blankFencedBlocks(content) {
|
|
166
|
+
const lines = content.split(LINE_SPLIT_RE);
|
|
167
|
+
const out = [];
|
|
168
|
+
let fenceChar = null;
|
|
169
|
+
let fenceLen = 0;
|
|
170
|
+
for (const line of lines) {
|
|
171
|
+
if (fenceChar) {
|
|
172
|
+
out.push('');
|
|
173
|
+
const close = line.match(FENCE_CLOSE_RE);
|
|
174
|
+
if (close && close[1][0] === fenceChar && close[1].length >= fenceLen) {
|
|
175
|
+
fenceChar = null;
|
|
176
|
+
fenceLen = 0;
|
|
177
|
+
}
|
|
178
|
+
continue;
|
|
179
|
+
}
|
|
180
|
+
// CommonMark: at most three SPACES of indent (a tab is four columns, so a
|
|
181
|
+
// tab-indented fence is indented code, not a fence), and a BACKTICK fence's
|
|
182
|
+
// info string may not contain a backtick. Accepting either opened a
|
|
183
|
+
// pseudo-fence that blanked real headings, so a stale section ran on past
|
|
184
|
+
// its own end - silent pass (`p082-b3-g4` finding 2).
|
|
185
|
+
const open = line.match(FENCE_OPEN_RE);
|
|
186
|
+
if (open && open[1][0] === '`' && open[2].includes('`')) {
|
|
187
|
+
out.push(line);
|
|
188
|
+
continue;
|
|
189
|
+
}
|
|
190
|
+
if (open) {
|
|
191
|
+
fenceChar = open[1][0];
|
|
192
|
+
fenceLen = open[1].length;
|
|
193
|
+
out.push('');
|
|
194
|
+
continue;
|
|
195
|
+
}
|
|
196
|
+
out.push(line);
|
|
197
|
+
}
|
|
198
|
+
return out;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
// All Markdown heading texts (any level), fence-aware.
|
|
202
|
+
//
|
|
203
|
+
// ⚠ These two used to carry their own full-line ATX expression — two of the four
|
|
204
|
+
// copies that disagreed with each other. They read the classification now, which
|
|
205
|
+
// means three behaviours arrived here for free rather than being argued about
|
|
206
|
+
// one at a time:
|
|
207
|
+
// - the 0-3 space indent, which was first left out here on the argument that
|
|
208
|
+
// packaged inputs are generated and never indented, and which measuring
|
|
209
|
+
// against the reference retired;
|
|
210
|
+
// - SETEXT headings, which neither of these ever saw. A `§ Heading` reference
|
|
211
|
+
// to one was reported dangling and a template section written that way
|
|
212
|
+
// counted as missing — both false positives, both invisible to the pins,
|
|
213
|
+
// because the pins only ever fed these functions ATX;
|
|
214
|
+
// - the empty-heading and closing-sequence rules, which now cannot drift from
|
|
215
|
+
// what the section walker uses, because there is nothing left to drift from.
|
|
216
|
+
function extractHeadings(content) {
|
|
217
|
+
const lines = blankFencedBlocks(content);
|
|
218
|
+
return classifyLines(lines).filter((c) => c.heading).map((c) => c.heading.text);
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
// H2 heading texts only, fence-aware — the section shape of a template.
|
|
222
|
+
// Level 2 covers both `## X` and a `---`-underlined setext heading.
|
|
223
|
+
function extractH2Headings(content) {
|
|
224
|
+
const lines = blankFencedBlocks(content);
|
|
225
|
+
return classifyLines(lines).filter((c) => c.heading && c.heading.level === 2).map((c) => c.heading.text);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
// `<file>.md § Heading` references whose file basename is `targetBasename`.
|
|
230
|
+
// A soft-wrapped heading name is captured by joining the next line; a match
|
|
231
|
+
// anchored beyond the current line is left to that line's own iteration.
|
|
232
|
+
function extractSectionRefs(content, targetBasename) {
|
|
233
|
+
// ⚠ `visibleTextLines`, not `blankFencedBlocks`: a `<file>.md § Heading`
|
|
234
|
+
// written inside an HTML comment was reported as a dangling reference, and the
|
|
235
|
+
// captured heading text carried the trailing `-->` into the message.
|
|
236
|
+
const lines = visibleTextLines(content);
|
|
237
|
+
const refs = [];
|
|
238
|
+
lines.forEach((line, i) => {
|
|
239
|
+
const firstLen = line.replace(/\s+$/, '').length;
|
|
240
|
+
const joined = line.replace(/\s+$/, '') + ' ' + (lines[i + 1] || '').replace(/^\s+/, '');
|
|
241
|
+
for (const m of joined.matchAll(/`?([A-Za-z0-9._/-]+\.md)`?\s*§\s*"?([^."`)\n]+)/g)) {
|
|
242
|
+
if (m.index > firstLen) continue;
|
|
243
|
+
const base = m[1].split('/').pop();
|
|
244
|
+
if (base !== targetBasename) continue;
|
|
245
|
+
refs.push({ line: i + 1, headingText: m[2].trim() });
|
|
246
|
+
}
|
|
247
|
+
});
|
|
248
|
+
return refs;
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
// ⚠⚠ `.trim()` HERE IS DELIBERATE, AND IT IS NOT THE `.trim()` DEFECT CLASS.
|
|
252
|
+
// This was briefly "fixed" to `stripSpaceTab` on the reasoning that one rule
|
|
253
|
+
// should have one spelling, and that broke real files — the measurement is in
|
|
254
|
+
// the pins below `normalizeHeading` in `test/upgrade-drift.mjs`. Two different
|
|
255
|
+
// questions were being read as one:
|
|
256
|
+
//
|
|
257
|
+
// `stripSpaceTab` answers **what the heading text IS** — block structure,
|
|
258
|
+
// spec-following, deliberately strict, because a U+00A0 is not whitespace to
|
|
259
|
+
// CommonMark and a section boundary must not move because of one.
|
|
260
|
+
//
|
|
261
|
+
// `normalizeHeading` answers **whether two heading strings NAME the same
|
|
262
|
+
// section** — an identity comparison over text a human retyped or pasted. It
|
|
263
|
+
// is tolerant BY CONSTRUCTION and already does three things CommonMark would
|
|
264
|
+
// never do: it strips ``, `*` and `_`, and (through `headingKey`) it folds
|
|
265
|
+
// case. Tightening it to the block-structure rule aligned the wrong function.
|
|
266
|
+
//
|
|
267
|
+
// What tightening it actually did: `## Ceremony Scaling (Project Application)`
|
|
268
|
+
// with a copy-pasted trailing U+00A0 stopped matching the fingerprint, so doctor
|
|
269
|
+
// reported `missing` on a file that carries the rule — and `missingTemplateSections`
|
|
270
|
+
// reported a section absent that was present. Loud, not silent, but wrong, and
|
|
271
|
+
// `CONVENTIONS_FINGERPRINTS` already records this exact false `missing` happening
|
|
272
|
+
// once before, for case rather than for whitespace. Reproduces for U+00A0, form
|
|
273
|
+
// feed, vertical tab, U+2002 and U+FEFF, leading and trailing.
|
|
274
|
+
//
|
|
275
|
+
// ⚠ `extractSectionRefs` also calls `.trim()` on the *reference* side of the
|
|
276
|
+
// `headingResolves` comparison. That is the same rule on both sides, which is
|
|
277
|
+
// the property to keep: tightening one side only is what made `§ Workflow_`
|
|
278
|
+
// stop resolving.
|
|
279
|
+
function normalizeHeading(text) {
|
|
280
|
+
return text.replace(/[`*_]/g, '').trim();
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
// Tolerant heading resolution: a reference resolves when it prefix-matches a
|
|
284
|
+
// real heading in either direction (covers soft wraps like "§ Workflow" for
|
|
285
|
+
// "Workflow Transparency" and shorthand references).
|
|
286
|
+
// ⚠ `headingKey`, not `normalizeHeading` — the difference is case folding, and
|
|
287
|
+
// this consumer used to be the odd one out. `CONVENTIONS_FINGERPRINTS` states
|
|
288
|
+
// the rule ("a heading an adopter retyped in different case is the same
|
|
289
|
+
// section") and `conventionsSectionBodies` implemented it, while this function
|
|
290
|
+
// and `missingTemplateSections` did not: `## filling the templates` was found by
|
|
291
|
+
// one and reported missing by the other, from the same document. One normaliser
|
|
292
|
+
// with three consumers, two of which skipped half of it.
|
|
293
|
+
function headingResolves(referenceText, headings) {
|
|
294
|
+
const want = headingKey(referenceText);
|
|
295
|
+
if (!want) return true;
|
|
296
|
+
return headings.some((heading) => {
|
|
297
|
+
const have = headingKey(heading);
|
|
298
|
+
// An EMPTY heading resolves nothing. Making empty ATX headings real (so they
|
|
299
|
+
// close a section, which CommonMark says they do) put `''` into this list,
|
|
300
|
+
// and `want.startsWith('')` is true for every reference — so one stray `## `
|
|
301
|
+
// in a guide silently satisfied every `<file> § Heading` cross-reference and
|
|
302
|
+
// doctor stopped reporting dangling ones (`p082-b3-g5` finding 1). A fix in
|
|
303
|
+
// one consumer of a shared list has to be checked against the others.
|
|
304
|
+
if (!have) return false;
|
|
305
|
+
return have.startsWith(want) || want.startsWith(have);
|
|
306
|
+
});
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
// Which of the template's H2 sections are missing from a filled document — the
|
|
310
|
+
// "created from an older template shape" signal.
|
|
311
|
+
function missingTemplateSections(templateContent, documentContent) {
|
|
312
|
+
// ⚠ `headingKey`, not `normalizeHeading`: same case-folding rule as
|
|
313
|
+
// `headingResolves` above and `conventionsSectionBodies` below. This one is
|
|
314
|
+
// user-facing — a case difference here produced doctor's "looks like an older
|
|
315
|
+
// `_index.md` template shape" on a document that has the section.
|
|
316
|
+
//
|
|
317
|
+
// ⚠⚠ A COUNTED multiset, not a `Set`. With a `Set`, a template carrying two
|
|
318
|
+
// H2s that fold to the same key — `## Filling the Templates` and
|
|
319
|
+
// `## FILLING THE TEMPLATES` — was fully satisfied by ONE of them appearing in
|
|
320
|
+
// the document, so a genuinely missing section went unreported. Silent, and it
|
|
321
|
+
// arrived with the case folding: before folding, the two keys were distinct.
|
|
322
|
+
// The packaged `_index.md` has no colliding pair today, so this is latent —
|
|
323
|
+
// which is exactly why it needs to be code and not a comment.
|
|
324
|
+
const have = new Map();
|
|
325
|
+
for (const heading of extractH2Headings(documentContent)) {
|
|
326
|
+
const key = headingKey(heading);
|
|
327
|
+
have.set(key, (have.get(key) || 0) + 1);
|
|
328
|
+
}
|
|
329
|
+
return extractH2Headings(templateContent).filter((heading) => {
|
|
330
|
+
const key = headingKey(heading);
|
|
331
|
+
const left = have.get(key) || 0;
|
|
332
|
+
if (left === 0) return true;
|
|
333
|
+
have.set(key, left - 1);
|
|
334
|
+
return false;
|
|
335
|
+
});
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
// True when `content` matches `templateContent` up to placeholder substitution:
|
|
339
|
+
// every single-line `{...}` placeholder in the template may match any text.
|
|
340
|
+
// Distinguishes "pristine current starter" from "edited or older starter"
|
|
341
|
+
// without knowing the values init substituted.
|
|
342
|
+
function matchesTemplateWithPlaceholders(content, templateContent) {
|
|
343
|
+
// Split/join through the module's single line-ending definition rather than
|
|
344
|
+
// a second hand-rolled CRLF replacement: a lone CR left the template and the
|
|
345
|
+
// document normalized differently, so a pristine starter compared as edited.
|
|
346
|
+
const normalize = (s) => String(s).split(LINE_SPLIT_RE).join('\n').replace(/\s+$/, '');
|
|
347
|
+
const parts = normalize(templateContent).split(/\{[^}\n]*\}/);
|
|
348
|
+
const pattern = `^${parts.map(escapeRegExp).join('[\\s\\S]*?')}$`;
|
|
349
|
+
try {
|
|
350
|
+
return new RegExp(pattern).test(normalize(content));
|
|
351
|
+
} catch {
|
|
352
|
+
return false;
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
function escapeRegExp(s) {
|
|
357
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
// PROPOSAL-078 phase 1: the table-cell formatting convention (P-072) ships
|
|
361
|
+
// as an HTML comment at every template head, so docs seeded before 0.13 —
|
|
362
|
+
// or written from scratch — hold tables with no in-file reminder and tend
|
|
363
|
+
// to grow run-on cell walls. Detection is info-only; doctor never edits
|
|
364
|
+
// user-authored specs. The snippet is matched anywhere in the file (the
|
|
365
|
+
// _index.md template places it mid-file, after its usage comment), and a
|
|
366
|
+
// table counts only outside code fences. The delimiter-row regex accepts
|
|
367
|
+
// any row with an internal pipe (`|---|---|`, `---|---`) plus the
|
|
368
|
+
// single-column form with BOTH outer pipes (`|---|`); a bare `---` is a
|
|
369
|
+
// thematic break / frontmatter fence and never matches — the one shape
|
|
370
|
+
// deliberately missed is a single-column delimiter written without outer
|
|
371
|
+
// pipes, which is indistinguishable from a thematic break anyway (the
|
|
372
|
+
// check is a nudge, not a gate).
|
|
373
|
+
const SPEC_FORMATTING_CONVENTION_SNIPPET = 'Formatting convention: keep table cells concise';
|
|
374
|
+
const TABLE_DELIMITER_ROW_RE = /^ {0,3}(?:\|?[ \t]*:?-{2,}:?[ \t]*(?:\|[ \t]*:?-{2,}:?[ \t]*)+\|?|\|[ \t]*:?-{2,}:?[ \t]*\|)[ \t]*$/;
|
|
375
|
+
|
|
376
|
+
// ⚠ This asks the classification for a table rather than grepping for a
|
|
377
|
+
// delimiter row, so it now agrees with the section walker about what a table is.
|
|
378
|
+
// The two used to answer differently: a delimiter row with no header line above
|
|
379
|
+
// it counted as a table HERE while the walker treated it as a thematic break or
|
|
380
|
+
// prose. Neither shape occurs in a packaged template — `test/upgrade-drift.mjs`
|
|
381
|
+
// pins the packaged verdicts — but "two places in this file decide the same
|
|
382
|
+
// question with different code" is the defect class, and it does not stop being
|
|
383
|
+
// the defect class because the current inputs happen not to hit it.
|
|
384
|
+
function hasTableWithoutConventionComment(content) {
|
|
385
|
+
if (String(content).includes(SPEC_FORMATTING_CONVENTION_SNIPPET)) {
|
|
386
|
+
return false;
|
|
387
|
+
}
|
|
388
|
+
const lines = blankFencedBlocks(String(content));
|
|
389
|
+
return classifyLines(lines).some((c) => c.type === 'table');
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
// --- PROPOSAL-082 G5 / PROPOSAL-083 §4: `_conventions.md` staleness -----------
|
|
393
|
+
//
|
|
394
|
+
// `_conventions.md` is user-owned. `dflow init` writes it once and neither init
|
|
395
|
+
// nor `configure-agents` ever rewrites it, so a project keeps the shape it was
|
|
396
|
+
// created with forever. Doctor reporting is the ONLY way an adopter learns their
|
|
397
|
+
// copy predates a contract change — which is why this reports and never edits.
|
|
398
|
+
//
|
|
399
|
+
// Every fingerprint is SECTION-LOCAL. A whole-file grep would pass the moment
|
|
400
|
+
// the phrase appeared anywhere in the document, including inside a section the
|
|
401
|
+
// adopter pasted in from somewhere else; the claim being checked is "THIS
|
|
402
|
+
// section carries this rule", so that is what gets read.
|
|
403
|
+
//
|
|
404
|
+
// Three states, and the middle one is the whole point:
|
|
405
|
+
//
|
|
406
|
+
// missing — the section is not in the file at all. Reported, never silently
|
|
407
|
+
// passed: MAINTAINERS § Cross-Model Review (P083 batch 2, g16) —
|
|
408
|
+
// "a gate that silently accepts absence is worse than no gate: it
|
|
409
|
+
// reports success". The previous attempt at this check was pulled
|
|
410
|
+
// for exactly that, because against the then-broken projection it
|
|
411
|
+
// could only ever report success.
|
|
412
|
+
// stale — section present, rule absent. This is the actionable one: the
|
|
413
|
+
// project's own conventions state something the shipped flows no
|
|
414
|
+
// longer do.
|
|
415
|
+
// current — section present, rule present.
|
|
416
|
+
// Heading detection has to cover more than `^#{1,6} `, because every shape it
|
|
417
|
+
// misses becomes a section that never ends — and an over-running body reports
|
|
418
|
+
// the WRONG section as current, which is the silent-success direction:
|
|
419
|
+
// - ATX indented 1-3 spaces is still a heading (CommonMark).
|
|
420
|
+
// - A setext heading (text line + `===` / `---` underline) is a heading, and
|
|
421
|
+
// missing it let a body run on into every following section. `---` is only
|
|
422
|
+
// setext when it directly follows a non-blank, non-heading line; otherwise
|
|
423
|
+
// it is a thematic break, which `_conventions.md` uses.
|
|
424
|
+
// Comparison is case-insensitive: a heading an adopter retyped in different
|
|
425
|
+
// case is the same section, and treating it as absent produced a false
|
|
426
|
+
// `missing` on a file that had the rule.
|
|
427
|
+
// A setext underline may only close a PARAGRAPH (CommonMark 4.3). After a list
|
|
428
|
+
// item, a table row, a blockquote or another heading it is not a heading at all
|
|
429
|
+
// — `- foo` / `---` is a list plus a thematic break. Treating those as headings
|
|
430
|
+
// ended sections early and, worse, made the caller pop a real content line out
|
|
431
|
+
// of the body: a marker sitting on the last list item or table row vanished and
|
|
432
|
+
// the section reported `stale`. That is the same false-warn-on-an-ordinary-edit
|
|
433
|
+
// class the soft-wrap fix removed.
|
|
434
|
+
//
|
|
435
|
+
// ⚠ A `---` directly after a plain paragraph line IS a setext heading, and the
|
|
436
|
+
// section really does end there — that is CommonMark, not a quirk of this
|
|
437
|
+
// parser. An adopter who wants a horizontal rule needs a blank line before it.
|
|
438
|
+
// ---------------------------------------------------------------------------
|
|
439
|
+
// THE BLOCK CLASSIFICATION PASS — PROPOSAL-082/083 debt 12.
|
|
440
|
+
//
|
|
441
|
+
// WHAT THIS REPLACED, AND WHY IT IS NOT A FIFTH PATCH. Five predicates used to
|
|
442
|
+
// live here — `isTableLine`, `closesOwnBlock`, `blockStartIndex`,
|
|
443
|
+
// `isParagraphLine` and `headingAt` — and each re-derived Markdown block
|
|
444
|
+
// structure from one line's shape, using backward scans and mutual recursion in
|
|
445
|
+
// place of the state a block parser keeps. But what a line IS depends on which
|
|
446
|
+
// block it sits inside, which a per-line predicate cannot see, so the five
|
|
447
|
+
// disagreed with one another. Six consecutive review rounds each found one more
|
|
448
|
+
// place where they did, and THREE of those rounds found a disagreement that the
|
|
449
|
+
// previous round's fix had introduced. Every widening of the arbiter's shape set
|
|
450
|
+
// found more: 30 -> 51 -> 57 -> 73 -> 79 shapes, each stage having measured
|
|
451
|
+
// clean at the stage before.
|
|
452
|
+
//
|
|
453
|
+
// So: one forward pass labels every line exactly once, and the former predicates
|
|
454
|
+
// become lookups into that label. CommonMark block parsing IS a forward pass;
|
|
455
|
+
// the backward scans existed only to reconstruct state that had been thrown
|
|
456
|
+
// away. A consequence worth having — the old `isParagraphLine` /
|
|
457
|
+
// `closesOwnBlock` mutual recursion was explicitly NOT linear (its own comment
|
|
458
|
+
// said so, and warned callers off large documents). This is O(lines).
|
|
459
|
+
//
|
|
460
|
+
// ⚠ THE RULE THAT REPLACES "do not add a fifth predicate": exactly one function
|
|
461
|
+
// decides block structure and this is it. Callers read labels. If you find
|
|
462
|
+
// yourself writing a regex against a raw line ANYWHERE else in order to decide
|
|
463
|
+
// what kind of line it is, that is this defect class growing back, and
|
|
464
|
+
// `test/upgrade-drift.mjs`'s class guard is meant to fail on it.
|
|
465
|
+
//
|
|
466
|
+
// ⚠ WHAT THIS DELIBERATELY DOES NOT IMPLEMENT — stated so the gaps are known
|
|
467
|
+
// rather than discovered by the next round:
|
|
468
|
+
// - HTML block type 7 (a complete tag whose name is not in the type-6 list,
|
|
469
|
+
// alone on a line). It is the only type that cannot interrupt a paragraph
|
|
470
|
+
// and recognizing it needs a real tag parser.
|
|
471
|
+
// `commonmark-differential.mjs` carries a row for it, so the divergence is
|
|
472
|
+
// MEASURED rather than assumed harmless.
|
|
473
|
+
// - GFM's rule that a delimiter row must have the same cell count as its
|
|
474
|
+
// header row. `test/upgrade-drift.mjs` pins `| A |` + `|---|---|` as a
|
|
475
|
+
// table, which is the looser reading, and tightening it would also move the
|
|
476
|
+
// P-078 formatting-convention check.
|
|
477
|
+
// - Nested container CONTENT. A blockquote or list item's interior is not
|
|
478
|
+
// parsed as its own block sequence. Knowing the document-level block is not
|
|
479
|
+
// a paragraph is enough for every question this module asks, and
|
|
480
|
+
// `_conventions.md` is a small hand-written file.
|
|
481
|
+
// - ⚠⚠ INLINE HTML COMMENTS — the one gap here whose direction is SILENT PASS,
|
|
482
|
+
// so it is the one that matters. This pass is LINE-level: a line that STARTS
|
|
483
|
+
// with `<!--` opens an HTML block and its text is correctly treated as unseen
|
|
484
|
+
// by a reader. A comment that begins part-way through a line is not a block
|
|
485
|
+
// at all — it is an inline span — and every consumer therefore counts its
|
|
486
|
+
// contents as live document text. Measured, with controls, on 2026-08-05:
|
|
487
|
+
//
|
|
488
|
+
// `prose <!-- cascade result is a floor -->` -> doctor reports `current`
|
|
489
|
+
// `- item <!-- ... -->` / `> quote <!-- ... -->` -> same
|
|
490
|
+
// `<!-- a --> <!-- cascade result is a floor -->` -> same (only the FIRST
|
|
491
|
+
// span is masked; the tail is never re-scanned)
|
|
492
|
+
//
|
|
493
|
+
// So a rule a user commented out mid-line still reads as present, which is
|
|
494
|
+
// the exact question `doctor` exists to answer. PRE-EXISTING — it was true
|
|
495
|
+
// before the visibility rule was added and the rule did not touch it; what
|
|
496
|
+
// the rule closed was the line-level half.
|
|
497
|
+
//
|
|
498
|
+
// ⚠ NOT fixed here on purpose (maintainer decision, 2026-08-05). Hiding
|
|
499
|
+
// spans needs an inline layer in a module that is deliberately line-level,
|
|
500
|
+
// and the obvious version is WRONG in the content-loss direction: a comment
|
|
501
|
+
// inside a CODE SPAN — `` `<!-- example -->` `` — is rendered and a reader
|
|
502
|
+
// DOES see it, so "mask everything between the delimiters" deletes visible
|
|
503
|
+
// content. That is a span tokenizer, with generated cases, as its own piece
|
|
504
|
+
// of work — not a fourth patch on a boundary predicate, which is how this
|
|
505
|
+
// module earned every defect it has. `p082-b3-k4` recommended exactly that.
|
|
506
|
+
// PROPOSAL-084 carries it as a detectable shape in the meantime: "this file
|
|
507
|
+
// contains an inline HTML comment" needs none of the boundary logic that
|
|
508
|
+
// could be wrong, which is what makes it warnable.
|
|
509
|
+
// ---------------------------------------------------------------------------
|
|
510
|
+
|
|
511
|
+
// HTML blocks (CommonMark 4.6). Types 1-5 pair a start with an end condition and
|
|
512
|
+
// close ON the line matching it — that line belongs to the block. Type 6 closes
|
|
513
|
+
// at the next blank line, which does NOT belong to it.
|
|
514
|
+
//
|
|
515
|
+
// ⚠ Type 2 is the one that forced this rewrite. The old code opened an HTML
|
|
516
|
+
// block on any `<!--` line but only ever closed a SINGLE-LINE `<!-- ... -->`, so
|
|
517
|
+
// a comment spanning two lines never closed: the section walker stopped inside
|
|
518
|
+
// it, dropped the list item that followed, and `findConventionsDrift` reported
|
|
519
|
+
// `ceremony-escalate-only:stale` against a file that was current — the
|
|
520
|
+
// content-loss direction (`p082-b3-g5` finding 4). It could not be patched
|
|
521
|
+
// per-line, because an HTML block's extent is cross-line state. That is the
|
|
522
|
+
// whole argument for this pass in one example.
|
|
523
|
+
// ⚠ NAMED because PROPOSAL-084's inline-comment detector also needs to know what
|
|
524
|
+
// a comment-block OPENER looks like, and the first draft of it hand-wrote
|
|
525
|
+
// `/^ {0,3}<!--/` a second time — a fresh copy of a block rule living outside the
|
|
526
|
+
// one place allowed to spell it, which is this module's whole recurring defect.
|
|
527
|
+
// One constant, two readers.
|
|
528
|
+
const HTML_COMMENT_OPEN_RE = /^ {0,3}<!--/;
|
|
529
|
+
|
|
530
|
+
// ⚠ The type-1 tag names, spelled ONCE. `HTML_BLOCK_TYPES` composes them into the
|
|
531
|
+
// start condition, `htmlBlockStart` derives the matching end tag from that capture,
|
|
532
|
+
// and `htmlBlockType7Line` needs the same list
|
|
533
|
+
// because CommonMark §4.6 EXCLUDES these four names from type 7 — type 1 wins.
|
|
534
|
+
// Naming it out of `HTML_BLOCK_TYPES` is the move `HTML_COMMENT_OPEN_RE` already
|
|
535
|
+
// made, for the same reason: the alternative is a third copy of the list, and the
|
|
536
|
+
// list being written twice inside this very constant is what let the type-7
|
|
537
|
+
// detector fire on `<pre/>` for a round (`p084-xv8` finding 4).
|
|
538
|
+
const HTML_TYPE1_TAGS = 'script|pre|style|textarea';
|
|
539
|
+
const HTML_TYPE1_NAME_RE = new RegExp(`^(?:${HTML_TYPE1_TAGS})$`, 'i');
|
|
540
|
+
|
|
541
|
+
const HTML_BLOCK_TYPES = [
|
|
542
|
+
[new RegExp(`^ {0,3}<(${HTML_TYPE1_TAGS})([ \\t>]|$)`, 'i'), null],
|
|
543
|
+
[HTML_COMMENT_OPEN_RE, /-->/],
|
|
544
|
+
[/^ {0,3}<\?/, /\?>/],
|
|
545
|
+
[/^ {0,3}<![A-Za-z]/, />/],
|
|
546
|
+
[/^ {0,3}<!\[CDATA\[/, /\]\]>/]
|
|
547
|
+
];
|
|
548
|
+
|
|
549
|
+
// Type 6's tag list, verbatim from the spec. Order inside the alternation does
|
|
550
|
+
// not matter because the tail anchors on a delimiter, so `p` cannot swallow the
|
|
551
|
+
// `param` case.
|
|
552
|
+
const HTML_BLOCK_TAGS = 'address|article|aside|base|basefont|blockquote|body|caption|center|col|colgroup|dd|details|dialog|dir|div|dl|dt|fieldset|figcaption|figure|footer|form|frame|frameset|h1|h2|h3|h4|h5|h6|head|header|hr|html|iframe|legend|li|link|main|menu|menuitem|nav|noframes|ol|optgroup|option|p|param|search|section|summary|table|tbody|td|tfoot|th|thead|title|tr|track|ul';
|
|
553
|
+
const HTML_BLOCK_TYPE6_RE = new RegExp(`^ {0,3}</?(?:${HTML_BLOCK_TAGS})(?:[ \\t]|/?>|$)`, 'i');
|
|
554
|
+
|
|
555
|
+
// Leaf-block openers. Spelled `[ \t]` throughout, never `\s`, for the reason
|
|
556
|
+
// given at `ATX_HEADING_RE`.
|
|
557
|
+
//
|
|
558
|
+
// ⚠ `\d{1,9}`: CommonMark caps an ordered marker at nine digits, so a ten-digit
|
|
559
|
+
// run is paragraph text and NOT a list (`p082-b3-g5` finding 3).
|
|
560
|
+
// ⚠ `([ \t]|$)` and not `[ \t]`: a marker with nothing after it is an EMPTY list
|
|
561
|
+
// item, not prose (`p082-b3-g1` finding 6).
|
|
562
|
+
// ⚠ The two marker shapes are spelled ONCE and the three predicates are built
|
|
563
|
+
// from them. The first draft of this block wrote `[-*+]` twice and
|
|
564
|
+
// `\d{1,9}[.)]` twice across four separate literals, which is the module's own
|
|
565
|
+
// defect class — change the digit cap in one and the "is this an empty item"
|
|
566
|
+
// test silently keeps the old one. Same reason `ATX_HEADING_RE` is a constant.
|
|
567
|
+
const BULLET_MARKER = '[-*+]';
|
|
568
|
+
// The delimiter set is its own fragment because THREE expressions need it and
|
|
569
|
+
// one of them used to hand-write it: `INTERRUPTING_ITEM_RE` spelled `1[.)]`
|
|
570
|
+
// while claiming to be derived, so widening `ORDERED_MARKER` would have left the
|
|
571
|
+
// paragraph-interrupt rule silently on the old set. Unbroken at the time, and
|
|
572
|
+
// load-bearing — which is the only reason it is worth fixing before it breaks.
|
|
573
|
+
const ORDERED_DELIM = '[.)]';
|
|
574
|
+
const ORDERED_MARKER = `\\d{1,9}${ORDERED_DELIM}`;
|
|
575
|
+
// Opens a list wherever it appears, including an EMPTY item.
|
|
576
|
+
const LIST_ITEM_RE = new RegExp(`^ {0,3}(?:${BULLET_MARKER}|${ORDERED_MARKER})([ \\t]|$)`);
|
|
577
|
+
// An item with nothing after the marker. CommonMark 5.2: a list may interrupt a
|
|
578
|
+
// paragraph only if its first line is non-blank, so this shape may not.
|
|
579
|
+
const EMPTY_LIST_ITEM_RE = new RegExp(`^ {0,3}(?:${BULLET_MARKER}|${ORDERED_MARKER})[ \\t]*$`);
|
|
580
|
+
// The markers that CAN interrupt a paragraph: any bullet, or an ordered marker
|
|
581
|
+
// starting at 1. Any other ordered marker is paragraph text.
|
|
582
|
+
const INTERRUPTING_ITEM_RE = new RegExp(`^ {0,3}(?:${BULLET_MARKER}|1${ORDERED_DELIM})([ \\t]|$)`);
|
|
583
|
+
|
|
584
|
+
// A BLANK LINE — spaces and tabs only (CommonMark 2.1).
|
|
585
|
+
//
|
|
586
|
+
// ⚠⚠ THIS EXISTS BECAUSE `.trim()` IS THE GUARDED DEFECT CLASS IN A DIFFERENT
|
|
587
|
+
// SPELLING, and the guard was blind to it. `\s` was swept out of every block
|
|
588
|
+
// predicate; `line.trim()` was not, and `.trim()` strips the SAME set — U+00A0,
|
|
589
|
+
// form feed, vertical tab, U+2000-200B, U+FEFF — while containing no `\s` for
|
|
590
|
+
// the guard to find. A line holding only a non-breaking space was therefore a
|
|
591
|
+
// blank line here and a paragraph to CommonMark, so `para` / NBSP / `---` made
|
|
592
|
+
// `---` a thematic break instead of a setext underline, the section ran on, and
|
|
593
|
+
// a stale `_conventions.md` reported `current`. SILENT PASS, the worst
|
|
594
|
+
// direction, measured. Four more shapes flipped the other way.
|
|
595
|
+
//
|
|
596
|
+
// The guard now bans `.trim(` in block-structure code for this reason. Strip
|
|
597
|
+
// with `stripSpaceTab` where a value needs trimming.
|
|
598
|
+
const BLANK_LINE_RE = /^[ \t]*$/;
|
|
599
|
+
|
|
600
|
+
// Strip leading and trailing SPACES AND TABS only — the CommonMark rule for
|
|
601
|
+
// heading content, and the same rule `parseAtxHeading` applies. The setext
|
|
602
|
+
// branch used to call `.trim()` here, so the two heading kinds disagreed about
|
|
603
|
+
// a trailing U+00A0: `## Title<NBSP>` kept it and `Title<NBSP>` / `---` did not,
|
|
604
|
+
// which made `isRecognizableDflowGuide` reject one form and accept the other.
|
|
605
|
+
//
|
|
606
|
+
// ⚠ This DIVERGES from commonmark.js, which calls JS `.trim()` in its inline
|
|
607
|
+
// parser and so strips U+00A0 too. The spec says spaces and tabs; the reference
|
|
608
|
+
// implementation is broader. We follow the spec.
|
|
609
|
+
//
|
|
610
|
+
// ⚠⚠ BE PRECISE ABOUT WHICH HALF IS IN THIS REPO. An earlier version of this
|
|
611
|
+
// comment said the divergence "is carried as a named row in
|
|
612
|
+
// `atx-differential.mjs`" — a file that **does not exist anywhere under this
|
|
613
|
+
// repo**. It lives in the maintainer's repo-external review folder, so a reader
|
|
614
|
+
// following that sentence finds nothing, and nothing in the tree re-derives the
|
|
615
|
+
// claim. The divergence is pinned HERE instead, in `test/upgrade-drift.mjs`
|
|
616
|
+
// pin (k).
|
|
617
|
+
//
|
|
618
|
+
// ⚠⚠ AND BE PRECISE ABOUT WHAT (k) COVERS, because the sentence that replaced
|
|
619
|
+
// the bad one was itself an over-claim: it said the pins ran "against every
|
|
620
|
+
// character the class covers" while (k) pinned **U+00A0 trailing, alone**. Form
|
|
621
|
+
// feed and vertical tab were pinned at the ATX *opener*, not at heading-text
|
|
622
|
+
// trimming, and no consumer consequence was pinned at all. Two corrections in a
|
|
623
|
+
// row, both widening a claim past what the tree checks — so what (k) covers
|
|
624
|
+
// today is stated exactly: five characters (U+00A0, form feed, vertical tab,
|
|
625
|
+
// U+2002, U+FEFF), both leading and trailing, on both heading kinds. Widen the
|
|
626
|
+
// claim only by widening the loop.
|
|
627
|
+
function stripSpaceTab(text) {
|
|
628
|
+
return text.replace(/^[ \t]+|[ \t]+$/g, '');
|
|
629
|
+
}
|
|
630
|
+
// Split for measuring a list item's CONTENT COLUMN: indent, marker, gap.
|
|
631
|
+
const LIST_ITEM_PREFIX_RE = new RegExp(`^( {0,3})(${BULLET_MARKER}|${ORDERED_MARKER})([ \\t]*)`);
|
|
632
|
+
const BLOCKQUOTE_RE = /^ {0,3}>/;
|
|
633
|
+
// (`INDENTED_CODE_RE` was deleted rather than fixed. It was the only raw-line
|
|
634
|
+
// spelling of a question `indentWidth` already answers correctly, and keeping a
|
|
635
|
+
// corrected copy would have left two expressions for one rule — the shape this
|
|
636
|
+
// module keeps paying for.)
|
|
637
|
+
const SETEXT_UNDERLINE_RE = /^ {0,3}(=+|-+)[ \t]*$/;
|
|
638
|
+
|
|
639
|
+
// Columns of leading whitespace, a tab advancing to the next multiple of four
|
|
640
|
+
// (CommonMark 2.2). Spaces and tabs only — anything else ends the indent, which
|
|
641
|
+
// is the `[ \t]`-not-`\s` rule in yet another place.
|
|
642
|
+
function indentWidth(line) {
|
|
643
|
+
let col = 0;
|
|
644
|
+
for (const ch of line) {
|
|
645
|
+
if (ch === ' ') col += 1;
|
|
646
|
+
else if (ch === '\t') col += 4 - (col % 4);
|
|
647
|
+
else break;
|
|
648
|
+
}
|
|
649
|
+
return col;
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
// The column a list item's CONTENT starts at — the indent, the marker, and the
|
|
653
|
+
// gap after it. CommonMark 5.2: a gap of 1-4 spaces sets the content column
|
|
654
|
+
// directly; an empty item, or a gap of five or more (which would start indented
|
|
655
|
+
// code), puts the content one column after the marker.
|
|
656
|
+
//
|
|
657
|
+
// ⚠ This exists because a blank line does NOT end a list. A line indented to at
|
|
658
|
+
// least this column after a blank is still the item's content, and reading it as
|
|
659
|
+
// a document-level paragraph made `- a` / blank / ` b` / `---` end the section
|
|
660
|
+
// where CommonMark reads a thematic break — content loss, the pop deleting the
|
|
661
|
+
// line that carries the rule. Measured against the reference; the first version
|
|
662
|
+
// of this pass closed every block on a blank line and got it wrong.
|
|
663
|
+
function listContentColumn(line) {
|
|
664
|
+
const m = line.match(LIST_ITEM_PREFIX_RE);
|
|
665
|
+
if (!m) return null;
|
|
666
|
+
const markerEnd = m[1].length + m[2].length;
|
|
667
|
+
let gap = 0;
|
|
668
|
+
for (const ch of m[3]) gap += ch === '\t' ? 4 - ((markerEnd + gap) % 4) : 1;
|
|
669
|
+
return gap === 0 || gap > 4 ? markerEnd + 1 : markerEnd + gap;
|
|
670
|
+
}
|
|
671
|
+
|
|
672
|
+
// Convert the classifier's visual column into the UTF-16 offset String#slice
|
|
673
|
+
// expects. A tab advances to a four-column stop but occupies one code unit; using
|
|
674
|
+
// the visual column as an offset cuts into nested content after `-\t` / `- \t`.
|
|
675
|
+
function stringIndexAtVisualColumn(line, targetColumn) {
|
|
676
|
+
let column = 0;
|
|
677
|
+
let index = 0;
|
|
678
|
+
while (index < line.length && column < targetColumn) {
|
|
679
|
+
column += line[index] === '\t' ? 4 - (column % 4) : 1;
|
|
680
|
+
index += 1;
|
|
681
|
+
}
|
|
682
|
+
return index;
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
// `{ endRe }` when the line opens an HTML block, else null. `endRe: null` means
|
|
686
|
+
// type 6 — closed by a blank line rather than by a pattern.
|
|
687
|
+
// ⚠ `invisible` answers a DIFFERENT question from the rest of this function, and
|
|
688
|
+
// keeping them apart matters: `endRe` is about where the block ENDS (block
|
|
689
|
+
// structure), `invisible` is about whether a reader ever SEES the text inside it
|
|
690
|
+
// (content). The module got the first right and never asked the second, so a
|
|
691
|
+
// rule commented out with `<!-- … -->` counted as present and a stale file
|
|
692
|
+
// reported `current` — silent pass, reachable by one edit a person actually
|
|
693
|
+
// makes (`p082-b3-k1`).
|
|
694
|
+
//
|
|
695
|
+
// Which types a reader cannot see, and why each:
|
|
696
|
+
// type 2 `<!-- -->` a comment. Never rendered.
|
|
697
|
+
// type 3 `<? ?>` a processing instruction. Never rendered.
|
|
698
|
+
// type 4 `<!DOCTYPE>` a declaration. Never rendered.
|
|
699
|
+
// type 5 `<![CDATA[ ]]>` ignored by HTML parsers.
|
|
700
|
+
// type 1 ONLY `script` and `style`. ⚠ `pre` and `textarea`
|
|
701
|
+
// share type 1's start condition but their text IS
|
|
702
|
+
// shown — treating all four alike would delete real
|
|
703
|
+
// convention prose out of a `<pre>` block, which is
|
|
704
|
+
// the content-loss direction.
|
|
705
|
+
// type 6 / type 7 ordinary tags; the text between them renders.
|
|
706
|
+
// The list is spelled here and nowhere else, for the same reason `ORDERED_DELIM`
|
|
707
|
+
// is: two copies of a rule is this module's defect class.
|
|
708
|
+
const HTML_TYPE1_INVISIBLE_TAGS = /^(script|style)$/i;
|
|
709
|
+
|
|
710
|
+
// ⚠⚠ A THIRD STATE, and missing it shipped a false positive AND a repair that
|
|
711
|
+
// made things worse (`p084-xv4` finding 1). `invisible` splits type 1 into "the
|
|
712
|
+
// reader sees nothing" (`script`, `style`) and "the reader sees the text"
|
|
713
|
+
// (`pre`, `textarea`) — but for an HTML COMMENT those two are not the same
|
|
714
|
+
// question, because it asks whether the reader sees the COMMENT, not the text.
|
|
715
|
+
// `<pre>` interior is parsed as HTML, so a comment in it is a comment and
|
|
716
|
+
// is NOT displayed → hiding really happens → worth reporting.
|
|
717
|
+
// `<textarea>` interior is RAW TEXT (RCDATA), so `<!-- x -->` is DISPLAYED
|
|
718
|
+
// verbatim, brackets and all → nothing is hidden from anyone →
|
|
719
|
+
// reporting it is noise, and the repair for it is actively
|
|
720
|
+
// harmful: moving the line below `</textarea>` turns a comment
|
|
721
|
+
// the reader could see into one they cannot.
|
|
722
|
+
// Matched against the shared type-1 opener rather than a fresh regex, for the
|
|
723
|
+
// same reason `HTML_COMMENT_OPEN_RE` exists — two copies of a rule is this
|
|
724
|
+
// module's defect class.
|
|
725
|
+
// ⚠⚠⚠ THERE IS NO TEXTAREA EXEMPTION HERE, AND THAT IS THE DESIGN DECISION —
|
|
726
|
+
// not an oversight, and not something to helpfully add back.
|
|
727
|
+
//
|
|
728
|
+
// `<textarea>` holds raw text, so a comment inside one IS displayed to the reader
|
|
729
|
+
// and reporting it is a false positive. Three rounds were spent trying to suppress
|
|
730
|
+
// exactly that, and **all three attempts produced a silent pass** — the failure this
|
|
731
|
+
// entire proposal exists to remove:
|
|
732
|
+
// `p084-xv4` → exemption added, keyed on the enclosing block.
|
|
733
|
+
// `p084-xv5` → it could not see a nested `<textarea>`; kept, advice neutralised.
|
|
734
|
+
// `p084-xv9` → it skipped the WHOLE LINE, so a comment after a same-line
|
|
735
|
+
// `</textarea>` was exempted although it really is hidden. Doctor
|
|
736
|
+
// read a policy out of it under `All checks passed`.
|
|
737
|
+
// `p084-xv10` → the column version accepted `</textarea >` while the classifier
|
|
738
|
+
// closes only on `</textarea>`. **Two consumers, one rule, no shared
|
|
739
|
+
// spelling** — this module's own oldest defect, reintroduced by the
|
|
740
|
+
// repair. A later exact close then suppressed the unclosed fallback,
|
|
741
|
+
// so nothing fired at all.
|
|
742
|
+
//
|
|
743
|
+
// This repo's rule is that a check rewritten three times is a design question, not
|
|
744
|
+
// a fourth patch. The design answer: **the exemption is not worth having.** It
|
|
745
|
+
// suppresses a rare false positive at the cost of a defect class that has landed
|
|
746
|
+
// three times running, in the one direction this module must never fail in. Without
|
|
747
|
+
// it a `<textarea>` comment is merely REPORTED — the noisy direction, which the
|
|
748
|
+
// module's own direction rule allows and which the maintainer has twice chosen
|
|
749
|
+
// (2026-08-09, for nested `<textarea>` and for container-nested fences).
|
|
750
|
+
//
|
|
751
|
+
// What replaces it is a sentence, not code: `comment-inside-container`'s action and
|
|
752
|
+
// both explainer pages open by telling the reader that a comment inside a
|
|
753
|
+
// `<textarea>` is already visible and should be left alone. That is pinned.
|
|
754
|
+
// ⚠ It also removes an inconsistency nobody could have explained: the document-level
|
|
755
|
+
// `<textarea>` was exempt while a nested one was not.
|
|
756
|
+
function htmlBlockStart(line) {
|
|
757
|
+
for (let t = 0; t < HTML_BLOCK_TYPES.length; t += 1) {
|
|
758
|
+
const [startRe, endRe] = HTML_BLOCK_TYPES[t];
|
|
759
|
+
const m = startRe.exec(line);
|
|
760
|
+
if (m) {
|
|
761
|
+
const invisible = t === 0 ? HTML_TYPE1_INVISIBLE_TAGS.test(m[1]) : true;
|
|
762
|
+
// Marked, which powers `dflow render`, keeps a type-1 raw block open until
|
|
763
|
+
// the tag that opened it closes. A shared `script|pre|style|textarea` end
|
|
764
|
+
// regex lets `</script>` close `<pre>` and makes doctor read hidden content.
|
|
765
|
+
const matchingEnd = t === 0 ? new RegExp(`<\\/${m[1]}>`, 'i') : endRe;
|
|
766
|
+
return { endRe: matchingEnd, invisible };
|
|
767
|
+
}
|
|
768
|
+
}
|
|
769
|
+
if (HTML_BLOCK_TYPE6_RE.test(line)) return { endRe: null, invisible: false };
|
|
770
|
+
return null;
|
|
771
|
+
}
|
|
772
|
+
|
|
773
|
+
// Remove explicit container openers until the line reaches the content that
|
|
774
|
+
// CommonMark parses inside them. Continuation indentation for an open outer list
|
|
775
|
+
// is removed by the caller first; this loop handles nested `> - ...` openers.
|
|
776
|
+
function stripNestedContainerOpeners(line, paragraphOpen = false, priorQuoteDepth = 0) {
|
|
777
|
+
let rest = line;
|
|
778
|
+
let quoteDepth = 0;
|
|
779
|
+
let listOpened = false;
|
|
780
|
+
while (true) {
|
|
781
|
+
const quote = /^ {0,3}>[ \t]?/.exec(rest);
|
|
782
|
+
if (quote) {
|
|
783
|
+
quoteDepth += 1;
|
|
784
|
+
rest = rest.slice(quote[0].length);
|
|
785
|
+
continue;
|
|
786
|
+
}
|
|
787
|
+
// Repeating the same explicit quote chain continues its inner block. Moving
|
|
788
|
+
// into or out of a quote chain starts a different block sequence instead.
|
|
789
|
+
if (quoteDepth !== priorQuoteDepth) paragraphOpen = false;
|
|
790
|
+
if (LIST_ITEM_RE.test(rest)) {
|
|
791
|
+
const canInterrupt = INTERRUPTING_ITEM_RE.test(rest) && !EMPTY_LIST_ITEM_RE.test(rest);
|
|
792
|
+
// An ordered marker beginning above 1 cannot interrupt an open paragraph.
|
|
793
|
+
// Leaving it in `rest` is load-bearing: `<details>` later on that same
|
|
794
|
+
// visible paragraph is not a raw-HTML opener.
|
|
795
|
+
if (paragraphOpen && !canInterrupt) return { content: rest, quoteDepth, listOpened };
|
|
796
|
+
const contentCol = listContentColumn(rest);
|
|
797
|
+
if (contentCol !== null) {
|
|
798
|
+
listOpened = true;
|
|
799
|
+
rest = rest.slice(stringIndexAtVisualColumn(rest, contentCol));
|
|
800
|
+
paragraphOpen = false;
|
|
801
|
+
continue;
|
|
802
|
+
}
|
|
803
|
+
}
|
|
804
|
+
return { content: rest, quoteDepth, listOpened };
|
|
805
|
+
}
|
|
806
|
+
}
|
|
807
|
+
|
|
808
|
+
function nestedParagraphOpens(line, rawHtml) {
|
|
809
|
+
if (rawHtml || BLANK_LINE_RE.test(line)) return false;
|
|
810
|
+
if (parseAtxHeading(line) || SETEXT_UNDERLINE_RE.test(line) || isThematicBreak(line)) return false;
|
|
811
|
+
if (FENCE_OPEN_RE.test(line) || indentWidth(line) >= 4) return false;
|
|
812
|
+
return true;
|
|
813
|
+
}
|
|
814
|
+
|
|
815
|
+
function nestedContainerContext(
|
|
816
|
+
line,
|
|
817
|
+
htmlState,
|
|
818
|
+
paragraphOpen = false,
|
|
819
|
+
quoteDepth = 0,
|
|
820
|
+
htmlInList = false,
|
|
821
|
+
outerList = false
|
|
822
|
+
) {
|
|
823
|
+
const stripped = stripNestedContainerOpeners(line, paragraphOpen, quoteDepth);
|
|
824
|
+
// A deeper quote remains inside the raw block opened by its outer quote, but
|
|
825
|
+
// moving to a shallower still-quoted line exits the inner block sequence.
|
|
826
|
+
// Carrying that state downward made visible escaped comments look hidden; the
|
|
827
|
+
// suggested column-0 repair would then hide text the reader could already see.
|
|
828
|
+
const activeHtml = stripped.quoteDepth < quoteDepth ? null : htmlState;
|
|
829
|
+
const html = nestedRawHtmlContext(stripped.content, activeHtml);
|
|
830
|
+
// Track what owns THIS raw-HTML state, not merely whether some earlier line
|
|
831
|
+
// contained a list marker. A markerless continuation may omit both the list
|
|
832
|
+
// indent and an outer quote marker; Marked still keeps raw HTML open only when
|
|
833
|
+
// the state belongs to a list item. A bare blockquote's state must instead be
|
|
834
|
+
// dropped at that boundary. When an existing state survives, its owner does;
|
|
835
|
+
// a newly opened state gets its owner from the current container chain.
|
|
836
|
+
const nextHtmlInList = html.next
|
|
837
|
+
? (activeHtml ? htmlInList : outerList || stripped.listOpened)
|
|
838
|
+
: false;
|
|
839
|
+
return {
|
|
840
|
+
rawHtml: html.rawHtml,
|
|
841
|
+
nextHtml: html.next,
|
|
842
|
+
nextParagraph: nestedParagraphOpens(stripped.content, html.rawHtml),
|
|
843
|
+
nextQuoteDepth: stripped.quoteDepth,
|
|
844
|
+
nextHtmlInList
|
|
845
|
+
};
|
|
846
|
+
}
|
|
847
|
+
|
|
848
|
+
// Track raw HTML after a list / blockquote prefix has been removed. The main
|
|
849
|
+
// classifier intentionally labels the OUTERMOST block; comment opener selection
|
|
850
|
+
// also needs to know whether inline Markdown syntax applies inside that block.
|
|
851
|
+
// Reuse htmlBlockStart and its end conditions rather than spelling HTML types a
|
|
852
|
+
// second time. `rawHtml` includes the opening/closing line; `next` is the state
|
|
853
|
+
// carried to the following line.
|
|
854
|
+
function nestedRawHtmlContext(line, state) {
|
|
855
|
+
if (state) {
|
|
856
|
+
if (!state.endRe) {
|
|
857
|
+
if (BLANK_LINE_RE.test(line)) return { rawHtml: false, next: null };
|
|
858
|
+
return { rawHtml: true, next: state };
|
|
859
|
+
}
|
|
860
|
+
return { rawHtml: true, next: state.endRe.test(line) ? null : state };
|
|
861
|
+
}
|
|
862
|
+
const html = htmlBlockStart(line);
|
|
863
|
+
if (!html) return { rawHtml: false, next: null };
|
|
864
|
+
const selfEnds = html.endRe && html.endRe.test(line);
|
|
865
|
+
return {
|
|
866
|
+
rawHtml: true,
|
|
867
|
+
next: selfEnds ? null : { endRe: html.endRe }
|
|
868
|
+
};
|
|
869
|
+
}
|
|
870
|
+
|
|
871
|
+
// One label per line, in one pass.
|
|
872
|
+
//
|
|
873
|
+
// Each entry is `{ type, heading }`. `type` is one of `blank`, `heading`,
|
|
874
|
+
// `thematic-break`, `paragraph`, `list`, `blockquote`, `code`, `html`, `table`.
|
|
875
|
+
// `heading` is null except on a heading line, where it is
|
|
876
|
+
// `{ level, text, setext, start }` — `start` being the first line of the
|
|
877
|
+
// underlined paragraph for a setext heading, and the heading's own index for an
|
|
878
|
+
// ATX one. Callers that pop a setext heading's text out of a body need `start`,
|
|
879
|
+
// and getting it from the classification is why a soft-wrapped setext heading
|
|
880
|
+
// now works: the old code assumed the text was exactly one line.
|
|
881
|
+
//
|
|
882
|
+
// ⚠ Fenced code is already gone by the time this runs — `blankFencedBlocks` has
|
|
883
|
+
// replaced those lines with empty ones, preserving positions. That is why there
|
|
884
|
+
// is no fence state here. Keep calling this on that function's output; handing
|
|
885
|
+
// it raw content would read headings out of fenced examples, which is the
|
|
886
|
+
// PROPOSAL-076 G1 defect.
|
|
887
|
+
function classifyLines(lines) {
|
|
888
|
+
const out = new Array(lines.length);
|
|
889
|
+
let open = null; // { kind, start, endRe } — the open DOCUMENT-LEVEL block
|
|
890
|
+
|
|
891
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
892
|
+
const line = lines[i];
|
|
893
|
+
|
|
894
|
+
// (1) Inside an HTML block. Its extent is exactly the cross-line state the
|
|
895
|
+
// old per-line predicates could not hold.
|
|
896
|
+
//
|
|
897
|
+
// ⚠⚠ `blockStart` — WHICH LINE OPENED THIS BLOCK. It is carried here because
|
|
898
|
+
// reconstructing it from the outside is not reliably possible, and that was
|
|
899
|
+
// established the expensive way: `checkGuideProjectContextFormat` needs it to
|
|
900
|
+
// tell an adopter which `<!--` or `<details>` is hiding their
|
|
901
|
+
// `## Project Context`, and THREE consecutive review rounds each defeated a
|
|
902
|
+
// reverse-scan approximation of it (`debt212223-xv1` F2, `xv2` F1, `xv3` F1).
|
|
903
|
+
// Walking back over contiguous `html` lines crosses into the block before it;
|
|
904
|
+
// and tag- or comment-shaped lines INSIDE an open block (`<p>x</p>`,
|
|
905
|
+
// `prose <!-- y`) are indistinguishable from openers when you only look at the
|
|
906
|
+
// line, because whether a line can open a block at all depends on cross-line
|
|
907
|
+
// state that only this loop holds. Three rounds on one predicate is this repo's
|
|
908
|
+
// own signal to stop patching and change the shape. The field is additive, like
|
|
909
|
+
// `unclosed` and `invisible` before it — every existing consumer reads `.type`
|
|
910
|
+
// and `.heading`.
|
|
911
|
+
if (open && open.kind === 'html') {
|
|
912
|
+
if (open.endRe) {
|
|
913
|
+
// ⚠ `visibleFrom`: an end condition can leave text AFTER it on the same
|
|
914
|
+
// line, and that text is rendered. `<!-- note --> rest of the line` puts
|
|
915
|
+
// `rest of the line` in the output, so blanking the whole line loses
|
|
916
|
+
// real content — measured as a SILENT PASS one commit after the
|
|
917
|
+
// visibility rule landed (`p082-b3-k3` finding 2): a retired string
|
|
918
|
+
// sitting after a `-->` stopped being reported.
|
|
919
|
+
const end = open.endRe.exec(line);
|
|
920
|
+
out[i] = { type: 'html', heading: null, invisible: open.invisible, blockStart: open.start, blockEndsOnBlank: false };
|
|
921
|
+
if (end) {
|
|
922
|
+
if (open.invisible) out[i].visibleFrom = end.index + end[0].length;
|
|
923
|
+
open = null;
|
|
924
|
+
}
|
|
925
|
+
continue;
|
|
926
|
+
}
|
|
927
|
+
if (BLANK_LINE_RE.test(line)) { out[i] = { type: 'blank', heading: null }; open = null; continue; }
|
|
928
|
+
// ⚠ THE SECOND OF THE TWO in-block assignments, and it is the one a
|
|
929
|
+
// whitespace-differing `replace_all` missed when `blockStart` was added: this
|
|
930
|
+
// branch serves blocks with NO end condition, which in this parser means
|
|
931
|
+
// **type 6 only** (`<details>`, `<div>`) — type 7 is deliberately not
|
|
932
|
+
// implemented, see the § above, and `<span>` classifies as a paragraph
|
|
933
|
+
// (`debt212223-xv5`, which probed it). Those are exactly the blocks the guide
|
|
934
|
+
// finding had to be able to name. Adding
|
|
935
|
+
// the field to the `endRe` branch alone left `<details>` reporting no opener
|
|
936
|
+
// at all, and the caller's fallback then pointed at the heading itself.
|
|
937
|
+
// ⚠⚠ `blockEndsOnBlank` rides along for the same reason `blockStart` does
|
|
938
|
+
// (`debt212223-y4` finding 1): **how you close a block is not visible from
|
|
939
|
+
// its opening line**. A caller that saw `<details>` and advised "close it"
|
|
940
|
+
// was wrong — `</details>` does not end a type-6 block, only a blank line
|
|
941
|
+
// does, measured — while `</script>` / `]]>` / `?>` really do end types 1/3/5.
|
|
942
|
+
// The distinction is `endRe`, which lives here and nowhere else, so it is
|
|
943
|
+
// carried out rather than re-derived at the caller. Re-deriving block
|
|
944
|
+
// structure outside this function is what cost three rounds already.
|
|
945
|
+
out[i] = { type: 'html', heading: null, invisible: open.invisible, blockStart: open.start, blockEndsOnBlank: true };
|
|
946
|
+
continue;
|
|
947
|
+
}
|
|
948
|
+
|
|
949
|
+
// (2) A blank line closes every document-level block — EXCEPT a list, whose
|
|
950
|
+
// item can continue after it (a "loose" list). Measured: closing the
|
|
951
|
+
// list here made `- a` / blank / ` b` / `---` end the section, while
|
|
952
|
+
// CommonMark keeps the item open and reads `---` as a thematic break.
|
|
953
|
+
// Content-loss direction — the walker pops the line carrying the rule.
|
|
954
|
+
// A BLOCKQUOTE really does close, which is not symmetry but measurement:
|
|
955
|
+
// `> q` / blank / ` more` ends the section in the reference too.
|
|
956
|
+
if (BLANK_LINE_RE.test(line)) {
|
|
957
|
+
out[i] = { type: 'blank', heading: null };
|
|
958
|
+
if (open && open.kind === 'list') {
|
|
959
|
+
open.blankSeen = true;
|
|
960
|
+
// A blank ends type-6 raw HTML but NOT type 1-5. Let the same transition
|
|
961
|
+
// function that owns those end conditions decide instead of clearing
|
|
962
|
+
// every nested block at the loose-list boundary.
|
|
963
|
+
open.nestedHtml = nestedRawHtmlContext('', open.nestedHtml).next;
|
|
964
|
+
if (!open.nestedHtml) open.nestedHtmlInList = false;
|
|
965
|
+
open.nestedParagraph = false;
|
|
966
|
+
open.nestedQuoteDepth = 0;
|
|
967
|
+
}
|
|
968
|
+
else open = null;
|
|
969
|
+
continue;
|
|
970
|
+
}
|
|
971
|
+
|
|
972
|
+
// (2.5) A line indented to the open list item's content column belongs to
|
|
973
|
+
// that item — including a heading, which is then a heading INSIDE the
|
|
974
|
+
// list and not a document-level one. This is checked before every
|
|
975
|
+
// block-start test below, because otherwise ` ## Heading` inside an
|
|
976
|
+
// item would end the section.
|
|
977
|
+
if (open && open.kind === 'list') {
|
|
978
|
+
if (indentWidth(line) >= open.contentCol) {
|
|
979
|
+
const nested = nestedContainerContext(
|
|
980
|
+
line.slice(stringIndexAtVisualColumn(line, open.contentCol)),
|
|
981
|
+
open.nestedHtml,
|
|
982
|
+
open.nestedParagraph,
|
|
983
|
+
open.nestedQuoteDepth,
|
|
984
|
+
open.nestedHtmlInList,
|
|
985
|
+
true
|
|
986
|
+
);
|
|
987
|
+
out[i] = { type: 'list', heading: null };
|
|
988
|
+
if (nested.rawHtml) out[i].rawHtml = true;
|
|
989
|
+
open.nestedHtml = nested.nextHtml;
|
|
990
|
+
open.nestedParagraph = nested.nextParagraph;
|
|
991
|
+
open.nestedQuoteDepth = nested.nextQuoteDepth;
|
|
992
|
+
open.nestedHtmlInList = nested.nextHtmlInList;
|
|
993
|
+
open.blankSeen = false;
|
|
994
|
+
continue;
|
|
995
|
+
}
|
|
996
|
+
// Not indented into the item, and a blank line intervened: the list ended
|
|
997
|
+
// at that blank and this line starts something new.
|
|
998
|
+
if (open.blankSeen) open = null;
|
|
999
|
+
}
|
|
1000
|
+
|
|
1001
|
+
const paraOpen = open !== null && open.kind === 'paragraph';
|
|
1002
|
+
|
|
1003
|
+
// (3) A setext underline — but ONLY when there is a paragraph to underline.
|
|
1004
|
+
// This single test is the arbitration the old code spread across
|
|
1005
|
+
// `closesOwnBlock`, `blockStartIndex` and `isParagraphLine`, which is
|
|
1006
|
+
// why they could contradict each other about the same line.
|
|
1007
|
+
// ⚠ It is tested BEFORE the thematic break on purpose: CommonMark 4.3
|
|
1008
|
+
// gives the setext heading precedence where both readings are possible,
|
|
1009
|
+
// so `para` + `---` is a heading while a bare `---` is a break.
|
|
1010
|
+
if (paraOpen && SETEXT_UNDERLINE_RE.test(line)) {
|
|
1011
|
+
const marker = line.match(SETEXT_UNDERLINE_RE)[1];
|
|
1012
|
+
const text = stripSpaceTab(lines.slice(open.start, i).map(stripSpaceTab).join(' '));
|
|
1013
|
+
out[i] = {
|
|
1014
|
+
type: 'heading',
|
|
1015
|
+
heading: { level: marker[0] === '=' ? 1 : 2, text, setext: true, start: open.start }
|
|
1016
|
+
};
|
|
1017
|
+
open = null;
|
|
1018
|
+
continue;
|
|
1019
|
+
}
|
|
1020
|
+
|
|
1021
|
+
// (4) Shapes that open their own block wherever they appear.
|
|
1022
|
+
const atx = parseAtxHeading(line);
|
|
1023
|
+
if (atx) {
|
|
1024
|
+
out[i] = { type: 'heading', heading: { ...atx, start: i } };
|
|
1025
|
+
open = null;
|
|
1026
|
+
continue;
|
|
1027
|
+
}
|
|
1028
|
+
|
|
1029
|
+
if (isThematicBreak(line)) {
|
|
1030
|
+
out[i] = { type: 'thematic-break', heading: null };
|
|
1031
|
+
open = null;
|
|
1032
|
+
continue;
|
|
1033
|
+
}
|
|
1034
|
+
|
|
1035
|
+
// HTML blocks of types 1-6 can all interrupt a paragraph, so this sits
|
|
1036
|
+
// above the continuation logic rather than inside the block-start branch.
|
|
1037
|
+
const html = htmlBlockStart(line);
|
|
1038
|
+
if (html) {
|
|
1039
|
+
out[i] = { type: 'html', heading: null, invisible: html.invisible, blockStart: i, blockEndsOnBlank: !html.endRe };
|
|
1040
|
+
// ⚠ The end condition is checked on the OPENING line too (CommonMark 4.6:
|
|
1041
|
+
// "the block ends at the first line matching the end condition", which
|
|
1042
|
+
// includes the line that started it). Leaving it out meant a single-line
|
|
1043
|
+
// `<!-- Seeded by Dflow. -->` — the comment at the head of every packaged
|
|
1044
|
+
// template — opened a block that never closed, and every heading in the
|
|
1045
|
+
// file after it disappeared. The existing pins caught it immediately,
|
|
1046
|
+
// which is the argument for having run them before widening anything.
|
|
1047
|
+
// Same rule on the opening line, which may also be the closing one:
|
|
1048
|
+
// `<!-- a --> b` renders ` b`.
|
|
1049
|
+
const selfEnd = html.endRe && html.endRe.exec(line);
|
|
1050
|
+
if (selfEnd) {
|
|
1051
|
+
if (html.invisible) out[i].visibleFrom = selfEnd.index + selfEnd[0].length;
|
|
1052
|
+
open = null;
|
|
1053
|
+
} else {
|
|
1054
|
+
open = { kind: 'html', start: i, endRe: html.endRe, invisible: html.invisible };
|
|
1055
|
+
}
|
|
1056
|
+
continue;
|
|
1057
|
+
}
|
|
1058
|
+
|
|
1059
|
+
if (BLOCKQUOTE_RE.test(line)) {
|
|
1060
|
+
const sameQuote = open && open.kind === 'blockquote';
|
|
1061
|
+
const nested = nestedContainerContext(
|
|
1062
|
+
line,
|
|
1063
|
+
sameQuote ? open.nestedHtml : null,
|
|
1064
|
+
sameQuote ? open.nestedParagraph : false,
|
|
1065
|
+
sameQuote ? open.nestedQuoteDepth : 0,
|
|
1066
|
+
sameQuote ? open.nestedHtmlInList : false
|
|
1067
|
+
);
|
|
1068
|
+
out[i] = { type: 'blockquote', heading: null };
|
|
1069
|
+
if (nested.rawHtml) out[i].rawHtml = true;
|
|
1070
|
+
open = {
|
|
1071
|
+
kind: 'blockquote',
|
|
1072
|
+
start: open && open.kind === 'blockquote' ? open.start : i,
|
|
1073
|
+
nestedHtml: nested.nextHtml,
|
|
1074
|
+
nestedParagraph: nested.nextParagraph,
|
|
1075
|
+
nestedQuoteDepth: nested.nextQuoteDepth,
|
|
1076
|
+
nestedHtmlInList: nested.nextHtmlInList
|
|
1077
|
+
};
|
|
1078
|
+
continue;
|
|
1079
|
+
}
|
|
1080
|
+
|
|
1081
|
+
// A list interrupts an open paragraph only when it is a bullet, or an
|
|
1082
|
+
// ordered marker starting at 1, AND its first line is non-blank
|
|
1083
|
+
// (CommonMark 5.2). Any other ordered marker — and an empty item — is
|
|
1084
|
+
// paragraph text. Reading those as block starts is the divergence `g1` found
|
|
1085
|
+
// by widening the arbiter, in the silent-pass direction.
|
|
1086
|
+
if (LIST_ITEM_RE.test(line)) {
|
|
1087
|
+
const canInterrupt = INTERRUPTING_ITEM_RE.test(line) && !EMPTY_LIST_ITEM_RE.test(line);
|
|
1088
|
+
if (!paraOpen || canInterrupt) {
|
|
1089
|
+
const contentCol = listContentColumn(line);
|
|
1090
|
+
const nested = nestedContainerContext(
|
|
1091
|
+
line.slice(stringIndexAtVisualColumn(line, contentCol)),
|
|
1092
|
+
null,
|
|
1093
|
+
false,
|
|
1094
|
+
0,
|
|
1095
|
+
false,
|
|
1096
|
+
true
|
|
1097
|
+
);
|
|
1098
|
+
out[i] = { type: 'list', heading: null };
|
|
1099
|
+
if (nested.rawHtml) out[i].rawHtml = true;
|
|
1100
|
+
// ⚠ `listContentColumn` cannot return null here, and the reason is worth
|
|
1101
|
+
// stating because the consequence if it ever did is severe:
|
|
1102
|
+
// `indentWidth(line) >= null` is `>= 0`, true for EVERY line, so the
|
|
1103
|
+
// whole rest of the document would be swallowed as list content —
|
|
1104
|
+
// silent pass. It cannot happen because `LIST_ITEM_PREFIX_RE` is built
|
|
1105
|
+
// from the same two marker fragments as `LIST_ITEM_RE` and is strictly
|
|
1106
|
+
// more permissive (its trailing `[ \t]*` matches empty where
|
|
1107
|
+
// `LIST_ITEM_RE` requires `[ \t]` or end of line), so anything this
|
|
1108
|
+
// branch accepts, it matches. Sharing the fragments is what makes that
|
|
1109
|
+
// an invariant rather than a hope — this module has broken exactly this
|
|
1110
|
+
// kind of unwritten relationship between two expressions five times.
|
|
1111
|
+
open = {
|
|
1112
|
+
kind: 'list',
|
|
1113
|
+
start: open && open.kind === 'list' ? open.start : i,
|
|
1114
|
+
// Each item sets the column its own content continues at.
|
|
1115
|
+
contentCol,
|
|
1116
|
+
blankSeen: false,
|
|
1117
|
+
nestedHtml: nested.nextHtml,
|
|
1118
|
+
nestedParagraph: nested.nextParagraph,
|
|
1119
|
+
nestedQuoteDepth: nested.nextQuoteDepth,
|
|
1120
|
+
nestedHtmlInList: nested.nextHtmlInList
|
|
1121
|
+
};
|
|
1122
|
+
continue;
|
|
1123
|
+
}
|
|
1124
|
+
// else: falls through to the paragraph continuation below.
|
|
1125
|
+
}
|
|
1126
|
+
|
|
1127
|
+
// (5) Continuation of an open container. Inside a list item or a blockquote
|
|
1128
|
+
// the document-level block is the container, not a paragraph, so no
|
|
1129
|
+
// setext underline can close anything here — which is precisely what
|
|
1130
|
+
// `blockStartIndex` used to walk backwards to work out.
|
|
1131
|
+
if (open && (open.kind === 'list' || open.kind === 'blockquote')) {
|
|
1132
|
+
out[i] = { type: open.kind, heading: null };
|
|
1133
|
+
// Raw HTML owned by a list item CAN cross a lazy continuation even when
|
|
1134
|
+
// its list indent — and an outer quote marker — is omitted. Marked keeps
|
|
1135
|
+
// the block open, so comment syntax on this line is still raw HTML. A bare
|
|
1136
|
+
// blockquote cannot cross that boundary; clearing it is the quote-exit
|
|
1137
|
+
// control that prevents visible escaped examples from being rejected.
|
|
1138
|
+
if (open.nestedHtml && open.nestedHtmlInList) {
|
|
1139
|
+
const lazy = nestedRawHtmlContext(line, open.nestedHtml);
|
|
1140
|
+
if (lazy.rawHtml) out[i].rawHtml = true;
|
|
1141
|
+
open.nestedHtml = lazy.next;
|
|
1142
|
+
if (!lazy.next) open.nestedHtmlInList = false;
|
|
1143
|
+
} else if (open.kind === 'blockquote') {
|
|
1144
|
+
open.nestedHtml = null;
|
|
1145
|
+
open.nestedHtmlInList = false;
|
|
1146
|
+
open.nestedParagraph = false;
|
|
1147
|
+
open.nestedQuoteDepth = 0;
|
|
1148
|
+
}
|
|
1149
|
+
continue;
|
|
1150
|
+
}
|
|
1151
|
+
|
|
1152
|
+
// (6) A table runs until a line that is not a row.
|
|
1153
|
+
// ⚠ NAMED GAP, DELIBERATELY NOT "FIXED": this branch has no indent test,
|
|
1154
|
+
// while (8) — which OPENS the table — now requires the header row to sit at
|
|
1155
|
+
// `indentWidth < 4`. So `| a | b |` / `|---|---|` / `<4SP>| c | d |` keeps
|
|
1156
|
+
// the third line in the table for us, where GFM ends the table and starts an
|
|
1157
|
+
// indented code block. The symmetric-looking change (adding an indent test
|
|
1158
|
+
// here) is NOT made, because **no reference available to this repo can say
|
|
1159
|
+
// which answer is right**, and shipping a change on an unarbitrated hunch is
|
|
1160
|
+
// how this module's boundary logic got into trouble in the first place:
|
|
1161
|
+
// - `commonmark@0.31.2` has no tables, so it never enters this state at
|
|
1162
|
+
// all and reads the whole run as one paragraph. It agrees with us here
|
|
1163
|
+
// for a reason that has nothing to do with the question.
|
|
1164
|
+
// - `marked`'s GFM lexer does answer, but it is measurably non-conformant
|
|
1165
|
+
// on exactly this boundary — for `prose` / `<4SP>prose` / `===` the spec
|
|
1166
|
+
// and `commonmark` both say setext H1 (an indented line cannot interrupt
|
|
1167
|
+
// a paragraph, so it is a lazy continuation) while `marked` says
|
|
1168
|
+
// paragraph. A reference that is wrong on the control cannot settle the
|
|
1169
|
+
// case.
|
|
1170
|
+
// Direction if GFM turns out to be right: content loss, the loud one.
|
|
1171
|
+
// Closing this needs a third reference (cmark-gfm), not more reasoning.
|
|
1172
|
+
if (open && open.kind === 'table') {
|
|
1173
|
+
if (line.includes('|')) { out[i] = { type: 'table', heading: null }; continue; }
|
|
1174
|
+
out[i] = { type: 'paragraph', heading: null };
|
|
1175
|
+
open = { kind: 'paragraph', start: i };
|
|
1176
|
+
continue;
|
|
1177
|
+
}
|
|
1178
|
+
|
|
1179
|
+
// (7) Indented code — which cannot interrupt a paragraph. After one, an
|
|
1180
|
+
// indented line is a lazy continuation and the paragraph stays open.
|
|
1181
|
+
// ⚠ `indentWidth(...) >= 4`, NOT a raw-line regex. This branch used
|
|
1182
|
+
// `INDENTED_CODE_RE = /^(\t| {4,})/`, which decides the SAME question as
|
|
1183
|
+
// `indentWidth` with different code — and they disagreed: a space followed
|
|
1184
|
+
// by a tab is column 4 (a tab advances to the next multiple of four), so
|
|
1185
|
+
// `indentWidth(' \tx') === 4` while the regex said false. Both directions
|
|
1186
|
+
// measured: `<SP><TAB>prose` became a paragraph, so a following `---` was a
|
|
1187
|
+
// setext underline and the pop deleted the body (content loss); and
|
|
1188
|
+
// `<SP><SP><SP><TAB>| a | b |` became a table header, so the section ran on
|
|
1189
|
+
// (silent pass). This module's own rule names it — a regex against a raw
|
|
1190
|
+
// line, anywhere else, to decide what kind of line it is.
|
|
1191
|
+
//
|
|
1192
|
+
// ⚠⚠ WHY WIDENING THIS BRANCH STOLE NOTHING FROM THE BRANCHES ABOVE IT, and
|
|
1193
|
+
// this is the part that was missing rather than wrong. `indentWidth >= 4` is
|
|
1194
|
+
// strictly wider than the regex it replaced, so it can only take lines that
|
|
1195
|
+
// used to reach branches further down — unless it also overlaps branch (4).
|
|
1196
|
+
// It cannot, and the reason is a property of every opener in this module,
|
|
1197
|
+
// not of this line: **every raw-prefix opener is `^ {0,3}` followed by a
|
|
1198
|
+
// character that is neither a space nor a tab**, which bounds any line one
|
|
1199
|
+
// of them matches at `indentWidth <= 3`; while `indentWidth >= 4` requires
|
|
1200
|
+
// four spaces, or a tab, inside the first four columns, which no `^ {0,3}X`
|
|
1201
|
+
// can match. So (4) and (7) partition cleanly.
|
|
1202
|
+
// ⚠ That is a property of the OTHER expressions, so it can be broken from a
|
|
1203
|
+
// distance: admit a tab into any `^ {0,3}` prefix — write `^[ \t]{0,3}`,
|
|
1204
|
+
// which looks like a tolerance improvement — and this branch starts eating
|
|
1205
|
+
// lines that opener was meant to claim, silently. `assertOpenerIndentPartition`
|
|
1206
|
+
// in `test/upgrade-drift.mjs` sweeps every opener against every indent form
|
|
1207
|
+
// so the break is a test failure and not a review round.
|
|
1208
|
+
if (!paraOpen && indentWidth(line) >= 4) {
|
|
1209
|
+
out[i] = { type: 'code', heading: null };
|
|
1210
|
+
open = { kind: 'code', start: open && open.kind === 'code' ? open.start : i };
|
|
1211
|
+
continue;
|
|
1212
|
+
}
|
|
1213
|
+
|
|
1214
|
+
// (8) A GFM table opens when a delimiter row sits directly under a
|
|
1215
|
+
// pipe-bearing paragraph line. Requiring the header is what keeps
|
|
1216
|
+
// ordinary prose containing a pipe — "the Command | Query split" — from
|
|
1217
|
+
// being read as a table, which an earlier `line.includes('|')` test got
|
|
1218
|
+
// wrong in the silent-success direction.
|
|
1219
|
+
// ⚠ The header row is the line DIRECTLY ABOVE the delimiter row, wherever
|
|
1220
|
+
// that line sits in the paragraph. Requiring it to be the paragraph's first
|
|
1221
|
+
// line (`open.start === i - 1`) was a regression against the predicate this
|
|
1222
|
+
// pass replaced, whose backward scan had no such restriction: one intro
|
|
1223
|
+
// sentence above a table — `Our rows:` / `| a | b |` / `|---|---|` — made the
|
|
1224
|
+
// whole run one paragraph, so a following `---` became a setext underline
|
|
1225
|
+
// and the pop deleted the ENTIRE section body. Total content loss for that
|
|
1226
|
+
// section, on input a hand-edited `_conventions.md` produces immediately.
|
|
1227
|
+
// Any earlier lines of the paragraph stay a paragraph; only the header row
|
|
1228
|
+
// and the delimiter row become the table.
|
|
1229
|
+
// ⚠⚠ THE HEADER ROW'S OWN INDENT IS CHECKED, and leaving it out was a
|
|
1230
|
+
// silent-pass defect this very branch introduced. `TABLE_DELIMITER_ROW_RE`
|
|
1231
|
+
// constrains the DELIMITER row to `^ {0,3}`, but the header test is only
|
|
1232
|
+
// `.includes('|')`. While the header had to be the paragraph's first line
|
|
1233
|
+
// that gap was unreachable — an indented line can only be a paragraph
|
|
1234
|
+
// CONTINUATION, since indented code cannot interrupt a paragraph — so
|
|
1235
|
+
// widening the rule opened it. A 4-column-indented header is indented code
|
|
1236
|
+
// to CommonMark and no table at all to GFM (`marked` emits no table token),
|
|
1237
|
+
// yet we read one, the `---` below stopped ending the section, and a stale
|
|
1238
|
+
// file reported `current`. Measured: five rows where both references
|
|
1239
|
+
// disagreed with us and neither with each other.
|
|
1240
|
+
// `i > 0` is deliberately gone: `paraOpen` cannot be true at i === 0 (`open`
|
|
1241
|
+
// starts null and is only assigned in the loop), so it never guarded
|
|
1242
|
+
// anything while reading like the bounds check that made this branch safe.
|
|
1243
|
+
// ⚠⚠ RELABELLING `out[i - 1]` RESTS ON AN INVARIANT NOTHING ELSE STATES:
|
|
1244
|
+
// when `paraOpen` is true at `i`, `out[i - 1]` is necessarily `'paragraph'`.
|
|
1245
|
+
// It holds because the only branches that leave `open.kind === 'paragraph'`
|
|
1246
|
+
// are (9) and (6)'s else-path, and both write `'paragraph'` to their own
|
|
1247
|
+
// line. So the overwrite below always replaces a label this loop just wrote
|
|
1248
|
+
// for the header row — never a `code`, `list` or `heading` line.
|
|
1249
|
+
// ⚠ A future branch that labels its line something else while leaving a
|
|
1250
|
+
// paragraph open would have that label silently overwritten as `table`, and
|
|
1251
|
+
// nothing here would notice. If you add such a branch, this line is the one
|
|
1252
|
+
// that breaks — and it breaks quietly, which is why the invariant is written
|
|
1253
|
+
// down instead of being left to be re-derived.
|
|
1254
|
+
if (paraOpen && indentWidth(lines[i - 1]) < 4 && lines[i - 1].includes('|') && TABLE_DELIMITER_ROW_RE.test(line)) {
|
|
1255
|
+
out[i - 1] = { type: 'table', heading: null };
|
|
1256
|
+
out[i] = { type: 'table', heading: null };
|
|
1257
|
+
open = { kind: 'table', start: i - 1 };
|
|
1258
|
+
continue;
|
|
1259
|
+
}
|
|
1260
|
+
|
|
1261
|
+
// (9) Paragraph: a continuation of the open one, or a new one.
|
|
1262
|
+
out[i] = { type: 'paragraph', heading: null };
|
|
1263
|
+
if (!paraOpen) open = { kind: 'paragraph', start: i };
|
|
1264
|
+
}
|
|
1265
|
+
|
|
1266
|
+
// ⚠⚠ AN HTML BLOCK STILL OPEN AT EOF IS MARKED, and this is not a parsing
|
|
1267
|
+
// nicety — it was a HIGH silent pass reachable on a file nobody had to
|
|
1268
|
+
// construct. HTML block types 1-5 close only on their end condition, so an
|
|
1269
|
+
// adopter who leaves one `<!--` unclosed mid-edit puts the ENTIRE REST OF THE
|
|
1270
|
+
// FILE inside that block. The parse is right; what was wrong was that
|
|
1271
|
+
// `conventionsSectionBodies` kept pushing those lines into the open section's
|
|
1272
|
+
// body, so the section swallowed every later section, reached a quoted copy
|
|
1273
|
+
// of the rule in an appendix, and a genuinely stale file reported `current`.
|
|
1274
|
+
// Measured through the real `doctor`: closing the comment produced two
|
|
1275
|
+
// warnings, leaving it open produced none.
|
|
1276
|
+
//
|
|
1277
|
+
// ⚠ Only HTML. A paragraph, list or code block open at EOF is ordinary — the
|
|
1278
|
+
// document simply ended. Fenced blocks never reach here because
|
|
1279
|
+
// `blankFencedBlocks` has already emptied them, and that immunity is
|
|
1280
|
+
// accidental rather than designed, which is exactly why this needed marking
|
|
1281
|
+
// rather than another special case.
|
|
1282
|
+
//
|
|
1283
|
+
// The flag is additive: every existing consumer reads `.type` and `.heading`.
|
|
1284
|
+
// ⚠⚠ `open.endRe` IS THE WHOLE POINT, and leaving it out was a regression this
|
|
1285
|
+
// very block introduced. Only types 1-5 close on an end condition; **type 6
|
|
1286
|
+
// closes on a blank line**, and `endRe` is null for it. A document ending in a
|
|
1287
|
+
// type-6 block therefore looked "never closed" — but a file that ends with a
|
|
1288
|
+
// newline supplies a trailing blank through the splitter, so the block closes
|
|
1289
|
+
// normally, while a file WITHOUT a trailing newline did not. Two files one
|
|
1290
|
+
// byte apart, and the shorter one got a spurious "unclosed HTML block"
|
|
1291
|
+
// warning plus a false `stale` on a section that plainly carries the rule.
|
|
1292
|
+
// Content loss, on `<div>…</div>` — ordinary hand-written HTML.
|
|
1293
|
+
// The comment above already said types 1-5; the code did not agree with it.
|
|
1294
|
+
if (open && open.kind === 'html' && open.endRe) {
|
|
1295
|
+
for (let i = open.start; i < out.length; i += 1) out[i].unclosed = true;
|
|
1296
|
+
}
|
|
1297
|
+
|
|
1298
|
+
return out;
|
|
1299
|
+
}
|
|
1300
|
+
|
|
1301
|
+
// ⚠ `toUpperCase`, not `toLowerCase`, and the difference is not cosmetic:
|
|
1302
|
+
// `headingResolves` PREFIX-matches on this key, and lowercasing is not
|
|
1303
|
+
// prefix-preserving. Greek capital sigma lowercases to a FINAL sigma at a word
|
|
1304
|
+
// end and a medial one elsewhere, so `ΑΣ` becomes `ας` while `ΑΣΒΓ` becomes
|
|
1305
|
+
// `ασβγ` — and `"ασβγ".startsWith("ας")` is false although the raw strings do
|
|
1306
|
+
// prefix-match. A `<file>.md § ΑΣ` reference to a heading `ΑΣΒΓ` was reported as
|
|
1307
|
+
// dangling. Uppercasing has no context-sensitive mapping in the default
|
|
1308
|
+
// case-conversion tables, so it preserves prefixes; `ß` → `SS` changes length
|
|
1309
|
+
// but preserves them too.
|
|
1310
|
+
// ⚠ Any fold works for the exact and Set comparisons; only the prefix consumer
|
|
1311
|
+
// constrains the choice. Keep ONE fold for all three rather than a
|
|
1312
|
+
// prefix-preserving one here and a convenient one there — that split is how the
|
|
1313
|
+
// three consumers of this normaliser drifted apart before.
|
|
1314
|
+
function headingKey(text) {
|
|
1315
|
+
return normalizeHeading(text).toUpperCase();
|
|
1316
|
+
}
|
|
1317
|
+
|
|
1318
|
+
// ALL bodies whose heading matches — `_conventions.md` is hand-edited and a
|
|
1319
|
+
// duplicated heading is ordinary. Taking only the first made a file that
|
|
1320
|
+
// carried the rule in its second copy report `stale`.
|
|
1321
|
+
function conventionsSectionBodies(content, heading) {
|
|
1322
|
+
// Blanked by the same expression `visibleTextLines` returns — see
|
|
1323
|
+
// `classifiedVisible`. Text a reader never sees is not this section's content.
|
|
1324
|
+
const { info, lines } = classifiedVisible(content);
|
|
1325
|
+
const target = headingKey(heading);
|
|
1326
|
+
const bodies = [];
|
|
1327
|
+
let level = 0;
|
|
1328
|
+
let body = null;
|
|
1329
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
1330
|
+
const h = info[i].heading;
|
|
1331
|
+
if (h) {
|
|
1332
|
+
// A setext heading's TEXT lines were already pushed into the open body —
|
|
1333
|
+
// take them back out. `h.start` is where the underlined paragraph began,
|
|
1334
|
+
// so this pops all of them.
|
|
1335
|
+
//
|
|
1336
|
+
// ⚠ The old version popped exactly ONE line and guarded it with
|
|
1337
|
+
// `body[body.length - 1].trim() === h.text`, because it assumed a setext
|
|
1338
|
+
// heading's text was a single line. A soft-wrapped one therefore left its
|
|
1339
|
+
// earlier lines sitting in the body as though they were content, and
|
|
1340
|
+
// reported the heading under only its last line — so a section titled
|
|
1341
|
+
// `Ceremony Scaling` / `(Project Application)` / `===` did not match the
|
|
1342
|
+
// fingerprint that names it. The classification knows the paragraph's
|
|
1343
|
+
// extent, so both halves are now right for free.
|
|
1344
|
+
if (body && h.setext) {
|
|
1345
|
+
for (let k = h.start; k < i && body.length > 0; k += 1) body.pop();
|
|
1346
|
+
}
|
|
1347
|
+
if (body && h.level <= level) {
|
|
1348
|
+
bodies.push(body.join('\n'));
|
|
1349
|
+
body = null;
|
|
1350
|
+
level = 0;
|
|
1351
|
+
}
|
|
1352
|
+
if (!body && headingKey(h.text) === target) {
|
|
1353
|
+
level = h.level;
|
|
1354
|
+
body = [];
|
|
1355
|
+
}
|
|
1356
|
+
continue;
|
|
1357
|
+
}
|
|
1358
|
+
// ⚠ A line inside an HTML block that never closed is NOT this section's
|
|
1359
|
+
// content — it is the interior of a block that swallowed the rest of the
|
|
1360
|
+
// file. Pushing it let a section reach text from a later section and report
|
|
1361
|
+
// `current` on a file that had genuinely lost the rule: silent pass, the
|
|
1362
|
+
// worst direction, and reachable from one unclosed `<!--`.
|
|
1363
|
+
// Stopping here is the conservative half of the fix — we only trust content
|
|
1364
|
+
// we can attribute to the section. The other half is telling the user why,
|
|
1365
|
+
// which `findConventionsDrift` does, because otherwise the resulting `stale`
|
|
1366
|
+
// is correct and inexplicable.
|
|
1367
|
+
if (info[i].unclosed) break;
|
|
1368
|
+
if (body) body.push(lines[i]);
|
|
1369
|
+
}
|
|
1370
|
+
if (body) bodies.push(body.join('\n'));
|
|
1371
|
+
return bodies;
|
|
1372
|
+
}
|
|
1373
|
+
|
|
1374
|
+
// The file's lines with everything a reader cannot see blanked out: fenced code
|
|
1375
|
+
// (already blanked on the way in) and the interior of an HTML block that renders
|
|
1376
|
+
// no text — a comment, a processing instruction, a declaration, CDATA, a
|
|
1377
|
+
// `<script>` or a `<style>`.
|
|
1378
|
+
//
|
|
1379
|
+
// ⚠⚠ WHY THIS EXISTS AS A FUNCTION RATHER THAN A TEST AT EACH CONSUMER. The
|
|
1380
|
+
// rewrite made `classifyLines` the single authority on *what kind of line this
|
|
1381
|
+
// is* — and then every CONTENT extractor went on reading the raw text without
|
|
1382
|
+
// asking. So three consumers each decided independently, by not deciding:
|
|
1383
|
+
// `findConventionsDrift` matched a fingerprint marker inside `<!-- … -->` and
|
|
1384
|
+
// certified a stale file as `current`; `parseContextLine` read a commented-out
|
|
1385
|
+
// machine-readable policy value as though it were live; `extractSectionRefs`
|
|
1386
|
+
// reported a reference inside a comment as a dangling link, with the `-->`
|
|
1387
|
+
// captured into the heading text (`p082-b3-k1`, all three reproduced).
|
|
1388
|
+
//
|
|
1389
|
+
// ⚠ The module was already INCONSISTENT about this, which is the tell that it
|
|
1390
|
+
// was never decided: a retired string below an *unclosed* comment was correctly
|
|
1391
|
+
// ignored (the section body stops there), while the same string inside a *closed*
|
|
1392
|
+
// comment counted. Same question, two answers, depending on whether the user
|
|
1393
|
+
// typed `-->`.
|
|
1394
|
+
//
|
|
1395
|
+
// BLANK, do not delete: `extractSectionRefs` reports 1-based line numbers and
|
|
1396
|
+
// `conventionsSectionBodies` pops setext text by index. This is the same
|
|
1397
|
+
// convention `blankFencedBlocks` established, for the same reason.
|
|
1398
|
+
//
|
|
1399
|
+
// ⚠ This is a real semantic decision, not a bug fix with an obvious answer:
|
|
1400
|
+
// it says a commented-out rule is ABSENT. `commonmark` agrees a reader never
|
|
1401
|
+
// sees it, and the section differential's whole question is "does a reader see
|
|
1402
|
+
// the marker in this section" — so the arbiter this module already trusts was
|
|
1403
|
+
// answering one question while the implementation answered another.
|
|
1404
|
+
// ⚠⚠ ONE APPLICATION OF THE RULE, not two. The first draft of this fix had
|
|
1405
|
+
// `visibleTextLines` blank the lines and `conventionsSectionBodies` repeat the
|
|
1406
|
+
// test inline (`info[i].invisible ? '' : line`) because it needed `info` anyway.
|
|
1407
|
+
// Both read the same flag, so the DECISION was single-sourced — but the two
|
|
1408
|
+
// expressions could still drift, and "one rule, two expressions" is the defect
|
|
1409
|
+
// class this module exists to hold shut. It was caught by measurement, not by
|
|
1410
|
+
// review: breaking the projection left the `conventionsSectionBodies` pins
|
|
1411
|
+
// green, which is precisely the "my pin cannot fail against the thing it names"
|
|
1412
|
+
// shape the mutation harness was built for.
|
|
1413
|
+
// ⚠ SPACES, not deletion, and the span rather than the line. Two reasons, both
|
|
1414
|
+
// paid for: `extractSectionRefs` compares a match's index against the line's
|
|
1415
|
+
// length, so shortening a line moves every column it reads; and an end condition
|
|
1416
|
+
// can leave rendered text after it on the same line (`visibleFrom`), which
|
|
1417
|
+
// whole-line blanking silently dropped.
|
|
1418
|
+
function classifiedVisible(content) {
|
|
1419
|
+
const lines = blankFencedBlocks(String(content));
|
|
1420
|
+
const info = classifyLines(lines);
|
|
1421
|
+
return {
|
|
1422
|
+
info,
|
|
1423
|
+
lines: lines.map((line, i) => {
|
|
1424
|
+
if (!info[i] || !info[i].invisible) return line;
|
|
1425
|
+
const from = info[i].visibleFrom;
|
|
1426
|
+
const hidden = from === undefined ? line.length : Math.min(from, line.length);
|
|
1427
|
+
return ' '.repeat(hidden) + line.slice(hidden);
|
|
1428
|
+
})
|
|
1429
|
+
};
|
|
1430
|
+
}
|
|
1431
|
+
|
|
1432
|
+
function visibleTextLines(content) {
|
|
1433
|
+
return classifiedVisible(content).lines;
|
|
1434
|
+
}
|
|
1435
|
+
|
|
1436
|
+
// The same content with every CODE line masked to spaces — same length, same
|
|
1437
|
+
// line count, so an index into the result is an index into the original.
|
|
1438
|
+
//
|
|
1439
|
+
// ⚠⚠ WHY A MASK AND NOT THE LINE ARRAY. `configure-agents` finds its managed
|
|
1440
|
+
// region with `indexOf` over the raw file and then SLICES the original by that
|
|
1441
|
+
// index, so anything that shortens a line moves the cut and corrupts the write.
|
|
1442
|
+
// A mask keeps every offset while removing the text that must not match.
|
|
1443
|
+
//
|
|
1444
|
+
// It exists because the write path — the only path in this tool that modifies a
|
|
1445
|
+
// user's file — decided a structural question by not asking: a user documenting
|
|
1446
|
+
// Dflow in their own `AGENTS.md`, with the marker comments shown inside a fenced
|
|
1447
|
+
// example, had that example treated as the managed region and OVERWRITTEN.
|
|
1448
|
+
// `runConfigureAgents` exited 0 and `doctor` then reported all clean
|
|
1449
|
+
// (`p082-b3-k3` finding 1). USER CONTENT LOSS, which is the worst thing this
|
|
1450
|
+
// codebase can do.
|
|
1451
|
+
//
|
|
1452
|
+
// ⚠ Fenced and indented code only. A marker inside a `<pre>` block is still
|
|
1453
|
+
// taken as a region boundary — `pre` renders its text, so it is not code to this
|
|
1454
|
+
// module, and closing that case needs a decision about what `<pre>` MEANS here
|
|
1455
|
+
// rather than another line in this function. Stated so the gap is known rather
|
|
1456
|
+
// than discovered.
|
|
1457
|
+
function maskCodeBlocks(content) {
|
|
1458
|
+
const s = String(content);
|
|
1459
|
+
const raw = s.split('\n');
|
|
1460
|
+
const blanked = blankFencedBlocks(s);
|
|
1461
|
+
const info = classifyLines(blanked);
|
|
1462
|
+
return raw.map((line, i) => {
|
|
1463
|
+
const isCode = blanked[i] !== line || (info[i] && info[i].type === 'code');
|
|
1464
|
+
return isCode ? ' '.repeat(line.length) : line;
|
|
1465
|
+
}).join('\n');
|
|
1466
|
+
}
|
|
1467
|
+
|
|
1468
|
+
// Where an HTML block that never closes begins, or -1. Reads the SAME
|
|
1469
|
+
// classification every other consumer reads rather than re-deciding it — the
|
|
1470
|
+
// module's one rule about not spelling a question twice.
|
|
1471
|
+
function unclosedHtmlBlockLine(content) {
|
|
1472
|
+
const lines = blankFencedBlocks(String(content));
|
|
1473
|
+
const info = classifyLines(lines);
|
|
1474
|
+
const at = info.findIndex((c) => c && c.unclosed);
|
|
1475
|
+
return at === -1 ? -1 : at + 1; // 1-based, for a message a human reads
|
|
1476
|
+
}
|
|
1477
|
+
|
|
1478
|
+
// ---------------------------------------------------------------------------
|
|
1479
|
+
// PROPOSAL-084 — UNCERTAINTY DETECTORS.
|
|
1480
|
+
//
|
|
1481
|
+
// These answer a different KIND of question from everything above, and that
|
|
1482
|
+
// difference is the whole reason the feature is safe to ship on top of a parser
|
|
1483
|
+
// with known gaps. The checks above ask "where does this block END" — the
|
|
1484
|
+
// question whose edge cases the `DELIBERATELY DOES NOT IMPLEMENT` list is about.
|
|
1485
|
+
// These ask "is this SHAPE present anywhere in the file", which needs no block
|
|
1486
|
+
// extent, no container interior, and no arbiter. So a detector cannot inherit the
|
|
1487
|
+
// defect it warns about. Three of the four review rounds that evaluated this
|
|
1488
|
+
// proposal wrote prototypes and measured the false-positive rate on real corpora
|
|
1489
|
+
// (0/140, 0/151, 0/645) rather than reasoning about it.
|
|
1490
|
+
//
|
|
1491
|
+
// ⚠ THE ONE PLACE THAT CLAIM NEEDS QUALIFYING, stated here rather than left for
|
|
1492
|
+
// a reviewer to find: `unterminatedCommentLine` is combined with
|
|
1493
|
+
// `unclosedHtmlBlockLine` by its caller, which DOES read the classification. That
|
|
1494
|
+
// is deliberate and it is a differential, not a dependency — "there is an
|
|
1495
|
+
// unterminated comment, yet the document-level scan did not flag one" is exactly
|
|
1496
|
+
// what it means for the opener to be inside a container. If the classification is
|
|
1497
|
+
// wrong the detector goes quiet or gets noisy; it cannot manufacture a new silent
|
|
1498
|
+
// pass, because the silent pass is the state that exists TODAY.
|
|
1499
|
+
//
|
|
1500
|
+
// ⚠ Every detector runs on `blankFencedBlocks` output. That masker is a
|
|
1501
|
+
// self-contained fence scanner — it never calls `classifyLines` — so masking
|
|
1502
|
+
// cannot import the boundary bugs either. Fenced examples of these shapes (this
|
|
1503
|
+
// repo's own docs are full of them) therefore do not fire.
|
|
1504
|
+
// ⚠⚠ THAT SENTENCE IS TRUE ONLY FOR A FENCE THE MASKER CAN SEE, and it was stated
|
|
1505
|
+
// without the qualifier for eight rounds (`p084-xv8` finding 5). The masker works
|
|
1506
|
+
// on RAW document lines: `FENCE_OPEN_RE` carries `^ {0,3}`, and nothing strips a
|
|
1507
|
+
// container prefix first. So it misses exactly two things — a fence opened inside a
|
|
1508
|
+
// block quote (`> ` before the backticks), and a fence whose RAW indent is four
|
|
1509
|
+
// spaces or more. A comment inside either still reports `comment-inside-container`.
|
|
1510
|
+
// ⚠ "Four or more spaces of raw indent" is the whole rule, and an earlier version
|
|
1511
|
+
// of this note said "a deep list item's content column", which is wrong twice
|
|
1512
|
+
// (`p084-xv9` finding 3): it happens under an ordinary `- item` as soon as the
|
|
1513
|
+
// fence is indented four, and it does NOT happen at two or three however deep the
|
|
1514
|
+
// list is. Describing an accepted defect inaccurately is worse than not describing
|
|
1515
|
+
// it — the next maintainer acts on the description.
|
|
1516
|
+
// ⚠⚠ THE DIRECTION RECORDED HERE WAS WRONG, AND THE DECISION WAS TAKEN ON IT
|
|
1517
|
+
// (`p084-y2` finding 1). This said the gap costs only NOISE. Measured: an unmasked
|
|
1518
|
+
// fence's content is counted as section body, so quoting an upstream rule as an
|
|
1519
|
+
// example inside a block quote — or at raw indent four — makes the rule look PRESENT
|
|
1520
|
+
// while the live copy of it has been changed. Reproduced end to end: `All checks
|
|
1521
|
+
// passed` over a `_conventions.md` whose rule was switched off. **It costs silence
|
|
1522
|
+
// too.** The maintainer accepted this gap on 2026-08-09 believing it was noise-only;
|
|
1523
|
+
// that premise is retired.
|
|
1524
|
+
//
|
|
1525
|
+
// ⚠⚠ THE DECISION WAS RETAKEN ON THE CORRECTED PREMISE (maintainer, 2026-08-10) AND
|
|
1526
|
+
// THE GAP IS STILL ACCEPTED — as a temporary accepted RISK, owner maintainer, to be
|
|
1527
|
+
// paid before the `doctor-reader-gaps` proposal closes. Not as a resolved defect.
|
|
1528
|
+
// What settled it is that the obvious repair is WORSE, and measurably so
|
|
1529
|
+
// (`p084-sol1`): excluding the quoted lines from the section body **fabricates
|
|
1530
|
+
// markers that were never in the file**. Removal is not monotonic for substring
|
|
1531
|
+
// matching — drop a line and its neighbours become adjacent, and the marker
|
|
1532
|
+
// comparison collapses whitespace, so
|
|
1533
|
+
// `The cascade result is` / <dropped line> / `a floor: rows may only escalate.`
|
|
1534
|
+
// reassembles into the very rule the file no longer states. Reproduced LF and CRLF:
|
|
1535
|
+
// a section that correctly reports `stale` today reports `current` after the repair.
|
|
1536
|
+
// Blanking the line instead of dropping it does not help, for the same reason.
|
|
1537
|
+
// ⚠ AND THE CLASS IS WIDER THAN FENCES, which is the part that makes a local patch
|
|
1538
|
+
// futile: substring matching cannot tell a RULE from PROSE ABOUT THAT RULE. An
|
|
1539
|
+
// indented example, a bare `> ` quotation and an ordinary sentence quoting the rule
|
|
1540
|
+
// all read as the live rule. The unmasked fence is one instance, not the class — so
|
|
1541
|
+
// closing it would move the next maintainer from a disclosed defect to an
|
|
1542
|
+
// undisclosed one, which is this proposal's own most expensive lesson.
|
|
1543
|
+
// The real repair is one walker emitting ordered evidence segments with HARD
|
|
1544
|
+
// boundaries that matching cannot cross, plus a declarative required/forbidden
|
|
1545
|
+
// direction policy — a redesign, out of this proposal's scope, tracked in
|
|
1546
|
+
// `planning/opt-in-backlog.md` under `doctor-reader-gaps`.
|
|
1547
|
+
// Pinned as-is so it cannot change without someone meeting the decision again.
|
|
1548
|
+
//
|
|
1549
|
+
// ⚠ DIRECTION RULE FOR ADDING ONE: a detector may be noisy, never silent. Where
|
|
1550
|
+
// the shape is ambiguous, do not fire. A false warning is visible and the user
|
|
1551
|
+
// can dismiss it; a missed shape restores exactly the silence this exists to
|
|
1552
|
+
// break. Gap D from the proposal is the worked example of the limit — its
|
|
1553
|
+
// prototype false-fired 5 times in 151 files, and it is deliberately NOT built,
|
|
1554
|
+
// because for that shape the warning really is worse than the gap.
|
|
1555
|
+
// ---------------------------------------------------------------------------
|
|
1556
|
+
|
|
1557
|
+
// Which positions on one line sit inside a CommonMark code span. Needed because
|
|
1558
|
+
// `` `<!-- x -->` `` IS rendered — a reader sees it — so flagging it would be a
|
|
1559
|
+
// plain false positive. Measured: the probe's I3 control.
|
|
1560
|
+
//
|
|
1561
|
+
// ⚠ Line-level, and that is a stated limit rather than an oversight: a code span
|
|
1562
|
+
// may legally straddle lines. There the mask under-covers, so a straddling span
|
|
1563
|
+
// could produce a NOISY hit, which is the acceptable direction.
|
|
1564
|
+
//
|
|
1565
|
+
// ⚠⚠ A BACKSLASH SUPPRESSES A BACKTICK ONLY WHILE NO SPAN IS OPEN, and getting
|
|
1566
|
+
// this wrong has produced a silent pass TWICE, from opposite sides
|
|
1567
|
+
// (`p084-xv2`, then `p084-xv3` on the repair). CommonMark §6.1: backslash escapes
|
|
1568
|
+
// do not work inside a code span — so whether a backtick "is escaped" depends on
|
|
1569
|
+
// whether a span is already open, which is the very thing this function computes.
|
|
1570
|
+
// The first repair asked the question as a PRE-PASS, filtering escaped backticks
|
|
1571
|
+
// out before pairing, and that is circular. It cost a new blocker: in
|
|
1572
|
+
// `` `foo\` <!-- rule --> `` the real closing delimiter was discarded as escaped,
|
|
1573
|
+
// the opener paired with a later run instead, and the mask swallowed a comment a
|
|
1574
|
+
// reader never sees while doctor went on reading it.
|
|
1575
|
+
//
|
|
1576
|
+
// So this is a single left-to-right scan, which is what an inline parser actually
|
|
1577
|
+
// does: an escape consumes the next character while looking for an OPENER, and the
|
|
1578
|
+
// search for the CLOSER is raw. `` `foo\` `` is a code span whose content is
|
|
1579
|
+
// `foo\`, exactly as the reference renderer has it.
|
|
1580
|
+
function codeSpanMask(line) {
|
|
1581
|
+
const mask = new Array(line.length).fill(false);
|
|
1582
|
+
let i = 0;
|
|
1583
|
+
while (i < line.length) {
|
|
1584
|
+
// No span is open here, so a backslash consumes the character after it and an
|
|
1585
|
+
// escaped backtick is plain text that opens nothing.
|
|
1586
|
+
if (line[i] === '\\') { i += 2; continue; }
|
|
1587
|
+
if (line[i] !== '`') { i += 1; continue; }
|
|
1588
|
+
let openEnd = i;
|
|
1589
|
+
while (openEnd < line.length && line[openEnd] === '`') openEnd += 1;
|
|
1590
|
+
const len = openEnd - i;
|
|
1591
|
+
// CommonMark pairs a backtick run with the next run of EQUAL length. No
|
|
1592
|
+
// backslash handling in here — see the note above; that is the whole fix.
|
|
1593
|
+
let closeStart = -1;
|
|
1594
|
+
for (let j = openEnd; j < line.length;) {
|
|
1595
|
+
if (line[j] !== '`') { j += 1; continue; }
|
|
1596
|
+
let k = j;
|
|
1597
|
+
while (k < line.length && line[k] === '`') k += 1;
|
|
1598
|
+
if (k - j === len) { closeStart = j; break; }
|
|
1599
|
+
j = k;
|
|
1600
|
+
}
|
|
1601
|
+
// An unpaired run is literal text. Resume just past it rather than past the
|
|
1602
|
+
// whole line, so a later run on the same line can still open a span.
|
|
1603
|
+
if (closeStart === -1) { i = openEnd; continue; }
|
|
1604
|
+
for (let p = i; p < closeStart + len; p += 1) mask[p] = true;
|
|
1605
|
+
i = closeStart + len;
|
|
1606
|
+
}
|
|
1607
|
+
return mask;
|
|
1608
|
+
}
|
|
1609
|
+
|
|
1610
|
+
// Every live `<!--` on a line, as column offsets. CommonMark parses code spans
|
|
1611
|
+
// and backslash escapes only in inline content. Inside a raw-HTML block the line
|
|
1612
|
+
// is passed through as HTML, so neither syntax can make an apparent opener
|
|
1613
|
+
// visible to the reader.
|
|
1614
|
+
function commentOpenersOutsideCode(line, inlineSyntaxApplies = true) {
|
|
1615
|
+
const mask = inlineSyntaxApplies ? codeSpanMask(line) : [];
|
|
1616
|
+
const out = [];
|
|
1617
|
+
for (let at = line.indexOf('<!--'); at !== -1; at = line.indexOf('<!--', at + 4)) {
|
|
1618
|
+
if (mask[at]) continue;
|
|
1619
|
+
// CommonMark backslash escaping is parity-based: an odd run escapes `<`, so
|
|
1620
|
+
// `\<!--` is visible text; an even run leaves `<` live, so `\\<!--` still
|
|
1621
|
+
// opens a comment after rendering one literal backslash. Count at the opener
|
|
1622
|
+
// rather than stripping escapes globally — codeSpanMask owns backticks, and
|
|
1623
|
+
// changing its input would move the offsets the two decisions share.
|
|
1624
|
+
let backslashes = 0;
|
|
1625
|
+
for (let p = at - 1; p >= 0 && line[p] === '\\'; p -= 1) backslashes += 1;
|
|
1626
|
+
if (!inlineSyntaxApplies || backslashes % 2 === 0) out.push(at);
|
|
1627
|
+
}
|
|
1628
|
+
return out;
|
|
1629
|
+
}
|
|
1630
|
+
|
|
1631
|
+
// GAP I and the container half of gap A. Both answer ONE question — **is there a
|
|
1632
|
+
// comment whose text this module still counts as live?** — and they are split only
|
|
1633
|
+
// because the two causes need different repair advice.
|
|
1634
|
+
//
|
|
1635
|
+
// ⚠⚠ THE ARBITER IS `classifyLines`, NOT A REGEX, AND THAT IS THE WHOLE FIX
|
|
1636
|
+
// (`p084-y1` findings 1, 2 and 5). The first version asked
|
|
1637
|
+
// `HTML_COMMENT_OPEN_RE.test(line)` — a DOCUMENT-LEVEL opener rule — to decide
|
|
1638
|
+
// whether a line's leading `<!--` opened a block. Inside a list item it does not:
|
|
1639
|
+
// `- x` / ` <!-- rule -->` is typed `list`, its text stays live, and a reader
|
|
1640
|
+
// sees nothing. That shape fired NOTHING, and it is exactly where the shipped
|
|
1641
|
+
// repair advice ("put the comment on a line of its own") sent the user — out of a
|
|
1642
|
+
// disclosed silent pass and into an undisclosed one. Restating a rule the
|
|
1643
|
+
// classifier already owns is this module's oldest defect; the module's own rule
|
|
1644
|
+
// says callers READ LABELS. So these read labels.
|
|
1645
|
+
//
|
|
1646
|
+
// ⚠ It also retires a detector. The previous `unclosed-html-in-container` fired on
|
|
1647
|
+
// "an unterminated comment that the document-level scan did not flag", which is
|
|
1648
|
+
// not the shape that hurts: measured, ANY later `-->` anywhere in the file —
|
|
1649
|
+
// including the one `unclosed-html-block`'s own action tells the user to add —
|
|
1650
|
+
// silenced it while the rule stayed hidden from readers. A check that keeps being
|
|
1651
|
+
// defeated a new way is mis-scoped, not under-specified. What actually matters is
|
|
1652
|
+
// "is the text hidden from a reader but not from us", and termination has nothing
|
|
1653
|
+
// to do with it.
|
|
1654
|
+
//
|
|
1655
|
+
// ⚠ Lines the classifier calls `code` are skipped in both: a reader sees indented
|
|
1656
|
+
// code verbatim, so there is no divergence to report and a warning there is the
|
|
1657
|
+
// pure noise gap D was rejected for.
|
|
1658
|
+
|
|
1659
|
+
function commentContext(content) {
|
|
1660
|
+
const lines = blankFencedBlocks(String(content));
|
|
1661
|
+
return { lines, info: classifyLines(lines) };
|
|
1662
|
+
}
|
|
1663
|
+
|
|
1664
|
+
// Everything that can legitimately sit between the start of a line and the start of
|
|
1665
|
+
// that line's CONTENT: indentation, block-quote markers, list markers. Built from
|
|
1666
|
+
// the same marker fragments the list predicates use, so widening a marker cannot
|
|
1667
|
+
// leave this behind — the reason `BULLET_MARKER` exists as a fragment at all.
|
|
1668
|
+
//
|
|
1669
|
+
// ⚠ It exists so `> <!-- rule -->` is judged as a comment that STARTS its content
|
|
1670
|
+
// (the container case) rather than as a mid-line one. Getting that wrong picks the
|
|
1671
|
+
// wrong id and prints repair advice for the wrong shape; both ids disclose, so the
|
|
1672
|
+
// cost is a confusing message rather than a silent pass.
|
|
1673
|
+
const CONTAINER_PREFIX_ONLY_RE = new RegExp(`^(?:[ \\t>]|${BULLET_MARKER}|${ORDERED_MARKER})*$`);
|
|
1674
|
+
function startsContent(line, at) {
|
|
1675
|
+
return CONTAINER_PREFIX_ONLY_RE.test(line.slice(0, at));
|
|
1676
|
+
}
|
|
1677
|
+
|
|
1678
|
+
// ⚠⚠ ONE ANSWER TO "IS THIS LINE INSIDE A CONTAINER WHOSE INTERIOR THIS MODULE
|
|
1679
|
+
// DOES NOT PARSE", read by every detector that needs it (`p084-xv3` finding 3).
|
|
1680
|
+
// The comment detectors spelled `list | blockquote | html` and the two boundary
|
|
1681
|
+
// detectors spelled `list | blockquote`, so a `<custom-element>` inside
|
|
1682
|
+
// `<details>` was container content to one and a document-level divergence to the
|
|
1683
|
+
// other — and only the second one warned about it. That is the same
|
|
1684
|
+
// two-rival-predicates shape `commentDisposition` exists to retire, one level up:
|
|
1685
|
+
// retiring it for the comment ids and leaving the neighbours to restate it is how
|
|
1686
|
+
// the class survived. Adding a container type now reaches every reader at once.
|
|
1687
|
+
//
|
|
1688
|
+
// ⚠ Deliberately NOT a block decision. Nothing here decides where a block starts
|
|
1689
|
+
// or ends — `classifyLines` has already done that, and this only reads the label
|
|
1690
|
+
// it produced. Being wrong here makes a warning noisy or absent; it cannot move a
|
|
1691
|
+
// boundary.
|
|
1692
|
+
const UNPARSED_CONTAINER_TYPES = ['list', 'blockquote', 'html'];
|
|
1693
|
+
function inUnparsedContainer(c) {
|
|
1694
|
+
return !!c && UNPARSED_CONTAINER_TYPES.includes(c.type);
|
|
1695
|
+
}
|
|
1696
|
+
|
|
1697
|
+
// GAP I — a comment that begins part-way through a line. It is an inline span, not
|
|
1698
|
+
// a block, so every consumer counts its contents as live document text while a
|
|
1699
|
+
// reader sees nothing. Two shapes: `prose <!-- rule -->`, and a SECOND comment on a
|
|
1700
|
+
// line that opens with one (`<!-- a --> <!-- rule -->`), because only the first
|
|
1701
|
+
// span is ever masked.
|
|
1702
|
+
// ⚠⚠ ONE FUNCTION DECIDES WHICH ID OWNS A LINE, and the two exported detectors are
|
|
1703
|
+
// views over it. Two independent predicates is how the first version left a hole:
|
|
1704
|
+
// `p084-xv1` found a comment at column 0 inside a `<details>` block that fell into
|
|
1705
|
+
// NEITHER id — the container detector skipped it because the line is typed `html`,
|
|
1706
|
+
// the inline detector skipped it because the comment starts the line's content, and
|
|
1707
|
+
// `parseContextLine` happily returned a policy value the reader cannot see. A
|
|
1708
|
+
// partition has to be written as a partition or its gaps are invisible.
|
|
1709
|
+
//
|
|
1710
|
+
// ⚠ `blockStart` is the field that makes this decidable, and it exists because the
|
|
1711
|
+
// previous batch stopped rebuilding what the classifier knows and had it carry the
|
|
1712
|
+
// answer instead. A comment line inside a type-6 block reports the ENCLOSING
|
|
1713
|
+
// block's start, not its own index — which is exactly the difference between "this
|
|
1714
|
+
// comment opened a block, so its text really is hidden" and "this comment is
|
|
1715
|
+
// sitting inside someone else's block, where its text is still read".
|
|
1716
|
+
//
|
|
1717
|
+
// Returns 'inline', 'container', or null (nothing to report on this line).
|
|
1718
|
+
function commentDisposition(lines, info, i) {
|
|
1719
|
+
const c = info[i];
|
|
1720
|
+
if (!c || c.type === 'code') return null; // indented code renders verbatim
|
|
1721
|
+
const line = lines[i];
|
|
1722
|
+
// ⚠⚠ ASK WHAT THE CHECKS ACTUALLY READ, rather than guessing from the line's
|
|
1723
|
+
// type. `invisible` is NOT a whole-line boolean — it pairs with `visibleFrom`,
|
|
1724
|
+
// and `classifiedVisible` blanks only up to that column. So `<!-- a --> <!-- b -->`
|
|
1725
|
+
// is `invisible: true, visibleFrom: 10`: the first span is masked and the tail is
|
|
1726
|
+
// live, which is the documented "only the FIRST span is masked" gap. Treating
|
|
1727
|
+
// `invisible` as "this line is hidden" silently dropped that whole shape — caught
|
|
1728
|
+
// by re-running the earlier round's own fixtures after this function was
|
|
1729
|
+
// rewritten, which is the only reason it did not ship.
|
|
1730
|
+
const hiddenUpTo = c.invisible ? (c.visibleFrom === undefined ? line.length : c.visibleFrom) : 0;
|
|
1731
|
+
// The openers whose text survives into what every check reads. One inside the
|
|
1732
|
+
// masked region cannot mislead anyone and is not a finding.
|
|
1733
|
+
// ⚠ A comment inside a `<textarea>` IS reported here, deliberately — see the
|
|
1734
|
+
// block headed `THERE IS NO TEXTAREA EXEMPTION HERE` for why that exemption was
|
|
1735
|
+
// removed rather than fixed a fourth time. (Grep that phrase — an earlier version of
|
|
1736
|
+
// this pointer named a FUNCTION and was ~700 lines off, `p084-y2` finding 5.)
|
|
1737
|
+
const live = commentOpenersOutsideCode(line, c.type !== 'html' && !c.rawHtml).filter((at) => at >= hiddenUpTo);
|
|
1738
|
+
if (live.length === 0) return null;
|
|
1739
|
+
// A prefix only counts as container prefix when the classifier agrees the line is
|
|
1740
|
+
// in a container. `2. <!-- x -->` continuing a paragraph LOOKS like an ordered
|
|
1741
|
+
// item and is prose (`p084-xv1`), so shape alone picks the wrong id and prints
|
|
1742
|
+
// repair advice about leaving a container the user is not in.
|
|
1743
|
+
return (inUnparsedContainer(c) && startsContent(line, live[0])) ? 'container' : 'inline';
|
|
1744
|
+
}
|
|
1745
|
+
|
|
1746
|
+
function inlineHtmlCommentLine(content) {
|
|
1747
|
+
const { lines, info } = commentContext(content);
|
|
1748
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
1749
|
+
if (commentDisposition(lines, info, i) === 'inline') return i + 1;
|
|
1750
|
+
}
|
|
1751
|
+
return -1;
|
|
1752
|
+
}
|
|
1753
|
+
|
|
1754
|
+
// The container half of GAP A — the comment DOES begin its line, but the line sits
|
|
1755
|
+
// inside a list item or block quote, whose interior this module does not parse as
|
|
1756
|
+
// its own block sequence. So no HTML block opens, the text stays live, and the
|
|
1757
|
+
// reader loses it. The repair differs from the inline case, which is why it has its
|
|
1758
|
+
// own id: moving such a comment "onto its own line" changes nothing.
|
|
1759
|
+
function containerHtmlCommentLine(content) {
|
|
1760
|
+
const { lines, info } = commentContext(content);
|
|
1761
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
1762
|
+
if (commentDisposition(lines, info, i) === 'container') return i + 1;
|
|
1763
|
+
}
|
|
1764
|
+
return -1;
|
|
1765
|
+
}
|
|
1766
|
+
|
|
1767
|
+
// GAP B — HTML block type 7 (a complete tag whose name is NOT in the type-6 list,
|
|
1768
|
+
// alone on a line). Deliberately not implemented: it is the only type that cannot
|
|
1769
|
+
// interrupt a paragraph and recognizing it needs a real tag parser.
|
|
1770
|
+
//
|
|
1771
|
+
// The trigger is narrow and comes from the measured differential rather than from
|
|
1772
|
+
// imagination: a bare `<custom-element>` line at a block start, directly followed
|
|
1773
|
+
// by a setext underline (`---` / `===`). Ours ends the section EARLY where the
|
|
1774
|
+
// reference continues it, so the usual user-visible failure is a false `stale`.
|
|
1775
|
+
// It is worth disclosing, because a `stale` a user cannot reproduce is what
|
|
1776
|
+
// teaches them to stop believing the tool.
|
|
1777
|
+
// ⚠⚠ BUT IT IS NOT ONLY LOUD, AND THIS COMMENT SAID IT WAS FOR MANY ROUNDS
|
|
1778
|
+
// (`p084-y2` finding 3; still stale here after that finding was fixed in the
|
|
1779
|
+
// shipped `detail` and both pages — `p084-sol2` finding 3 caught the leftover).
|
|
1780
|
+
// Ending the section early also DROPS ITS TAIL, and the `CONVENTIONS_RETIRED`
|
|
1781
|
+
// fingerprints report on what is PRESENT in a body — so a retired row below the
|
|
1782
|
+
// shape stops being seen and its finding disappears. Measured: control reports
|
|
1783
|
+
// `ceremony-ef-tweak-t3:retired`, the same row under a `<my-widget>` / `---`
|
|
1784
|
+
// pair reports nothing. Say "unknown in BOTH directions", never "loud".
|
|
1785
|
+
// ⚠ The malformed-table shape has the mirror-image of this defect for the same
|
|
1786
|
+
// reason — they are one mechanism (a moved section boundary) that used to wear
|
|
1787
|
+
// two ids. It no longer has a detector; see GAP C below for why.
|
|
1788
|
+
// ⚠⚠ KEEP THE TWO HALVES OF THIS ID'S `detail` APART, because they say opposite
|
|
1789
|
+
// things and an earlier version of this very comment ran them together. For
|
|
1790
|
+
// type 7 the BOUNDARY moves ONE way — the section ends EARLIER than the reader's
|
|
1791
|
+
// does — while the RESULTS are unknown in BOTH directions, because ending early
|
|
1792
|
+
// drops the tail and a retired row down there stops being seen. "Both directions"
|
|
1793
|
+
// is true of the results and false of the boundary. Do not "fix" the shipped
|
|
1794
|
+
// `detail` into saying the boundary moves both ways; that is a different id's
|
|
1795
|
+
// shape, and this distinction has now been got wrong twice in opposite
|
|
1796
|
+
// directions (`doctor-table-boundary-account` in `planning/opt-in-backlog.md`).
|
|
1797
|
+
// ⚠⚠ A CLOSING TAG OPENS TYPE 7 TOO, and leaving it out made the id's own docs
|
|
1798
|
+
// describe a shape wider than the detector found (`p084-xv5` finding 2).
|
|
1799
|
+
// CommonMark's start condition is "a complete open tag … **or a complete closing
|
|
1800
|
+
// tag**, followed by only whitespace or the end of the line". Measured:
|
|
1801
|
+
// `</my-widget>` above `---` is one raw HTML block to the reference renderer and a
|
|
1802
|
+
// setext heading to this module — the divergence the id exists to disclose — and
|
|
1803
|
+
// the detector returned -1 on it. Undisclosed, which is the direction that matters.
|
|
1804
|
+
// ⚠⚠ THE TWO FORMS NEED DIFFERENT TAILS, and sharing them was a false positive
|
|
1805
|
+
// (`p084-xv6` finding 2). A closing tag may carry NEITHER attributes nor a `/`:
|
|
1806
|
+
// CommonMark's "complete closing tag" is `</`, a name, optional whitespace, `>`.
|
|
1807
|
+
// So `</my-widget attr>` and `</my-widget/>` are ordinary paragraph text, `---`
|
|
1808
|
+
// below them really is a setext heading, and this module ALREADY agrees with the
|
|
1809
|
+
// spec there — nothing diverges, and firing was noise on a shape that is not a tag.
|
|
1810
|
+
// ⚠ The tag-name shape is still spelled ONCE, as a fragment, which is what the
|
|
1811
|
+
// "no second nearly-identical regex" rule actually asks for — the previous version
|
|
1812
|
+
// obeyed the letter of that rule by sharing a tail that does not belong to both.
|
|
1813
|
+
// ⚠ `m[1] || m[2]`: the two branches capture into different groups. A reader
|
|
1814
|
+
// changing this must keep the detector's `name` lookup in step.
|
|
1815
|
+
//
|
|
1816
|
+
// ⚠⚠ THE OPEN-TAG TAIL IS THE SPEC'S GRAMMAR, NOT AN APPROXIMATION OF IT, and the
|
|
1817
|
+
// approximation it replaces produced the exact failure this whole proposal exists
|
|
1818
|
+
// to prevent (`p084-xv7`, blocker). The tail used to be `(?:[ \t][^<>]*)?` — "any
|
|
1819
|
+
// run without angle brackets" — which forbids `>`. CommonMark's raw-HTML grammar
|
|
1820
|
+
// ALLOWS `>` inside a quoted attribute value, so `<my-widget title="a > b">` is a
|
|
1821
|
+
// complete open tag, starts HTML block type 7, and swallows the `---` below it.
|
|
1822
|
+
// This module reads that `---` as a setext heading instead. Measured end to end in
|
|
1823
|
+
// a real project: the divergence was there, no detector fired, and doctor printed
|
|
1824
|
+
// `All checks passed` over a section that was genuinely stale underneath.
|
|
1825
|
+
// The same approximation was loose in the other direction — `<my-widget =x>` and
|
|
1826
|
+
// `<my-widget "x">` are not tags at all, this module agrees with the spec about
|
|
1827
|
+
// them, and the detector was firing anyway.
|
|
1828
|
+
//
|
|
1829
|
+
// So the productions are transcribed rather than guessed, each named after the
|
|
1830
|
+
// spec's own name for it. `[ \t]` and never `\s`, for the reason given at
|
|
1831
|
+
// `ATX_HEADING_RE`; a line-level detector cannot see the spec's multi-line
|
|
1832
|
+
// whitespace and does not need to.
|
|
1833
|
+
const BARE_TAG_NAME = '[A-Za-z][A-Za-z0-9-]*';
|
|
1834
|
+
const HTML_ATTR_NAME = '[A-Za-z_:][A-Za-z0-9_.:-]*';
|
|
1835
|
+
// unquoted | single-quoted | double-quoted. The quoted forms are why `>` cannot be
|
|
1836
|
+
// excluded wholesale.
|
|
1837
|
+
const HTML_ATTR_VALUE = '(?:[^ \\t"\'=<>`]+|\'[^\']*\'|"[^"]*")';
|
|
1838
|
+
const HTML_ATTR = `(?:[ \\t]+${HTML_ATTR_NAME}(?:[ \\t]*=[ \\t]*${HTML_ATTR_VALUE})?)`;
|
|
1839
|
+
const BARE_TAG_LINE_RE = new RegExp(
|
|
1840
|
+
`^ {0,3}<(?:(${BARE_TAG_NAME})${HTML_ATTR}*[ \\t]*\\/?|\\/(${BARE_TAG_NAME})[ \\t]*)>[ \\t]*$`
|
|
1841
|
+
);
|
|
1842
|
+
function htmlBlockType7Line(content) {
|
|
1843
|
+
const lines = blankFencedBlocks(String(content));
|
|
1844
|
+
const info = classifyLines(lines);
|
|
1845
|
+
const type6 = new RegExp(`^(?:${HTML_BLOCK_TAGS})$`, 'i');
|
|
1846
|
+
for (let i = 0; i < lines.length - 1; i += 1) {
|
|
1847
|
+
const m = lines[i].match(BARE_TAG_LINE_RE);
|
|
1848
|
+
const openName = m && m[1]; // complete open tag
|
|
1849
|
+
const closeName = m && m[2]; // complete closing tag
|
|
1850
|
+
const name = openName || closeName;
|
|
1851
|
+
if (!name || type6.test(name)) continue;
|
|
1852
|
+
// ⚠⚠ CommonMark §4.6 attaches the name exclusion to the OPEN-TAG form only:
|
|
1853
|
+
// "a complete open tag (with any tag name other than pre, script, style, or
|
|
1854
|
+
// textarea) **or a complete closing tag**". So `<pre/>` is not type 7 and
|
|
1855
|
+
// `</pre>` is. Relying on the classifier for this was not enough
|
|
1856
|
+
// (`p084-xv8` finding 4): `htmlBlockStart` recognises `<pre>` — the name
|
|
1857
|
+
// followed by space, `>` or end of line — but NOT `<pre/>`, where a `/`
|
|
1858
|
+
// follows the name, so that line stayed `paragraph`, escaped the container
|
|
1859
|
+
// skip, and got reported as a divergence that does not exist.
|
|
1860
|
+
if (openName && HTML_TYPE1_NAME_RE.test(openName)) continue;
|
|
1861
|
+
if (!SETEXT_UNDERLINE_RE.test(lines[i + 1])) continue;
|
|
1862
|
+
// ⚠ Inside a container nothing at DOCUMENT level moves, so the divergence this
|
|
1863
|
+
// id describes does not occur there (`p084-xv2`, widened to HTML blocks by
|
|
1864
|
+
// `p084-xv3`). The shape is about a section boundary shifting; a
|
|
1865
|
+
// `<custom-element>` sitting in a list — or inside `<details>` — is container
|
|
1866
|
+
// content to both readings. Firing there is noise on legitimate content.
|
|
1867
|
+
// ⚠ At document level this shape's line is typed `paragraph`, never `html`
|
|
1868
|
+
// (the module does not implement type 7 — that is the whole gap), so reading
|
|
1869
|
+
// the shared container predicate here cannot silence the detector's real case.
|
|
1870
|
+
if (inUnparsedContainer(info[i])) continue;
|
|
1871
|
+
// ⚠ THE BLOCK-START TEST IS NOT OPTIONAL, and leaving it out was a false
|
|
1872
|
+
// positive both doc pages already excluded in words ("standing alone at the
|
|
1873
|
+
// START of a block") — `p084-y1` finding 7. Type 7 is defined by being the one
|
|
1874
|
+
// HTML block that CANNOT interrupt a paragraph, so a bare tag continuing a
|
|
1875
|
+
// paragraph is not type 7 at all: `some prose` / `<my-widget>` / `---` is a
|
|
1876
|
+
// setext heading to CommonMark and this module already agrees with that. Firing
|
|
1877
|
+
// there reports a divergence that does not exist.
|
|
1878
|
+
if (i > 0 && info[i - 1] && info[i - 1].type === 'paragraph') continue;
|
|
1879
|
+
return i + 1;
|
|
1880
|
+
}
|
|
1881
|
+
return -1;
|
|
1882
|
+
}
|
|
1883
|
+
|
|
1884
|
+
// GAP C — a table whose delimiter row carries a different number of cells from
|
|
1885
|
+
// its header. GFM treats that pair as ordinary prose (it is not a table at all)
|
|
1886
|
+
// while `classifyLines` opens a table, so the two can disagree about where a
|
|
1887
|
+
// section ends, and `conventionsSectionBodies` then reads a different span than
|
|
1888
|
+
// the reader sees.
|
|
1889
|
+
//
|
|
1890
|
+
// ⚠⚠ THERE IS DELIBERATELY NO DETECTOR FOR THIS (user decision, 2026-08-12), and
|
|
1891
|
+
// the reason is worth the space because the obvious next move is to write one.
|
|
1892
|
+
// P-084 shipped one and then spent SIX review rounds on it. Every round found a
|
|
1893
|
+
// document it stayed silent on, and silence is the direction that prints
|
|
1894
|
+
// `All checks passed` over a file that has drifted:
|
|
1895
|
+
// `p084gate-x7` the bare mismatched pair
|
|
1896
|
+
// `p084gate-y3` a multi-line prose run before a dash underline
|
|
1897
|
+
// `p084gate-y4` the whole equals-underline family
|
|
1898
|
+
// `x13`/`y5` a prose run whose FIRST line is underline-shaped
|
|
1899
|
+
// `x14`/`y6` a prose run whose first line is INDENTED
|
|
1900
|
+
// `p084gate-y7` a delimiter row written with single hyphens, which GFM
|
|
1901
|
+
// accepts and `TABLE_DELIMITER_ROW_RE` does not
|
|
1902
|
+
// The first five were an enumeration inside the detector, and removing that
|
|
1903
|
+
// enumeration did not help: the sixth was the same failure living in the
|
|
1904
|
+
// delimiter-row recogniser instead. **An enumeration cannot be finished from the
|
|
1905
|
+
// inside**, and each unfinished edge fails silently.
|
|
1906
|
+
// ⚠ What settled it was a measurement rather than fatigue. `p084gate-y6`
|
|
1907
|
+
// reproduced the identical silent false clean with a delimiter row whose cell
|
|
1908
|
+
// count was CORRECT. So the harm is section-boundary disagreement, the detector's
|
|
1909
|
+
// scope was cell-count mismatch, and the second is a strict subset of the first —
|
|
1910
|
+
// no version of this detector, narrow or broad, could close the harm it was
|
|
1911
|
+
// written for.
|
|
1912
|
+
// ⚠⚠ The durable fix is a different instrument: `marked` is already a runtime
|
|
1913
|
+
// dependency of this package (`dflow render` uses it), so the section boundary
|
|
1914
|
+
// can be taken FROM the renderer rather than guessed alongside it and compared
|
|
1915
|
+
// shape by shape. That is a design change with its own review. It is recorded,
|
|
1916
|
+
// with owner and gate, as `doctor-section-boundary-arbiter` in
|
|
1917
|
+
// `planning/opt-in-backlog.md`, and both explainer pages name this gap under
|
|
1918
|
+
// "shapes that are known and deliberately not reported".
|
|
1919
|
+
// ⚠ `TABLE_DELIMITER_ROW_RE` stays — `classifyLines` and
|
|
1920
|
+
// `hasTableWithoutConventionComment` both use it. Its `-{2,}` is narrower than
|
|
1921
|
+
// GFM's `-{1,}` (`p084gate-y7`); that is now a classifier property with no
|
|
1922
|
+
// detector resting on it, and widening it would change what counts as a table
|
|
1923
|
+
// across this module.
|
|
1924
|
+
// Kept for callers that only need the first match.
|
|
1925
|
+
function conventionsSectionBody(content, heading) {
|
|
1926
|
+
const bodies = conventionsSectionBodies(content, heading);
|
|
1927
|
+
return bodies.length === 0 ? null : bodies[0];
|
|
1928
|
+
}
|
|
1929
|
+
|
|
1930
|
+
// Each `marker` is verified present in BOTH packaged templates (or, where the
|
|
1931
|
+
// section is edition-specific, in the edition that has it).
|
|
1932
|
+
// `test/upgrade-drift.mjs` re-derives that from the packaged templates rather
|
|
1933
|
+
// than trusting this comment.
|
|
1934
|
+
//
|
|
1935
|
+
// ⚠ BE PRECISE ABOUT WHICH HALF IS MECHANICAL. This sentence used to also claim
|
|
1936
|
+
// each marker is "absent from the pre-P082 shape", with the same
|
|
1937
|
+
// "re-derives that" clause covering both halves. Only the presence half is
|
|
1938
|
+
// re-derived; nothing in the tree checks the absence half (`p082-b3-g1`
|
|
1939
|
+
// finding 4). The risk it leaves is real and worth naming: a marker phrase that
|
|
1940
|
+
// ALREADY existed in the old text would make an un-migrated file report
|
|
1941
|
+
// `current` — silence about a stale file, the false-negative direction.
|
|
1942
|
+
// It is not mechanized because the only source for the old shape is git
|
|
1943
|
+
// history, and a test that shells out to git cannot run from the published
|
|
1944
|
+
// tarball, where there is no `.git`. So it stays a review obligation: when you
|
|
1945
|
+
// add a marker, check it against the pre-P082 text yourself.
|
|
1946
|
+
// `missingLevel` is per-fingerprint on purpose. `### SPEC-ID Format` was
|
|
1947
|
+
// re-parented by 59e0eb2 and no released version ever projected it, so EVERY
|
|
1948
|
+
// existing project is legitimately `missing` there — info. The other two
|
|
1949
|
+
// sections HAVE always shipped, so their absence means the file was edited or
|
|
1950
|
+
// predates them, which is worth a warn. A blanket info also mis-reported a
|
|
1951
|
+
// merely RENAMED heading as "projects created before this section shipped do
|
|
1952
|
+
// not have it at all", which is a false statement about a file that has it.
|
|
1953
|
+
const CONVENTIONS_FINGERPRINTS = [
|
|
1954
|
+
{
|
|
1955
|
+
id: 'ceremony-escalate-only',
|
|
1956
|
+
heading: 'Ceremony Scaling (Project Application)',
|
|
1957
|
+
marker: 'cascade result is a floor',
|
|
1958
|
+
editions: null,
|
|
1959
|
+
missingLevel: 'warn',
|
|
1960
|
+
rule: 'the escalate-only rule (project rows may raise a tier, never lower it)',
|
|
1961
|
+
consequence: 'Without it the table reads as a free re-classification surface, so a project row can lower a tier the cascade raised — the drift PROPOSAL-082 closed.'
|
|
1962
|
+
},
|
|
1963
|
+
{
|
|
1964
|
+
id: 'spec-files-no-br-families',
|
|
1965
|
+
heading: 'Filling the Templates',
|
|
1966
|
+
marker: 'no-BR family variants',
|
|
1967
|
+
editions: null,
|
|
1968
|
+
missingLevel: 'warn',
|
|
1969
|
+
rule: 'the no-BR family variants of `lightweight-spec.md`',
|
|
1970
|
+
consequence: 'Without them the table implies a T2 always carries a BR delta, so a presentation / contract / operational / performance change gets a fabricated BR-ID to satisfy a rule that no longer exists.'
|
|
1971
|
+
},
|
|
1972
|
+
{
|
|
1973
|
+
id: 'spec-id-minimal-host',
|
|
1974
|
+
heading: 'SPEC-ID Format',
|
|
1975
|
+
marker: 'Minimal (zero-phase) host exception',
|
|
1976
|
+
editions: null,
|
|
1977
|
+
missingLevel: 'info',
|
|
1978
|
+
neverProjected: true,
|
|
1979
|
+
rule: 'the minimal (zero-phase) host exception',
|
|
1980
|
+
consequence: 'Without it the section still says the SPEC-ID appears in the first phase-spec filename and the branch name, which a zero-phase host has no way to satisfy — and a functional-bug host carries a BUG-NUMBER branch instead.'
|
|
1981
|
+
}
|
|
1982
|
+
];
|
|
1983
|
+
|
|
1984
|
+
// The second fingerprint kind: strings PROPOSAL-082 retired, which must not
|
|
1985
|
+
// still be sitting in an adopter's file. These need no new upstream contract —
|
|
1986
|
+
// the evidence is that batch 1 deleted them from both packaged templates — so
|
|
1987
|
+
// unlike the gf DDD-Modeling-Depth fingerprint (which has no marker to pin and
|
|
1988
|
+
// is recorded as debt 10), there is nothing blocking them.
|
|
1989
|
+
//
|
|
1990
|
+
// Every string below is verified zero-match in both packaged templates AND both
|
|
1991
|
+
// tutorial fixtures; `test/upgrade-drift.mjs` re-derives that rather than
|
|
1992
|
+
// trusting this comment. A retired string that still matched the current
|
|
1993
|
+
// template would fire on every project forever.
|
|
1994
|
+
const CONVENTIONS_RETIRED = [
|
|
1995
|
+
{
|
|
1996
|
+
id: 'ceremony-ef-tweak-t3',
|
|
1997
|
+
heading: 'Ceremony Scaling (Project Application)',
|
|
1998
|
+
retired: 'T3 if no Domain change',
|
|
1999
|
+
editions: null,
|
|
2000
|
+
rule: 'the retired "EF configuration tweak → T3 if no Domain change" example row',
|
|
2001
|
+
consequence: 'Under the ordered cascade an Infrastructure mapping tweak is not automatically T3 — the packaged template now uses that row to show a project ESCALATING it, and the old row teaches the opposite.'
|
|
2002
|
+
},
|
|
2003
|
+
{
|
|
2004
|
+
id: 'ceremony-query-only-t2',
|
|
2005
|
+
heading: 'Ceremony Scaling (Project Application)',
|
|
2006
|
+
// Anchored through the TIER cell, not just the situation. The situation
|
|
2007
|
+
// alone flagged `| Adding a Query only (no write) | T1 (project convention) |
|
|
2008
|
+
// We escalate all new reads |` — a row an adopter migrated CORRECTLY into
|
|
2009
|
+
// the form the current template teaches — and then told them to overwrite it.
|
|
2010
|
+
// The claim is about a row that restates a bare tier, so the anchor has to
|
|
2011
|
+
// include the bare tier. Sibling `ceremony-ef-tweak-t3` never had this
|
|
2012
|
+
// problem because its anchor was already the tier cell.
|
|
2013
|
+
retired: 'Adding a Query only (no write)} | T2',
|
|
2014
|
+
editions: null,
|
|
2015
|
+
rule: 'the retired "Adding a Query only (no write) → T2" example row',
|
|
2016
|
+
consequence: 'The cascade decides this now; the row restated a tier instead of recording a project decision, which is the drift the escalate-only rule was added to stop.'
|
|
2017
|
+
},
|
|
2018
|
+
{
|
|
2019
|
+
id: 'ceremony-criteria-table',
|
|
2020
|
+
heading: 'Ceremony Scaling (Project Application)',
|
|
2021
|
+
// Anchored on enough of the retired sentence to be implausible as adopter
|
|
2022
|
+
// prose. Two shorter anchors were tried and both trip on legitimate text:
|
|
2023
|
+
// `criteria table` matches "Our own criteria table below records the
|
|
2024
|
+
// escalations we apply", and `for the full criteria table` still matches
|
|
2025
|
+
// "see that document for the full criteria table" — a generic five-word
|
|
2026
|
+
// English construction. The retired text was "See `AI-AGENT-GUIDE.md`
|
|
2027
|
+
// § Ceremony Scaling for the full criteria table.", soft-wrapped between
|
|
2028
|
+
// "full" and "criteria"; `normalizeForMarker` collapses that, so including
|
|
2029
|
+
// the § name costs nothing and no longer matches ordinary sentences.
|
|
2030
|
+
retired: 'Ceremony Scaling for the full criteria table',
|
|
2031
|
+
editions: null,
|
|
2032
|
+
rule: 'the retired pointer to a "criteria table" in AI-AGENT-GUIDE.md',
|
|
2033
|
+
consequence: 'That table was replaced by the ordered cascade. A file still pointing at it sends the reader to a section that no longer exists under that name.'
|
|
2034
|
+
},
|
|
2035
|
+
{
|
|
2036
|
+
id: 'spec-files-br-delta-equation',
|
|
2037
|
+
heading: 'Filling the Templates',
|
|
2038
|
+
retired: 'bug fix / small tweak with BR Delta',
|
|
2039
|
+
editions: null,
|
|
2040
|
+
rule: 'the retired "T2 Light — bug fix / small tweak with BR Delta" row',
|
|
2041
|
+
consequence: 'It equates T2 with carrying a BR Delta, which the no-BR families retired — the same equation that made people fabricate BR-IDs.'
|
|
2042
|
+
}
|
|
2043
|
+
];
|
|
2044
|
+
|
|
2045
|
+
// Markers are matched against whitespace-normalized text. `_conventions.md` is
|
|
2046
|
+
// user-owned and reflowing a paragraph is an ordinary edit; matching raw made a
|
|
2047
|
+
// soft-wrap inside the marker report a false `stale`.
|
|
2048
|
+
function normalizeForMarker(text) {
|
|
2049
|
+
return String(text).replace(/\s+/g, ' ');
|
|
2050
|
+
}
|
|
2051
|
+
|
|
2052
|
+
// ⚠ INHERENT LIMIT, stated rather than papered over: these are heuristic
|
|
2053
|
+
// substring checks (PROPOSAL-082 specifies them as such — 字串偵測, following
|
|
2054
|
+
// the P058 starter-drift precedent). A substring cannot tell a rule from a
|
|
2055
|
+
// sentence ABOUT the rule, so text like "we removed the old rule that the
|
|
2056
|
+
// cascade result is a floor" satisfies the fingerprint and reports `current`.
|
|
2057
|
+
// That is a false negative on a deliberately-worded file, not on drift, and no
|
|
2058
|
+
// amount of matching harder fixes it — deciding it needs to read meaning.
|
|
2059
|
+
// Do not describe this check as proving a project's conventions are correct;
|
|
2060
|
+
// it detects the shapes that arise from a file predating a contract change.
|
|
2061
|
+
//
|
|
2062
|
+
// ⚠ The `retired` kind fails in the OPPOSITE direction, and the two must not be
|
|
2063
|
+
// conflated. A `marker` that is too short yields a false NEGATIVE (silence); a
|
|
2064
|
+
// `retired` string that is too short yields a false POSITIVE — doctor telling a
|
|
2065
|
+
// developer their own sentence is retired Dflow text. So a retired entry must
|
|
2066
|
+
// be anchored on enough of the original sentence to be implausible as adopter
|
|
2067
|
+
// prose, and `test/upgrade-drift.mjs` asserts each is zero-match against both
|
|
2068
|
+
// packaged templates and both tutorial fixtures.
|
|
2069
|
+
//
|
|
2070
|
+
// Returns one entry per fingerprint that is not `current`. `edition` selects
|
|
2071
|
+
// edition-specific fingerprints; pass null to evaluate only the shared ones.
|
|
2072
|
+
// Whether an entry applies to the edition being checked. `editions: null` means
|
|
2073
|
+
// "shared, applies everywhere"; a list means the entry is only meaningful for
|
|
2074
|
+
// those editions and a caller that did not resolve an edition must not see it.
|
|
2075
|
+
//
|
|
2076
|
+
// Extracted from the two identical inline conditions below because every shipped
|
|
2077
|
+
// entry currently has `editions: null`, which makes the non-null path dead and
|
|
2078
|
+
// therefore unproven (`p082-b3-g1` finding 5). It is NOT speculative — debt 10
|
|
2079
|
+
// (the greenfield DDD-Modeling-Depth fingerprint) is a known future
|
|
2080
|
+
// edition-specific entry, blocked only on an upstream marker to pin. So the
|
|
2081
|
+
// branch is kept and tested here instead of being deleted and rewritten
|
|
2082
|
+
// untested at the moment someone is busy thinking about something else.
|
|
2083
|
+
function fingerprintAppliesTo(fp, edition) {
|
|
2084
|
+
if (!fp.editions) return true;
|
|
2085
|
+
return Boolean(edition) && fp.editions.includes(edition);
|
|
2086
|
+
}
|
|
2087
|
+
|
|
2088
|
+
function findConventionsDrift(content, edition = null) {
|
|
2089
|
+
// No early return for empty input. It used to short-circuit to `[]`, which
|
|
2090
|
+
// made a whitespace-only file report NOTHING while a one-character file
|
|
2091
|
+
// reported all three sections missing — the emptier file getting the cleaner
|
|
2092
|
+
// verdict, and the gate silently accepting absence, which is the exact thing
|
|
2093
|
+
// this module says it exists to prevent. Empty content genuinely has every
|
|
2094
|
+
// section missing, so it says so; the caller collapses that into one finding.
|
|
2095
|
+
const text = String(content || '');
|
|
2096
|
+
const drift = [];
|
|
2097
|
+
for (const fp of CONVENTIONS_FINGERPRINTS) {
|
|
2098
|
+
if (!fingerprintAppliesTo(fp, edition)) continue;
|
|
2099
|
+
const bodies = conventionsSectionBodies(text, fp.heading);
|
|
2100
|
+
const base = {
|
|
2101
|
+
id: fp.id,
|
|
2102
|
+
heading: fp.heading,
|
|
2103
|
+
rule: fp.rule,
|
|
2104
|
+
consequence: fp.consequence,
|
|
2105
|
+
neverProjected: Boolean(fp.neverProjected)
|
|
2106
|
+
};
|
|
2107
|
+
if (bodies.length === 0) {
|
|
2108
|
+
drift.push({ ...base, state: 'missing', level: fp.missingLevel || 'warn' });
|
|
2109
|
+
continue;
|
|
2110
|
+
}
|
|
2111
|
+
const marker = normalizeForMarker(fp.marker);
|
|
2112
|
+
// Any copy of the section carrying the rule is enough.
|
|
2113
|
+
if (!bodies.some((body) => normalizeForMarker(body).includes(marker))) {
|
|
2114
|
+
drift.push({ ...base, state: 'stale', level: 'warn' });
|
|
2115
|
+
}
|
|
2116
|
+
}
|
|
2117
|
+
for (const fp of CONVENTIONS_RETIRED) {
|
|
2118
|
+
if (!fingerprintAppliesTo(fp, edition)) continue;
|
|
2119
|
+
const bodies = conventionsSectionBodies(text, fp.heading);
|
|
2120
|
+
// A section that is absent cannot carry a retired string. Its absence is
|
|
2121
|
+
// already reported by the `requires` fingerprint on the same heading, so
|
|
2122
|
+
// saying nothing here is not silence about an unchecked state.
|
|
2123
|
+
if (bodies.length === 0) continue;
|
|
2124
|
+
const retired = normalizeForMarker(fp.retired);
|
|
2125
|
+
if (bodies.some((body) => normalizeForMarker(body).includes(retired))) {
|
|
2126
|
+
drift.push({
|
|
2127
|
+
id: fp.id,
|
|
2128
|
+
state: 'retired',
|
|
2129
|
+
level: 'warn',
|
|
2130
|
+
heading: fp.heading,
|
|
2131
|
+
rule: fp.rule,
|
|
2132
|
+
consequence: fp.consequence,
|
|
2133
|
+
neverProjected: false
|
|
2134
|
+
});
|
|
2135
|
+
}
|
|
2136
|
+
}
|
|
2137
|
+
return drift;
|
|
2138
|
+
}
|
|
2139
|
+
|
|
2140
|
+
// --- PROPOSAL-092: shape markers and template skeletons ------------------------
|
|
2141
|
+
//
|
|
2142
|
+
// Every template an adopter's document is created from carries one line naming
|
|
2143
|
+
// its track, its template and the number of the shape it has:
|
|
2144
|
+
//
|
|
2145
|
+
// <!-- dflow-shape: greenfield/rules.md 1 — keep this line: dflow doctor reads it -->
|
|
2146
|
+
//
|
|
2147
|
+
// The document inherits the line when it is created from the template, so
|
|
2148
|
+
// doctor can compare the number with the template's current one. Same number:
|
|
2149
|
+
// every difference between the document and the template is the adopter's own
|
|
2150
|
+
// decision, and nothing is reported — which is the distinction a comparison
|
|
2151
|
+
// alone cannot make (six OBTS `rules.md` files written one consistent way, not
|
|
2152
|
+
// the template's). Older number: the registry (`lib/doc-shapes.json`) says what
|
|
2153
|
+
// changed between the two numbers.
|
|
2154
|
+
//
|
|
2155
|
+
// ⚠ Only the three fields are read. Whatever follows the number up to `-->` is
|
|
2156
|
+
// the note for the reader and is not validated: an AI that re-words the note has
|
|
2157
|
+
// not changed what the line records.
|
|
2158
|
+
const crypto = require('node:crypto');
|
|
2159
|
+
|
|
2160
|
+
const SHAPE_MARKER_MENTION = 'dflow-shape:';
|
|
2161
|
+
// The note after the number stops at a line break of either kind: in a file
|
|
2162
|
+
// with lone-CR line breaks it would otherwise swallow the next line — another
|
|
2163
|
+
// marker included — and pass the doc as read.
|
|
2164
|
+
const SHAPE_MARKER_RE = /^<!-- dflow-shape: ([a-z]+)\/([A-Za-z0-9_.-]+\.md) ([1-9][0-9]*)(?: [^\r\n]*)? -->[ \t]*$/;
|
|
2165
|
+
|
|
2166
|
+
// Stage one of two: every line that mentions `dflow-shape:`, wherever it sits —
|
|
2167
|
+
// in a fence, a comment, a list, a quote, an HTML block.
|
|
2168
|
+
// ⚠ Doctor does not decide which of those lines is live (PROPOSAL-092 D2).
|
|
2169
|
+
// Deciding means reading every container exactly the way `dflow render`
|
|
2170
|
+
// (`marked`) reads it, and the two readings are hard to keep aligned. A mention
|
|
2171
|
+
// anywhere but the standard position makes the doc unreadable instead: saying
|
|
2172
|
+
// "cannot tell" never reports a doc doctor did not read as passing.
|
|
2173
|
+
// ⚠ Finding and validating stay separate: validating while finding would let one
|
|
2174
|
+
// valid line plus one broken line count as "exactly one marker", and would let a
|
|
2175
|
+
// doc whose only marker is broken count as unmarked.
|
|
2176
|
+
function findShapeMarkerLines(content) {
|
|
2177
|
+
const found = [];
|
|
2178
|
+
String(content).split('\n').forEach((line, i) => {
|
|
2179
|
+
if (line.includes(SHAPE_MARKER_MENTION)) found.push({ line: i + 1, text: line.replace(/\r$/, '') });
|
|
2180
|
+
});
|
|
2181
|
+
return found;
|
|
2182
|
+
}
|
|
2183
|
+
|
|
2184
|
+
// The one line a marker is read from: line 1, or the line after the
|
|
2185
|
+
// frontmatter's closing `---` — the frontmatter rule `dflow render` uses, which
|
|
2186
|
+
// is also where every template puts its marker. 1-based.
|
|
2187
|
+
// ⚠ Found in the text as it is, a byte-order mark included: render does not
|
|
2188
|
+
// skip one, so a BOM in front of `---` leaves the file with no frontmatter — for
|
|
2189
|
+
// render, and so for doctor. Skipping it here alone would read a marker that
|
|
2190
|
+
// render shows as body text.
|
|
2191
|
+
function shapeMarkerStandardLine(content) {
|
|
2192
|
+
return frontmatterLineCount(String(content).split('\n')) + 1;
|
|
2193
|
+
}
|
|
2194
|
+
|
|
2195
|
+
// Stage two. `absent`: no line mentions it. `ok`: exactly one line mentions it,
|
|
2196
|
+
// that line is the standard one, and it parses. Anything else is `unreadable`,
|
|
2197
|
+
// with the reason — `multiple`, `misplaced` (`standard` is the line it belongs
|
|
2198
|
+
// on; `bomHidesFrontmatter` when a BOM is what hides the frontmatter above it)
|
|
2199
|
+
// or `malformed`. Whether the track and template name exist is the caller's
|
|
2200
|
+
// question: it needs the registry, and this module does no I/O.
|
|
2201
|
+
function readShapeMarker(content) {
|
|
2202
|
+
const raw = String(content);
|
|
2203
|
+
// A BOM is not a line: line numbers and the marker's own text are read without it.
|
|
2204
|
+
const text = raw.replace(/^\uFEFF/, '');
|
|
2205
|
+
const found = findShapeMarkerLines(text);
|
|
2206
|
+
if (found.length === 0) return { state: 'absent' };
|
|
2207
|
+
if (found.length > 1) {
|
|
2208
|
+
return { state: 'unreadable', reason: 'multiple', lines: found.map((f) => f.line) };
|
|
2209
|
+
}
|
|
2210
|
+
const standard = shapeMarkerStandardLine(raw);
|
|
2211
|
+
if (found[0].line !== standard) {
|
|
2212
|
+
const misplaced = { state: 'unreadable', reason: 'misplaced', lines: [found[0].line], standard };
|
|
2213
|
+
if (text !== raw && frontmatterLineCount(text.split('\n')) > 0) misplaced.bomHidesFrontmatter = true;
|
|
2214
|
+
return misplaced;
|
|
2215
|
+
}
|
|
2216
|
+
const m = found[0].text.match(SHAPE_MARKER_RE);
|
|
2217
|
+
const number = m ? Number(m[3]) : NaN;
|
|
2218
|
+
if (!m || !Number.isSafeInteger(number)) {
|
|
2219
|
+
return { state: 'unreadable', reason: 'malformed', lines: [found[0].line] };
|
|
2220
|
+
}
|
|
2221
|
+
return { state: 'ok', track: m[1], template: m[2], number, line: found[0].line };
|
|
2222
|
+
}
|
|
2223
|
+
|
|
2224
|
+
// The number of lines the frontmatter occupies, fences included; 0 when there
|
|
2225
|
+
// is none. The SAME rule as `splitFrontmatter` in `lib/render.js`, which the
|
|
2226
|
+
// marker placement was measured against: the document's first line starts with
|
|
2227
|
+
// `---`, and the frontmatter ends at the next line that is `---` once trimmed;
|
|
2228
|
+
// with no such line there is no frontmatter. `test/doc-shapes.mjs` pins the two
|
|
2229
|
+
// against each other on every packaged template and on the line-ending cases.
|
|
2230
|
+
// ⚠ Callers pass the text split on `\n` ONLY — render's split, not
|
|
2231
|
+
// `LINE_SPLIT_RE`. With the module's own splitter a lone-CR file had
|
|
2232
|
+
// frontmatter here and none in render, so "the same rule" was two rules.
|
|
2233
|
+
// ⚠ Nothing else in this module knows about frontmatter, and `classifyLines`
|
|
2234
|
+
// reads it as ordinary blocks: a commented field like `# follow-up-of:` is an
|
|
2235
|
+
// H1 to it and the closing `---` underlines an H2. So the skeleton cuts it off
|
|
2236
|
+
// before classifying anything.
|
|
2237
|
+
function frontmatterLineCount(lines) {
|
|
2238
|
+
if (lines.length === 0 || !lines[0].startsWith('---')) return 0;
|
|
2239
|
+
for (let i = 1; i < lines.length; i += 1) {
|
|
2240
|
+
if (lines[i].trim() === '---') return i + 1;
|
|
2241
|
+
}
|
|
2242
|
+
return 0;
|
|
2243
|
+
}
|
|
2244
|
+
|
|
2245
|
+
// A field the template carries commented out: `# name: …`, indented or not,
|
|
2246
|
+
// with the name spelled as a key would be (quoted, dotted), as long as it is
|
|
2247
|
+
// one token — `# a note: …` in prose is not a field.
|
|
2248
|
+
const FRONTMATTER_COMMENTED_FIELD_RE = /^[ \t]*#[ \t]*("[^"\n]*"|'[^'\n]*'|[A-Za-z][A-Za-z0-9_.-]*)[ \t]*:/;
|
|
2249
|
+
// A `{…}` placeholder anywhere in a heading makes it an example, not a section
|
|
2250
|
+
// the template requires (`## {Feature Area 1}`, `### FL-01: {流程名稱}`).
|
|
2251
|
+
const SHAPE_PLACEHOLDER_RE = /\{[^}\n]*\}/;
|
|
2252
|
+
|
|
2253
|
+
function shapeHeadingText(text) {
|
|
2254
|
+
return String(text)
|
|
2255
|
+
.replace(/<!--[\s\S]*?-->/g, '')
|
|
2256
|
+
.replace(/<!--.*$/, '')
|
|
2257
|
+
.replace(/[ \t]+/g, ' ')
|
|
2258
|
+
.trim();
|
|
2259
|
+
}
|
|
2260
|
+
|
|
2261
|
+
// The frontmatter's field lines and the body after it, cut the way `dflow
|
|
2262
|
+
// render` cuts them (see `frontmatterLineCount`).
|
|
2263
|
+
function splitShapeFrontmatter(content) {
|
|
2264
|
+
const text = String(content);
|
|
2265
|
+
const lines = text.split('\n');
|
|
2266
|
+
const count = frontmatterLineCount(lines);
|
|
2267
|
+
return {
|
|
2268
|
+
fields: count === 0 ? [] : lines.slice(1, count - 1).map((l) => l.replace(/\r$/, '')),
|
|
2269
|
+
body: count === 0 ? text : lines.slice(count).join('\n')
|
|
2270
|
+
};
|
|
2271
|
+
}
|
|
2272
|
+
|
|
2273
|
+
// A frontmatter field is whatever `splitFrontmatter` in `lib/render.js` reads
|
|
2274
|
+
// as a key — the text before the line's first `:`, trimmed, on a line that has
|
|
2275
|
+
// one and does not start with `#` — so a field render shows on the meta card
|
|
2276
|
+
// (`"review-owner":`, `review.owner:`) is never missing from the skeleton.
|
|
2277
|
+
// `test/doc-shapes.mjs` pins the two against each other. Null for any other line.
|
|
2278
|
+
function frontmatterFieldName(line) {
|
|
2279
|
+
if (line.trimStart().startsWith('#')) return null;
|
|
2280
|
+
const colon = line.indexOf(':');
|
|
2281
|
+
return colon === -1 ? null : line.slice(0, colon).trim();
|
|
2282
|
+
}
|
|
2283
|
+
|
|
2284
|
+
function tableHeaderCells(line) {
|
|
2285
|
+
// Comments go before the split: a `|` inside one is not a column boundary.
|
|
2286
|
+
let row = line.replace(/<!--[\s\S]*?-->/g, '').trim();
|
|
2287
|
+
if (row.startsWith('|')) row = row.slice(1);
|
|
2288
|
+
if (row.endsWith('|') && !row.endsWith('\\|')) row = row.slice(0, -1);
|
|
2289
|
+
return row.split(/(?<!\\)\|/).map((cell) => shapeHeadingText(cell));
|
|
2290
|
+
}
|
|
2291
|
+
|
|
2292
|
+
// The shape of a template: what a shape number stands for (PROPOSAL-092 D6).
|
|
2293
|
+
// Items are small arrays so they compare, sort and print without a parser:
|
|
2294
|
+
//
|
|
2295
|
+
// ['field', name] frontmatter field
|
|
2296
|
+
// ['field', name, 'optional'] field the template carries commented out
|
|
2297
|
+
// ['h2', heading]
|
|
2298
|
+
// ['h3', h2, heading] h2 is '' when the H3 sits above any H2
|
|
2299
|
+
// ['column', h2, h3, table, position, heading]
|
|
2300
|
+
// a table's header cell; `table` counts
|
|
2301
|
+
// from 1 within that (h2, h3) section, and
|
|
2302
|
+
// `position` from 1 within the row
|
|
2303
|
+
// ['note', h2, h3, digest] the `>` notes of that section
|
|
2304
|
+
// ['comment', h2, h3, digest] its HTML comments: every one that starts
|
|
2305
|
+
// a line's content — at the top level or in
|
|
2306
|
+
// a list item, multi-line ones included —
|
|
2307
|
+
// and the ones in its heading line
|
|
2308
|
+
//
|
|
2309
|
+
// The first four are the skeleton; notes, comments and the heading order are
|
|
2310
|
+
// the shape's notes-and-order half — a change there does not change the doc's
|
|
2311
|
+
// structure, and doctor reports it as such. A note or comment digest is taken
|
|
2312
|
+
// over its text with `>` and line breaks folded to single spaces, so re-wrapping
|
|
2313
|
+
// is not a change. A comment counts as its `<!-- … -->` span only: prose after
|
|
2314
|
+
// `-->` on the same line is prose. A comment in a `>` block is part of that
|
|
2315
|
+
// note. The marker line is not a comment of the shape.
|
|
2316
|
+
//
|
|
2317
|
+
// ⚠ The column POSITION is part of the item: a table's columns are read by
|
|
2318
|
+
// position, so swapping `Upstream` and `Downstream` reverses what every row
|
|
2319
|
+
// means while leaving the set of header names the same. Items compare
|
|
2320
|
+
// order-free; the heading order is compared on its own (`shapeOrderChanges`).
|
|
2321
|
+
//
|
|
2322
|
+
// Left out: H1 and H4 and below, prose outside notes and comments, fixed label
|
|
2323
|
+
// text, table rows, and every `##` / `###` heading with a placeholder —
|
|
2324
|
+
// together with everything under it, down to the next heading at its level or
|
|
2325
|
+
// above. A table or comment under an H4 is recorded under the nearest H2 / H3,
|
|
2326
|
+
// placeholder H4 or not: an H4 is not part of the skeleton, so it cannot take
|
|
2327
|
+
// what follows it out of the shape either.
|
|
2328
|
+
// `variable` lists the sections a template lets the author replace (`[h2]` or
|
|
2329
|
+
// `[h2, h3]`); those and everything under them are left out as well.
|
|
2330
|
+
// ⚠ `test/doc-shapes.mjs` holds this reading against `dflow render`'s parser on
|
|
2331
|
+
// every registered template: the headings, tables, notes and comments render
|
|
2332
|
+
// sees must be the ones recorded here. A `>` block inside a list item is one
|
|
2333
|
+
// render sees and this does not read, so a template may not carry one.
|
|
2334
|
+
function extractShapeSkeleton(content, options) {
|
|
2335
|
+
const { items, sections } = extractShapeSections(content, options);
|
|
2336
|
+
for (const t of sections) {
|
|
2337
|
+
if (t.note.length > 0) items.push(['note', t.h2, t.h3, shapeTextDigest(t.note)]);
|
|
2338
|
+
if (t.comment.length > 0) items.push(['comment', t.h2, t.h3, shapeTextDigest(t.comment)]);
|
|
2339
|
+
}
|
|
2340
|
+
return items;
|
|
2341
|
+
}
|
|
2342
|
+
|
|
2343
|
+
// The `<!-- … -->` spans in `text` that a shape records — never the prose around
|
|
2344
|
+
// them. An unterminated one runs to the end of the text. The marker is not one.
|
|
2345
|
+
function shapeCommentSpans(text) {
|
|
2346
|
+
return (String(text).match(/<!--[\s\S]*?(?:-->|$)/g) || [])
|
|
2347
|
+
.filter((span) => !span.includes(SHAPE_MARKER_MENTION));
|
|
2348
|
+
}
|
|
2349
|
+
|
|
2350
|
+
// The skeleton items, and per section the raw note and comment texts its
|
|
2351
|
+
// digests are taken over (`extractShapeSkeleton` adds the digests).
|
|
2352
|
+
function extractShapeSections(content, { variable = [] } = {}) {
|
|
2353
|
+
const { fields, body: bodyText } = splitShapeFrontmatter(content);
|
|
2354
|
+
const items = [];
|
|
2355
|
+
for (const line of fields) {
|
|
2356
|
+
const name = frontmatterFieldName(line);
|
|
2357
|
+
if (name !== null) { items.push(['field', name]); continue; }
|
|
2358
|
+
const m = line.match(FRONTMATTER_COMMENTED_FIELD_RE);
|
|
2359
|
+
if (m) items.push(['field', m[1], 'optional']);
|
|
2360
|
+
}
|
|
2361
|
+
const variableKeys = new Set(variable.map((v) => JSON.stringify(v)));
|
|
2362
|
+
const body = blankFencedBlocks(bodyText);
|
|
2363
|
+
const classes = classifyLines(body);
|
|
2364
|
+
let h2 = '';
|
|
2365
|
+
let h3 = '';
|
|
2366
|
+
let placeholderLevel = 0;
|
|
2367
|
+
let skipped = false;
|
|
2368
|
+
const tables = new Map();
|
|
2369
|
+
// Per section, in the order sections first appear: its notes and comments.
|
|
2370
|
+
const texts = new Map();
|
|
2371
|
+
const textsHere = () => {
|
|
2372
|
+
const key = JSON.stringify([h2, h3]);
|
|
2373
|
+
if (!texts.has(key)) texts.set(key, { h2, h3, note: [], comment: [] });
|
|
2374
|
+
return texts.get(key);
|
|
2375
|
+
};
|
|
2376
|
+
const commentStillOpen = (t) => t.lastIndexOf('<!--') > t.lastIndexOf('-->');
|
|
2377
|
+
for (let i = 0; i < body.length; i += 1) {
|
|
2378
|
+
const c = classes[i];
|
|
2379
|
+
if (c.heading) {
|
|
2380
|
+
const { level } = c.heading;
|
|
2381
|
+
const text = shapeHeadingText(c.heading.text);
|
|
2382
|
+
if (placeholderLevel && level <= placeholderLevel) placeholderLevel = 0;
|
|
2383
|
+
if (placeholderLevel) continue;
|
|
2384
|
+
if ((level === 2 || level === 3) && SHAPE_PLACEHOLDER_RE.test(text)) {
|
|
2385
|
+
placeholderLevel = level;
|
|
2386
|
+
continue;
|
|
2387
|
+
}
|
|
2388
|
+
if (level === 1) {
|
|
2389
|
+
h2 = '';
|
|
2390
|
+
h3 = '';
|
|
2391
|
+
skipped = false;
|
|
2392
|
+
} else if (level === 2) {
|
|
2393
|
+
h2 = text;
|
|
2394
|
+
h3 = '';
|
|
2395
|
+
skipped = variableKeys.has(JSON.stringify([h2]));
|
|
2396
|
+
if (!skipped) items.push(['h2', h2]);
|
|
2397
|
+
} else if (level === 3) {
|
|
2398
|
+
h3 = text;
|
|
2399
|
+
skipped = variableKeys.has(JSON.stringify([h2])) || variableKeys.has(JSON.stringify([h2, h3]));
|
|
2400
|
+
if (!skipped) items.push(['h3', h2, h3]);
|
|
2401
|
+
}
|
|
2402
|
+
if (!skipped) textsHere().comment.push(...shapeCommentSpans(c.heading.text));
|
|
2403
|
+
continue;
|
|
2404
|
+
}
|
|
2405
|
+
if (placeholderLevel || skipped) continue;
|
|
2406
|
+
if (c.type === 'blockquote') {
|
|
2407
|
+
// The line is a blockquote by `classifyLines`; its `>` markers go — all of
|
|
2408
|
+
// them, so re-wrapping a nested `> >` note is not a change either.
|
|
2409
|
+
const part = body[i].trimStart().replace(/^(?:>[ \t]?)+/, '');
|
|
2410
|
+
const notes = textsHere().note;
|
|
2411
|
+
if (i > 0 && classes[i - 1].type === 'blockquote' && notes.length > 0) notes[notes.length - 1] += `\n${part}`;
|
|
2412
|
+
else notes.push(part);
|
|
2413
|
+
continue;
|
|
2414
|
+
}
|
|
2415
|
+
// An HTML block: the comments in it, whatever else it holds.
|
|
2416
|
+
if (c.type === 'html' && c.blockStart === i) {
|
|
2417
|
+
let end = i;
|
|
2418
|
+
while (end + 1 < body.length && classes[end + 1].type === 'html' && classes[end + 1].blockStart === i) end += 1;
|
|
2419
|
+
textsHere().comment.push(...shapeCommentSpans(body.slice(i, end + 1).join('\n')));
|
|
2420
|
+
i = end;
|
|
2421
|
+
continue;
|
|
2422
|
+
}
|
|
2423
|
+
// A comment that starts a list line's content (` <!-- … -->` under an item,
|
|
2424
|
+
// or `- <!-- … -->`): a block of its own inside the item, as render reads it.
|
|
2425
|
+
// It runs on until it closes, across blank lines, as a comment does.
|
|
2426
|
+
if (c.type === 'list') {
|
|
2427
|
+
const at = body[i].indexOf('<!--');
|
|
2428
|
+
if (at >= 0 && startsContent(body[i], at) && !body[i].slice(0, at).includes('>')) {
|
|
2429
|
+
let end = i;
|
|
2430
|
+
let chunk = body[i].slice(at);
|
|
2431
|
+
while (commentStillOpen(chunk) && end + 1 < body.length && (classes[end + 1].type === 'list' || classes[end + 1].type === 'blank')) {
|
|
2432
|
+
end += 1;
|
|
2433
|
+
chunk += `\n${body[end]}`;
|
|
2434
|
+
}
|
|
2435
|
+
textsHere().comment.push(...shapeCommentSpans(chunk));
|
|
2436
|
+
i = end;
|
|
2437
|
+
continue;
|
|
2438
|
+
}
|
|
2439
|
+
}
|
|
2440
|
+
const startsTable = c.type === 'table' && (i === 0 || classes[i - 1].type !== 'table');
|
|
2441
|
+
if (!startsTable) continue;
|
|
2442
|
+
const section = JSON.stringify([h2, h3]);
|
|
2443
|
+
const table = (tables.get(section) || 0) + 1;
|
|
2444
|
+
tables.set(section, table);
|
|
2445
|
+
tableHeaderCells(body[i]).forEach((cell, k) => items.push(['column', h2, h3, table, k + 1, cell]));
|
|
2446
|
+
}
|
|
2447
|
+
return { items, sections: [...texts.values()] };
|
|
2448
|
+
}
|
|
2449
|
+
|
|
2450
|
+
// A section's notes or comments, folded so that re-wrapping changes nothing:
|
|
2451
|
+
// every run of whitespace is one space. The order of the parts is kept.
|
|
2452
|
+
function shapeTextDigest(parts) {
|
|
2453
|
+
const folded = parts.map((p) => String(p).replace(/\s+/g, ' ').trim()).join('\n');
|
|
2454
|
+
return crypto.createHash('sha256').update(folded, 'utf8').digest('hex').slice(0, 16);
|
|
2455
|
+
}
|
|
2456
|
+
|
|
2457
|
+
// The heading order a shape records: the H2s, and the H3s under each H2 ('' for
|
|
2458
|
+
// H3s above the first H2). The `null` key holds the H2 list itself.
|
|
2459
|
+
function shapeHeadingOrder(skeleton) {
|
|
2460
|
+
const order = new Map([[null, []]]);
|
|
2461
|
+
for (const item of skeleton) {
|
|
2462
|
+
if (item[0] === 'h2') order.get(null).push(item[1]);
|
|
2463
|
+
else if (item[0] === 'h3') {
|
|
2464
|
+
if (!order.has(item[1])) order.set(item[1], []);
|
|
2465
|
+
order.get(item[1]).push(item[2]);
|
|
2466
|
+
}
|
|
2467
|
+
}
|
|
2468
|
+
return order;
|
|
2469
|
+
}
|
|
2470
|
+
|
|
2471
|
+
// Where the headings both shapes have stand in a different order: `null` for
|
|
2472
|
+
// the H2s, an H2's name for the H3s under it. A heading only one shape has is
|
|
2473
|
+
// left out — adding or removing a section is a change of its own, not a
|
|
2474
|
+
// reordering.
|
|
2475
|
+
function shapeOrderChanges(from, to) {
|
|
2476
|
+
const before = shapeHeadingOrder(from);
|
|
2477
|
+
const after = shapeHeadingOrder(to);
|
|
2478
|
+
const shared = (list, other) => [...new Set(list.filter((h) => other.includes(h)))];
|
|
2479
|
+
const changed = [];
|
|
2480
|
+
for (const [parent, list] of before) {
|
|
2481
|
+
const other = after.get(parent);
|
|
2482
|
+
if (!other) continue;
|
|
2483
|
+
if (JSON.stringify(shared(list, other)) !== JSON.stringify(shared(other, list))) changed.push(parent);
|
|
2484
|
+
}
|
|
2485
|
+
return changed;
|
|
2486
|
+
}
|
|
2487
|
+
|
|
2488
|
+
function describeShapeOrder(parent) {
|
|
2489
|
+
if (parent === null) return 'the order of the `##` sections';
|
|
2490
|
+
return parent ? `the order of the \`###\` sections under \`## ${parent}\`` : 'the order of the `###` sections above the first `##`';
|
|
2491
|
+
}
|
|
2492
|
+
|
|
2493
|
+
// Multiset difference: what `to` has that `from` does not, and the reverse.
|
|
2494
|
+
function skeletonDifference(from, to) {
|
|
2495
|
+
const left = new Map();
|
|
2496
|
+
for (const item of from) {
|
|
2497
|
+
const key = JSON.stringify(item);
|
|
2498
|
+
left.set(key, (left.get(key) || 0) + 1);
|
|
2499
|
+
}
|
|
2500
|
+
const added = [];
|
|
2501
|
+
for (const item of to) {
|
|
2502
|
+
const key = JSON.stringify(item);
|
|
2503
|
+
const n = left.get(key) || 0;
|
|
2504
|
+
if (n > 0) left.set(key, n - 1);
|
|
2505
|
+
else added.push(item);
|
|
2506
|
+
}
|
|
2507
|
+
const removed = [];
|
|
2508
|
+
for (const [key, n] of left) {
|
|
2509
|
+
for (let k = 0; k < n; k += 1) removed.push(JSON.parse(key));
|
|
2510
|
+
}
|
|
2511
|
+
return { added, removed };
|
|
2512
|
+
}
|
|
2513
|
+
|
|
2514
|
+
// The items order-free, plus the heading order: a section moved without being
|
|
2515
|
+
// renamed is a new shape too (its notes-and-order half). `test/doc-shapes.mjs`
|
|
2516
|
+
// recomputes this for every registered number, which is what makes a rewrite of
|
|
2517
|
+
// a published number visible.
|
|
2518
|
+
function shapeDigest(skeleton) {
|
|
2519
|
+
const canonical = skeleton.map((item) => JSON.stringify(item)).sort().join('\n');
|
|
2520
|
+
const order = JSON.stringify([...shapeHeadingOrder(skeleton)]);
|
|
2521
|
+
return crypto.createHash('sha256').update(`${canonical}\n${order}`, 'utf8').digest('hex');
|
|
2522
|
+
}
|
|
2523
|
+
|
|
2524
|
+
// Notes, comments and heading order: the part of a shape that does not change
|
|
2525
|
+
// a doc's structure (PROPOSAL-092 D4's third class).
|
|
2526
|
+
function isShapeTextItem(item) {
|
|
2527
|
+
return Array.isArray(item) && (item[0] === 'note' || item[0] === 'comment');
|
|
2528
|
+
}
|
|
2529
|
+
|
|
2530
|
+
function describeShapeItem(item) {
|
|
2531
|
+
const [kind] = item;
|
|
2532
|
+
if (kind === 'field') return `frontmatter field \`${item[1]}:\`${item[2] === 'optional' ? ' (optional)' : ''}`;
|
|
2533
|
+
if (kind === 'h2') return `\`## ${item[1]}\``;
|
|
2534
|
+
if (kind === 'h3') return item[1] ? `\`### ${item[2]}\` under \`## ${item[1]}\`` : `\`### ${item[2]}\``;
|
|
2535
|
+
if (kind === 'column') {
|
|
2536
|
+
const where = [item[1] && `\`## ${item[1]}\``, item[2] && `\`### ${item[2]}\``].filter(Boolean).join(' › ') || 'the top of the document';
|
|
2537
|
+
return `column \`${item[5]}\` (position ${item[4]}) in table ${item[3]} under ${where}`;
|
|
2538
|
+
}
|
|
2539
|
+
if (isShapeTextItem(item)) {
|
|
2540
|
+
const where = [item[1] && `\`## ${item[1]}\``, item[2] && `\`### ${item[2]}\``].filter(Boolean).join(' › ');
|
|
2541
|
+
const what = kind === 'note' ? 'the `>` notes' : 'the HTML comments';
|
|
2542
|
+
return where ? `${what} under ${where}` : `${what} at the top of the document`;
|
|
2543
|
+
}
|
|
2544
|
+
return JSON.stringify(item);
|
|
2545
|
+
}
|
|
2546
|
+
|
|
2547
|
+
// `pattern` is relative to `dflow/specs/`; `*` matches within one path segment.
|
|
2548
|
+
function shapePathMatches(pattern, relPath) {
|
|
2549
|
+
const re = new RegExp(`^${pattern.split('*').map(escapeRegExp).join('[^/]*')}$`);
|
|
2550
|
+
return re.test(relPath);
|
|
2551
|
+
}
|
|
2552
|
+
|
|
2553
|
+
// The data rows of the first table under the H2 named `heading`, or null when
|
|
2554
|
+
// there is no such section or it holds no table. Used to recognise a zero-phase
|
|
2555
|
+
// host from its `_index.md`: an empty `## Phase Specs` table is what the flows
|
|
2556
|
+
// themselves test (`finish-feature-flow.md` Step 1 certifies a host that carries
|
|
2557
|
+
// a phase as phase-bearing).
|
|
2558
|
+
function sectionTableRowCount(content, heading) {
|
|
2559
|
+
const body = blankFencedBlocks(splitShapeFrontmatter(content).body);
|
|
2560
|
+
const classes = classifyLines(body);
|
|
2561
|
+
const want = headingKey(heading);
|
|
2562
|
+
let inside = false;
|
|
2563
|
+
for (let i = 0; i < body.length; i += 1) {
|
|
2564
|
+
const c = classes[i];
|
|
2565
|
+
if (c.heading && c.heading.level <= 2) {
|
|
2566
|
+
if (inside) return null;
|
|
2567
|
+
inside = c.heading.level === 2 && headingKey(c.heading.text) === want;
|
|
2568
|
+
continue;
|
|
2569
|
+
}
|
|
2570
|
+
if (!inside || c.type !== 'table') continue;
|
|
2571
|
+
let end = i;
|
|
2572
|
+
while (end + 1 < body.length && classes[end + 1].type === 'table') end += 1;
|
|
2573
|
+
return Math.max(0, end - i - 1);
|
|
2574
|
+
}
|
|
2575
|
+
return null;
|
|
2576
|
+
}
|
|
2577
|
+
|
|
2578
|
+
module.exports = {
|
|
2579
|
+
GIT_POLICY_LINE_RE,
|
|
2580
|
+
AI_COMMIT_MARKER_LINE_RE,
|
|
2581
|
+
PROSE_LANGUAGE_LINE_RE,
|
|
2582
|
+
TECH_STACK_ROW_RE,
|
|
2583
|
+
MIGRATION_CONTEXT_ROW_RE,
|
|
2584
|
+
GIT_POLICY_VALUES,
|
|
2585
|
+
AI_COMMIT_MARKER_VALUES,
|
|
2586
|
+
parseContextLine,
|
|
2587
|
+
blankFencedBlocks,
|
|
2588
|
+
extractHeadings,
|
|
2589
|
+
extractH2Headings,
|
|
2590
|
+
extractSectionRefs,
|
|
2591
|
+
headingResolves,
|
|
2592
|
+
missingTemplateSections,
|
|
2593
|
+
matchesTemplateWithPlaceholders,
|
|
2594
|
+
SPEC_FORMATTING_CONVENTION_SNIPPET,
|
|
2595
|
+
hasTableWithoutConventionComment,
|
|
2596
|
+
conventionsSectionBody,
|
|
2597
|
+
conventionsSectionBodies,
|
|
2598
|
+
CONVENTIONS_FINGERPRINTS,
|
|
2599
|
+
CONVENTIONS_RETIRED,
|
|
2600
|
+
// Exported so the edition filter can be tested directly. Every shipped entry
|
|
2601
|
+
// has `editions: null`, so `findConventionsDrift` alone can never exercise the
|
|
2602
|
+
// non-null path — a test written through it would prove nothing.
|
|
2603
|
+
fingerprintAppliesTo,
|
|
2604
|
+
isThematicBreak,
|
|
2605
|
+
// The single block-structure decision point (debt 12). Exported so
|
|
2606
|
+
// `lib/init.js` can locate headings through it instead of hand-rolling a
|
|
2607
|
+
// fourth ATX expression, and so `test/upgrade-drift.mjs` can assert labels
|
|
2608
|
+
// directly rather than only through what a section body happens to contain.
|
|
2609
|
+
parseAtxHeading,
|
|
2610
|
+
classifyLines,
|
|
2611
|
+
visibleTextLines,
|
|
2612
|
+
maskCodeBlocks,
|
|
2613
|
+
unclosedHtmlBlockLine,
|
|
2614
|
+
// Exported so `lib/init.js` reads the rule instead of writing a third copy of
|
|
2615
|
+
// `/^[ \t]*$/` inline. The commit that introduced this constant claimed it and
|
|
2616
|
+
// `stripSpaceTab` were "the only two spellings now"; they were not, because
|
|
2617
|
+
// neither was reachable from the other file.
|
|
2618
|
+
BLANK_LINE_RE,
|
|
2619
|
+
stripSpaceTab,
|
|
2620
|
+
findConventionsDrift,
|
|
2621
|
+
// PROPOSAL-084 uncertainty detectors. Each returns a 1-based line number or -1
|
|
2622
|
+
// and asks only whether a SHAPE is present, never where a block ends — see the
|
|
2623
|
+
// section header above them for why that is what makes them safe to run on top
|
|
2624
|
+
// of a parser with named gaps. Exported individually (rather than behind one
|
|
2625
|
+
// aggregate) so `test/upgrade-drift.mjs` can pin each shape and each
|
|
2626
|
+
// false-positive control against the detector that owns it: an aggregate that
|
|
2627
|
+
// returns "something fired" cannot fail for the right reason.
|
|
2628
|
+
inlineHtmlCommentLine,
|
|
2629
|
+
containerHtmlCommentLine,
|
|
2630
|
+
htmlBlockType7Line,
|
|
2631
|
+
// Exported for the false-positive controls specifically: the code-span rule is
|
|
2632
|
+
// the boundary that keeps `` `<!-- x -->` `` — which a reader DOES see — out of
|
|
2633
|
+
// the inline-comment detector, and it deserves a direct pin rather than being
|
|
2634
|
+
// tested only through its two consumers.
|
|
2635
|
+
codeSpanMask,
|
|
2636
|
+
// PROPOSAL-092.
|
|
2637
|
+
findShapeMarkerLines,
|
|
2638
|
+
readShapeMarker,
|
|
2639
|
+
shapeMarkerStandardLine,
|
|
2640
|
+
frontmatterLineCount,
|
|
2641
|
+
extractShapeSkeleton,
|
|
2642
|
+
extractShapeSections,
|
|
2643
|
+
shapeCommentSpans,
|
|
2644
|
+
shapeHeadingText,
|
|
2645
|
+
skeletonDifference,
|
|
2646
|
+
shapeHeadingOrder,
|
|
2647
|
+
shapeOrderChanges,
|
|
2648
|
+
shapeDigest,
|
|
2649
|
+
isShapeTextItem,
|
|
2650
|
+
describeShapeItem,
|
|
2651
|
+
describeShapeOrder,
|
|
2652
|
+
shapePathMatches,
|
|
2653
|
+
sectionTableRowCount
|
|
2654
|
+
};
|