dflow-sdd-ddd 0.13.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/CHANGELOG.md +824 -1
  2. package/CONTRIBUTING.md +16 -10
  3. package/README.en.md +156 -200
  4. package/README.md +89 -144
  5. package/TEMPLATE-COVERAGE.md +15 -8
  6. package/TEMPLATE-LANGUAGE-GLOSSARY.md +15 -1
  7. package/bin/dflow.js +36 -4
  8. package/docs/commands.en.md +110 -0
  9. package/docs/commands.md +101 -0
  10. package/docs/doctor-uncertainty.en.md +212 -0
  11. package/docs/doctor-uncertainty.md +212 -0
  12. package/docs/evaluating-dflow.en.md +29 -11
  13. package/docs/evaluating-dflow.md +8 -6
  14. package/docs/npm-publish-checklist.md +3 -1
  15. package/docs/release-versioning-policy.md +8 -2
  16. package/docs/upgrading.en.md +196 -0
  17. package/docs/upgrading.md +197 -0
  18. package/docs/using-with-claude-code.en.md +25 -10
  19. package/docs/using-with-claude-code.md +20 -7
  20. package/docs/using-with-codex.en.md +18 -6
  21. package/docs/using-with-codex.md +16 -5
  22. package/docs/using-with-github-copilot.en.md +25 -10
  23. package/docs/using-with-github-copilot.md +21 -8
  24. package/lib/doc-shapes.json +997 -0
  25. package/lib/doctor-checks.js +2654 -0
  26. package/lib/init.js +3583 -107
  27. package/lib/render-diagrams.js +1474 -0
  28. package/lib/render.js +865 -49
  29. package/package.json +2 -2
  30. package/templates/brownfield/references/drift-verification.md +4 -0
  31. package/templates/brownfield/references/finish-feature-flow.md +635 -88
  32. package/templates/brownfield/references/finish-feature-follow-up.md +60 -0
  33. package/templates/brownfield/references/finish-feature-minimal-host.md +406 -0
  34. package/templates/brownfield/references/finish-feature-post-hoc-hotfix.md +95 -0
  35. package/templates/brownfield/references/git-integration.md +160 -15
  36. package/templates/brownfield/references/init-project-flow.md +26 -4
  37. package/templates/brownfield/references/modify-existing-flow.md +412 -87
  38. package/templates/brownfield/references/modify-existing-follow-up.md +121 -0
  39. package/templates/brownfield/references/modify-existing-post-hoc-hotfix.md +82 -0
  40. package/templates/brownfield/references/new-feature-flow.md +61 -6
  41. package/templates/brownfield/references/new-phase-flow.md +57 -7
  42. package/templates/brownfield/references/pr-review-checklist.md +303 -10
  43. package/templates/brownfield/scaffolding/AI-AGENT-GUIDE.md +158 -34
  44. package/templates/brownfield/scaffolding/CLAUDE-md-snippet.md +1 -1
  45. package/templates/brownfield/scaffolding/Git-principles-gitflow.md +75 -6
  46. package/templates/brownfield/scaffolding/Git-principles-trunk.md +82 -8
  47. package/templates/brownfield/scaffolding/_conventions.md +50 -28
  48. package/templates/brownfield/scaffolding/_overview.md +1 -0
  49. package/templates/brownfield/templates/_index.md +151 -7
  50. package/templates/brownfield/templates/analysis.md +79 -0
  51. package/templates/brownfield/templates/behavior.md +1 -0
  52. package/templates/brownfield/templates/context-definition.md +1 -0
  53. package/templates/brownfield/templates/context-map.md +2 -1
  54. package/templates/brownfield/templates/glossary.md +1 -0
  55. package/templates/brownfield/templates/lightweight-spec.md +154 -11
  56. package/templates/brownfield/templates/models.md +1 -0
  57. package/templates/brownfield/templates/phase-spec.md +9 -1
  58. package/templates/brownfield/templates/rules.md +1 -0
  59. package/templates/brownfield/templates/tech-debt.md +1 -0
  60. package/templates/common/references/ddd-modeling-guide.md +33 -16
  61. package/templates/{greenfield → common}/references/dflow-feedback-flow.md +2 -1
  62. package/templates/common/references/flow-rationale-registry.md +130 -0
  63. package/templates/common/skill/SKILL.md +13 -11
  64. package/templates/greenfield/references/drift-verification.md +4 -0
  65. package/templates/greenfield/references/finish-feature-flow.md +625 -89
  66. package/templates/greenfield/references/finish-feature-follow-up.md +60 -0
  67. package/templates/greenfield/references/finish-feature-minimal-host.md +363 -0
  68. package/templates/greenfield/references/finish-feature-post-hoc-hotfix.md +95 -0
  69. package/templates/greenfield/references/git-integration.md +148 -15
  70. package/templates/greenfield/references/init-project-flow.md +28 -8
  71. package/templates/greenfield/references/modify-existing-flow.md +378 -85
  72. package/templates/greenfield/references/modify-existing-follow-up.md +103 -0
  73. package/templates/greenfield/references/modify-existing-post-hoc-hotfix.md +82 -0
  74. package/templates/greenfield/references/new-feature-flow.md +67 -4
  75. package/templates/greenfield/references/new-phase-flow.md +56 -7
  76. package/templates/greenfield/references/pr-review-checklist.md +287 -8
  77. package/templates/greenfield/scaffolding/AI-AGENT-GUIDE.md +153 -32
  78. package/templates/greenfield/scaffolding/CLAUDE-md-snippet.md +5 -2
  79. package/templates/greenfield/scaffolding/Git-principles-gitflow.md +74 -6
  80. package/templates/greenfield/scaffolding/Git-principles-trunk.md +88 -12
  81. package/templates/greenfield/scaffolding/_conventions.md +50 -28
  82. package/templates/greenfield/scaffolding/_overview.md +6 -2
  83. package/templates/greenfield/templates/_index.md +137 -7
  84. package/templates/greenfield/templates/aggregate-design.md +1 -0
  85. package/templates/greenfield/templates/analysis.md +79 -0
  86. package/templates/greenfield/templates/behavior.md +1 -0
  87. package/templates/greenfield/templates/context-definition.md +1 -0
  88. package/templates/greenfield/templates/context-map.md +2 -1
  89. package/templates/greenfield/templates/events.md +4 -1
  90. package/templates/greenfield/templates/glossary.md +1 -0
  91. package/templates/greenfield/templates/lightweight-spec.md +154 -11
  92. package/templates/greenfield/templates/models.md +1 -0
  93. package/templates/greenfield/templates/phase-spec.md +9 -1
  94. package/templates/greenfield/templates/rules.md +1 -0
  95. package/templates/greenfield/templates/tech-debt.md +1 -0
  96. package/templates/brownfield/references/dflow-feedback-flow.md +0 -251
@@ -0,0 +1,2654 @@
1
+ // PROPOSAL-058: pure, shippable drift-detection helpers for `dflow doctor`.
2
+ //
3
+ // The dev-only cross-ref resolver (scripts/check-cross-refs.mjs, PROPOSAL-055)
4
+ // is not part of the npm package, so the runtime checks reimplement the narrow
5
+ // subset doctor needs: fence-aware heading extraction, "<file> § Heading"
6
+ // reference extraction with soft-wrap joining, tolerant heading matching, and
7
+ // template-shape comparison. Everything here is I/O-free — callers read the
8
+ // files and pass contents — which keeps the checks unit-testable.
9
+
10
+ 'use strict';
11
+
12
+ // Machine-readable context lines. Shared with lib/init.js inference so the
13
+ // doctor "machine format" checks can never drift from what inference actually
14
+ // parses: inferGitPolicy / inferAiCommitMarker / inferProseLanguage read
15
+ // _conventions.md; inferTechStackSummary / inferMigrationContext read the
16
+ // `| Tech stack |` / `| Migration / legacy context |` rows of the guide's
17
+ // "## Project Context" table (PROPOSAL-076 — no packaged _overview.md template
18
+ // ever carried those rows).
19
+ const GIT_POLICY_LINE_RE = /Selected Git policy:\s*`([^`]+)`/;
20
+ const AI_COMMIT_MARKER_LINE_RE = /AI commit marker:\s*`([^`]+)`/;
21
+ const PROSE_LANGUAGE_LINE_RE = /Project prose language:\s*`([^`]+)`/;
22
+ // The row values accept Markdown-escaped pipes (`\|`) so a cell like
23
+ // `Node \| Express` is captured whole; parseContextLine unescapes them
24
+ // (PROPOSAL-076 gate G1 — a bare `[^|\n]` capture silently truncated at the
25
+ // escaped pipe while doctor still called the row machine-readable).
26
+ const TECH_STACK_ROW_RE = /\|\s*Tech stack\s*\|\s*((?:\\\||[^|\n])+?)\s*\|/i;
27
+ const MIGRATION_CONTEXT_ROW_RE = /\|\s*Migration \/ legacy context\s*\|\s*((?:\\\||[^|\n])+?)\s*\|/i;
28
+
29
+ const GIT_POLICY_VALUES = new Set(['gitflow', 'trunk']);
30
+ const AI_COMMIT_MARKER_VALUES = new Set(['none', 'co-authored-by', 'prefix']);
31
+
32
+ // Trimmed capture of a machine-readable context line, or null when the line is
33
+ // absent or its value is whitespace-only. Inference and the doctor checks must
34
+ // both parse through this helper: a value doctor accepts has to be the exact
35
+ // value configure-agents consumes (e.g. a stray space inside the backticks
36
+ // would otherwise pass doctor's trimmed validation yet miss strict comparisons
37
+ // like buildSubstitutionMap's `gitflow` check downstream).
38
+ function parseContextLine(content, re) {
39
+ // ⚠ Read only text a reader can see. A commented-out
40
+ // `<!-- AI commit marker: `prefix` -->` was parsed as though it were the live
41
+ // setting, so doctor reported the policy satisfied by a line the user had
42
+ // deliberately switched off — silent pass (`p082-b3-k1`). Callers pass whole
43
+ // files and single rows alike; a row with no HTML block in it is unchanged.
44
+ //
45
+ // ⚠⚠ THE TYPE CHECK IS PART OF THAT FIX, NOT TIDINESS. This function used to
46
+ // reach `content.match(re)` directly, so a caller handing it `undefined` got a
47
+ // loud TypeError. Routing through the projection would have swallowed that:
48
+ // `classifiedVisible` calls `String(content)`, turning `undefined` into the
49
+ // four-character text "undefined", which matches nothing and returns null —
50
+ // and null here means "the policy line is absent", a finding about the user's
51
+ // file caused by a bug in ours. Trading a crash for a silent wrong answer is
52
+ // the exact direction this module exists to avoid, so the crash is kept.
53
+ if (typeof content !== 'string') {
54
+ throw new TypeError(`parseContextLine expects a string, got ${typeof content}`);
55
+ }
56
+ const match = visibleTextLines(content).join('\n').match(re);
57
+ if (!match) return null;
58
+ // Unescape Markdown-escaped pipes captured by the table-row patterns; the
59
+ // backtick context lines never contain `\|`, so this is a no-op for them.
60
+ const value = match[1].replace(/\\\|/g, '|').trim();
61
+ return value === '' ? null : value;
62
+ }
63
+
64
+ // CommonMark's line-ending definition, and the reason it is a named constant:
65
+ // every splitter in this file must use the SAME one. `/\r?\n/` handled LF and
66
+ // CRLF but not a lone CR, while the sibling checks in `lib/init.js` match with
67
+ // the `m` flag — and JS's `m` treats a lone `\r` as a line terminator. So a
68
+ // CR-only file was one giant line to this module and a normal file to them, and
69
+ // the two reported different things about it (`p082-b3-g1` finding 3).
70
+ const LINE_SPLIT_RE = /\r\n|\r|\n/;
71
+
72
+ // A thematic break: three or more `-`, `_` or `*`, optionally separated by
73
+ // spaces or tabs.
74
+ //
75
+ // ⚠ `-` is a thematic break at 3+ and a setext underline at ANY length, so the
76
+ // two roles overlap and callers must not treat this as "is an underline".
77
+ // `=` is NOT here on purpose: a lone `=+` run is not a thematic break at all —
78
+ // at a block start it is ordinary paragraph text, which is exactly the
79
+ // asymmetry the old combined `(=+|-+)` pattern erased.
80
+ //
81
+ // ⚠ WHICH of the two roles wins is not decided here, and that is the shape of
82
+ // the whole rewrite. `classifyLines` decides it from block context, in one
83
+ // place: when a document-level paragraph is open the setext reading wins
84
+ // (CommonMark 4.3); otherwise this one does. Two predicates each answering that
85
+ // question from a line's own shape is what produced this module's defect class.
86
+ //
87
+ // An ATX heading's opening sequence is 1-6 hashes followed by a space or tab.
88
+ // `[ \t]` and NOT `\s`: CommonMark accepts only a space or tab there, while JS
89
+ // `\s` also matches U+00A0, form feed and vertical tab. A line sitting in that
90
+ // gap was neither a paragraph nor a heading, so a following `---` could not
91
+ // close the section and it ran on — silent pass (`p082-b3-g3`, `p082-b3-g4`).
92
+ // U+00A0 is the realistic input: it arrives by copy-paste from a web page or a
93
+ // word processor.
94
+ //
95
+ // ⚠ There is no literal U+00A0 anywhere in this file. The paragraph above used
96
+ // to demonstrate the defect with a real one, which is indistinguishable from a
97
+ // space in a diff and which an editor may silently normalize away.
98
+ // `test/upgrade-drift.mjs` builds these characters with `String.fromCharCode`
99
+ // for precisely that reason, and this file had been ignoring its own advice.
100
+ // `test/upgrade-drift.mjs` now asserts the absence rather than trusting this
101
+ // sentence.
102
+ //
103
+ // ⚠ There is exactly ONE ATX expression in this module now, and that is a
104
+ // deliberate outcome of the debt-12 rewrite. There used to be FOUR — this
105
+ // constant plus a full-line copy inside `headingAt`, `extractHeadings` and
106
+ // `extractH2Headings` — and three separate review rounds found them disagreeing
107
+ // with each other. A shared constant that only two of the four copies read is
108
+ // not a single source of truth.
109
+ const ATX_HEADING_RE = /^ {0,3}(#{1,6})([ \t].*)?$/;
110
+
111
+ // The `{ level, text }` of an ATX heading line, or null.
112
+ //
113
+ // ⚠ The closing sequence is stripped in a SECOND step rather than by an
114
+ // optional group in the same expression, and that is a correctness fix, not a
115
+ // style preference. The old one-expression form
116
+ // `(?:[ \t]+(.*?))?(?:[ \t]+#+)?[ \t]*$` gave `## #` the text `#`: the lazy
117
+ // capture backtracks into the closing sequence, so the group meant to absorb it
118
+ // never gets the chance. CommonMark says that heading's content is EMPTY. Two
119
+ // steps cannot express the ambiguity, so they simply get it right — while still
120
+ // keeping `## Git Policy#` as the text `Git Policy#`, because the closing
121
+ // sequence must be PRECEDED by whitespace (`p082-b3-g2` finding 2 paid for
122
+ // that, and `test/upgrade-drift.mjs` pins both).
123
+ //
124
+ // An EMPTY heading (`#`, `## `, `## #`) is a real heading and really does end a
125
+ // section, so it is reported as one. `headingResolves` must go on refusing to
126
+ // resolve a reference against it — `''` prefix-matches every reference, which
127
+ // silently satisfied every cross-ref in the guide (`p082-b3-g5` finding 1).
128
+ function parseAtxHeading(line) {
129
+ const m = line.match(ATX_HEADING_RE);
130
+ if (!m) return null;
131
+ const rest = (m[2] || '').replace(/[ \t]+#+[ \t]*$/, '');
132
+ // Through `stripSpaceTab`, not a second copy of the same expression: this used
133
+ // to spell the rule inline while the setext branch used `.trim()`, and the two
134
+ // heading kinds then disagreed about a trailing U+00A0.
135
+ return { level: m[1].length, text: stripSpaceTab(rest), setext: false };
136
+ }
137
+
138
+ function isThematicBreak(line) {
139
+ return /^ {0,3}(-[ \t]*){3,}$/.test(line)
140
+ || /^ {0,3}(\*[ \t]*){3,}$/.test(line)
141
+ || /^ {0,3}(_[ \t]*){3,}$/.test(line);
142
+ }
143
+
144
+ // A fenced-code marker: three or more backticks, or three or more tildes.
145
+ const FENCE_MARKER = '(```+|~~~+)';
146
+ const FENCE_OPEN_RE = new RegExp(`^ {0,3}${FENCE_MARKER}(.*)$`);
147
+ const FENCE_CLOSE_RE = new RegExp(`^ {0,3}${FENCE_MARKER}[ \\t]*$`);
148
+
149
+ // Blank out fenced code blocks (``` / ~~~) line-by-line so headings and § refs
150
+ // inside examples are never extracted. Line positions are preserved. CommonMark
151
+ // close rules (PROPOSAL-076 gates G4/G6): a block closes only on the same fence
152
+ // character repeated at least the opening length, indented at most three
153
+ // spaces, with nothing but whitespace after — so a three-backtick line inside a
154
+ // four-backtick example, or an info-string line like ```js inside an open
155
+ // fence, is content and does not end the block early.
156
+ //
157
+ // This is the module's ONLY line splitter; everything downstream works on its
158
+ // output, which is what makes `LINE_SPLIT_RE` a single point of control.
159
+ //
160
+ // ⚠ The fence marker is spelled ONCE, in `FENCE_MARKER`. The open and close
161
+ // tests are genuinely different rules — a close must have nothing after it,
162
+ // while an open takes an info string — but they must agree about what a fence
163
+ // LOOKS like, and two literals cannot be relied on to. Same reason as
164
+ // `ATX_HEADING_RE` and `BULLET_MARKER`.
165
+ function blankFencedBlocks(content) {
166
+ const lines = content.split(LINE_SPLIT_RE);
167
+ const out = [];
168
+ let fenceChar = null;
169
+ let fenceLen = 0;
170
+ for (const line of lines) {
171
+ if (fenceChar) {
172
+ out.push('');
173
+ const close = line.match(FENCE_CLOSE_RE);
174
+ if (close && close[1][0] === fenceChar && close[1].length >= fenceLen) {
175
+ fenceChar = null;
176
+ fenceLen = 0;
177
+ }
178
+ continue;
179
+ }
180
+ // CommonMark: at most three SPACES of indent (a tab is four columns, so a
181
+ // tab-indented fence is indented code, not a fence), and a BACKTICK fence's
182
+ // info string may not contain a backtick. Accepting either opened a
183
+ // pseudo-fence that blanked real headings, so a stale section ran on past
184
+ // its own end - silent pass (`p082-b3-g4` finding 2).
185
+ const open = line.match(FENCE_OPEN_RE);
186
+ if (open && open[1][0] === '`' && open[2].includes('`')) {
187
+ out.push(line);
188
+ continue;
189
+ }
190
+ if (open) {
191
+ fenceChar = open[1][0];
192
+ fenceLen = open[1].length;
193
+ out.push('');
194
+ continue;
195
+ }
196
+ out.push(line);
197
+ }
198
+ return out;
199
+ }
200
+
201
+ // All Markdown heading texts (any level), fence-aware.
202
+ //
203
+ // ⚠ These two used to carry their own full-line ATX expression — two of the four
204
+ // copies that disagreed with each other. They read the classification now, which
205
+ // means three behaviours arrived here for free rather than being argued about
206
+ // one at a time:
207
+ // - the 0-3 space indent, which was first left out here on the argument that
208
+ // packaged inputs are generated and never indented, and which measuring
209
+ // against the reference retired;
210
+ // - SETEXT headings, which neither of these ever saw. A `§ Heading` reference
211
+ // to one was reported dangling and a template section written that way
212
+ // counted as missing — both false positives, both invisible to the pins,
213
+ // because the pins only ever fed these functions ATX;
214
+ // - the empty-heading and closing-sequence rules, which now cannot drift from
215
+ // what the section walker uses, because there is nothing left to drift from.
216
+ function extractHeadings(content) {
217
+ const lines = blankFencedBlocks(content);
218
+ return classifyLines(lines).filter((c) => c.heading).map((c) => c.heading.text);
219
+ }
220
+
221
+ // H2 heading texts only, fence-aware — the section shape of a template.
222
+ // Level 2 covers both `## X` and a `---`-underlined setext heading.
223
+ function extractH2Headings(content) {
224
+ const lines = blankFencedBlocks(content);
225
+ return classifyLines(lines).filter((c) => c.heading && c.heading.level === 2).map((c) => c.heading.text);
226
+ }
227
+
228
+
229
+ // `<file>.md § Heading` references whose file basename is `targetBasename`.
230
+ // A soft-wrapped heading name is captured by joining the next line; a match
231
+ // anchored beyond the current line is left to that line's own iteration.
232
+ function extractSectionRefs(content, targetBasename) {
233
+ // ⚠ `visibleTextLines`, not `blankFencedBlocks`: a `<file>.md § Heading`
234
+ // written inside an HTML comment was reported as a dangling reference, and the
235
+ // captured heading text carried the trailing `-->` into the message.
236
+ const lines = visibleTextLines(content);
237
+ const refs = [];
238
+ lines.forEach((line, i) => {
239
+ const firstLen = line.replace(/\s+$/, '').length;
240
+ const joined = line.replace(/\s+$/, '') + ' ' + (lines[i + 1] || '').replace(/^\s+/, '');
241
+ for (const m of joined.matchAll(/`?([A-Za-z0-9._/-]+\.md)`?\s*§\s*"?([^."`)\n]+)/g)) {
242
+ if (m.index > firstLen) continue;
243
+ const base = m[1].split('/').pop();
244
+ if (base !== targetBasename) continue;
245
+ refs.push({ line: i + 1, headingText: m[2].trim() });
246
+ }
247
+ });
248
+ return refs;
249
+ }
250
+
251
+ // ⚠⚠ `.trim()` HERE IS DELIBERATE, AND IT IS NOT THE `.trim()` DEFECT CLASS.
252
+ // This was briefly "fixed" to `stripSpaceTab` on the reasoning that one rule
253
+ // should have one spelling, and that broke real files — the measurement is in
254
+ // the pins below `normalizeHeading` in `test/upgrade-drift.mjs`. Two different
255
+ // questions were being read as one:
256
+ //
257
+ // `stripSpaceTab` answers **what the heading text IS** — block structure,
258
+ // spec-following, deliberately strict, because a U+00A0 is not whitespace to
259
+ // CommonMark and a section boundary must not move because of one.
260
+ //
261
+ // `normalizeHeading` answers **whether two heading strings NAME the same
262
+ // section** — an identity comparison over text a human retyped or pasted. It
263
+ // is tolerant BY CONSTRUCTION and already does three things CommonMark would
264
+ // never do: it strips ``, `*` and `_`, and (through `headingKey`) it folds
265
+ // case. Tightening it to the block-structure rule aligned the wrong function.
266
+ //
267
+ // What tightening it actually did: `## Ceremony Scaling (Project Application)`
268
+ // with a copy-pasted trailing U+00A0 stopped matching the fingerprint, so doctor
269
+ // reported `missing` on a file that carries the rule — and `missingTemplateSections`
270
+ // reported a section absent that was present. Loud, not silent, but wrong, and
271
+ // `CONVENTIONS_FINGERPRINTS` already records this exact false `missing` happening
272
+ // once before, for case rather than for whitespace. Reproduces for U+00A0, form
273
+ // feed, vertical tab, U+2002 and U+FEFF, leading and trailing.
274
+ //
275
+ // ⚠ `extractSectionRefs` also calls `.trim()` on the *reference* side of the
276
+ // `headingResolves` comparison. That is the same rule on both sides, which is
277
+ // the property to keep: tightening one side only is what made `§ Workflow_`
278
+ // stop resolving.
279
+ function normalizeHeading(text) {
280
+ return text.replace(/[`*_]/g, '').trim();
281
+ }
282
+
283
+ // Tolerant heading resolution: a reference resolves when it prefix-matches a
284
+ // real heading in either direction (covers soft wraps like "§ Workflow" for
285
+ // "Workflow Transparency" and shorthand references).
286
+ // ⚠ `headingKey`, not `normalizeHeading` — the difference is case folding, and
287
+ // this consumer used to be the odd one out. `CONVENTIONS_FINGERPRINTS` states
288
+ // the rule ("a heading an adopter retyped in different case is the same
289
+ // section") and `conventionsSectionBodies` implemented it, while this function
290
+ // and `missingTemplateSections` did not: `## filling the templates` was found by
291
+ // one and reported missing by the other, from the same document. One normaliser
292
+ // with three consumers, two of which skipped half of it.
293
+ function headingResolves(referenceText, headings) {
294
+ const want = headingKey(referenceText);
295
+ if (!want) return true;
296
+ return headings.some((heading) => {
297
+ const have = headingKey(heading);
298
+ // An EMPTY heading resolves nothing. Making empty ATX headings real (so they
299
+ // close a section, which CommonMark says they do) put `''` into this list,
300
+ // and `want.startsWith('')` is true for every reference — so one stray `## `
301
+ // in a guide silently satisfied every `<file> § Heading` cross-reference and
302
+ // doctor stopped reporting dangling ones (`p082-b3-g5` finding 1). A fix in
303
+ // one consumer of a shared list has to be checked against the others.
304
+ if (!have) return false;
305
+ return have.startsWith(want) || want.startsWith(have);
306
+ });
307
+ }
308
+
309
+ // Which of the template's H2 sections are missing from a filled document — the
310
+ // "created from an older template shape" signal.
311
+ function missingTemplateSections(templateContent, documentContent) {
312
+ // ⚠ `headingKey`, not `normalizeHeading`: same case-folding rule as
313
+ // `headingResolves` above and `conventionsSectionBodies` below. This one is
314
+ // user-facing — a case difference here produced doctor's "looks like an older
315
+ // `_index.md` template shape" on a document that has the section.
316
+ //
317
+ // ⚠⚠ A COUNTED multiset, not a `Set`. With a `Set`, a template carrying two
318
+ // H2s that fold to the same key — `## Filling the Templates` and
319
+ // `## FILLING THE TEMPLATES` — was fully satisfied by ONE of them appearing in
320
+ // the document, so a genuinely missing section went unreported. Silent, and it
321
+ // arrived with the case folding: before folding, the two keys were distinct.
322
+ // The packaged `_index.md` has no colliding pair today, so this is latent —
323
+ // which is exactly why it needs to be code and not a comment.
324
+ const have = new Map();
325
+ for (const heading of extractH2Headings(documentContent)) {
326
+ const key = headingKey(heading);
327
+ have.set(key, (have.get(key) || 0) + 1);
328
+ }
329
+ return extractH2Headings(templateContent).filter((heading) => {
330
+ const key = headingKey(heading);
331
+ const left = have.get(key) || 0;
332
+ if (left === 0) return true;
333
+ have.set(key, left - 1);
334
+ return false;
335
+ });
336
+ }
337
+
338
+ // True when `content` matches `templateContent` up to placeholder substitution:
339
+ // every single-line `{...}` placeholder in the template may match any text.
340
+ // Distinguishes "pristine current starter" from "edited or older starter"
341
+ // without knowing the values init substituted.
342
+ function matchesTemplateWithPlaceholders(content, templateContent) {
343
+ // Split/join through the module's single line-ending definition rather than
344
+ // a second hand-rolled CRLF replacement: a lone CR left the template and the
345
+ // document normalized differently, so a pristine starter compared as edited.
346
+ const normalize = (s) => String(s).split(LINE_SPLIT_RE).join('\n').replace(/\s+$/, '');
347
+ const parts = normalize(templateContent).split(/\{[^}\n]*\}/);
348
+ const pattern = `^${parts.map(escapeRegExp).join('[\\s\\S]*?')}$`;
349
+ try {
350
+ return new RegExp(pattern).test(normalize(content));
351
+ } catch {
352
+ return false;
353
+ }
354
+ }
355
+
356
+ function escapeRegExp(s) {
357
+ return s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
358
+ }
359
+
360
+ // PROPOSAL-078 phase 1: the table-cell formatting convention (P-072) ships
361
+ // as an HTML comment at every template head, so docs seeded before 0.13 —
362
+ // or written from scratch — hold tables with no in-file reminder and tend
363
+ // to grow run-on cell walls. Detection is info-only; doctor never edits
364
+ // user-authored specs. The snippet is matched anywhere in the file (the
365
+ // _index.md template places it mid-file, after its usage comment), and a
366
+ // table counts only outside code fences. The delimiter-row regex accepts
367
+ // any row with an internal pipe (`|---|---|`, `---|---`) plus the
368
+ // single-column form with BOTH outer pipes (`|---|`); a bare `---` is a
369
+ // thematic break / frontmatter fence and never matches — the one shape
370
+ // deliberately missed is a single-column delimiter written without outer
371
+ // pipes, which is indistinguishable from a thematic break anyway (the
372
+ // check is a nudge, not a gate).
373
+ const SPEC_FORMATTING_CONVENTION_SNIPPET = 'Formatting convention: keep table cells concise';
374
+ const TABLE_DELIMITER_ROW_RE = /^ {0,3}(?:\|?[ \t]*:?-{2,}:?[ \t]*(?:\|[ \t]*:?-{2,}:?[ \t]*)+\|?|\|[ \t]*:?-{2,}:?[ \t]*\|)[ \t]*$/;
375
+
376
+ // ⚠ This asks the classification for a table rather than grepping for a
377
+ // delimiter row, so it now agrees with the section walker about what a table is.
378
+ // The two used to answer differently: a delimiter row with no header line above
379
+ // it counted as a table HERE while the walker treated it as a thematic break or
380
+ // prose. Neither shape occurs in a packaged template — `test/upgrade-drift.mjs`
381
+ // pins the packaged verdicts — but "two places in this file decide the same
382
+ // question with different code" is the defect class, and it does not stop being
383
+ // the defect class because the current inputs happen not to hit it.
384
+ function hasTableWithoutConventionComment(content) {
385
+ if (String(content).includes(SPEC_FORMATTING_CONVENTION_SNIPPET)) {
386
+ return false;
387
+ }
388
+ const lines = blankFencedBlocks(String(content));
389
+ return classifyLines(lines).some((c) => c.type === 'table');
390
+ }
391
+
392
+ // --- PROPOSAL-082 G5 / PROPOSAL-083 §4: `_conventions.md` staleness -----------
393
+ //
394
+ // `_conventions.md` is user-owned. `dflow init` writes it once and neither init
395
+ // nor `configure-agents` ever rewrites it, so a project keeps the shape it was
396
+ // created with forever. Doctor reporting is the ONLY way an adopter learns their
397
+ // copy predates a contract change — which is why this reports and never edits.
398
+ //
399
+ // Every fingerprint is SECTION-LOCAL. A whole-file grep would pass the moment
400
+ // the phrase appeared anywhere in the document, including inside a section the
401
+ // adopter pasted in from somewhere else; the claim being checked is "THIS
402
+ // section carries this rule", so that is what gets read.
403
+ //
404
+ // Three states, and the middle one is the whole point:
405
+ //
406
+ // missing — the section is not in the file at all. Reported, never silently
407
+ // passed: MAINTAINERS § Cross-Model Review (P083 batch 2, g16) —
408
+ // "a gate that silently accepts absence is worse than no gate: it
409
+ // reports success". The previous attempt at this check was pulled
410
+ // for exactly that, because against the then-broken projection it
411
+ // could only ever report success.
412
+ // stale — section present, rule absent. This is the actionable one: the
413
+ // project's own conventions state something the shipped flows no
414
+ // longer do.
415
+ // current — section present, rule present.
416
+ // Heading detection has to cover more than `^#{1,6} `, because every shape it
417
+ // misses becomes a section that never ends — and an over-running body reports
418
+ // the WRONG section as current, which is the silent-success direction:
419
+ // - ATX indented 1-3 spaces is still a heading (CommonMark).
420
+ // - A setext heading (text line + `===` / `---` underline) is a heading, and
421
+ // missing it let a body run on into every following section. `---` is only
422
+ // setext when it directly follows a non-blank, non-heading line; otherwise
423
+ // it is a thematic break, which `_conventions.md` uses.
424
+ // Comparison is case-insensitive: a heading an adopter retyped in different
425
+ // case is the same section, and treating it as absent produced a false
426
+ // `missing` on a file that had the rule.
427
+ // A setext underline may only close a PARAGRAPH (CommonMark 4.3). After a list
428
+ // item, a table row, a blockquote or another heading it is not a heading at all
429
+ // — `- foo` / `---` is a list plus a thematic break. Treating those as headings
430
+ // ended sections early and, worse, made the caller pop a real content line out
431
+ // of the body: a marker sitting on the last list item or table row vanished and
432
+ // the section reported `stale`. That is the same false-warn-on-an-ordinary-edit
433
+ // class the soft-wrap fix removed.
434
+ //
435
+ // ⚠ A `---` directly after a plain paragraph line IS a setext heading, and the
436
+ // section really does end there — that is CommonMark, not a quirk of this
437
+ // parser. An adopter who wants a horizontal rule needs a blank line before it.
438
+ // ---------------------------------------------------------------------------
439
+ // THE BLOCK CLASSIFICATION PASS — PROPOSAL-082/083 debt 12.
440
+ //
441
+ // WHAT THIS REPLACED, AND WHY IT IS NOT A FIFTH PATCH. Five predicates used to
442
+ // live here — `isTableLine`, `closesOwnBlock`, `blockStartIndex`,
443
+ // `isParagraphLine` and `headingAt` — and each re-derived Markdown block
444
+ // structure from one line's shape, using backward scans and mutual recursion in
445
+ // place of the state a block parser keeps. But what a line IS depends on which
446
+ // block it sits inside, which a per-line predicate cannot see, so the five
447
+ // disagreed with one another. Six consecutive review rounds each found one more
448
+ // place where they did, and THREE of those rounds found a disagreement that the
449
+ // previous round's fix had introduced. Every widening of the arbiter's shape set
450
+ // found more: 30 -> 51 -> 57 -> 73 -> 79 shapes, each stage having measured
451
+ // clean at the stage before.
452
+ //
453
+ // So: one forward pass labels every line exactly once, and the former predicates
454
+ // become lookups into that label. CommonMark block parsing IS a forward pass;
455
+ // the backward scans existed only to reconstruct state that had been thrown
456
+ // away. A consequence worth having — the old `isParagraphLine` /
457
+ // `closesOwnBlock` mutual recursion was explicitly NOT linear (its own comment
458
+ // said so, and warned callers off large documents). This is O(lines).
459
+ //
460
+ // ⚠ THE RULE THAT REPLACES "do not add a fifth predicate": exactly one function
461
+ // decides block structure and this is it. Callers read labels. If you find
462
+ // yourself writing a regex against a raw line ANYWHERE else in order to decide
463
+ // what kind of line it is, that is this defect class growing back, and
464
+ // `test/upgrade-drift.mjs`'s class guard is meant to fail on it.
465
+ //
466
+ // ⚠ WHAT THIS DELIBERATELY DOES NOT IMPLEMENT — stated so the gaps are known
467
+ // rather than discovered by the next round:
468
+ // - HTML block type 7 (a complete tag whose name is not in the type-6 list,
469
+ // alone on a line). It is the only type that cannot interrupt a paragraph
470
+ // and recognizing it needs a real tag parser.
471
+ // `commonmark-differential.mjs` carries a row for it, so the divergence is
472
+ // MEASURED rather than assumed harmless.
473
+ // - GFM's rule that a delimiter row must have the same cell count as its
474
+ // header row. `test/upgrade-drift.mjs` pins `| A |` + `|---|---|` as a
475
+ // table, which is the looser reading, and tightening it would also move the
476
+ // P-078 formatting-convention check.
477
+ // - Nested container CONTENT. A blockquote or list item's interior is not
478
+ // parsed as its own block sequence. Knowing the document-level block is not
479
+ // a paragraph is enough for every question this module asks, and
480
+ // `_conventions.md` is a small hand-written file.
481
+ // - ⚠⚠ INLINE HTML COMMENTS — the one gap here whose direction is SILENT PASS,
482
+ // so it is the one that matters. This pass is LINE-level: a line that STARTS
483
+ // with `<!--` opens an HTML block and its text is correctly treated as unseen
484
+ // by a reader. A comment that begins part-way through a line is not a block
485
+ // at all — it is an inline span — and every consumer therefore counts its
486
+ // contents as live document text. Measured, with controls, on 2026-08-05:
487
+ //
488
+ // `prose <!-- cascade result is a floor -->` -> doctor reports `current`
489
+ // `- item <!-- ... -->` / `> quote <!-- ... -->` -> same
490
+ // `<!-- a --> <!-- cascade result is a floor -->` -> same (only the FIRST
491
+ // span is masked; the tail is never re-scanned)
492
+ //
493
+ // So a rule a user commented out mid-line still reads as present, which is
494
+ // the exact question `doctor` exists to answer. PRE-EXISTING — it was true
495
+ // before the visibility rule was added and the rule did not touch it; what
496
+ // the rule closed was the line-level half.
497
+ //
498
+ // ⚠ NOT fixed here on purpose (maintainer decision, 2026-08-05). Hiding
499
+ // spans needs an inline layer in a module that is deliberately line-level,
500
+ // and the obvious version is WRONG in the content-loss direction: a comment
501
+ // inside a CODE SPAN — `` `<!-- example -->` `` — is rendered and a reader
502
+ // DOES see it, so "mask everything between the delimiters" deletes visible
503
+ // content. That is a span tokenizer, with generated cases, as its own piece
504
+ // of work — not a fourth patch on a boundary predicate, which is how this
505
+ // module earned every defect it has. `p082-b3-k4` recommended exactly that.
506
+ // PROPOSAL-084 carries it as a detectable shape in the meantime: "this file
507
+ // contains an inline HTML comment" needs none of the boundary logic that
508
+ // could be wrong, which is what makes it warnable.
509
+ // ---------------------------------------------------------------------------
510
+
511
+ // HTML blocks (CommonMark 4.6). Types 1-5 pair a start with an end condition and
512
+ // close ON the line matching it — that line belongs to the block. Type 6 closes
513
+ // at the next blank line, which does NOT belong to it.
514
+ //
515
+ // ⚠ Type 2 is the one that forced this rewrite. The old code opened an HTML
516
+ // block on any `<!--` line but only ever closed a SINGLE-LINE `<!-- ... -->`, so
517
+ // a comment spanning two lines never closed: the section walker stopped inside
518
+ // it, dropped the list item that followed, and `findConventionsDrift` reported
519
+ // `ceremony-escalate-only:stale` against a file that was current — the
520
+ // content-loss direction (`p082-b3-g5` finding 4). It could not be patched
521
+ // per-line, because an HTML block's extent is cross-line state. That is the
522
+ // whole argument for this pass in one example.
523
+ // ⚠ NAMED because PROPOSAL-084's inline-comment detector also needs to know what
524
+ // a comment-block OPENER looks like, and the first draft of it hand-wrote
525
+ // `/^ {0,3}<!--/` a second time — a fresh copy of a block rule living outside the
526
+ // one place allowed to spell it, which is this module's whole recurring defect.
527
+ // One constant, two readers.
528
+ const HTML_COMMENT_OPEN_RE = /^ {0,3}<!--/;
529
+
530
+ // ⚠ The type-1 tag names, spelled ONCE. `HTML_BLOCK_TYPES` composes them into the
531
+ // start condition, `htmlBlockStart` derives the matching end tag from that capture,
532
+ // and `htmlBlockType7Line` needs the same list
533
+ // because CommonMark §4.6 EXCLUDES these four names from type 7 — type 1 wins.
534
+ // Naming it out of `HTML_BLOCK_TYPES` is the move `HTML_COMMENT_OPEN_RE` already
535
+ // made, for the same reason: the alternative is a third copy of the list, and the
536
+ // list being written twice inside this very constant is what let the type-7
537
+ // detector fire on `<pre/>` for a round (`p084-xv8` finding 4).
538
+ const HTML_TYPE1_TAGS = 'script|pre|style|textarea';
539
+ const HTML_TYPE1_NAME_RE = new RegExp(`^(?:${HTML_TYPE1_TAGS})$`, 'i');
540
+
541
+ const HTML_BLOCK_TYPES = [
542
+ [new RegExp(`^ {0,3}<(${HTML_TYPE1_TAGS})([ \\t>]|$)`, 'i'), null],
543
+ [HTML_COMMENT_OPEN_RE, /-->/],
544
+ [/^ {0,3}<\?/, /\?>/],
545
+ [/^ {0,3}<![A-Za-z]/, />/],
546
+ [/^ {0,3}<!\[CDATA\[/, /\]\]>/]
547
+ ];
548
+
549
+ // Type 6's tag list, verbatim from the spec. Order inside the alternation does
550
+ // not matter because the tail anchors on a delimiter, so `p` cannot swallow the
551
+ // `param` case.
552
+ const HTML_BLOCK_TAGS = 'address|article|aside|base|basefont|blockquote|body|caption|center|col|colgroup|dd|details|dialog|dir|div|dl|dt|fieldset|figcaption|figure|footer|form|frame|frameset|h1|h2|h3|h4|h5|h6|head|header|hr|html|iframe|legend|li|link|main|menu|menuitem|nav|noframes|ol|optgroup|option|p|param|search|section|summary|table|tbody|td|tfoot|th|thead|title|tr|track|ul';
553
+ const HTML_BLOCK_TYPE6_RE = new RegExp(`^ {0,3}</?(?:${HTML_BLOCK_TAGS})(?:[ \\t]|/?>|$)`, 'i');
554
+
555
+ // Leaf-block openers. Spelled `[ \t]` throughout, never `\s`, for the reason
556
+ // given at `ATX_HEADING_RE`.
557
+ //
558
+ // ⚠ `\d{1,9}`: CommonMark caps an ordered marker at nine digits, so a ten-digit
559
+ // run is paragraph text and NOT a list (`p082-b3-g5` finding 3).
560
+ // ⚠ `([ \t]|$)` and not `[ \t]`: a marker with nothing after it is an EMPTY list
561
+ // item, not prose (`p082-b3-g1` finding 6).
562
+ // ⚠ The two marker shapes are spelled ONCE and the three predicates are built
563
+ // from them. The first draft of this block wrote `[-*+]` twice and
564
+ // `\d{1,9}[.)]` twice across four separate literals, which is the module's own
565
+ // defect class — change the digit cap in one and the "is this an empty item"
566
+ // test silently keeps the old one. Same reason `ATX_HEADING_RE` is a constant.
567
+ const BULLET_MARKER = '[-*+]';
568
+ // The delimiter set is its own fragment because THREE expressions need it and
569
+ // one of them used to hand-write it: `INTERRUPTING_ITEM_RE` spelled `1[.)]`
570
+ // while claiming to be derived, so widening `ORDERED_MARKER` would have left the
571
+ // paragraph-interrupt rule silently on the old set. Unbroken at the time, and
572
+ // load-bearing — which is the only reason it is worth fixing before it breaks.
573
+ const ORDERED_DELIM = '[.)]';
574
+ const ORDERED_MARKER = `\\d{1,9}${ORDERED_DELIM}`;
575
+ // Opens a list wherever it appears, including an EMPTY item.
576
+ const LIST_ITEM_RE = new RegExp(`^ {0,3}(?:${BULLET_MARKER}|${ORDERED_MARKER})([ \\t]|$)`);
577
+ // An item with nothing after the marker. CommonMark 5.2: a list may interrupt a
578
+ // paragraph only if its first line is non-blank, so this shape may not.
579
+ const EMPTY_LIST_ITEM_RE = new RegExp(`^ {0,3}(?:${BULLET_MARKER}|${ORDERED_MARKER})[ \\t]*$`);
580
+ // The markers that CAN interrupt a paragraph: any bullet, or an ordered marker
581
+ // starting at 1. Any other ordered marker is paragraph text.
582
+ const INTERRUPTING_ITEM_RE = new RegExp(`^ {0,3}(?:${BULLET_MARKER}|1${ORDERED_DELIM})([ \\t]|$)`);
583
+
584
+ // A BLANK LINE — spaces and tabs only (CommonMark 2.1).
585
+ //
586
+ // ⚠⚠ THIS EXISTS BECAUSE `.trim()` IS THE GUARDED DEFECT CLASS IN A DIFFERENT
587
+ // SPELLING, and the guard was blind to it. `\s` was swept out of every block
588
+ // predicate; `line.trim()` was not, and `.trim()` strips the SAME set — U+00A0,
589
+ // form feed, vertical tab, U+2000-200B, U+FEFF — while containing no `\s` for
590
+ // the guard to find. A line holding only a non-breaking space was therefore a
591
+ // blank line here and a paragraph to CommonMark, so `para` / NBSP / `---` made
592
+ // `---` a thematic break instead of a setext underline, the section ran on, and
593
+ // a stale `_conventions.md` reported `current`. SILENT PASS, the worst
594
+ // direction, measured. Four more shapes flipped the other way.
595
+ //
596
+ // The guard now bans `.trim(` in block-structure code for this reason. Strip
597
+ // with `stripSpaceTab` where a value needs trimming.
598
+ const BLANK_LINE_RE = /^[ \t]*$/;
599
+
600
+ // Strip leading and trailing SPACES AND TABS only — the CommonMark rule for
601
+ // heading content, and the same rule `parseAtxHeading` applies. The setext
602
+ // branch used to call `.trim()` here, so the two heading kinds disagreed about
603
+ // a trailing U+00A0: `## Title<NBSP>` kept it and `Title<NBSP>` / `---` did not,
604
+ // which made `isRecognizableDflowGuide` reject one form and accept the other.
605
+ //
606
+ // ⚠ This DIVERGES from commonmark.js, which calls JS `.trim()` in its inline
607
+ // parser and so strips U+00A0 too. The spec says spaces and tabs; the reference
608
+ // implementation is broader. We follow the spec.
609
+ //
610
+ // ⚠⚠ BE PRECISE ABOUT WHICH HALF IS IN THIS REPO. An earlier version of this
611
+ // comment said the divergence "is carried as a named row in
612
+ // `atx-differential.mjs`" — a file that **does not exist anywhere under this
613
+ // repo**. It lives in the maintainer's repo-external review folder, so a reader
614
+ // following that sentence finds nothing, and nothing in the tree re-derives the
615
+ // claim. The divergence is pinned HERE instead, in `test/upgrade-drift.mjs`
616
+ // pin (k).
617
+ //
618
+ // ⚠⚠ AND BE PRECISE ABOUT WHAT (k) COVERS, because the sentence that replaced
619
+ // the bad one was itself an over-claim: it said the pins ran "against every
620
+ // character the class covers" while (k) pinned **U+00A0 trailing, alone**. Form
621
+ // feed and vertical tab were pinned at the ATX *opener*, not at heading-text
622
+ // trimming, and no consumer consequence was pinned at all. Two corrections in a
623
+ // row, both widening a claim past what the tree checks — so what (k) covers
624
+ // today is stated exactly: five characters (U+00A0, form feed, vertical tab,
625
+ // U+2002, U+FEFF), both leading and trailing, on both heading kinds. Widen the
626
+ // claim only by widening the loop.
627
+ function stripSpaceTab(text) {
628
+ return text.replace(/^[ \t]+|[ \t]+$/g, '');
629
+ }
630
+ // Split for measuring a list item's CONTENT COLUMN: indent, marker, gap.
631
+ const LIST_ITEM_PREFIX_RE = new RegExp(`^( {0,3})(${BULLET_MARKER}|${ORDERED_MARKER})([ \\t]*)`);
632
+ const BLOCKQUOTE_RE = /^ {0,3}>/;
633
+ // (`INDENTED_CODE_RE` was deleted rather than fixed. It was the only raw-line
634
+ // spelling of a question `indentWidth` already answers correctly, and keeping a
635
+ // corrected copy would have left two expressions for one rule — the shape this
636
+ // module keeps paying for.)
637
+ const SETEXT_UNDERLINE_RE = /^ {0,3}(=+|-+)[ \t]*$/;
638
+
639
+ // Columns of leading whitespace, a tab advancing to the next multiple of four
640
+ // (CommonMark 2.2). Spaces and tabs only — anything else ends the indent, which
641
+ // is the `[ \t]`-not-`\s` rule in yet another place.
642
+ function indentWidth(line) {
643
+ let col = 0;
644
+ for (const ch of line) {
645
+ if (ch === ' ') col += 1;
646
+ else if (ch === '\t') col += 4 - (col % 4);
647
+ else break;
648
+ }
649
+ return col;
650
+ }
651
+
652
+ // The column a list item's CONTENT starts at — the indent, the marker, and the
653
+ // gap after it. CommonMark 5.2: a gap of 1-4 spaces sets the content column
654
+ // directly; an empty item, or a gap of five or more (which would start indented
655
+ // code), puts the content one column after the marker.
656
+ //
657
+ // ⚠ This exists because a blank line does NOT end a list. A line indented to at
658
+ // least this column after a blank is still the item's content, and reading it as
659
+ // a document-level paragraph made `- a` / blank / ` b` / `---` end the section
660
+ // where CommonMark reads a thematic break — content loss, the pop deleting the
661
+ // line that carries the rule. Measured against the reference; the first version
662
+ // of this pass closed every block on a blank line and got it wrong.
663
+ function listContentColumn(line) {
664
+ const m = line.match(LIST_ITEM_PREFIX_RE);
665
+ if (!m) return null;
666
+ const markerEnd = m[1].length + m[2].length;
667
+ let gap = 0;
668
+ for (const ch of m[3]) gap += ch === '\t' ? 4 - ((markerEnd + gap) % 4) : 1;
669
+ return gap === 0 || gap > 4 ? markerEnd + 1 : markerEnd + gap;
670
+ }
671
+
672
+ // Convert the classifier's visual column into the UTF-16 offset String#slice
673
+ // expects. A tab advances to a four-column stop but occupies one code unit; using
674
+ // the visual column as an offset cuts into nested content after `-\t` / `- \t`.
675
+ function stringIndexAtVisualColumn(line, targetColumn) {
676
+ let column = 0;
677
+ let index = 0;
678
+ while (index < line.length && column < targetColumn) {
679
+ column += line[index] === '\t' ? 4 - (column % 4) : 1;
680
+ index += 1;
681
+ }
682
+ return index;
683
+ }
684
+
685
+ // `{ endRe }` when the line opens an HTML block, else null. `endRe: null` means
686
+ // type 6 — closed by a blank line rather than by a pattern.
687
+ // ⚠ `invisible` answers a DIFFERENT question from the rest of this function, and
688
+ // keeping them apart matters: `endRe` is about where the block ENDS (block
689
+ // structure), `invisible` is about whether a reader ever SEES the text inside it
690
+ // (content). The module got the first right and never asked the second, so a
691
+ // rule commented out with `<!-- … -->` counted as present and a stale file
692
+ // reported `current` — silent pass, reachable by one edit a person actually
693
+ // makes (`p082-b3-k1`).
694
+ //
695
+ // Which types a reader cannot see, and why each:
696
+ // type 2 `<!-- -->` a comment. Never rendered.
697
+ // type 3 `<? ?>` a processing instruction. Never rendered.
698
+ // type 4 `<!DOCTYPE>` a declaration. Never rendered.
699
+ // type 5 `<![CDATA[ ]]>` ignored by HTML parsers.
700
+ // type 1 ONLY `script` and `style`. ⚠ `pre` and `textarea`
701
+ // share type 1's start condition but their text IS
702
+ // shown — treating all four alike would delete real
703
+ // convention prose out of a `<pre>` block, which is
704
+ // the content-loss direction.
705
+ // type 6 / type 7 ordinary tags; the text between them renders.
706
+ // The list is spelled here and nowhere else, for the same reason `ORDERED_DELIM`
707
+ // is: two copies of a rule is this module's defect class.
708
+ const HTML_TYPE1_INVISIBLE_TAGS = /^(script|style)$/i;
709
+
710
+ // ⚠⚠ A THIRD STATE, and missing it shipped a false positive AND a repair that
711
+ // made things worse (`p084-xv4` finding 1). `invisible` splits type 1 into "the
712
+ // reader sees nothing" (`script`, `style`) and "the reader sees the text"
713
+ // (`pre`, `textarea`) — but for an HTML COMMENT those two are not the same
714
+ // question, because it asks whether the reader sees the COMMENT, not the text.
715
+ // `<pre>` interior is parsed as HTML, so a comment in it is a comment and
716
+ // is NOT displayed → hiding really happens → worth reporting.
717
+ // `<textarea>` interior is RAW TEXT (RCDATA), so `<!-- x -->` is DISPLAYED
718
+ // verbatim, brackets and all → nothing is hidden from anyone →
719
+ // reporting it is noise, and the repair for it is actively
720
+ // harmful: moving the line below `</textarea>` turns a comment
721
+ // the reader could see into one they cannot.
722
+ // Matched against the shared type-1 opener rather than a fresh regex, for the
723
+ // same reason `HTML_COMMENT_OPEN_RE` exists — two copies of a rule is this
724
+ // module's defect class.
725
+ // ⚠⚠⚠ THERE IS NO TEXTAREA EXEMPTION HERE, AND THAT IS THE DESIGN DECISION —
726
+ // not an oversight, and not something to helpfully add back.
727
+ //
728
+ // `<textarea>` holds raw text, so a comment inside one IS displayed to the reader
729
+ // and reporting it is a false positive. Three rounds were spent trying to suppress
730
+ // exactly that, and **all three attempts produced a silent pass** — the failure this
731
+ // entire proposal exists to remove:
732
+ // `p084-xv4` → exemption added, keyed on the enclosing block.
733
+ // `p084-xv5` → it could not see a nested `<textarea>`; kept, advice neutralised.
734
+ // `p084-xv9` → it skipped the WHOLE LINE, so a comment after a same-line
735
+ // `</textarea>` was exempted although it really is hidden. Doctor
736
+ // read a policy out of it under `All checks passed`.
737
+ // `p084-xv10` → the column version accepted `</textarea >` while the classifier
738
+ // closes only on `</textarea>`. **Two consumers, one rule, no shared
739
+ // spelling** — this module's own oldest defect, reintroduced by the
740
+ // repair. A later exact close then suppressed the unclosed fallback,
741
+ // so nothing fired at all.
742
+ //
743
+ // This repo's rule is that a check rewritten three times is a design question, not
744
+ // a fourth patch. The design answer: **the exemption is not worth having.** It
745
+ // suppresses a rare false positive at the cost of a defect class that has landed
746
+ // three times running, in the one direction this module must never fail in. Without
747
+ // it a `<textarea>` comment is merely REPORTED — the noisy direction, which the
748
+ // module's own direction rule allows and which the maintainer has twice chosen
749
+ // (2026-08-09, for nested `<textarea>` and for container-nested fences).
750
+ //
751
+ // What replaces it is a sentence, not code: `comment-inside-container`'s action and
752
+ // both explainer pages open by telling the reader that a comment inside a
753
+ // `<textarea>` is already visible and should be left alone. That is pinned.
754
+ // ⚠ It also removes an inconsistency nobody could have explained: the document-level
755
+ // `<textarea>` was exempt while a nested one was not.
756
+ function htmlBlockStart(line) {
757
+ for (let t = 0; t < HTML_BLOCK_TYPES.length; t += 1) {
758
+ const [startRe, endRe] = HTML_BLOCK_TYPES[t];
759
+ const m = startRe.exec(line);
760
+ if (m) {
761
+ const invisible = t === 0 ? HTML_TYPE1_INVISIBLE_TAGS.test(m[1]) : true;
762
+ // Marked, which powers `dflow render`, keeps a type-1 raw block open until
763
+ // the tag that opened it closes. A shared `script|pre|style|textarea` end
764
+ // regex lets `</script>` close `<pre>` and makes doctor read hidden content.
765
+ const matchingEnd = t === 0 ? new RegExp(`<\\/${m[1]}>`, 'i') : endRe;
766
+ return { endRe: matchingEnd, invisible };
767
+ }
768
+ }
769
+ if (HTML_BLOCK_TYPE6_RE.test(line)) return { endRe: null, invisible: false };
770
+ return null;
771
+ }
772
+
773
+ // Remove explicit container openers until the line reaches the content that
774
+ // CommonMark parses inside them. Continuation indentation for an open outer list
775
+ // is removed by the caller first; this loop handles nested `> - ...` openers.
776
+ function stripNestedContainerOpeners(line, paragraphOpen = false, priorQuoteDepth = 0) {
777
+ let rest = line;
778
+ let quoteDepth = 0;
779
+ let listOpened = false;
780
+ while (true) {
781
+ const quote = /^ {0,3}>[ \t]?/.exec(rest);
782
+ if (quote) {
783
+ quoteDepth += 1;
784
+ rest = rest.slice(quote[0].length);
785
+ continue;
786
+ }
787
+ // Repeating the same explicit quote chain continues its inner block. Moving
788
+ // into or out of a quote chain starts a different block sequence instead.
789
+ if (quoteDepth !== priorQuoteDepth) paragraphOpen = false;
790
+ if (LIST_ITEM_RE.test(rest)) {
791
+ const canInterrupt = INTERRUPTING_ITEM_RE.test(rest) && !EMPTY_LIST_ITEM_RE.test(rest);
792
+ // An ordered marker beginning above 1 cannot interrupt an open paragraph.
793
+ // Leaving it in `rest` is load-bearing: `<details>` later on that same
794
+ // visible paragraph is not a raw-HTML opener.
795
+ if (paragraphOpen && !canInterrupt) return { content: rest, quoteDepth, listOpened };
796
+ const contentCol = listContentColumn(rest);
797
+ if (contentCol !== null) {
798
+ listOpened = true;
799
+ rest = rest.slice(stringIndexAtVisualColumn(rest, contentCol));
800
+ paragraphOpen = false;
801
+ continue;
802
+ }
803
+ }
804
+ return { content: rest, quoteDepth, listOpened };
805
+ }
806
+ }
807
+
808
+ function nestedParagraphOpens(line, rawHtml) {
809
+ if (rawHtml || BLANK_LINE_RE.test(line)) return false;
810
+ if (parseAtxHeading(line) || SETEXT_UNDERLINE_RE.test(line) || isThematicBreak(line)) return false;
811
+ if (FENCE_OPEN_RE.test(line) || indentWidth(line) >= 4) return false;
812
+ return true;
813
+ }
814
+
815
+ function nestedContainerContext(
816
+ line,
817
+ htmlState,
818
+ paragraphOpen = false,
819
+ quoteDepth = 0,
820
+ htmlInList = false,
821
+ outerList = false
822
+ ) {
823
+ const stripped = stripNestedContainerOpeners(line, paragraphOpen, quoteDepth);
824
+ // A deeper quote remains inside the raw block opened by its outer quote, but
825
+ // moving to a shallower still-quoted line exits the inner block sequence.
826
+ // Carrying that state downward made visible escaped comments look hidden; the
827
+ // suggested column-0 repair would then hide text the reader could already see.
828
+ const activeHtml = stripped.quoteDepth < quoteDepth ? null : htmlState;
829
+ const html = nestedRawHtmlContext(stripped.content, activeHtml);
830
+ // Track what owns THIS raw-HTML state, not merely whether some earlier line
831
+ // contained a list marker. A markerless continuation may omit both the list
832
+ // indent and an outer quote marker; Marked still keeps raw HTML open only when
833
+ // the state belongs to a list item. A bare blockquote's state must instead be
834
+ // dropped at that boundary. When an existing state survives, its owner does;
835
+ // a newly opened state gets its owner from the current container chain.
836
+ const nextHtmlInList = html.next
837
+ ? (activeHtml ? htmlInList : outerList || stripped.listOpened)
838
+ : false;
839
+ return {
840
+ rawHtml: html.rawHtml,
841
+ nextHtml: html.next,
842
+ nextParagraph: nestedParagraphOpens(stripped.content, html.rawHtml),
843
+ nextQuoteDepth: stripped.quoteDepth,
844
+ nextHtmlInList
845
+ };
846
+ }
847
+
848
+ // Track raw HTML after a list / blockquote prefix has been removed. The main
849
+ // classifier intentionally labels the OUTERMOST block; comment opener selection
850
+ // also needs to know whether inline Markdown syntax applies inside that block.
851
+ // Reuse htmlBlockStart and its end conditions rather than spelling HTML types a
852
+ // second time. `rawHtml` includes the opening/closing line; `next` is the state
853
+ // carried to the following line.
854
+ function nestedRawHtmlContext(line, state) {
855
+ if (state) {
856
+ if (!state.endRe) {
857
+ if (BLANK_LINE_RE.test(line)) return { rawHtml: false, next: null };
858
+ return { rawHtml: true, next: state };
859
+ }
860
+ return { rawHtml: true, next: state.endRe.test(line) ? null : state };
861
+ }
862
+ const html = htmlBlockStart(line);
863
+ if (!html) return { rawHtml: false, next: null };
864
+ const selfEnds = html.endRe && html.endRe.test(line);
865
+ return {
866
+ rawHtml: true,
867
+ next: selfEnds ? null : { endRe: html.endRe }
868
+ };
869
+ }
870
+
871
+ // One label per line, in one pass.
872
+ //
873
+ // Each entry is `{ type, heading }`. `type` is one of `blank`, `heading`,
874
+ // `thematic-break`, `paragraph`, `list`, `blockquote`, `code`, `html`, `table`.
875
+ // `heading` is null except on a heading line, where it is
876
+ // `{ level, text, setext, start }` — `start` being the first line of the
877
+ // underlined paragraph for a setext heading, and the heading's own index for an
878
+ // ATX one. Callers that pop a setext heading's text out of a body need `start`,
879
+ // and getting it from the classification is why a soft-wrapped setext heading
880
+ // now works: the old code assumed the text was exactly one line.
881
+ //
882
+ // ⚠ Fenced code is already gone by the time this runs — `blankFencedBlocks` has
883
+ // replaced those lines with empty ones, preserving positions. That is why there
884
+ // is no fence state here. Keep calling this on that function's output; handing
885
+ // it raw content would read headings out of fenced examples, which is the
886
+ // PROPOSAL-076 G1 defect.
887
+ function classifyLines(lines) {
888
+ const out = new Array(lines.length);
889
+ let open = null; // { kind, start, endRe } — the open DOCUMENT-LEVEL block
890
+
891
+ for (let i = 0; i < lines.length; i += 1) {
892
+ const line = lines[i];
893
+
894
+ // (1) Inside an HTML block. Its extent is exactly the cross-line state the
895
+ // old per-line predicates could not hold.
896
+ //
897
+ // ⚠⚠ `blockStart` — WHICH LINE OPENED THIS BLOCK. It is carried here because
898
+ // reconstructing it from the outside is not reliably possible, and that was
899
+ // established the expensive way: `checkGuideProjectContextFormat` needs it to
900
+ // tell an adopter which `<!--` or `<details>` is hiding their
901
+ // `## Project Context`, and THREE consecutive review rounds each defeated a
902
+ // reverse-scan approximation of it (`debt212223-xv1` F2, `xv2` F1, `xv3` F1).
903
+ // Walking back over contiguous `html` lines crosses into the block before it;
904
+ // and tag- or comment-shaped lines INSIDE an open block (`<p>x</p>`,
905
+ // `prose <!-- y`) are indistinguishable from openers when you only look at the
906
+ // line, because whether a line can open a block at all depends on cross-line
907
+ // state that only this loop holds. Three rounds on one predicate is this repo's
908
+ // own signal to stop patching and change the shape. The field is additive, like
909
+ // `unclosed` and `invisible` before it — every existing consumer reads `.type`
910
+ // and `.heading`.
911
+ if (open && open.kind === 'html') {
912
+ if (open.endRe) {
913
+ // ⚠ `visibleFrom`: an end condition can leave text AFTER it on the same
914
+ // line, and that text is rendered. `<!-- note --> rest of the line` puts
915
+ // `rest of the line` in the output, so blanking the whole line loses
916
+ // real content — measured as a SILENT PASS one commit after the
917
+ // visibility rule landed (`p082-b3-k3` finding 2): a retired string
918
+ // sitting after a `-->` stopped being reported.
919
+ const end = open.endRe.exec(line);
920
+ out[i] = { type: 'html', heading: null, invisible: open.invisible, blockStart: open.start, blockEndsOnBlank: false };
921
+ if (end) {
922
+ if (open.invisible) out[i].visibleFrom = end.index + end[0].length;
923
+ open = null;
924
+ }
925
+ continue;
926
+ }
927
+ if (BLANK_LINE_RE.test(line)) { out[i] = { type: 'blank', heading: null }; open = null; continue; }
928
+ // ⚠ THE SECOND OF THE TWO in-block assignments, and it is the one a
929
+ // whitespace-differing `replace_all` missed when `blockStart` was added: this
930
+ // branch serves blocks with NO end condition, which in this parser means
931
+ // **type 6 only** (`<details>`, `<div>`) — type 7 is deliberately not
932
+ // implemented, see the § above, and `<span>` classifies as a paragraph
933
+ // (`debt212223-xv5`, which probed it). Those are exactly the blocks the guide
934
+ // finding had to be able to name. Adding
935
+ // the field to the `endRe` branch alone left `<details>` reporting no opener
936
+ // at all, and the caller's fallback then pointed at the heading itself.
937
+ // ⚠⚠ `blockEndsOnBlank` rides along for the same reason `blockStart` does
938
+ // (`debt212223-y4` finding 1): **how you close a block is not visible from
939
+ // its opening line**. A caller that saw `<details>` and advised "close it"
940
+ // was wrong — `</details>` does not end a type-6 block, only a blank line
941
+ // does, measured — while `</script>` / `]]>` / `?>` really do end types 1/3/5.
942
+ // The distinction is `endRe`, which lives here and nowhere else, so it is
943
+ // carried out rather than re-derived at the caller. Re-deriving block
944
+ // structure outside this function is what cost three rounds already.
945
+ out[i] = { type: 'html', heading: null, invisible: open.invisible, blockStart: open.start, blockEndsOnBlank: true };
946
+ continue;
947
+ }
948
+
949
+ // (2) A blank line closes every document-level block — EXCEPT a list, whose
950
+ // item can continue after it (a "loose" list). Measured: closing the
951
+ // list here made `- a` / blank / ` b` / `---` end the section, while
952
+ // CommonMark keeps the item open and reads `---` as a thematic break.
953
+ // Content-loss direction — the walker pops the line carrying the rule.
954
+ // A BLOCKQUOTE really does close, which is not symmetry but measurement:
955
+ // `> q` / blank / ` more` ends the section in the reference too.
956
+ if (BLANK_LINE_RE.test(line)) {
957
+ out[i] = { type: 'blank', heading: null };
958
+ if (open && open.kind === 'list') {
959
+ open.blankSeen = true;
960
+ // A blank ends type-6 raw HTML but NOT type 1-5. Let the same transition
961
+ // function that owns those end conditions decide instead of clearing
962
+ // every nested block at the loose-list boundary.
963
+ open.nestedHtml = nestedRawHtmlContext('', open.nestedHtml).next;
964
+ if (!open.nestedHtml) open.nestedHtmlInList = false;
965
+ open.nestedParagraph = false;
966
+ open.nestedQuoteDepth = 0;
967
+ }
968
+ else open = null;
969
+ continue;
970
+ }
971
+
972
+ // (2.5) A line indented to the open list item's content column belongs to
973
+ // that item — including a heading, which is then a heading INSIDE the
974
+ // list and not a document-level one. This is checked before every
975
+ // block-start test below, because otherwise ` ## Heading` inside an
976
+ // item would end the section.
977
+ if (open && open.kind === 'list') {
978
+ if (indentWidth(line) >= open.contentCol) {
979
+ const nested = nestedContainerContext(
980
+ line.slice(stringIndexAtVisualColumn(line, open.contentCol)),
981
+ open.nestedHtml,
982
+ open.nestedParagraph,
983
+ open.nestedQuoteDepth,
984
+ open.nestedHtmlInList,
985
+ true
986
+ );
987
+ out[i] = { type: 'list', heading: null };
988
+ if (nested.rawHtml) out[i].rawHtml = true;
989
+ open.nestedHtml = nested.nextHtml;
990
+ open.nestedParagraph = nested.nextParagraph;
991
+ open.nestedQuoteDepth = nested.nextQuoteDepth;
992
+ open.nestedHtmlInList = nested.nextHtmlInList;
993
+ open.blankSeen = false;
994
+ continue;
995
+ }
996
+ // Not indented into the item, and a blank line intervened: the list ended
997
+ // at that blank and this line starts something new.
998
+ if (open.blankSeen) open = null;
999
+ }
1000
+
1001
+ const paraOpen = open !== null && open.kind === 'paragraph';
1002
+
1003
+ // (3) A setext underline — but ONLY when there is a paragraph to underline.
1004
+ // This single test is the arbitration the old code spread across
1005
+ // `closesOwnBlock`, `blockStartIndex` and `isParagraphLine`, which is
1006
+ // why they could contradict each other about the same line.
1007
+ // ⚠ It is tested BEFORE the thematic break on purpose: CommonMark 4.3
1008
+ // gives the setext heading precedence where both readings are possible,
1009
+ // so `para` + `---` is a heading while a bare `---` is a break.
1010
+ if (paraOpen && SETEXT_UNDERLINE_RE.test(line)) {
1011
+ const marker = line.match(SETEXT_UNDERLINE_RE)[1];
1012
+ const text = stripSpaceTab(lines.slice(open.start, i).map(stripSpaceTab).join(' '));
1013
+ out[i] = {
1014
+ type: 'heading',
1015
+ heading: { level: marker[0] === '=' ? 1 : 2, text, setext: true, start: open.start }
1016
+ };
1017
+ open = null;
1018
+ continue;
1019
+ }
1020
+
1021
+ // (4) Shapes that open their own block wherever they appear.
1022
+ const atx = parseAtxHeading(line);
1023
+ if (atx) {
1024
+ out[i] = { type: 'heading', heading: { ...atx, start: i } };
1025
+ open = null;
1026
+ continue;
1027
+ }
1028
+
1029
+ if (isThematicBreak(line)) {
1030
+ out[i] = { type: 'thematic-break', heading: null };
1031
+ open = null;
1032
+ continue;
1033
+ }
1034
+
1035
+ // HTML blocks of types 1-6 can all interrupt a paragraph, so this sits
1036
+ // above the continuation logic rather than inside the block-start branch.
1037
+ const html = htmlBlockStart(line);
1038
+ if (html) {
1039
+ out[i] = { type: 'html', heading: null, invisible: html.invisible, blockStart: i, blockEndsOnBlank: !html.endRe };
1040
+ // ⚠ The end condition is checked on the OPENING line too (CommonMark 4.6:
1041
+ // "the block ends at the first line matching the end condition", which
1042
+ // includes the line that started it). Leaving it out meant a single-line
1043
+ // `<!-- Seeded by Dflow. -->` — the comment at the head of every packaged
1044
+ // template — opened a block that never closed, and every heading in the
1045
+ // file after it disappeared. The existing pins caught it immediately,
1046
+ // which is the argument for having run them before widening anything.
1047
+ // Same rule on the opening line, which may also be the closing one:
1048
+ // `<!-- a --> b` renders ` b`.
1049
+ const selfEnd = html.endRe && html.endRe.exec(line);
1050
+ if (selfEnd) {
1051
+ if (html.invisible) out[i].visibleFrom = selfEnd.index + selfEnd[0].length;
1052
+ open = null;
1053
+ } else {
1054
+ open = { kind: 'html', start: i, endRe: html.endRe, invisible: html.invisible };
1055
+ }
1056
+ continue;
1057
+ }
1058
+
1059
+ if (BLOCKQUOTE_RE.test(line)) {
1060
+ const sameQuote = open && open.kind === 'blockquote';
1061
+ const nested = nestedContainerContext(
1062
+ line,
1063
+ sameQuote ? open.nestedHtml : null,
1064
+ sameQuote ? open.nestedParagraph : false,
1065
+ sameQuote ? open.nestedQuoteDepth : 0,
1066
+ sameQuote ? open.nestedHtmlInList : false
1067
+ );
1068
+ out[i] = { type: 'blockquote', heading: null };
1069
+ if (nested.rawHtml) out[i].rawHtml = true;
1070
+ open = {
1071
+ kind: 'blockquote',
1072
+ start: open && open.kind === 'blockquote' ? open.start : i,
1073
+ nestedHtml: nested.nextHtml,
1074
+ nestedParagraph: nested.nextParagraph,
1075
+ nestedQuoteDepth: nested.nextQuoteDepth,
1076
+ nestedHtmlInList: nested.nextHtmlInList
1077
+ };
1078
+ continue;
1079
+ }
1080
+
1081
+ // A list interrupts an open paragraph only when it is a bullet, or an
1082
+ // ordered marker starting at 1, AND its first line is non-blank
1083
+ // (CommonMark 5.2). Any other ordered marker — and an empty item — is
1084
+ // paragraph text. Reading those as block starts is the divergence `g1` found
1085
+ // by widening the arbiter, in the silent-pass direction.
1086
+ if (LIST_ITEM_RE.test(line)) {
1087
+ const canInterrupt = INTERRUPTING_ITEM_RE.test(line) && !EMPTY_LIST_ITEM_RE.test(line);
1088
+ if (!paraOpen || canInterrupt) {
1089
+ const contentCol = listContentColumn(line);
1090
+ const nested = nestedContainerContext(
1091
+ line.slice(stringIndexAtVisualColumn(line, contentCol)),
1092
+ null,
1093
+ false,
1094
+ 0,
1095
+ false,
1096
+ true
1097
+ );
1098
+ out[i] = { type: 'list', heading: null };
1099
+ if (nested.rawHtml) out[i].rawHtml = true;
1100
+ // ⚠ `listContentColumn` cannot return null here, and the reason is worth
1101
+ // stating because the consequence if it ever did is severe:
1102
+ // `indentWidth(line) >= null` is `>= 0`, true for EVERY line, so the
1103
+ // whole rest of the document would be swallowed as list content —
1104
+ // silent pass. It cannot happen because `LIST_ITEM_PREFIX_RE` is built
1105
+ // from the same two marker fragments as `LIST_ITEM_RE` and is strictly
1106
+ // more permissive (its trailing `[ \t]*` matches empty where
1107
+ // `LIST_ITEM_RE` requires `[ \t]` or end of line), so anything this
1108
+ // branch accepts, it matches. Sharing the fragments is what makes that
1109
+ // an invariant rather than a hope — this module has broken exactly this
1110
+ // kind of unwritten relationship between two expressions five times.
1111
+ open = {
1112
+ kind: 'list',
1113
+ start: open && open.kind === 'list' ? open.start : i,
1114
+ // Each item sets the column its own content continues at.
1115
+ contentCol,
1116
+ blankSeen: false,
1117
+ nestedHtml: nested.nextHtml,
1118
+ nestedParagraph: nested.nextParagraph,
1119
+ nestedQuoteDepth: nested.nextQuoteDepth,
1120
+ nestedHtmlInList: nested.nextHtmlInList
1121
+ };
1122
+ continue;
1123
+ }
1124
+ // else: falls through to the paragraph continuation below.
1125
+ }
1126
+
1127
+ // (5) Continuation of an open container. Inside a list item or a blockquote
1128
+ // the document-level block is the container, not a paragraph, so no
1129
+ // setext underline can close anything here — which is precisely what
1130
+ // `blockStartIndex` used to walk backwards to work out.
1131
+ if (open && (open.kind === 'list' || open.kind === 'blockquote')) {
1132
+ out[i] = { type: open.kind, heading: null };
1133
+ // Raw HTML owned by a list item CAN cross a lazy continuation even when
1134
+ // its list indent — and an outer quote marker — is omitted. Marked keeps
1135
+ // the block open, so comment syntax on this line is still raw HTML. A bare
1136
+ // blockquote cannot cross that boundary; clearing it is the quote-exit
1137
+ // control that prevents visible escaped examples from being rejected.
1138
+ if (open.nestedHtml && open.nestedHtmlInList) {
1139
+ const lazy = nestedRawHtmlContext(line, open.nestedHtml);
1140
+ if (lazy.rawHtml) out[i].rawHtml = true;
1141
+ open.nestedHtml = lazy.next;
1142
+ if (!lazy.next) open.nestedHtmlInList = false;
1143
+ } else if (open.kind === 'blockquote') {
1144
+ open.nestedHtml = null;
1145
+ open.nestedHtmlInList = false;
1146
+ open.nestedParagraph = false;
1147
+ open.nestedQuoteDepth = 0;
1148
+ }
1149
+ continue;
1150
+ }
1151
+
1152
+ // (6) A table runs until a line that is not a row.
1153
+ // ⚠ NAMED GAP, DELIBERATELY NOT "FIXED": this branch has no indent test,
1154
+ // while (8) — which OPENS the table — now requires the header row to sit at
1155
+ // `indentWidth < 4`. So `| a | b |` / `|---|---|` / `<4SP>| c | d |` keeps
1156
+ // the third line in the table for us, where GFM ends the table and starts an
1157
+ // indented code block. The symmetric-looking change (adding an indent test
1158
+ // here) is NOT made, because **no reference available to this repo can say
1159
+ // which answer is right**, and shipping a change on an unarbitrated hunch is
1160
+ // how this module's boundary logic got into trouble in the first place:
1161
+ // - `commonmark@0.31.2` has no tables, so it never enters this state at
1162
+ // all and reads the whole run as one paragraph. It agrees with us here
1163
+ // for a reason that has nothing to do with the question.
1164
+ // - `marked`'s GFM lexer does answer, but it is measurably non-conformant
1165
+ // on exactly this boundary — for `prose` / `<4SP>prose` / `===` the spec
1166
+ // and `commonmark` both say setext H1 (an indented line cannot interrupt
1167
+ // a paragraph, so it is a lazy continuation) while `marked` says
1168
+ // paragraph. A reference that is wrong on the control cannot settle the
1169
+ // case.
1170
+ // Direction if GFM turns out to be right: content loss, the loud one.
1171
+ // Closing this needs a third reference (cmark-gfm), not more reasoning.
1172
+ if (open && open.kind === 'table') {
1173
+ if (line.includes('|')) { out[i] = { type: 'table', heading: null }; continue; }
1174
+ out[i] = { type: 'paragraph', heading: null };
1175
+ open = { kind: 'paragraph', start: i };
1176
+ continue;
1177
+ }
1178
+
1179
+ // (7) Indented code — which cannot interrupt a paragraph. After one, an
1180
+ // indented line is a lazy continuation and the paragraph stays open.
1181
+ // ⚠ `indentWidth(...) >= 4`, NOT a raw-line regex. This branch used
1182
+ // `INDENTED_CODE_RE = /^(\t| {4,})/`, which decides the SAME question as
1183
+ // `indentWidth` with different code — and they disagreed: a space followed
1184
+ // by a tab is column 4 (a tab advances to the next multiple of four), so
1185
+ // `indentWidth(' \tx') === 4` while the regex said false. Both directions
1186
+ // measured: `<SP><TAB>prose` became a paragraph, so a following `---` was a
1187
+ // setext underline and the pop deleted the body (content loss); and
1188
+ // `<SP><SP><SP><TAB>| a | b |` became a table header, so the section ran on
1189
+ // (silent pass). This module's own rule names it — a regex against a raw
1190
+ // line, anywhere else, to decide what kind of line it is.
1191
+ //
1192
+ // ⚠⚠ WHY WIDENING THIS BRANCH STOLE NOTHING FROM THE BRANCHES ABOVE IT, and
1193
+ // this is the part that was missing rather than wrong. `indentWidth >= 4` is
1194
+ // strictly wider than the regex it replaced, so it can only take lines that
1195
+ // used to reach branches further down — unless it also overlaps branch (4).
1196
+ // It cannot, and the reason is a property of every opener in this module,
1197
+ // not of this line: **every raw-prefix opener is `^ {0,3}` followed by a
1198
+ // character that is neither a space nor a tab**, which bounds any line one
1199
+ // of them matches at `indentWidth <= 3`; while `indentWidth >= 4` requires
1200
+ // four spaces, or a tab, inside the first four columns, which no `^ {0,3}X`
1201
+ // can match. So (4) and (7) partition cleanly.
1202
+ // ⚠ That is a property of the OTHER expressions, so it can be broken from a
1203
+ // distance: admit a tab into any `^ {0,3}` prefix — write `^[ \t]{0,3}`,
1204
+ // which looks like a tolerance improvement — and this branch starts eating
1205
+ // lines that opener was meant to claim, silently. `assertOpenerIndentPartition`
1206
+ // in `test/upgrade-drift.mjs` sweeps every opener against every indent form
1207
+ // so the break is a test failure and not a review round.
1208
+ if (!paraOpen && indentWidth(line) >= 4) {
1209
+ out[i] = { type: 'code', heading: null };
1210
+ open = { kind: 'code', start: open && open.kind === 'code' ? open.start : i };
1211
+ continue;
1212
+ }
1213
+
1214
+ // (8) A GFM table opens when a delimiter row sits directly under a
1215
+ // pipe-bearing paragraph line. Requiring the header is what keeps
1216
+ // ordinary prose containing a pipe — "the Command | Query split" — from
1217
+ // being read as a table, which an earlier `line.includes('|')` test got
1218
+ // wrong in the silent-success direction.
1219
+ // ⚠ The header row is the line DIRECTLY ABOVE the delimiter row, wherever
1220
+ // that line sits in the paragraph. Requiring it to be the paragraph's first
1221
+ // line (`open.start === i - 1`) was a regression against the predicate this
1222
+ // pass replaced, whose backward scan had no such restriction: one intro
1223
+ // sentence above a table — `Our rows:` / `| a | b |` / `|---|---|` — made the
1224
+ // whole run one paragraph, so a following `---` became a setext underline
1225
+ // and the pop deleted the ENTIRE section body. Total content loss for that
1226
+ // section, on input a hand-edited `_conventions.md` produces immediately.
1227
+ // Any earlier lines of the paragraph stay a paragraph; only the header row
1228
+ // and the delimiter row become the table.
1229
+ // ⚠⚠ THE HEADER ROW'S OWN INDENT IS CHECKED, and leaving it out was a
1230
+ // silent-pass defect this very branch introduced. `TABLE_DELIMITER_ROW_RE`
1231
+ // constrains the DELIMITER row to `^ {0,3}`, but the header test is only
1232
+ // `.includes('|')`. While the header had to be the paragraph's first line
1233
+ // that gap was unreachable — an indented line can only be a paragraph
1234
+ // CONTINUATION, since indented code cannot interrupt a paragraph — so
1235
+ // widening the rule opened it. A 4-column-indented header is indented code
1236
+ // to CommonMark and no table at all to GFM (`marked` emits no table token),
1237
+ // yet we read one, the `---` below stopped ending the section, and a stale
1238
+ // file reported `current`. Measured: five rows where both references
1239
+ // disagreed with us and neither with each other.
1240
+ // `i > 0` is deliberately gone: `paraOpen` cannot be true at i === 0 (`open`
1241
+ // starts null and is only assigned in the loop), so it never guarded
1242
+ // anything while reading like the bounds check that made this branch safe.
1243
+ // ⚠⚠ RELABELLING `out[i - 1]` RESTS ON AN INVARIANT NOTHING ELSE STATES:
1244
+ // when `paraOpen` is true at `i`, `out[i - 1]` is necessarily `'paragraph'`.
1245
+ // It holds because the only branches that leave `open.kind === 'paragraph'`
1246
+ // are (9) and (6)'s else-path, and both write `'paragraph'` to their own
1247
+ // line. So the overwrite below always replaces a label this loop just wrote
1248
+ // for the header row — never a `code`, `list` or `heading` line.
1249
+ // ⚠ A future branch that labels its line something else while leaving a
1250
+ // paragraph open would have that label silently overwritten as `table`, and
1251
+ // nothing here would notice. If you add such a branch, this line is the one
1252
+ // that breaks — and it breaks quietly, which is why the invariant is written
1253
+ // down instead of being left to be re-derived.
1254
+ if (paraOpen && indentWidth(lines[i - 1]) < 4 && lines[i - 1].includes('|') && TABLE_DELIMITER_ROW_RE.test(line)) {
1255
+ out[i - 1] = { type: 'table', heading: null };
1256
+ out[i] = { type: 'table', heading: null };
1257
+ open = { kind: 'table', start: i - 1 };
1258
+ continue;
1259
+ }
1260
+
1261
+ // (9) Paragraph: a continuation of the open one, or a new one.
1262
+ out[i] = { type: 'paragraph', heading: null };
1263
+ if (!paraOpen) open = { kind: 'paragraph', start: i };
1264
+ }
1265
+
1266
+ // ⚠⚠ AN HTML BLOCK STILL OPEN AT EOF IS MARKED, and this is not a parsing
1267
+ // nicety — it was a HIGH silent pass reachable on a file nobody had to
1268
+ // construct. HTML block types 1-5 close only on their end condition, so an
1269
+ // adopter who leaves one `<!--` unclosed mid-edit puts the ENTIRE REST OF THE
1270
+ // FILE inside that block. The parse is right; what was wrong was that
1271
+ // `conventionsSectionBodies` kept pushing those lines into the open section's
1272
+ // body, so the section swallowed every later section, reached a quoted copy
1273
+ // of the rule in an appendix, and a genuinely stale file reported `current`.
1274
+ // Measured through the real `doctor`: closing the comment produced two
1275
+ // warnings, leaving it open produced none.
1276
+ //
1277
+ // ⚠ Only HTML. A paragraph, list or code block open at EOF is ordinary — the
1278
+ // document simply ended. Fenced blocks never reach here because
1279
+ // `blankFencedBlocks` has already emptied them, and that immunity is
1280
+ // accidental rather than designed, which is exactly why this needed marking
1281
+ // rather than another special case.
1282
+ //
1283
+ // The flag is additive: every existing consumer reads `.type` and `.heading`.
1284
+ // ⚠⚠ `open.endRe` IS THE WHOLE POINT, and leaving it out was a regression this
1285
+ // very block introduced. Only types 1-5 close on an end condition; **type 6
1286
+ // closes on a blank line**, and `endRe` is null for it. A document ending in a
1287
+ // type-6 block therefore looked "never closed" — but a file that ends with a
1288
+ // newline supplies a trailing blank through the splitter, so the block closes
1289
+ // normally, while a file WITHOUT a trailing newline did not. Two files one
1290
+ // byte apart, and the shorter one got a spurious "unclosed HTML block"
1291
+ // warning plus a false `stale` on a section that plainly carries the rule.
1292
+ // Content loss, on `<div>…</div>` — ordinary hand-written HTML.
1293
+ // The comment above already said types 1-5; the code did not agree with it.
1294
+ if (open && open.kind === 'html' && open.endRe) {
1295
+ for (let i = open.start; i < out.length; i += 1) out[i].unclosed = true;
1296
+ }
1297
+
1298
+ return out;
1299
+ }
1300
+
1301
+ // ⚠ `toUpperCase`, not `toLowerCase`, and the difference is not cosmetic:
1302
+ // `headingResolves` PREFIX-matches on this key, and lowercasing is not
1303
+ // prefix-preserving. Greek capital sigma lowercases to a FINAL sigma at a word
1304
+ // end and a medial one elsewhere, so `ΑΣ` becomes `ας` while `ΑΣΒΓ` becomes
1305
+ // `ασβγ` — and `"ασβγ".startsWith("ας")` is false although the raw strings do
1306
+ // prefix-match. A `<file>.md § ΑΣ` reference to a heading `ΑΣΒΓ` was reported as
1307
+ // dangling. Uppercasing has no context-sensitive mapping in the default
1308
+ // case-conversion tables, so it preserves prefixes; `ß` → `SS` changes length
1309
+ // but preserves them too.
1310
+ // ⚠ Any fold works for the exact and Set comparisons; only the prefix consumer
1311
+ // constrains the choice. Keep ONE fold for all three rather than a
1312
+ // prefix-preserving one here and a convenient one there — that split is how the
1313
+ // three consumers of this normaliser drifted apart before.
1314
+ function headingKey(text) {
1315
+ return normalizeHeading(text).toUpperCase();
1316
+ }
1317
+
1318
+ // ALL bodies whose heading matches — `_conventions.md` is hand-edited and a
1319
+ // duplicated heading is ordinary. Taking only the first made a file that
1320
+ // carried the rule in its second copy report `stale`.
1321
+ function conventionsSectionBodies(content, heading) {
1322
+ // Blanked by the same expression `visibleTextLines` returns — see
1323
+ // `classifiedVisible`. Text a reader never sees is not this section's content.
1324
+ const { info, lines } = classifiedVisible(content);
1325
+ const target = headingKey(heading);
1326
+ const bodies = [];
1327
+ let level = 0;
1328
+ let body = null;
1329
+ for (let i = 0; i < lines.length; i += 1) {
1330
+ const h = info[i].heading;
1331
+ if (h) {
1332
+ // A setext heading's TEXT lines were already pushed into the open body —
1333
+ // take them back out. `h.start` is where the underlined paragraph began,
1334
+ // so this pops all of them.
1335
+ //
1336
+ // ⚠ The old version popped exactly ONE line and guarded it with
1337
+ // `body[body.length - 1].trim() === h.text`, because it assumed a setext
1338
+ // heading's text was a single line. A soft-wrapped one therefore left its
1339
+ // earlier lines sitting in the body as though they were content, and
1340
+ // reported the heading under only its last line — so a section titled
1341
+ // `Ceremony Scaling` / `(Project Application)` / `===` did not match the
1342
+ // fingerprint that names it. The classification knows the paragraph's
1343
+ // extent, so both halves are now right for free.
1344
+ if (body && h.setext) {
1345
+ for (let k = h.start; k < i && body.length > 0; k += 1) body.pop();
1346
+ }
1347
+ if (body && h.level <= level) {
1348
+ bodies.push(body.join('\n'));
1349
+ body = null;
1350
+ level = 0;
1351
+ }
1352
+ if (!body && headingKey(h.text) === target) {
1353
+ level = h.level;
1354
+ body = [];
1355
+ }
1356
+ continue;
1357
+ }
1358
+ // ⚠ A line inside an HTML block that never closed is NOT this section's
1359
+ // content — it is the interior of a block that swallowed the rest of the
1360
+ // file. Pushing it let a section reach text from a later section and report
1361
+ // `current` on a file that had genuinely lost the rule: silent pass, the
1362
+ // worst direction, and reachable from one unclosed `<!--`.
1363
+ // Stopping here is the conservative half of the fix — we only trust content
1364
+ // we can attribute to the section. The other half is telling the user why,
1365
+ // which `findConventionsDrift` does, because otherwise the resulting `stale`
1366
+ // is correct and inexplicable.
1367
+ if (info[i].unclosed) break;
1368
+ if (body) body.push(lines[i]);
1369
+ }
1370
+ if (body) bodies.push(body.join('\n'));
1371
+ return bodies;
1372
+ }
1373
+
1374
+ // The file's lines with everything a reader cannot see blanked out: fenced code
1375
+ // (already blanked on the way in) and the interior of an HTML block that renders
1376
+ // no text — a comment, a processing instruction, a declaration, CDATA, a
1377
+ // `<script>` or a `<style>`.
1378
+ //
1379
+ // ⚠⚠ WHY THIS EXISTS AS A FUNCTION RATHER THAN A TEST AT EACH CONSUMER. The
1380
+ // rewrite made `classifyLines` the single authority on *what kind of line this
1381
+ // is* — and then every CONTENT extractor went on reading the raw text without
1382
+ // asking. So three consumers each decided independently, by not deciding:
1383
+ // `findConventionsDrift` matched a fingerprint marker inside `<!-- … -->` and
1384
+ // certified a stale file as `current`; `parseContextLine` read a commented-out
1385
+ // machine-readable policy value as though it were live; `extractSectionRefs`
1386
+ // reported a reference inside a comment as a dangling link, with the `-->`
1387
+ // captured into the heading text (`p082-b3-k1`, all three reproduced).
1388
+ //
1389
+ // ⚠ The module was already INCONSISTENT about this, which is the tell that it
1390
+ // was never decided: a retired string below an *unclosed* comment was correctly
1391
+ // ignored (the section body stops there), while the same string inside a *closed*
1392
+ // comment counted. Same question, two answers, depending on whether the user
1393
+ // typed `-->`.
1394
+ //
1395
+ // BLANK, do not delete: `extractSectionRefs` reports 1-based line numbers and
1396
+ // `conventionsSectionBodies` pops setext text by index. This is the same
1397
+ // convention `blankFencedBlocks` established, for the same reason.
1398
+ //
1399
+ // ⚠ This is a real semantic decision, not a bug fix with an obvious answer:
1400
+ // it says a commented-out rule is ABSENT. `commonmark` agrees a reader never
1401
+ // sees it, and the section differential's whole question is "does a reader see
1402
+ // the marker in this section" — so the arbiter this module already trusts was
1403
+ // answering one question while the implementation answered another.
1404
+ // ⚠⚠ ONE APPLICATION OF THE RULE, not two. The first draft of this fix had
1405
+ // `visibleTextLines` blank the lines and `conventionsSectionBodies` repeat the
1406
+ // test inline (`info[i].invisible ? '' : line`) because it needed `info` anyway.
1407
+ // Both read the same flag, so the DECISION was single-sourced — but the two
1408
+ // expressions could still drift, and "one rule, two expressions" is the defect
1409
+ // class this module exists to hold shut. It was caught by measurement, not by
1410
+ // review: breaking the projection left the `conventionsSectionBodies` pins
1411
+ // green, which is precisely the "my pin cannot fail against the thing it names"
1412
+ // shape the mutation harness was built for.
1413
+ // ⚠ SPACES, not deletion, and the span rather than the line. Two reasons, both
1414
+ // paid for: `extractSectionRefs` compares a match's index against the line's
1415
+ // length, so shortening a line moves every column it reads; and an end condition
1416
+ // can leave rendered text after it on the same line (`visibleFrom`), which
1417
+ // whole-line blanking silently dropped.
1418
+ function classifiedVisible(content) {
1419
+ const lines = blankFencedBlocks(String(content));
1420
+ const info = classifyLines(lines);
1421
+ return {
1422
+ info,
1423
+ lines: lines.map((line, i) => {
1424
+ if (!info[i] || !info[i].invisible) return line;
1425
+ const from = info[i].visibleFrom;
1426
+ const hidden = from === undefined ? line.length : Math.min(from, line.length);
1427
+ return ' '.repeat(hidden) + line.slice(hidden);
1428
+ })
1429
+ };
1430
+ }
1431
+
1432
+ function visibleTextLines(content) {
1433
+ return classifiedVisible(content).lines;
1434
+ }
1435
+
1436
+ // The same content with every CODE line masked to spaces — same length, same
1437
+ // line count, so an index into the result is an index into the original.
1438
+ //
1439
+ // ⚠⚠ WHY A MASK AND NOT THE LINE ARRAY. `configure-agents` finds its managed
1440
+ // region with `indexOf` over the raw file and then SLICES the original by that
1441
+ // index, so anything that shortens a line moves the cut and corrupts the write.
1442
+ // A mask keeps every offset while removing the text that must not match.
1443
+ //
1444
+ // It exists because the write path — the only path in this tool that modifies a
1445
+ // user's file — decided a structural question by not asking: a user documenting
1446
+ // Dflow in their own `AGENTS.md`, with the marker comments shown inside a fenced
1447
+ // example, had that example treated as the managed region and OVERWRITTEN.
1448
+ // `runConfigureAgents` exited 0 and `doctor` then reported all clean
1449
+ // (`p082-b3-k3` finding 1). USER CONTENT LOSS, which is the worst thing this
1450
+ // codebase can do.
1451
+ //
1452
+ // ⚠ Fenced and indented code only. A marker inside a `<pre>` block is still
1453
+ // taken as a region boundary — `pre` renders its text, so it is not code to this
1454
+ // module, and closing that case needs a decision about what `<pre>` MEANS here
1455
+ // rather than another line in this function. Stated so the gap is known rather
1456
+ // than discovered.
1457
+ function maskCodeBlocks(content) {
1458
+ const s = String(content);
1459
+ const raw = s.split('\n');
1460
+ const blanked = blankFencedBlocks(s);
1461
+ const info = classifyLines(blanked);
1462
+ return raw.map((line, i) => {
1463
+ const isCode = blanked[i] !== line || (info[i] && info[i].type === 'code');
1464
+ return isCode ? ' '.repeat(line.length) : line;
1465
+ }).join('\n');
1466
+ }
1467
+
1468
+ // Where an HTML block that never closes begins, or -1. Reads the SAME
1469
+ // classification every other consumer reads rather than re-deciding it — the
1470
+ // module's one rule about not spelling a question twice.
1471
+ function unclosedHtmlBlockLine(content) {
1472
+ const lines = blankFencedBlocks(String(content));
1473
+ const info = classifyLines(lines);
1474
+ const at = info.findIndex((c) => c && c.unclosed);
1475
+ return at === -1 ? -1 : at + 1; // 1-based, for a message a human reads
1476
+ }
1477
+
1478
+ // ---------------------------------------------------------------------------
1479
+ // PROPOSAL-084 — UNCERTAINTY DETECTORS.
1480
+ //
1481
+ // These answer a different KIND of question from everything above, and that
1482
+ // difference is the whole reason the feature is safe to ship on top of a parser
1483
+ // with known gaps. The checks above ask "where does this block END" — the
1484
+ // question whose edge cases the `DELIBERATELY DOES NOT IMPLEMENT` list is about.
1485
+ // These ask "is this SHAPE present anywhere in the file", which needs no block
1486
+ // extent, no container interior, and no arbiter. So a detector cannot inherit the
1487
+ // defect it warns about. Three of the four review rounds that evaluated this
1488
+ // proposal wrote prototypes and measured the false-positive rate on real corpora
1489
+ // (0/140, 0/151, 0/645) rather than reasoning about it.
1490
+ //
1491
+ // ⚠ THE ONE PLACE THAT CLAIM NEEDS QUALIFYING, stated here rather than left for
1492
+ // a reviewer to find: `unterminatedCommentLine` is combined with
1493
+ // `unclosedHtmlBlockLine` by its caller, which DOES read the classification. That
1494
+ // is deliberate and it is a differential, not a dependency — "there is an
1495
+ // unterminated comment, yet the document-level scan did not flag one" is exactly
1496
+ // what it means for the opener to be inside a container. If the classification is
1497
+ // wrong the detector goes quiet or gets noisy; it cannot manufacture a new silent
1498
+ // pass, because the silent pass is the state that exists TODAY.
1499
+ //
1500
+ // ⚠ Every detector runs on `blankFencedBlocks` output. That masker is a
1501
+ // self-contained fence scanner — it never calls `classifyLines` — so masking
1502
+ // cannot import the boundary bugs either. Fenced examples of these shapes (this
1503
+ // repo's own docs are full of them) therefore do not fire.
1504
+ // ⚠⚠ THAT SENTENCE IS TRUE ONLY FOR A FENCE THE MASKER CAN SEE, and it was stated
1505
+ // without the qualifier for eight rounds (`p084-xv8` finding 5). The masker works
1506
+ // on RAW document lines: `FENCE_OPEN_RE` carries `^ {0,3}`, and nothing strips a
1507
+ // container prefix first. So it misses exactly two things — a fence opened inside a
1508
+ // block quote (`> ` before the backticks), and a fence whose RAW indent is four
1509
+ // spaces or more. A comment inside either still reports `comment-inside-container`.
1510
+ // ⚠ "Four or more spaces of raw indent" is the whole rule, and an earlier version
1511
+ // of this note said "a deep list item's content column", which is wrong twice
1512
+ // (`p084-xv9` finding 3): it happens under an ordinary `- item` as soon as the
1513
+ // fence is indented four, and it does NOT happen at two or three however deep the
1514
+ // list is. Describing an accepted defect inaccurately is worse than not describing
1515
+ // it — the next maintainer acts on the description.
1516
+ // ⚠⚠ THE DIRECTION RECORDED HERE WAS WRONG, AND THE DECISION WAS TAKEN ON IT
1517
+ // (`p084-y2` finding 1). This said the gap costs only NOISE. Measured: an unmasked
1518
+ // fence's content is counted as section body, so quoting an upstream rule as an
1519
+ // example inside a block quote — or at raw indent four — makes the rule look PRESENT
1520
+ // while the live copy of it has been changed. Reproduced end to end: `All checks
1521
+ // passed` over a `_conventions.md` whose rule was switched off. **It costs silence
1522
+ // too.** The maintainer accepted this gap on 2026-08-09 believing it was noise-only;
1523
+ // that premise is retired.
1524
+ //
1525
+ // ⚠⚠ THE DECISION WAS RETAKEN ON THE CORRECTED PREMISE (maintainer, 2026-08-10) AND
1526
+ // THE GAP IS STILL ACCEPTED — as a temporary accepted RISK, owner maintainer, to be
1527
+ // paid before the `doctor-reader-gaps` proposal closes. Not as a resolved defect.
1528
+ // What settled it is that the obvious repair is WORSE, and measurably so
1529
+ // (`p084-sol1`): excluding the quoted lines from the section body **fabricates
1530
+ // markers that were never in the file**. Removal is not monotonic for substring
1531
+ // matching — drop a line and its neighbours become adjacent, and the marker
1532
+ // comparison collapses whitespace, so
1533
+ // `The cascade result is` / <dropped line> / `a floor: rows may only escalate.`
1534
+ // reassembles into the very rule the file no longer states. Reproduced LF and CRLF:
1535
+ // a section that correctly reports `stale` today reports `current` after the repair.
1536
+ // Blanking the line instead of dropping it does not help, for the same reason.
1537
+ // ⚠ AND THE CLASS IS WIDER THAN FENCES, which is the part that makes a local patch
1538
+ // futile: substring matching cannot tell a RULE from PROSE ABOUT THAT RULE. An
1539
+ // indented example, a bare `> ` quotation and an ordinary sentence quoting the rule
1540
+ // all read as the live rule. The unmasked fence is one instance, not the class — so
1541
+ // closing it would move the next maintainer from a disclosed defect to an
1542
+ // undisclosed one, which is this proposal's own most expensive lesson.
1543
+ // The real repair is one walker emitting ordered evidence segments with HARD
1544
+ // boundaries that matching cannot cross, plus a declarative required/forbidden
1545
+ // direction policy — a redesign, out of this proposal's scope, tracked in
1546
+ // `planning/opt-in-backlog.md` under `doctor-reader-gaps`.
1547
+ // Pinned as-is so it cannot change without someone meeting the decision again.
1548
+ //
1549
+ // ⚠ DIRECTION RULE FOR ADDING ONE: a detector may be noisy, never silent. Where
1550
+ // the shape is ambiguous, do not fire. A false warning is visible and the user
1551
+ // can dismiss it; a missed shape restores exactly the silence this exists to
1552
+ // break. Gap D from the proposal is the worked example of the limit — its
1553
+ // prototype false-fired 5 times in 151 files, and it is deliberately NOT built,
1554
+ // because for that shape the warning really is worse than the gap.
1555
+ // ---------------------------------------------------------------------------
1556
+
1557
+ // Which positions on one line sit inside a CommonMark code span. Needed because
1558
+ // `` `<!-- x -->` `` IS rendered — a reader sees it — so flagging it would be a
1559
+ // plain false positive. Measured: the probe's I3 control.
1560
+ //
1561
+ // ⚠ Line-level, and that is a stated limit rather than an oversight: a code span
1562
+ // may legally straddle lines. There the mask under-covers, so a straddling span
1563
+ // could produce a NOISY hit, which is the acceptable direction.
1564
+ //
1565
+ // ⚠⚠ A BACKSLASH SUPPRESSES A BACKTICK ONLY WHILE NO SPAN IS OPEN, and getting
1566
+ // this wrong has produced a silent pass TWICE, from opposite sides
1567
+ // (`p084-xv2`, then `p084-xv3` on the repair). CommonMark §6.1: backslash escapes
1568
+ // do not work inside a code span — so whether a backtick "is escaped" depends on
1569
+ // whether a span is already open, which is the very thing this function computes.
1570
+ // The first repair asked the question as a PRE-PASS, filtering escaped backticks
1571
+ // out before pairing, and that is circular. It cost a new blocker: in
1572
+ // `` `foo\` <!-- rule --> `` the real closing delimiter was discarded as escaped,
1573
+ // the opener paired with a later run instead, and the mask swallowed a comment a
1574
+ // reader never sees while doctor went on reading it.
1575
+ //
1576
+ // So this is a single left-to-right scan, which is what an inline parser actually
1577
+ // does: an escape consumes the next character while looking for an OPENER, and the
1578
+ // search for the CLOSER is raw. `` `foo\` `` is a code span whose content is
1579
+ // `foo\`, exactly as the reference renderer has it.
1580
+ function codeSpanMask(line) {
1581
+ const mask = new Array(line.length).fill(false);
1582
+ let i = 0;
1583
+ while (i < line.length) {
1584
+ // No span is open here, so a backslash consumes the character after it and an
1585
+ // escaped backtick is plain text that opens nothing.
1586
+ if (line[i] === '\\') { i += 2; continue; }
1587
+ if (line[i] !== '`') { i += 1; continue; }
1588
+ let openEnd = i;
1589
+ while (openEnd < line.length && line[openEnd] === '`') openEnd += 1;
1590
+ const len = openEnd - i;
1591
+ // CommonMark pairs a backtick run with the next run of EQUAL length. No
1592
+ // backslash handling in here — see the note above; that is the whole fix.
1593
+ let closeStart = -1;
1594
+ for (let j = openEnd; j < line.length;) {
1595
+ if (line[j] !== '`') { j += 1; continue; }
1596
+ let k = j;
1597
+ while (k < line.length && line[k] === '`') k += 1;
1598
+ if (k - j === len) { closeStart = j; break; }
1599
+ j = k;
1600
+ }
1601
+ // An unpaired run is literal text. Resume just past it rather than past the
1602
+ // whole line, so a later run on the same line can still open a span.
1603
+ if (closeStart === -1) { i = openEnd; continue; }
1604
+ for (let p = i; p < closeStart + len; p += 1) mask[p] = true;
1605
+ i = closeStart + len;
1606
+ }
1607
+ return mask;
1608
+ }
1609
+
1610
+ // Every live `<!--` on a line, as column offsets. CommonMark parses code spans
1611
+ // and backslash escapes only in inline content. Inside a raw-HTML block the line
1612
+ // is passed through as HTML, so neither syntax can make an apparent opener
1613
+ // visible to the reader.
1614
+ function commentOpenersOutsideCode(line, inlineSyntaxApplies = true) {
1615
+ const mask = inlineSyntaxApplies ? codeSpanMask(line) : [];
1616
+ const out = [];
1617
+ for (let at = line.indexOf('<!--'); at !== -1; at = line.indexOf('<!--', at + 4)) {
1618
+ if (mask[at]) continue;
1619
+ // CommonMark backslash escaping is parity-based: an odd run escapes `<`, so
1620
+ // `\<!--` is visible text; an even run leaves `<` live, so `\\<!--` still
1621
+ // opens a comment after rendering one literal backslash. Count at the opener
1622
+ // rather than stripping escapes globally — codeSpanMask owns backticks, and
1623
+ // changing its input would move the offsets the two decisions share.
1624
+ let backslashes = 0;
1625
+ for (let p = at - 1; p >= 0 && line[p] === '\\'; p -= 1) backslashes += 1;
1626
+ if (!inlineSyntaxApplies || backslashes % 2 === 0) out.push(at);
1627
+ }
1628
+ return out;
1629
+ }
1630
+
1631
+ // GAP I and the container half of gap A. Both answer ONE question — **is there a
1632
+ // comment whose text this module still counts as live?** — and they are split only
1633
+ // because the two causes need different repair advice.
1634
+ //
1635
+ // ⚠⚠ THE ARBITER IS `classifyLines`, NOT A REGEX, AND THAT IS THE WHOLE FIX
1636
+ // (`p084-y1` findings 1, 2 and 5). The first version asked
1637
+ // `HTML_COMMENT_OPEN_RE.test(line)` — a DOCUMENT-LEVEL opener rule — to decide
1638
+ // whether a line's leading `<!--` opened a block. Inside a list item it does not:
1639
+ // `- x` / ` <!-- rule -->` is typed `list`, its text stays live, and a reader
1640
+ // sees nothing. That shape fired NOTHING, and it is exactly where the shipped
1641
+ // repair advice ("put the comment on a line of its own") sent the user — out of a
1642
+ // disclosed silent pass and into an undisclosed one. Restating a rule the
1643
+ // classifier already owns is this module's oldest defect; the module's own rule
1644
+ // says callers READ LABELS. So these read labels.
1645
+ //
1646
+ // ⚠ It also retires a detector. The previous `unclosed-html-in-container` fired on
1647
+ // "an unterminated comment that the document-level scan did not flag", which is
1648
+ // not the shape that hurts: measured, ANY later `-->` anywhere in the file —
1649
+ // including the one `unclosed-html-block`'s own action tells the user to add —
1650
+ // silenced it while the rule stayed hidden from readers. A check that keeps being
1651
+ // defeated a new way is mis-scoped, not under-specified. What actually matters is
1652
+ // "is the text hidden from a reader but not from us", and termination has nothing
1653
+ // to do with it.
1654
+ //
1655
+ // ⚠ Lines the classifier calls `code` are skipped in both: a reader sees indented
1656
+ // code verbatim, so there is no divergence to report and a warning there is the
1657
+ // pure noise gap D was rejected for.
1658
+
1659
+ function commentContext(content) {
1660
+ const lines = blankFencedBlocks(String(content));
1661
+ return { lines, info: classifyLines(lines) };
1662
+ }
1663
+
1664
+ // Everything that can legitimately sit between the start of a line and the start of
1665
+ // that line's CONTENT: indentation, block-quote markers, list markers. Built from
1666
+ // the same marker fragments the list predicates use, so widening a marker cannot
1667
+ // leave this behind — the reason `BULLET_MARKER` exists as a fragment at all.
1668
+ //
1669
+ // ⚠ It exists so `> <!-- rule -->` is judged as a comment that STARTS its content
1670
+ // (the container case) rather than as a mid-line one. Getting that wrong picks the
1671
+ // wrong id and prints repair advice for the wrong shape; both ids disclose, so the
1672
+ // cost is a confusing message rather than a silent pass.
1673
+ const CONTAINER_PREFIX_ONLY_RE = new RegExp(`^(?:[ \\t>]|${BULLET_MARKER}|${ORDERED_MARKER})*$`);
1674
+ function startsContent(line, at) {
1675
+ return CONTAINER_PREFIX_ONLY_RE.test(line.slice(0, at));
1676
+ }
1677
+
1678
+ // ⚠⚠ ONE ANSWER TO "IS THIS LINE INSIDE A CONTAINER WHOSE INTERIOR THIS MODULE
1679
+ // DOES NOT PARSE", read by every detector that needs it (`p084-xv3` finding 3).
1680
+ // The comment detectors spelled `list | blockquote | html` and the two boundary
1681
+ // detectors spelled `list | blockquote`, so a `<custom-element>` inside
1682
+ // `<details>` was container content to one and a document-level divergence to the
1683
+ // other — and only the second one warned about it. That is the same
1684
+ // two-rival-predicates shape `commentDisposition` exists to retire, one level up:
1685
+ // retiring it for the comment ids and leaving the neighbours to restate it is how
1686
+ // the class survived. Adding a container type now reaches every reader at once.
1687
+ //
1688
+ // ⚠ Deliberately NOT a block decision. Nothing here decides where a block starts
1689
+ // or ends — `classifyLines` has already done that, and this only reads the label
1690
+ // it produced. Being wrong here makes a warning noisy or absent; it cannot move a
1691
+ // boundary.
1692
+ const UNPARSED_CONTAINER_TYPES = ['list', 'blockquote', 'html'];
1693
+ function inUnparsedContainer(c) {
1694
+ return !!c && UNPARSED_CONTAINER_TYPES.includes(c.type);
1695
+ }
1696
+
1697
+ // GAP I — a comment that begins part-way through a line. It is an inline span, not
1698
+ // a block, so every consumer counts its contents as live document text while a
1699
+ // reader sees nothing. Two shapes: `prose <!-- rule -->`, and a SECOND comment on a
1700
+ // line that opens with one (`<!-- a --> <!-- rule -->`), because only the first
1701
+ // span is ever masked.
1702
+ // ⚠⚠ ONE FUNCTION DECIDES WHICH ID OWNS A LINE, and the two exported detectors are
1703
+ // views over it. Two independent predicates is how the first version left a hole:
1704
+ // `p084-xv1` found a comment at column 0 inside a `<details>` block that fell into
1705
+ // NEITHER id — the container detector skipped it because the line is typed `html`,
1706
+ // the inline detector skipped it because the comment starts the line's content, and
1707
+ // `parseContextLine` happily returned a policy value the reader cannot see. A
1708
+ // partition has to be written as a partition or its gaps are invisible.
1709
+ //
1710
+ // ⚠ `blockStart` is the field that makes this decidable, and it exists because the
1711
+ // previous batch stopped rebuilding what the classifier knows and had it carry the
1712
+ // answer instead. A comment line inside a type-6 block reports the ENCLOSING
1713
+ // block's start, not its own index — which is exactly the difference between "this
1714
+ // comment opened a block, so its text really is hidden" and "this comment is
1715
+ // sitting inside someone else's block, where its text is still read".
1716
+ //
1717
+ // Returns 'inline', 'container', or null (nothing to report on this line).
1718
+ function commentDisposition(lines, info, i) {
1719
+ const c = info[i];
1720
+ if (!c || c.type === 'code') return null; // indented code renders verbatim
1721
+ const line = lines[i];
1722
+ // ⚠⚠ ASK WHAT THE CHECKS ACTUALLY READ, rather than guessing from the line's
1723
+ // type. `invisible` is NOT a whole-line boolean — it pairs with `visibleFrom`,
1724
+ // and `classifiedVisible` blanks only up to that column. So `<!-- a --> <!-- b -->`
1725
+ // is `invisible: true, visibleFrom: 10`: the first span is masked and the tail is
1726
+ // live, which is the documented "only the FIRST span is masked" gap. Treating
1727
+ // `invisible` as "this line is hidden" silently dropped that whole shape — caught
1728
+ // by re-running the earlier round's own fixtures after this function was
1729
+ // rewritten, which is the only reason it did not ship.
1730
+ const hiddenUpTo = c.invisible ? (c.visibleFrom === undefined ? line.length : c.visibleFrom) : 0;
1731
+ // The openers whose text survives into what every check reads. One inside the
1732
+ // masked region cannot mislead anyone and is not a finding.
1733
+ // ⚠ A comment inside a `<textarea>` IS reported here, deliberately — see the
1734
+ // block headed `THERE IS NO TEXTAREA EXEMPTION HERE` for why that exemption was
1735
+ // removed rather than fixed a fourth time. (Grep that phrase — an earlier version of
1736
+ // this pointer named a FUNCTION and was ~700 lines off, `p084-y2` finding 5.)
1737
+ const live = commentOpenersOutsideCode(line, c.type !== 'html' && !c.rawHtml).filter((at) => at >= hiddenUpTo);
1738
+ if (live.length === 0) return null;
1739
+ // A prefix only counts as container prefix when the classifier agrees the line is
1740
+ // in a container. `2. <!-- x -->` continuing a paragraph LOOKS like an ordered
1741
+ // item and is prose (`p084-xv1`), so shape alone picks the wrong id and prints
1742
+ // repair advice about leaving a container the user is not in.
1743
+ return (inUnparsedContainer(c) && startsContent(line, live[0])) ? 'container' : 'inline';
1744
+ }
1745
+
1746
+ function inlineHtmlCommentLine(content) {
1747
+ const { lines, info } = commentContext(content);
1748
+ for (let i = 0; i < lines.length; i += 1) {
1749
+ if (commentDisposition(lines, info, i) === 'inline') return i + 1;
1750
+ }
1751
+ return -1;
1752
+ }
1753
+
1754
+ // The container half of GAP A — the comment DOES begin its line, but the line sits
1755
+ // inside a list item or block quote, whose interior this module does not parse as
1756
+ // its own block sequence. So no HTML block opens, the text stays live, and the
1757
+ // reader loses it. The repair differs from the inline case, which is why it has its
1758
+ // own id: moving such a comment "onto its own line" changes nothing.
1759
+ function containerHtmlCommentLine(content) {
1760
+ const { lines, info } = commentContext(content);
1761
+ for (let i = 0; i < lines.length; i += 1) {
1762
+ if (commentDisposition(lines, info, i) === 'container') return i + 1;
1763
+ }
1764
+ return -1;
1765
+ }
1766
+
1767
+ // GAP B — HTML block type 7 (a complete tag whose name is NOT in the type-6 list,
1768
+ // alone on a line). Deliberately not implemented: it is the only type that cannot
1769
+ // interrupt a paragraph and recognizing it needs a real tag parser.
1770
+ //
1771
+ // The trigger is narrow and comes from the measured differential rather than from
1772
+ // imagination: a bare `<custom-element>` line at a block start, directly followed
1773
+ // by a setext underline (`---` / `===`). Ours ends the section EARLY where the
1774
+ // reference continues it, so the usual user-visible failure is a false `stale`.
1775
+ // It is worth disclosing, because a `stale` a user cannot reproduce is what
1776
+ // teaches them to stop believing the tool.
1777
+ // ⚠⚠ BUT IT IS NOT ONLY LOUD, AND THIS COMMENT SAID IT WAS FOR MANY ROUNDS
1778
+ // (`p084-y2` finding 3; still stale here after that finding was fixed in the
1779
+ // shipped `detail` and both pages — `p084-sol2` finding 3 caught the leftover).
1780
+ // Ending the section early also DROPS ITS TAIL, and the `CONVENTIONS_RETIRED`
1781
+ // fingerprints report on what is PRESENT in a body — so a retired row below the
1782
+ // shape stops being seen and its finding disappears. Measured: control reports
1783
+ // `ceremony-ef-tweak-t3:retired`, the same row under a `<my-widget>` / `---`
1784
+ // pair reports nothing. Say "unknown in BOTH directions", never "loud".
1785
+ // ⚠ The malformed-table shape has the mirror-image of this defect for the same
1786
+ // reason — they are one mechanism (a moved section boundary) that used to wear
1787
+ // two ids. It no longer has a detector; see GAP C below for why.
1788
+ // ⚠⚠ KEEP THE TWO HALVES OF THIS ID'S `detail` APART, because they say opposite
1789
+ // things and an earlier version of this very comment ran them together. For
1790
+ // type 7 the BOUNDARY moves ONE way — the section ends EARLIER than the reader's
1791
+ // does — while the RESULTS are unknown in BOTH directions, because ending early
1792
+ // drops the tail and a retired row down there stops being seen. "Both directions"
1793
+ // is true of the results and false of the boundary. Do not "fix" the shipped
1794
+ // `detail` into saying the boundary moves both ways; that is a different id's
1795
+ // shape, and this distinction has now been got wrong twice in opposite
1796
+ // directions (`doctor-table-boundary-account` in `planning/opt-in-backlog.md`).
1797
+ // ⚠⚠ A CLOSING TAG OPENS TYPE 7 TOO, and leaving it out made the id's own docs
1798
+ // describe a shape wider than the detector found (`p084-xv5` finding 2).
1799
+ // CommonMark's start condition is "a complete open tag … **or a complete closing
1800
+ // tag**, followed by only whitespace or the end of the line". Measured:
1801
+ // `</my-widget>` above `---` is one raw HTML block to the reference renderer and a
1802
+ // setext heading to this module — the divergence the id exists to disclose — and
1803
+ // the detector returned -1 on it. Undisclosed, which is the direction that matters.
1804
+ // ⚠⚠ THE TWO FORMS NEED DIFFERENT TAILS, and sharing them was a false positive
1805
+ // (`p084-xv6` finding 2). A closing tag may carry NEITHER attributes nor a `/`:
1806
+ // CommonMark's "complete closing tag" is `</`, a name, optional whitespace, `>`.
1807
+ // So `</my-widget attr>` and `</my-widget/>` are ordinary paragraph text, `---`
1808
+ // below them really is a setext heading, and this module ALREADY agrees with the
1809
+ // spec there — nothing diverges, and firing was noise on a shape that is not a tag.
1810
+ // ⚠ The tag-name shape is still spelled ONCE, as a fragment, which is what the
1811
+ // "no second nearly-identical regex" rule actually asks for — the previous version
1812
+ // obeyed the letter of that rule by sharing a tail that does not belong to both.
1813
+ // ⚠ `m[1] || m[2]`: the two branches capture into different groups. A reader
1814
+ // changing this must keep the detector's `name` lookup in step.
1815
+ //
1816
+ // ⚠⚠ THE OPEN-TAG TAIL IS THE SPEC'S GRAMMAR, NOT AN APPROXIMATION OF IT, and the
1817
+ // approximation it replaces produced the exact failure this whole proposal exists
1818
+ // to prevent (`p084-xv7`, blocker). The tail used to be `(?:[ \t][^<>]*)?` — "any
1819
+ // run without angle brackets" — which forbids `>`. CommonMark's raw-HTML grammar
1820
+ // ALLOWS `>` inside a quoted attribute value, so `<my-widget title="a > b">` is a
1821
+ // complete open tag, starts HTML block type 7, and swallows the `---` below it.
1822
+ // This module reads that `---` as a setext heading instead. Measured end to end in
1823
+ // a real project: the divergence was there, no detector fired, and doctor printed
1824
+ // `All checks passed` over a section that was genuinely stale underneath.
1825
+ // The same approximation was loose in the other direction — `<my-widget =x>` and
1826
+ // `<my-widget "x">` are not tags at all, this module agrees with the spec about
1827
+ // them, and the detector was firing anyway.
1828
+ //
1829
+ // So the productions are transcribed rather than guessed, each named after the
1830
+ // spec's own name for it. `[ \t]` and never `\s`, for the reason given at
1831
+ // `ATX_HEADING_RE`; a line-level detector cannot see the spec's multi-line
1832
+ // whitespace and does not need to.
1833
+ const BARE_TAG_NAME = '[A-Za-z][A-Za-z0-9-]*';
1834
+ const HTML_ATTR_NAME = '[A-Za-z_:][A-Za-z0-9_.:-]*';
1835
+ // unquoted | single-quoted | double-quoted. The quoted forms are why `>` cannot be
1836
+ // excluded wholesale.
1837
+ const HTML_ATTR_VALUE = '(?:[^ \\t"\'=<>`]+|\'[^\']*\'|"[^"]*")';
1838
+ const HTML_ATTR = `(?:[ \\t]+${HTML_ATTR_NAME}(?:[ \\t]*=[ \\t]*${HTML_ATTR_VALUE})?)`;
1839
+ const BARE_TAG_LINE_RE = new RegExp(
1840
+ `^ {0,3}<(?:(${BARE_TAG_NAME})${HTML_ATTR}*[ \\t]*\\/?|\\/(${BARE_TAG_NAME})[ \\t]*)>[ \\t]*$`
1841
+ );
1842
+ function htmlBlockType7Line(content) {
1843
+ const lines = blankFencedBlocks(String(content));
1844
+ const info = classifyLines(lines);
1845
+ const type6 = new RegExp(`^(?:${HTML_BLOCK_TAGS})$`, 'i');
1846
+ for (let i = 0; i < lines.length - 1; i += 1) {
1847
+ const m = lines[i].match(BARE_TAG_LINE_RE);
1848
+ const openName = m && m[1]; // complete open tag
1849
+ const closeName = m && m[2]; // complete closing tag
1850
+ const name = openName || closeName;
1851
+ if (!name || type6.test(name)) continue;
1852
+ // ⚠⚠ CommonMark §4.6 attaches the name exclusion to the OPEN-TAG form only:
1853
+ // "a complete open tag (with any tag name other than pre, script, style, or
1854
+ // textarea) **or a complete closing tag**". So `<pre/>` is not type 7 and
1855
+ // `</pre>` is. Relying on the classifier for this was not enough
1856
+ // (`p084-xv8` finding 4): `htmlBlockStart` recognises `<pre>` — the name
1857
+ // followed by space, `>` or end of line — but NOT `<pre/>`, where a `/`
1858
+ // follows the name, so that line stayed `paragraph`, escaped the container
1859
+ // skip, and got reported as a divergence that does not exist.
1860
+ if (openName && HTML_TYPE1_NAME_RE.test(openName)) continue;
1861
+ if (!SETEXT_UNDERLINE_RE.test(lines[i + 1])) continue;
1862
+ // ⚠ Inside a container nothing at DOCUMENT level moves, so the divergence this
1863
+ // id describes does not occur there (`p084-xv2`, widened to HTML blocks by
1864
+ // `p084-xv3`). The shape is about a section boundary shifting; a
1865
+ // `<custom-element>` sitting in a list — or inside `<details>` — is container
1866
+ // content to both readings. Firing there is noise on legitimate content.
1867
+ // ⚠ At document level this shape's line is typed `paragraph`, never `html`
1868
+ // (the module does not implement type 7 — that is the whole gap), so reading
1869
+ // the shared container predicate here cannot silence the detector's real case.
1870
+ if (inUnparsedContainer(info[i])) continue;
1871
+ // ⚠ THE BLOCK-START TEST IS NOT OPTIONAL, and leaving it out was a false
1872
+ // positive both doc pages already excluded in words ("standing alone at the
1873
+ // START of a block") — `p084-y1` finding 7. Type 7 is defined by being the one
1874
+ // HTML block that CANNOT interrupt a paragraph, so a bare tag continuing a
1875
+ // paragraph is not type 7 at all: `some prose` / `<my-widget>` / `---` is a
1876
+ // setext heading to CommonMark and this module already agrees with that. Firing
1877
+ // there reports a divergence that does not exist.
1878
+ if (i > 0 && info[i - 1] && info[i - 1].type === 'paragraph') continue;
1879
+ return i + 1;
1880
+ }
1881
+ return -1;
1882
+ }
1883
+
1884
+ // GAP C — a table whose delimiter row carries a different number of cells from
1885
+ // its header. GFM treats that pair as ordinary prose (it is not a table at all)
1886
+ // while `classifyLines` opens a table, so the two can disagree about where a
1887
+ // section ends, and `conventionsSectionBodies` then reads a different span than
1888
+ // the reader sees.
1889
+ //
1890
+ // ⚠⚠ THERE IS DELIBERATELY NO DETECTOR FOR THIS (user decision, 2026-08-12), and
1891
+ // the reason is worth the space because the obvious next move is to write one.
1892
+ // P-084 shipped one and then spent SIX review rounds on it. Every round found a
1893
+ // document it stayed silent on, and silence is the direction that prints
1894
+ // `All checks passed` over a file that has drifted:
1895
+ // `p084gate-x7` the bare mismatched pair
1896
+ // `p084gate-y3` a multi-line prose run before a dash underline
1897
+ // `p084gate-y4` the whole equals-underline family
1898
+ // `x13`/`y5` a prose run whose FIRST line is underline-shaped
1899
+ // `x14`/`y6` a prose run whose first line is INDENTED
1900
+ // `p084gate-y7` a delimiter row written with single hyphens, which GFM
1901
+ // accepts and `TABLE_DELIMITER_ROW_RE` does not
1902
+ // The first five were an enumeration inside the detector, and removing that
1903
+ // enumeration did not help: the sixth was the same failure living in the
1904
+ // delimiter-row recogniser instead. **An enumeration cannot be finished from the
1905
+ // inside**, and each unfinished edge fails silently.
1906
+ // ⚠ What settled it was a measurement rather than fatigue. `p084gate-y6`
1907
+ // reproduced the identical silent false clean with a delimiter row whose cell
1908
+ // count was CORRECT. So the harm is section-boundary disagreement, the detector's
1909
+ // scope was cell-count mismatch, and the second is a strict subset of the first —
1910
+ // no version of this detector, narrow or broad, could close the harm it was
1911
+ // written for.
1912
+ // ⚠⚠ The durable fix is a different instrument: `marked` is already a runtime
1913
+ // dependency of this package (`dflow render` uses it), so the section boundary
1914
+ // can be taken FROM the renderer rather than guessed alongside it and compared
1915
+ // shape by shape. That is a design change with its own review. It is recorded,
1916
+ // with owner and gate, as `doctor-section-boundary-arbiter` in
1917
+ // `planning/opt-in-backlog.md`, and both explainer pages name this gap under
1918
+ // "shapes that are known and deliberately not reported".
1919
+ // ⚠ `TABLE_DELIMITER_ROW_RE` stays — `classifyLines` and
1920
+ // `hasTableWithoutConventionComment` both use it. Its `-{2,}` is narrower than
1921
+ // GFM's `-{1,}` (`p084gate-y7`); that is now a classifier property with no
1922
+ // detector resting on it, and widening it would change what counts as a table
1923
+ // across this module.
1924
+ // Kept for callers that only need the first match.
1925
+ function conventionsSectionBody(content, heading) {
1926
+ const bodies = conventionsSectionBodies(content, heading);
1927
+ return bodies.length === 0 ? null : bodies[0];
1928
+ }
1929
+
1930
+ // Each `marker` is verified present in BOTH packaged templates (or, where the
1931
+ // section is edition-specific, in the edition that has it).
1932
+ // `test/upgrade-drift.mjs` re-derives that from the packaged templates rather
1933
+ // than trusting this comment.
1934
+ //
1935
+ // ⚠ BE PRECISE ABOUT WHICH HALF IS MECHANICAL. This sentence used to also claim
1936
+ // each marker is "absent from the pre-P082 shape", with the same
1937
+ // "re-derives that" clause covering both halves. Only the presence half is
1938
+ // re-derived; nothing in the tree checks the absence half (`p082-b3-g1`
1939
+ // finding 4). The risk it leaves is real and worth naming: a marker phrase that
1940
+ // ALREADY existed in the old text would make an un-migrated file report
1941
+ // `current` — silence about a stale file, the false-negative direction.
1942
+ // It is not mechanized because the only source for the old shape is git
1943
+ // history, and a test that shells out to git cannot run from the published
1944
+ // tarball, where there is no `.git`. So it stays a review obligation: when you
1945
+ // add a marker, check it against the pre-P082 text yourself.
1946
+ // `missingLevel` is per-fingerprint on purpose. `### SPEC-ID Format` was
1947
+ // re-parented by 59e0eb2 and no released version ever projected it, so EVERY
1948
+ // existing project is legitimately `missing` there — info. The other two
1949
+ // sections HAVE always shipped, so their absence means the file was edited or
1950
+ // predates them, which is worth a warn. A blanket info also mis-reported a
1951
+ // merely RENAMED heading as "projects created before this section shipped do
1952
+ // not have it at all", which is a false statement about a file that has it.
1953
+ const CONVENTIONS_FINGERPRINTS = [
1954
+ {
1955
+ id: 'ceremony-escalate-only',
1956
+ heading: 'Ceremony Scaling (Project Application)',
1957
+ marker: 'cascade result is a floor',
1958
+ editions: null,
1959
+ missingLevel: 'warn',
1960
+ rule: 'the escalate-only rule (project rows may raise a tier, never lower it)',
1961
+ consequence: 'Without it the table reads as a free re-classification surface, so a project row can lower a tier the cascade raised — the drift PROPOSAL-082 closed.'
1962
+ },
1963
+ {
1964
+ id: 'spec-files-no-br-families',
1965
+ heading: 'Filling the Templates',
1966
+ marker: 'no-BR family variants',
1967
+ editions: null,
1968
+ missingLevel: 'warn',
1969
+ rule: 'the no-BR family variants of `lightweight-spec.md`',
1970
+ consequence: 'Without them the table implies a T2 always carries a BR delta, so a presentation / contract / operational / performance change gets a fabricated BR-ID to satisfy a rule that no longer exists.'
1971
+ },
1972
+ {
1973
+ id: 'spec-id-minimal-host',
1974
+ heading: 'SPEC-ID Format',
1975
+ marker: 'Minimal (zero-phase) host exception',
1976
+ editions: null,
1977
+ missingLevel: 'info',
1978
+ neverProjected: true,
1979
+ rule: 'the minimal (zero-phase) host exception',
1980
+ consequence: 'Without it the section still says the SPEC-ID appears in the first phase-spec filename and the branch name, which a zero-phase host has no way to satisfy — and a functional-bug host carries a BUG-NUMBER branch instead.'
1981
+ }
1982
+ ];
1983
+
1984
+ // The second fingerprint kind: strings PROPOSAL-082 retired, which must not
1985
+ // still be sitting in an adopter's file. These need no new upstream contract —
1986
+ // the evidence is that batch 1 deleted them from both packaged templates — so
1987
+ // unlike the gf DDD-Modeling-Depth fingerprint (which has no marker to pin and
1988
+ // is recorded as debt 10), there is nothing blocking them.
1989
+ //
1990
+ // Every string below is verified zero-match in both packaged templates AND both
1991
+ // tutorial fixtures; `test/upgrade-drift.mjs` re-derives that rather than
1992
+ // trusting this comment. A retired string that still matched the current
1993
+ // template would fire on every project forever.
1994
+ const CONVENTIONS_RETIRED = [
1995
+ {
1996
+ id: 'ceremony-ef-tweak-t3',
1997
+ heading: 'Ceremony Scaling (Project Application)',
1998
+ retired: 'T3 if no Domain change',
1999
+ editions: null,
2000
+ rule: 'the retired "EF configuration tweak → T3 if no Domain change" example row',
2001
+ consequence: 'Under the ordered cascade an Infrastructure mapping tweak is not automatically T3 — the packaged template now uses that row to show a project ESCALATING it, and the old row teaches the opposite.'
2002
+ },
2003
+ {
2004
+ id: 'ceremony-query-only-t2',
2005
+ heading: 'Ceremony Scaling (Project Application)',
2006
+ // Anchored through the TIER cell, not just the situation. The situation
2007
+ // alone flagged `| Adding a Query only (no write) | T1 (project convention) |
2008
+ // We escalate all new reads |` — a row an adopter migrated CORRECTLY into
2009
+ // the form the current template teaches — and then told them to overwrite it.
2010
+ // The claim is about a row that restates a bare tier, so the anchor has to
2011
+ // include the bare tier. Sibling `ceremony-ef-tweak-t3` never had this
2012
+ // problem because its anchor was already the tier cell.
2013
+ retired: 'Adding a Query only (no write)} | T2',
2014
+ editions: null,
2015
+ rule: 'the retired "Adding a Query only (no write) → T2" example row',
2016
+ consequence: 'The cascade decides this now; the row restated a tier instead of recording a project decision, which is the drift the escalate-only rule was added to stop.'
2017
+ },
2018
+ {
2019
+ id: 'ceremony-criteria-table',
2020
+ heading: 'Ceremony Scaling (Project Application)',
2021
+ // Anchored on enough of the retired sentence to be implausible as adopter
2022
+ // prose. Two shorter anchors were tried and both trip on legitimate text:
2023
+ // `criteria table` matches "Our own criteria table below records the
2024
+ // escalations we apply", and `for the full criteria table` still matches
2025
+ // "see that document for the full criteria table" — a generic five-word
2026
+ // English construction. The retired text was "See `AI-AGENT-GUIDE.md`
2027
+ // § Ceremony Scaling for the full criteria table.", soft-wrapped between
2028
+ // "full" and "criteria"; `normalizeForMarker` collapses that, so including
2029
+ // the § name costs nothing and no longer matches ordinary sentences.
2030
+ retired: 'Ceremony Scaling for the full criteria table',
2031
+ editions: null,
2032
+ rule: 'the retired pointer to a "criteria table" in AI-AGENT-GUIDE.md',
2033
+ consequence: 'That table was replaced by the ordered cascade. A file still pointing at it sends the reader to a section that no longer exists under that name.'
2034
+ },
2035
+ {
2036
+ id: 'spec-files-br-delta-equation',
2037
+ heading: 'Filling the Templates',
2038
+ retired: 'bug fix / small tweak with BR Delta',
2039
+ editions: null,
2040
+ rule: 'the retired "T2 Light — bug fix / small tweak with BR Delta" row',
2041
+ consequence: 'It equates T2 with carrying a BR Delta, which the no-BR families retired — the same equation that made people fabricate BR-IDs.'
2042
+ }
2043
+ ];
2044
+
2045
+ // Markers are matched against whitespace-normalized text. `_conventions.md` is
2046
+ // user-owned and reflowing a paragraph is an ordinary edit; matching raw made a
2047
+ // soft-wrap inside the marker report a false `stale`.
2048
+ function normalizeForMarker(text) {
2049
+ return String(text).replace(/\s+/g, ' ');
2050
+ }
2051
+
2052
+ // ⚠ INHERENT LIMIT, stated rather than papered over: these are heuristic
2053
+ // substring checks (PROPOSAL-082 specifies them as such — 字串偵測, following
2054
+ // the P058 starter-drift precedent). A substring cannot tell a rule from a
2055
+ // sentence ABOUT the rule, so text like "we removed the old rule that the
2056
+ // cascade result is a floor" satisfies the fingerprint and reports `current`.
2057
+ // That is a false negative on a deliberately-worded file, not on drift, and no
2058
+ // amount of matching harder fixes it — deciding it needs to read meaning.
2059
+ // Do not describe this check as proving a project's conventions are correct;
2060
+ // it detects the shapes that arise from a file predating a contract change.
2061
+ //
2062
+ // ⚠ The `retired` kind fails in the OPPOSITE direction, and the two must not be
2063
+ // conflated. A `marker` that is too short yields a false NEGATIVE (silence); a
2064
+ // `retired` string that is too short yields a false POSITIVE — doctor telling a
2065
+ // developer their own sentence is retired Dflow text. So a retired entry must
2066
+ // be anchored on enough of the original sentence to be implausible as adopter
2067
+ // prose, and `test/upgrade-drift.mjs` asserts each is zero-match against both
2068
+ // packaged templates and both tutorial fixtures.
2069
+ //
2070
+ // Returns one entry per fingerprint that is not `current`. `edition` selects
2071
+ // edition-specific fingerprints; pass null to evaluate only the shared ones.
2072
+ // Whether an entry applies to the edition being checked. `editions: null` means
2073
+ // "shared, applies everywhere"; a list means the entry is only meaningful for
2074
+ // those editions and a caller that did not resolve an edition must not see it.
2075
+ //
2076
+ // Extracted from the two identical inline conditions below because every shipped
2077
+ // entry currently has `editions: null`, which makes the non-null path dead and
2078
+ // therefore unproven (`p082-b3-g1` finding 5). It is NOT speculative — debt 10
2079
+ // (the greenfield DDD-Modeling-Depth fingerprint) is a known future
2080
+ // edition-specific entry, blocked only on an upstream marker to pin. So the
2081
+ // branch is kept and tested here instead of being deleted and rewritten
2082
+ // untested at the moment someone is busy thinking about something else.
2083
+ function fingerprintAppliesTo(fp, edition) {
2084
+ if (!fp.editions) return true;
2085
+ return Boolean(edition) && fp.editions.includes(edition);
2086
+ }
2087
+
2088
+ function findConventionsDrift(content, edition = null) {
2089
+ // No early return for empty input. It used to short-circuit to `[]`, which
2090
+ // made a whitespace-only file report NOTHING while a one-character file
2091
+ // reported all three sections missing — the emptier file getting the cleaner
2092
+ // verdict, and the gate silently accepting absence, which is the exact thing
2093
+ // this module says it exists to prevent. Empty content genuinely has every
2094
+ // section missing, so it says so; the caller collapses that into one finding.
2095
+ const text = String(content || '');
2096
+ const drift = [];
2097
+ for (const fp of CONVENTIONS_FINGERPRINTS) {
2098
+ if (!fingerprintAppliesTo(fp, edition)) continue;
2099
+ const bodies = conventionsSectionBodies(text, fp.heading);
2100
+ const base = {
2101
+ id: fp.id,
2102
+ heading: fp.heading,
2103
+ rule: fp.rule,
2104
+ consequence: fp.consequence,
2105
+ neverProjected: Boolean(fp.neverProjected)
2106
+ };
2107
+ if (bodies.length === 0) {
2108
+ drift.push({ ...base, state: 'missing', level: fp.missingLevel || 'warn' });
2109
+ continue;
2110
+ }
2111
+ const marker = normalizeForMarker(fp.marker);
2112
+ // Any copy of the section carrying the rule is enough.
2113
+ if (!bodies.some((body) => normalizeForMarker(body).includes(marker))) {
2114
+ drift.push({ ...base, state: 'stale', level: 'warn' });
2115
+ }
2116
+ }
2117
+ for (const fp of CONVENTIONS_RETIRED) {
2118
+ if (!fingerprintAppliesTo(fp, edition)) continue;
2119
+ const bodies = conventionsSectionBodies(text, fp.heading);
2120
+ // A section that is absent cannot carry a retired string. Its absence is
2121
+ // already reported by the `requires` fingerprint on the same heading, so
2122
+ // saying nothing here is not silence about an unchecked state.
2123
+ if (bodies.length === 0) continue;
2124
+ const retired = normalizeForMarker(fp.retired);
2125
+ if (bodies.some((body) => normalizeForMarker(body).includes(retired))) {
2126
+ drift.push({
2127
+ id: fp.id,
2128
+ state: 'retired',
2129
+ level: 'warn',
2130
+ heading: fp.heading,
2131
+ rule: fp.rule,
2132
+ consequence: fp.consequence,
2133
+ neverProjected: false
2134
+ });
2135
+ }
2136
+ }
2137
+ return drift;
2138
+ }
2139
+
2140
+ // --- PROPOSAL-092: shape markers and template skeletons ------------------------
2141
+ //
2142
+ // Every template an adopter's document is created from carries one line naming
2143
+ // its track, its template and the number of the shape it has:
2144
+ //
2145
+ // <!-- dflow-shape: greenfield/rules.md 1 — keep this line: dflow doctor reads it -->
2146
+ //
2147
+ // The document inherits the line when it is created from the template, so
2148
+ // doctor can compare the number with the template's current one. Same number:
2149
+ // every difference between the document and the template is the adopter's own
2150
+ // decision, and nothing is reported — which is the distinction a comparison
2151
+ // alone cannot make (six OBTS `rules.md` files written one consistent way, not
2152
+ // the template's). Older number: the registry (`lib/doc-shapes.json`) says what
2153
+ // changed between the two numbers.
2154
+ //
2155
+ // ⚠ Only the three fields are read. Whatever follows the number up to `-->` is
2156
+ // the note for the reader and is not validated: an AI that re-words the note has
2157
+ // not changed what the line records.
2158
+ const crypto = require('node:crypto');
2159
+
2160
+ const SHAPE_MARKER_MENTION = 'dflow-shape:';
2161
+ // The note after the number stops at a line break of either kind: in a file
2162
+ // with lone-CR line breaks it would otherwise swallow the next line — another
2163
+ // marker included — and pass the doc as read.
2164
+ const SHAPE_MARKER_RE = /^<!-- dflow-shape: ([a-z]+)\/([A-Za-z0-9_.-]+\.md) ([1-9][0-9]*)(?: [^\r\n]*)? -->[ \t]*$/;
2165
+
2166
+ // Stage one of two: every line that mentions `dflow-shape:`, wherever it sits —
2167
+ // in a fence, a comment, a list, a quote, an HTML block.
2168
+ // ⚠ Doctor does not decide which of those lines is live (PROPOSAL-092 D2).
2169
+ // Deciding means reading every container exactly the way `dflow render`
2170
+ // (`marked`) reads it, and the two readings are hard to keep aligned. A mention
2171
+ // anywhere but the standard position makes the doc unreadable instead: saying
2172
+ // "cannot tell" never reports a doc doctor did not read as passing.
2173
+ // ⚠ Finding and validating stay separate: validating while finding would let one
2174
+ // valid line plus one broken line count as "exactly one marker", and would let a
2175
+ // doc whose only marker is broken count as unmarked.
2176
+ function findShapeMarkerLines(content) {
2177
+ const found = [];
2178
+ String(content).split('\n').forEach((line, i) => {
2179
+ if (line.includes(SHAPE_MARKER_MENTION)) found.push({ line: i + 1, text: line.replace(/\r$/, '') });
2180
+ });
2181
+ return found;
2182
+ }
2183
+
2184
+ // The one line a marker is read from: line 1, or the line after the
2185
+ // frontmatter's closing `---` — the frontmatter rule `dflow render` uses, which
2186
+ // is also where every template puts its marker. 1-based.
2187
+ // ⚠ Found in the text as it is, a byte-order mark included: render does not
2188
+ // skip one, so a BOM in front of `---` leaves the file with no frontmatter — for
2189
+ // render, and so for doctor. Skipping it here alone would read a marker that
2190
+ // render shows as body text.
2191
+ function shapeMarkerStandardLine(content) {
2192
+ return frontmatterLineCount(String(content).split('\n')) + 1;
2193
+ }
2194
+
2195
+ // Stage two. `absent`: no line mentions it. `ok`: exactly one line mentions it,
2196
+ // that line is the standard one, and it parses. Anything else is `unreadable`,
2197
+ // with the reason — `multiple`, `misplaced` (`standard` is the line it belongs
2198
+ // on; `bomHidesFrontmatter` when a BOM is what hides the frontmatter above it)
2199
+ // or `malformed`. Whether the track and template name exist is the caller's
2200
+ // question: it needs the registry, and this module does no I/O.
2201
+ function readShapeMarker(content) {
2202
+ const raw = String(content);
2203
+ // A BOM is not a line: line numbers and the marker's own text are read without it.
2204
+ const text = raw.replace(/^\uFEFF/, '');
2205
+ const found = findShapeMarkerLines(text);
2206
+ if (found.length === 0) return { state: 'absent' };
2207
+ if (found.length > 1) {
2208
+ return { state: 'unreadable', reason: 'multiple', lines: found.map((f) => f.line) };
2209
+ }
2210
+ const standard = shapeMarkerStandardLine(raw);
2211
+ if (found[0].line !== standard) {
2212
+ const misplaced = { state: 'unreadable', reason: 'misplaced', lines: [found[0].line], standard };
2213
+ if (text !== raw && frontmatterLineCount(text.split('\n')) > 0) misplaced.bomHidesFrontmatter = true;
2214
+ return misplaced;
2215
+ }
2216
+ const m = found[0].text.match(SHAPE_MARKER_RE);
2217
+ const number = m ? Number(m[3]) : NaN;
2218
+ if (!m || !Number.isSafeInteger(number)) {
2219
+ return { state: 'unreadable', reason: 'malformed', lines: [found[0].line] };
2220
+ }
2221
+ return { state: 'ok', track: m[1], template: m[2], number, line: found[0].line };
2222
+ }
2223
+
2224
+ // The number of lines the frontmatter occupies, fences included; 0 when there
2225
+ // is none. The SAME rule as `splitFrontmatter` in `lib/render.js`, which the
2226
+ // marker placement was measured against: the document's first line starts with
2227
+ // `---`, and the frontmatter ends at the next line that is `---` once trimmed;
2228
+ // with no such line there is no frontmatter. `test/doc-shapes.mjs` pins the two
2229
+ // against each other on every packaged template and on the line-ending cases.
2230
+ // ⚠ Callers pass the text split on `\n` ONLY — render's split, not
2231
+ // `LINE_SPLIT_RE`. With the module's own splitter a lone-CR file had
2232
+ // frontmatter here and none in render, so "the same rule" was two rules.
2233
+ // ⚠ Nothing else in this module knows about frontmatter, and `classifyLines`
2234
+ // reads it as ordinary blocks: a commented field like `# follow-up-of:` is an
2235
+ // H1 to it and the closing `---` underlines an H2. So the skeleton cuts it off
2236
+ // before classifying anything.
2237
+ function frontmatterLineCount(lines) {
2238
+ if (lines.length === 0 || !lines[0].startsWith('---')) return 0;
2239
+ for (let i = 1; i < lines.length; i += 1) {
2240
+ if (lines[i].trim() === '---') return i + 1;
2241
+ }
2242
+ return 0;
2243
+ }
2244
+
2245
+ // A field the template carries commented out: `# name: …`, indented or not,
2246
+ // with the name spelled as a key would be (quoted, dotted), as long as it is
2247
+ // one token — `# a note: …` in prose is not a field.
2248
+ const FRONTMATTER_COMMENTED_FIELD_RE = /^[ \t]*#[ \t]*("[^"\n]*"|'[^'\n]*'|[A-Za-z][A-Za-z0-9_.-]*)[ \t]*:/;
2249
+ // A `{…}` placeholder anywhere in a heading makes it an example, not a section
2250
+ // the template requires (`## {Feature Area 1}`, `### FL-01: {流程名稱}`).
2251
+ const SHAPE_PLACEHOLDER_RE = /\{[^}\n]*\}/;
2252
+
2253
+ function shapeHeadingText(text) {
2254
+ return String(text)
2255
+ .replace(/<!--[\s\S]*?-->/g, '')
2256
+ .replace(/<!--.*$/, '')
2257
+ .replace(/[ \t]+/g, ' ')
2258
+ .trim();
2259
+ }
2260
+
2261
+ // The frontmatter's field lines and the body after it, cut the way `dflow
2262
+ // render` cuts them (see `frontmatterLineCount`).
2263
+ function splitShapeFrontmatter(content) {
2264
+ const text = String(content);
2265
+ const lines = text.split('\n');
2266
+ const count = frontmatterLineCount(lines);
2267
+ return {
2268
+ fields: count === 0 ? [] : lines.slice(1, count - 1).map((l) => l.replace(/\r$/, '')),
2269
+ body: count === 0 ? text : lines.slice(count).join('\n')
2270
+ };
2271
+ }
2272
+
2273
+ // A frontmatter field is whatever `splitFrontmatter` in `lib/render.js` reads
2274
+ // as a key — the text before the line's first `:`, trimmed, on a line that has
2275
+ // one and does not start with `#` — so a field render shows on the meta card
2276
+ // (`"review-owner":`, `review.owner:`) is never missing from the skeleton.
2277
+ // `test/doc-shapes.mjs` pins the two against each other. Null for any other line.
2278
+ function frontmatterFieldName(line) {
2279
+ if (line.trimStart().startsWith('#')) return null;
2280
+ const colon = line.indexOf(':');
2281
+ return colon === -1 ? null : line.slice(0, colon).trim();
2282
+ }
2283
+
2284
+ function tableHeaderCells(line) {
2285
+ // Comments go before the split: a `|` inside one is not a column boundary.
2286
+ let row = line.replace(/<!--[\s\S]*?-->/g, '').trim();
2287
+ if (row.startsWith('|')) row = row.slice(1);
2288
+ if (row.endsWith('|') && !row.endsWith('\\|')) row = row.slice(0, -1);
2289
+ return row.split(/(?<!\\)\|/).map((cell) => shapeHeadingText(cell));
2290
+ }
2291
+
2292
+ // The shape of a template: what a shape number stands for (PROPOSAL-092 D6).
2293
+ // Items are small arrays so they compare, sort and print without a parser:
2294
+ //
2295
+ // ['field', name] frontmatter field
2296
+ // ['field', name, 'optional'] field the template carries commented out
2297
+ // ['h2', heading]
2298
+ // ['h3', h2, heading] h2 is '' when the H3 sits above any H2
2299
+ // ['column', h2, h3, table, position, heading]
2300
+ // a table's header cell; `table` counts
2301
+ // from 1 within that (h2, h3) section, and
2302
+ // `position` from 1 within the row
2303
+ // ['note', h2, h3, digest] the `>` notes of that section
2304
+ // ['comment', h2, h3, digest] its HTML comments: every one that starts
2305
+ // a line's content — at the top level or in
2306
+ // a list item, multi-line ones included —
2307
+ // and the ones in its heading line
2308
+ //
2309
+ // The first four are the skeleton; notes, comments and the heading order are
2310
+ // the shape's notes-and-order half — a change there does not change the doc's
2311
+ // structure, and doctor reports it as such. A note or comment digest is taken
2312
+ // over its text with `>` and line breaks folded to single spaces, so re-wrapping
2313
+ // is not a change. A comment counts as its `<!-- … -->` span only: prose after
2314
+ // `-->` on the same line is prose. A comment in a `>` block is part of that
2315
+ // note. The marker line is not a comment of the shape.
2316
+ //
2317
+ // ⚠ The column POSITION is part of the item: a table's columns are read by
2318
+ // position, so swapping `Upstream` and `Downstream` reverses what every row
2319
+ // means while leaving the set of header names the same. Items compare
2320
+ // order-free; the heading order is compared on its own (`shapeOrderChanges`).
2321
+ //
2322
+ // Left out: H1 and H4 and below, prose outside notes and comments, fixed label
2323
+ // text, table rows, and every `##` / `###` heading with a placeholder —
2324
+ // together with everything under it, down to the next heading at its level or
2325
+ // above. A table or comment under an H4 is recorded under the nearest H2 / H3,
2326
+ // placeholder H4 or not: an H4 is not part of the skeleton, so it cannot take
2327
+ // what follows it out of the shape either.
2328
+ // `variable` lists the sections a template lets the author replace (`[h2]` or
2329
+ // `[h2, h3]`); those and everything under them are left out as well.
2330
+ // ⚠ `test/doc-shapes.mjs` holds this reading against `dflow render`'s parser on
2331
+ // every registered template: the headings, tables, notes and comments render
2332
+ // sees must be the ones recorded here. A `>` block inside a list item is one
2333
+ // render sees and this does not read, so a template may not carry one.
2334
+ function extractShapeSkeleton(content, options) {
2335
+ const { items, sections } = extractShapeSections(content, options);
2336
+ for (const t of sections) {
2337
+ if (t.note.length > 0) items.push(['note', t.h2, t.h3, shapeTextDigest(t.note)]);
2338
+ if (t.comment.length > 0) items.push(['comment', t.h2, t.h3, shapeTextDigest(t.comment)]);
2339
+ }
2340
+ return items;
2341
+ }
2342
+
2343
+ // The `<!-- … -->` spans in `text` that a shape records — never the prose around
2344
+ // them. An unterminated one runs to the end of the text. The marker is not one.
2345
+ function shapeCommentSpans(text) {
2346
+ return (String(text).match(/<!--[\s\S]*?(?:-->|$)/g) || [])
2347
+ .filter((span) => !span.includes(SHAPE_MARKER_MENTION));
2348
+ }
2349
+
2350
+ // The skeleton items, and per section the raw note and comment texts its
2351
+ // digests are taken over (`extractShapeSkeleton` adds the digests).
2352
+ function extractShapeSections(content, { variable = [] } = {}) {
2353
+ const { fields, body: bodyText } = splitShapeFrontmatter(content);
2354
+ const items = [];
2355
+ for (const line of fields) {
2356
+ const name = frontmatterFieldName(line);
2357
+ if (name !== null) { items.push(['field', name]); continue; }
2358
+ const m = line.match(FRONTMATTER_COMMENTED_FIELD_RE);
2359
+ if (m) items.push(['field', m[1], 'optional']);
2360
+ }
2361
+ const variableKeys = new Set(variable.map((v) => JSON.stringify(v)));
2362
+ const body = blankFencedBlocks(bodyText);
2363
+ const classes = classifyLines(body);
2364
+ let h2 = '';
2365
+ let h3 = '';
2366
+ let placeholderLevel = 0;
2367
+ let skipped = false;
2368
+ const tables = new Map();
2369
+ // Per section, in the order sections first appear: its notes and comments.
2370
+ const texts = new Map();
2371
+ const textsHere = () => {
2372
+ const key = JSON.stringify([h2, h3]);
2373
+ if (!texts.has(key)) texts.set(key, { h2, h3, note: [], comment: [] });
2374
+ return texts.get(key);
2375
+ };
2376
+ const commentStillOpen = (t) => t.lastIndexOf('<!--') > t.lastIndexOf('-->');
2377
+ for (let i = 0; i < body.length; i += 1) {
2378
+ const c = classes[i];
2379
+ if (c.heading) {
2380
+ const { level } = c.heading;
2381
+ const text = shapeHeadingText(c.heading.text);
2382
+ if (placeholderLevel && level <= placeholderLevel) placeholderLevel = 0;
2383
+ if (placeholderLevel) continue;
2384
+ if ((level === 2 || level === 3) && SHAPE_PLACEHOLDER_RE.test(text)) {
2385
+ placeholderLevel = level;
2386
+ continue;
2387
+ }
2388
+ if (level === 1) {
2389
+ h2 = '';
2390
+ h3 = '';
2391
+ skipped = false;
2392
+ } else if (level === 2) {
2393
+ h2 = text;
2394
+ h3 = '';
2395
+ skipped = variableKeys.has(JSON.stringify([h2]));
2396
+ if (!skipped) items.push(['h2', h2]);
2397
+ } else if (level === 3) {
2398
+ h3 = text;
2399
+ skipped = variableKeys.has(JSON.stringify([h2])) || variableKeys.has(JSON.stringify([h2, h3]));
2400
+ if (!skipped) items.push(['h3', h2, h3]);
2401
+ }
2402
+ if (!skipped) textsHere().comment.push(...shapeCommentSpans(c.heading.text));
2403
+ continue;
2404
+ }
2405
+ if (placeholderLevel || skipped) continue;
2406
+ if (c.type === 'blockquote') {
2407
+ // The line is a blockquote by `classifyLines`; its `>` markers go — all of
2408
+ // them, so re-wrapping a nested `> >` note is not a change either.
2409
+ const part = body[i].trimStart().replace(/^(?:>[ \t]?)+/, '');
2410
+ const notes = textsHere().note;
2411
+ if (i > 0 && classes[i - 1].type === 'blockquote' && notes.length > 0) notes[notes.length - 1] += `\n${part}`;
2412
+ else notes.push(part);
2413
+ continue;
2414
+ }
2415
+ // An HTML block: the comments in it, whatever else it holds.
2416
+ if (c.type === 'html' && c.blockStart === i) {
2417
+ let end = i;
2418
+ while (end + 1 < body.length && classes[end + 1].type === 'html' && classes[end + 1].blockStart === i) end += 1;
2419
+ textsHere().comment.push(...shapeCommentSpans(body.slice(i, end + 1).join('\n')));
2420
+ i = end;
2421
+ continue;
2422
+ }
2423
+ // A comment that starts a list line's content (` <!-- … -->` under an item,
2424
+ // or `- <!-- … -->`): a block of its own inside the item, as render reads it.
2425
+ // It runs on until it closes, across blank lines, as a comment does.
2426
+ if (c.type === 'list') {
2427
+ const at = body[i].indexOf('<!--');
2428
+ if (at >= 0 && startsContent(body[i], at) && !body[i].slice(0, at).includes('>')) {
2429
+ let end = i;
2430
+ let chunk = body[i].slice(at);
2431
+ while (commentStillOpen(chunk) && end + 1 < body.length && (classes[end + 1].type === 'list' || classes[end + 1].type === 'blank')) {
2432
+ end += 1;
2433
+ chunk += `\n${body[end]}`;
2434
+ }
2435
+ textsHere().comment.push(...shapeCommentSpans(chunk));
2436
+ i = end;
2437
+ continue;
2438
+ }
2439
+ }
2440
+ const startsTable = c.type === 'table' && (i === 0 || classes[i - 1].type !== 'table');
2441
+ if (!startsTable) continue;
2442
+ const section = JSON.stringify([h2, h3]);
2443
+ const table = (tables.get(section) || 0) + 1;
2444
+ tables.set(section, table);
2445
+ tableHeaderCells(body[i]).forEach((cell, k) => items.push(['column', h2, h3, table, k + 1, cell]));
2446
+ }
2447
+ return { items, sections: [...texts.values()] };
2448
+ }
2449
+
2450
+ // A section's notes or comments, folded so that re-wrapping changes nothing:
2451
+ // every run of whitespace is one space. The order of the parts is kept.
2452
+ function shapeTextDigest(parts) {
2453
+ const folded = parts.map((p) => String(p).replace(/\s+/g, ' ').trim()).join('\n');
2454
+ return crypto.createHash('sha256').update(folded, 'utf8').digest('hex').slice(0, 16);
2455
+ }
2456
+
2457
+ // The heading order a shape records: the H2s, and the H3s under each H2 ('' for
2458
+ // H3s above the first H2). The `null` key holds the H2 list itself.
2459
+ function shapeHeadingOrder(skeleton) {
2460
+ const order = new Map([[null, []]]);
2461
+ for (const item of skeleton) {
2462
+ if (item[0] === 'h2') order.get(null).push(item[1]);
2463
+ else if (item[0] === 'h3') {
2464
+ if (!order.has(item[1])) order.set(item[1], []);
2465
+ order.get(item[1]).push(item[2]);
2466
+ }
2467
+ }
2468
+ return order;
2469
+ }
2470
+
2471
+ // Where the headings both shapes have stand in a different order: `null` for
2472
+ // the H2s, an H2's name for the H3s under it. A heading only one shape has is
2473
+ // left out — adding or removing a section is a change of its own, not a
2474
+ // reordering.
2475
+ function shapeOrderChanges(from, to) {
2476
+ const before = shapeHeadingOrder(from);
2477
+ const after = shapeHeadingOrder(to);
2478
+ const shared = (list, other) => [...new Set(list.filter((h) => other.includes(h)))];
2479
+ const changed = [];
2480
+ for (const [parent, list] of before) {
2481
+ const other = after.get(parent);
2482
+ if (!other) continue;
2483
+ if (JSON.stringify(shared(list, other)) !== JSON.stringify(shared(other, list))) changed.push(parent);
2484
+ }
2485
+ return changed;
2486
+ }
2487
+
2488
+ function describeShapeOrder(parent) {
2489
+ if (parent === null) return 'the order of the `##` sections';
2490
+ return parent ? `the order of the \`###\` sections under \`## ${parent}\`` : 'the order of the `###` sections above the first `##`';
2491
+ }
2492
+
2493
+ // Multiset difference: what `to` has that `from` does not, and the reverse.
2494
+ function skeletonDifference(from, to) {
2495
+ const left = new Map();
2496
+ for (const item of from) {
2497
+ const key = JSON.stringify(item);
2498
+ left.set(key, (left.get(key) || 0) + 1);
2499
+ }
2500
+ const added = [];
2501
+ for (const item of to) {
2502
+ const key = JSON.stringify(item);
2503
+ const n = left.get(key) || 0;
2504
+ if (n > 0) left.set(key, n - 1);
2505
+ else added.push(item);
2506
+ }
2507
+ const removed = [];
2508
+ for (const [key, n] of left) {
2509
+ for (let k = 0; k < n; k += 1) removed.push(JSON.parse(key));
2510
+ }
2511
+ return { added, removed };
2512
+ }
2513
+
2514
+ // The items order-free, plus the heading order: a section moved without being
2515
+ // renamed is a new shape too (its notes-and-order half). `test/doc-shapes.mjs`
2516
+ // recomputes this for every registered number, which is what makes a rewrite of
2517
+ // a published number visible.
2518
+ function shapeDigest(skeleton) {
2519
+ const canonical = skeleton.map((item) => JSON.stringify(item)).sort().join('\n');
2520
+ const order = JSON.stringify([...shapeHeadingOrder(skeleton)]);
2521
+ return crypto.createHash('sha256').update(`${canonical}\n${order}`, 'utf8').digest('hex');
2522
+ }
2523
+
2524
+ // Notes, comments and heading order: the part of a shape that does not change
2525
+ // a doc's structure (PROPOSAL-092 D4's third class).
2526
+ function isShapeTextItem(item) {
2527
+ return Array.isArray(item) && (item[0] === 'note' || item[0] === 'comment');
2528
+ }
2529
+
2530
+ function describeShapeItem(item) {
2531
+ const [kind] = item;
2532
+ if (kind === 'field') return `frontmatter field \`${item[1]}:\`${item[2] === 'optional' ? ' (optional)' : ''}`;
2533
+ if (kind === 'h2') return `\`## ${item[1]}\``;
2534
+ if (kind === 'h3') return item[1] ? `\`### ${item[2]}\` under \`## ${item[1]}\`` : `\`### ${item[2]}\``;
2535
+ if (kind === 'column') {
2536
+ const where = [item[1] && `\`## ${item[1]}\``, item[2] && `\`### ${item[2]}\``].filter(Boolean).join(' › ') || 'the top of the document';
2537
+ return `column \`${item[5]}\` (position ${item[4]}) in table ${item[3]} under ${where}`;
2538
+ }
2539
+ if (isShapeTextItem(item)) {
2540
+ const where = [item[1] && `\`## ${item[1]}\``, item[2] && `\`### ${item[2]}\``].filter(Boolean).join(' › ');
2541
+ const what = kind === 'note' ? 'the `>` notes' : 'the HTML comments';
2542
+ return where ? `${what} under ${where}` : `${what} at the top of the document`;
2543
+ }
2544
+ return JSON.stringify(item);
2545
+ }
2546
+
2547
+ // `pattern` is relative to `dflow/specs/`; `*` matches within one path segment.
2548
+ function shapePathMatches(pattern, relPath) {
2549
+ const re = new RegExp(`^${pattern.split('*').map(escapeRegExp).join('[^/]*')}$`);
2550
+ return re.test(relPath);
2551
+ }
2552
+
2553
+ // The data rows of the first table under the H2 named `heading`, or null when
2554
+ // there is no such section or it holds no table. Used to recognise a zero-phase
2555
+ // host from its `_index.md`: an empty `## Phase Specs` table is what the flows
2556
+ // themselves test (`finish-feature-flow.md` Step 1 certifies a host that carries
2557
+ // a phase as phase-bearing).
2558
+ function sectionTableRowCount(content, heading) {
2559
+ const body = blankFencedBlocks(splitShapeFrontmatter(content).body);
2560
+ const classes = classifyLines(body);
2561
+ const want = headingKey(heading);
2562
+ let inside = false;
2563
+ for (let i = 0; i < body.length; i += 1) {
2564
+ const c = classes[i];
2565
+ if (c.heading && c.heading.level <= 2) {
2566
+ if (inside) return null;
2567
+ inside = c.heading.level === 2 && headingKey(c.heading.text) === want;
2568
+ continue;
2569
+ }
2570
+ if (!inside || c.type !== 'table') continue;
2571
+ let end = i;
2572
+ while (end + 1 < body.length && classes[end + 1].type === 'table') end += 1;
2573
+ return Math.max(0, end - i - 1);
2574
+ }
2575
+ return null;
2576
+ }
2577
+
2578
+ module.exports = {
2579
+ GIT_POLICY_LINE_RE,
2580
+ AI_COMMIT_MARKER_LINE_RE,
2581
+ PROSE_LANGUAGE_LINE_RE,
2582
+ TECH_STACK_ROW_RE,
2583
+ MIGRATION_CONTEXT_ROW_RE,
2584
+ GIT_POLICY_VALUES,
2585
+ AI_COMMIT_MARKER_VALUES,
2586
+ parseContextLine,
2587
+ blankFencedBlocks,
2588
+ extractHeadings,
2589
+ extractH2Headings,
2590
+ extractSectionRefs,
2591
+ headingResolves,
2592
+ missingTemplateSections,
2593
+ matchesTemplateWithPlaceholders,
2594
+ SPEC_FORMATTING_CONVENTION_SNIPPET,
2595
+ hasTableWithoutConventionComment,
2596
+ conventionsSectionBody,
2597
+ conventionsSectionBodies,
2598
+ CONVENTIONS_FINGERPRINTS,
2599
+ CONVENTIONS_RETIRED,
2600
+ // Exported so the edition filter can be tested directly. Every shipped entry
2601
+ // has `editions: null`, so `findConventionsDrift` alone can never exercise the
2602
+ // non-null path — a test written through it would prove nothing.
2603
+ fingerprintAppliesTo,
2604
+ isThematicBreak,
2605
+ // The single block-structure decision point (debt 12). Exported so
2606
+ // `lib/init.js` can locate headings through it instead of hand-rolling a
2607
+ // fourth ATX expression, and so `test/upgrade-drift.mjs` can assert labels
2608
+ // directly rather than only through what a section body happens to contain.
2609
+ parseAtxHeading,
2610
+ classifyLines,
2611
+ visibleTextLines,
2612
+ maskCodeBlocks,
2613
+ unclosedHtmlBlockLine,
2614
+ // Exported so `lib/init.js` reads the rule instead of writing a third copy of
2615
+ // `/^[ \t]*$/` inline. The commit that introduced this constant claimed it and
2616
+ // `stripSpaceTab` were "the only two spellings now"; they were not, because
2617
+ // neither was reachable from the other file.
2618
+ BLANK_LINE_RE,
2619
+ stripSpaceTab,
2620
+ findConventionsDrift,
2621
+ // PROPOSAL-084 uncertainty detectors. Each returns a 1-based line number or -1
2622
+ // and asks only whether a SHAPE is present, never where a block ends — see the
2623
+ // section header above them for why that is what makes them safe to run on top
2624
+ // of a parser with named gaps. Exported individually (rather than behind one
2625
+ // aggregate) so `test/upgrade-drift.mjs` can pin each shape and each
2626
+ // false-positive control against the detector that owns it: an aggregate that
2627
+ // returns "something fired" cannot fail for the right reason.
2628
+ inlineHtmlCommentLine,
2629
+ containerHtmlCommentLine,
2630
+ htmlBlockType7Line,
2631
+ // Exported for the false-positive controls specifically: the code-span rule is
2632
+ // the boundary that keeps `` `<!-- x -->` `` — which a reader DOES see — out of
2633
+ // the inline-comment detector, and it deserves a direct pin rather than being
2634
+ // tested only through its two consumers.
2635
+ codeSpanMask,
2636
+ // PROPOSAL-092.
2637
+ findShapeMarkerLines,
2638
+ readShapeMarker,
2639
+ shapeMarkerStandardLine,
2640
+ frontmatterLineCount,
2641
+ extractShapeSkeleton,
2642
+ extractShapeSections,
2643
+ shapeCommentSpans,
2644
+ shapeHeadingText,
2645
+ skeletonDifference,
2646
+ shapeHeadingOrder,
2647
+ shapeOrderChanges,
2648
+ shapeDigest,
2649
+ isShapeTextItem,
2650
+ describeShapeItem,
2651
+ describeShapeOrder,
2652
+ shapePathMatches,
2653
+ sectionTableRowCount
2654
+ };