@dzhechkov/skills-feature-adr 1.5.10 → 1.5.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.dz-manifest.json CHANGED
@@ -13,7 +13,7 @@
13
13
  },
14
14
  {
15
15
  "path": "README.md",
16
- "sha256": "b1dbcb7b1f0f3d95bdda9f9568f65e1f78b306b2cb64e42979b1981aeadc6047"
16
+ "sha256": "e72e7ae2a5adc046598feb5de0bcac114d5aa961571f57d2aa530186c1dbb4b3"
17
17
  },
18
18
  {
19
19
  "path": "bin/cli.js",
@@ -25,7 +25,7 @@
25
25
  },
26
26
  {
27
27
  "path": "package.json",
28
- "sha256": "e376fe73743953ad7f0b7525e22f815ea949450fe7ea784f89d1e61bcc62ce6f"
28
+ "sha256": "8b0f75ec75e5f2a4e121c65a625abe6c9a8fb4b12732349293b78232348b2234"
29
29
  },
30
30
  {
31
31
  "path": "src/cli.js",
@@ -121,7 +121,7 @@
121
121
  },
122
122
  {
123
123
  "path": "templates/.claude/skills/feature-adr/modules/00-complexity-router.md",
124
- "sha256": "a1275bd960ddff1c61352f96d8b7a4249371190766f08d1778f5a54f98350950"
124
+ "sha256": "6a98ca7af49aabcb2e14a081815f1167740ac4dba2e0cbb938c87aceac47bfa3"
125
125
  },
126
126
  {
127
127
  "path": "templates/.claude/skills/feature-adr/modules/01-requirements.md",
@@ -149,15 +149,15 @@
149
149
  },
150
150
  {
151
151
  "path": "templates/.claude/skills/feature-adr/modules/06-implementation-plan.md",
152
- "sha256": "c624e0ade0314ca07913ba64e092d4655f86201856e8188c3e616b2320226c1e"
152
+ "sha256": "5c8d4c79d5329702b8c75b9afb36ae335aa63c99e74045c0afa98e12658ee975"
153
153
  },
154
154
  {
155
155
  "path": "templates/.claude/skills/feature-adr/modules/07-code.md",
156
- "sha256": "c427a0abacb4bc1d969d9f47739ce2aeff286a66da73eab67bf042d5c4dc9a59"
156
+ "sha256": "fb054591e86d55cee184eab96b5f0440053ada8fa9edf619e4a66ec5ac371587"
157
157
  },
158
158
  {
159
159
  "path": "templates/.claude/skills/feature-adr/modules/08-qe.md",
160
- "sha256": "ea8fb546acb7f09816d2fc9f9189a86f2400b8cec91c9ba4b82b68a8e32920bb"
160
+ "sha256": "7973676cc9a5a6d8429c0bcdbe163bb5dc16e2d2f28a72698e44ad72d234e199"
161
161
  },
162
162
  {
163
163
  "path": "templates/.claude/skills/feature-adr/modules/09-fleet-qe.md",
@@ -249,7 +249,11 @@
249
249
  },
250
250
  {
251
251
  "path": "templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs",
252
- "sha256": "cd2d4e05d1b61be5485c9836abdebdb00b5f4085a14f3c71126cc45d4e08fe3b"
252
+ "sha256": "08e69d2e63350fe2dc64483226fcd30901b77a8e930ca6280ef71181c67675b4"
253
+ },
254
+ {
255
+ "path": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
256
+ "sha256": "82c3c48c25f400f3a64d1caba3d47357f3db87460f0746f88706f901eb6ff689"
253
257
  },
254
258
  {
255
259
  "path": "templates/.claude/skills/frontend-design/LICENSE.txt",
@@ -313,7 +317,7 @@
313
317
  },
314
318
  {
315
319
  "path": "templates/.claude/workflows/feature-adr.js",
316
- "sha256": "ef6a119466e504307116b12af3369029c13738ad6c40cc7a345d7d81834eca72"
320
+ "sha256": "8900ff6bc777a078aa4e37d8e28a2ffbc612187354995f726daf93e6f87d3109"
317
321
  },
318
322
  {
319
323
  "path": "templates/lib/memory-protocol.md",
@@ -325,5 +329,5 @@
325
329
  }
326
330
  ]
327
331
  },
328
- "signature": "2ltA45AgleqbKopq4ritI5XTO7YtiM/vDWxo6fNIFBYrlz+sz9ZAFSxyNOEN3KYs8uwlgmef1I1sb1wvMIgxAw=="
332
+ "signature": "lPrCKcYVN87Vvd2+Xm6HFSVQdv8eu1AXQ1yGKm42Vad53oNafZvw3gF50Jm4pG1FkKd9vJ1SH/0zHJVNZi9bAQ=="
329
333
  }
package/README.md CHANGED
@@ -698,7 +698,7 @@ sufficiency + honesty, overengineering, silent decisions, runtime consistency, s
698
698
  `dz challenge --plan <plan.md>` or the `challenge-panel` skill; scaffold the degradations registry via
699
699
  `dz feature-adr-setup --from-spec <spec with {"degradations":true}> --apply`.
700
700
 
701
- ### ADR quality gate (Step 3 generates → Step 8 enforces)
701
+ ### ADR quality review + Confirmation file gate (Step 3 generates → Step 8 checks)
702
702
 
703
703
  Step 3 and Step 8 share an ADR best-practices contract distilled from the
704
704
  [architecture-decision-record monograph](https://github.com/architecture-decision-record/architecture-decision-record):
@@ -709,12 +709,17 @@ Step 3 and Step 8 share an ADR best-practices contract distilled from the
709
709
  links + after-action review, a **`## Confirmation`** stanza (method, monitoring, success metric, owner)
710
710
  naming the load-bearing property, and a **`## Links`** traceability block. Template weight is tier-mapped:
711
711
  S/M → Nygard/ITD-lightweight, L/XL → MADR + Confirmation.
712
- - **Step 8** runs a **13-point ADR fitness checklist** (`qe-code-reviewer`) against every generated ADR and
713
- **fails the gate** on any miss — decision-shaped title, controlled-vocabulary Status + reversibility,
712
+ - **Step 8** runs a **13-point advisory ADR fitness checklist** (`qe-code-reviewer`) against every generated ADR —
713
+ decision-shaped title, controlled-vocabulary Status + reversibility,
714
714
  neutral Context-before-Decision, symmetric options, driver-mapped rationale, concrete/testable decision,
715
715
  negative consequences, traceability links, no placeholder, and **rejects explainer-masquerading-as-ADR**.
716
- The **Confirmation→test link is load-bearing**: if the named safety property has no automated test the ADR
717
- grades no better than C.
716
+ Those judgment-based items remain findings; they do not independently force the workflow verdict.
717
+ - The **one mandatory gate** runs after Step 7.5 and before the QE verdict: every test-file path named
718
+ under an ADR heading beginning with `## Confirmation` must exist as a readable regular file. Missing
719
+ files or unreadable/unparseable paths force a non-passing Step-8 grade while the independent QE review
720
+ still runs. A feature with no ADR prints `пропущено: ADR нет, проверять нечего` and is not failed.
721
+ `dz discrimination-check` and `dz mutation-gate` remain advisory: existence does not prove that a test
722
+ actually discriminates the load-bearing property.
718
723
 
719
724
  The pipeline **dog-foods** this: a harness test runs the gate against feature-adr's own generated ADR, so a
720
725
  Step-3↔Step-8 drift fails CI rather than shipping.
@@ -1174,6 +1179,64 @@ pre-code probe that returns nothing no longer becomes an all-null baseline that
1174
1179
  "every target changed".
1175
1180
 
1176
1181
 
1182
+ `next` — **the plan inherits requirements by contract, not by goodwill** (feature `plan-inherits-requirements`,
1183
+ staged: not yet versioned or published). Two swarms plus a cross-family check measured that the implementation
1184
+ plan referenced only 95 of 339 requirement ids across 8 M/L features (28 %), lost one requirement without a trace
1185
+ and introduced one contradiction, while the norm "every FR-N → a task" lived only in the module text — neither
1186
+ the planner prompt nor the K2 gate enforced it. Now:
1187
+
1188
+ - **C8 — requirement coverage.** The K2 gate reads `01_requirements.md`, extracts every id declared at line start
1189
+ in the four corpus shapes (heading / bold / list / table; `FR-N`, `NFR-N`, `AC-N`, `C-N`, with an optional
1190
+ letter group and dotted sub-number — measured over 394 requirement files) and checks each by word boundary in
1191
+ the plan. By default a WARN with the exact count; under `--require-requirements` (which the pipeline passes)
1192
+ every missing id is its own FAIL line. An absent `01_requirements.md` is a WARN naming the absence, never a
1193
+ skip; a file declaring no ids in the contract shapes says so. NAMED LIMIT, in the same form as C1's: C8 is a
1194
+ grep — a prose mention satisfies it; "mentioned but not tasked" is not caught.
1195
+ - **C1 counts decisions by heading, not by filename.** `# ADR-NNN` / `## ADR-NNN` headings inside each ADR file
1196
+ are the decisions the plan owes a task; a file holding four decisions now yields four checks, not one. No
1197
+ heading ⇒ filename prefix with a WARN.
1198
+ - **An unclosed code fence on the declaration side is NOT-ESTABLISHED.** Masking it silently dropped every id
1199
+ after it (measured); restoring it silently established a heading that was only example code. Neither silent
1200
+ reading is honest, so the gate refuses with the file named — close the fence, rerun. The corpus has 0 such
1201
+ files out of 831.
1202
+ - **The Step-6 planner is told the inputs by NAME** (`01_requirements.md`, every `03_adr/*.md`,
1203
+ `05_architecture.md`, `03.5_ideation_report.md` / `04_domain_model.md` when present) and the C8 contract in the
1204
+ same clause as C1/C2/C4 — the lesson that a gate whose contract is not named in the authoring prompt produces
1205
+ refusal after refusal.
1206
+ - **ONE automatic repair round.** A genuine script-verdict FAIL re-dispatches the planner with the FAIL lines
1207
+ ("close EXACTLY these gaps, keep everything else"). Because that sentence is a prompt and not a guarantee, the
1208
+ round is bracketed: the plan is backed up first (no backup ⇒ no repair; a stale `.pre-repair` or a symlinked
1209
+ plan refuses), snapshots before/after compare byte length, every `EXPECTED_CODE_TARGETS` line and every task
1210
+ heading line WITH multiplicity, a snapshot that did not complete REJECTS (never fails open), the re-gate runs
1211
+ before the backup is archived, and a rejected or still-failing repair is restored from the backup with the
1212
+ restore PROVEN by POSIX `cksum` + length. `planGateAttempts` counts gate runs; `planRepair` carries the
1213
+ outcome; a `plan-repair` ledger row is written either way. NAMED LIMIT: task bodies are not proven preserved
1214
+ by any metric — a repair that keeps every heading, every target and 80 % of the bytes while gutting prose is
1215
+ undetectable by construction.
1216
+ - The pure halves (`shellQuote`, `planBackupCmd`, `planRestoreCmd`, `planArchiveBackupCmd`, `planSnapshotCmd`,
1217
+ `snapshotBlock`, `snapshotNumber`, `parsePlanSnapshot`) live in `@dzhechkov/harness-core` and are body-pinned
1218
+ against the inline copies by the drift guard. Three Codex review rounds (C, C, D) — every finding either fixed
1219
+ or named above; the full account is in `features/plan-inherits-requirements/08_qe_report.md`.
1220
+
1221
+ Also staged: **the coder now gets the same recall lane the planner already had** (feature
1222
+ `coder-reads-and-recall`). `1.5.9` wired advisory decision-point micro-recall into Step 3 (ADR
1223
+ selection) and Step 6 (plan routing) only — Step 7 (Code) never called `prepareDecisionRecall` at
1224
+ all, so any lesson reaching the coder was a night-shift human pasting it into the brief by hand.
1225
+ `drDecisionShape` now knows a third decision kind, `code-implementation` → `step-7` /
1226
+ `feature-adr-decision-code-implementation`, kept apart from the plan's own `step-6` bandit context.
1227
+ The call sits INSIDE the code stage's checkpoint, before any of the three places that read the
1228
+ coder's prompt (the Claude dispatch, the Codex dispatch, and the training-pair capture) — so a
1229
+ resumed stage neither re-spends the recall nor loses it, and all three see the identical
1230
+ recall-augmented text. Separately measured (Step 0 of this feature, instrument: host workflow
1231
+ records → the coding-stage agent's own tool-call transcript): a comment added 24.08 naming the plan,
1232
+ every ADR, the architecture doc, requirements and the domain model by file name moved how often
1233
+ Claude-family coders opened `01_requirements.md` from 39 % (11 of 28 runs) before the change to 71 %
1234
+ (5 of 7) after — a real jump, but on **n = 7**, not a controlled comparison, and it says nothing
1235
+ about Codex-family coders: they were 77 % of the post-change sample and are invisible to this
1236
+ instrument (a Codex coder's transcript carries exactly one entry, the dispatch itself). Full method,
1237
+ the two false reads caught before the number was trusted, and the raw counts are in
1238
+ `features/coder-reads-and-recall/00_complexity_assessment.md`.
1239
+
1177
1240
  `1.5.3` — **the workflow stops crashing on the way into Step 7.** `1.5.2` shipped a workflow that
1178
1241
  CALLED three helpers it never defined — `changeSetProbeCmd`, `parseHashProbe`, `changedFromHashes`
1179
1242
  (5 call sites, 0 definitions). `QE_SCOPE` defaults to `uncommitted`, so the guarded branch was true
@@ -1211,3 +1274,11 @@ Also in this release, both halves of the K2 plan-completeness gate that field us
1211
1274
  `dispatchOutcomes` отчёта. Причина отказа остаётся у outcome отказавшей ступени, а intent следующего
1212
1275
  fallback получает нейтральную причину `fallback-rung`. Регион `stage-line` принадлежит генератору
1213
1276
  `gen-loop-blobs` и не правится руками.
1277
+
1278
+ ### Shared Markdown masking in feature-adr gates
1279
+
1280
+ The standalone plan-completeness gate ships with `markdown-masker.mjs`, copied byte-for-byte from
1281
+ harness-core's `src/markdown-masker.ts`. It runs without a core build. Amendment checks, swarm briefs
1282
+ and K2 share the parser while retaining their existing unclosed-block and indentation policies.
1283
+ The four-space indented-code gap remains open for amendment checks and K2; swarm briefs retain their
1284
+ existing masking of indented code. Versions are unchanged in this staged change.
package/bin/cli.js CHANGED
File without changes
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dzhechkov/skills-feature-adr",
3
- "version": "1.5.10",
3
+ "version": "1.5.12",
4
4
  "description": "Adaptive Feature Development skill pack for Claude Code — 11-step pipeline with Complexity Router (S/M/L/XL), ADR-driven architecture, 15 agentic-qe skills, multi-agent fleet QE. Supports --full-qe, --full-qe-extended, --with-learning, and --knowledge-extractor modes.",
5
5
  "bin": {
6
6
  "skills-feature-adr": "./bin/cli.js"
@@ -15,6 +15,10 @@
15
15
  ".dz-manifest.json",
16
16
  "sbom.json"
17
17
  ],
18
+ "scripts": {
19
+ "test": "node --test \"test/**/*.test.js\"",
20
+ "prepack": "node -e \"const fs=require('fs');const bad=['.claude','.skills-feature-adr.json'].filter(p=>fs.existsSync(p));if(bad.length){console.error('prepack guard: stray init artifacts in package dir: '+bad.join(', ')+' — remove before packing');process.exit(1)}\""
21
+ },
18
22
  "keywords": [
19
23
  "claude",
20
24
  "claude-code",
@@ -59,8 +63,5 @@
59
63
  },
60
64
  "publishConfig": {
61
65
  "access": "public"
62
- },
63
- "scripts": {
64
- "test": "node --test \"test/**/*.test.js\""
65
66
  }
66
- }
67
+ }
package/sbom.json CHANGED
@@ -35,7 +35,7 @@
35
35
  "hashes": [
36
36
  {
37
37
  "alg": "SHA-256",
38
- "content": "b1dbcb7b1f0f3d95bdda9f9568f65e1f78b306b2cb64e42979b1981aeadc6047"
38
+ "content": "e72e7ae2a5adc046598feb5de0bcac114d5aa961571f57d2aa530186c1dbb4b3"
39
39
  }
40
40
  ]
41
41
  },
@@ -69,7 +69,7 @@
69
69
  },
70
70
  {
71
71
  "name": "dz:canonical-json-sha256-v2",
72
- "value": "e376fe73743953ad7f0b7525e22f815ea949450fe7ea784f89d1e61bcc62ce6f"
72
+ "value": "8b0f75ec75e5f2a4e121c65a625abe6c9a8fb4b12732349293b78232348b2234"
73
73
  }
74
74
  ]
75
75
  },
@@ -309,7 +309,7 @@
309
309
  "hashes": [
310
310
  {
311
311
  "alg": "SHA-256",
312
- "content": "a1275bd960ddff1c61352f96d8b7a4249371190766f08d1778f5a54f98350950"
312
+ "content": "6a98ca7af49aabcb2e14a081815f1167740ac4dba2e0cbb938c87aceac47bfa3"
313
313
  }
314
314
  ]
315
315
  },
@@ -379,7 +379,7 @@
379
379
  "hashes": [
380
380
  {
381
381
  "alg": "SHA-256",
382
- "content": "c624e0ade0314ca07913ba64e092d4655f86201856e8188c3e616b2320226c1e"
382
+ "content": "5c8d4c79d5329702b8c75b9afb36ae335aa63c99e74045c0afa98e12658ee975"
383
383
  }
384
384
  ]
385
385
  },
@@ -389,7 +389,7 @@
389
389
  "hashes": [
390
390
  {
391
391
  "alg": "SHA-256",
392
- "content": "c427a0abacb4bc1d969d9f47739ce2aeff286a66da73eab67bf042d5c4dc9a59"
392
+ "content": "fb054591e86d55cee184eab96b5f0440053ada8fa9edf619e4a66ec5ac371587"
393
393
  }
394
394
  ]
395
395
  },
@@ -399,7 +399,7 @@
399
399
  "hashes": [
400
400
  {
401
401
  "alg": "SHA-256",
402
- "content": "ea8fb546acb7f09816d2fc9f9189a86f2400b8cec91c9ba4b82b68a8e32920bb"
402
+ "content": "7973676cc9a5a6d8429c0bcdbe163bb5dc16e2d2f28a72698e44ad72d234e199"
403
403
  }
404
404
  ]
405
405
  },
@@ -629,7 +629,17 @@
629
629
  "hashes": [
630
630
  {
631
631
  "alg": "SHA-256",
632
- "content": "cd2d4e05d1b61be5485c9836abdebdb00b5f4085a14f3c71126cc45d4e08fe3b"
632
+ "content": "08e69d2e63350fe2dc64483226fcd30901b77a8e930ca6280ef71181c67675b4"
633
+ }
634
+ ]
635
+ },
636
+ {
637
+ "type": "file",
638
+ "name": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
639
+ "hashes": [
640
+ {
641
+ "alg": "SHA-256",
642
+ "content": "82c3c48c25f400f3a64d1caba3d47357f3db87460f0746f88706f901eb6ff689"
633
643
  }
634
644
  ]
635
645
  },
@@ -789,7 +799,7 @@
789
799
  "hashes": [
790
800
  {
791
801
  "alg": "SHA-256",
792
- "content": "ef6a119466e504307116b12af3369029c13738ad6c40cc7a345d7d81834eca72"
802
+ "content": "8900ff6bc777a078aa4e37d8e28a2ffbc612187354995f726daf93e6f87d3109"
793
803
  }
794
804
  ]
795
805
  },
@@ -103,19 +103,39 @@ Runs after tier classification, only when `{AGENTIC_QE_MODE}` = `direct` | `dire
103
103
 
104
104
  Step 1 folds `{LEARNED_PATTERNS}` into the requirements brief as "lessons from previous features" — advisory context only, never requirements themselves. Why here: Step 0 is the one point where recall can be keyed to the feature's phase and domain — per-prompt hook auto-injection cannot see which pipeline step is running.
105
105
 
106
+ ### 7. Task Kind Classification (experiment-envelope, ADR-001)
107
+
108
+ Classify the feature as exactly ONE of the six task kinds below, and write `Task kind: <x>` as its
109
+ own line in the artifact. This feeds the experiment envelope the conveyor builds after this step —
110
+ downstream learning reads the kind, not a paraphrase, so the line must use one of the exact words.
111
+
112
+ | Task kind | Criterion |
113
+ |-----------|-----------|
114
+ | `feature` | A genuinely new capability, adapter, command, or skill did not exist before this run. |
115
+ | `bugfix` | The change corrects observed incorrect behavior in existing code — a defect, not an absence. |
116
+ | `refactor` | Behavior is unchanged; the change restructures, renames, or simplifies existing code. |
117
+ | `tooling` | The change is to build/CI/dev-tooling/scripts rather than to the product's own runtime behavior. |
118
+ | `docs` | The change is documentation-only (README, ADR prose, comments) with no code delta. |
119
+ | `research` | The deliverable is a finding or a design decision, not a shipped code change. |
120
+
121
+ If `args.taskKind` was passed explicitly by the caller, it OVERRIDES this classification — record
122
+ both: `Task kind: <forced value> (forced by the caller)` as the kind of record, and your own
123
+ classification separately if it differs, the same pattern Step 0 already uses for a forced tier.
124
+
106
125
  ## Output
107
126
 
108
127
  Set the following variables:
109
128
 
110
129
  ```
111
130
  {COMPLEXITY_TIER} = S | M | L | XL
131
+ {TASK_KIND} = feature | bugfix | refactor | tooling | docs | research
112
132
  {ACTIVE_STEPS} = [0, 1, 6, 7, 8] # example for S
113
133
  {TIME_BUDGET} = { requirements: 2, planning: 0, implementation: 10, qe: 3 }
114
134
  {DIMENSION_SCORES} = { files: 1, domains: 1, integrations: 1, breaking: 1, models: 1, crosscutting: 1 }
115
135
  {LEARNED_PATTERNS} = [ ... ] # top-3 by confidence from memory_query; [] in reference mode / no hits / error
116
136
  ```
117
137
 
118
- Create artifact: `features/<slug>/00_complexity_assessment.md`
138
+ Create artifact: `features/<slug>/00_complexity_assessment.md`, including the `Task kind: <x>` line.
119
139
 
120
140
  ## Checkpoint 0 Format
121
141
 
@@ -127,7 +127,9 @@ For each group, identify:
127
127
 
128
128
  Before finalizing the plan, validate completeness:
129
129
 
130
- 1. Cross-reference every `{REQUIREMENT}` (FR-N) → at least one TASK covers it
130
+ 1. Cross-reference every `{REQUIREMENT}` (FR-N) → at least one TASK covers it (C8 — the gate checks
131
+ this by identifier: WARN with a count by default, FAIL under `--require-requirements`, which the
132
+ pipeline passes)
131
133
  2. Cross-reference every `{ADR_DECISION}` → at least one TASK implements it
132
134
  3. Cross-reference every critical risk from `{QUALITY_RISKS}` → mitigation in some TASK
133
135
  4. Name every acid token `A<n>` from `00_complexity_assessment.md` VERBATIM in the plan (C4), each
@@ -233,11 +235,16 @@ C2 recognises JS/TS, pytest, Go, Rust, JVM and .NET test paths, extensible per p
233
235
 
234
236
  Never proceed on a non-zero exit, and never treat empty output as a pass — the last line
235
237
  (`K2 plan-completeness: PASS|FAIL|NOT-ESTABLISHED`) is the verdict, and its absence is not one.
236
- What it checks: C1 every ADR has a plan task citing it · C2 every ADR Confirmation test path is named
237
- in the plan · C3 the `EXPECTED_CODE_TARGETS:` block parses line by line · C4 the feature's declared
238
- acid corpus is named · C5 (WARN) the `Inputs read:` line. An S-tier run with no `03_adr/` skips C1/C2
239
- with a note (it cannot be failed for ADRs it never had) — unless the plan itself cites `ADR-<n>`,
240
- which is NOT-ESTABLISHED. C1 is a grep: it catches "forgot entirely", not "mentioned but not tasked".
238
+ What it checks: C1 every ADR **decision** (a `# ADR-NNN` / `## ADR-NNN` heading INSIDE the file, not
239
+ just the filename prefix — a file with several headings owes several plan citations) has a plan task
240
+ citing it · C2 every ADR Confirmation test path is named in the plan · C3 the `EXPECTED_CODE_TARGETS:`
241
+ block parses line by line · C4 the feature's declared acid corpus is named · C5 (WARN) the
242
+ `Inputs read:` line · C8 every requirement id declared in `01_requirements.md` (`FR-N`, `NFR-N`,
243
+ `AC-N`, `C-N`) is cited by the plan by word boundary — WARN with a count by default, FAIL per missing
244
+ id under `--require-requirements` (the pipeline passes this flag; a bare interactive run of the
245
+ script does not). An S-tier run with no `03_adr/` skips C1/C2 with a note (it cannot be failed for
246
+ ADRs it never had) — unless the plan itself cites `ADR-<n>`, which is NOT-ESTABLISHED. C1 is a grep:
247
+ it catches "forgot entirely", not "mentioned but not tasked".
241
248
 
242
249
  Pass the run's tier so the check cannot be dodged: `--tier=S|M|L|XL`. An M/L/XL feature with no
243
250
  `03_adr/` FAILS C1/C2 (an M+ feature owes ADRs); only `--tier=S` — or no tier at all, and then the
@@ -20,6 +20,7 @@ opus (complex code generation)
20
20
  - `{ADR_DECISIONS}` from Step 3 (M+)
21
21
  - `{DOMAIN_MODEL}` from Step 4 (L/XL)
22
22
  - Codebase context (existing patterns, conventions)
23
+ - {LEARNED_PATTERNS} for Step 7 — the decision-recall block (≤3 lessons), appended to the coder prompt by the pipeline
23
24
 
24
25
  ## Protocol
25
26
 
@@ -311,6 +311,51 @@ Compile all findings into a structured report:
311
311
  ✅ READY FOR MERGE | ❌ NEEDS FIXES | ⚠️ CONDITIONAL APPROVAL
312
312
  ```
313
313
 
314
+ ### 7.1 Findings ledger (machine-readable)
315
+
316
+ Prose is for people; `dz score`/`dz recap` need a machine-readable surface too (qe-findings-record,
317
+ ADR-001). Write BOTH of these into `08_qe_report.md`, in addition to the prose report above:
318
+
319
+ 1. Exactly ONE line, in the PROSE body of the file (never inside a fenced code block, an indented
320
+ code block, a `>` blockquote, or an HTML comment — the parser masks all four before it looks):
321
+ `QE-VERDICT: <A|A-|A+|B|B+|B-|C|C+|C-|D>` — the letter grade you gave above, machine-parseable,
322
+ ASCII hyphen for the sign (the parser also accepts U+2212 as the same sign; write ASCII). Two such
323
+ lines make the report `ambiguous`, never "last wins" — write it once. A line that starts with
324
+ `QE-VERDICT` but gets the grammar wrong (wrong dash, trailing prose, wrong case, no colon) is worse
325
+ than writing nothing: the parser reports `invalid` and REFUSES to fall back to guessing a grade
326
+ from prose — get the one line right rather than close.
327
+ 2. A `## Findings ledger` section — this exact heading, occurring EXACTLY ONCE in the file, at the
328
+ top level (never inside a fence/indented block/blockquote/comment either) — with a table directly
329
+ under it, under this EXACT header (copy it verbatim):
330
+
331
+ ```markdown
332
+ ## Findings ledger
333
+
334
+ | Finding | Severity | Status | Round | Author | Title |
335
+ |---------|----------|--------|-------|--------|-------|
336
+ | F1 | HIGH | fixed | 1 | codex | escaping bug in the token scanner |
337
+ | F2 | MEDIUM | open | 1 | lead | style nit, not blocking |
338
+ ```
339
+
340
+ A table found ANYWHERE else — before the heading, after the section ends at the next heading, with
341
+ no heading in the file, or with the heading duplicated — is REFUSED as `outside ledger section`,
342
+ never parsed as the real ledger; a second table found INSIDE the section is REFUSED as `duplicate
343
+ table`. Both refusals name how many of the rejected table's rows were ignored.
344
+
345
+ Closed dictionaries — a row using anything else is REFUSED by the parser, never coerced to the
346
+ nearest known value:
347
+ - **Severity**: `BLOCKER` `CRITICAL` `HIGH` `MEDIUM` `LOW` `INFO`
348
+ - **Status**: `confirmed` `fixed` `partial` `refuted` `named-limit` `open`
349
+ - **Round**: an integer ≥ 1 (which review round raised it)
350
+ - **Author**: `codex` `claude` `lead`
351
+ - **Finding**: a short id token with no spaces (`F1`, `R2-3`, …); **Title**: free text — a literal
352
+ `|` inside Title must be escaped as `\|`, or wrapped in inline code (`` `a|b` ``), or it fragments
353
+ the row into extra columns and the row is refused as malformed.
354
+
355
+ An empty (header-only) table is read as `hollow: true` — worse than no table at all, because it
356
+ claims a ledger exists and says nothing. If there are no findings, omit the section entirely rather
357
+ than writing an empty table.
358
+
314
359
  ### 8. QE Pattern Store (Direct Mode only)
315
360
 
316
361
  When `{AGENTIC_QE_MODE}` = `direct` | `direct-extended`, after the gap loop is closed and the verdict is set:
@@ -3,10 +3,14 @@
3
3
  // Generalized from features/wave1-instrument-repair/check-plan-completeness.mjs (that copy is the
4
4
  // historical artifact of its run and stays untouched); this one is parameterized by feature dir.
5
5
  //
6
- // USAGE: node .claude/skills/feature-adr/scripts/check-plan-completeness.mjs [<feature-dir>] [--tier=M] [--acid=A1,A2]
6
+ // USAGE: node .claude/skills/feature-adr/scripts/check-plan-completeness.mjs [<feature-dir>] [--tier=M] [--acid=A1,A2] [--require-requirements]
7
7
  // <feature-dir> defaults to the current working directory.
8
8
  // --tier=S|M|L|XL closes the ADR-less dodge (see S-TIER HONESTY); omitting it keeps the
9
9
  // heuristic, and the skip note then names the dodge out loud.
10
+ // --require-requirements promotes C8 (below) from a WARN-with-count to a per-id FAIL; the
11
+ // pipeline passes this flag, so the check is introduced two-shot (ADR-001 plan-inherits-
12
+ // requirements): WARN first so the corpus can be measured without repainting every green
13
+ // fixture red, FAIL once the planner prompt names the contract (feature-adr.js Step 6).
10
14
  //
11
15
  // VERDICT CONTRACT (unchanged from the proven copy — never a silent pass):
12
16
  // PASS exit 0 last line: `K2 plan-completeness: PASS (...)`
@@ -21,6 +25,12 @@
21
25
  // 3dbd2851-adjacent) must parse task structure. Kept honest here: C1 catches "forgot entirely",
22
26
  // not "mentioned but not tasked".
23
27
  //
28
+ // KNOWN LIMITATION (fix round 1, 2026-09-16, same class as C1 above): C8 is also a grep — a PROSE
29
+ // mention of "FR-3" satisfies it exactly like a task reference. Kept honest here too: C8 catches
30
+ // "forgot entirely", not "mentioned but not tasked". Masking the PLAN side (not just the 01/ADR
31
+ // side) for C1 and C8 together, so a prose mention stops satisfying either check, is a separate
32
+ // backlog item — filed by the lead, not chased here.
33
+ //
24
34
  // Checks:
25
35
  // C1 every ADR file in 03_adr/ has >=1 task line in 06_implementation_plan.md citing it (ADR-00N)
26
36
  // C2 every Confirmation-numbered check in each ADR is named in the plan (by its test-file path)
@@ -29,6 +39,23 @@
29
39
  // — SFDIPOT condition: line-level validation, reject-with-reason, not just block presence
30
40
  // C4 the plan names the feature's OWN acid corpus (see "acid corpus" below)
31
41
  // C5 the plan has an 'Inputs read:' line naming 03_adr, 05_architecture (wave-2 seam, cheap here)
42
+ // C8 every requirement id DECLARED in 01_requirements.md (FR-N, NFR-N, AC-N, C-N, with an optional
43
+ // letter group FR-AN and an optional fraction FR-N.N) is CITED by the plan, by word boundary —
44
+ // set difference over identifiers, exactly like C1, never text similarity. WARN with a count by
45
+ // default; FAIL per missing id under --require-requirements (see USAGE above). 01 absent or
46
+ // declaring no ids in the four contract shapes ⇒ WARN, same honesty discipline as C4's absent
47
+ // 00_complexity_assessment.md.
48
+ // NAMED LIMIT (C-1, measured over 394 corpus files 2026-09-16): forms outside FR|NFR|AC|C — bare
49
+ // `RN`/`QN` registries, an id with no hyphen (`FR1`) — are NOT caught. Minority forms in the
50
+ // corpus; widening the regex "just in case" would manufacture false WARN/FAIL on real plans, so
51
+ // the limit is named here rather than chased.
52
+ // NAMED LIMIT (fix round 1, 2026-09-16): the `C-N` shape is read as a Constraint requirement id
53
+ // by this regex — the repo convention this gate trusts. A defect table uses `D-N` (`| D1 |`),
54
+ // never `C-N`; a table of open defects mislabeled with `C-N` rows would be misread as
55
+ // requirement ids. That is a corpus-naming-convention limit, not a bug this gate works around.
56
+ // Declared ids are read from 01 THROUGH `maskMarkdown` (fenced code and HTML comments stripped
57
+ // first, same reader C6 already uses below), so an id quoted inside an example or a comment is
58
+ // never mistaken for a declaration.
32
59
  //
33
60
  // S-TIER HONESTY (no 03_adr/): an S-tier run legitimately has no ADR files, and forcing it to fail a
34
61
  // plan gate it can never satisfy would make the gate a nuisance to route around. So:
@@ -51,6 +78,7 @@
51
78
  // the feature's own `00_complexity_assessment.md` acid-case table (rows shaped `| A<N> | … |`), or
52
79
  // supplied explicitly with `--acid=T1,T2,…`. If neither establishes a corpus, C4 is SKIPPED-with-note
53
80
  // (a feature that declared no acid cases cannot be failed for not naming them).
81
+ import { maskMarkdown } from './markdown-masker.mjs';
54
82
  import { readFileSync, readdirSync, existsSync } from 'node:fs';
55
83
  import { isAbsolute, join, resolve } from 'node:path';
56
84
 
@@ -59,9 +87,11 @@ const acidArg = argv.find((a) => a.startsWith('--acid='));
59
87
  const tierArg = argv.find((a) => a.startsWith('--tier='));
60
88
  const TIER = tierArg ? tierArg.slice('--tier='.length).trim().toUpperCase() : null;
61
89
  const TIER_REQUIRES_ADR = TIER === 'M' || TIER === 'L' || TIER === 'XL';
90
+ const REQUIRE_REQUIREMENTS = argv.includes('--require-requirements');
62
91
  const dirArg = argv.find((a) => !a.startsWith('--'));
63
92
  const FDIR = resolve(dirArg && dirArg !== '' ? (isAbsolute(dirArg) ? dirArg : join(process.cwd(), dirArg)) : process.cwd());
64
93
  const planPath = join(FDIR, '06_implementation_plan.md');
94
+ const requirementsPath = join(FDIR, '01_requirements.md');
65
95
  const adrDir = join(FDIR, '03_adr');
66
96
  const complexityPath = join(FDIR, '00_complexity_assessment.md');
67
97
 
@@ -77,6 +107,23 @@ const out = (s) => console.log(s);
77
107
  const notEstablished = (why) => { out(`K2 plan-completeness: NOT-ESTABLISHED — ${safe(why)}`); process.exit(3); };
78
108
  let failures = [], warnings = [], skips = [];
79
109
 
110
+ // Lead delta after Codex rounds 2+3 (2026-09-16, plan-inherits-requirements): on the DECLARATION side
111
+ // (01_requirements.md ids, ADR headings) an UNCLOSED fence is AMBIGUOUS INPUT — CommonMark reads it as
112
+ // code to EOF, a human reads it as prose that forgot a backtick. Round 2 measured that masking it
113
+ // silently DROPPED every id after it (FR-2 vanished, no line said so); round 3 showed that restoring
114
+ // it silently ESTABLISHED a heading that was only example code. Neither silent reading is honest, so
115
+ // the gate does what it does for every other input it cannot read: NOT-ESTABLISHED, naming the file —
116
+ // "close the fence" is a one-character fix, and the corpus has ZERO such files today (MEASURED
117
+ // 2026-09-16 11:58 UTC over 831 requirements/ADR files). The detector is the two masking modes
118
+ // disagreeing: an unclosed fence with an empty or whitespace-only tail changes no id and needs no
119
+ // verdict — a NAMED LIMIT of the proxy, not a gap in it.
120
+ function maskDeclarations(text, label) {
121
+ const strict = maskMarkdown(text, { unclosed: 'mask' });
122
+ const restored = maskMarkdown(text, { unclosed: 'restore' });
123
+ if (strict !== restored) notEstablished(`${label} has an UNCLOSED code fence — an ambiguous declaration input establishes nothing (close the fence and rerun)`);
124
+ return strict;
125
+ }
126
+
80
127
  if (!existsSync(FDIR)) notEstablished(`feature dir absent: ${FDIR}`);
81
128
  if (!existsSync(planPath)) notEstablished('06_implementation_plan.md absent');
82
129
  const plan = readFileSync(planPath, 'utf-8');
@@ -165,9 +212,35 @@ if (adrFiles.length === 0 && TIER_REQUIRES_ADR) {
165
212
  skips.push(`C1: no 03_adr/ and the plan claims no ADR work — ADR-coverage check SKIPPED${TIER === 'S' ? ' (--tier=S, the legitimate S-tier shape)' : ' (NO --tier supplied: an M/L/XL run that simply never wrote 03_adr/ would dodge C1/C2 here — pass --tier to close it)'}`);
166
213
  skips.push('C2: no 03_adr/ — Confirmation-test coverage check SKIPPED');
167
214
  } else {
168
- // C1 — ADR ids referenced by plan tasks
215
+ // C1 — ADR ids referenced by plan tasks (FR-4). Decisions are counted by HEADING inside the file
216
+ // (`# ADR-NNN` / `## ADR-NNN`, multiline), not by the filename prefix: a file named `001-004-*.md`
217
+ // that actually contains four decisions used to be checked as ONE. A file with no such heading
218
+ // falls back to the old filename-prefix path, WARNed so the fallback is never silent.
219
+ // Fix round 1 (2026-09-16): the file is read THROUGH `maskMarkdown` first (fenced code and HTML
220
+ // comments blanked, positions preserved) — a heading printed inside a ```-fenced example is not a
221
+ // real decision. Up to 3 leading spaces of indentation are tolerated (CommonMark's own limit for a
222
+ // line to still be a heading; 4+ spaces is indented code and is already blanked by the mask).
223
+ // NAMED LIMIT: a nested `### ADR-002` heading cited INSIDE ADR-001's own file as a cross-reference
224
+ // (rather than living in ADR-002's own file) is still counted as a decision owed a task — a corpus
225
+ // rarity, and telling "own decision" from "cross-reference" needs semantic parsing, the same class
226
+ // of limit the C1 grep above already accepts.
227
+ const ADR_HEADING_RE = /^ {0,3}#{1,6}\s*ADR-(\d+)\b/gm;
169
228
  for (const f of adrFiles) {
229
+ let adrTextForHeadings = null;
230
+ try { adrTextForHeadings = maskDeclarations(readFileSync(join(adrDir, f), 'utf-8'), `C1: ${safe(f)}`); } catch { adrTextForHeadings = null; }
231
+ const headingNums = adrTextForHeadings === null
232
+ ? []
233
+ : [...new Set([...adrTextForHeadings.matchAll(ADR_HEADING_RE)].map((mm) => mm[1]))];
234
+ if (headingNums.length > 0) {
235
+ for (const n of headingNums) {
236
+ const id = `ADR-${n}`;
237
+ const re = new RegExp(`ADR-0*${Number(n)}\\b`);
238
+ if (!re.test(plan)) failures.push(`C1: ${id} (${safe(f)}) has NO task in the plan referencing it`);
239
+ }
240
+ continue;
241
+ }
170
242
  const m = f.match(/^(\d{3})-/); if (!m) { warnings.push(`C1: unparseable ADR filename ${safe(f)}`); continue; }
243
+ warnings.push(`C1: ${safe(f)} has no ADR-NNN heading — falling back to the filename prefix`);
171
244
  const id = `ADR-${m[1]}`;
172
245
  const re = new RegExp(`ADR-0*${Number(m[1])}\\b`);
173
246
  if (!re.test(plan)) failures.push(`C1: ${id} (${safe(f)}) has NO task in the plan referencing it`);
@@ -319,21 +392,7 @@ else for (const t of acidTokens) if (!new RegExp(`\\b${t.replace(/[.*+?^${}()|[\
319
392
  // `## Amendments` could become the section heading, and a fenced example row could either open a
320
393
  // phantom amendment or hand a real testless one someone else's marker. Third fence-blindness
321
394
  // found in a checker today, so it is closed here by construction rather than by care.
322
- const rawLines = plan.split('\n');
323
- const planLines = [];
324
- {
325
- let fence = null;
326
- for (const line of rawLines) {
327
- const open = /^ {0,3}(```+|~~~+)/.exec(line);
328
- if (fence === null && open) { fence = open[1][0]; planLines.push(''); continue; }
329
- if (fence !== null) {
330
- planLines.push('');
331
- if (new RegExp('^ {0,3}' + fence + '{3,}\\s*$').test(line)) fence = null;
332
- continue;
333
- }
334
- planLines.push(line);
335
- }
336
- }
395
+ const planLines = maskMarkdown(plan, { unclosed: 'mask' }).split('\n');
337
396
  let sectionStart = -1, sectionEnd = -1, cursor = 0;
338
397
  for (const pl of planLines) {
339
398
  // The SAME heading shape amendment-trace.ts accepts: up to three leading spaces, two to four
@@ -445,6 +504,42 @@ else for (const t of acidTokens) if (!new RegExp(`\\b${t.replace(/[.*+?^${}()|[\
445
504
  }
446
505
  }
447
506
 
507
+ // ── C8 requirement-id coverage (FR-1/2/3, ADR-001 plan-inherits-requirements) ───────────────────
508
+ // Set difference over IDENTIFIERS, exactly like C1 — never text similarity. Declared id forms are
509
+ // measured from the corpus (see the USAGE-block comment above for the named C-1 limit): `FR-N`,
510
+ // `NFR-N`, `AC-N`, `C-N`, an optional letter group (`FR-AN`) and an optional fraction (`FR-N.N`),
511
+ // declared at the START of a line as a heading, a bold run, a list item, or a table cell.
512
+ {
513
+ const REQ_ID_DECL_RE = /^\s*(?:#{1,6}\s*|[-*]\s+|\|\s*)?\**\s*((?:FR|NFR|AC|C)-[A-Z]?\d+(?:\.\d+)?)\b/gm;
514
+ const escapeReqId = (s) => String(s).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
515
+ if (!existsSync(requirementsPath)) {
516
+ warnings.push('C8: 01_requirements.md is ABSENT — requirement coverage had no input to read');
517
+ } else {
518
+ // Fix round 1 (2026-09-16): masked THROUGH maskMarkdown first, exactly like C1's heading scan —
519
+ // an id inside a ```-fenced example or an HTML comment is not a declaration.
520
+ const requirementsText = maskDeclarations(readFileSync(requirementsPath, 'utf-8'), 'C8: 01_requirements.md');
521
+ const reqIds = [...new Set([...requirementsText.matchAll(REQ_ID_DECL_RE)].map((mm) => mm[1]))];
522
+ if (reqIds.length === 0) {
523
+ warnings.push('C8: 01_requirements.md declares NO requirement ids in the contract shapes (FR-N / NFR-N / AC-N / C-N at line start) — nothing to cover');
524
+ } else {
525
+ const missing = reqIds.filter((id) => !new RegExp('\\b' + escapeReqId(id) + '\\b').test(plan));
526
+ if (missing.length === 0) {
527
+ // Silence is not a verdict (K7): a check that ran and found nothing wrong must still print,
528
+ // or a reader cannot tell "C8 ran clean" from "C8 never ran". Deliberately NOT a PASS/FAIL/
529
+ // NOT-ESTABLISHED word — parsePlanGateVerdict anchors on the LAST such word, and a stray one
530
+ // mid-stream is exactly the G-F1 forgery class this script's `safe()` already defends against.
531
+ out(`NOTE C8: all ${reqIds.length} requirement ids referenced`);
532
+ } else if (REQUIRE_REQUIREMENTS) {
533
+ for (const id of missing) failures.push(`C8: ${id} (01_requirements.md) is not referenced by the plan`);
534
+ } else {
535
+ const shown = missing.slice(0, 12);
536
+ const more = missing.length > 12 ? `, …and ${missing.length - 12} more` : '';
537
+ warnings.push(`C8: ${missing.length} of ${reqIds.length} requirement ids are not referenced by the plan: ${shown.join(', ')}${more}`);
538
+ }
539
+ }
540
+ }
541
+ }
542
+
448
543
  // C5 — Inputs read line
449
544
  if (!/Inputs read:/i.test(plan)) warnings.push('C5: no "Inputs read:" line (wave-2 seam, WARN only)');
450
545
  else for (const need of ['03_adr','05_architecture']) if (!plan.includes(need)) warnings.push(`C5: Inputs read line missing ${need}`);