@dzhechkov/skills-feature-adr 1.5.13 → 1.5.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.dz-manifest.json CHANGED
@@ -5,7 +5,7 @@
5
5
  "files": [
6
6
  {
7
7
  "path": "CHANGELOG.md",
8
- "sha256": "88e0389016f697b90731ee8743a35e523f0ccb079426d59861d894f33e516392"
8
+ "sha256": "26fa03f89862d03ebfc6447c866cb5d4dcbd5591ae4f8fb3a3ea08b6c586e19d"
9
9
  },
10
10
  {
11
11
  "path": "LICENSE",
@@ -13,7 +13,7 @@
13
13
  },
14
14
  {
15
15
  "path": "README.md",
16
- "sha256": "619a231352a602c736ce6eec1e5fa79a056dc2f5f80f6b35eed236f9670f7337"
16
+ "sha256": "8c96e82b41fb8479befb56797a367b3cc2119cc5b022b18fdd18f129ef8cd0f0"
17
17
  },
18
18
  {
19
19
  "path": "bin/cli.js",
@@ -25,7 +25,7 @@
25
25
  },
26
26
  {
27
27
  "path": "package.json",
28
- "sha256": "07537ebbd424c668a37a42b670c033a4d3134a9e7e13803f426aa31149a85b58"
28
+ "sha256": "0cdd048a5aacd52294f72b5903b5a94c049ad61680dbcac532d5cb3fbd45c736"
29
29
  },
30
30
  {
31
31
  "path": "src/cli.js",
@@ -37,7 +37,7 @@
37
37
  },
38
38
  {
39
39
  "path": "src/commands/init.js",
40
- "sha256": "fbc3853435048c5b632119bf37cc702c4c305cae7de656548891295f6b9e8947"
40
+ "sha256": "07ff9d955422979ea67aa5aa454c13303f5ae644efb32e5cc6f359a509638347"
41
41
  },
42
42
  {
43
43
  "path": "src/commands/list.js",
@@ -157,7 +157,7 @@
157
157
  },
158
158
  {
159
159
  "path": "templates/.claude/skills/feature-adr/modules/08-qe.md",
160
- "sha256": "7ab9b7b1523289986a1f51b4cbbebf4410d96ad4a153f22fa8613e8a0f4f5fa7"
160
+ "sha256": "5355e2b36ab13b6acb65abe5b206685915ef2b08191b4a3130574620012b9135"
161
161
  },
162
162
  {
163
163
  "path": "templates/.claude/skills/feature-adr/modules/09-fleet-qe.md",
@@ -249,7 +249,7 @@
249
249
  },
250
250
  {
251
251
  "path": "templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs",
252
- "sha256": "08e69d2e63350fe2dc64483226fcd30901b77a8e930ca6280ef71181c67675b4"
252
+ "sha256": "eb77822f121ade10bd6265b0f0145ec38507a07d2cb72ba5fa8336c962096bda"
253
253
  },
254
254
  {
255
255
  "path": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
@@ -317,7 +317,7 @@
317
317
  },
318
318
  {
319
319
  "path": "templates/.claude/workflows/feature-adr.js",
320
- "sha256": "e96c5280ad21b604036cc912418c9627b0ff0ae8ca1436c58dcff951ba9a034b"
320
+ "sha256": "6d1bb93be6fae296281ac8f9fc2f266e5f09336cf6e183912fab5ff3bd0d3fd8"
321
321
  },
322
322
  {
323
323
  "path": "templates/lib/memory-protocol.md",
@@ -329,5 +329,5 @@
329
329
  }
330
330
  ]
331
331
  },
332
- "signature": "CfTJgsBbekrZE+IbzaQFLofuRDZ3B2CJTqPk6gjeOiLXhChXiPGkqxaWKCT+hdQTy6ORB7fu6vfMx37rwJf6Cg=="
332
+ "signature": "+DKyJg37w6aHqhg/UhCpQ+y+w/e66HEEr9hgOv/TttSIMc6TokGEbvJpxEw2pdu19Q0SHlA/o6pSucM+h3KEDw=="
333
333
  }
package/CHANGELOG.md CHANGED
@@ -1,5 +1,19 @@
1
1
  # Changelog
2
2
 
3
+ ## [Unreleased]
4
+
5
+ ### Fixed — `init --force` больше не стирает локальную правку молча
6
+
7
+ - Из трёх путей записи `init --force` был ЕДИНСТВЕННЫМ, который перезаписывал локально
8
+ изменённый файл без резервной копии и без строки в отчёте — при том, что баннер успеха
9
+ рекомендует именно эту команду. `update` сохраняет правку по трёхсторонней сверке,
10
+ `update --force` кладёт `.bak` с первого дня; расходился только `init --force`.
11
+ - Теперь `init --force` копирует изменённый файл в `<file>.bak` ПЕРЕД перезаписью и называет
12
+ его в отчёте. Копия делается только если байты ОТЛИЧАЮТСЯ от шаблона: `.bak`, совпадающий
13
+ с шаблоном, — чистый мусор. `--dry-run --force` называет будущую копию и ничего не пишет.
14
+ - Поведение самого `--force` не смягчено: файл по-прежнему перезаписывается, это его смысл.
15
+ Менялось только то, что правка перестала исчезать бесследно.
16
+
3
17
  ## [1.5.1] - 2026-08-21
4
18
 
5
19
  ### Changed — the Step-8 amendment gate is a COMMAND, and the durable writers are witnessed
package/README.md CHANGED
@@ -63,7 +63,9 @@ npx @dzhechkov/skills-feature-adr init # Install core components
63
63
  npx @dzhechkov/skills-feature-adr init --with-learning # + reward learning
64
64
  npx @dzhechkov/skills-feature-adr init --knowledge-extractor # + knowledge extractor
65
65
  npx @dzhechkov/skills-feature-adr init --with-learning --knowledge-extractor # + both
66
- npx @dzhechkov/skills-feature-adr init --force # Overwrite existing files
66
+ npx @dzhechkov/skills-feature-adr init --force # Overwrite existing files (a locally
67
+ # changed file is copied to <file>.bak first,
68
+ # and the copy is named in the report)
67
69
  npx @dzhechkov/skills-feature-adr init --dry-run # Preview without making changes
68
70
  npx @dzhechkov/skills-feature-adr update # Update to latest version
69
71
  npx @dzhechkov/skills-feature-adr remove # Clean uninstall
@@ -106,6 +108,13 @@ ARCHITECTURE → IMPLEMENTATION → CODE → QE → FLEET QE
106
108
  # Full protocols + 6 extra skills, up to 7 fleet QE agents
107
109
  ```
108
110
 
111
+ ### Checkpoint reads have one source (v1.5.14)
112
+
113
+ `loadCheckpoints` in the bundled `feature-adr.js` now builds its read command with the checkpoints blob's own
114
+ `checkpointReadCmd(FDIR)` instead of a hand-written duplicate, so the helper with the speaking name is the one the pipeline
115
+ runs. Behaviour is unchanged: for directory names with spaces, quotes and missing directories the shell output is
116
+ byte-identical (proved against the old command). A wiring test fails if the duplicate is re-inlined.
117
+
109
118
  ### Advisory micro-recall at two decision points (v1.5.9, staged)
110
119
 
111
120
  The workflow makes one bounded decision-local recall attempt immediately before the live Step 3
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dzhechkov/skills-feature-adr",
3
- "version": "1.5.13",
3
+ "version": "1.5.14",
4
4
  "description": "Adaptive Feature Development skill pack for Claude Code — 11-step pipeline with Complexity Router (S/M/L/XL), ADR-driven architecture, 15 agentic-qe skills, multi-agent fleet QE. Supports --full-qe, --full-qe-extended, --with-learning, and --knowledge-extractor modes.",
5
5
  "bin": {
6
6
  "skills-feature-adr": "./bin/cli.js"
package/sbom.json CHANGED
@@ -15,7 +15,7 @@
15
15
  "hashes": [
16
16
  {
17
17
  "alg": "SHA-256",
18
- "content": "88e0389016f697b90731ee8743a35e523f0ccb079426d59861d894f33e516392"
18
+ "content": "26fa03f89862d03ebfc6447c866cb5d4dcbd5591ae4f8fb3a3ea08b6c586e19d"
19
19
  }
20
20
  ]
21
21
  },
@@ -35,7 +35,7 @@
35
35
  "hashes": [
36
36
  {
37
37
  "alg": "SHA-256",
38
- "content": "619a231352a602c736ce6eec1e5fa79a056dc2f5f80f6b35eed236f9670f7337"
38
+ "content": "8c96e82b41fb8479befb56797a367b3cc2119cc5b022b18fdd18f129ef8cd0f0"
39
39
  }
40
40
  ]
41
41
  },
@@ -69,7 +69,7 @@
69
69
  },
70
70
  {
71
71
  "name": "dz:canonical-json-sha256-v2",
72
- "value": "07537ebbd424c668a37a42b670c033a4d3134a9e7e13803f426aa31149a85b58"
72
+ "value": "0cdd048a5aacd52294f72b5903b5a94c049ad61680dbcac532d5cb3fbd45c736"
73
73
  }
74
74
  ]
75
75
  },
@@ -99,7 +99,7 @@
99
99
  "hashes": [
100
100
  {
101
101
  "alg": "SHA-256",
102
- "content": "fbc3853435048c5b632119bf37cc702c4c305cae7de656548891295f6b9e8947"
102
+ "content": "07ff9d955422979ea67aa5aa454c13303f5ae644efb32e5cc6f359a509638347"
103
103
  }
104
104
  ]
105
105
  },
@@ -399,7 +399,7 @@
399
399
  "hashes": [
400
400
  {
401
401
  "alg": "SHA-256",
402
- "content": "7ab9b7b1523289986a1f51b4cbbebf4410d96ad4a153f22fa8613e8a0f4f5fa7"
402
+ "content": "5355e2b36ab13b6acb65abe5b206685915ef2b08191b4a3130574620012b9135"
403
403
  }
404
404
  ]
405
405
  },
@@ -629,7 +629,7 @@
629
629
  "hashes": [
630
630
  {
631
631
  "alg": "SHA-256",
632
- "content": "08e69d2e63350fe2dc64483226fcd30901b77a8e930ca6280ef71181c67675b4"
632
+ "content": "eb77822f121ade10bd6265b0f0145ec38507a07d2cb72ba5fa8336c962096bda"
633
633
  }
634
634
  ]
635
635
  },
@@ -799,7 +799,7 @@
799
799
  "hashes": [
800
800
  {
801
801
  "alg": "SHA-256",
802
- "content": "e96c5280ad21b604036cc912418c9627b0ff0ae8ca1436c58dcff951ba9a034b"
802
+ "content": "6d1bb93be6fae296281ac8f9fc2f266e5f09336cf6e183912fab5ff3bd0d3fd8"
803
803
  }
804
804
  ]
805
805
  },
@@ -54,14 +54,15 @@ function showKeysariumIntegration(keysariumManifest) {
54
54
 
55
55
  // Copies a component file-by-file with per-file overwrite protection:
56
56
  // - destination missing -> write, record in `written`
57
- // - destination exists + --force -> overwrite, record in `written`
57
+ // - destination exists + --force -> BACK UP to a .bak sibling if the bytes differ,
58
+ // then overwrite; record in `written` (+ `backedUp`)
58
59
  // - destination exists, no --force -> do NOT write, record in `preserved`
59
60
  // In dry-run mode nothing is written, but the same written/preserved
60
61
  // classification is produced.
61
62
  // Returns { missing, fileCount } where `missing` means the template source
62
63
  // was absent on disk and `fileCount` is how many files the template provides.
63
64
  function installComponent(key, comp, templatesDir, targetDir, opts) {
64
- const { force, dryRun, written, preserved, hashes } = opts;
65
+ const { force, dryRun, written, preserved, backedUp, hashes } = opts;
65
66
  const src = path.join(templatesDir, comp.src);
66
67
  const destRoot = path.join(targetDir, comp.src);
67
68
 
@@ -89,12 +90,20 @@ function installComponent(key, comp, templatesDir, targetDir, opts) {
89
90
  }
90
91
 
91
92
  for (const entry of entries) {
92
- if (fileExists(entry.destFile) && !force) {
93
+ const existed = fileExists(entry.destFile);
94
+ if (existed && !force) {
93
95
  preserved.push(entry.rel);
94
96
  continue;
95
97
  }
98
+ // `update --force` has backed edits up to a `.bak` sibling since day one; `init --force` was
99
+ // the ONE path that overwrote silently — and the success banner recommends exactly that
100
+ // command, so a locally-edited workflow could vanish with no notice and no copy.
101
+ // Only DIFFERING bytes are backed up: a `.bak` identical to the template is pure litter.
102
+ const differs = existed && hashFile(entry.destFile) !== hashFile(entry.srcFile);
103
+ if (differs && backedUp) backedUp.push(entry.rel);
96
104
  if (!dryRun) {
97
105
  ensureDir(path.dirname(entry.destFile));
106
+ if (differs) fs.copyFileSync(entry.destFile, `${entry.destFile}.bak`);
98
107
  fs.copyFileSync(entry.srcFile, entry.destFile);
99
108
  // Record the SHA-256 of the TEMPLATE bytes we just installed (not the
100
109
  // dest) as this file's baseline. This makes baseline == mine immediately
@@ -110,6 +119,23 @@ function installComponent(key, comp, templatesDir, targetDir, opts) {
110
119
  return { missing: false, fileCount: entries.length };
111
120
  }
112
121
 
122
+ // Print the block of locally-changed files that were (or would be) backed up before --force
123
+ // overwrote them. Silence here is what the field report caught: the operator had no way to learn
124
+ // that their edit was gone, let alone where the copy is.
125
+ function printBackedUpBlock(backedUp, dryRun) {
126
+ if (backedUp.length === 0) return;
127
+ const MAX_SHOWN = 10; // `printPreservedBlock` держит свою копию — она объявлена в его теле
128
+ console.log('');
129
+ const verb = dryRun ? 'would be backed up' : 'backed up';
130
+ warn(`${backedUp.length} locally-changed file(s) ${verb} to a .bak sibling before being overwritten:`);
131
+ for (const rel of backedUp.slice(0, MAX_SHOWN)) {
132
+ console.log(dim(` ${rel} -> ${rel}.bak`));
133
+ }
134
+ if (backedUp.length > MAX_SHOWN) {
135
+ console.log(dim(` …and ${backedUp.length - MAX_SHOWN} more`));
136
+ }
137
+ }
138
+
113
139
  // Print the block of pre-existing files that were (or would be) preserved
114
140
  function printPreservedBlock(preserved, dryRun) {
115
141
  console.log('');
@@ -233,6 +259,7 @@ async function run(options) {
233
259
  const installedFiles = []; // files written (or would-write in dry-run)
234
260
  const installedHashes = {}; // rel -> sha256 of the TEMPLATE bytes installed (baseline)
235
261
  const preservedFiles = []; // pre-existing files NOT overwritten (no --force)
262
+ const backedUpFiles = []; // locally-changed files copied to .bak before --force overwrote them
236
263
  const completedKeys = []; // component keys processed so far (for partial manifest)
237
264
  const installedOptionalKeys = [];
238
265
  let stepNum = 0;
@@ -246,7 +273,7 @@ async function run(options) {
246
273
 
247
274
  installComponent(key, comp, templatesDir, targetDir, {
248
275
  force, dryRun, written: installedFiles, preserved: preservedFiles,
249
- hashes: installedHashes,
276
+ backedUp: backedUpFiles, hashes: installedHashes,
250
277
  });
251
278
  completedKeys.push(key);
252
279
  }
@@ -262,7 +289,7 @@ async function run(options) {
262
289
 
263
290
  const res = installComponent(key, comp, templatesDir, targetDir, {
264
291
  force, dryRun, written: installedFiles, preserved: preservedFiles,
265
- hashes: installedHashes,
292
+ backedUp: backedUpFiles, hashes: installedHashes,
266
293
  });
267
294
  if (!res.missing && res.fileCount > 0) {
268
295
  installedOptionalKeys.push(key);
@@ -296,6 +323,10 @@ async function run(options) {
296
323
  if (dryRun) {
297
324
  console.log('');
298
325
  info(`Dry run: ${installedFiles.length} file(s) would be written, ${preservedFiles.length} pre-existing file(s) would be preserved.`);
326
+ // Отчёт о копиях стоит СНАРУЖИ стража сохранённых: при --force сохранять нечего, список
327
+ // пуст, и внутри стража блок был бы недостижим ровно в том случае, ради которого написан.
328
+ // Пустой список функция отсекает сама.
329
+ printBackedUpBlock(backedUpFiles, true);
299
330
  if (preservedFiles.length > 0) {
300
331
  printPreservedBlock(preservedFiles, true);
301
332
  }
@@ -318,10 +349,11 @@ async function run(options) {
318
349
  process.exit(0);
319
350
  }
320
351
 
352
+ printBackedUpBlock(backedUpFiles, false); // снаружи: при --force preservedFiles пуст
321
353
  if (preservedFiles.length > 0) {
322
354
  printPreservedBlock(preservedFiles, false);
323
- console.log('');
324
355
  }
356
+ if (preservedFiles.length > 0 || backedUpFiles.length > 0) console.log('');
325
357
 
326
358
  // ── f) Write manifest ──────────────────────────────────────────────────
327
359
  const pkgPath = path.resolve(__dirname, '../../package.json');
@@ -205,7 +205,9 @@ list), grep for unfinished-stub markers: `TODO` / `FIXME` / `HACK` / `XXX` / `PL
205
205
  file:line, UNLESS the line carries an inline `no-stubs: <reason>` waiver WITH a non-empty reason, or
206
206
  `.dz/guard.json` `stubWaivers` lists the path WITH a reason. A REASONLESS waiver is itself a HIGH gap,
207
207
  never an exemption. Cross-check mechanically: `dz guard check --op publish --json` runs the same scan
208
- as the SOFT `no-stubs` rule over the working-tree diff. When you QUOTE a marker in `08_qe_report.md`,
208
+ as the SOFT `no-stubs` rule over the working-tree diff.
209
+ After the Step-7 code has landed, run `dz guard check --op code --json` and treat a HARD `block` verdict
210
+ as a HIGH finding naming the drifted file. When you QUOTE a marker in `08_qe_report.md`,
209
211
  backtick it so the report itself scans clean (the claim-check forbidden-phrase convention). Record the
210
212
  verdict in the ADR Fitness section.
211
213
 
@@ -311,6 +313,9 @@ Compile all findings into a structured report:
311
313
  ✅ READY FOR MERGE | ❌ NEEDS FIXES | ⚠️ CONDITIONAL APPROVAL
312
314
  ```
313
315
 
316
+ Settle a cross-family re-QE debt with `dz reqe --slug <s> --done --report <08b>`.
317
+ For `dz reqe --done`, exit 3 means the review settled but named BLOCKER/HIGH findings — stop and surface them to the owner.
318
+
314
319
  ### 7.1 Findings ledger (machine-readable)
315
320
 
316
321
  Prose is for people; `dz score`/`dz recap` need a machine-readable surface too (qe-findings-record,
@@ -373,9 +378,13 @@ resume guard at all before this).
373
378
  Otherwise — QE ran fresh in this pass — after `08_qe_report.md` is written and the grade is final, run:
374
379
 
375
380
  ```bash
376
- dz feature-adr-record --kind ledger --stage qe --slug <slug> --row '<json>' --auto --json
381
+ dz feature-adr-record --kind ledger --stage qe --slug <slug> --row '<json>' --json
377
382
  ```
378
383
 
384
+ fix-round-1 (Codex r1 HIGH finding 4, ADR-001 D5): no `--auto` here. `--auto` is the trusted marker
385
+ of an AUTOMATED pipeline run and requires an experiment envelope that the manual path does not have;
386
+ this plain-mode row is written without it.
387
+
379
388
  `<json>` carries the same fields the ultracode pipeline writes for this row: `reviewer` (the model
380
389
  that reviewed, or `null` if unknown — never guessed), `reviewerFamily` (`claude`|`codex`|`null` when
381
390
  the reviewer identity is not one of the two known families — never guessed as `claude`), `qeRole`
@@ -31,12 +31,18 @@
31
31
  // side) for C1 and C8 together, so a prose mention stops satisfying either check, is a separate
32
32
  // backlog item — filed by the lead, not chased here.
33
33
  //
34
+ // KNOWN LIMITATION (C9): only byte-identical copies are visible before the edit. Already drifted
35
+ // or intentionally different pinned copies remain the identity tests' job; the frozen
36
+ // features/wave1-instrument-repair/check-plan-completeness.mjs is a real example C9 cannot see.
37
+ //
34
38
  // Checks:
35
39
  // C1 every ADR file in 03_adr/ has >=1 task line in 06_implementation_plan.md citing it (ADR-00N)
36
40
  // C2 every Confirmation-numbered check in each ADR is named in the plan (by its test-file path)
37
41
  // C3 the plan carries an EXPECTED_CODE_TARGETS: block, non-empty, and EVERY line parses to a
38
42
  // plausible repo-relative path (no spaces unless quoted, no traversal, no markdown residue)
39
43
  // — SFDIPOT condition: line-level validation, reject-with-reason, not just block presence
44
+ // C9 every tracked byte-identical twin of a target is also listed or explicitly waived with a
45
+ // reason; size-first narrowing, fixed exclusions, one FAIL per missing (target, twin) pair
40
46
  // C4 the plan names the feature's OWN acid corpus (see "acid corpus" below)
41
47
  // C5 the plan has an 'Inputs read:' line naming 03_adr, 05_architecture (wave-2 seam, cheap here)
42
48
  // C8 every requirement id DECLARED in 01_requirements.md (FR-N, NFR-N, AC-N, C-N, with an optional
@@ -79,7 +85,9 @@
79
85
  // supplied explicitly with `--acid=T1,T2,…`. If neither establishes a corpus, C4 is SKIPPED-with-note
80
86
  // (a feature that declared no acid cases cannot be failed for not naming them).
81
87
  import { maskMarkdown } from './markdown-masker.mjs';
82
- import { readFileSync, readdirSync, existsSync } from 'node:fs';
88
+ import { readFileSync, readdirSync, existsSync, statSync } from 'node:fs';
89
+ import { execFileSync } from 'node:child_process';
90
+ import { createHash } from 'node:crypto';
83
91
  import { isAbsolute, join, resolve } from 'node:path';
84
92
 
85
93
  const argv = process.argv.slice(2);
@@ -335,6 +343,7 @@ function classifyTargetPath(path) {
335
343
 
336
344
  // C3 — EXPECTED_CODE_TARGETS block, line-level validation
337
345
  const blockM = plan.match(/EXPECTED_CODE_TARGETS:\s*\n((?:\s*[-*]\s*.+\n?)+)/);
346
+ const listedTargets = new Set();
338
347
  if (!blockM) failures.push('C3: no EXPECTED_CODE_TARGETS: block in the plan');
339
348
  else {
340
349
  const lines = blockM[1].split('\n').map(s => s.trim()).filter(Boolean);
@@ -343,6 +352,81 @@ else {
343
352
  const path = ln.replace(/^[-*]\s*/, '').replace(/`/g, '').trim();
344
353
  const reasons = classifyTargetPath(path);
345
354
  if (reasons.length) failures.push(`C3: target line rejected: "${safe(ln)}" — ${reasons.join(', ')}`);
355
+ else listedTargets.add(path);
356
+ }
357
+ }
358
+
359
+ // C9 — every byte-identical copy is listed or carries a reasoned waiver (ADR-001).
360
+ const TWIN_EXCLUDED_PREFIXES = ['out/', '.claude/worktrees/', 'node_modules/'];
361
+ {
362
+ const excluded = (path) => TWIN_EXCLUDED_PREFIXES.some((prefix) => path.startsWith(prefix) || path.includes('/' + prefix));
363
+ const waivers = [];
364
+ for (const line of maskMarkdown(plan, { unclosed: 'mask' }).split('\n')) {
365
+ if (!/^\s*(?:[-*]\s*)?TWIN_NOT_A_TARGET:/.test(line)) continue;
366
+ const match = line.match(/^\s*(?:[-*]\s*)?TWIN_NOT_A_TARGET:\s*`?([^\s`]+)`?\s*(?:—|--)\s*(\S.*)$/);
367
+ if (!match) {
368
+ failures.push(`C9: waiver "${safe(line.trim())}" carries no reason — a waiver without a reason is an allowlist entry`);
369
+ } else waivers.push({ path: match[1], used: false });
370
+ }
371
+
372
+ let tracked = null;
373
+ try {
374
+ tracked = execFileSync('git', ['ls-files', '-z'], {
375
+ cwd: process.cwd(), encoding: 'utf-8', maxBuffer: 16 * 1024 * 1024,
376
+ stdio: ['ignore', 'pipe', 'pipe'],
377
+ }).split('\0').filter(Boolean);
378
+ } catch (error) {
379
+ warnings.push(`C9: twin scan had no input — git ls-files failed in ${safe(process.cwd())}: ${safe(error.message)}`);
380
+ }
381
+ if (tracked !== null) {
382
+ // Stat the universe first; only size matches ever reach readFileSync / md5.
383
+ const sizes = new Map();
384
+ for (const path of new Set([...tracked, ...listedTargets])) {
385
+ if (excluded(path)) continue;
386
+ try {
387
+ const stat = statSync(path);
388
+ if (stat.isFile()) sizes.set(path, stat.size);
389
+ } catch (error) {
390
+ // Missing targets are new files; tracked files can also have been deleted locally.
391
+ if (error.code !== 'ENOENT' && error.code !== 'ENOTDIR') warnings.push(`C9: cannot stat ${safe(path)}: ${safe(error.message)}`);
392
+ }
393
+ }
394
+ const targetSizes = new Set([...listedTargets].map((path) => sizes.get(path)).filter((size) => size > 0));
395
+ const hashes = new Map();
396
+ const twinsByHash = new Map();
397
+ for (const [path, size] of sizes) {
398
+ if (!targetSizes.has(size)) continue;
399
+ try { hashes.set(path, createHash('md5').update(readFileSync(path)).digest('hex')); }
400
+ catch (error) { warnings.push(`C9: cannot hash ${safe(path)}: ${safe(error.message)}`); }
401
+ }
402
+ for (const path of tracked) {
403
+ const hash = hashes.get(path);
404
+ if (hash === undefined) continue;
405
+ if (!twinsByHash.has(hash)) twinsByHash.set(hash, []);
406
+ twinsByHash.get(hash).push(path);
407
+ }
408
+ let unlisted = 0;
409
+ for (const target of listedTargets) {
410
+ for (const twin of twinsByHash.get(hashes.get(target)) ?? []) {
411
+ if (twin === target) continue;
412
+ let waived = false;
413
+ for (const waiver of waivers) {
414
+ if (waiver.path.endsWith('/') ? twin.startsWith(waiver.path) : twin === waiver.path) {
415
+ waiver.used = true;
416
+ waived = true;
417
+ }
418
+ }
419
+ if (listedTargets.has(twin) || waived) continue;
420
+ unlisted++;
421
+ // Keep pairs separate: safe() truncates each echoed value at 300 characters.
422
+ failures.push(`C9: target ${safe(target)} has a byte-identical twin the plan does not list: ${safe(twin)} — list it or waive it (TWIN_NOT_A_TARGET: ${safe(twin)} — <why>)`);
423
+ }
424
+ }
425
+ for (const waiver of waivers) if (!waiver.used) warnings.push(`C9: unused waiver ${safe(waiver.path)}`);
426
+ if (unlisted === 0) {
427
+ const onDisk = [...listedTargets].filter((path) => sizes.has(path)).length;
428
+ out(`NOTE C9: twin scan over ${listedTargets.size} target(s), ${onDisk} on disk, 0 unlisted twins (byte-identical only; drifted copies need identity tests)`);
429
+ }
346
430
  }
347
431
  }
348
432
 
@@ -417,7 +501,21 @@ else for (const t of acidTokens) if (!new RegExp(`\\b${t.replace(/[.*+?^${}()|[\
417
501
  // bullet-less row was invisible here while being a row there (so it evaded this check), and
418
502
  // this side lacked the range guard, so «AM-1..AM-4 are covered elsewhere» was falsely
419
503
  // reported as a definition. One shape, one meaning.
420
- const isDef = /^\s*(?:[-*|]\s*)?\*{0,2}AM-(?:CP-)?\d+\*{0,2}\b(?!\s*\.)/.test(pl);
504
+ // ЯЧЕЙКА ТАБЛИЦЫ ВНЕ СЕКЦИИ — НЕ ОПРЕДЕЛЕНИЕ. Внутри `## Amendments` строка таблицы
505
+ // законно объявляет поправку, поэтому там `|` остаётся допустимым началом строки. Снаружи
506
+ // же таблица — это СВОДКА, ссылающаяся на поправки, определённые в другом месте, и читать
507
+ // её как определение значит отказывать плану за оглавление.
508
+ // ИЗМЕРЕНО 2026-09-05: четыре ложных FAIL за один день на четырёх разных фичах
509
+ // (narrated-error-must-be-taught, fa-phase-statusline, core-boundary-guard,
510
+ // run-registry-liveness); каждый стоил ручной правки ведущего и перезапуска конвейера,
511
+ // то есть 5–10 минут и один цикл роутера. Один планировщик дошёл до того, что вписал в
512
+ // план оговорку «no line of this paragraph may begin with AM-» — прибор начал
513
+ // диктовать людям форму прозы, а это уже не проверка, а суеверие.
514
+ // ОДИН механизм, а не два: `|` УБРАН из класса начал строки. Первая редакция этой правки
515
+ // добавляла ещё и отдельную проверку `isTableRow`, и мутация, снимавшая только её,
516
+ // оставалась ЭКВИВАЛЕНТНОЙ — 77 тестов из 77 зелёные при «снятой» починке. Дублирующая
517
+ // защита не усиливает гейт, она прячет от мутационной пробы, что именно держит свойство.
518
+ const isDef = /^\s*(?:[-*]\s*)?\*{0,2}AM-(?:CP-)?\d+\*{0,2}\b(?!\s*\.)/.test(pl);
421
519
  const inSection = sectionStart >= 0 && cursor2 >= sectionStart && cursor2 < sectionEnd;
422
520
  if (isDef && !inSection) {
423
521
  const tok = (/AM-(?:CP-)?\d+/.exec(pl) || ['AM-?'])[0];
@@ -479,7 +577,10 @@ else for (const t of acidTokens) if (!new RegExp(`\\b${t.replace(/[.*+?^${}()|[\
479
577
  }
480
578
  if (superseded) break;
481
579
  }
482
- if (!hasTest && !superseded) failures.push(`C6: ${defs[k].id} carries neither \`\u2192 test <name>\` nor \`superseded by AM-N\` — an amendment without a confirmation is a wish, and a retracted one must say its successor`);
580
+ // Сообщение называет ТУ ЖЕ форму, которую требует MARK выше. Прежняя редакция обещала
581
+ // простое `→ test <name>`, а проверка требовала имя и файл в обратных кавычках — человек,
582
+ // написавший ровно то, что просило сообщение, получал отказ снова и не понимал, почему.
583
+ if (!hasTest && !superseded) failures.push(`C6: ${defs[k].id} carries neither \`\u2192 test \`<имя>\` in \`<файл>\`\` (обратные кавычки обязательны, как и слово in) nor \`superseded by AM-N\` — an amendment without a confirmation is a wish, and a retracted one must say its successor`);
483
584
  }
484
585
  }
485
586
  }
@@ -16,6 +16,49 @@ export const meta = {
16
16
  ],
17
17
  }
18
18
 
19
+ // Self-contained mirror of harness-core/src/reqe.ts shouldEmitReqeDebt. TASK-3 executes both
20
+ // predicates against the shared cause table; keep branch order and decision text identical.
21
+ function reqeEmitDecision(input) {
22
+ const familyOf = function (spec) { return /codex|gpt|openai/i.test(String(spec ?? '')) ? 'openai' : 'claude' }
23
+ const coderFam = familyOf(input.coderUsed)
24
+ const qeFam = familyOf(input.qeReviewerUsed)
25
+ if (coderFam !== qeFam) {
26
+ return { emit: false, cause: null, degraded: false, reason: 'cross-family QE ran' }
27
+ }
28
+ if (input.routingRequested === false) {
29
+ return { emit: false, cause: null, degraded: false,
30
+ reason: 'same-family by configuration (cross-family never requested)' }
31
+ }
32
+ if (/\(usage-switched\)/.test(String(input.qeModelLabel ?? ''))) {
33
+ return {
34
+ emit: true, cause: 'usage-switched', degraded: false,
35
+ reason:
36
+ 'usage-switched self-review: Step-8 QE ran on the coder’s own family (' + coderFam +
37
+ ') under the limit override — the cross-model guard was suspended (FR-2.9)',
38
+ }
39
+ }
40
+ if (input.bridge?.rungState === 'probe-failed') {
41
+ return { emit: true, cause: 'probe-failed', degraded: false,
42
+ reason: 'codex probe found no usable id; review ran on the coder’s own family (' + coderFam + ')' }
43
+ }
44
+ if (input.bridge?.rungState === 'dispatched' || input.bridge?.rungState === 'refused-before-dispatch' ||
45
+ input.qeReviewerUsed === 'codex-fallback') {
46
+ return { emit: true, cause: 'same-family-fallback', degraded: false,
47
+ reason: 'same-family fallback: ' + (input.bridge?.decline ?? input.bridge?.rungReason ?? 'cross-family reviewer unavailable') +
48
+ '; review ran on the coder’s own family (' + coderFam + ')' }
49
+ }
50
+ if (input.bridge == null) {
51
+ return { emit: false, cause: null, degraded: true,
52
+ reason: "cause undeterminable (pre-change checkpoint) — re-run with resume:'never' to classify" }
53
+ }
54
+ if (input.routingRequested === true && input.bridge.rungState === 'pending') {
55
+ return { emit: true, cause: 'same-family-pinned', degraded: false,
56
+ reason: 'same-family by explicit qe pin (args.models.qe) while cross-family routing was requested' }
57
+ }
58
+ return { emit: false, cause: null, degraded: true,
59
+ reason: 'cause undeterminable (unknown rung state or routing request)' }
60
+ }
61
+
19
62
  // Finalize on every normal return and on a caught runtime error; a killed host cannot finalize.
20
63
  let finishRunRegistry = null
21
64
  let registryOutcome = 'errored'
@@ -26,7 +69,6 @@ const A = typeof args === 'string' ? JSON.parse(args) : (args || {})
26
69
  const SLUG = A.slug || 'feature'
27
70
  const DESC = A.description || ''
28
71
  const CODE_HINT = A.code || '(discover from the description)'
29
- const MODE = A.mode || 'full-qe-extended'
30
72
  const STOP_AFTER = A.stopAfter || null
31
73
  // fa-phase-statusline (ADR-001 D2): tier holder for the ckpt-side phase-start fa-records. The real
32
74
  // tier variable initializes only AFTER the router stage (TDZ — reading it from the router's own
@@ -207,6 +249,40 @@ function parseRoundCommandJson(raw) {
207
249
  return null
208
250
  }
209
251
 
252
+ // ablation-c-start (ADR-001 D6, fix-round-1 BLOCKER #2): the run's arm is NEVER accepted from a
253
+ // caller-supplied option — that let a run execute reference mode (or claim "direct" while actually
254
+ // running "reference") with zero journal entry, bypassing the lottery entirely. If args.experiment
255
+ // is given, the arm is RESOLVED here from the SAME assignments journal `dz experiment assign`
256
+ // already wrote to BEFORE this run was ever dispatched, via `dz experiment resolve` — never
257
+ // invented, never trusted from an option the caller could set to anything.
258
+ const EXPERIMENT = typeof A.experiment === 'string' ? A.experiment.trim() : ''
259
+ const EXPERIMENT_TASK_ID = typeof A.taskId === 'string' ? A.taskId.trim() : ''
260
+ // direct -> the pipeline's own default full pipeline; reference -> the reduced-QE arm. A caller
261
+ // that also passes args.mode must AGREE with what the resolved arm implies, or the run refuses
262
+ // (assignment-conflict) rather than silently letting the mode win over the arm.
263
+ const ABLATION_ARM_EXPECTED_MODE = { direct: 'full-qe-extended', reference: 'reference' }
264
+ let ABLATION_ARM = null
265
+ let ABLATION_PROPENSITY = null
266
+ if (EXPERIMENT !== '') {
267
+ if (EXPERIMENT_TASK_ID === '') {
268
+ throw new Error('feature-adr: args.experiment=' + JSON.stringify(EXPERIMENT) + ' but args.taskId is missing — the arm can only be RESOLVED from an existing assignment, never invented (ADR-001 D6, assignment-missing).')
269
+ }
270
+ const resolveCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' experiment resolve --experiment ' + shq(EXPERIMENT) + ' --task ' + shq(EXPERIMENT_TASK_ID) + ' --project ' + shq(REPO) + ' --json'
271
+ const resolveOut = await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and return its stdout VERBATIM, no commentary: ' + resolveCmd, { label: 'resolve-experiment-arm', phase: 'Route', model: 'haiku', effort: 'low' })
272
+ const resolved = parseRoundCommandJson(resolveOut)
273
+ if (!resolved || resolved.status !== 'resolved' || (resolved.arm !== 'direct' && resolved.arm !== 'reference')) {
274
+ const reason = (resolved && typeof resolved.status === 'string') ? resolved.status : 'assignment-missing'
275
+ throw new Error('feature-adr: could not resolve an assignment for task ' + JSON.stringify(EXPERIMENT_TASK_ID) + ' in experiment ' + JSON.stringify(EXPERIMENT) + ' (' + reason + ') — run `dz experiment assign` for this task BEFORE dispatching it (ADR-001 D6, assignment-missing).')
276
+ }
277
+ ABLATION_ARM = resolved.arm
278
+ ABLATION_PROPENSITY = (typeof resolved.propensity === 'number') ? resolved.propensity : null
279
+ const expectedMode = ABLATION_ARM_EXPECTED_MODE[ABLATION_ARM]
280
+ if (typeof A.mode === 'string' && A.mode !== '' && A.mode !== expectedMode) {
281
+ throw new Error('feature-adr: the resolved arm ' + JSON.stringify(ABLATION_ARM) + ' for task ' + JSON.stringify(EXPERIMENT_TASK_ID) + ' implies mode ' + JSON.stringify(expectedMode) + ', but args.mode=' + JSON.stringify(A.mode) + ' disagrees — refusing rather than letting one silently override the other (ADR-001 D6, assignment-conflict).')
282
+ }
283
+ }
284
+ const MODE = ABLATION_ARM !== null ? ABLATION_ARM_EXPECTED_MODE[ABLATION_ARM] : (A.mode || 'full-qe-extended')
285
+
210
286
  // ── Durable checkpoints + resume (backlog 49e4a95b) — inline mirror of ──
211
287
  // ── harness-core/src/feature-adr-checkpoints.ts (the workflow is self-contained, no imports) ──
212
288
  // After each expensive stage a cheap effort-low agent appends {stage, inputHash, result} to
@@ -226,7 +302,7 @@ const CKPT_FILE = FDIR + '/.fa-state/checkpoints.jsonl'
226
302
  // M10 Stage-A, feature loop-designer). This region is now a GENERATED BLOB (regen-diff-gated by
227
303
  // loop-blobs-regen.test.ts): edit the canonical TS FIRST, run node scripts/gen-loop-blobs.mjs,
228
304
  // then re-splice. The value-pinned wiring tests in feature-adr-checkpoints.test.ts stay the net.
229
- // ── BEGIN BLOB checkpoints@1.2.0 sha256:a44560c6036fd143b7a3f125fec00fa8b91c3c06ac5b9e1ecfb83d27a5211e9b src=packages/@dzhechkov/harness-core/src/feature-adr-checkpoints.ts ──
305
+ // ── BEGIN BLOB checkpoints@1.2.0 sha256:b602357649042d6b8aa68b5ca5593f7f5c483fdb36a287742e7b27bda8a2a6c7 src=packages/@dzhechkov/harness-core/src/feature-adr-checkpoints.ts ──
230
306
  const CHECKPOINT_STAGES = ['router', 'design', 'plan', 'code', 'qe', 'fleet'];
231
307
  const STAGE_ARTIFACTS = {
232
308
  router: '00_complexity_assessment.md',
@@ -300,12 +376,12 @@ function decideCheckpointResume(opts) {
300
376
  }
301
377
  return { resume: true, reason: 'resumed' };
302
378
  }
303
- function serializeCheckpoint(stage, inputHash, result) {
379
+ function serializeCheckpoint(stage, inputHash, result, ts) {
304
380
  if (result === null || result === undefined)
305
381
  return null;
306
382
  let line;
307
383
  try {
308
- line = JSON.stringify({ stage, inputHash, result });
384
+ line = JSON.stringify({ stage, inputHash, result, ...(ts === undefined ? {} : { ts }) });
309
385
  }
310
386
  catch {
311
387
  return null;
@@ -315,6 +391,7 @@ function serializeCheckpoint(stage, inputHash, result) {
315
391
  return line;
316
392
  }
317
393
  const CHECKPOINT_LS_SENTINEL = '---FA-CKPT-LS---';
394
+ const CHECKPOINT_TIMESTAMP = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{1,3})?Z$/;
318
395
  function parseCheckpointRead(text) {
319
396
  const out = { entries: {}, listing: new Set(), malformedLines: 0 };
320
397
  const raw = String(text ?? '');
@@ -329,7 +406,8 @@ function parseCheckpointRead(text) {
329
406
  try {
330
407
  const e = JSON.parse(t);
331
408
  if (e && typeof e === 'object' && typeof e.stage === 'string' && typeof e.inputHash === 'string' && 'result' in e && e.result !== null && e.result !== undefined) {
332
- out.entries[e.stage] = e;
409
+ const { ts, ...entry } = e;
410
+ out.entries[e.stage] = typeof ts === 'string' && CHECKPOINT_TIMESTAMP.test(ts) ? { ...entry, ts } : entry;
333
411
  }
334
412
  else {
335
413
  if (e && typeof e === 'object' && typeof e.stage === 'string')
@@ -360,7 +438,11 @@ function checkpointReadCmd(fdirAbs) {
360
438
  function checkpointAppendCmd(fdirAbs, line) {
361
439
  const dir = shellQuote(fdirAbs + '/.fa-state');
362
440
  const file = shellQuote(fdirAbs + '/.fa-state/checkpoints.jsonl');
363
- return 'mkdir -p ' + dir + " && printf '%s\\n' " + shellQuote(line) + ' >> ' + file;
441
+ const plain = 'mkdir -p ' + dir + " && printf '%s\\n' " + shellQuote(line) + ' >> ' + file;
442
+ if (typeof line !== 'string' || !line.startsWith('{'))
443
+ return plain;
444
+ return 'ts=$(date -u +%Y-%m-%dT%H:%M:%SZ); mkdir -p ' + dir
445
+ + ' && { printf \'{"ts":"%s",\' "$ts"; printf \'%s\\n\' ' + shellQuote(line.slice(1)) + '; } >> ' + file;
364
446
  }
365
447
  function parseArtifactProbe(opts) {
366
448
  if (opts.stdout === null || opts.stdout === undefined)
@@ -406,7 +488,7 @@ let CKPT_LISTING = new Set()
406
488
  const resumedStages = []
407
489
  async function loadCheckpoints(phaseName) {
408
490
  if (!CHECKPOINTS_ON) return
409
- const readCmd = 'cat ' + shq(CKPT_FILE) + ' 2>/dev/null || true; echo ' + shq(CKPT_LS_SENTINEL) + '; cd ' + shq(FDIR) + ' 2>/dev/null && find . -maxdepth 2 -type f 2>/dev/null | sed "s|^\\./||" || true'
491
+ const readCmd = checkpointReadCmd(FDIR)
410
492
  const readOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and return its stdout VERBATIM (it may be empty) with NO code fences and NO commentary: ' + readCmd, { label: 'ckpt:read', phase: phaseName, effort: 'low' })
411
493
  const raw = String(readOut == null ? '' : readOut)
412
494
  // LINE-ANCHORED sentinel: a sentinel string INSIDE a recorded result shares its line with JSON
@@ -893,6 +975,70 @@ function qeFindingsSummary(gaps) {
893
975
  }
894
976
  return { findings: gaps.length, findingsBySeverity: bySeverity, findingsSource: 'gaps' }
895
977
  }
978
+ // instrument-round-b T1/A1 (ADR-001 D1): a self-report can ONLY come from the QE agent's OWN return
979
+ // object, keyed to a NAMED source — never derived from MODE (mode is an intent the run was
980
+ // configured with, not an event that happened). A missing or non-boolean aqeInvoked is honestly
981
+ // 'not-reported', never guessed true/false from context. Pure, never throws.
982
+ // fix-round-1 (Codex r1 MEDIUM finding 6): `aqeEvidence` used to accept ANY nonempty string,
983
+ // including pure whitespace. Now TRIMMED; blank-after-trim keeps `aqeInvoked` (the agent DID answer
984
+ // the invoked/not-invoked question) but the evidence itself is honestly `null` with
985
+ // `aqeEvidenceStatus:'blank'`. A non-blank string is checked against a SOFT form (does it look like
986
+ // an MCP tool name or an `aqe ` command?) — a fabricated tool name is still indistinguishable from a
987
+ // truthful one (a named, honest limit, not a fixable gap: see the module README), so this can only
988
+ // catch the OBVIOUSLY wrong shape, never a convincing lie. The value is kept either way; only the
989
+ // status is marked 'unrecognized', never discarded.
990
+ function aqeSelfReport(qe) {
991
+ if (!qe || typeof qe !== 'object' || typeof qe.aqeInvoked !== 'boolean') {
992
+ return { aqeInvoked: null, aqeInvokedSource: 'not-reported', aqeEvidence: null }
993
+ }
994
+ const trimmed = typeof qe.aqeEvidence === 'string' ? qe.aqeEvidence.trim() : ''
995
+ if (trimmed === '') {
996
+ return { aqeInvoked: qe.aqeInvoked, aqeInvokedSource: 'qe-self-report', aqeEvidence: null, aqeEvidenceStatus: 'blank' }
997
+ }
998
+ // Lead delta after Codex r2 (#6 partial): a substring test called 'garbagemcp__agentic-qe__garbage'
999
+ // recognized evidence. The shape is now ANCHORED: the whole trimmed string is either an MCP tool
1000
+ // name (mcp__agentic-qe__<tool>) or an aqe command line starting with 'aqe '.
1001
+ const looksReal = /^mcp__agentic-qe__[a-z][a-z0-9_]*$/.test(trimmed) || /^aqe [\w-]/.test(trimmed)
1002
+ return {
1003
+ aqeInvoked: qe.aqeInvoked,
1004
+ aqeInvokedSource: 'qe-self-report',
1005
+ aqeEvidence: trimmed,
1006
+ aqeEvidenceStatus: looksReal ? 'recognized' : 'unrecognized',
1007
+ }
1008
+ }
1009
+ // instrument-round-b T2/A2/A7 (ADR-001 D2): `arm`/`propensity` are COPIES of
1010
+ // envelope.chosen.mode/envelope.policy.propensity, promoted to top-level ledger-row fields so a
1011
+ // consumer never has to parse the nested envelope object — copied at write time so the two can never
1012
+ // disagree by construction. No envelope (the router never completed) -> both null, armSource names
1013
+ // why. Pure, never throws.
1014
+ // fix-round-1 (Codex r1 HIGH finding 5): an envelope object present but carrying neither a valid
1015
+ // `chosen.mode` NOR a valid `policy.propensity` (e.g. `{}`) used to be labeled `armSource:'envelope'`
1016
+ // despite yielding two nulls — a validated-looking label on an unvalidated object. Now honestly
1017
+ // `'invalid-envelope'` unless AT LEAST one of the two fields actually resolved to a real value.
1018
+ function envelopeArmFields(envelope) {
1019
+ if (!envelope || typeof envelope !== 'object') return { arm: null, propensity: null, armSource: 'no-envelope' }
1020
+ const chosen = envelope.chosen
1021
+ const policy = envelope.policy
1022
+ const arm = (chosen && typeof chosen.mode === 'string' && chosen.mode !== '') ? chosen.mode : null
1023
+ const propensity = (policy && typeof policy.propensity === 'number') ? policy.propensity : null
1024
+ const armSource = (arm !== null || propensity !== null) ? 'envelope' : 'invalid-envelope'
1025
+ return { arm: arm, propensity: propensity, armSource: armSource }
1026
+ }
1027
+ // fix-round-1 (Codex r1 HIGH finding 5): keys a per-stage `extra` is FORBIDDEN from carrying —
1028
+ // each one is either derived from the SAME canonical ENVELOPE the row already stamps (arm,
1029
+ // propensity, armSource) or IS that canonical value (envelope) — stripped before the spread so an
1030
+ // `extra` that named one of these could never make the row's TOP-LEVEL fields disagree with its
1031
+ // NESTED envelope. Defense in depth: appendRunCostRow also restates `envelope: ENVELOPE` and
1032
+ // re-derives arm/propensity/armSource AFTER the (now-sanitized) spread.
1033
+ function sanitizeLedgerExtra(extra) {
1034
+ const RESERVED_LEDGER_EXTRA_KEYS = ['arm', 'propensity', 'armSource', 'envelope']
1035
+ if (!extra || typeof extra !== 'object') return {}
1036
+ const out = {}
1037
+ for (const k in extra) {
1038
+ if (Object.prototype.hasOwnProperty.call(extra, k) && RESERVED_LEDGER_EXTRA_KEYS.indexOf(k) === -1) out[k] = extra[k]
1039
+ }
1040
+ return out
1041
+ }
896
1042
  function tpText(v) { if (typeof v === 'string') return v; if (v === null || v === undefined) return ''; try { const s = JSON.stringify(v); return typeof s === 'string' ? s : String(v) } catch (e) { return String(v) } }
897
1043
  function tpBudget(raw) {
898
1044
  try {
@@ -1139,7 +1285,18 @@ async function appendRunCostRow(stage, phaseName, outcome, extra) {
1139
1285
  // the payload carries no runId), so passing it here is a strict reliability improvement.
1140
1286
  runId: (typeof RUN_ID === 'string' && RUN_ID !== '') ? RUN_ID : null,
1141
1287
  // aqe-ledger-row T1/NFR-1: additive-only — spreads nothing when `extra` is absent.
1142
- ...(extra && typeof extra === 'object' ? extra : {}),
1288
+ // fix-round-1 (Codex r1 HIGH finding 5): reserved keys are STRIPPED from `extra` first
1289
+ // (sanitizeLedgerExtra), which is what makes the single `envelope: ENVELOPE` above canonical —
1290
+ // nothing the spread carries can shadow it. Lead delta after Codex r2: the belt-and-braces
1291
+ // RESTATEMENT that used to sit here was removed — a second `envelope` key in the same literal
1292
+ // is invisible at runtime but visible to the key-order guard (feature-adr-training-pairs),
1293
+ // which reads the literal's keys, so the defense in depth read as a contract break.
1294
+ ...sanitizeLedgerExtra(extra),
1295
+ // instrument-round-b T2/FR-2/A2/A7 (ADR-001 D2): arm/propensity/armSource land AFTER the
1296
+ // per-stage `extra` spread — the last fields on every row, exactly like NFR-1 requires. Derived
1297
+ // from the SAME `ENVELOPE` just restated above, so the row's nested envelope and its top-level
1298
+ // arm/propensity/armSource can never disagree (Codex r1 finding 5).
1299
+ ...envelopeArmFields(ENVELOPE),
1143
1300
  })
1144
1301
  // WITNESSED WRITE (ADR-001): the subagent RUNS a command with data arguments; it is no longer
1145
1302
  // handed a shell pipeline with the row baked in. The command refuses a malformed row, stamps the
@@ -1147,8 +1304,18 @@ async function appendRunCostRow(stage, phaseName, outcome, extra) {
1147
1304
  // re-reading the tail. A courier could do none of those three.
1148
1305
  // fix-round-1/F2: --auto is the TRUSTED CLI-level marker (harness-core run-records.ts) — the
1149
1306
  // written row's auto:true no longer depends solely on the JSON payload remembering the field.
1307
+ // instrument-round-b FR-4/A5 (ADR-001 D4), fix-round-1 (Codex r1 HIGH finding 3): a NAMED,
1308
+ // SCOPED allowance — `--allow-incomplete tokens,minutes --incomplete-reason
1309
+ // sandbox-metrics-unavailable` — replaces the removed blanket `--no-strict`. This row's `tokens`/
1310
+ // `minutes` are LEGITIMATELY incomplete by construction, every single call: the sandbox has no
1311
+ // clock and budget.spent() exposes only a partial OUTPUT-token delta (tokensOut, a DIFFERENT
1312
+ // field than the `tokens`/`minutes` completeness checks), never the real total — that total is
1313
+ // visible only in the Workflow COMPLETION NOTIFICATION, which the running script cannot see (see
1314
+ // the HONESTY comment above the row literal). Naming exactly these two fields means if a THIRD
1315
+ // one (e.g. `envelope`) ever went missing too, the write would refuse rather than silently
1316
+ // widen what "legitimately incomplete" covers.
1150
1317
  const cmd = DZ + ' feature-adr-record --kind ledger --stage ' + shq(stage) + ' --project ' + shq(REPO)
1151
- + ' --row ' + shq(line) + ' --auto --json'
1318
+ + ' --row ' + shq(line) + ' --auto --allow-incomplete tokens,minutes --incomplete-reason sandbox-metrics-unavailable --json'
1152
1319
  const out = await dispatchAgent(newRung(), 'Run this command via your Bash tool and reply with only its stdout: ' + cmd, { label: 'ledger:append', phase: phaseName, effort: 'low' })
1153
1320
  const readback = String(out == null ? '' : out)
1154
1321
  if (!/"verdict"\s*:\s*"written"/.test(readback)) {
@@ -1638,7 +1805,14 @@ function buildExperimentEnvelopeInline(input) {
1638
1805
  // this mirror used to alias them directly, so a caller mutating its own arms.stages object
1639
1806
  // after calling this function would silently mutate the built envelope too.
1640
1807
  arms: { mode: input.arms.mode.slice(), stages: mergeOpts({}, input.arms.stages) },
1641
- chosen: { mode: input.chosen.mode, stages: mergeOpts({}, input.chosen.stages), overrides: mergeOpts({}, input.chosen.overrides === undefined ? {} : input.chosen.overrides) },
1808
+ chosen: {
1809
+ mode: input.chosen.mode,
1810
+ stages: mergeOpts({}, input.chosen.stages),
1811
+ overrides: mergeOpts({}, input.chosen.overrides === undefined ? {} : input.chosen.overrides),
1812
+ // ablation-c-start (ADR-001, T3): mirror of the core builder — carried through only
1813
+ // when the caller actually set it, so an unset qeMode never appears in the JSON.
1814
+ ...(input.chosen.qeMode !== undefined ? { qeMode: input.chosen.qeMode } : {}),
1815
+ },
1642
1816
  policy: { name: input.policy.name, version: input.policy.version, propensity: input.policy.propensity },
1643
1817
  evaluator: { family: input.evaluator.family, model: input.evaluator.model, source: input.evaluator.source },
1644
1818
  }
@@ -1748,6 +1922,12 @@ function validateExperimentEnvelopeInline(value) {
1748
1922
  return { ok: false, reason: 'chosen.stages.' + stage + ': "' + spec + '" is not a member of arms.stages.' + stage + ' (' + offered.join('|') + ') and not declared in chosen.overrides' }
1749
1923
  }
1750
1924
  }
1925
+ // ablation-c-start (ADR-001, T3): qeMode is OPTIONAL — absent on every run this feature
1926
+ // does not touch — but when present must be a non-empty string, same shape rule every
1927
+ // other envelope field gets (mirror of the core validator).
1928
+ if (chosen.qeMode !== undefined && !isNonEmptyStringEnv(chosen.qeMode)) {
1929
+ return { ok: false, reason: 'chosen.qeMode: expected a non-empty string when present' }
1930
+ }
1751
1931
  if (!isPlainObjectEnv(v.policy)) return { ok: false, reason: 'policy: expected an object' }
1752
1932
  const policy = v.policy
1753
1933
  if (!isNonEmptyStringEnv(policy.name)) return { ok: false, reason: 'policy.name: expected a non-empty string' }
@@ -3735,7 +3915,7 @@ const ARTIFACT = { type: 'object', additionalProperties: false, required: ['wrot
3735
3915
  // (hasManifest + the who-injected report). The BIG per-stage guidance content is fetched by each stage
3736
3916
  // agent directly from `dz project-skills` (never threaded through a model → fidelity preserved).
3737
3917
  const PROJECT_SKILLS = { type: 'object', additionalProperties: false, required: ['hasManifest', 'report'], properties: { hasManifest: { type: 'boolean' }, report: { type: 'string' } } }
3738
- const QE = { type: 'object', additionalProperties: false, required: ['grade', 'gaps', 'codeTestsAdequate', 'docTestsPresent'], properties: { grade: { type: 'string' }, codeTestsAdequate: { type: 'boolean' }, docTestsPresent: { type: 'boolean' }, gaps: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['sev', 'what'], properties: { sev: { type: 'string' }, what: { type: 'string' } } } }, claimCheck: { type: 'object', additionalProperties: false, properties: { findings: { type: 'number' }, high: { type: 'number' }, medium: { type: 'number' } } }, roundLessons: { type: 'array', items: { type: 'string' } }, roundNoNewKnowledge: { type: 'string' } } }
3918
+ const QE = { type: 'object', additionalProperties: false, required: ['grade', 'gaps', 'codeTestsAdequate', 'docTestsPresent'], properties: { grade: { type: 'string' }, codeTestsAdequate: { type: 'boolean' }, docTestsPresent: { type: 'boolean' }, gaps: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['sev', 'what'], properties: { sev: { type: 'string' }, what: { type: 'string' } } } }, claimCheck: { type: 'object', additionalProperties: false, properties: { findings: { type: 'number' }, high: { type: 'number' }, medium: { type: 'number' } } }, roundLessons: { type: 'array', items: { type: 'string' } }, roundNoNewKnowledge: { type: 'string' }, aqeInvoked: { type: 'boolean' }, aqeEvidence: { type: 'string' } } }
3739
3919
  const CONFIRMATION_FILE_GATE = { type: 'object', additionalProperties: false, required: ['verdict', 'missing', 'checked', 'reason'], properties: { verdict: { type: 'string', enum: ['pass', 'fail', 'skipped', 'refused'] }, missing: { type: 'array', items: { type: 'string' } }, checked: { type: 'array', items: { type: 'string' } }, reason: { type: 'string' } } }
3740
3920
 
3741
3921
  function normalizeConfirmationFileGate(raw) {
@@ -3783,7 +3963,7 @@ const MUTATION_GATE = 'MUTATION GATE (feature ha-mutation-gate — run alongside
3783
3963
  // same scan runs mechanically at publish time as the SOFT `no-stubs` guard rule.
3784
3964
  const STUB_RX = '(^|[^A-Za-z0-9_])(' + ['TO' + 'DO', 'FIX' + 'ME', 'HA' + 'CK', 'XX' + 'X', 'PLACE' + 'HOLDER'].join('|') + ')([^A-Za-z0-9_]|$)'
3785
3965
  const STUB_PHRASE = 'imple' + 'ment later'
3786
- const NO_STUBS_GATE = 'NO-STUBS GATE (backlog 0b403a0106103901 — layer 1 of the cost-of-detection ladder): over the files THIS RUN touched (the Step-7 change list; for a Codex coder, the landed-barrier file list), via Bash run EXACTLY `grep -nE \'' + STUB_RX + '\' <touched files>` (case-SENSITIVE — never add -i) plus `grep -niE \'' + STUB_PHRASE.replace(' ', '[[:space:]]+') + '\' <touched files>`. ANY match = the task shipped incomplete → HIGH gap naming file:line, UNLESS the line carries an inline `no-stubs: <reason>` waiver WITH a non-empty reason, or `.dz/guard.json` stubWaivers lists the path WITH a reason — a REASONLESS waiver is itself a HIGH gap, never an exemption. Cross-check mechanically: `dz guard check --op publish --json` runs the same scan as the SOFT `no-stubs` rule over the working-tree diff. When you QUOTE a marker in 08_qe_report.md, backtick it so the report itself scans clean (the same convention as the claim-check forbidden-phrase escape). Record the verdict in the 08_qe_report.md ADR Fitness section.'
3966
+ const NO_STUBS_GATE = 'NO-STUBS GATE (backlog 0b403a0106103901 — layer 1 of the cost-of-detection ladder): over the files THIS RUN touched (the Step-7 change list; for a Codex coder, the landed-barrier file list), via Bash run EXACTLY `grep -nE \'' + STUB_RX + '\' <touched files>` (case-SENSITIVE — never add -i) plus `grep -niE \'' + STUB_PHRASE.replace(' ', '[[:space:]]+') + '\' <touched files>`. ANY match = the task shipped incomplete → HIGH gap naming file:line, UNLESS the line carries an inline `no-stubs: <reason>` waiver WITH a non-empty reason, or `.dz/guard.json` stubWaivers lists the path WITH a reason — a REASONLESS waiver is itself a HIGH gap, never an exemption. Cross-check mechanically: `dz guard check --op publish --json` runs the same scan as the SOFT `no-stubs` rule over the working-tree diff. After the Step-7 code has landed, run `dz guard check --op code --json` and treat a HARD `block` verdict as a HIGH finding naming the drifted file. When you QUOTE a marker in 08_qe_report.md, backtick it so the report itself scans clean (the same convention as the claim-check forbidden-phrase escape). Record the verdict in the 08_qe_report.md ADR Fitness section.'
3787
3967
  const DISCRIMINATION_GATE = '\u00a742 TEST-DISCRIMINATION GATE (run right after asserting the property has a test): the ADR Confirmation names `Required automated check: <test file>` for the load-bearing property. Prove that test DISCRIMINATES \u2014 via Bash run EXACTLY `' + DZ + ' discrimination-check --test <that test file> --base HEAD --json` (the PINNED workspace bin, never bare `dz` — the global install measurably lags the workspace) (the Step-7 feature diff is UNCOMMITTED, so HEAD is the pre-feature base). Parse the JSON: read `perTest[]` (each row carries verdict + reason), `findings[]` (ALL entries, not only the first), `measurementValid`, and `primaryAction` \u2014 the singular `finding` is a DEPRECATED alias; do not consume it. The SEVEN verdicts and the required QE action for each: `DISCRIMINATES` (assertion-red at base, execution-evidenced) = PASS. `DISCRIMINATES_VIA_ERROR` (evidenced load-error at base + evidenced pass at tip) = PASS \u2014 note the inference. `NON_DISCRIMINATING` (evidenced pass at base \u2014 a proven false green) \u2192 HIGH gap "property test does not discriminate: <file>"; advisory, not an automatic blocker. `TEST_FILE_ABSENT` (the named test is not a regular file) \u2192 HIGH gap; action create-missing-test; NEVER a pass. `LOAD_ERROR_AT_BOTH_REVS` (the instrument could not execute the test at either rev \u2014 zero signal) \u2192 HIGH gap; action fix-runner-invocation. `FAILS_AT_TIP` (the feature\'s own test is red WITH the feature present) \u2192 HIGH gap; action fix-red-feature-test \u2014 grade the feature code accordingly. `CANNOT_ISOLATE` (no established observation; the row\'s `reason` is one of no-execution-evidence | unrecognised-runner-output | no-tests-executed | inconsistent-evidence | tip-control-missing | tip-evidence-missing | timeout) \u2192 HIGH gap NAMING the reason; action per `primaryAction` (map-a-test or fix-runner-invocation). `measurementValid` false or \'partial\' means the instrument did not (fully) measure \u2014 report it verbatim; never convert a degraded reading into a pass. Record every verdict + reason in the 08_qe_report.md ADR Fitness section. If `discrimination-check` is unavailable at the pinned path, errors, or overruns its window \u2192 record a HIGH gap `discrimination gate INCONCLUSIVE: <unavailable|error|timeout>` (backlog 52d0ed08: an instrument that could not run is never a pass and never applicable-by-silence). Still never abort the run.'
3788
3968
  // P2 (amendment-confirmation-discipline, fa-improvements 2026-07-18): amendments are where the SHARPEST design
3789
3969
  // corrections land (challenge-panel/QCSD) and were the least-tested — prose deltas with no proving test. Every
@@ -4078,8 +4258,17 @@ ENVELOPE = buildExperimentEnvelopeInline({
4078
4258
  treeSha: ENVELOPE_TREE_SHA,
4079
4259
  treeShaReason: ENVELOPE_TREE_SHA_REASON,
4080
4260
  arms: { mode: ['same-family', 'cross-family'], stages: envelopeArmsStages },
4081
- chosen: { mode: ENVELOPE_CHOSEN_MODE, stages: envelopeChosenStages, overrides: envelopeChosenOverrides },
4082
- policy: { name: 'routing-tables', version: POLICY_VERSION, propensity: null },
4261
+ // ablation-c-start (ADR-001, T3; fix-round-1 BLOCKER #2): a non-null ABLATION_ARM puts the arm
4262
+ // in chosen.qeMode (a field SEPARATE from chosen.mode's same-family/cross-family axis — mode
4263
+ // keeps its old meaning untouched) and renames the policy so a reader of the ledger's embedded
4264
+ // envelope can tell an ablation-c run from a routing-tables run at a glance. ABLATION_ARM and
4265
+ // ABLATION_PROPENSITY are RESOLVED above from the assignments journal via `dz experiment
4266
+ // resolve` — never taken from a caller-supplied option — so this line cannot be used to bypass
4267
+ // the lottery. ?? 0.5 covers only a missing/invalid propensity on the resolved record alongside
4268
+ // a present arm, never a missing arm. Without an arm, both lines stay byte-identical to before
4269
+ // this feature (NFR-2).
4270
+ chosen: { mode: ENVELOPE_CHOSEN_MODE, stages: envelopeChosenStages, overrides: envelopeChosenOverrides, ...(ABLATION_ARM ? { qeMode: ABLATION_ARM } : {}) },
4271
+ policy: ABLATION_ARM ? { name: 'ablation-c', version: POLICY_VERSION, propensity: (typeof ABLATION_PROPENSITY === 'number') ? ABLATION_PROPENSITY : 0.5 } : { name: 'routing-tables', version: POLICY_VERSION, propensity: null },
4083
4272
  evaluator: { family: ENVELOPE_QE_FAMILY, model: ENVELOPE_QE_SPEC, source: 'planned' },
4084
4273
  })
4085
4274
  // fix-round-1/F1: the router's training pair, captured here — AFTER ENVELOPE is real — so it gets
@@ -4766,7 +4955,17 @@ if (stopHere) {
4766
4955
  const rows = cpFindings.map((f, i) => '- AM-CP-' + (i + 1) + ' [' + f.severity + '] ' + String(f.title || '').replace(/[\r\n`]/g, ' ').slice(0, 160) + ' \u2192 test `названный кодером при реализации — заменить на имя реального теста` (panel ' + String(f.c || '') + ')').join('\n')
4767
4956
  const marker = '<!-- challenge-panel amendments appended ' + fnv1a64(rows) + ' -->'
4768
4957
  const planPath = FDIR + '/06_implementation_plan.md'
4769
- const appendCmd = 'cd ' + shq(REPO) + ' && grep -qF ' + shq(marker) + ' ' + shq(planPath) + ' && echo CP-DUP || { grep -q "^## Amendments" ' + shq(planPath) + ' || printf "\n## Amendments\n" >> ' + shq(planPath) + '; printf "%s\n%s\n" ' + shq(marker) + ' ' + shq(rows) + ' >> ' + shq(planPath) + '; echo CP-APPENDED; }'
4958
+ // The rows must land INSIDE `## Amendments`, not at EOF. MEASURED 2026-09-20 (backlog d48505cc,
4959
+ // which stood as an unverified inference until then): with `## Amendments` at line 13 and a later
4960
+ // `## Risks` at 17, the old `>> plan` put AM-CP-1 at line 21 and K2 C6 answered "AM-CP-1 is
4961
+ // DEFINED outside the `## Amendments` section". A plan whose Amendments section happens to be
4962
+ // last worked by luck, not by construction. The awk pass inserts before the NEXT `## ` heading
4963
+ // after the section, and falls back to EOF when the section IS last; the `a` flag is set AFTER
4964
+ // the heading test so the `## Amendments` line itself never triggers the insert. Idempotence is
4965
+ // unchanged: the marker grep still short-circuits to CP-DUP.
4966
+ const insertAwk = 'awk -v m=' + shq(marker) + ' -v r=' + shq(rows)
4967
+ + ' \'BEGIN{a=0;d=0} { if(a&&!d&&/^## /){print m; print r; d=1; a=0} if($0 ~ /^## Amendments/)a=1; print } END{if(a&&!d){print m; print r}}\' '
4968
+ const appendCmd = 'cd ' + shq(REPO) + ' && grep -qF ' + shq(marker) + ' ' + shq(planPath) + ' && echo CP-DUP || { grep -q "^## Amendments" ' + shq(planPath) + ' || printf "\n## Amendments\n" >> ' + shq(planPath) + '; ' + insertAwk + shq(planPath) + ' > ' + shq(planPath) + '.cp-tmp && mv ' + shq(planPath) + '.cp-tmp ' + shq(planPath) + '; echo CP-APPENDED; }'
4770
4969
  const cpOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and reply with ONLY its stdout: ' + appendCmd, { label: 'challenge:append-amendments', phase: 'Plan', effort: 'low' })
4771
4970
  log('challenge panel \u2192 plan amendments: ' + (/CP-APPENDED/.test(String(cpOut || '')) ? cpFindings.length + ' AM-CP row(s) appended' : /CP-DUP/.test(String(cpOut || '')) ? 'already appended (idempotent)' : 'NOT appended (probe answered: ' + String(cpOut || '').slice(0, 80) + ')'))
4772
4971
  }
@@ -5115,7 +5314,7 @@ const confirmationGateLine = confirmationFileGate.verdict === 'skipped'
5115
5314
  log(confirmationGateLine)
5116
5315
  const confirmationGateNote = ' MANDATORY CONFIRMATION FILE GATE RESULT: `' + confirmationGateLine + '`. Write that as a separate line in 08_qe_report.md. The independent QE review MUST still run. If the gate verdict is fail or refused, the final Step-8 grade cannot be A or B; the workflow also enforces that after the reviewer returns. This gate proves only existence/readability; all other ADR checklist items remain advisory.'
5117
5316
  await usageProbe('QE')
5118
- const qePrompt = 'Step 8 (QE - brutal-honesty review, agentic-qe) of /feature-adr for "' + DESC + '" (' + SLUG + '). Adversarially review the SHIPPED code (read it): correctness, edge cases, error handling, and the LOAD-BEARING property the ADR named (ASSERT it has a test that DISCRIMINATES - the recurring lesson: a test that would still pass with the protection deleted is documentation, not a gate). Run this ADR gate before final grading: ' + ADR_FITNESS_CHECKLIST + ' ' + DISCRIMINATION_GATE + ' ' + MUTATION_GATE + ' ' + NO_STUBS_GATE + ' ' + AMENDMENT_GATE + ' Grade A/B/C/D honestly. Assess code-test adequacy + doc-test presence. List CONFIRMED gaps with severity. Write ' + FDIR + '/08_qe_report.md with the primary findings under the exact heading `## Primary QE pass` and an ADR Fitness Checklist section showing PASS/FAIL per ADR and evidence for the Confirmation-linked test. MANDATORY SELF-LEARNING STORE (close the loop, never skip): compare every candidate lesson against the Step-0 recalled LEARNED patterns above. Teach ONLY lessons NOT covered by Step-0 recall. On overlap, run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate; if --reinforce is unavailable, skip the duplicate teach and report `reinforced existing pattern <id>` in the QE report. Store every genuinely new lesson in the CANONICAL BRAIN store at `' + BRAIN + '` so it is NOT lost to a target repo you may have cd`d into. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal) — the `cd ' + BRAIN + ' &&` prefix + `--project ' + BRAIN + '` pin guarantee the lesson lands in the brain regardless of your CWD. Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` (run it verbatim via Bash, do not skip). Do NOT teach trivia or invent gaps. In the return object set roundLessons to the teach:<id> receipts successfully written in this Step 8; when there were none, return roundLessons:[] and a non-empty roundNoNewKnowledge reason derived from this review/reinforcement decision. AUTHORING-TIME CLAIM-CHECK (Deliverable of claim-check-authoring-time): after writing ' + FDIR + '/08_qe_report.md, run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash, parse the {ok, findings, scanned} JSON, and report claimCheck: {findings: N, high: N, medium: N} (counts by severity) in your return object. TAG EVERY QUANTITATIVE CLAIM you write in the report using the convention the checker recognizes as honest — write "1131 tests pass (MEASURED — `npx vitest run`)", never a bare "1131 tests pass" — and where you QUOTE a forbidden phrase as an example (e.g. the retracted "100% passing" framing), backtick the literal so it reads as code, not an assertion, so your own compliant report scans clean. Return {grade, gaps, codeTestsAdequate, docTestsPresent, claimCheck, roundLessons, roundNoNewKnowledge}.' + ABSOLUTE_PATH_NOTE + FINDINGS_LEDGER_NOTE + landedNote + wqNote + confirmationGateNote + PS_GUIDANCE('qe')
5317
+ const qePrompt = 'Step 8 (QE - brutal-honesty review, agentic-qe) of /feature-adr for "' + DESC + '" (' + SLUG + '). Adversarially review the SHIPPED code (read it): correctness, edge cases, error handling, and the LOAD-BEARING property the ADR named (ASSERT it has a test that DISCRIMINATES - the recurring lesson: a test that would still pass with the protection deleted is documentation, not a gate). Run this ADR gate before final grading: ' + ADR_FITNESS_CHECKLIST + ' ' + DISCRIMINATION_GATE + ' ' + MUTATION_GATE + ' ' + NO_STUBS_GATE + ' ' + AMENDMENT_GATE + ' Grade A/B/C/D honestly. Assess code-test adequacy + doc-test presence. List CONFIRMED gaps with severity. Write ' + FDIR + '/08_qe_report.md with the primary findings under the exact heading `## Primary QE pass` and an ADR Fitness Checklist section showing PASS/FAIL per ADR and evidence for the Confirmation-linked test. MANDATORY SELF-LEARNING STORE (close the loop, never skip): compare every candidate lesson against the Step-0 recalled LEARNED patterns above. Teach ONLY lessons NOT covered by Step-0 recall. On overlap, run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate; if --reinforce is unavailable, skip the duplicate teach and report `reinforced existing pattern <id>` in the QE report. Store every genuinely new lesson in the CANONICAL BRAIN store at `' + BRAIN + '` so it is NOT lost to a target repo you may have cd`d into. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal) — the `cd ' + BRAIN + ' &&` prefix + `--project ' + BRAIN + '` pin guarantee the lesson lands in the brain regardless of your CWD. Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` (run it verbatim via Bash, do not skip). Do NOT teach trivia or invent gaps. In the return object set roundLessons to the teach:<id> receipts successfully written in this Step 8; when there were none, return roundLessons:[] and a non-empty roundNoNewKnowledge reason derived from this review/reinforcement decision. AUTHORING-TIME CLAIM-CHECK (Deliverable of claim-check-authoring-time): after writing ' + FDIR + '/08_qe_report.md, run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash, parse the {ok, findings, scanned} JSON, and report claimCheck: {findings: N, high: N, medium: N} (counts by severity) in your return object. TAG EVERY QUANTITATIVE CLAIM you write in the report using the convention the checker recognizes as honest — write "1131 tests pass (MEASURED — `npx vitest run`)", never a bare "1131 tests pass" — and where you QUOTE a forbidden phrase as an example (e.g. the retracted "100% passing" framing), backtick the literal so it reads as code, not an assertion, so your own compliant report scans clean. AQE SELF-REPORT (instrument-round-b T1/A1, ADR-001 D1): state honestly in your return object whether you actually invoked LIVE agentic-qe in THIS QE pass — aqeInvoked: true or false — and name aqeEvidence: the exact MCP tool you called (e.g. mcp__agentic-qe__quality_assess) or the exact command you ran (e.g. aqe quality assess). This is a SELF-REPORT of what YOU did, never an inference from the run MODE (' + MODE + ') — the mode names an intent this run was configured with, not proof that live agentic-qe was actually called this pass. If you did not call live agentic-qe this pass, say aqeInvoked:false and name what you used instead in aqeEvidence. Return {grade, gaps, codeTestsAdequate, docTestsPresent, claimCheck, roundLessons, roundNoNewKnowledge, aqeInvoked, aqeEvidence}.' + ABSOLUTE_PATH_NOTE + FINDINGS_LEDGER_NOTE + landedNote + wqNote + confirmationGateNote + PS_GUIDANCE('qe')
5119
5318
  // CROSS-MODEL QE (load-bearing): resolveStageModel('qe') derives the OTHER family than the resolved
5120
5319
  // coder when args.models.qe is unset (coder-codex ⇒ opus; coder-Claude ⇒ codex, or opus if codex absent).
5121
5320
  // An explicit args.models.qe wins. A Claude qe spec is merged onto the qe-code-reviewer base (role
@@ -5499,12 +5698,17 @@ if (qe !== null && qe2Spec !== null) {
5499
5698
  }
5500
5699
  }
5501
5700
  if (qe === null) return null
5502
- return { qe: qe, qeReviewerUsed: qeReviewerUsed, modelUsed: modelsUsed.qe, qe2: qe2, qe2ModelUsed: modelsUsed.qe2 || null }
5503
- }, { validate: function (r) { return !!(r && typeof r === 'object' && r.qe && typeof r.qe === 'object' && typeof r.qeReviewerUsed === 'string') } })
5701
+ return { qe: qe, qeReviewerUsed: qeReviewerUsed, modelUsed: modelsUsed.qe, qe2: qe2, qe2ModelUsed: modelsUsed.qe2 || null,
5702
+ bridge: { happened: crossFamilyQeReport ? crossFamilyQeReport.happened : null, rungState: qeCodexRung ? qeCodexRung.state : null, rungReason: qeCodexRung ? qeCodexRung.reason : null, decline: lastCodexDecline } }
5703
+ // Missing bridge is valid for pre-change checkpoints; the debt decision logs the degradation.
5704
+ }, { validate: function (r) { return !!(r && typeof r === 'object' && r.qe && typeof r.qe === 'object' && typeof r.qeReviewerUsed === 'string' && (r.bridge == null || (typeof r.bridge === 'object' && !Array.isArray(r.bridge)))) } })
5504
5705
  qe = qeStage ? qeStage.qe : null
5505
5706
  let qeReviewerUsed = qeStage ? qeStage.qeReviewerUsed : 'claude'
5506
5707
  if (qeStage && qeStage.modelUsed) modelsUsed.qe = qeStage.modelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
5507
5708
  if (qeStage && qeStage.qe2ModelUsed) modelsUsed.qe2 = qeStage.qe2ModelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
5709
+ if (qeStage && qeStage.bridge == null && resumedStages.indexOf('qe') !== -1) {
5710
+ log('re-QE debt: resumed pre-change checkpoint without bridge — only usage-switched can be classified from the model label')
5711
+ }
5508
5712
 
5509
5713
  // aqe-ledger-row T3/FR-1/FR-3/A1/A3/A4: the QE step's OWN autorow — who reviewed, how many
5510
5714
  // findings, cross-family or not — so the instrument's own footprint in the ledger stops being
@@ -5536,6 +5740,10 @@ if (resumedStages.indexOf('qe') === -1) {
5536
5740
  claimCheck: qeClaimCheck,
5537
5741
  crossFamily: qeCrossFamily,
5538
5742
  qeScope: qeScopeRow,
5743
+ // instrument-round-b T1/FR-1/A1 (ADR-001 D1): last fields on the qe row's own extra — the
5744
+ // agent's self-report of whether it actually invoked live agentic-qe this pass, never derived
5745
+ // from MODE.
5746
+ ...aqeSelfReport(qe),
5539
5747
  })
5540
5748
  } else {
5541
5749
  log('run-cost ledger: qe row skipped — stage resumed')
@@ -5576,19 +5784,17 @@ await capturePairs('code', 'QE', [{ input: codePrompt, output: codeStage, evalua
5576
5784
  await capturePairs('qe', 'QE', [{ input: qePrompt, output: qeStage ? qeStage.qe : null, evaluation: { grade: qe ? qe.grade : null, gradedBy: qeReviewerUsed, lessonsInjected: [] }, provenance: { model: String(modelsUsed.qe || ''), family: tpFamily(qeReviewerUsed), role: 'reviewer' } }])
5577
5785
 
5578
5786
  // ── re-QE debt emission (backlog 6b40e667) — mirror of harness-core/src/reqe.ts ──
5579
- // The cross-model guard was consciously SUSPENDED when the usage override made coder and QE the
5580
- // same family (FR-2.9). Record that as a machine DEBT (features/<slug>/.fa-state/reqe-due.json) so
5581
- // `dz reqe` / `dz usage` surface it after limits reset — a doc instruction on the weakest detection
5582
- // layer becomes a fact on disk. Emitted ONLY for the actual same-family-under-override case: a
5583
- // switch that kept cross-family QE, or the codex-unavailable Claude belt (no override), creates no
5584
- // debt. A RESUMED qe never re-emits (the original run emitted; a settlement must not be clobbered).
5787
+ // Record same-family reviews after usage switches, failed probes, fallbacks or explicit QE pins
5788
+ // as machine debt. Checkpoint-carried bridge facts preserve the cause on resume; routing OFF is
5789
+ // excluded explicitly. Missing old checkpoint facts degrade visibly. The guards below preserve
5790
+ // existing debt and this run's settlement when a resumed QE retries the write.
5585
5791
  let reqeDue = false
5586
5792
  {
5587
5793
  const reqeFamOf = function (s) { return /codex|gpt|openai/i.test(String(s || '')) ? 'openai' : 'claude' }
5588
- const qeLabel = String(modelsUsed.qe || '')
5589
- if (/\(usage-switched\)/.test(qeLabel) && reqeFamOf(coderUsed) === reqeFamOf(qeReviewerUsed)) {
5794
+ const reqeDecision = reqeEmitDecision({ coderUsed: coderUsed, qeReviewerUsed: qeReviewerUsed, qeModelLabel: modelsUsed.qe, routingRequested: routingRequested, bridge: qeStage ? qeStage.bridge : undefined })
5795
+ if (reqeDecision.emit) {
5590
5796
  reqeDue = true
5591
- const reqeDebt = { schema: 'reqe-due-1', slug: SLUG, coderFamily: reqeFamOf(coderUsed), qeFamily: reqeFamOf(qeReviewerUsed), qeGrade: (qe && qe.grade) ? String(qe.grade) : null, reason: 'usage-switched self-review: Step-8 QE ran on the coder’s own family under the limit override (FR-2.9)', emittedAt: null, runStamp: qeHash }
5797
+ const reqeDebt = { schema: 'reqe-due-1', slug: SLUG, coderFamily: reqeFamOf(coderUsed), qeFamily: reqeFamOf(qeReviewerUsed), qeGrade: (qe && qe.grade) ? String(qe.grade) : null, cause: reqeDecision.cause, bridge: qeStage ? qeStage.bridge : undefined, reason: reqeDecision.reason, emittedAt: null, runStamp: qeHash }
5592
5798
  // IDEMPOTENT + VERIFIED (reqe QE #2 + r2 #2/#3): a RESUMED qe re-runs this block (the original
5593
5799
  // run may have died between the qe checkpoint and this write), but: an existing due file is
5594
5800
  // never clobbered; a settlement blocks re-emission ONLY when it carries THIS run's runStamp (an
@@ -5602,8 +5808,10 @@ let reqeDue = false
5602
5808
  const emitText = String(emitOut || '')
5603
5809
  if (/REQE-SETTLED-THIS-RUN/.test(emitText)) log('re-QE debt: THIS run’s debt was already settled — not re-opened')
5604
5810
  else if (/REQE-EXISTS/.test(emitText)) log('re-QE debt: already recorded for ' + SLUG + ' — not overwritten')
5605
- else if (emitText.indexOf('reqe-due-1') !== -1) log('re-QE DEBT recorded: Step-8 ran same-family under the usage override — after limits reset run `dz reqe --slug ' + SLUG + '` for the independent cross-family pass')
5811
+ else if (emitText.indexOf('reqe-due-1') !== -1) log('re-QE DEBT recorded: Step-8 ran same-family; cause=' + reqeDecision.cause + ' — run `dz reqe --slug ' + SLUG + '` for the independent cross-family pass')
5606
5812
  else log('re-QE debt write UNVERIFIED (agent returned no readback) — the debt may be missing on disk; reqeDue=true is still reported, record it manually via features/' + SLUG + '/.fa-state/reqe-due.json')
5813
+ } else {
5814
+ log('re-QE debt: ' + (reqeDecision.degraded ? '' : 'none — ') + reqeDecision.reason)
5607
5815
  }
5608
5816
  }
5609
5817