@dzhechkov/skills-feature-adr 1.5.10 → 1.5.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.dz-manifest.json CHANGED
@@ -13,7 +13,7 @@
13
13
  },
14
14
  {
15
15
  "path": "README.md",
16
- "sha256": "b1dbcb7b1f0f3d95bdda9f9568f65e1f78b306b2cb64e42979b1981aeadc6047"
16
+ "sha256": "889d62c5dfafb53c0664867f0882e5712e47eb8aa7a44c6db46d082e693ed58d"
17
17
  },
18
18
  {
19
19
  "path": "bin/cli.js",
@@ -25,7 +25,7 @@
25
25
  },
26
26
  {
27
27
  "path": "package.json",
28
- "sha256": "e376fe73743953ad7f0b7525e22f815ea949450fe7ea784f89d1e61bcc62ce6f"
28
+ "sha256": "d13930ae60a1d849d08b798dbb3b3ed4fb5d0b50c4c3910fc76a455dd9e6ce31"
29
29
  },
30
30
  {
31
31
  "path": "src/cli.js",
@@ -249,7 +249,11 @@
249
249
  },
250
250
  {
251
251
  "path": "templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs",
252
- "sha256": "cd2d4e05d1b61be5485c9836abdebdb00b5f4085a14f3c71126cc45d4e08fe3b"
252
+ "sha256": "cbe70a524f6ce7987b5e3961ee412b92c24c00d84066d438c759659d656efaa2"
253
+ },
254
+ {
255
+ "path": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
256
+ "sha256": "82c3c48c25f400f3a64d1caba3d47357f3db87460f0746f88706f901eb6ff689"
253
257
  },
254
258
  {
255
259
  "path": "templates/.claude/skills/frontend-design/LICENSE.txt",
@@ -313,7 +317,7 @@
313
317
  },
314
318
  {
315
319
  "path": "templates/.claude/workflows/feature-adr.js",
316
- "sha256": "ef6a119466e504307116b12af3369029c13738ad6c40cc7a345d7d81834eca72"
320
+ "sha256": "22416f105c1b988c07f6cb9002a507b54b07842b5b4bb109b7e300ff52272603"
317
321
  },
318
322
  {
319
323
  "path": "templates/lib/memory-protocol.md",
@@ -325,5 +329,5 @@
325
329
  }
326
330
  ]
327
331
  },
328
- "signature": "2ltA45AgleqbKopq4ritI5XTO7YtiM/vDWxo6fNIFBYrlz+sz9ZAFSxyNOEN3KYs8uwlgmef1I1sb1wvMIgxAw=="
332
+ "signature": "WT7wArJY3aKuIMMVgbBzFO+Rqw7qfcaVh8UQ9MENSltqdTJTFAr/0TBOS5A89UAfczkluCrkTtyUvPOTQvaNCw=="
329
333
  }
package/README.md CHANGED
@@ -698,7 +698,7 @@ sufficiency + honesty, overengineering, silent decisions, runtime consistency, s
698
698
  `dz challenge --plan <plan.md>` or the `challenge-panel` skill; scaffold the degradations registry via
699
699
  `dz feature-adr-setup --from-spec <spec with {"degradations":true}> --apply`.
700
700
 
701
- ### ADR quality gate (Step 3 generates → Step 8 enforces)
701
+ ### ADR quality review + Confirmation file gate (Step 3 generates → Step 8 checks)
702
702
 
703
703
  Step 3 and Step 8 share an ADR best-practices contract distilled from the
704
704
  [architecture-decision-record monograph](https://github.com/architecture-decision-record/architecture-decision-record):
@@ -709,12 +709,17 @@ Step 3 and Step 8 share an ADR best-practices contract distilled from the
709
709
  links + after-action review, a **`## Confirmation`** stanza (method, monitoring, success metric, owner)
710
710
  naming the load-bearing property, and a **`## Links`** traceability block. Template weight is tier-mapped:
711
711
  S/M → Nygard/ITD-lightweight, L/XL → MADR + Confirmation.
712
- - **Step 8** runs a **13-point ADR fitness checklist** (`qe-code-reviewer`) against every generated ADR and
713
- **fails the gate** on any miss — decision-shaped title, controlled-vocabulary Status + reversibility,
712
+ - **Step 8** runs a **13-point advisory ADR fitness checklist** (`qe-code-reviewer`) against every generated ADR —
713
+ decision-shaped title, controlled-vocabulary Status + reversibility,
714
714
  neutral Context-before-Decision, symmetric options, driver-mapped rationale, concrete/testable decision,
715
715
  negative consequences, traceability links, no placeholder, and **rejects explainer-masquerading-as-ADR**.
716
- The **Confirmation→test link is load-bearing**: if the named safety property has no automated test the ADR
717
- grades no better than C.
716
+ Those judgment-based items remain findings; they do not independently force the workflow verdict.
717
+ - The **one mandatory gate** runs after Step 7.5 and before the QE verdict: every test-file path named
718
+ under an ADR heading beginning with `## Confirmation` must exist as a readable regular file. Missing
719
+ files or unreadable/unparseable paths force a non-passing Step-8 grade while the independent QE review
720
+ still runs. A feature with no ADR prints `пропущено: ADR нет, проверять нечего` and is not failed.
721
+ `dz discrimination-check` and `dz mutation-gate` remain advisory: existence does not prove that a test
722
+ actually discriminates the load-bearing property.
718
723
 
719
724
  The pipeline **dog-foods** this: a harness test runs the gate against feature-adr's own generated ADR, so a
720
725
  Step-3↔Step-8 drift fails CI rather than shipping.
@@ -1211,3 +1216,11 @@ Also in this release, both halves of the K2 plan-completeness gate that field us
1211
1216
  `dispatchOutcomes` отчёта. Причина отказа остаётся у outcome отказавшей ступени, а intent следующего
1212
1217
  fallback получает нейтральную причину `fallback-rung`. Регион `stage-line` принадлежит генератору
1213
1218
  `gen-loop-blobs` и не правится руками.
1219
+
1220
+ ### Shared Markdown masking in feature-adr gates
1221
+
1222
+ The standalone plan-completeness gate ships with `markdown-masker.mjs`, copied byte-for-byte from
1223
+ harness-core's `src/markdown-masker.ts`. It runs without a core build. Amendment checks, swarm briefs
1224
+ and K2 share the parser while retaining their existing unclosed-block and indentation policies.
1225
+ The four-space indented-code gap remains open for amendment checks and K2; swarm briefs retain their
1226
+ existing masking of indented code. Versions are unchanged in this staged change.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dzhechkov/skills-feature-adr",
3
- "version": "1.5.10",
3
+ "version": "1.5.11",
4
4
  "description": "Adaptive Feature Development skill pack for Claude Code — 11-step pipeline with Complexity Router (S/M/L/XL), ADR-driven architecture, 15 agentic-qe skills, multi-agent fleet QE. Supports --full-qe, --full-qe-extended, --with-learning, and --knowledge-extractor modes.",
5
5
  "bin": {
6
6
  "skills-feature-adr": "./bin/cli.js"
package/sbom.json CHANGED
@@ -35,7 +35,7 @@
35
35
  "hashes": [
36
36
  {
37
37
  "alg": "SHA-256",
38
- "content": "b1dbcb7b1f0f3d95bdda9f9568f65e1f78b306b2cb64e42979b1981aeadc6047"
38
+ "content": "889d62c5dfafb53c0664867f0882e5712e47eb8aa7a44c6db46d082e693ed58d"
39
39
  }
40
40
  ]
41
41
  },
@@ -69,7 +69,7 @@
69
69
  },
70
70
  {
71
71
  "name": "dz:canonical-json-sha256-v2",
72
- "value": "e376fe73743953ad7f0b7525e22f815ea949450fe7ea784f89d1e61bcc62ce6f"
72
+ "value": "d13930ae60a1d849d08b798dbb3b3ed4fb5d0b50c4c3910fc76a455dd9e6ce31"
73
73
  }
74
74
  ]
75
75
  },
@@ -629,7 +629,17 @@
629
629
  "hashes": [
630
630
  {
631
631
  "alg": "SHA-256",
632
- "content": "cd2d4e05d1b61be5485c9836abdebdb00b5f4085a14f3c71126cc45d4e08fe3b"
632
+ "content": "cbe70a524f6ce7987b5e3961ee412b92c24c00d84066d438c759659d656efaa2"
633
+ }
634
+ ]
635
+ },
636
+ {
637
+ "type": "file",
638
+ "name": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
639
+ "hashes": [
640
+ {
641
+ "alg": "SHA-256",
642
+ "content": "82c3c48c25f400f3a64d1caba3d47357f3db87460f0746f88706f901eb6ff689"
633
643
  }
634
644
  ]
635
645
  },
@@ -789,7 +799,7 @@
789
799
  "hashes": [
790
800
  {
791
801
  "alg": "SHA-256",
792
- "content": "ef6a119466e504307116b12af3369029c13738ad6c40cc7a345d7d81834eca72"
802
+ "content": "22416f105c1b988c07f6cb9002a507b54b07842b5b4bb109b7e300ff52272603"
793
803
  }
794
804
  ]
795
805
  },
@@ -51,6 +51,7 @@
51
51
  // the feature's own `00_complexity_assessment.md` acid-case table (rows shaped `| A<N> | … |`), or
52
52
  // supplied explicitly with `--acid=T1,T2,…`. If neither establishes a corpus, C4 is SKIPPED-with-note
53
53
  // (a feature that declared no acid cases cannot be failed for not naming them).
54
+ import { maskMarkdown } from './markdown-masker.mjs';
54
55
  import { readFileSync, readdirSync, existsSync } from 'node:fs';
55
56
  import { isAbsolute, join, resolve } from 'node:path';
56
57
 
@@ -319,21 +320,7 @@ else for (const t of acidTokens) if (!new RegExp(`\\b${t.replace(/[.*+?^${}()|[\
319
320
  // `## Amendments` could become the section heading, and a fenced example row could either open a
320
321
  // phantom amendment or hand a real testless one someone else's marker. Third fence-blindness
321
322
  // found in a checker today, so it is closed here by construction rather than by care.
322
- const rawLines = plan.split('\n');
323
- const planLines = [];
324
- {
325
- let fence = null;
326
- for (const line of rawLines) {
327
- const open = /^ {0,3}(```+|~~~+)/.exec(line);
328
- if (fence === null && open) { fence = open[1][0]; planLines.push(''); continue; }
329
- if (fence !== null) {
330
- planLines.push('');
331
- if (new RegExp('^ {0,3}' + fence + '{3,}\\s*$').test(line)) fence = null;
332
- continue;
333
- }
334
- planLines.push(line);
335
- }
336
- }
323
+ const planLines = maskMarkdown(plan, { unclosed: 'mask' }).split('\n');
337
324
  let sectionStart = -1, sectionEnd = -1, cursor = 0;
338
325
  for (const pl of planLines) {
339
326
  // The SAME heading shape amendment-trace.ts accepts: up to three leading spaces, two to four
@@ -0,0 +1,109 @@
1
+ /**
2
+ * Canonical Markdown block masker; copied byte-for-byte as markdown-masker.mjs beside K2.
3
+ * Keep this file valid JavaScript (inferred TS types, no build needed by the copy).
4
+ *
5
+ * CommonMark fences and type-2 HTML comments; same UTF-16 length and newline positions.
6
+ * Inline code cannot open an HTML block. HTML delimiters within a block are consumed left to
7
+ * right, including close/reopen on one line. This is not a complete CommonMark parser.
8
+ *
9
+ * Reader policies are deliberate: amendment-trace restores unclosed blocks; brief/K2 hide them.
10
+ * Four-space code is still unsupported by default. ONLY brief keeps its pre-existing policy.
11
+ * Containers, tab indentation and HTML block types other than comments remain unsupported.
12
+ * Callbacks expose line facts; callers own semantic diagnostics and list-barrier representation.
13
+ */
14
+ export function maskMarkdown(md = '', {
15
+ unclosed = 'restore',
16
+ indentedCode = false,
17
+ inlineComments = false,
18
+ onMasked = (_line = 0) => {},
19
+ onDisputed = (_line = 0) => {},
20
+ } = {}) {
21
+ const lines = md.split('\n');
22
+ const out = lines.slice();
23
+ const masked = lines.map(() => false);
24
+ let state = 'text';
25
+ let marker = '';
26
+ let markerLength = 0;
27
+ let openedAt = -1;
28
+ let nestedOpener = false;
29
+ let disputed = false;
30
+ for (let i = 0; i < lines.length; i++) {
31
+ const line = lines[i] ?? '';
32
+ const blank = () => { out[i] = ' '.repeat(line.length); masked[i] = true; };
33
+ if (state === 'fence') {
34
+ blank();
35
+ const close = /^ {0,3}(`{3,}|~{3,})\s*$/.exec(line);
36
+ if (close) {
37
+ const run = close[1] ?? '';
38
+ if (run[0] === marker && run.length >= markerLength) {
39
+ state = 'text';
40
+ openedAt = -1;
41
+ }
42
+ }
43
+ continue;
44
+ }
45
+ if (state === 'text') {
46
+ const fence = /^ {0,3}(`{3,}|~{3,})([^\n]*)$/.exec(line);
47
+ if (fence && !(fence[1]?.[0] === '`' && fence[2]?.includes('`'))) {
48
+ const run = fence[1] ?? '';
49
+ state = 'fence';
50
+ marker = run[0] ?? '';
51
+ markerLength = run.length;
52
+ openedAt = i;
53
+ blank();
54
+ continue;
55
+ }
56
+ if (indentedCode && /^ {4,}/.test(line)) {
57
+ blank();
58
+ continue;
59
+ }
60
+ }
61
+ // Outside HTML only a block opener counts: code-spanned delimiters in prose are inert.
62
+ // Inside HTML, backticks are literal and do not protect a delimiter.
63
+ const htmlStart = /^ {0,3}<!--/.test(line);
64
+ const wasDisputed = disputed;
65
+ // Ambiguity is diagnostic state, never permission to scan otherwise-visible prose.
66
+ // Brief deliberately keeps that diagnostic across blocks until the author's outer close;
67
+ // block-only readers must not hide an unrelated arrow or opener because it is pending.
68
+ if (state === 'html' || htmlStart || inlineComments) {
69
+ let maskLine = state === 'html' || htmlStart;
70
+ const tokens = [...line.matchAll(/`+|<!--|-->/g)];
71
+ for (let t = 0; t < tokens.length; t++) {
72
+ const token = tokens[t];
73
+ if (!token) continue;
74
+ if (token[0][0] === '`') {
75
+ // Code spans require an EXACT matching run; a shorter embedded run is literal.
76
+ // HTML blocks treat backticks literally. Inline-comment support is brief's existing
77
+ // policy; other readers only enter this scan at a block opener or while disputed.
78
+ if (state !== 'html') {
79
+ const end = tokens.findIndex((next, j) => j > t && next[0] === token[0]);
80
+ if (end >= 0) t = end;
81
+ }
82
+ continue;
83
+ }
84
+ if (token[0] === '<!--') {
85
+ if (state === 'html') nestedOpener = true;
86
+ else { state = 'html'; openedAt = i; }
87
+ maskLine = true;
88
+ } else if (state === 'html') {
89
+ state = 'text';
90
+ openedAt = -1;
91
+ if (nestedOpener) { disputed = true; nestedOpener = false; }
92
+ } else if (disputed) {
93
+ disputed = false;
94
+ maskLine = true;
95
+ }
96
+ }
97
+ if (maskLine) blank();
98
+ }
99
+ if (wasDisputed || disputed) onDisputed(i);
100
+ }
101
+ if (openedAt >= 0 && unclosed === 'restore') {
102
+ for (let i = openedAt; i < lines.length; i++) {
103
+ out[i] = lines[i] ?? '';
104
+ masked[i] = false;
105
+ }
106
+ }
107
+ for (let i = 0; i < masked.length; i++) if (masked[i]) onMasked(i);
108
+ return out.join('\n');
109
+ }
@@ -16,6 +16,12 @@ export const meta = {
16
16
  ],
17
17
  }
18
18
 
19
+ // Finalize on every normal return and on a caught runtime error; a killed host cannot finalize.
20
+ let finishRunRegistry = null
21
+ let registryOutcome = 'errored'
22
+ try {
23
+ const RUNTIME_SPENT = (typeof budget === 'object' && budget && typeof budget.spent === 'function') ? function () { return budget.spent() } : null
24
+ let spentAtPrevRow = RUNTIME_SPENT ? RUNTIME_SPENT() : null
19
25
  const A = typeof args === 'string' ? JSON.parse(args) : (args || {})
20
26
  const SLUG = A.slug || 'feature'
21
27
  const DESC = A.description || ''
@@ -135,15 +141,35 @@ function normalizeDzBin(raw, ws) {
135
141
  const base = (typeof ws === 'string' && ws.length > 0) ? ws.replace(/\/+$/, '') : ''
136
142
  return base === '' ? r : base + '/' + r
137
143
  }
138
- const DZ_RAW = A.dzBin || 'dz'
139
- // The probe above runs only when args.repo needed resolving; a relative dzBin needs WS too, so run
140
- // it here in exactly that case — a bare `dz` (the common case) pays zero extra agent calls.
144
+ // In the hub, PATH may name an older published CLI. Resolve the workspace first so an omitted
145
+ // args.dzBin can prefer the build that belongs to this checkout.
146
+ if (A.workspace !== undefined && A.workspace !== null) WS = assertAbsoluteNoTraversal(A.workspace, 'workspace')
147
+ if ((A.dzBin === undefined || A.dzBin === null || A.dzBin === '') && WS === null) {
148
+ WS = await probeSessionCwd('resolve-workspace-dz')
149
+ }
150
+ const WORKSPACE_DZ = WS === null ? null : WS.replace(/\/+$/, '') + '/packages/@dzhechkov/harness-cli/dist/bin.js'
151
+ let DZ_RAW = A.dzBin
152
+ let DZ_VERSION = 'not probed (dzBin given)'
153
+ if (DZ_RAW === undefined || DZ_RAW === null || DZ_RAW === '') {
154
+ const dzProbeCmd = WORKSPACE_DZ === null
155
+ ? "printf 'DZ_PATH_FALLBACK '; dz --version"
156
+ : 'if [ -f ' + shq(WORKSPACE_DZ) + " ]; then printf 'DZ_WORKSPACE_BUILD '; " + shq(WORKSPACE_DZ) + " --version; else printf 'DZ_PATH_FALLBACK '; dz --version; fi"
157
+ const localDzOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and reply with ONLY its stdout: ' + dzProbeCmd, { label: 'resolve-workspace-dz:select-version', phase: 'Route', model: 'haiku', effort: 'low' })
158
+ const localDzText = String(localDzOut || '').trim()
159
+ const workspaceDzExists = localDzText.indexOf('DZ_WORKSPACE_BUILD ') === 0
160
+ DZ_RAW = workspaceDzExists ? WORKSPACE_DZ : 'dz'
161
+ const dzVersionText = localDzText.replace(/^DZ_(?:WORKSPACE_BUILD|PATH_FALLBACK)\s*/, '')
162
+ const dzVersionLines = dzVersionText.split(/\r?\n/).map(function (line) { return line.trim() }).filter(Boolean)
163
+ DZ_VERSION = dzVersionLines.length > 0 ? dzVersionLines[dzVersionLines.length - 1].slice(0, 80) : 'unknown'
164
+ }
165
+ // A relative dzBin still needs WS so it resolves identically after every downstream cd. The combined
166
+ // selection/version probe above runs only when dzBin is absent; a supplied absolute dzBin costs no call.
141
167
  if (DZ_RAW.indexOf('/') >= 0 && DZ_RAW.charAt(0) !== '/' && WS === null) {
142
168
  WS = await probeSessionCwd('resolve-ws')
143
169
  if (!WS) throw new Error('feature-adr: args.dzBin ' + JSON.stringify(DZ_RAW) + ' is relative and the workspace root could not be resolved after 2 attempts; refusing to splice a path that would resolve differently in every cd\'d command. Pass an absolute args.dzBin.')
144
170
  }
145
171
  const DZ = normalizeDzBin(DZ_RAW, WS)
146
- log('dz binary: ' + DZ)
172
+ log('dz binary: ' + DZ + ' (' + DZ_VERSION + ')')
147
173
  // args.gateScript — an explicit ABSOLUTE path to the K2 gate script (ADR-002 candidate 1). Validated
148
174
  // HERE, at invocation time, so a bad value fails at the same layer the pure half fails rather than
149
175
  // two layers later inside an emitted shell command.
@@ -152,7 +178,6 @@ const GATE_SCRIPT_ARG = (A.gateScript === undefined || A.gateScript === null) ?
152
178
  // shell fallback WS=$(pwd -P) runs in the GATE AGENT own cwd. On a run against an external repo it
153
179
  // equalled REPO, so the workspace candidate pointed at the target repo and the skill installed in
154
180
  // the workspace was never found - NOT-ESTABLISHED, exit 3, Step 7 never ran. args.workspace pins it.
155
- if (A.workspace !== undefined && A.workspace !== null) WS = assertAbsoluteNoTraversal(A.workspace, 'workspace')
156
181
  // CANONICAL BRAIN store: the self-learning loop (Step-0 recall → Step-8 teach) MUST read+write ONE
157
182
  // shared pattern store so lessons never fragment into a target repo's .dz when the Step-7 coder cd's
158
183
  // away. BRAIN defaults to the workspace root (REPO) — so an OMITTED args.brain is behaviorally inert
@@ -170,6 +195,14 @@ const DZ_RECALL = (terms) => 'cd ' + BRAIN + ' && ' + DZ + ' recall "' + terms +
170
195
  const DZ_TEACH = (lesson, reward, domain) =>
171
196
  'cd ' + BRAIN + ' && ' + DZ + ' teach "' + lesson + '" --reward ' + reward + ' --domain ' + domain + ' --project ' + BRAIN
172
197
 
198
+ function parseRoundCommandJson(raw) {
199
+ const lines = String(raw === null || raw === undefined ? '' : raw).split(/\r?\n/).map(function (line) { return line.trim() }).filter(Boolean)
200
+ for (let i = lines.length - 1; i >= 0; i--) {
201
+ try { return JSON.parse(lines[i]) } catch { /* a chatty shell courier may add non-JSON lines */ }
202
+ }
203
+ return null
204
+ }
205
+
173
206
  // ── Durable checkpoints + resume (backlog 49e4a95b) — inline mirror of ──
174
207
  // ── harness-core/src/feature-adr-checkpoints.ts (the workflow is self-contained, no imports) ──
175
208
  // After each expensive stage a cheap effort-low agent appends {stage, inputHash, result} to
@@ -435,7 +468,11 @@ async function withCheckpoint(stage, phaseName, inputHash, runFn, ckptOpts) {
435
468
  else if (stage === 'fleet') faNext = 'done'
436
469
  }
437
470
  const faTier = (stage === 'router' && result && typeof result.tier === 'string') ? result.tier : FA_TIER.v
438
- const faRecordCmd = faNext === null ? null : (DZ + ' statusline --fa-record --slug ' + shq(SLUG) + ' --step ' + shq(faNext) + (faTier ? ' --tier ' + shq(faTier) : '') + ' --mode ' + shq(MODE) + ' --project ' + shq(REPO))
471
+ const faRunIdFile = FDIR + '/.fa-state/run-id'
472
+ const faMintRunId = stage === 'router'
473
+ ? ('mkdir -p ' + shq(FDIR + '/.fa-state') + ' 2>/dev/null || true; date +%s%3N > ' + shq(faRunIdFile) + ' 2>/dev/null || true; ')
474
+ : ''
475
+ const faRecordCmd = faNext === null ? null : (faMintRunId + DZ + ' statusline --fa-record --slug ' + shq(SLUG) + ' --step ' + shq(faNext) + (faTier ? ' --tier ' + shq(faTier) : '') + ' --mode ' + shq(MODE) + ' --project ' + shq(REPO) + ' --run-id "$(cat ' + shq(faRunIdFile) + ' 2>/dev/null || true)"')
439
476
  var faReported = false
440
477
  if (CHECKPOINTS_ON && result !== null && result !== undefined && !partial && persistable) {
441
478
  let line = null
@@ -970,7 +1007,14 @@ async function appendRunCostRow(stage, phaseName, outcome) {
970
1007
  // Like capturePairs, a ledger failure is a logged SECONDARY event that can NEVER fail the run;
971
1008
  // the whole body therefore rides one best-effort try/catch and never rethrows.
972
1009
  try {
973
- // HONESTY: tokens and minutes and agents come from the Workflow COMPLETION NOTIFICATION, which the running script CANNOT see. So the automated row MUST write null for them — never an estimate, never a guess, never a fabricated number. The operator still enriches tokens/minutes afterwards.
1010
+ const spentNow = RUNTIME_SPENT ? RUNTIME_SPENT() : null
1011
+ const tokensOut = Number.isFinite(spentNow) && Number.isFinite(spentAtPrevRow)
1012
+ ? spentNow - spentAtPrevRow
1013
+ : null
1014
+ if (Number.isFinite(spentNow)) spentAtPrevRow = spentNow
1015
+ // HONESTY: budget.spent() exposes only cumulative OUTPUT tokens, so tokensOut is that measured
1016
+ // partial cost. Complete token cost, minutes and agents come from the Workflow COMPLETION NOTIFICATION,
1017
+ // which the running script CANNOT see. Keep their fields null: never an estimate, never a guess, never a fabricated number.
974
1018
  const line = JSON.stringify({
975
1019
  slug: (typeof SLUG === 'string' && SLUG !== '') ? SLUG : null,
976
1020
  stage: (typeof stage === 'string' && stage !== '') ? stage : null,
@@ -978,6 +1022,9 @@ async function appendRunCostRow(stage, phaseName, outcome) {
978
1022
  tokens: null,
979
1023
  minutes: null,
980
1024
  agents: null,
1025
+ tokensOut: tokensOut,
1026
+ tokensOutSource: tokensOut === null ? 'unavailable' : 'budget.spent',
1027
+ costNote: 'total tokens visible only in the completion notification; minutes not measurable inside the sandbox',
981
1028
  coder: (typeof coderUsed === 'string' && coderUsed !== '') ? coderUsed : null,
982
1029
  grade: (qe && typeof qe.grade === 'string' && qe.grade !== '') ? qe.grade : null,
983
1030
  outcome: (typeof outcome === 'string' && outcome !== '') ? outcome : null,
@@ -1098,9 +1145,9 @@ function stageReason(stage, decision) {
1098
1145
  if (decision.reason === 'explicit-models' && AUTOCOST[stage]) return 'auto-cost'
1099
1146
  return decision.reason
1100
1147
  }
1101
- const KNOWN_CODEX = { 'auto': 1, 'gpt-5.5': 1, 'gpt-5.6': 1, 'gpt-5.6-luna': 1, 'gpt-5.6-terra': 1, 'gpt-5.6-sol': 1 }
1148
+ const KNOWN_CODEX = { 'auto': 1, 'gpt-5.5': 1, 'gpt-5.6': 1, 'gpt-5.6-luna': 1, 'gpt-5.6-terra': 1, 'gpt-5.6-sol': 1, 'gpt-6-astra': 1 }
1102
1149
  // The allowlist is not an availability check — probe every id before every run; ids drift in both directions.
1103
- const CODEX_TIERS = { flagship: 'gpt-5.6-sol', workhorse: 'gpt-5.6-terra', 'high-volume': 'gpt-5.6-luna' }
1150
+ const CODEX_TIERS = { premium: 'gpt-6-astra', flagship: 'gpt-5.6-sol', workhorse: 'gpt-5.6-terra', 'high-volume': 'gpt-5.6-luna' }
1104
1151
  const CLAUDE_NAMES = { fable: 1, opus: 1, sonnet: 1, haiku: 1 }
1105
1152
  const VALID_REASONING = { none: 1, minimal: 1, low: 1, medium: 1, high: 1, xhigh: 1, max: 1 }
1106
1153
  const DEFAULT_MODELS = { router: 'fable', requirements: 'sonnet', research: 'sonnet', adr: 'opus', ideation: 'sonnet', ddd: 'opus', architecture: 'opus', plan: 'sonnet', code: null, qe: null, fleet: 'sonnet' }
@@ -1238,9 +1285,26 @@ function budgetTable(primary, mode) {
1238
1285
  codexHalf = { ...ROUTING_TABLES.claude.codex[mode.codex], qe: A.codexAvailable === false ? 'opus' : qeSpec }
1239
1286
  } else {
1240
1287
  const normal = mode.codex === 'normal'
1241
- const id = codexIdForTier(normal ? 'flagship' : 'workhorse')
1242
- const design = 'codex:' + id + ':' + (normal ? 'high' : 'medium')
1243
- codexHalf = { requirements: design, research: design, adr: design, ideation: design, ddd: design, architecture: design, plan: 'codex:' + id + ':' + (normal ? 'high' : 'low'), code: 'codex:' + id + ':medium' }
1288
+ // Тир берётся из FA_TIER.v, а НЕ из A.tier: A.tier несёт только ЯВНО переданный аргумент,
1289
+ // а обычный прогон узнаёт свой размер от нулевого шага (роутера), который кладёт его сюда
1290
+ // на строке FA_TIER.v = tier. Читая A.tier, ячейка плана никогда не поднималась до
1291
+ // премиального яруса в обычном прогоне — матрица работала наполовину (ИЗМЕРЕНО 2026-09-09).
1292
+ const largePlan = FA_TIER.v === 'L' || FA_TIER.v === 'XL'
1293
+ const work = 'codex:' + codexIdForTier(normal ? 'flagship' : 'workhorse') + ':high'
1294
+ const evidence = 'codex:' + codexIdForTier(normal ? 'workhorse' : 'high-volume') + ':medium'
1295
+ codexHalf = {
1296
+ router: evidence,
1297
+ requirements: 'codex:' + codexIdForTier(normal ? 'flagship' : 'workhorse') + ':medium',
1298
+ research: evidence,
1299
+ adr: 'codex:' + codexIdForTier(normal ? 'premium' : 'flagship') + ':high',
1300
+ ideation: work,
1301
+ ddd: work,
1302
+ architecture: 'codex:' + codexIdForTier(normal ? 'premium' : 'flagship') + ':high',
1303
+ plan: largePlan ? 'codex:' + codexIdForTier(normal ? 'premium' : 'flagship') + ':high' : work,
1304
+ code: work,
1305
+ fleet: work,
1306
+ }
1307
+
1244
1308
  }
1245
1309
  return { ...claudeHalf, ...codexHalf }
1246
1310
  }
@@ -2237,17 +2301,18 @@ const CODE_LANDING_CEILING_ENV = 'DZ_FEATURE_ADR_CODE_LANDING_CEILING_MS'
2237
2301
  const CODEX_COMPANION_SCRIPT = '/root/.claude/plugins/cache/openai-codex/codex/1.0.5/scripts/codex-companion.mjs'
2238
2302
  const CODEX_COMPANION_STATE_ROOT = '/root/.claude/plugins/data/codex-openai-codex/state'
2239
2303
 
2240
- function decideCodeLandingLiveness(input) {
2304
+ function decideCodeLandingLiveness(input, pidProbe) {
2241
2305
  const status = typeof input.companionStatus === 'string' ? input.companionStatus.trim().toLowerCase() : ''
2242
2306
  const elapsedMs = Number.isFinite(input.elapsedMs) ? Math.max(0, input.elapsedMs) : 0
2243
2307
  const ceilingMs = Number.isFinite(input.ceilingMs) && input.ceilingMs > 0 ? input.ceilingMs : DEFAULT_CODE_LANDING_CEILING_MS
2308
+ const recordedPidAlive = input.recordedPidAlive === undefined ? (input.recordedPid === undefined ? null : pidProbe(input.recordedPid)) : input.recordedPidAlive
2244
2309
  const live = status === 'running' || status === 'queued'
2245
2310
  const terminal = status === 'completed' || status === 'failed' || status === 'cancelled'
2246
2311
 
2247
- if (live && input.recordedPidAlive === false) {
2312
+ if (live && recordedPidAlive === false) {
2248
2313
  return { verdict: 'dead-worker', reason: 'recorded-pid-absent' }
2249
2314
  }
2250
- if (live && input.recordedPidAlive === true) {
2315
+ if (live && recordedPidAlive === true) {
2251
2316
  if (elapsedMs >= ceilingMs) return { verdict: 'inconclusive', reason: 'ceiling-exceeded' }
2252
2317
  return { verdict: 'coder-running', reason: 'recorded-pid-alive' }
2253
2318
  }
@@ -2753,7 +2818,8 @@ function codeLandingLivenessProbeCmd(repo, plan, baselineAbsPath, jobId, waitSec
2753
2818
  // cannot answer) — a different event from an expired window, and the only one with a cure.
2754
2819
  // Unreadable stays 'unknown', which the verdict treats as no evidence, never as a clean exit.
2755
2820
  'tf=unknown; if [ -f "$state" ]; then if grep -q \'"touchedFiles": *\\[ *\\]\' "$state"; then tf=0; elif grep -q \'"touchedFiles"\' "$state"; then tf=1; fi; fi; ' +
2756
- 'if [ -n "$pid" ]; then if ps -p "$pid" -o pid= >/dev/null 2>&1; then pid_alive=true; else pid_alive=false; fi; fi; fi; fi; ' +
2821
+ // Replaces ps -p "$pid": the CLI uses the shared probePid; inaccessible stays unknown.
2822
+ 'if [ -n "$pid" ]; then pid_alive=$(' + DZ + ' runs --probe-pid "$pid"); case "$pid_alive" in true|false|unknown) :;; *) pid_alive=unknown;; esac; fi; fi; fi; ' +
2757
2823
  'echo "CODEX-LIVENESS-SIGNAL companion=$companion pid-alive=$pid_alive targets-changed=$target elapsed-ms=$elapsed_ms ceiling-ms=$ceiling_ms start-ms=$start_ms touched-files=$tf"; ' +
2758
2824
  'printf "%s\n" "$landing" | head -40; rm -rf "$sc"'
2759
2825
  )
@@ -3056,7 +3122,34 @@ const ARTIFACT = { type: 'object', additionalProperties: false, required: ['wrot
3056
3122
  // (hasManifest + the who-injected report). The BIG per-stage guidance content is fetched by each stage
3057
3123
  // agent directly from `dz project-skills` (never threaded through a model → fidelity preserved).
3058
3124
  const PROJECT_SKILLS = { type: 'object', additionalProperties: false, required: ['hasManifest', 'report'], properties: { hasManifest: { type: 'boolean' }, report: { type: 'string' } } }
3059
- const QE = { type: 'object', additionalProperties: false, required: ['grade', 'gaps', 'codeTestsAdequate', 'docTestsPresent'], properties: { grade: { type: 'string' }, codeTestsAdequate: { type: 'boolean' }, docTestsPresent: { type: 'boolean' }, gaps: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['sev', 'what'], properties: { sev: { type: 'string' }, what: { type: 'string' } } } }, claimCheck: { type: 'object', additionalProperties: false, properties: { findings: { type: 'number' }, high: { type: 'number' }, medium: { type: 'number' } } } } }
3125
+ const QE = { type: 'object', additionalProperties: false, required: ['grade', 'gaps', 'codeTestsAdequate', 'docTestsPresent'], properties: { grade: { type: 'string' }, codeTestsAdequate: { type: 'boolean' }, docTestsPresent: { type: 'boolean' }, gaps: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['sev', 'what'], properties: { sev: { type: 'string' }, what: { type: 'string' } } } }, claimCheck: { type: 'object', additionalProperties: false, properties: { findings: { type: 'number' }, high: { type: 'number' }, medium: { type: 'number' } } }, roundLessons: { type: 'array', items: { type: 'string' } }, roundNoNewKnowledge: { type: 'string' } } }
3126
+ const CONFIRMATION_FILE_GATE = { type: 'object', additionalProperties: false, required: ['verdict', 'missing', 'checked', 'reason'], properties: { verdict: { type: 'string', enum: ['pass', 'fail', 'skipped', 'refused'] }, missing: { type: 'array', items: { type: 'string' } }, checked: { type: 'array', items: { type: 'string' } }, reason: { type: 'string' } } }
3127
+
3128
+ function normalizeConfirmationFileGate(raw) {
3129
+ if (!raw || typeof raw !== 'object') return { verdict: 'refused', missing: [], checked: [], reason: 'confirmation-file gate agent returned no readable result' }
3130
+ const verdict = raw.verdict
3131
+ if (verdict !== 'pass' && verdict !== 'fail' && verdict !== 'skipped' && verdict !== 'refused') return { verdict: 'refused', missing: [], checked: [], reason: 'confirmation-file gate returned an invalid verdict' }
3132
+ return {
3133
+ verdict: verdict,
3134
+ missing: Array.isArray(raw.missing) ? raw.missing.map(function (x) { return String(x) }) : [],
3135
+ checked: Array.isArray(raw.checked) ? raw.checked.map(function (x) { return String(x) }) : [],
3136
+ reason: typeof raw.reason === 'string' ? raw.reason : ''
3137
+ }
3138
+ }
3139
+
3140
+ function enforceConfirmationFileGate(qe, gate) {
3141
+ if (!qe || typeof qe !== 'object') return qe
3142
+ const out = Object.assign({}, qe, { confirmationFileGate: gate })
3143
+ if (gate.verdict !== 'fail' && gate.verdict !== 'refused') return out
3144
+ const what = gate.verdict === 'fail'
3145
+ ? 'confirmation file gate FAIL — missing: ' + gate.missing.join(', ')
3146
+ : 'confirmation file gate REFUSED — ' + gate.reason
3147
+ const gaps = Array.isArray(out.gaps) ? out.gaps.slice() : []
3148
+ if (!gaps.some(function (g) { return g && g.what === what })) gaps.push({ sev: 'HIGH', what: what })
3149
+ out.gaps = gaps
3150
+ out.grade = String(out.grade || '').trim().toUpperCase() === 'D' ? 'D' : 'C'
3151
+ return out
3152
+ }
3060
3153
  const ADR_TEMPLATE_GUIDE = 'ADR best-practices for Step 3: emit exactly one decision per ADR with the invariant core Title, Status, Context, Decision, Consequences. Template weight is tier-routed: S/M use Nygard/ITD-lightweight form but still include decision drivers, considered options, rationale, consequences, and Confirmation; L/XL use MADR structure plus an NHS Wales Confirmation stanza. Confirmation MUST name verification method, monitoring, success metric, and owner, and its load-bearing safety property MUST be tied to a Step-8 automated test/fitness function. Use status vocabulary proposed/accepted/rejected/deprecated/superseded plus a reversibility clause. Context must be neutral and appear before Decision. Considered Options must include rejected options with symmetric pros/cons. Rationale points must map to stated drivers and explain why losers were rejected. Consequences must include positive and negative outcomes/accepted downsides, follow-up ADR links, an after-action review schedule, and supersession discipline: supersession mints a new ADR and never edits accepted/rejected ADR content in place. Decision must be concrete/testable with exact names, versions, formats, paths, commands, or APIs. Reject explainer-masquerading-as-ADR: a domain overview with no concrete Decision is not an ADR. File names under 03_adr MUST be sequential NNN-{decision-slug}.md with lowercase kebab-case, dateless, ticketless slugs (the auto-001 ADR tracks the feature slug, so the present-tense imperative signal lives in the ADR Title; model-named additional ADRs use imperative slugs). Add a ## Links traceability block (requirements, driving use case, related ADRs) and a one-line provenance note (model-generated, edited for clarity); for a long ADR include a top-of-file table of contents.'
3061
3154
  const ADR_FITNESS_CHECKLIST = 'ADR fitness checklist for Step 8: read every ' + FDIR + '/03_adr/NNN-*.md ADR and fail the QE gate for any miss. Required checks: (1) filename is 03_adr/NNN-{decision-slug}.md where the slug is lowercase kebab-case, imperative, dateless, and ticketless; (2) title is decision-shaped and the ADR records one decision only; (3) Status is non-empty controlled vocabulary proposed/accepted/rejected/deprecated/superseded and includes a reversibility/revisit clause; (4) Context is neutral, problem-first, and appears before Decision; (5) Decision Drivers are stated and ranked/weighted; (6) Considered Options include the chosen and rejected options, each with symmetric pros and cons; (7) Rationale maps each point back to a driver and explains why rejected options lost; (8) Decision is concrete/testable with exact names, versions, formats, paths, commands, or APIs; (9) Consequences include positive and negative outcomes/accepted downsides, follow-up ADR links, and an after-action review schedule; (10) Confirmation names verification method, monitoring, success metric, and owner, then links the load-bearing safety property to an automated test/fitness function; (11) no placeholder text, template hints, raw generation scaffolding, or fake Markdown structure; (12) reject explainer-masquerading-as-ADR: describing a space with no concrete Decision is a blocker; (13) a Related/Links traceability block maps the ADR to its requirements, driving use case, and related ADRs. The ADR Confirmation check is load-bearing: assert the named safety property has a test that DISCRIMINATES — a real test by file/name that would go RED if the protection were deleted (the discrimination + mutation gates below are the proof; a test that would still pass with the protection deleted is documentation, not a gate); if absent, grade no better than C and record a blocker gap.'
3062
3155
  // §42 test-discrimination gate (feature step8-discrimination-gate, grounded in cve-bench/evaluate.mjs). Asserting
@@ -3088,8 +3181,55 @@ const AMENDMENT_RULE = 'AMENDMENT CONFIRMATION DISCIPLINE (every amendment is a
3088
3181
  const AMENDMENT_GATE = 'AMENDMENT GATE (P2): do NOT judge this yourself — RUN the check and report what it says. Via Bash run EXACTLY `' + DZ + ' amendment-check --slug ' + SLUG + ' --json` (add `--feature-dir ' + FDIR + '` if the slug does not resolve from your CWD). Parse the JSON and report `amendments: {outcome, counts, reasons}` in your return object. outcome `pass` or `skip` clears the gate; `fail` is a HIGH gap and every reason must be quoted verbatim into the QE report; `not-established` means the check could not be run or the grammar matched nothing — that is NEVER a pass, report it as inconclusive with the tool error. Empty stdout, a crash, or a missing `dz` is `not-established`, not a clean gate. This check proves each amendment RESOLVES to a real test; it does NOT prove the test discriminates — vacuity stays with the discrimination gate above. ' +
3089
3182
  'IO-ON-PURE-PATH + FIXTURE-SWAP HUNT (P5): in the test diff, hunt for replacements of broken/unbound fixtures with healthy ones — the old fixture was probably a NEGATIVE CONTROL proving a path was I/O-free; each such swap requires a compensating negative resource-down test. If the code diff adds I/O (DB/network/file) to a previously-pure path — especially startup/lifespan/health — require a negative resource-down test (broken/unbound resource → the path degrades per its declared contract: fail-open for advisory, explicit fail-fast for load-bearing). Missing → HIGH gap.'
3090
3183
 
3184
+ // ── BEGIN BLOB run-registry@1.0.0 sha256:a4187c278260e8f8285fc494da2a6d82923efe4465fabba20ce1d3141df7fef3 src=packages/@dzhechkov/harness-core/src/run-registry.ts ──
3185
+ function runRecordCommand(dz, root, event, runId, slug, pid, parentRunId, outcome) {
3186
+ const quote = (s) => "'" + s.replace(/'/g, "'\\''") + "'";
3187
+ let cmd = dz + ' runs-record --project ' + quote(root) + ' --event ' + quote(event) + (runId ? ' --run-id ' + quote(runId) : '');
3188
+ if (event === 'started') {
3189
+ cmd += ' --kind feature-adr --slug ' + quote(slug);
3190
+ cmd += ' --pid ' + quote(pid === null ? 'host' : String(pid));
3191
+ if (parentRunId)
3192
+ cmd += ' --parent-run-id ' + quote(parentRunId);
3193
+ }
3194
+ if (event === 'finished')
3195
+ cmd += ' --outcome ' + quote(outcome);
3196
+ return cmd + ' --json';
3197
+ }
3198
+ // ── END BLOB run-registry ──
3199
+
3200
+ // Run registry: the courier executes the CLI because the sandbox has no filesystem.
3201
+ let registryRunId = ''
3202
+ let registryPhase = 'Router'
3203
+ async function recordRegistryEvent(event, phaseName, outcome) {
3204
+ registryPhase = phaseName
3205
+ if (event !== 'started' && !registryRunId) return
3206
+ try {
3207
+ const cmd = runRecordCommand(DZ, REPO, event, registryRunId, SLUG,
3208
+ A.runPid === undefined ? null : A.runPid, A.parentRunId || null, outcome || '')
3209
+ const out = await dispatchAgent(newRung(), 'Run EXACTLY this command via Bash and return ONLY its stdout: ' + cmd,
3210
+ { label: 'runs-record:' + event + ':' + phaseName, phase: phaseName, effort: 'low' })
3211
+ let receipt = null
3212
+ try { receipt = typeof out === 'string' ? JSON.parse(out.trim()) : out } catch { /* unverified below */ }
3213
+ if (!receipt || receipt.status !== 'written' || receipt.event !== event ||
3214
+ typeof receipt.runId !== 'string' || !receipt.runId || (registryRunId && receipt.runId !== registryRunId)) {
3215
+ registryOutcome = 'unverified'
3216
+ log('run registry: ' + event + ' UNVERIFIED — ' + (receipt && receipt.reason ? String(receipt.reason) : String(out)))
3217
+ return
3218
+ }
3219
+ if (event === 'started') registryRunId = receipt.runId
3220
+ } catch (error) {
3221
+ registryOutcome = 'unverified'
3222
+ log('run registry: ' + event + ' UNVERIFIED — ' + String(error))
3223
+ }
3224
+ }
3225
+ finishRunRegistry = async function () {
3226
+ await recordRegistryEvent('finished', registryPhase, registryOutcome)
3227
+ }
3228
+ await recordRegistryEvent('started', 'Router')
3229
+
3091
3230
  // Step 0: Router + MANDATORY self-learning recall
3092
3231
  phase('Router')
3232
+ await recordRegistryEvent('heartbeat', 'Router')
3093
3233
 
3094
3234
  // W1 (backlog 848853a0): REPO must be the git TOPLEVEL. Both measured incidents were a REPO
3095
3235
  // pointing INSIDE the repository (packages/@dzhechkov/health-advisor) — artifacts then scatter
@@ -3110,6 +3250,7 @@ else if (wrootTop !== wrootHere) {
3110
3250
  log('REPO ROOT MISMATCH: REPO canonicalizes to ' + wrootHere + ' but the git toplevel is ' + wrootTop + ' — refusing before any design spend (the measured incident class: artifacts scattered into a subdirectory features/)')
3111
3251
  const repoRootMismatchGates = {}
3112
3252
  const repoRootMismatchOutcome = runOutcomeOf({ phase: 'repo-root-mismatch', gates: repoRootMismatchGates })
3253
+ if (registryOutcome !== 'unverified') registryOutcome = repoRootMismatchOutcome
3113
3254
  return { phase: 'repo-root-mismatch', outcome: repoRootMismatchOutcome, repo: REPO, repoCanonical: wrootHere, gitToplevel: wrootTop, cure: 'invoke with args.repo=' + wrootTop + ' (or run from the repository root)' }
3114
3255
  }
3115
3256
  await loadCheckpoints('Router')
@@ -3148,6 +3289,8 @@ FA_TIER.v = tier // fa-phase-statusline: from here every ckpt-side fa-record car
3148
3289
  // Outer completion state starts absent so the plan-only ledger row can report null honestly.
3149
3290
  let coderUsed = null
3150
3291
  let qe = null
3292
+ let pipelineRound = null
3293
+ let roundClosed = false
3151
3294
  const LEARNED = router ? router.rationale : 'none recalled'
3152
3295
  const isMplus = tier === 'M' || tier === 'L' || tier === 'XL'
3153
3296
  const isLplus = tier === 'L' || tier === 'XL'
@@ -3191,6 +3334,26 @@ if (autoCostStages.length > 0) {
3191
3334
  // most visible moment. Uses the workspace bin (PATH-independent). Best-effort — never blocks.
3192
3335
  if (resumedStages.indexOf('router') === -1) await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and report its stdout verbatim — do nothing else, do not summarize: ' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 0 recall" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored 0 --mode ' + MODE + ' --project ' + REPO, { label: 'fa-record:step0', phase: 'Router', effort: 'low' })
3193
3336
 
3337
+ // A whole feature-adr run is one outer round. Cost remains in the unchanged per-stage ledger rows;
3338
+ // the round records only the outcome. State lives in REPO because the command cd's there, while
3339
+ // recall reads the canonical BRAIN and carries the same fa:<slug> attribution as DZ_RECALL.
3340
+ const roundOwnerArg = registryRunId
3341
+ ? ' --owner-run ' + shq(registryRunId)
3342
+ : ''
3343
+ if (!registryRunId) log('round owner: no registry run id (registry write failed) — falling back to explicit owner')
3344
+ const roundOpenCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' round open --slug ' + shq(SLUG) + ' --round auto --topic ' + shq(DESC) + ' --project ' + shq(BRAIN) + ' --run fa:' + SLUG + roundOwnerArg + ' --json'
3345
+ const roundOpenOut = await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and return its stdout VERBATIM, nothing else: ' + roundOpenCmd, { label: 'round:open', phase: 'Router', effort: 'low' })
3346
+ const roundOpenReceipt = parseRoundCommandJson(roundOpenOut)
3347
+ pipelineRound = Number(roundOpenReceipt && roundOpenReceipt.state ? roundOpenReceipt.state.round : (roundOpenReceipt ? roundOpenReceipt.round : NaN))
3348
+ if (!Number.isInteger(pipelineRound) || pipelineRound < 1) {
3349
+ pipelineRound = null
3350
+ log('round open refused: ' + (roundOpenReceipt && roundOpenReceipt.message ? roundOpenReceipt.message : String(roundOpenOut || 'no JSON receipt')))
3351
+ } else if (!roundOpenReceipt.state) {
3352
+ // A resumed run sees the still-open state and receives the selected auto number in the refusal.
3353
+ // Keep that number so Step 8 can close the original round, but never call the refusal an open.
3354
+ log('round open refused: ' + String(roundOpenReceipt.message || 'round already open'))
3355
+ }
3356
+
3194
3357
  // R1 product-architecture-lens (ADR-001 Decision 3): forward-looking сверка of THIS feature vs the LIVE
3195
3358
  // product map + vision. NON-BLOCKING/soft by design — it LOGS {signal,confidence} so a real command
3196
3359
  // duplication or vision-boundary tension is visible at Step 0; the hard-stop call stays the user's (a
@@ -3241,6 +3404,7 @@ const PS_GUIDANCE = (stage) => POLY.hasManifest
3241
3404
 
3242
3405
  // Steps 1-5: Design (tier-gated thunks built explicitly - no inline ternary-null)
3243
3406
  phase('Design')
3407
+ await recordRegistryEvent('heartbeat', 'Design')
3244
3408
  await usageProbe('Design')
3245
3409
  const designThunks = []
3246
3410
  const reqExtra = isLplus ? ' Also write ' + FDIR + '/02_research.md (codebase patterns + external analogues; read the repo for the closest existing implementation to mirror).' : ''
@@ -3482,6 +3646,7 @@ if (!fanVerdict.complete) {
3482
3646
  // (coderUsed/qe are the outer bindings, both still null here, so the row reports null honestly.)
3483
3647
  const designIncompleteGates = { design: fanVerdict.reason === 'probe-not-established' ? 'not-established' : 'incomplete', plan: 'not-run', planCompleteness: 'not-run', challengePanel: 'not-run', code: 'not-run', qe: 'not-run' }
3484
3648
  const designIncompleteOutcome = runOutcomeOf({ phase: 'design-incomplete', gates: designIncompleteGates })
3649
+ if (registryOutcome !== 'unverified') registryOutcome = designIncompleteOutcome
3485
3650
  await appendRunCostRow('design-gate', 'Design', designIncompleteOutcome)
3486
3651
  return { tier: tier, phase: 'design-incomplete', outcome: designIncompleteOutcome, slug: SLUG, artifactsDir: FDIR, missingSubstages: fanVerdict.missingSubstages, missingArtifacts: fanVerdict.missingArtifacts, reason: fanVerdict.reason, modelsUsed: modelsUsed, dispatchOutcomes: dispatchOutcomes, gates: designIncompleteGates, resumedStages: resumedStages, checkpointing: CHECKPOINTS_ON ? RESUME_MODE : 'off', trainingPairs: CAPTURE_PAIRS ? TP_DIR : 'off', captureFailures: captureFailures, recordFailures: recordFailures, decisionRecallFailures: decisionRecallFailures, usageEvents: usageEvents, usageThreshold: USAGE_THRESHOLD, polymorphism: POLY.hasManifest ? POLY.report : null, note: 'REFUSED at the Step-5/6 boundary: ' + what + ', so the design is incomplete and Step 6 was NOT dispatched. Planning off a partial design produces a plan with no ADR behind it. ' + repair + ' If a sibling died on a Claude limit, add usage-adaptive routing or route that stage to Codex first (args.models). To rebuild the whole design from scratch instead, re-invoke with args.resume=\'never\'.' }
3487
3652
  }
@@ -3491,6 +3656,7 @@ if (!fanVerdict.complete) {
3491
3656
  // the codex:codex-rescue runtime and GRACEFULLY FALL BACK to the default (Claude) planner if Codex is
3492
3657
  // unavailable/errors — the pipeline never blocks on Codex.
3493
3658
  phase('Plan')
3659
+ await recordRegistryEvent('heartbeat', 'Plan')
3494
3660
  await usageProbe('Plan')
3495
3661
  const planPrompt = 'Step 6 (SPARC-GOAP implementation plan) of /feature-adr for "' + DESC + '" (' + SLUG + ', tier ' + tier + '). Given the requirements + ADR + architecture in ' + FDIR + ', decompose into milestones + concrete tasks with success metrics. Write ' + FDIR + '/06_implementation_plan.md. END the plan with a trailing `EXPECTED_CODE_TARGETS:` block listing, one per line as `- <repo-relative path>`, EVERY production/test/config/doc file Step 7 is expected to create or modify. This block is machine-read by the Step-7.5 landing barrier: only paths it ESTABLISHES can ever count as landed, so an absent or unpollable block makes the barrier verdict INCONCLUSIVE. List only real targets outside features/, .dz/, .agentic-qe/ and roam/. The K2 plan-completeness gate blocks Step 7 until the plan satisfies these too, so write them in as you author, not afterwards: (C1) every ADR under 03_adr/ is cited as `ADR-<n>` by the task that implements it; (C2) every test path named in an ADR Confirmation stanza appears verbatim in the plan, bound to the task that writes it; (C4) every acid token `A<n>` from 00_complexity_assessment.md is named verbatim, bound to its owning task and to the test that proves the refusal. If any corrections from Step 3.5 (a CONDITIONAL verdict) or other sources are folded into this plan, carry them in a `## Amendments` section. ' + AMENDMENT_RULE + ' Return wrote[] + summary.' + ABSOLUTE_PATH_NOTE + WRITE_DISCIPLINE
3496
3662
  const planContext = buildDecisionContext({ slug: SLUG, decisionKind: 'plan-route-selection', description: DESC, tier: tier, codeHint: CODE_HINT, upstreamDigest: fnv1a64(JSON.stringify(design === undefined ? null : design)) })
@@ -3686,6 +3852,7 @@ if (planGate.verdict !== 'pass') {
3686
3852
  // checkpoint deliberately: an incomplete plan is not something to steer, it is something to fix.
3687
3853
  const planGateFailedGates = { plan: (plan ? 'produced' : 'missing'), planCompleteness: planGate.verdict, challengePanel: 'not-run', code: 'not-run', qe: 'not-run' }
3688
3854
  const planGateFailedOutcome = runOutcomeOf({ phase: 'plan-gate-failed', gates: planGateFailedGates })
3855
+ if (registryOutcome !== 'unverified') registryOutcome = planGateFailedOutcome
3689
3856
  await appendRunCostRow('plan-gate', 'Plan', planGateFailedOutcome)
3690
3857
  return { tier: tier, phase: 'plan-gate-failed', outcome: planGateFailedOutcome, slug: SLUG, artifactsDir: FDIR, planner: (plan ? plan.planner : null), plan: (plan ? plan.summary : null), modelsUsed: modelsUsed, dispatchOutcomes: dispatchOutcomes, planGate: planGate, gates: planGateFailedGates, resumedStages: resumedStages, checkpointing: CHECKPOINTS_ON ? RESUME_MODE : 'off', trainingPairs: CAPTURE_PAIRS ? TP_DIR : 'off', captureFailures: captureFailures, recordFailures: recordFailures, decisionRecallFailures: decisionRecallFailures, usageEvents: usageEvents, usageThreshold: USAGE_THRESHOLD, polymorphism: POLY.hasManifest ? POLY.report : null, note: refusalNoteFor(planGate, SLUG) }
3691
3858
  }
@@ -3728,6 +3895,7 @@ if (stopHere) {
3728
3895
  // presence), never from prose, so a skipped gate shows as 'not-run' instead of being silently forgotten.
3729
3896
  const planGates = { plan: (plan ? 'produced' : 'missing'), planCompleteness: planGate.verdict, challengePanel: (challengeVerdict ? 'ran' : 'not-run'), code: 'not-run', qe: 'not-run' }
3730
3897
  const checkpointAfterPlanOutcome = runOutcomeOf({ phase: 'checkpoint-after-plan', gates: planGates })
3898
+ if (registryOutcome !== 'unverified') registryOutcome = checkpointAfterPlanOutcome
3731
3899
  await appendRunCostRow('plan', 'Plan', checkpointAfterPlanOutcome)
3732
3900
  return { tier: tier, phase: 'checkpoint-after-plan', outcome: checkpointAfterPlanOutcome, artifactsDir: FDIR, planner: (plan ? plan.planner : null), plan: (plan ? plan.summary : null), modelsUsed: plannedModels, dispatchOutcomes: dispatchOutcomes, challengeVerdict: challengeVerdict, gates: planGates, resumedStages: resumedStages, checkpointing: CHECKPOINTS_ON ? RESUME_MODE : 'off', trainingPairs: CAPTURE_PAIRS ? TP_DIR : 'off', captureFailures: captureFailures, recordFailures: recordFailures, decisionRecallFailures: decisionRecallFailures, usageEvents: usageEvents, usageThreshold: USAGE_THRESHOLD, polymorphism: POLY.hasManifest ? POLY.report : null, note: 'L/XL checkpoint - review the ADR + plan (+ the planned code/qe/fleet models) + the challenge panel verdict (advisory) + the gates line, then re-invoke with args.stopAfter="none" to implement + QE (durable checkpoints make the re-invoke resume router+design+plan instead of re-running them). Present the gates map as a `🚦 Gates:` line in the checkpoint banner, rendering the planCompleteness entry as `K2 plan-completeness ✓` (pass) / `✗` (fail) / `inconclusive`.' }
3733
3901
  }
@@ -3764,6 +3932,8 @@ if (QE_SCOPE === 'uncommitted') {
3764
3932
  }
3765
3933
 
3766
3934
  phase('Code')
3935
+
3936
+ await recordRegistryEvent('heartbeat', 'Code')
3767
3937
  await usageProbe('Code')
3768
3938
  // GATE-ANSWERED preamble. MEASURED 2026-08-31 (job task-mtgrlrq7, feature storage-auth-classes):
3769
3939
  // a Codex coder exited status=completed, exit 0, touchedFiles=[] because it ASKED the routing
@@ -3977,6 +4147,7 @@ if (resumedStages.indexOf('code') !== -1 && landedNote !== '') {
3977
4147
 
3978
4148
  // Step 8: QE (brutal-honesty, agentic-qe) + MANDATORY teach
3979
4149
  phase('QE')
4150
+ await recordRegistryEvent('heartbeat', 'QE')
3980
4151
 
3981
4152
  // ── Writer-quiescence probe (feature qe-writer-quiescence, backlog 700b46a4) ─────────────────────
3982
4153
  // Step-8 used to grade a MOVING tree (crossrt-1: a background worker wrote AFTER the verdict,
@@ -4017,8 +4188,20 @@ log('Step 8 writer-quiescence: ' + writerQuiescence.verdict + ' (windows: ' + (w
4017
4188
  const wqNote = writerQuiescence.verdict === 'quiet'
4018
4189
  ? ' WRITER-QUIESCENCE: quiet (' + writerQuiescence.note + ').'
4019
4190
  : ' WRITER-QUIESCENCE GATE (MANDATORY to acknowledge): ' + writerQuiescence.note + ' State this standing explicitly in 08_qe_report.md next to the grade.'
4191
+ const confirmationGatePrompt = 'STEP-8 CONFIRMATION FILE GATE. You are the shell because this workflow sandbox has no filesystem API. Work under repo ' + REPO + ' and inspect only direct canonical ADR Markdown files in ' + FDIR + '/03_adr/NNN-*.md. If 03_adr is absent or has zero direct ADR files, return verdict=skipped, reason="ADR нет, проверять нечего", missing=[], checked=[]. Otherwise read every ADR. Find its unique H2 whose line starts with `## Confirmation` (the suffix may be Russian), extract EVERY repo-relative test-file path named in that section, and refuse if the section/path cannot be parsed. For every extracted path run shell checks from ' + REPO + ': missing (`test -e` is false) goes in missing; an existing path that is not a readable regular file (`test -f`, `test -r`, and an actual read) returns verdict=refused with the exact path and reason. Never turn an unreadable path, directory, parse failure, empty stdout, or tool error into skipped. If missing is non-empty return verdict=fail; otherwise pass. This gate proves ONLY file existence/readability; do not enforce the other ADR checklist items. Return exactly {verdict, missing, checked, reason}.'
4192
+ const confirmationGateRaw = await dispatchAgent(newRung(), confirmationGatePrompt, { label: 'qe:confirmation-files', phase: 'QE', effort: 'low', schema: CONFIRMATION_FILE_GATE })
4193
+ const confirmationFileGate = normalizeConfirmationFileGate(confirmationGateRaw)
4194
+ const confirmationGateLine = confirmationFileGate.verdict === 'skipped'
4195
+ ? 'Confirmation file gate: пропущено: ADR нет, проверять нечего'
4196
+ : confirmationFileGate.verdict === 'pass'
4197
+ ? 'Confirmation file gate: PASS — checked ' + confirmationFileGate.checked.join(', ')
4198
+ : confirmationFileGate.verdict === 'fail'
4199
+ ? 'Confirmation file gate: FAIL — missing ' + confirmationFileGate.missing.join(', ')
4200
+ : 'Confirmation file gate: REFUSED — ' + confirmationFileGate.reason
4201
+ log(confirmationGateLine)
4202
+ const confirmationGateNote = ' MANDATORY CONFIRMATION FILE GATE RESULT: `' + confirmationGateLine + '`. Write that as a separate line in 08_qe_report.md. The independent QE review MUST still run. If the gate verdict is fail or refused, the final Step-8 grade cannot be A or B; the workflow also enforces that after the reviewer returns. This gate proves only existence/readability; all other ADR checklist items remain advisory.'
4020
4203
  await usageProbe('QE')
4021
- const qePrompt = 'Step 8 (QE - brutal-honesty review, agentic-qe) of /feature-adr for "' + DESC + '" (' + SLUG + '). Adversarially review the SHIPPED code (read it): correctness, edge cases, error handling, and the LOAD-BEARING property the ADR named (ASSERT it has a test that DISCRIMINATES - the recurring lesson: a test that would still pass with the protection deleted is documentation, not a gate). Run this ADR gate before final grading: ' + ADR_FITNESS_CHECKLIST + ' ' + DISCRIMINATION_GATE + ' ' + MUTATION_GATE + ' ' + NO_STUBS_GATE + ' ' + AMENDMENT_GATE + ' Grade A/B/C/D honestly. Assess code-test adequacy + doc-test presence. List CONFIRMED gaps with severity. Write ' + FDIR + '/08_qe_report.md with the primary findings under the exact heading `## Primary QE pass` and an ADR Fitness Checklist section showing PASS/FAIL per ADR and evidence for the Confirmation-linked test. MANDATORY SELF-LEARNING STORE (close the loop, never skip): compare every candidate lesson against the Step-0 recalled LEARNED patterns above. Teach ONLY lessons NOT covered by Step-0 recall. On overlap, run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate; if --reinforce is unavailable, skip the duplicate teach and report `reinforced existing pattern <id>` in the QE report. Store every genuinely new lesson in the CANONICAL BRAIN store at `' + BRAIN + '` so it is NOT lost to a target repo you may have cd`d into. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal) — the `cd ' + BRAIN + ' &&` prefix + `--project ' + BRAIN + '` pin guarantee the lesson lands in the brain regardless of your CWD. Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` (run it verbatim via Bash, do not skip). Do NOT teach trivia or invent gaps. AUTHORING-TIME CLAIM-CHECK (Deliverable of claim-check-authoring-time): after writing ' + FDIR + '/08_qe_report.md, run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash, parse the {ok, findings, scanned} JSON, and report claimCheck: {findings: N, high: N, medium: N} (counts by severity) in your return object. TAG EVERY QUANTITATIVE CLAIM you write in the report using the convention the checker recognizes as honest — write "1131 tests pass (MEASURED — `npx vitest run`)", never a bare "1131 tests pass" — and where you QUOTE a forbidden phrase as an example (e.g. the retracted "100% passing" framing), backtick the literal so it reads as code, not an assertion, so your own compliant report scans clean. Return {grade, gaps, codeTestsAdequate, docTestsPresent, claimCheck}.' + ABSOLUTE_PATH_NOTE + landedNote + wqNote + PS_GUIDANCE('qe')
4204
+ const qePrompt = 'Step 8 (QE - brutal-honesty review, agentic-qe) of /feature-adr for "' + DESC + '" (' + SLUG + '). Adversarially review the SHIPPED code (read it): correctness, edge cases, error handling, and the LOAD-BEARING property the ADR named (ASSERT it has a test that DISCRIMINATES - the recurring lesson: a test that would still pass with the protection deleted is documentation, not a gate). Run this ADR gate before final grading: ' + ADR_FITNESS_CHECKLIST + ' ' + DISCRIMINATION_GATE + ' ' + MUTATION_GATE + ' ' + NO_STUBS_GATE + ' ' + AMENDMENT_GATE + ' Grade A/B/C/D honestly. Assess code-test adequacy + doc-test presence. List CONFIRMED gaps with severity. Write ' + FDIR + '/08_qe_report.md with the primary findings under the exact heading `## Primary QE pass` and an ADR Fitness Checklist section showing PASS/FAIL per ADR and evidence for the Confirmation-linked test. MANDATORY SELF-LEARNING STORE (close the loop, never skip): compare every candidate lesson against the Step-0 recalled LEARNED patterns above. Teach ONLY lessons NOT covered by Step-0 recall. On overlap, run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate; if --reinforce is unavailable, skip the duplicate teach and report `reinforced existing pattern <id>` in the QE report. Store every genuinely new lesson in the CANONICAL BRAIN store at `' + BRAIN + '` so it is NOT lost to a target repo you may have cd`d into. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal) — the `cd ' + BRAIN + ' &&` prefix + `--project ' + BRAIN + '` pin guarantee the lesson lands in the brain regardless of your CWD. Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` (run it verbatim via Bash, do not skip). Do NOT teach trivia or invent gaps. In the return object set roundLessons to the teach:<id> receipts successfully written in this Step 8; when there were none, return roundLessons:[] and a non-empty roundNoNewKnowledge reason derived from this review/reinforcement decision. AUTHORING-TIME CLAIM-CHECK (Deliverable of claim-check-authoring-time): after writing ' + FDIR + '/08_qe_report.md, run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash, parse the {ok, findings, scanned} JSON, and report claimCheck: {findings: N, high: N, medium: N} (counts by severity) in your return object. TAG EVERY QUANTITATIVE CLAIM you write in the report using the convention the checker recognizes as honest — write "1131 tests pass (MEASURED — `npx vitest run`)", never a bare "1131 tests pass" — and where you QUOTE a forbidden phrase as an example (e.g. the retracted "100% passing" framing), backtick the literal so it reads as code, not an assertion, so your own compliant report scans clean. Return {grade, gaps, codeTestsAdequate, docTestsPresent, claimCheck, roundLessons, roundNoNewKnowledge}.' + ABSOLUTE_PATH_NOTE + landedNote + wqNote + confirmationGateNote + PS_GUIDANCE('qe')
4022
4205
  // CROSS-MODEL QE (load-bearing): resolveStageModel('qe') derives the OTHER family than the resolved
4023
4206
  // coder when args.models.qe is unset (coder-codex ⇒ opus; coder-Claude ⇒ codex, or opus if codex absent).
4024
4207
  // An explicit args.models.qe wins. A Claude qe spec is merged onto the qe-code-reviewer base (role
@@ -4049,7 +4232,7 @@ const qe2Spec = qePrecisionPassSpec(PRIMARY, BUDGET_MODE, tier)
4049
4232
  // re-teach (the original run already stored its lessons — replaying teach would double-store).
4050
4233
  // R6: the review SCOPE is part of what a QE verdict is about, so it enters the hash — a resume must
4051
4234
  // not present a verdict obtained over one scope as if it had been obtained over another.
4052
- const qeHash = ckptHash('qe', [fnv1a64(JSON.stringify(codeStage === undefined ? null : codeStage)), tier, DESC, QE_REVIEWER, MODELS.qe === undefined ? null : MODELS.qe, CODEX_MODEL, coderUsed, PRIMARY, BUDGET_MODE, qe2Spec, POLY.hasManifest, fnv1a64(String(POLY.report || '')), usageOverride, QE_SCOPE, QE_SCOPE_REF])
4235
+ const qeHash = ckptHash('qe', [fnv1a64(JSON.stringify(codeStage === undefined ? null : codeStage)), tier, DESC, QE_REVIEWER, MODELS.qe === undefined ? null : MODELS.qe, CODEX_MODEL, coderUsed, PRIMARY, BUDGET_MODE, qe2Spec, POLY.hasManifest, fnv1a64(String(POLY.report || '')), usageOverride, QE_SCOPE, QE_SCOPE_REF, confirmationFileGate])
4053
4236
  let crossFamilyQeReport = null
4054
4237
  const qeStage = await withCheckpoint('qe', 'QE', qeHash, async () => {
4055
4238
  let qe = null
@@ -4184,7 +4367,7 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
4184
4367
  if (codexQe) {
4185
4368
  // gaps come from what the reviewer ACTUALLY found; gradeSource says whether the letter was
4186
4369
  // STATED by the reviewer or DERIVED from its findings, because mode A cannot be asked for one.
4187
- qe = { grade: codexQe.grade, gaps: codexQe.findings, codeTestsAdequate: null, docTestsPresent: null, summary: String(codexQe.text).slice(0, 1500), gradeSource: codexQe.gradeSource, qeScope: { mode: codexQe.mode, ref: codexQe.scopeRef, files: codexQe.files } }
4370
+ qe = enforceConfirmationFileGate({ grade: codexQe.grade, gaps: codexQe.findings, codeTestsAdequate: null, docTestsPresent: null, summary: String(codexQe.text).slice(0, 1500), gradeSource: codexQe.gradeSource, qeScope: { mode: codexQe.mode, ref: codexQe.scopeRef, files: codexQe.files } }, confirmationFileGate)
4188
4371
  qeReviewerUsed = qeIsCodex ? 'codex' : 'codex-fallback'
4189
4372
  log('QE: cross-family review by codex, mode ' + codexQe.mode + ' (scope ' + codexQe.scopeRef + ', grade ' + codexQe.grade + ' ' + codexQe.gradeSource + ', ' + codexQe.elapsedSeconds + 's)')
4190
4373
  // ARTIFACT SCRIBE. The old dispatch handed Codex the whole Step-8 prompt, so the reviewer itself
@@ -4199,7 +4382,7 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
4199
4382
  // with no 08_qe_report.md at all. Named by cross-family review of b6973199. The verdict itself
4200
4383
  // is real (Codex produced it), so a failed transcription DEGRADES the run rather than voiding
4201
4384
  // it — but it must be visible, and it must never read as a clean QE.
4202
- const scribePrompt = 'Step 8 (QE) of /feature-adr for "' + DESC + '" (' + SLUG + '). The independent cross-family review has ALREADY BEEN DONE, by Codex. You are the SCRIBE, not the reviewer: RECORD it, do NOT re-grade it, do NOT soften it, do NOT add a verdict of your own, and do NOT mark anything resolved that the reviewer flagged. The grade is ' + codexQe.grade + ' and it is FINAL.\n\nWrite ' + FDIR + '/08_qe_report.md with: (1) the grade ' + codexQe.grade + ' stated verbatim; (2) HOW it was obtained — dispatch mode ' + codexQe.mode + ', scope ' + codexQe.scopeRef + ', wall-clock ' + codexQe.elapsedSeconds + 's, gradeSource ' + codexQe.gradeSource + ' (a DERIVED grade means the reviewer could not be asked for a letter and it was computed from the severities it reported — say so plainly); (3) the reviewer text below, verbatim, under the exact heading `## Primary QE pass`; (4) an ADR Fitness Checklist section with PASS/FAIL per ADR and the evidence pointer for the Confirmation-linked test.\n\nREVIEWER TEXT (verbatim, do not edit or summarise):\n' + String(codexQe.text) + '\n\nMANDATORY SELF-LEARNING STORE (close the loop, never skip): compare candidate lessons against the Step-0 recalled LEARNED patterns. Teach ONLY lessons NOT already covered; on overlap run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal). Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` verbatim via Bash. Finally run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash and TAG every quantitative claim you write the way the checker recognises as honest.' + ABSOLUTE_PATH_NOTE
4385
+ const scribePrompt = 'Step 8 (QE) of /feature-adr for "' + DESC + '" (' + SLUG + '). The independent cross-family review has ALREADY BEEN DONE, by Codex. You are the SCRIBE, not the reviewer: RECORD it, do NOT re-grade it, do NOT soften it, do NOT add a verdict of your own, and do NOT mark anything resolved that the reviewer flagged. The reviewer grade was ' + codexQe.grade + '; the final Step-8 grade after the deterministic Confirmation file gate is ' + qe.grade + ' and it is FINAL: the file gate can only LOWER a reviewer grade, never raise it, and nothing after this point may change it.\n\nWrite ' + FDIR + '/08_qe_report.md with: (1) the final grade ' + qe.grade + ' stated verbatim; (2) HOW it was obtained — reviewer grade ' + codexQe.grade + ', dispatch mode ' + codexQe.mode + ', scope ' + codexQe.scopeRef + ', wall-clock ' + codexQe.elapsedSeconds + 's, gradeSource ' + codexQe.gradeSource + ' (a DERIVED grade means the reviewer could not be asked for a letter and it was computed from the severities it reported — say so plainly); (3) the reviewer text below, verbatim, under the exact heading `## Primary QE pass`; (4) an ADR Fitness Checklist section with PASS/FAIL per ADR and the evidence pointer for the Confirmation-linked test; (5) this gate receipt as a separate line: `' + confirmationGateLine + '`.\n\nREVIEWER TEXT (verbatim, do not edit or summarise):\n' + String(codexQe.text) + '\n\nMANDATORY SELF-LEARNING STORE (close the loop, never skip): compare candidate lessons against the Step-0 recalled LEARNED patterns. Teach ONLY lessons NOT already covered; on overlap run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal). Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` verbatim via Bash. Finally run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash and TAG every quantitative claim you write the way the checker recognises as honest.' + ABSOLUTE_PATH_NOTE
4203
4386
  // WITNESS THE REWRITE, not the existence. On a re-QE or a resume with the same slug an OLD
4204
4387
  // 08_qe_report.md is already sitting there, and an existence probe reports that stale file as
4205
4388
  // landed — so a scribe that wrote nothing still marked the new verdict recorded, and the stage
@@ -4254,6 +4437,7 @@ if (qe === null && qeIsCodex) {
4254
4437
  if (qe) { qeReviewerUsed = 'claude'; crossFamilyQeReport = cfBelt.report }
4255
4438
  else modelsUsed.qe = cfBelt.label + ' (no deliverable)'
4256
4439
  }
4440
+ qe = enforceConfirmationFileGate(qe, confirmationFileGate)
4257
4441
  // A-normal L/XL only: Sonnet is the recall-oriented primary reviewer; Opus is a SECOND,
4258
4442
  // independent precision pass. It is advisory but real — never a table-only half-wire — and its
4259
4443
  // provenance stays separate in both the return object and 08_qe_report.md.
@@ -4305,6 +4489,26 @@ let qeReviewerUsed = qeStage ? qeStage.qeReviewerUsed : 'claude'
4305
4489
  if (qeStage && qeStage.modelUsed) modelsUsed.qe = qeStage.modelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
4306
4490
  if (qeStage && qeStage.qe2ModelUsed) modelsUsed.qe2 = qeStage.qe2ModelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
4307
4491
 
4492
+ // Step 8 has completed its teach/reinforce work and written 08_qe_report.md. Closing telemetry is
4493
+ // secondary: refusal is loud and reflected in roundClosed, but it never overturns the feature run.
4494
+ if (qe && pipelineRound !== null) {
4495
+ const roundGrade = String(qe.grade || '').trim().toUpperCase()
4496
+ const roundOutcome = ['A', 'A-', 'B+', 'B'].indexOf(roundGrade) !== -1 ? 'shipped' : (['C', 'D'].indexOf(roundGrade) !== -1 ? 'refuted' : null)
4497
+ if (roundOutcome === null) {
4498
+ log('round close refused: unsupported Step 8 grade ' + JSON.stringify(roundGrade))
4499
+ } else {
4500
+ const roundLessonIds = Array.isArray(qe.roundLessons) ? qe.roundLessons.filter(function (id) { return /^teach:[a-z0-9]+$/i.test(String(id)) }) : []
4501
+ const roundLearningArg = roundLessonIds.length > 0
4502
+ ? roundLessonIds.map(function (id) { return ' --lesson ' + shq(String(id)) }).join('')
4503
+ : ' --no-new-knowledge ' + shq((typeof qe.roundNoNewKnowledge === 'string' && qe.roundNoNewKnowledge.trim() !== '') ? qe.roundNoNewKnowledge.trim() : 'Step 8 returned no new teach receipt for this round')
4504
+ const roundCloseCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' round close --slug ' + shq(SLUG) + ' --round ' + pipelineRound + ' --outcome ' + roundOutcome + ' --reason ' + shq('grade ' + roundGrade) + roundLearningArg + ' --project ' + shq(BRAIN) + ' --no-cost --json'
4505
+ const roundCloseOut = await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and return its stdout VERBATIM, nothing else: ' + roundCloseCmd, { label: 'round:close', phase: 'QE', effort: 'low' })
4506
+ const roundCloseReceipt = parseRoundCommandJson(roundCloseOut)
4507
+ roundClosed = !!(roundCloseReceipt && roundCloseReceipt.marker && roundCloseReceipt.row && roundCloseReceipt.row.stage === 'round')
4508
+ if (!roundClosed) log('round close refused: ' + (roundCloseReceipt && roundCloseReceipt.message ? roundCloseReceipt.message : String(roundCloseOut || 'no JSON receipt')))
4509
+ }
4510
+ }
4511
+
4308
4512
  // Step 8 claim-gate: fold the QE agent's reported claim-check counts into an additive result field.
4309
4513
  const claimGate = step8ClaimGate(qe && qe.claimCheck ? qe.claimCheck : null)
4310
4514
  log(claimGate.note)
@@ -4411,6 +4615,7 @@ if (Object.keys(AUTOCOST).length > 0 && resumedStages.indexOf('qe') === -1) {
4411
4615
  let fleet = 'skipped (S/M)'
4412
4616
  if (isLplus) {
4413
4617
  phase('FleetQE')
4618
+ await recordRegistryEvent('heartbeat', 'FleetQE')
4414
4619
  await usageProbe('FleetQE')
4415
4620
  const fleetDecision = resolveStageDecision('fleet')
4416
4621
  const fleetModel = fleetDecision.opts
@@ -4469,6 +4674,7 @@ let delivery = null
4469
4674
  if (DELIVERY_ON) {
4470
4675
  try {
4471
4676
  phase('Delivery')
4677
+ await recordRegistryEvent('heartbeat', 'Delivery')
4472
4678
  await usageProbe('Delivery')
4473
4679
  // QE-D#2: cross-family is judged against the ACTUAL coder (coderUsed — the code stage already ran), and
4474
4680
  // codex PLANES are unsupported in v1: a plane is a data-returning schema stage, and the codex wrapper
@@ -4650,6 +4856,7 @@ const finalGates = {
4650
4856
  // DERIVED from the barrier's machine verdict, not from "is there a result object". A codex run
4651
4857
  // whose barrier came back INCONCLUSIVE used to render exactly like a clean synchronous one.
4652
4858
  code: (codeStage === null || codeStage === undefined || !code ? 'missing' : (codeStage.landingStatus === 'synchronous' ? 'produced' : (codeStage.landingStatus === 'landed' ? 'landed' : (codeStage.landingStatus === 'inconclusive' ? 'inconclusive' : 'not-landed')))),
4859
+ confirmationFiles: confirmationFileGate.verdict,
4653
4860
  qe: (qe ? (qe.grade || 'ran') : 'not-run'),
4654
4861
  claimCheck: (qe && qe.claimCheck ? (qe.claimCheck.high > 0 ? 'high-findings' : 'clean') : 'not-run'),
4655
4862
  fleet: (isLplus ? (fleet ? 'ran' : 'not-run') : 'n/a'),
@@ -4664,6 +4871,7 @@ await appendRunCostRow('full', (isLplus ? 'FleetQE' : 'QE'), finalOutcome)
4664
4871
  // time) hit SCORE-EXISTS and froze the failed attempt's 0/N forever. An unfinished run is simply
4665
4872
  // not scored; the resume that finishes the work scores it.
4666
4873
  const score = finalOutcome === 'completed' ? await autoScore(qeHash) : null
4874
+ if (registryOutcome !== 'unverified') registryOutcome = finalOutcome
4667
4875
  return {
4668
4876
  slug: SLUG, tier: tier, mode: MODE, artifactsDir: FDIR,
4669
4877
  outcome: finalOutcome,
@@ -4671,6 +4879,7 @@ return {
4671
4879
  design: design.filter(Boolean).map((d) => d.wrote).flat(),
4672
4880
  codeWrote: code ? code.wrote : [],
4673
4881
  qeGrade: qe ? qe.grade : null,
4882
+ roundClosed: roundClosed,
4674
4883
  score: score,
4675
4884
  gaps: qe ? qe.gaps : [],
4676
4885
  codeTestsAdequate: qe ? qe.codeTestsAdequate : null,
@@ -4696,9 +4905,16 @@ return {
4696
4905
  crossFamilyQe: crossFamilyQeReport,
4697
4906
  qeReportWritten: (qe && typeof qe.qeReportWritten === 'boolean') ? qe.qeReportWritten : null,
4698
4907
  claimGate: claimGate,
4908
+ confirmationFileGate: confirmationFileGate,
4699
4909
  autoCost: Object.keys(AUTOCOST).length ? AUTOCOST : null,
4700
4910
  // P4 (checkpoint-gate-line): DERIVED gate map for the final banner — from actual run state, never prose.
4701
4911
  gates: finalGates,
4702
4912
  delivery: delivery,
4703
4913
  promiseTags: tags,
4704
4914
  }
4915
+ } finally {
4916
+ if (finishRunRegistry) {
4917
+ try { await finishRunRegistry() }
4918
+ catch (error) { registryOutcome = 'unverified'; log('run registry: finished UNVERIFIED — ' + String(error)) }
4919
+ }
4920
+ }