@dzhechkov/skills-feature-adr 1.5.10 → 1.5.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +9 -5
- package/README.md +18 -5
- package/package.json +1 -1
- package/sbom.json +14 -4
- package/templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs +2 -15
- package/templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs +109 -0
- package/templates/.claude/workflows/feature-adr.js +237 -21
package/.dz-manifest.json
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
},
|
|
14
14
|
{
|
|
15
15
|
"path": "README.md",
|
|
16
|
-
"sha256": "
|
|
16
|
+
"sha256": "889d62c5dfafb53c0664867f0882e5712e47eb8aa7a44c6db46d082e693ed58d"
|
|
17
17
|
},
|
|
18
18
|
{
|
|
19
19
|
"path": "bin/cli.js",
|
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
},
|
|
26
26
|
{
|
|
27
27
|
"path": "package.json",
|
|
28
|
-
"sha256": "
|
|
28
|
+
"sha256": "d13930ae60a1d849d08b798dbb3b3ed4fb5d0b50c4c3910fc76a455dd9e6ce31"
|
|
29
29
|
},
|
|
30
30
|
{
|
|
31
31
|
"path": "src/cli.js",
|
|
@@ -249,7 +249,11 @@
|
|
|
249
249
|
},
|
|
250
250
|
{
|
|
251
251
|
"path": "templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs",
|
|
252
|
-
"sha256": "
|
|
252
|
+
"sha256": "cbe70a524f6ce7987b5e3961ee412b92c24c00d84066d438c759659d656efaa2"
|
|
253
|
+
},
|
|
254
|
+
{
|
|
255
|
+
"path": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
|
|
256
|
+
"sha256": "82c3c48c25f400f3a64d1caba3d47357f3db87460f0746f88706f901eb6ff689"
|
|
253
257
|
},
|
|
254
258
|
{
|
|
255
259
|
"path": "templates/.claude/skills/frontend-design/LICENSE.txt",
|
|
@@ -313,7 +317,7 @@
|
|
|
313
317
|
},
|
|
314
318
|
{
|
|
315
319
|
"path": "templates/.claude/workflows/feature-adr.js",
|
|
316
|
-
"sha256": "
|
|
320
|
+
"sha256": "22416f105c1b988c07f6cb9002a507b54b07842b5b4bb109b7e300ff52272603"
|
|
317
321
|
},
|
|
318
322
|
{
|
|
319
323
|
"path": "templates/lib/memory-protocol.md",
|
|
@@ -325,5 +329,5 @@
|
|
|
325
329
|
}
|
|
326
330
|
]
|
|
327
331
|
},
|
|
328
|
-
"signature": "
|
|
332
|
+
"signature": "WT7wArJY3aKuIMMVgbBzFO+Rqw7qfcaVh8UQ9MENSltqdTJTFAr/0TBOS5A89UAfczkluCrkTtyUvPOTQvaNCw=="
|
|
329
333
|
}
|
package/README.md
CHANGED
|
@@ -698,7 +698,7 @@ sufficiency + honesty, overengineering, silent decisions, runtime consistency, s
|
|
|
698
698
|
`dz challenge --plan <plan.md>` or the `challenge-panel` skill; scaffold the degradations registry via
|
|
699
699
|
`dz feature-adr-setup --from-spec <spec with {"degradations":true}> --apply`.
|
|
700
700
|
|
|
701
|
-
### ADR quality gate (Step 3 generates → Step 8
|
|
701
|
+
### ADR quality review + Confirmation file gate (Step 3 generates → Step 8 checks)
|
|
702
702
|
|
|
703
703
|
Step 3 and Step 8 share an ADR best-practices contract distilled from the
|
|
704
704
|
[architecture-decision-record monograph](https://github.com/architecture-decision-record/architecture-decision-record):
|
|
@@ -709,12 +709,17 @@ Step 3 and Step 8 share an ADR best-practices contract distilled from the
|
|
|
709
709
|
links + after-action review, a **`## Confirmation`** stanza (method, monitoring, success metric, owner)
|
|
710
710
|
naming the load-bearing property, and a **`## Links`** traceability block. Template weight is tier-mapped:
|
|
711
711
|
S/M → Nygard/ITD-lightweight, L/XL → MADR + Confirmation.
|
|
712
|
-
- **Step 8** runs a **13-point ADR fitness checklist** (`qe-code-reviewer`) against every generated ADR
|
|
713
|
-
|
|
712
|
+
- **Step 8** runs a **13-point advisory ADR fitness checklist** (`qe-code-reviewer`) against every generated ADR —
|
|
713
|
+
decision-shaped title, controlled-vocabulary Status + reversibility,
|
|
714
714
|
neutral Context-before-Decision, symmetric options, driver-mapped rationale, concrete/testable decision,
|
|
715
715
|
negative consequences, traceability links, no placeholder, and **rejects explainer-masquerading-as-ADR**.
|
|
716
|
-
|
|
717
|
-
|
|
716
|
+
Those judgment-based items remain findings; they do not independently force the workflow verdict.
|
|
717
|
+
- The **one mandatory gate** runs after Step 7.5 and before the QE verdict: every test-file path named
|
|
718
|
+
under an ADR heading beginning with `## Confirmation` must exist as a readable regular file. Missing
|
|
719
|
+
files or unreadable/unparseable paths force a non-passing Step-8 grade while the independent QE review
|
|
720
|
+
still runs. A feature with no ADR prints `пропущено: ADR нет, проверять нечего` and is not failed.
|
|
721
|
+
`dz discrimination-check` and `dz mutation-gate` remain advisory: existence does not prove that a test
|
|
722
|
+
actually discriminates the load-bearing property.
|
|
718
723
|
|
|
719
724
|
The pipeline **dog-foods** this: a harness test runs the gate against feature-adr's own generated ADR, so a
|
|
720
725
|
Step-3↔Step-8 drift fails CI rather than shipping.
|
|
@@ -1211,3 +1216,11 @@ Also in this release, both halves of the K2 plan-completeness gate that field us
|
|
|
1211
1216
|
`dispatchOutcomes` отчёта. Причина отказа остаётся у outcome отказавшей ступени, а intent следующего
|
|
1212
1217
|
fallback получает нейтральную причину `fallback-rung`. Регион `stage-line` принадлежит генератору
|
|
1213
1218
|
`gen-loop-blobs` и не правится руками.
|
|
1219
|
+
|
|
1220
|
+
### Shared Markdown masking in feature-adr gates
|
|
1221
|
+
|
|
1222
|
+
The standalone plan-completeness gate ships with `markdown-masker.mjs`, copied byte-for-byte from
|
|
1223
|
+
harness-core's `src/markdown-masker.ts`. It runs without a core build. Amendment checks, swarm briefs
|
|
1224
|
+
and K2 share the parser while retaining their existing unclosed-block and indentation policies.
|
|
1225
|
+
The four-space indented-code gap remains open for amendment checks and K2; swarm briefs retain their
|
|
1226
|
+
existing masking of indented code. Versions are unchanged in this staged change.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@dzhechkov/skills-feature-adr",
|
|
3
|
-
"version": "1.5.
|
|
3
|
+
"version": "1.5.11",
|
|
4
4
|
"description": "Adaptive Feature Development skill pack for Claude Code — 11-step pipeline with Complexity Router (S/M/L/XL), ADR-driven architecture, 15 agentic-qe skills, multi-agent fleet QE. Supports --full-qe, --full-qe-extended, --with-learning, and --knowledge-extractor modes.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"skills-feature-adr": "./bin/cli.js"
|
package/sbom.json
CHANGED
|
@@ -35,7 +35,7 @@
|
|
|
35
35
|
"hashes": [
|
|
36
36
|
{
|
|
37
37
|
"alg": "SHA-256",
|
|
38
|
-
"content": "
|
|
38
|
+
"content": "889d62c5dfafb53c0664867f0882e5712e47eb8aa7a44c6db46d082e693ed58d"
|
|
39
39
|
}
|
|
40
40
|
]
|
|
41
41
|
},
|
|
@@ -69,7 +69,7 @@
|
|
|
69
69
|
},
|
|
70
70
|
{
|
|
71
71
|
"name": "dz:canonical-json-sha256-v2",
|
|
72
|
-
"value": "
|
|
72
|
+
"value": "d13930ae60a1d849d08b798dbb3b3ed4fb5d0b50c4c3910fc76a455dd9e6ce31"
|
|
73
73
|
}
|
|
74
74
|
]
|
|
75
75
|
},
|
|
@@ -629,7 +629,17 @@
|
|
|
629
629
|
"hashes": [
|
|
630
630
|
{
|
|
631
631
|
"alg": "SHA-256",
|
|
632
|
-
"content": "
|
|
632
|
+
"content": "cbe70a524f6ce7987b5e3961ee412b92c24c00d84066d438c759659d656efaa2"
|
|
633
|
+
}
|
|
634
|
+
]
|
|
635
|
+
},
|
|
636
|
+
{
|
|
637
|
+
"type": "file",
|
|
638
|
+
"name": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
|
|
639
|
+
"hashes": [
|
|
640
|
+
{
|
|
641
|
+
"alg": "SHA-256",
|
|
642
|
+
"content": "82c3c48c25f400f3a64d1caba3d47357f3db87460f0746f88706f901eb6ff689"
|
|
633
643
|
}
|
|
634
644
|
]
|
|
635
645
|
},
|
|
@@ -789,7 +799,7 @@
|
|
|
789
799
|
"hashes": [
|
|
790
800
|
{
|
|
791
801
|
"alg": "SHA-256",
|
|
792
|
-
"content": "
|
|
802
|
+
"content": "22416f105c1b988c07f6cb9002a507b54b07842b5b4bb109b7e300ff52272603"
|
|
793
803
|
}
|
|
794
804
|
]
|
|
795
805
|
},
|
|
@@ -51,6 +51,7 @@
|
|
|
51
51
|
// the feature's own `00_complexity_assessment.md` acid-case table (rows shaped `| A<N> | … |`), or
|
|
52
52
|
// supplied explicitly with `--acid=T1,T2,…`. If neither establishes a corpus, C4 is SKIPPED-with-note
|
|
53
53
|
// (a feature that declared no acid cases cannot be failed for not naming them).
|
|
54
|
+
import { maskMarkdown } from './markdown-masker.mjs';
|
|
54
55
|
import { readFileSync, readdirSync, existsSync } from 'node:fs';
|
|
55
56
|
import { isAbsolute, join, resolve } from 'node:path';
|
|
56
57
|
|
|
@@ -319,21 +320,7 @@ else for (const t of acidTokens) if (!new RegExp(`\\b${t.replace(/[.*+?^${}()|[\
|
|
|
319
320
|
// `## Amendments` could become the section heading, and a fenced example row could either open a
|
|
320
321
|
// phantom amendment or hand a real testless one someone else's marker. Third fence-blindness
|
|
321
322
|
// found in a checker today, so it is closed here by construction rather than by care.
|
|
322
|
-
const
|
|
323
|
-
const planLines = [];
|
|
324
|
-
{
|
|
325
|
-
let fence = null;
|
|
326
|
-
for (const line of rawLines) {
|
|
327
|
-
const open = /^ {0,3}(```+|~~~+)/.exec(line);
|
|
328
|
-
if (fence === null && open) { fence = open[1][0]; planLines.push(''); continue; }
|
|
329
|
-
if (fence !== null) {
|
|
330
|
-
planLines.push('');
|
|
331
|
-
if (new RegExp('^ {0,3}' + fence + '{3,}\\s*$').test(line)) fence = null;
|
|
332
|
-
continue;
|
|
333
|
-
}
|
|
334
|
-
planLines.push(line);
|
|
335
|
-
}
|
|
336
|
-
}
|
|
323
|
+
const planLines = maskMarkdown(plan, { unclosed: 'mask' }).split('\n');
|
|
337
324
|
let sectionStart = -1, sectionEnd = -1, cursor = 0;
|
|
338
325
|
for (const pl of planLines) {
|
|
339
326
|
// The SAME heading shape amendment-trace.ts accepts: up to three leading spaces, two to four
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Canonical Markdown block masker; copied byte-for-byte as markdown-masker.mjs beside K2.
|
|
3
|
+
* Keep this file valid JavaScript (inferred TS types, no build needed by the copy).
|
|
4
|
+
*
|
|
5
|
+
* CommonMark fences and type-2 HTML comments; same UTF-16 length and newline positions.
|
|
6
|
+
* Inline code cannot open an HTML block. HTML delimiters within a block are consumed left to
|
|
7
|
+
* right, including close/reopen on one line. This is not a complete CommonMark parser.
|
|
8
|
+
*
|
|
9
|
+
* Reader policies are deliberate: amendment-trace restores unclosed blocks; brief/K2 hide them.
|
|
10
|
+
* Four-space code is still unsupported by default. ONLY brief keeps its pre-existing policy.
|
|
11
|
+
* Containers, tab indentation and HTML block types other than comments remain unsupported.
|
|
12
|
+
* Callbacks expose line facts; callers own semantic diagnostics and list-barrier representation.
|
|
13
|
+
*/
|
|
14
|
+
export function maskMarkdown(md = '', {
|
|
15
|
+
unclosed = 'restore',
|
|
16
|
+
indentedCode = false,
|
|
17
|
+
inlineComments = false,
|
|
18
|
+
onMasked = (_line = 0) => {},
|
|
19
|
+
onDisputed = (_line = 0) => {},
|
|
20
|
+
} = {}) {
|
|
21
|
+
const lines = md.split('\n');
|
|
22
|
+
const out = lines.slice();
|
|
23
|
+
const masked = lines.map(() => false);
|
|
24
|
+
let state = 'text';
|
|
25
|
+
let marker = '';
|
|
26
|
+
let markerLength = 0;
|
|
27
|
+
let openedAt = -1;
|
|
28
|
+
let nestedOpener = false;
|
|
29
|
+
let disputed = false;
|
|
30
|
+
for (let i = 0; i < lines.length; i++) {
|
|
31
|
+
const line = lines[i] ?? '';
|
|
32
|
+
const blank = () => { out[i] = ' '.repeat(line.length); masked[i] = true; };
|
|
33
|
+
if (state === 'fence') {
|
|
34
|
+
blank();
|
|
35
|
+
const close = /^ {0,3}(`{3,}|~{3,})\s*$/.exec(line);
|
|
36
|
+
if (close) {
|
|
37
|
+
const run = close[1] ?? '';
|
|
38
|
+
if (run[0] === marker && run.length >= markerLength) {
|
|
39
|
+
state = 'text';
|
|
40
|
+
openedAt = -1;
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
continue;
|
|
44
|
+
}
|
|
45
|
+
if (state === 'text') {
|
|
46
|
+
const fence = /^ {0,3}(`{3,}|~{3,})([^\n]*)$/.exec(line);
|
|
47
|
+
if (fence && !(fence[1]?.[0] === '`' && fence[2]?.includes('`'))) {
|
|
48
|
+
const run = fence[1] ?? '';
|
|
49
|
+
state = 'fence';
|
|
50
|
+
marker = run[0] ?? '';
|
|
51
|
+
markerLength = run.length;
|
|
52
|
+
openedAt = i;
|
|
53
|
+
blank();
|
|
54
|
+
continue;
|
|
55
|
+
}
|
|
56
|
+
if (indentedCode && /^ {4,}/.test(line)) {
|
|
57
|
+
blank();
|
|
58
|
+
continue;
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
// Outside HTML only a block opener counts: code-spanned delimiters in prose are inert.
|
|
62
|
+
// Inside HTML, backticks are literal and do not protect a delimiter.
|
|
63
|
+
const htmlStart = /^ {0,3}<!--/.test(line);
|
|
64
|
+
const wasDisputed = disputed;
|
|
65
|
+
// Ambiguity is diagnostic state, never permission to scan otherwise-visible prose.
|
|
66
|
+
// Brief deliberately keeps that diagnostic across blocks until the author's outer close;
|
|
67
|
+
// block-only readers must not hide an unrelated arrow or opener because it is pending.
|
|
68
|
+
if (state === 'html' || htmlStart || inlineComments) {
|
|
69
|
+
let maskLine = state === 'html' || htmlStart;
|
|
70
|
+
const tokens = [...line.matchAll(/`+|<!--|-->/g)];
|
|
71
|
+
for (let t = 0; t < tokens.length; t++) {
|
|
72
|
+
const token = tokens[t];
|
|
73
|
+
if (!token) continue;
|
|
74
|
+
if (token[0][0] === '`') {
|
|
75
|
+
// Code spans require an EXACT matching run; a shorter embedded run is literal.
|
|
76
|
+
// HTML blocks treat backticks literally. Inline-comment support is brief's existing
|
|
77
|
+
// policy; other readers only enter this scan at a block opener or while disputed.
|
|
78
|
+
if (state !== 'html') {
|
|
79
|
+
const end = tokens.findIndex((next, j) => j > t && next[0] === token[0]);
|
|
80
|
+
if (end >= 0) t = end;
|
|
81
|
+
}
|
|
82
|
+
continue;
|
|
83
|
+
}
|
|
84
|
+
if (token[0] === '<!--') {
|
|
85
|
+
if (state === 'html') nestedOpener = true;
|
|
86
|
+
else { state = 'html'; openedAt = i; }
|
|
87
|
+
maskLine = true;
|
|
88
|
+
} else if (state === 'html') {
|
|
89
|
+
state = 'text';
|
|
90
|
+
openedAt = -1;
|
|
91
|
+
if (nestedOpener) { disputed = true; nestedOpener = false; }
|
|
92
|
+
} else if (disputed) {
|
|
93
|
+
disputed = false;
|
|
94
|
+
maskLine = true;
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
if (maskLine) blank();
|
|
98
|
+
}
|
|
99
|
+
if (wasDisputed || disputed) onDisputed(i);
|
|
100
|
+
}
|
|
101
|
+
if (openedAt >= 0 && unclosed === 'restore') {
|
|
102
|
+
for (let i = openedAt; i < lines.length; i++) {
|
|
103
|
+
out[i] = lines[i] ?? '';
|
|
104
|
+
masked[i] = false;
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
for (let i = 0; i < masked.length; i++) if (masked[i]) onMasked(i);
|
|
108
|
+
return out.join('\n');
|
|
109
|
+
}
|
|
@@ -16,6 +16,12 @@ export const meta = {
|
|
|
16
16
|
],
|
|
17
17
|
}
|
|
18
18
|
|
|
19
|
+
// Finalize on every normal return and on a caught runtime error; a killed host cannot finalize.
|
|
20
|
+
let finishRunRegistry = null
|
|
21
|
+
let registryOutcome = 'errored'
|
|
22
|
+
try {
|
|
23
|
+
const RUNTIME_SPENT = (typeof budget === 'object' && budget && typeof budget.spent === 'function') ? function () { return budget.spent() } : null
|
|
24
|
+
let spentAtPrevRow = RUNTIME_SPENT ? RUNTIME_SPENT() : null
|
|
19
25
|
const A = typeof args === 'string' ? JSON.parse(args) : (args || {})
|
|
20
26
|
const SLUG = A.slug || 'feature'
|
|
21
27
|
const DESC = A.description || ''
|
|
@@ -135,15 +141,35 @@ function normalizeDzBin(raw, ws) {
|
|
|
135
141
|
const base = (typeof ws === 'string' && ws.length > 0) ? ws.replace(/\/+$/, '') : ''
|
|
136
142
|
return base === '' ? r : base + '/' + r
|
|
137
143
|
}
|
|
138
|
-
|
|
139
|
-
//
|
|
140
|
-
|
|
144
|
+
// In the hub, PATH may name an older published CLI. Resolve the workspace first so an omitted
|
|
145
|
+
// args.dzBin can prefer the build that belongs to this checkout.
|
|
146
|
+
if (A.workspace !== undefined && A.workspace !== null) WS = assertAbsoluteNoTraversal(A.workspace, 'workspace')
|
|
147
|
+
if ((A.dzBin === undefined || A.dzBin === null || A.dzBin === '') && WS === null) {
|
|
148
|
+
WS = await probeSessionCwd('resolve-workspace-dz')
|
|
149
|
+
}
|
|
150
|
+
const WORKSPACE_DZ = WS === null ? null : WS.replace(/\/+$/, '') + '/packages/@dzhechkov/harness-cli/dist/bin.js'
|
|
151
|
+
let DZ_RAW = A.dzBin
|
|
152
|
+
let DZ_VERSION = 'not probed (dzBin given)'
|
|
153
|
+
if (DZ_RAW === undefined || DZ_RAW === null || DZ_RAW === '') {
|
|
154
|
+
const dzProbeCmd = WORKSPACE_DZ === null
|
|
155
|
+
? "printf 'DZ_PATH_FALLBACK '; dz --version"
|
|
156
|
+
: 'if [ -f ' + shq(WORKSPACE_DZ) + " ]; then printf 'DZ_WORKSPACE_BUILD '; " + shq(WORKSPACE_DZ) + " --version; else printf 'DZ_PATH_FALLBACK '; dz --version; fi"
|
|
157
|
+
const localDzOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and reply with ONLY its stdout: ' + dzProbeCmd, { label: 'resolve-workspace-dz:select-version', phase: 'Route', model: 'haiku', effort: 'low' })
|
|
158
|
+
const localDzText = String(localDzOut || '').trim()
|
|
159
|
+
const workspaceDzExists = localDzText.indexOf('DZ_WORKSPACE_BUILD ') === 0
|
|
160
|
+
DZ_RAW = workspaceDzExists ? WORKSPACE_DZ : 'dz'
|
|
161
|
+
const dzVersionText = localDzText.replace(/^DZ_(?:WORKSPACE_BUILD|PATH_FALLBACK)\s*/, '')
|
|
162
|
+
const dzVersionLines = dzVersionText.split(/\r?\n/).map(function (line) { return line.trim() }).filter(Boolean)
|
|
163
|
+
DZ_VERSION = dzVersionLines.length > 0 ? dzVersionLines[dzVersionLines.length - 1].slice(0, 80) : 'unknown'
|
|
164
|
+
}
|
|
165
|
+
// A relative dzBin still needs WS so it resolves identically after every downstream cd. The combined
|
|
166
|
+
// selection/version probe above runs only when dzBin is absent; a supplied absolute dzBin costs no call.
|
|
141
167
|
if (DZ_RAW.indexOf('/') >= 0 && DZ_RAW.charAt(0) !== '/' && WS === null) {
|
|
142
168
|
WS = await probeSessionCwd('resolve-ws')
|
|
143
169
|
if (!WS) throw new Error('feature-adr: args.dzBin ' + JSON.stringify(DZ_RAW) + ' is relative and the workspace root could not be resolved after 2 attempts; refusing to splice a path that would resolve differently in every cd\'d command. Pass an absolute args.dzBin.')
|
|
144
170
|
}
|
|
145
171
|
const DZ = normalizeDzBin(DZ_RAW, WS)
|
|
146
|
-
log('dz binary: ' + DZ)
|
|
172
|
+
log('dz binary: ' + DZ + ' (' + DZ_VERSION + ')')
|
|
147
173
|
// args.gateScript — an explicit ABSOLUTE path to the K2 gate script (ADR-002 candidate 1). Validated
|
|
148
174
|
// HERE, at invocation time, so a bad value fails at the same layer the pure half fails rather than
|
|
149
175
|
// two layers later inside an emitted shell command.
|
|
@@ -152,7 +178,6 @@ const GATE_SCRIPT_ARG = (A.gateScript === undefined || A.gateScript === null) ?
|
|
|
152
178
|
// shell fallback WS=$(pwd -P) runs in the GATE AGENT own cwd. On a run against an external repo it
|
|
153
179
|
// equalled REPO, so the workspace candidate pointed at the target repo and the skill installed in
|
|
154
180
|
// the workspace was never found - NOT-ESTABLISHED, exit 3, Step 7 never ran. args.workspace pins it.
|
|
155
|
-
if (A.workspace !== undefined && A.workspace !== null) WS = assertAbsoluteNoTraversal(A.workspace, 'workspace')
|
|
156
181
|
// CANONICAL BRAIN store: the self-learning loop (Step-0 recall → Step-8 teach) MUST read+write ONE
|
|
157
182
|
// shared pattern store so lessons never fragment into a target repo's .dz when the Step-7 coder cd's
|
|
158
183
|
// away. BRAIN defaults to the workspace root (REPO) — so an OMITTED args.brain is behaviorally inert
|
|
@@ -170,6 +195,14 @@ const DZ_RECALL = (terms) => 'cd ' + BRAIN + ' && ' + DZ + ' recall "' + terms +
|
|
|
170
195
|
const DZ_TEACH = (lesson, reward, domain) =>
|
|
171
196
|
'cd ' + BRAIN + ' && ' + DZ + ' teach "' + lesson + '" --reward ' + reward + ' --domain ' + domain + ' --project ' + BRAIN
|
|
172
197
|
|
|
198
|
+
function parseRoundCommandJson(raw) {
|
|
199
|
+
const lines = String(raw === null || raw === undefined ? '' : raw).split(/\r?\n/).map(function (line) { return line.trim() }).filter(Boolean)
|
|
200
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
201
|
+
try { return JSON.parse(lines[i]) } catch { /* a chatty shell courier may add non-JSON lines */ }
|
|
202
|
+
}
|
|
203
|
+
return null
|
|
204
|
+
}
|
|
205
|
+
|
|
173
206
|
// ── Durable checkpoints + resume (backlog 49e4a95b) — inline mirror of ──
|
|
174
207
|
// ── harness-core/src/feature-adr-checkpoints.ts (the workflow is self-contained, no imports) ──
|
|
175
208
|
// After each expensive stage a cheap effort-low agent appends {stage, inputHash, result} to
|
|
@@ -435,7 +468,11 @@ async function withCheckpoint(stage, phaseName, inputHash, runFn, ckptOpts) {
|
|
|
435
468
|
else if (stage === 'fleet') faNext = 'done'
|
|
436
469
|
}
|
|
437
470
|
const faTier = (stage === 'router' && result && typeof result.tier === 'string') ? result.tier : FA_TIER.v
|
|
438
|
-
const
|
|
471
|
+
const faRunIdFile = FDIR + '/.fa-state/run-id'
|
|
472
|
+
const faMintRunId = stage === 'router'
|
|
473
|
+
? ('mkdir -p ' + shq(FDIR + '/.fa-state') + ' 2>/dev/null || true; date +%s%3N > ' + shq(faRunIdFile) + ' 2>/dev/null || true; ')
|
|
474
|
+
: ''
|
|
475
|
+
const faRecordCmd = faNext === null ? null : (faMintRunId + DZ + ' statusline --fa-record --slug ' + shq(SLUG) + ' --step ' + shq(faNext) + (faTier ? ' --tier ' + shq(faTier) : '') + ' --mode ' + shq(MODE) + ' --project ' + shq(REPO) + ' --run-id "$(cat ' + shq(faRunIdFile) + ' 2>/dev/null || true)"')
|
|
439
476
|
var faReported = false
|
|
440
477
|
if (CHECKPOINTS_ON && result !== null && result !== undefined && !partial && persistable) {
|
|
441
478
|
let line = null
|
|
@@ -970,7 +1007,14 @@ async function appendRunCostRow(stage, phaseName, outcome) {
|
|
|
970
1007
|
// Like capturePairs, a ledger failure is a logged SECONDARY event that can NEVER fail the run;
|
|
971
1008
|
// the whole body therefore rides one best-effort try/catch and never rethrows.
|
|
972
1009
|
try {
|
|
973
|
-
|
|
1010
|
+
const spentNow = RUNTIME_SPENT ? RUNTIME_SPENT() : null
|
|
1011
|
+
const tokensOut = Number.isFinite(spentNow) && Number.isFinite(spentAtPrevRow)
|
|
1012
|
+
? spentNow - spentAtPrevRow
|
|
1013
|
+
: null
|
|
1014
|
+
if (Number.isFinite(spentNow)) spentAtPrevRow = spentNow
|
|
1015
|
+
// HONESTY: budget.spent() exposes only cumulative OUTPUT tokens, so tokensOut is that measured
|
|
1016
|
+
// partial cost. Complete token cost, minutes and agents come from the Workflow COMPLETION NOTIFICATION,
|
|
1017
|
+
// which the running script CANNOT see. Keep their fields null: never an estimate, never a guess, never a fabricated number.
|
|
974
1018
|
const line = JSON.stringify({
|
|
975
1019
|
slug: (typeof SLUG === 'string' && SLUG !== '') ? SLUG : null,
|
|
976
1020
|
stage: (typeof stage === 'string' && stage !== '') ? stage : null,
|
|
@@ -978,6 +1022,9 @@ async function appendRunCostRow(stage, phaseName, outcome) {
|
|
|
978
1022
|
tokens: null,
|
|
979
1023
|
minutes: null,
|
|
980
1024
|
agents: null,
|
|
1025
|
+
tokensOut: tokensOut,
|
|
1026
|
+
tokensOutSource: tokensOut === null ? 'unavailable' : 'budget.spent',
|
|
1027
|
+
costNote: 'total tokens visible only in the completion notification; minutes not measurable inside the sandbox',
|
|
981
1028
|
coder: (typeof coderUsed === 'string' && coderUsed !== '') ? coderUsed : null,
|
|
982
1029
|
grade: (qe && typeof qe.grade === 'string' && qe.grade !== '') ? qe.grade : null,
|
|
983
1030
|
outcome: (typeof outcome === 'string' && outcome !== '') ? outcome : null,
|
|
@@ -1098,9 +1145,9 @@ function stageReason(stage, decision) {
|
|
|
1098
1145
|
if (decision.reason === 'explicit-models' && AUTOCOST[stage]) return 'auto-cost'
|
|
1099
1146
|
return decision.reason
|
|
1100
1147
|
}
|
|
1101
|
-
const KNOWN_CODEX = { 'auto': 1, 'gpt-5.5': 1, 'gpt-5.6': 1, 'gpt-5.6-luna': 1, 'gpt-5.6-terra': 1, 'gpt-5.6-sol': 1 }
|
|
1148
|
+
const KNOWN_CODEX = { 'auto': 1, 'gpt-5.5': 1, 'gpt-5.6': 1, 'gpt-5.6-luna': 1, 'gpt-5.6-terra': 1, 'gpt-5.6-sol': 1, 'gpt-6-astra': 1 }
|
|
1102
1149
|
// The allowlist is not an availability check — probe every id before every run; ids drift in both directions.
|
|
1103
|
-
const CODEX_TIERS = { flagship: 'gpt-5.6-sol', workhorse: 'gpt-5.6-terra', 'high-volume': 'gpt-5.6-luna' }
|
|
1150
|
+
const CODEX_TIERS = { premium: 'gpt-6-astra', flagship: 'gpt-5.6-sol', workhorse: 'gpt-5.6-terra', 'high-volume': 'gpt-5.6-luna' }
|
|
1104
1151
|
const CLAUDE_NAMES = { fable: 1, opus: 1, sonnet: 1, haiku: 1 }
|
|
1105
1152
|
const VALID_REASONING = { none: 1, minimal: 1, low: 1, medium: 1, high: 1, xhigh: 1, max: 1 }
|
|
1106
1153
|
const DEFAULT_MODELS = { router: 'fable', requirements: 'sonnet', research: 'sonnet', adr: 'opus', ideation: 'sonnet', ddd: 'opus', architecture: 'opus', plan: 'sonnet', code: null, qe: null, fleet: 'sonnet' }
|
|
@@ -1238,9 +1285,26 @@ function budgetTable(primary, mode) {
|
|
|
1238
1285
|
codexHalf = { ...ROUTING_TABLES.claude.codex[mode.codex], qe: A.codexAvailable === false ? 'opus' : qeSpec }
|
|
1239
1286
|
} else {
|
|
1240
1287
|
const normal = mode.codex === 'normal'
|
|
1241
|
-
|
|
1242
|
-
|
|
1243
|
-
|
|
1288
|
+
// Тир берётся из FA_TIER.v, а НЕ из A.tier: A.tier несёт только ЯВНО переданный аргумент,
|
|
1289
|
+
// а обычный прогон узнаёт свой размер от нулевого шага (роутера), который кладёт его сюда
|
|
1290
|
+
// на строке FA_TIER.v = tier. Читая A.tier, ячейка плана никогда не поднималась до
|
|
1291
|
+
// премиального яруса в обычном прогоне — матрица работала наполовину (ИЗМЕРЕНО 2026-09-09).
|
|
1292
|
+
const largePlan = FA_TIER.v === 'L' || FA_TIER.v === 'XL'
|
|
1293
|
+
const work = 'codex:' + codexIdForTier(normal ? 'flagship' : 'workhorse') + ':high'
|
|
1294
|
+
const evidence = 'codex:' + codexIdForTier(normal ? 'workhorse' : 'high-volume') + ':medium'
|
|
1295
|
+
codexHalf = {
|
|
1296
|
+
router: evidence,
|
|
1297
|
+
requirements: 'codex:' + codexIdForTier(normal ? 'flagship' : 'workhorse') + ':medium',
|
|
1298
|
+
research: evidence,
|
|
1299
|
+
adr: 'codex:' + codexIdForTier(normal ? 'premium' : 'flagship') + ':high',
|
|
1300
|
+
ideation: work,
|
|
1301
|
+
ddd: work,
|
|
1302
|
+
architecture: 'codex:' + codexIdForTier(normal ? 'premium' : 'flagship') + ':high',
|
|
1303
|
+
plan: largePlan ? 'codex:' + codexIdForTier(normal ? 'premium' : 'flagship') + ':high' : work,
|
|
1304
|
+
code: work,
|
|
1305
|
+
fleet: work,
|
|
1306
|
+
}
|
|
1307
|
+
|
|
1244
1308
|
}
|
|
1245
1309
|
return { ...claudeHalf, ...codexHalf }
|
|
1246
1310
|
}
|
|
@@ -2237,17 +2301,18 @@ const CODE_LANDING_CEILING_ENV = 'DZ_FEATURE_ADR_CODE_LANDING_CEILING_MS'
|
|
|
2237
2301
|
const CODEX_COMPANION_SCRIPT = '/root/.claude/plugins/cache/openai-codex/codex/1.0.5/scripts/codex-companion.mjs'
|
|
2238
2302
|
const CODEX_COMPANION_STATE_ROOT = '/root/.claude/plugins/data/codex-openai-codex/state'
|
|
2239
2303
|
|
|
2240
|
-
function decideCodeLandingLiveness(input) {
|
|
2304
|
+
function decideCodeLandingLiveness(input, pidProbe) {
|
|
2241
2305
|
const status = typeof input.companionStatus === 'string' ? input.companionStatus.trim().toLowerCase() : ''
|
|
2242
2306
|
const elapsedMs = Number.isFinite(input.elapsedMs) ? Math.max(0, input.elapsedMs) : 0
|
|
2243
2307
|
const ceilingMs = Number.isFinite(input.ceilingMs) && input.ceilingMs > 0 ? input.ceilingMs : DEFAULT_CODE_LANDING_CEILING_MS
|
|
2308
|
+
const recordedPidAlive = input.recordedPidAlive === undefined ? (input.recordedPid === undefined ? null : pidProbe(input.recordedPid)) : input.recordedPidAlive
|
|
2244
2309
|
const live = status === 'running' || status === 'queued'
|
|
2245
2310
|
const terminal = status === 'completed' || status === 'failed' || status === 'cancelled'
|
|
2246
2311
|
|
|
2247
|
-
if (live &&
|
|
2312
|
+
if (live && recordedPidAlive === false) {
|
|
2248
2313
|
return { verdict: 'dead-worker', reason: 'recorded-pid-absent' }
|
|
2249
2314
|
}
|
|
2250
|
-
if (live &&
|
|
2315
|
+
if (live && recordedPidAlive === true) {
|
|
2251
2316
|
if (elapsedMs >= ceilingMs) return { verdict: 'inconclusive', reason: 'ceiling-exceeded' }
|
|
2252
2317
|
return { verdict: 'coder-running', reason: 'recorded-pid-alive' }
|
|
2253
2318
|
}
|
|
@@ -2753,7 +2818,8 @@ function codeLandingLivenessProbeCmd(repo, plan, baselineAbsPath, jobId, waitSec
|
|
|
2753
2818
|
// cannot answer) — a different event from an expired window, and the only one with a cure.
|
|
2754
2819
|
// Unreadable stays 'unknown', which the verdict treats as no evidence, never as a clean exit.
|
|
2755
2820
|
'tf=unknown; if [ -f "$state" ]; then if grep -q \'"touchedFiles": *\\[ *\\]\' "$state"; then tf=0; elif grep -q \'"touchedFiles"\' "$state"; then tf=1; fi; fi; ' +
|
|
2756
|
-
|
|
2821
|
+
// Replaces ps -p "$pid": the CLI uses the shared probePid; inaccessible stays unknown.
|
|
2822
|
+
'if [ -n "$pid" ]; then pid_alive=$(' + DZ + ' runs --probe-pid "$pid"); case "$pid_alive" in true|false|unknown) :;; *) pid_alive=unknown;; esac; fi; fi; fi; ' +
|
|
2757
2823
|
'echo "CODEX-LIVENESS-SIGNAL companion=$companion pid-alive=$pid_alive targets-changed=$target elapsed-ms=$elapsed_ms ceiling-ms=$ceiling_ms start-ms=$start_ms touched-files=$tf"; ' +
|
|
2758
2824
|
'printf "%s\n" "$landing" | head -40; rm -rf "$sc"'
|
|
2759
2825
|
)
|
|
@@ -3056,7 +3122,34 @@ const ARTIFACT = { type: 'object', additionalProperties: false, required: ['wrot
|
|
|
3056
3122
|
// (hasManifest + the who-injected report). The BIG per-stage guidance content is fetched by each stage
|
|
3057
3123
|
// agent directly from `dz project-skills` (never threaded through a model → fidelity preserved).
|
|
3058
3124
|
const PROJECT_SKILLS = { type: 'object', additionalProperties: false, required: ['hasManifest', 'report'], properties: { hasManifest: { type: 'boolean' }, report: { type: 'string' } } }
|
|
3059
|
-
const QE = { type: 'object', additionalProperties: false, required: ['grade', 'gaps', 'codeTestsAdequate', 'docTestsPresent'], properties: { grade: { type: 'string' }, codeTestsAdequate: { type: 'boolean' }, docTestsPresent: { type: 'boolean' }, gaps: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['sev', 'what'], properties: { sev: { type: 'string' }, what: { type: 'string' } } } }, claimCheck: { type: 'object', additionalProperties: false, properties: { findings: { type: 'number' }, high: { type: 'number' }, medium: { type: 'number' } } } } }
|
|
3125
|
+
const QE = { type: 'object', additionalProperties: false, required: ['grade', 'gaps', 'codeTestsAdequate', 'docTestsPresent'], properties: { grade: { type: 'string' }, codeTestsAdequate: { type: 'boolean' }, docTestsPresent: { type: 'boolean' }, gaps: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['sev', 'what'], properties: { sev: { type: 'string' }, what: { type: 'string' } } } }, claimCheck: { type: 'object', additionalProperties: false, properties: { findings: { type: 'number' }, high: { type: 'number' }, medium: { type: 'number' } } }, roundLessons: { type: 'array', items: { type: 'string' } }, roundNoNewKnowledge: { type: 'string' } } }
|
|
3126
|
+
const CONFIRMATION_FILE_GATE = { type: 'object', additionalProperties: false, required: ['verdict', 'missing', 'checked', 'reason'], properties: { verdict: { type: 'string', enum: ['pass', 'fail', 'skipped', 'refused'] }, missing: { type: 'array', items: { type: 'string' } }, checked: { type: 'array', items: { type: 'string' } }, reason: { type: 'string' } } }
|
|
3127
|
+
|
|
3128
|
+
function normalizeConfirmationFileGate(raw) {
|
|
3129
|
+
if (!raw || typeof raw !== 'object') return { verdict: 'refused', missing: [], checked: [], reason: 'confirmation-file gate agent returned no readable result' }
|
|
3130
|
+
const verdict = raw.verdict
|
|
3131
|
+
if (verdict !== 'pass' && verdict !== 'fail' && verdict !== 'skipped' && verdict !== 'refused') return { verdict: 'refused', missing: [], checked: [], reason: 'confirmation-file gate returned an invalid verdict' }
|
|
3132
|
+
return {
|
|
3133
|
+
verdict: verdict,
|
|
3134
|
+
missing: Array.isArray(raw.missing) ? raw.missing.map(function (x) { return String(x) }) : [],
|
|
3135
|
+
checked: Array.isArray(raw.checked) ? raw.checked.map(function (x) { return String(x) }) : [],
|
|
3136
|
+
reason: typeof raw.reason === 'string' ? raw.reason : ''
|
|
3137
|
+
}
|
|
3138
|
+
}
|
|
3139
|
+
|
|
3140
|
+
function enforceConfirmationFileGate(qe, gate) {
|
|
3141
|
+
if (!qe || typeof qe !== 'object') return qe
|
|
3142
|
+
const out = Object.assign({}, qe, { confirmationFileGate: gate })
|
|
3143
|
+
if (gate.verdict !== 'fail' && gate.verdict !== 'refused') return out
|
|
3144
|
+
const what = gate.verdict === 'fail'
|
|
3145
|
+
? 'confirmation file gate FAIL — missing: ' + gate.missing.join(', ')
|
|
3146
|
+
: 'confirmation file gate REFUSED — ' + gate.reason
|
|
3147
|
+
const gaps = Array.isArray(out.gaps) ? out.gaps.slice() : []
|
|
3148
|
+
if (!gaps.some(function (g) { return g && g.what === what })) gaps.push({ sev: 'HIGH', what: what })
|
|
3149
|
+
out.gaps = gaps
|
|
3150
|
+
out.grade = String(out.grade || '').trim().toUpperCase() === 'D' ? 'D' : 'C'
|
|
3151
|
+
return out
|
|
3152
|
+
}
|
|
3060
3153
|
const ADR_TEMPLATE_GUIDE = 'ADR best-practices for Step 3: emit exactly one decision per ADR with the invariant core Title, Status, Context, Decision, Consequences. Template weight is tier-routed: S/M use Nygard/ITD-lightweight form but still include decision drivers, considered options, rationale, consequences, and Confirmation; L/XL use MADR structure plus an NHS Wales Confirmation stanza. Confirmation MUST name verification method, monitoring, success metric, and owner, and its load-bearing safety property MUST be tied to a Step-8 automated test/fitness function. Use status vocabulary proposed/accepted/rejected/deprecated/superseded plus a reversibility clause. Context must be neutral and appear before Decision. Considered Options must include rejected options with symmetric pros/cons. Rationale points must map to stated drivers and explain why losers were rejected. Consequences must include positive and negative outcomes/accepted downsides, follow-up ADR links, an after-action review schedule, and supersession discipline: supersession mints a new ADR and never edits accepted/rejected ADR content in place. Decision must be concrete/testable with exact names, versions, formats, paths, commands, or APIs. Reject explainer-masquerading-as-ADR: a domain overview with no concrete Decision is not an ADR. File names under 03_adr MUST be sequential NNN-{decision-slug}.md with lowercase kebab-case, dateless, ticketless slugs (the auto-001 ADR tracks the feature slug, so the present-tense imperative signal lives in the ADR Title; model-named additional ADRs use imperative slugs). Add a ## Links traceability block (requirements, driving use case, related ADRs) and a one-line provenance note (model-generated, edited for clarity); for a long ADR include a top-of-file table of contents.'
|
|
3061
3154
|
const ADR_FITNESS_CHECKLIST = 'ADR fitness checklist for Step 8: read every ' + FDIR + '/03_adr/NNN-*.md ADR and fail the QE gate for any miss. Required checks: (1) filename is 03_adr/NNN-{decision-slug}.md where the slug is lowercase kebab-case, imperative, dateless, and ticketless; (2) title is decision-shaped and the ADR records one decision only; (3) Status is non-empty controlled vocabulary proposed/accepted/rejected/deprecated/superseded and includes a reversibility/revisit clause; (4) Context is neutral, problem-first, and appears before Decision; (5) Decision Drivers are stated and ranked/weighted; (6) Considered Options include the chosen and rejected options, each with symmetric pros and cons; (7) Rationale maps each point back to a driver and explains why rejected options lost; (8) Decision is concrete/testable with exact names, versions, formats, paths, commands, or APIs; (9) Consequences include positive and negative outcomes/accepted downsides, follow-up ADR links, and an after-action review schedule; (10) Confirmation names verification method, monitoring, success metric, and owner, then links the load-bearing safety property to an automated test/fitness function; (11) no placeholder text, template hints, raw generation scaffolding, or fake Markdown structure; (12) reject explainer-masquerading-as-ADR: describing a space with no concrete Decision is a blocker; (13) a Related/Links traceability block maps the ADR to its requirements, driving use case, and related ADRs. The ADR Confirmation check is load-bearing: assert the named safety property has a test that DISCRIMINATES — a real test by file/name that would go RED if the protection were deleted (the discrimination + mutation gates below are the proof; a test that would still pass with the protection deleted is documentation, not a gate); if absent, grade no better than C and record a blocker gap.'
|
|
3062
3155
|
// §42 test-discrimination gate (feature step8-discrimination-gate, grounded in cve-bench/evaluate.mjs). Asserting
|
|
@@ -3088,8 +3181,55 @@ const AMENDMENT_RULE = 'AMENDMENT CONFIRMATION DISCIPLINE (every amendment is a
|
|
|
3088
3181
|
const AMENDMENT_GATE = 'AMENDMENT GATE (P2): do NOT judge this yourself — RUN the check and report what it says. Via Bash run EXACTLY `' + DZ + ' amendment-check --slug ' + SLUG + ' --json` (add `--feature-dir ' + FDIR + '` if the slug does not resolve from your CWD). Parse the JSON and report `amendments: {outcome, counts, reasons}` in your return object. outcome `pass` or `skip` clears the gate; `fail` is a HIGH gap and every reason must be quoted verbatim into the QE report; `not-established` means the check could not be run or the grammar matched nothing — that is NEVER a pass, report it as inconclusive with the tool error. Empty stdout, a crash, or a missing `dz` is `not-established`, not a clean gate. This check proves each amendment RESOLVES to a real test; it does NOT prove the test discriminates — vacuity stays with the discrimination gate above. ' +
|
|
3089
3182
|
'IO-ON-PURE-PATH + FIXTURE-SWAP HUNT (P5): in the test diff, hunt for replacements of broken/unbound fixtures with healthy ones — the old fixture was probably a NEGATIVE CONTROL proving a path was I/O-free; each such swap requires a compensating negative resource-down test. If the code diff adds I/O (DB/network/file) to a previously-pure path — especially startup/lifespan/health — require a negative resource-down test (broken/unbound resource → the path degrades per its declared contract: fail-open for advisory, explicit fail-fast for load-bearing). Missing → HIGH gap.'
|
|
3090
3183
|
|
|
3184
|
+
// ── BEGIN BLOB run-registry@1.0.0 sha256:a4187c278260e8f8285fc494da2a6d82923efe4465fabba20ce1d3141df7fef3 src=packages/@dzhechkov/harness-core/src/run-registry.ts ──
|
|
3185
|
+
function runRecordCommand(dz, root, event, runId, slug, pid, parentRunId, outcome) {
|
|
3186
|
+
const quote = (s) => "'" + s.replace(/'/g, "'\\''") + "'";
|
|
3187
|
+
let cmd = dz + ' runs-record --project ' + quote(root) + ' --event ' + quote(event) + (runId ? ' --run-id ' + quote(runId) : '');
|
|
3188
|
+
if (event === 'started') {
|
|
3189
|
+
cmd += ' --kind feature-adr --slug ' + quote(slug);
|
|
3190
|
+
cmd += ' --pid ' + quote(pid === null ? 'host' : String(pid));
|
|
3191
|
+
if (parentRunId)
|
|
3192
|
+
cmd += ' --parent-run-id ' + quote(parentRunId);
|
|
3193
|
+
}
|
|
3194
|
+
if (event === 'finished')
|
|
3195
|
+
cmd += ' --outcome ' + quote(outcome);
|
|
3196
|
+
return cmd + ' --json';
|
|
3197
|
+
}
|
|
3198
|
+
// ── END BLOB run-registry ──
|
|
3199
|
+
|
|
3200
|
+
// Run registry: the courier executes the CLI because the sandbox has no filesystem.
|
|
3201
|
+
let registryRunId = ''
|
|
3202
|
+
let registryPhase = 'Router'
|
|
3203
|
+
async function recordRegistryEvent(event, phaseName, outcome) {
|
|
3204
|
+
registryPhase = phaseName
|
|
3205
|
+
if (event !== 'started' && !registryRunId) return
|
|
3206
|
+
try {
|
|
3207
|
+
const cmd = runRecordCommand(DZ, REPO, event, registryRunId, SLUG,
|
|
3208
|
+
A.runPid === undefined ? null : A.runPid, A.parentRunId || null, outcome || '')
|
|
3209
|
+
const out = await dispatchAgent(newRung(), 'Run EXACTLY this command via Bash and return ONLY its stdout: ' + cmd,
|
|
3210
|
+
{ label: 'runs-record:' + event + ':' + phaseName, phase: phaseName, effort: 'low' })
|
|
3211
|
+
let receipt = null
|
|
3212
|
+
try { receipt = typeof out === 'string' ? JSON.parse(out.trim()) : out } catch { /* unverified below */ }
|
|
3213
|
+
if (!receipt || receipt.status !== 'written' || receipt.event !== event ||
|
|
3214
|
+
typeof receipt.runId !== 'string' || !receipt.runId || (registryRunId && receipt.runId !== registryRunId)) {
|
|
3215
|
+
registryOutcome = 'unverified'
|
|
3216
|
+
log('run registry: ' + event + ' UNVERIFIED — ' + (receipt && receipt.reason ? String(receipt.reason) : String(out)))
|
|
3217
|
+
return
|
|
3218
|
+
}
|
|
3219
|
+
if (event === 'started') registryRunId = receipt.runId
|
|
3220
|
+
} catch (error) {
|
|
3221
|
+
registryOutcome = 'unverified'
|
|
3222
|
+
log('run registry: ' + event + ' UNVERIFIED — ' + String(error))
|
|
3223
|
+
}
|
|
3224
|
+
}
|
|
3225
|
+
finishRunRegistry = async function () {
|
|
3226
|
+
await recordRegistryEvent('finished', registryPhase, registryOutcome)
|
|
3227
|
+
}
|
|
3228
|
+
await recordRegistryEvent('started', 'Router')
|
|
3229
|
+
|
|
3091
3230
|
// Step 0: Router + MANDATORY self-learning recall
|
|
3092
3231
|
phase('Router')
|
|
3232
|
+
await recordRegistryEvent('heartbeat', 'Router')
|
|
3093
3233
|
|
|
3094
3234
|
// W1 (backlog 848853a0): REPO must be the git TOPLEVEL. Both measured incidents were a REPO
|
|
3095
3235
|
// pointing INSIDE the repository (packages/@dzhechkov/health-advisor) — artifacts then scatter
|
|
@@ -3110,6 +3250,7 @@ else if (wrootTop !== wrootHere) {
|
|
|
3110
3250
|
log('REPO ROOT MISMATCH: REPO canonicalizes to ' + wrootHere + ' but the git toplevel is ' + wrootTop + ' — refusing before any design spend (the measured incident class: artifacts scattered into a subdirectory features/)')
|
|
3111
3251
|
const repoRootMismatchGates = {}
|
|
3112
3252
|
const repoRootMismatchOutcome = runOutcomeOf({ phase: 'repo-root-mismatch', gates: repoRootMismatchGates })
|
|
3253
|
+
if (registryOutcome !== 'unverified') registryOutcome = repoRootMismatchOutcome
|
|
3113
3254
|
return { phase: 'repo-root-mismatch', outcome: repoRootMismatchOutcome, repo: REPO, repoCanonical: wrootHere, gitToplevel: wrootTop, cure: 'invoke with args.repo=' + wrootTop + ' (or run from the repository root)' }
|
|
3114
3255
|
}
|
|
3115
3256
|
await loadCheckpoints('Router')
|
|
@@ -3148,6 +3289,8 @@ FA_TIER.v = tier // fa-phase-statusline: from here every ckpt-side fa-record car
|
|
|
3148
3289
|
// Outer completion state starts absent so the plan-only ledger row can report null honestly.
|
|
3149
3290
|
let coderUsed = null
|
|
3150
3291
|
let qe = null
|
|
3292
|
+
let pipelineRound = null
|
|
3293
|
+
let roundClosed = false
|
|
3151
3294
|
const LEARNED = router ? router.rationale : 'none recalled'
|
|
3152
3295
|
const isMplus = tier === 'M' || tier === 'L' || tier === 'XL'
|
|
3153
3296
|
const isLplus = tier === 'L' || tier === 'XL'
|
|
@@ -3191,6 +3334,26 @@ if (autoCostStages.length > 0) {
|
|
|
3191
3334
|
// most visible moment. Uses the workspace bin (PATH-independent). Best-effort — never blocks.
|
|
3192
3335
|
if (resumedStages.indexOf('router') === -1) await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and report its stdout verbatim — do nothing else, do not summarize: ' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 0 recall" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored 0 --mode ' + MODE + ' --project ' + REPO, { label: 'fa-record:step0', phase: 'Router', effort: 'low' })
|
|
3193
3336
|
|
|
3337
|
+
// A whole feature-adr run is one outer round. Cost remains in the unchanged per-stage ledger rows;
|
|
3338
|
+
// the round records only the outcome. State lives in REPO because the command cd's there, while
|
|
3339
|
+
// recall reads the canonical BRAIN and carries the same fa:<slug> attribution as DZ_RECALL.
|
|
3340
|
+
const roundOwnerArg = registryRunId
|
|
3341
|
+
? ' --owner-run ' + shq(registryRunId)
|
|
3342
|
+
: ''
|
|
3343
|
+
if (!registryRunId) log('round owner: no registry run id (registry write failed) — falling back to explicit owner')
|
|
3344
|
+
const roundOpenCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' round open --slug ' + shq(SLUG) + ' --round auto --topic ' + shq(DESC) + ' --project ' + shq(BRAIN) + ' --run fa:' + SLUG + roundOwnerArg + ' --json'
|
|
3345
|
+
const roundOpenOut = await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and return its stdout VERBATIM, nothing else: ' + roundOpenCmd, { label: 'round:open', phase: 'Router', effort: 'low' })
|
|
3346
|
+
const roundOpenReceipt = parseRoundCommandJson(roundOpenOut)
|
|
3347
|
+
pipelineRound = Number(roundOpenReceipt && roundOpenReceipt.state ? roundOpenReceipt.state.round : (roundOpenReceipt ? roundOpenReceipt.round : NaN))
|
|
3348
|
+
if (!Number.isInteger(pipelineRound) || pipelineRound < 1) {
|
|
3349
|
+
pipelineRound = null
|
|
3350
|
+
log('round open refused: ' + (roundOpenReceipt && roundOpenReceipt.message ? roundOpenReceipt.message : String(roundOpenOut || 'no JSON receipt')))
|
|
3351
|
+
} else if (!roundOpenReceipt.state) {
|
|
3352
|
+
// A resumed run sees the still-open state and receives the selected auto number in the refusal.
|
|
3353
|
+
// Keep that number so Step 8 can close the original round, but never call the refusal an open.
|
|
3354
|
+
log('round open refused: ' + String(roundOpenReceipt.message || 'round already open'))
|
|
3355
|
+
}
|
|
3356
|
+
|
|
3194
3357
|
// R1 product-architecture-lens (ADR-001 Decision 3): forward-looking сверка of THIS feature vs the LIVE
|
|
3195
3358
|
// product map + vision. NON-BLOCKING/soft by design — it LOGS {signal,confidence} so a real command
|
|
3196
3359
|
// duplication or vision-boundary tension is visible at Step 0; the hard-stop call stays the user's (a
|
|
@@ -3241,6 +3404,7 @@ const PS_GUIDANCE = (stage) => POLY.hasManifest
|
|
|
3241
3404
|
|
|
3242
3405
|
// Steps 1-5: Design (tier-gated thunks built explicitly - no inline ternary-null)
|
|
3243
3406
|
phase('Design')
|
|
3407
|
+
await recordRegistryEvent('heartbeat', 'Design')
|
|
3244
3408
|
await usageProbe('Design')
|
|
3245
3409
|
const designThunks = []
|
|
3246
3410
|
const reqExtra = isLplus ? ' Also write ' + FDIR + '/02_research.md (codebase patterns + external analogues; read the repo for the closest existing implementation to mirror).' : ''
|
|
@@ -3482,6 +3646,7 @@ if (!fanVerdict.complete) {
|
|
|
3482
3646
|
// (coderUsed/qe are the outer bindings, both still null here, so the row reports null honestly.)
|
|
3483
3647
|
const designIncompleteGates = { design: fanVerdict.reason === 'probe-not-established' ? 'not-established' : 'incomplete', plan: 'not-run', planCompleteness: 'not-run', challengePanel: 'not-run', code: 'not-run', qe: 'not-run' }
|
|
3484
3648
|
const designIncompleteOutcome = runOutcomeOf({ phase: 'design-incomplete', gates: designIncompleteGates })
|
|
3649
|
+
if (registryOutcome !== 'unverified') registryOutcome = designIncompleteOutcome
|
|
3485
3650
|
await appendRunCostRow('design-gate', 'Design', designIncompleteOutcome)
|
|
3486
3651
|
return { tier: tier, phase: 'design-incomplete', outcome: designIncompleteOutcome, slug: SLUG, artifactsDir: FDIR, missingSubstages: fanVerdict.missingSubstages, missingArtifacts: fanVerdict.missingArtifacts, reason: fanVerdict.reason, modelsUsed: modelsUsed, dispatchOutcomes: dispatchOutcomes, gates: designIncompleteGates, resumedStages: resumedStages, checkpointing: CHECKPOINTS_ON ? RESUME_MODE : 'off', trainingPairs: CAPTURE_PAIRS ? TP_DIR : 'off', captureFailures: captureFailures, recordFailures: recordFailures, decisionRecallFailures: decisionRecallFailures, usageEvents: usageEvents, usageThreshold: USAGE_THRESHOLD, polymorphism: POLY.hasManifest ? POLY.report : null, note: 'REFUSED at the Step-5/6 boundary: ' + what + ', so the design is incomplete and Step 6 was NOT dispatched. Planning off a partial design produces a plan with no ADR behind it. ' + repair + ' If a sibling died on a Claude limit, add usage-adaptive routing or route that stage to Codex first (args.models). To rebuild the whole design from scratch instead, re-invoke with args.resume=\'never\'.' }
|
|
3487
3652
|
}
|
|
@@ -3491,6 +3656,7 @@ if (!fanVerdict.complete) {
|
|
|
3491
3656
|
// the codex:codex-rescue runtime and GRACEFULLY FALL BACK to the default (Claude) planner if Codex is
|
|
3492
3657
|
// unavailable/errors — the pipeline never blocks on Codex.
|
|
3493
3658
|
phase('Plan')
|
|
3659
|
+
await recordRegistryEvent('heartbeat', 'Plan')
|
|
3494
3660
|
await usageProbe('Plan')
|
|
3495
3661
|
const planPrompt = 'Step 6 (SPARC-GOAP implementation plan) of /feature-adr for "' + DESC + '" (' + SLUG + ', tier ' + tier + '). Given the requirements + ADR + architecture in ' + FDIR + ', decompose into milestones + concrete tasks with success metrics. Write ' + FDIR + '/06_implementation_plan.md. END the plan with a trailing `EXPECTED_CODE_TARGETS:` block listing, one per line as `- <repo-relative path>`, EVERY production/test/config/doc file Step 7 is expected to create or modify. This block is machine-read by the Step-7.5 landing barrier: only paths it ESTABLISHES can ever count as landed, so an absent or unpollable block makes the barrier verdict INCONCLUSIVE. List only real targets outside features/, .dz/, .agentic-qe/ and roam/. The K2 plan-completeness gate blocks Step 7 until the plan satisfies these too, so write them in as you author, not afterwards: (C1) every ADR under 03_adr/ is cited as `ADR-<n>` by the task that implements it; (C2) every test path named in an ADR Confirmation stanza appears verbatim in the plan, bound to the task that writes it; (C4) every acid token `A<n>` from 00_complexity_assessment.md is named verbatim, bound to its owning task and to the test that proves the refusal. If any corrections from Step 3.5 (a CONDITIONAL verdict) or other sources are folded into this plan, carry them in a `## Amendments` section. ' + AMENDMENT_RULE + ' Return wrote[] + summary.' + ABSOLUTE_PATH_NOTE + WRITE_DISCIPLINE
|
|
3496
3662
|
const planContext = buildDecisionContext({ slug: SLUG, decisionKind: 'plan-route-selection', description: DESC, tier: tier, codeHint: CODE_HINT, upstreamDigest: fnv1a64(JSON.stringify(design === undefined ? null : design)) })
|
|
@@ -3686,6 +3852,7 @@ if (planGate.verdict !== 'pass') {
|
|
|
3686
3852
|
// checkpoint deliberately: an incomplete plan is not something to steer, it is something to fix.
|
|
3687
3853
|
const planGateFailedGates = { plan: (plan ? 'produced' : 'missing'), planCompleteness: planGate.verdict, challengePanel: 'not-run', code: 'not-run', qe: 'not-run' }
|
|
3688
3854
|
const planGateFailedOutcome = runOutcomeOf({ phase: 'plan-gate-failed', gates: planGateFailedGates })
|
|
3855
|
+
if (registryOutcome !== 'unverified') registryOutcome = planGateFailedOutcome
|
|
3689
3856
|
await appendRunCostRow('plan-gate', 'Plan', planGateFailedOutcome)
|
|
3690
3857
|
return { tier: tier, phase: 'plan-gate-failed', outcome: planGateFailedOutcome, slug: SLUG, artifactsDir: FDIR, planner: (plan ? plan.planner : null), plan: (plan ? plan.summary : null), modelsUsed: modelsUsed, dispatchOutcomes: dispatchOutcomes, planGate: planGate, gates: planGateFailedGates, resumedStages: resumedStages, checkpointing: CHECKPOINTS_ON ? RESUME_MODE : 'off', trainingPairs: CAPTURE_PAIRS ? TP_DIR : 'off', captureFailures: captureFailures, recordFailures: recordFailures, decisionRecallFailures: decisionRecallFailures, usageEvents: usageEvents, usageThreshold: USAGE_THRESHOLD, polymorphism: POLY.hasManifest ? POLY.report : null, note: refusalNoteFor(planGate, SLUG) }
|
|
3691
3858
|
}
|
|
@@ -3728,6 +3895,7 @@ if (stopHere) {
|
|
|
3728
3895
|
// presence), never from prose, so a skipped gate shows as 'not-run' instead of being silently forgotten.
|
|
3729
3896
|
const planGates = { plan: (plan ? 'produced' : 'missing'), planCompleteness: planGate.verdict, challengePanel: (challengeVerdict ? 'ran' : 'not-run'), code: 'not-run', qe: 'not-run' }
|
|
3730
3897
|
const checkpointAfterPlanOutcome = runOutcomeOf({ phase: 'checkpoint-after-plan', gates: planGates })
|
|
3898
|
+
if (registryOutcome !== 'unverified') registryOutcome = checkpointAfterPlanOutcome
|
|
3731
3899
|
await appendRunCostRow('plan', 'Plan', checkpointAfterPlanOutcome)
|
|
3732
3900
|
return { tier: tier, phase: 'checkpoint-after-plan', outcome: checkpointAfterPlanOutcome, artifactsDir: FDIR, planner: (plan ? plan.planner : null), plan: (plan ? plan.summary : null), modelsUsed: plannedModels, dispatchOutcomes: dispatchOutcomes, challengeVerdict: challengeVerdict, gates: planGates, resumedStages: resumedStages, checkpointing: CHECKPOINTS_ON ? RESUME_MODE : 'off', trainingPairs: CAPTURE_PAIRS ? TP_DIR : 'off', captureFailures: captureFailures, recordFailures: recordFailures, decisionRecallFailures: decisionRecallFailures, usageEvents: usageEvents, usageThreshold: USAGE_THRESHOLD, polymorphism: POLY.hasManifest ? POLY.report : null, note: 'L/XL checkpoint - review the ADR + plan (+ the planned code/qe/fleet models) + the challenge panel verdict (advisory) + the gates line, then re-invoke with args.stopAfter="none" to implement + QE (durable checkpoints make the re-invoke resume router+design+plan instead of re-running them). Present the gates map as a `🚦 Gates:` line in the checkpoint banner, rendering the planCompleteness entry as `K2 plan-completeness ✓` (pass) / `✗` (fail) / `inconclusive`.' }
|
|
3733
3901
|
}
|
|
@@ -3764,6 +3932,8 @@ if (QE_SCOPE === 'uncommitted') {
|
|
|
3764
3932
|
}
|
|
3765
3933
|
|
|
3766
3934
|
phase('Code')
|
|
3935
|
+
|
|
3936
|
+
await recordRegistryEvent('heartbeat', 'Code')
|
|
3767
3937
|
await usageProbe('Code')
|
|
3768
3938
|
// GATE-ANSWERED preamble. MEASURED 2026-08-31 (job task-mtgrlrq7, feature storage-auth-classes):
|
|
3769
3939
|
// a Codex coder exited status=completed, exit 0, touchedFiles=[] because it ASKED the routing
|
|
@@ -3977,6 +4147,7 @@ if (resumedStages.indexOf('code') !== -1 && landedNote !== '') {
|
|
|
3977
4147
|
|
|
3978
4148
|
// Step 8: QE (brutal-honesty, agentic-qe) + MANDATORY teach
|
|
3979
4149
|
phase('QE')
|
|
4150
|
+
await recordRegistryEvent('heartbeat', 'QE')
|
|
3980
4151
|
|
|
3981
4152
|
// ── Writer-quiescence probe (feature qe-writer-quiescence, backlog 700b46a4) ─────────────────────
|
|
3982
4153
|
// Step-8 used to grade a MOVING tree (crossrt-1: a background worker wrote AFTER the verdict,
|
|
@@ -4017,8 +4188,20 @@ log('Step 8 writer-quiescence: ' + writerQuiescence.verdict + ' (windows: ' + (w
|
|
|
4017
4188
|
const wqNote = writerQuiescence.verdict === 'quiet'
|
|
4018
4189
|
? ' WRITER-QUIESCENCE: quiet (' + writerQuiescence.note + ').'
|
|
4019
4190
|
: ' WRITER-QUIESCENCE GATE (MANDATORY to acknowledge): ' + writerQuiescence.note + ' State this standing explicitly in 08_qe_report.md next to the grade.'
|
|
4191
|
+
const confirmationGatePrompt = 'STEP-8 CONFIRMATION FILE GATE. You are the shell because this workflow sandbox has no filesystem API. Work under repo ' + REPO + ' and inspect only direct canonical ADR Markdown files in ' + FDIR + '/03_adr/NNN-*.md. If 03_adr is absent or has zero direct ADR files, return verdict=skipped, reason="ADR нет, проверять нечего", missing=[], checked=[]. Otherwise read every ADR. Find its unique H2 whose line starts with `## Confirmation` (the suffix may be Russian), extract EVERY repo-relative test-file path named in that section, and refuse if the section/path cannot be parsed. For every extracted path run shell checks from ' + REPO + ': missing (`test -e` is false) goes in missing; an existing path that is not a readable regular file (`test -f`, `test -r`, and an actual read) returns verdict=refused with the exact path and reason. Never turn an unreadable path, directory, parse failure, empty stdout, or tool error into skipped. If missing is non-empty return verdict=fail; otherwise pass. This gate proves ONLY file existence/readability; do not enforce the other ADR checklist items. Return exactly {verdict, missing, checked, reason}.'
|
|
4192
|
+
const confirmationGateRaw = await dispatchAgent(newRung(), confirmationGatePrompt, { label: 'qe:confirmation-files', phase: 'QE', effort: 'low', schema: CONFIRMATION_FILE_GATE })
|
|
4193
|
+
const confirmationFileGate = normalizeConfirmationFileGate(confirmationGateRaw)
|
|
4194
|
+
const confirmationGateLine = confirmationFileGate.verdict === 'skipped'
|
|
4195
|
+
? 'Confirmation file gate: пропущено: ADR нет, проверять нечего'
|
|
4196
|
+
: confirmationFileGate.verdict === 'pass'
|
|
4197
|
+
? 'Confirmation file gate: PASS — checked ' + confirmationFileGate.checked.join(', ')
|
|
4198
|
+
: confirmationFileGate.verdict === 'fail'
|
|
4199
|
+
? 'Confirmation file gate: FAIL — missing ' + confirmationFileGate.missing.join(', ')
|
|
4200
|
+
: 'Confirmation file gate: REFUSED — ' + confirmationFileGate.reason
|
|
4201
|
+
log(confirmationGateLine)
|
|
4202
|
+
const confirmationGateNote = ' MANDATORY CONFIRMATION FILE GATE RESULT: `' + confirmationGateLine + '`. Write that as a separate line in 08_qe_report.md. The independent QE review MUST still run. If the gate verdict is fail or refused, the final Step-8 grade cannot be A or B; the workflow also enforces that after the reviewer returns. This gate proves only existence/readability; all other ADR checklist items remain advisory.'
|
|
4020
4203
|
await usageProbe('QE')
|
|
4021
|
-
const qePrompt = 'Step 8 (QE - brutal-honesty review, agentic-qe) of /feature-adr for "' + DESC + '" (' + SLUG + '). Adversarially review the SHIPPED code (read it): correctness, edge cases, error handling, and the LOAD-BEARING property the ADR named (ASSERT it has a test that DISCRIMINATES - the recurring lesson: a test that would still pass with the protection deleted is documentation, not a gate). Run this ADR gate before final grading: ' + ADR_FITNESS_CHECKLIST + ' ' + DISCRIMINATION_GATE + ' ' + MUTATION_GATE + ' ' + NO_STUBS_GATE + ' ' + AMENDMENT_GATE + ' Grade A/B/C/D honestly. Assess code-test adequacy + doc-test presence. List CONFIRMED gaps with severity. Write ' + FDIR + '/08_qe_report.md with the primary findings under the exact heading `## Primary QE pass` and an ADR Fitness Checklist section showing PASS/FAIL per ADR and evidence for the Confirmation-linked test. MANDATORY SELF-LEARNING STORE (close the loop, never skip): compare every candidate lesson against the Step-0 recalled LEARNED patterns above. Teach ONLY lessons NOT covered by Step-0 recall. On overlap, run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate; if --reinforce is unavailable, skip the duplicate teach and report `reinforced existing pattern <id>` in the QE report. Store every genuinely new lesson in the CANONICAL BRAIN store at `' + BRAIN + '` so it is NOT lost to a target repo you may have cd`d into. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal) — the `cd ' + BRAIN + ' &&` prefix + `--project ' + BRAIN + '` pin guarantee the lesson lands in the brain regardless of your CWD. Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` (run it verbatim via Bash, do not skip). Do NOT teach trivia or invent gaps. AUTHORING-TIME CLAIM-CHECK (Deliverable of claim-check-authoring-time): after writing ' + FDIR + '/08_qe_report.md, run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash, parse the {ok, findings, scanned} JSON, and report claimCheck: {findings: N, high: N, medium: N} (counts by severity) in your return object. TAG EVERY QUANTITATIVE CLAIM you write in the report using the convention the checker recognizes as honest — write "1131 tests pass (MEASURED — `npx vitest run`)", never a bare "1131 tests pass" — and where you QUOTE a forbidden phrase as an example (e.g. the retracted "100% passing" framing), backtick the literal so it reads as code, not an assertion, so your own compliant report scans clean. Return {grade, gaps, codeTestsAdequate, docTestsPresent, claimCheck}.' + ABSOLUTE_PATH_NOTE + landedNote + wqNote + PS_GUIDANCE('qe')
|
|
4204
|
+
const qePrompt = 'Step 8 (QE - brutal-honesty review, agentic-qe) of /feature-adr for "' + DESC + '" (' + SLUG + '). Adversarially review the SHIPPED code (read it): correctness, edge cases, error handling, and the LOAD-BEARING property the ADR named (ASSERT it has a test that DISCRIMINATES - the recurring lesson: a test that would still pass with the protection deleted is documentation, not a gate). Run this ADR gate before final grading: ' + ADR_FITNESS_CHECKLIST + ' ' + DISCRIMINATION_GATE + ' ' + MUTATION_GATE + ' ' + NO_STUBS_GATE + ' ' + AMENDMENT_GATE + ' Grade A/B/C/D honestly. Assess code-test adequacy + doc-test presence. List CONFIRMED gaps with severity. Write ' + FDIR + '/08_qe_report.md with the primary findings under the exact heading `## Primary QE pass` and an ADR Fitness Checklist section showing PASS/FAIL per ADR and evidence for the Confirmation-linked test. MANDATORY SELF-LEARNING STORE (close the loop, never skip): compare every candidate lesson against the Step-0 recalled LEARNED patterns above. Teach ONLY lessons NOT covered by Step-0 recall. On overlap, run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate; if --reinforce is unavailable, skip the duplicate teach and report `reinforced existing pattern <id>` in the QE report. Store every genuinely new lesson in the CANONICAL BRAIN store at `' + BRAIN + '` so it is NOT lost to a target repo you may have cd`d into. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal) — the `cd ' + BRAIN + ' &&` prefix + `--project ' + BRAIN + '` pin guarantee the lesson lands in the brain regardless of your CWD. Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` (run it verbatim via Bash, do not skip). Do NOT teach trivia or invent gaps. In the return object set roundLessons to the teach:<id> receipts successfully written in this Step 8; when there were none, return roundLessons:[] and a non-empty roundNoNewKnowledge reason derived from this review/reinforcement decision. AUTHORING-TIME CLAIM-CHECK (Deliverable of claim-check-authoring-time): after writing ' + FDIR + '/08_qe_report.md, run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash, parse the {ok, findings, scanned} JSON, and report claimCheck: {findings: N, high: N, medium: N} (counts by severity) in your return object. TAG EVERY QUANTITATIVE CLAIM you write in the report using the convention the checker recognizes as honest — write "1131 tests pass (MEASURED — `npx vitest run`)", never a bare "1131 tests pass" — and where you QUOTE a forbidden phrase as an example (e.g. the retracted "100% passing" framing), backtick the literal so it reads as code, not an assertion, so your own compliant report scans clean. Return {grade, gaps, codeTestsAdequate, docTestsPresent, claimCheck, roundLessons, roundNoNewKnowledge}.' + ABSOLUTE_PATH_NOTE + landedNote + wqNote + confirmationGateNote + PS_GUIDANCE('qe')
|
|
4022
4205
|
// CROSS-MODEL QE (load-bearing): resolveStageModel('qe') derives the OTHER family than the resolved
|
|
4023
4206
|
// coder when args.models.qe is unset (coder-codex ⇒ opus; coder-Claude ⇒ codex, or opus if codex absent).
|
|
4024
4207
|
// An explicit args.models.qe wins. A Claude qe spec is merged onto the qe-code-reviewer base (role
|
|
@@ -4049,7 +4232,7 @@ const qe2Spec = qePrecisionPassSpec(PRIMARY, BUDGET_MODE, tier)
|
|
|
4049
4232
|
// re-teach (the original run already stored its lessons — replaying teach would double-store).
|
|
4050
4233
|
// R6: the review SCOPE is part of what a QE verdict is about, so it enters the hash — a resume must
|
|
4051
4234
|
// not present a verdict obtained over one scope as if it had been obtained over another.
|
|
4052
|
-
const qeHash = ckptHash('qe', [fnv1a64(JSON.stringify(codeStage === undefined ? null : codeStage)), tier, DESC, QE_REVIEWER, MODELS.qe === undefined ? null : MODELS.qe, CODEX_MODEL, coderUsed, PRIMARY, BUDGET_MODE, qe2Spec, POLY.hasManifest, fnv1a64(String(POLY.report || '')), usageOverride, QE_SCOPE, QE_SCOPE_REF])
|
|
4235
|
+
const qeHash = ckptHash('qe', [fnv1a64(JSON.stringify(codeStage === undefined ? null : codeStage)), tier, DESC, QE_REVIEWER, MODELS.qe === undefined ? null : MODELS.qe, CODEX_MODEL, coderUsed, PRIMARY, BUDGET_MODE, qe2Spec, POLY.hasManifest, fnv1a64(String(POLY.report || '')), usageOverride, QE_SCOPE, QE_SCOPE_REF, confirmationFileGate])
|
|
4053
4236
|
let crossFamilyQeReport = null
|
|
4054
4237
|
const qeStage = await withCheckpoint('qe', 'QE', qeHash, async () => {
|
|
4055
4238
|
let qe = null
|
|
@@ -4184,7 +4367,7 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
|
|
|
4184
4367
|
if (codexQe) {
|
|
4185
4368
|
// gaps come from what the reviewer ACTUALLY found; gradeSource says whether the letter was
|
|
4186
4369
|
// STATED by the reviewer or DERIVED from its findings, because mode A cannot be asked for one.
|
|
4187
|
-
qe = { grade: codexQe.grade, gaps: codexQe.findings, codeTestsAdequate: null, docTestsPresent: null, summary: String(codexQe.text).slice(0, 1500), gradeSource: codexQe.gradeSource, qeScope: { mode: codexQe.mode, ref: codexQe.scopeRef, files: codexQe.files } }
|
|
4370
|
+
qe = enforceConfirmationFileGate({ grade: codexQe.grade, gaps: codexQe.findings, codeTestsAdequate: null, docTestsPresent: null, summary: String(codexQe.text).slice(0, 1500), gradeSource: codexQe.gradeSource, qeScope: { mode: codexQe.mode, ref: codexQe.scopeRef, files: codexQe.files } }, confirmationFileGate)
|
|
4188
4371
|
qeReviewerUsed = qeIsCodex ? 'codex' : 'codex-fallback'
|
|
4189
4372
|
log('QE: cross-family review by codex, mode ' + codexQe.mode + ' (scope ' + codexQe.scopeRef + ', grade ' + codexQe.grade + ' ' + codexQe.gradeSource + ', ' + codexQe.elapsedSeconds + 's)')
|
|
4190
4373
|
// ARTIFACT SCRIBE. The old dispatch handed Codex the whole Step-8 prompt, so the reviewer itself
|
|
@@ -4199,7 +4382,7 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
|
|
|
4199
4382
|
// with no 08_qe_report.md at all. Named by cross-family review of b6973199. The verdict itself
|
|
4200
4383
|
// is real (Codex produced it), so a failed transcription DEGRADES the run rather than voiding
|
|
4201
4384
|
// it — but it must be visible, and it must never read as a clean QE.
|
|
4202
|
-
const scribePrompt = 'Step 8 (QE) of /feature-adr for "' + DESC + '" (' + SLUG + '). The independent cross-family review has ALREADY BEEN DONE, by Codex. You are the SCRIBE, not the reviewer: RECORD it, do NOT re-grade it, do NOT soften it, do NOT add a verdict of your own, and do NOT mark anything resolved that the reviewer flagged. The grade
|
|
4385
|
+
const scribePrompt = 'Step 8 (QE) of /feature-adr for "' + DESC + '" (' + SLUG + '). The independent cross-family review has ALREADY BEEN DONE, by Codex. You are the SCRIBE, not the reviewer: RECORD it, do NOT re-grade it, do NOT soften it, do NOT add a verdict of your own, and do NOT mark anything resolved that the reviewer flagged. The reviewer grade was ' + codexQe.grade + '; the final Step-8 grade after the deterministic Confirmation file gate is ' + qe.grade + ' and it is FINAL: the file gate can only LOWER a reviewer grade, never raise it, and nothing after this point may change it.\n\nWrite ' + FDIR + '/08_qe_report.md with: (1) the final grade ' + qe.grade + ' stated verbatim; (2) HOW it was obtained — reviewer grade ' + codexQe.grade + ', dispatch mode ' + codexQe.mode + ', scope ' + codexQe.scopeRef + ', wall-clock ' + codexQe.elapsedSeconds + 's, gradeSource ' + codexQe.gradeSource + ' (a DERIVED grade means the reviewer could not be asked for a letter and it was computed from the severities it reported — say so plainly); (3) the reviewer text below, verbatim, under the exact heading `## Primary QE pass`; (4) an ADR Fitness Checklist section with PASS/FAIL per ADR and the evidence pointer for the Confirmation-linked test; (5) this gate receipt as a separate line: `' + confirmationGateLine + '`.\n\nREVIEWER TEXT (verbatim, do not edit or summarise):\n' + String(codexQe.text) + '\n\nMANDATORY SELF-LEARNING STORE (close the loop, never skip): compare candidate lessons against the Step-0 recalled LEARNED patterns. Teach ONLY lessons NOT already covered; on overlap run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal). Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` verbatim via Bash. Finally run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash and TAG every quantitative claim you write the way the checker recognises as honest.' + ABSOLUTE_PATH_NOTE
|
|
4203
4386
|
// WITNESS THE REWRITE, not the existence. On a re-QE or a resume with the same slug an OLD
|
|
4204
4387
|
// 08_qe_report.md is already sitting there, and an existence probe reports that stale file as
|
|
4205
4388
|
// landed — so a scribe that wrote nothing still marked the new verdict recorded, and the stage
|
|
@@ -4254,6 +4437,7 @@ if (qe === null && qeIsCodex) {
|
|
|
4254
4437
|
if (qe) { qeReviewerUsed = 'claude'; crossFamilyQeReport = cfBelt.report }
|
|
4255
4438
|
else modelsUsed.qe = cfBelt.label + ' (no deliverable)'
|
|
4256
4439
|
}
|
|
4440
|
+
qe = enforceConfirmationFileGate(qe, confirmationFileGate)
|
|
4257
4441
|
// A-normal L/XL only: Sonnet is the recall-oriented primary reviewer; Opus is a SECOND,
|
|
4258
4442
|
// independent precision pass. It is advisory but real — never a table-only half-wire — and its
|
|
4259
4443
|
// provenance stays separate in both the return object and 08_qe_report.md.
|
|
@@ -4305,6 +4489,26 @@ let qeReviewerUsed = qeStage ? qeStage.qeReviewerUsed : 'claude'
|
|
|
4305
4489
|
if (qeStage && qeStage.modelUsed) modelsUsed.qe = qeStage.modelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
|
|
4306
4490
|
if (qeStage && qeStage.qe2ModelUsed) modelsUsed.qe2 = qeStage.qe2ModelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
|
|
4307
4491
|
|
|
4492
|
+
// Step 8 has completed its teach/reinforce work and written 08_qe_report.md. Closing telemetry is
|
|
4493
|
+
// secondary: refusal is loud and reflected in roundClosed, but it never overturns the feature run.
|
|
4494
|
+
if (qe && pipelineRound !== null) {
|
|
4495
|
+
const roundGrade = String(qe.grade || '').trim().toUpperCase()
|
|
4496
|
+
const roundOutcome = ['A', 'A-', 'B+', 'B'].indexOf(roundGrade) !== -1 ? 'shipped' : (['C', 'D'].indexOf(roundGrade) !== -1 ? 'refuted' : null)
|
|
4497
|
+
if (roundOutcome === null) {
|
|
4498
|
+
log('round close refused: unsupported Step 8 grade ' + JSON.stringify(roundGrade))
|
|
4499
|
+
} else {
|
|
4500
|
+
const roundLessonIds = Array.isArray(qe.roundLessons) ? qe.roundLessons.filter(function (id) { return /^teach:[a-z0-9]+$/i.test(String(id)) }) : []
|
|
4501
|
+
const roundLearningArg = roundLessonIds.length > 0
|
|
4502
|
+
? roundLessonIds.map(function (id) { return ' --lesson ' + shq(String(id)) }).join('')
|
|
4503
|
+
: ' --no-new-knowledge ' + shq((typeof qe.roundNoNewKnowledge === 'string' && qe.roundNoNewKnowledge.trim() !== '') ? qe.roundNoNewKnowledge.trim() : 'Step 8 returned no new teach receipt for this round')
|
|
4504
|
+
const roundCloseCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' round close --slug ' + shq(SLUG) + ' --round ' + pipelineRound + ' --outcome ' + roundOutcome + ' --reason ' + shq('grade ' + roundGrade) + roundLearningArg + ' --project ' + shq(BRAIN) + ' --no-cost --json'
|
|
4505
|
+
const roundCloseOut = await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and return its stdout VERBATIM, nothing else: ' + roundCloseCmd, { label: 'round:close', phase: 'QE', effort: 'low' })
|
|
4506
|
+
const roundCloseReceipt = parseRoundCommandJson(roundCloseOut)
|
|
4507
|
+
roundClosed = !!(roundCloseReceipt && roundCloseReceipt.marker && roundCloseReceipt.row && roundCloseReceipt.row.stage === 'round')
|
|
4508
|
+
if (!roundClosed) log('round close refused: ' + (roundCloseReceipt && roundCloseReceipt.message ? roundCloseReceipt.message : String(roundCloseOut || 'no JSON receipt')))
|
|
4509
|
+
}
|
|
4510
|
+
}
|
|
4511
|
+
|
|
4308
4512
|
// Step 8 claim-gate: fold the QE agent's reported claim-check counts into an additive result field.
|
|
4309
4513
|
const claimGate = step8ClaimGate(qe && qe.claimCheck ? qe.claimCheck : null)
|
|
4310
4514
|
log(claimGate.note)
|
|
@@ -4411,6 +4615,7 @@ if (Object.keys(AUTOCOST).length > 0 && resumedStages.indexOf('qe') === -1) {
|
|
|
4411
4615
|
let fleet = 'skipped (S/M)'
|
|
4412
4616
|
if (isLplus) {
|
|
4413
4617
|
phase('FleetQE')
|
|
4618
|
+
await recordRegistryEvent('heartbeat', 'FleetQE')
|
|
4414
4619
|
await usageProbe('FleetQE')
|
|
4415
4620
|
const fleetDecision = resolveStageDecision('fleet')
|
|
4416
4621
|
const fleetModel = fleetDecision.opts
|
|
@@ -4469,6 +4674,7 @@ let delivery = null
|
|
|
4469
4674
|
if (DELIVERY_ON) {
|
|
4470
4675
|
try {
|
|
4471
4676
|
phase('Delivery')
|
|
4677
|
+
await recordRegistryEvent('heartbeat', 'Delivery')
|
|
4472
4678
|
await usageProbe('Delivery')
|
|
4473
4679
|
// QE-D#2: cross-family is judged against the ACTUAL coder (coderUsed — the code stage already ran), and
|
|
4474
4680
|
// codex PLANES are unsupported in v1: a plane is a data-returning schema stage, and the codex wrapper
|
|
@@ -4650,6 +4856,7 @@ const finalGates = {
|
|
|
4650
4856
|
// DERIVED from the barrier's machine verdict, not from "is there a result object". A codex run
|
|
4651
4857
|
// whose barrier came back INCONCLUSIVE used to render exactly like a clean synchronous one.
|
|
4652
4858
|
code: (codeStage === null || codeStage === undefined || !code ? 'missing' : (codeStage.landingStatus === 'synchronous' ? 'produced' : (codeStage.landingStatus === 'landed' ? 'landed' : (codeStage.landingStatus === 'inconclusive' ? 'inconclusive' : 'not-landed')))),
|
|
4859
|
+
confirmationFiles: confirmationFileGate.verdict,
|
|
4653
4860
|
qe: (qe ? (qe.grade || 'ran') : 'not-run'),
|
|
4654
4861
|
claimCheck: (qe && qe.claimCheck ? (qe.claimCheck.high > 0 ? 'high-findings' : 'clean') : 'not-run'),
|
|
4655
4862
|
fleet: (isLplus ? (fleet ? 'ran' : 'not-run') : 'n/a'),
|
|
@@ -4664,6 +4871,7 @@ await appendRunCostRow('full', (isLplus ? 'FleetQE' : 'QE'), finalOutcome)
|
|
|
4664
4871
|
// time) hit SCORE-EXISTS and froze the failed attempt's 0/N forever. An unfinished run is simply
|
|
4665
4872
|
// not scored; the resume that finishes the work scores it.
|
|
4666
4873
|
const score = finalOutcome === 'completed' ? await autoScore(qeHash) : null
|
|
4874
|
+
if (registryOutcome !== 'unverified') registryOutcome = finalOutcome
|
|
4667
4875
|
return {
|
|
4668
4876
|
slug: SLUG, tier: tier, mode: MODE, artifactsDir: FDIR,
|
|
4669
4877
|
outcome: finalOutcome,
|
|
@@ -4671,6 +4879,7 @@ return {
|
|
|
4671
4879
|
design: design.filter(Boolean).map((d) => d.wrote).flat(),
|
|
4672
4880
|
codeWrote: code ? code.wrote : [],
|
|
4673
4881
|
qeGrade: qe ? qe.grade : null,
|
|
4882
|
+
roundClosed: roundClosed,
|
|
4674
4883
|
score: score,
|
|
4675
4884
|
gaps: qe ? qe.gaps : [],
|
|
4676
4885
|
codeTestsAdequate: qe ? qe.codeTestsAdequate : null,
|
|
@@ -4696,9 +4905,16 @@ return {
|
|
|
4696
4905
|
crossFamilyQe: crossFamilyQeReport,
|
|
4697
4906
|
qeReportWritten: (qe && typeof qe.qeReportWritten === 'boolean') ? qe.qeReportWritten : null,
|
|
4698
4907
|
claimGate: claimGate,
|
|
4908
|
+
confirmationFileGate: confirmationFileGate,
|
|
4699
4909
|
autoCost: Object.keys(AUTOCOST).length ? AUTOCOST : null,
|
|
4700
4910
|
// P4 (checkpoint-gate-line): DERIVED gate map for the final banner — from actual run state, never prose.
|
|
4701
4911
|
gates: finalGates,
|
|
4702
4912
|
delivery: delivery,
|
|
4703
4913
|
promiseTags: tags,
|
|
4704
4914
|
}
|
|
4915
|
+
} finally {
|
|
4916
|
+
if (finishRunRegistry) {
|
|
4917
|
+
try { await finishRunRegistry() }
|
|
4918
|
+
catch (error) { registryOutcome = 'unverified'; log('run registry: finished UNVERIFIED — ' + String(error)) }
|
|
4919
|
+
}
|
|
4920
|
+
}
|