@dzhechkov/skills-feature-adr 1.5.13 → 1.5.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +8 -8
- package/CHANGELOG.md +14 -0
- package/README.md +10 -1
- package/package.json +1 -1
- package/sbom.json +7 -7
- package/src/commands/init.js +38 -6
- package/templates/.claude/skills/feature-adr/modules/08-qe.md +11 -2
- package/templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs +104 -3
- package/templates/.claude/workflows/feature-adr.js +236 -28
package/.dz-manifest.json
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
"files": [
|
|
6
6
|
{
|
|
7
7
|
"path": "CHANGELOG.md",
|
|
8
|
-
"sha256": "
|
|
8
|
+
"sha256": "26fa03f89862d03ebfc6447c866cb5d4dcbd5591ae4f8fb3a3ea08b6c586e19d"
|
|
9
9
|
},
|
|
10
10
|
{
|
|
11
11
|
"path": "LICENSE",
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
},
|
|
14
14
|
{
|
|
15
15
|
"path": "README.md",
|
|
16
|
-
"sha256": "
|
|
16
|
+
"sha256": "8c96e82b41fb8479befb56797a367b3cc2119cc5b022b18fdd18f129ef8cd0f0"
|
|
17
17
|
},
|
|
18
18
|
{
|
|
19
19
|
"path": "bin/cli.js",
|
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
},
|
|
26
26
|
{
|
|
27
27
|
"path": "package.json",
|
|
28
|
-
"sha256": "
|
|
28
|
+
"sha256": "0cdd048a5aacd52294f72b5903b5a94c049ad61680dbcac532d5cb3fbd45c736"
|
|
29
29
|
},
|
|
30
30
|
{
|
|
31
31
|
"path": "src/cli.js",
|
|
@@ -37,7 +37,7 @@
|
|
|
37
37
|
},
|
|
38
38
|
{
|
|
39
39
|
"path": "src/commands/init.js",
|
|
40
|
-
"sha256": "
|
|
40
|
+
"sha256": "07ff9d955422979ea67aa5aa454c13303f5ae644efb32e5cc6f359a509638347"
|
|
41
41
|
},
|
|
42
42
|
{
|
|
43
43
|
"path": "src/commands/list.js",
|
|
@@ -157,7 +157,7 @@
|
|
|
157
157
|
},
|
|
158
158
|
{
|
|
159
159
|
"path": "templates/.claude/skills/feature-adr/modules/08-qe.md",
|
|
160
|
-
"sha256": "
|
|
160
|
+
"sha256": "5355e2b36ab13b6acb65abe5b206685915ef2b08191b4a3130574620012b9135"
|
|
161
161
|
},
|
|
162
162
|
{
|
|
163
163
|
"path": "templates/.claude/skills/feature-adr/modules/09-fleet-qe.md",
|
|
@@ -249,7 +249,7 @@
|
|
|
249
249
|
},
|
|
250
250
|
{
|
|
251
251
|
"path": "templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs",
|
|
252
|
-
"sha256": "
|
|
252
|
+
"sha256": "eb77822f121ade10bd6265b0f0145ec38507a07d2cb72ba5fa8336c962096bda"
|
|
253
253
|
},
|
|
254
254
|
{
|
|
255
255
|
"path": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
|
|
@@ -317,7 +317,7 @@
|
|
|
317
317
|
},
|
|
318
318
|
{
|
|
319
319
|
"path": "templates/.claude/workflows/feature-adr.js",
|
|
320
|
-
"sha256": "
|
|
320
|
+
"sha256": "6d1bb93be6fae296281ac8f9fc2f266e5f09336cf6e183912fab5ff3bd0d3fd8"
|
|
321
321
|
},
|
|
322
322
|
{
|
|
323
323
|
"path": "templates/lib/memory-protocol.md",
|
|
@@ -329,5 +329,5 @@
|
|
|
329
329
|
}
|
|
330
330
|
]
|
|
331
331
|
},
|
|
332
|
-
"signature": "
|
|
332
|
+
"signature": "+DKyJg37w6aHqhg/UhCpQ+y+w/e66HEEr9hgOv/TttSIMc6TokGEbvJpxEw2pdu19Q0SHlA/o6pSucM+h3KEDw=="
|
|
333
333
|
}
|
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,19 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [Unreleased]
|
|
4
|
+
|
|
5
|
+
### Fixed — `init --force` больше не стирает локальную правку молча
|
|
6
|
+
|
|
7
|
+
- Из трёх путей записи `init --force` был ЕДИНСТВЕННЫМ, который перезаписывал локально
|
|
8
|
+
изменённый файл без резервной копии и без строки в отчёте — при том, что баннер успеха
|
|
9
|
+
рекомендует именно эту команду. `update` сохраняет правку по трёхсторонней сверке,
|
|
10
|
+
`update --force` кладёт `.bak` с первого дня; расходился только `init --force`.
|
|
11
|
+
- Теперь `init --force` копирует изменённый файл в `<file>.bak` ПЕРЕД перезаписью и называет
|
|
12
|
+
его в отчёте. Копия делается только если байты ОТЛИЧАЮТСЯ от шаблона: `.bak`, совпадающий
|
|
13
|
+
с шаблоном, — чистый мусор. `--dry-run --force` называет будущую копию и ничего не пишет.
|
|
14
|
+
- Поведение самого `--force` не смягчено: файл по-прежнему перезаписывается, это его смысл.
|
|
15
|
+
Менялось только то, что правка перестала исчезать бесследно.
|
|
16
|
+
|
|
3
17
|
## [1.5.1] - 2026-08-21
|
|
4
18
|
|
|
5
19
|
### Changed — the Step-8 amendment gate is a COMMAND, and the durable writers are witnessed
|
package/README.md
CHANGED
|
@@ -63,7 +63,9 @@ npx @dzhechkov/skills-feature-adr init # Install core components
|
|
|
63
63
|
npx @dzhechkov/skills-feature-adr init --with-learning # + reward learning
|
|
64
64
|
npx @dzhechkov/skills-feature-adr init --knowledge-extractor # + knowledge extractor
|
|
65
65
|
npx @dzhechkov/skills-feature-adr init --with-learning --knowledge-extractor # + both
|
|
66
|
-
npx @dzhechkov/skills-feature-adr init --force # Overwrite existing files
|
|
66
|
+
npx @dzhechkov/skills-feature-adr init --force # Overwrite existing files (a locally
|
|
67
|
+
# changed file is copied to <file>.bak first,
|
|
68
|
+
# and the copy is named in the report)
|
|
67
69
|
npx @dzhechkov/skills-feature-adr init --dry-run # Preview without making changes
|
|
68
70
|
npx @dzhechkov/skills-feature-adr update # Update to latest version
|
|
69
71
|
npx @dzhechkov/skills-feature-adr remove # Clean uninstall
|
|
@@ -106,6 +108,13 @@ ARCHITECTURE → IMPLEMENTATION → CODE → QE → FLEET QE
|
|
|
106
108
|
# Full protocols + 6 extra skills, up to 7 fleet QE agents
|
|
107
109
|
```
|
|
108
110
|
|
|
111
|
+
### Checkpoint reads have one source (v1.5.14)
|
|
112
|
+
|
|
113
|
+
`loadCheckpoints` in the bundled `feature-adr.js` now builds its read command with the checkpoints blob's own
|
|
114
|
+
`checkpointReadCmd(FDIR)` instead of a hand-written duplicate, so the helper with the speaking name is the one the pipeline
|
|
115
|
+
runs. Behaviour is unchanged: for directory names with spaces, quotes and missing directories the shell output is
|
|
116
|
+
byte-identical (proved against the old command). A wiring test fails if the duplicate is re-inlined.
|
|
117
|
+
|
|
109
118
|
### Advisory micro-recall at two decision points (v1.5.9, staged)
|
|
110
119
|
|
|
111
120
|
The workflow makes one bounded decision-local recall attempt immediately before the live Step 3
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@dzhechkov/skills-feature-adr",
|
|
3
|
-
"version": "1.5.
|
|
3
|
+
"version": "1.5.14",
|
|
4
4
|
"description": "Adaptive Feature Development skill pack for Claude Code — 11-step pipeline with Complexity Router (S/M/L/XL), ADR-driven architecture, 15 agentic-qe skills, multi-agent fleet QE. Supports --full-qe, --full-qe-extended, --with-learning, and --knowledge-extractor modes.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"skills-feature-adr": "./bin/cli.js"
|
package/sbom.json
CHANGED
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
"hashes": [
|
|
16
16
|
{
|
|
17
17
|
"alg": "SHA-256",
|
|
18
|
-
"content": "
|
|
18
|
+
"content": "26fa03f89862d03ebfc6447c866cb5d4dcbd5591ae4f8fb3a3ea08b6c586e19d"
|
|
19
19
|
}
|
|
20
20
|
]
|
|
21
21
|
},
|
|
@@ -35,7 +35,7 @@
|
|
|
35
35
|
"hashes": [
|
|
36
36
|
{
|
|
37
37
|
"alg": "SHA-256",
|
|
38
|
-
"content": "
|
|
38
|
+
"content": "8c96e82b41fb8479befb56797a367b3cc2119cc5b022b18fdd18f129ef8cd0f0"
|
|
39
39
|
}
|
|
40
40
|
]
|
|
41
41
|
},
|
|
@@ -69,7 +69,7 @@
|
|
|
69
69
|
},
|
|
70
70
|
{
|
|
71
71
|
"name": "dz:canonical-json-sha256-v2",
|
|
72
|
-
"value": "
|
|
72
|
+
"value": "0cdd048a5aacd52294f72b5903b5a94c049ad61680dbcac532d5cb3fbd45c736"
|
|
73
73
|
}
|
|
74
74
|
]
|
|
75
75
|
},
|
|
@@ -99,7 +99,7 @@
|
|
|
99
99
|
"hashes": [
|
|
100
100
|
{
|
|
101
101
|
"alg": "SHA-256",
|
|
102
|
-
"content": "
|
|
102
|
+
"content": "07ff9d955422979ea67aa5aa454c13303f5ae644efb32e5cc6f359a509638347"
|
|
103
103
|
}
|
|
104
104
|
]
|
|
105
105
|
},
|
|
@@ -399,7 +399,7 @@
|
|
|
399
399
|
"hashes": [
|
|
400
400
|
{
|
|
401
401
|
"alg": "SHA-256",
|
|
402
|
-
"content": "
|
|
402
|
+
"content": "5355e2b36ab13b6acb65abe5b206685915ef2b08191b4a3130574620012b9135"
|
|
403
403
|
}
|
|
404
404
|
]
|
|
405
405
|
},
|
|
@@ -629,7 +629,7 @@
|
|
|
629
629
|
"hashes": [
|
|
630
630
|
{
|
|
631
631
|
"alg": "SHA-256",
|
|
632
|
-
"content": "
|
|
632
|
+
"content": "eb77822f121ade10bd6265b0f0145ec38507a07d2cb72ba5fa8336c962096bda"
|
|
633
633
|
}
|
|
634
634
|
]
|
|
635
635
|
},
|
|
@@ -799,7 +799,7 @@
|
|
|
799
799
|
"hashes": [
|
|
800
800
|
{
|
|
801
801
|
"alg": "SHA-256",
|
|
802
|
-
"content": "
|
|
802
|
+
"content": "6d1bb93be6fae296281ac8f9fc2f266e5f09336cf6e183912fab5ff3bd0d3fd8"
|
|
803
803
|
}
|
|
804
804
|
]
|
|
805
805
|
},
|
package/src/commands/init.js
CHANGED
|
@@ -54,14 +54,15 @@ function showKeysariumIntegration(keysariumManifest) {
|
|
|
54
54
|
|
|
55
55
|
// Copies a component file-by-file with per-file overwrite protection:
|
|
56
56
|
// - destination missing -> write, record in `written`
|
|
57
|
-
// - destination exists + --force ->
|
|
57
|
+
// - destination exists + --force -> BACK UP to a .bak sibling if the bytes differ,
|
|
58
|
+
// then overwrite; record in `written` (+ `backedUp`)
|
|
58
59
|
// - destination exists, no --force -> do NOT write, record in `preserved`
|
|
59
60
|
// In dry-run mode nothing is written, but the same written/preserved
|
|
60
61
|
// classification is produced.
|
|
61
62
|
// Returns { missing, fileCount } where `missing` means the template source
|
|
62
63
|
// was absent on disk and `fileCount` is how many files the template provides.
|
|
63
64
|
function installComponent(key, comp, templatesDir, targetDir, opts) {
|
|
64
|
-
const { force, dryRun, written, preserved, hashes } = opts;
|
|
65
|
+
const { force, dryRun, written, preserved, backedUp, hashes } = opts;
|
|
65
66
|
const src = path.join(templatesDir, comp.src);
|
|
66
67
|
const destRoot = path.join(targetDir, comp.src);
|
|
67
68
|
|
|
@@ -89,12 +90,20 @@ function installComponent(key, comp, templatesDir, targetDir, opts) {
|
|
|
89
90
|
}
|
|
90
91
|
|
|
91
92
|
for (const entry of entries) {
|
|
92
|
-
|
|
93
|
+
const existed = fileExists(entry.destFile);
|
|
94
|
+
if (existed && !force) {
|
|
93
95
|
preserved.push(entry.rel);
|
|
94
96
|
continue;
|
|
95
97
|
}
|
|
98
|
+
// `update --force` has backed edits up to a `.bak` sibling since day one; `init --force` was
|
|
99
|
+
// the ONE path that overwrote silently — and the success banner recommends exactly that
|
|
100
|
+
// command, so a locally-edited workflow could vanish with no notice and no copy.
|
|
101
|
+
// Only DIFFERING bytes are backed up: a `.bak` identical to the template is pure litter.
|
|
102
|
+
const differs = existed && hashFile(entry.destFile) !== hashFile(entry.srcFile);
|
|
103
|
+
if (differs && backedUp) backedUp.push(entry.rel);
|
|
96
104
|
if (!dryRun) {
|
|
97
105
|
ensureDir(path.dirname(entry.destFile));
|
|
106
|
+
if (differs) fs.copyFileSync(entry.destFile, `${entry.destFile}.bak`);
|
|
98
107
|
fs.copyFileSync(entry.srcFile, entry.destFile);
|
|
99
108
|
// Record the SHA-256 of the TEMPLATE bytes we just installed (not the
|
|
100
109
|
// dest) as this file's baseline. This makes baseline == mine immediately
|
|
@@ -110,6 +119,23 @@ function installComponent(key, comp, templatesDir, targetDir, opts) {
|
|
|
110
119
|
return { missing: false, fileCount: entries.length };
|
|
111
120
|
}
|
|
112
121
|
|
|
122
|
+
// Print the block of locally-changed files that were (or would be) backed up before --force
|
|
123
|
+
// overwrote them. Silence here is what the field report caught: the operator had no way to learn
|
|
124
|
+
// that their edit was gone, let alone where the copy is.
|
|
125
|
+
function printBackedUpBlock(backedUp, dryRun) {
|
|
126
|
+
if (backedUp.length === 0) return;
|
|
127
|
+
const MAX_SHOWN = 10; // `printPreservedBlock` держит свою копию — она объявлена в его теле
|
|
128
|
+
console.log('');
|
|
129
|
+
const verb = dryRun ? 'would be backed up' : 'backed up';
|
|
130
|
+
warn(`${backedUp.length} locally-changed file(s) ${verb} to a .bak sibling before being overwritten:`);
|
|
131
|
+
for (const rel of backedUp.slice(0, MAX_SHOWN)) {
|
|
132
|
+
console.log(dim(` ${rel} -> ${rel}.bak`));
|
|
133
|
+
}
|
|
134
|
+
if (backedUp.length > MAX_SHOWN) {
|
|
135
|
+
console.log(dim(` …and ${backedUp.length - MAX_SHOWN} more`));
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
113
139
|
// Print the block of pre-existing files that were (or would be) preserved
|
|
114
140
|
function printPreservedBlock(preserved, dryRun) {
|
|
115
141
|
console.log('');
|
|
@@ -233,6 +259,7 @@ async function run(options) {
|
|
|
233
259
|
const installedFiles = []; // files written (or would-write in dry-run)
|
|
234
260
|
const installedHashes = {}; // rel -> sha256 of the TEMPLATE bytes installed (baseline)
|
|
235
261
|
const preservedFiles = []; // pre-existing files NOT overwritten (no --force)
|
|
262
|
+
const backedUpFiles = []; // locally-changed files copied to .bak before --force overwrote them
|
|
236
263
|
const completedKeys = []; // component keys processed so far (for partial manifest)
|
|
237
264
|
const installedOptionalKeys = [];
|
|
238
265
|
let stepNum = 0;
|
|
@@ -246,7 +273,7 @@ async function run(options) {
|
|
|
246
273
|
|
|
247
274
|
installComponent(key, comp, templatesDir, targetDir, {
|
|
248
275
|
force, dryRun, written: installedFiles, preserved: preservedFiles,
|
|
249
|
-
hashes: installedHashes,
|
|
276
|
+
backedUp: backedUpFiles, hashes: installedHashes,
|
|
250
277
|
});
|
|
251
278
|
completedKeys.push(key);
|
|
252
279
|
}
|
|
@@ -262,7 +289,7 @@ async function run(options) {
|
|
|
262
289
|
|
|
263
290
|
const res = installComponent(key, comp, templatesDir, targetDir, {
|
|
264
291
|
force, dryRun, written: installedFiles, preserved: preservedFiles,
|
|
265
|
-
hashes: installedHashes,
|
|
292
|
+
backedUp: backedUpFiles, hashes: installedHashes,
|
|
266
293
|
});
|
|
267
294
|
if (!res.missing && res.fileCount > 0) {
|
|
268
295
|
installedOptionalKeys.push(key);
|
|
@@ -296,6 +323,10 @@ async function run(options) {
|
|
|
296
323
|
if (dryRun) {
|
|
297
324
|
console.log('');
|
|
298
325
|
info(`Dry run: ${installedFiles.length} file(s) would be written, ${preservedFiles.length} pre-existing file(s) would be preserved.`);
|
|
326
|
+
// Отчёт о копиях стоит СНАРУЖИ стража сохранённых: при --force сохранять нечего, список
|
|
327
|
+
// пуст, и внутри стража блок был бы недостижим ровно в том случае, ради которого написан.
|
|
328
|
+
// Пустой список функция отсекает сама.
|
|
329
|
+
printBackedUpBlock(backedUpFiles, true);
|
|
299
330
|
if (preservedFiles.length > 0) {
|
|
300
331
|
printPreservedBlock(preservedFiles, true);
|
|
301
332
|
}
|
|
@@ -318,10 +349,11 @@ async function run(options) {
|
|
|
318
349
|
process.exit(0);
|
|
319
350
|
}
|
|
320
351
|
|
|
352
|
+
printBackedUpBlock(backedUpFiles, false); // снаружи: при --force preservedFiles пуст
|
|
321
353
|
if (preservedFiles.length > 0) {
|
|
322
354
|
printPreservedBlock(preservedFiles, false);
|
|
323
|
-
console.log('');
|
|
324
355
|
}
|
|
356
|
+
if (preservedFiles.length > 0 || backedUpFiles.length > 0) console.log('');
|
|
325
357
|
|
|
326
358
|
// ── f) Write manifest ──────────────────────────────────────────────────
|
|
327
359
|
const pkgPath = path.resolve(__dirname, '../../package.json');
|
|
@@ -205,7 +205,9 @@ list), grep for unfinished-stub markers: `TODO` / `FIXME` / `HACK` / `XXX` / `PL
|
|
|
205
205
|
file:line, UNLESS the line carries an inline `no-stubs: <reason>` waiver WITH a non-empty reason, or
|
|
206
206
|
`.dz/guard.json` `stubWaivers` lists the path WITH a reason. A REASONLESS waiver is itself a HIGH gap,
|
|
207
207
|
never an exemption. Cross-check mechanically: `dz guard check --op publish --json` runs the same scan
|
|
208
|
-
as the SOFT `no-stubs` rule over the working-tree diff.
|
|
208
|
+
as the SOFT `no-stubs` rule over the working-tree diff.
|
|
209
|
+
After the Step-7 code has landed, run `dz guard check --op code --json` and treat a HARD `block` verdict
|
|
210
|
+
as a HIGH finding naming the drifted file. When you QUOTE a marker in `08_qe_report.md`,
|
|
209
211
|
backtick it so the report itself scans clean (the claim-check forbidden-phrase convention). Record the
|
|
210
212
|
verdict in the ADR Fitness section.
|
|
211
213
|
|
|
@@ -311,6 +313,9 @@ Compile all findings into a structured report:
|
|
|
311
313
|
✅ READY FOR MERGE | ❌ NEEDS FIXES | ⚠️ CONDITIONAL APPROVAL
|
|
312
314
|
```
|
|
313
315
|
|
|
316
|
+
Settle a cross-family re-QE debt with `dz reqe --slug <s> --done --report <08b>`.
|
|
317
|
+
For `dz reqe --done`, exit 3 means the review settled but named BLOCKER/HIGH findings — stop and surface them to the owner.
|
|
318
|
+
|
|
314
319
|
### 7.1 Findings ledger (machine-readable)
|
|
315
320
|
|
|
316
321
|
Prose is for people; `dz score`/`dz recap` need a machine-readable surface too (qe-findings-record,
|
|
@@ -373,9 +378,13 @@ resume guard at all before this).
|
|
|
373
378
|
Otherwise — QE ran fresh in this pass — after `08_qe_report.md` is written and the grade is final, run:
|
|
374
379
|
|
|
375
380
|
```bash
|
|
376
|
-
dz feature-adr-record --kind ledger --stage qe --slug <slug> --row '<json>' --
|
|
381
|
+
dz feature-adr-record --kind ledger --stage qe --slug <slug> --row '<json>' --json
|
|
377
382
|
```
|
|
378
383
|
|
|
384
|
+
fix-round-1 (Codex r1 HIGH finding 4, ADR-001 D5): no `--auto` here. `--auto` is the trusted marker
|
|
385
|
+
of an AUTOMATED pipeline run and requires an experiment envelope that the manual path does not have;
|
|
386
|
+
this plain-mode row is written without it.
|
|
387
|
+
|
|
379
388
|
`<json>` carries the same fields the ultracode pipeline writes for this row: `reviewer` (the model
|
|
380
389
|
that reviewed, or `null` if unknown — never guessed), `reviewerFamily` (`claude`|`codex`|`null` when
|
|
381
390
|
the reviewer identity is not one of the two known families — never guessed as `claude`), `qeRole`
|
|
@@ -31,12 +31,18 @@
|
|
|
31
31
|
// side) for C1 and C8 together, so a prose mention stops satisfying either check, is a separate
|
|
32
32
|
// backlog item — filed by the lead, not chased here.
|
|
33
33
|
//
|
|
34
|
+
// KNOWN LIMITATION (C9): only byte-identical copies are visible before the edit. Already drifted
|
|
35
|
+
// or intentionally different pinned copies remain the identity tests' job; the frozen
|
|
36
|
+
// features/wave1-instrument-repair/check-plan-completeness.mjs is a real example C9 cannot see.
|
|
37
|
+
//
|
|
34
38
|
// Checks:
|
|
35
39
|
// C1 every ADR file in 03_adr/ has >=1 task line in 06_implementation_plan.md citing it (ADR-00N)
|
|
36
40
|
// C2 every Confirmation-numbered check in each ADR is named in the plan (by its test-file path)
|
|
37
41
|
// C3 the plan carries an EXPECTED_CODE_TARGETS: block, non-empty, and EVERY line parses to a
|
|
38
42
|
// plausible repo-relative path (no spaces unless quoted, no traversal, no markdown residue)
|
|
39
43
|
// — SFDIPOT condition: line-level validation, reject-with-reason, not just block presence
|
|
44
|
+
// C9 every tracked byte-identical twin of a target is also listed or explicitly waived with a
|
|
45
|
+
// reason; size-first narrowing, fixed exclusions, one FAIL per missing (target, twin) pair
|
|
40
46
|
// C4 the plan names the feature's OWN acid corpus (see "acid corpus" below)
|
|
41
47
|
// C5 the plan has an 'Inputs read:' line naming 03_adr, 05_architecture (wave-2 seam, cheap here)
|
|
42
48
|
// C8 every requirement id DECLARED in 01_requirements.md (FR-N, NFR-N, AC-N, C-N, with an optional
|
|
@@ -79,7 +85,9 @@
|
|
|
79
85
|
// supplied explicitly with `--acid=T1,T2,…`. If neither establishes a corpus, C4 is SKIPPED-with-note
|
|
80
86
|
// (a feature that declared no acid cases cannot be failed for not naming them).
|
|
81
87
|
import { maskMarkdown } from './markdown-masker.mjs';
|
|
82
|
-
import { readFileSync, readdirSync, existsSync } from 'node:fs';
|
|
88
|
+
import { readFileSync, readdirSync, existsSync, statSync } from 'node:fs';
|
|
89
|
+
import { execFileSync } from 'node:child_process';
|
|
90
|
+
import { createHash } from 'node:crypto';
|
|
83
91
|
import { isAbsolute, join, resolve } from 'node:path';
|
|
84
92
|
|
|
85
93
|
const argv = process.argv.slice(2);
|
|
@@ -335,6 +343,7 @@ function classifyTargetPath(path) {
|
|
|
335
343
|
|
|
336
344
|
// C3 — EXPECTED_CODE_TARGETS block, line-level validation
|
|
337
345
|
const blockM = plan.match(/EXPECTED_CODE_TARGETS:\s*\n((?:\s*[-*]\s*.+\n?)+)/);
|
|
346
|
+
const listedTargets = new Set();
|
|
338
347
|
if (!blockM) failures.push('C3: no EXPECTED_CODE_TARGETS: block in the plan');
|
|
339
348
|
else {
|
|
340
349
|
const lines = blockM[1].split('\n').map(s => s.trim()).filter(Boolean);
|
|
@@ -343,6 +352,81 @@ else {
|
|
|
343
352
|
const path = ln.replace(/^[-*]\s*/, '').replace(/`/g, '').trim();
|
|
344
353
|
const reasons = classifyTargetPath(path);
|
|
345
354
|
if (reasons.length) failures.push(`C3: target line rejected: "${safe(ln)}" — ${reasons.join(', ')}`);
|
|
355
|
+
else listedTargets.add(path);
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
// C9 — every byte-identical copy is listed or carries a reasoned waiver (ADR-001).
|
|
360
|
+
const TWIN_EXCLUDED_PREFIXES = ['out/', '.claude/worktrees/', 'node_modules/'];
|
|
361
|
+
{
|
|
362
|
+
const excluded = (path) => TWIN_EXCLUDED_PREFIXES.some((prefix) => path.startsWith(prefix) || path.includes('/' + prefix));
|
|
363
|
+
const waivers = [];
|
|
364
|
+
for (const line of maskMarkdown(plan, { unclosed: 'mask' }).split('\n')) {
|
|
365
|
+
if (!/^\s*(?:[-*]\s*)?TWIN_NOT_A_TARGET:/.test(line)) continue;
|
|
366
|
+
const match = line.match(/^\s*(?:[-*]\s*)?TWIN_NOT_A_TARGET:\s*`?([^\s`]+)`?\s*(?:—|--)\s*(\S.*)$/);
|
|
367
|
+
if (!match) {
|
|
368
|
+
failures.push(`C9: waiver "${safe(line.trim())}" carries no reason — a waiver without a reason is an allowlist entry`);
|
|
369
|
+
} else waivers.push({ path: match[1], used: false });
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
let tracked = null;
|
|
373
|
+
try {
|
|
374
|
+
tracked = execFileSync('git', ['ls-files', '-z'], {
|
|
375
|
+
cwd: process.cwd(), encoding: 'utf-8', maxBuffer: 16 * 1024 * 1024,
|
|
376
|
+
stdio: ['ignore', 'pipe', 'pipe'],
|
|
377
|
+
}).split('\0').filter(Boolean);
|
|
378
|
+
} catch (error) {
|
|
379
|
+
warnings.push(`C9: twin scan had no input — git ls-files failed in ${safe(process.cwd())}: ${safe(error.message)}`);
|
|
380
|
+
}
|
|
381
|
+
if (tracked !== null) {
|
|
382
|
+
// Stat the universe first; only size matches ever reach readFileSync / md5.
|
|
383
|
+
const sizes = new Map();
|
|
384
|
+
for (const path of new Set([...tracked, ...listedTargets])) {
|
|
385
|
+
if (excluded(path)) continue;
|
|
386
|
+
try {
|
|
387
|
+
const stat = statSync(path);
|
|
388
|
+
if (stat.isFile()) sizes.set(path, stat.size);
|
|
389
|
+
} catch (error) {
|
|
390
|
+
// Missing targets are new files; tracked files can also have been deleted locally.
|
|
391
|
+
if (error.code !== 'ENOENT' && error.code !== 'ENOTDIR') warnings.push(`C9: cannot stat ${safe(path)}: ${safe(error.message)}`);
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
const targetSizes = new Set([...listedTargets].map((path) => sizes.get(path)).filter((size) => size > 0));
|
|
395
|
+
const hashes = new Map();
|
|
396
|
+
const twinsByHash = new Map();
|
|
397
|
+
for (const [path, size] of sizes) {
|
|
398
|
+
if (!targetSizes.has(size)) continue;
|
|
399
|
+
try { hashes.set(path, createHash('md5').update(readFileSync(path)).digest('hex')); }
|
|
400
|
+
catch (error) { warnings.push(`C9: cannot hash ${safe(path)}: ${safe(error.message)}`); }
|
|
401
|
+
}
|
|
402
|
+
for (const path of tracked) {
|
|
403
|
+
const hash = hashes.get(path);
|
|
404
|
+
if (hash === undefined) continue;
|
|
405
|
+
if (!twinsByHash.has(hash)) twinsByHash.set(hash, []);
|
|
406
|
+
twinsByHash.get(hash).push(path);
|
|
407
|
+
}
|
|
408
|
+
let unlisted = 0;
|
|
409
|
+
for (const target of listedTargets) {
|
|
410
|
+
for (const twin of twinsByHash.get(hashes.get(target)) ?? []) {
|
|
411
|
+
if (twin === target) continue;
|
|
412
|
+
let waived = false;
|
|
413
|
+
for (const waiver of waivers) {
|
|
414
|
+
if (waiver.path.endsWith('/') ? twin.startsWith(waiver.path) : twin === waiver.path) {
|
|
415
|
+
waiver.used = true;
|
|
416
|
+
waived = true;
|
|
417
|
+
}
|
|
418
|
+
}
|
|
419
|
+
if (listedTargets.has(twin) || waived) continue;
|
|
420
|
+
unlisted++;
|
|
421
|
+
// Keep pairs separate: safe() truncates each echoed value at 300 characters.
|
|
422
|
+
failures.push(`C9: target ${safe(target)} has a byte-identical twin the plan does not list: ${safe(twin)} — list it or waive it (TWIN_NOT_A_TARGET: ${safe(twin)} — <why>)`);
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
for (const waiver of waivers) if (!waiver.used) warnings.push(`C9: unused waiver ${safe(waiver.path)}`);
|
|
426
|
+
if (unlisted === 0) {
|
|
427
|
+
const onDisk = [...listedTargets].filter((path) => sizes.has(path)).length;
|
|
428
|
+
out(`NOTE C9: twin scan over ${listedTargets.size} target(s), ${onDisk} on disk, 0 unlisted twins (byte-identical only; drifted copies need identity tests)`);
|
|
429
|
+
}
|
|
346
430
|
}
|
|
347
431
|
}
|
|
348
432
|
|
|
@@ -417,7 +501,21 @@ else for (const t of acidTokens) if (!new RegExp(`\\b${t.replace(/[.*+?^${}()|[\
|
|
|
417
501
|
// bullet-less row was invisible here while being a row there (so it evaded this check), and
|
|
418
502
|
// this side lacked the range guard, so «AM-1..AM-4 are covered elsewhere» was falsely
|
|
419
503
|
// reported as a definition. One shape, one meaning.
|
|
420
|
-
|
|
504
|
+
// ЯЧЕЙКА ТАБЛИЦЫ ВНЕ СЕКЦИИ — НЕ ОПРЕДЕЛЕНИЕ. Внутри `## Amendments` строка таблицы
|
|
505
|
+
// законно объявляет поправку, поэтому там `|` остаётся допустимым началом строки. Снаружи
|
|
506
|
+
// же таблица — это СВОДКА, ссылающаяся на поправки, определённые в другом месте, и читать
|
|
507
|
+
// её как определение значит отказывать плану за оглавление.
|
|
508
|
+
// ИЗМЕРЕНО 2026-09-05: четыре ложных FAIL за один день на четырёх разных фичах
|
|
509
|
+
// (narrated-error-must-be-taught, fa-phase-statusline, core-boundary-guard,
|
|
510
|
+
// run-registry-liveness); каждый стоил ручной правки ведущего и перезапуска конвейера,
|
|
511
|
+
// то есть 5–10 минут и один цикл роутера. Один планировщик дошёл до того, что вписал в
|
|
512
|
+
// план оговорку «no line of this paragraph may begin with AM-» — прибор начал
|
|
513
|
+
// диктовать людям форму прозы, а это уже не проверка, а суеверие.
|
|
514
|
+
// ОДИН механизм, а не два: `|` УБРАН из класса начал строки. Первая редакция этой правки
|
|
515
|
+
// добавляла ещё и отдельную проверку `isTableRow`, и мутация, снимавшая только её,
|
|
516
|
+
// оставалась ЭКВИВАЛЕНТНОЙ — 77 тестов из 77 зелёные при «снятой» починке. Дублирующая
|
|
517
|
+
// защита не усиливает гейт, она прячет от мутационной пробы, что именно держит свойство.
|
|
518
|
+
const isDef = /^\s*(?:[-*]\s*)?\*{0,2}AM-(?:CP-)?\d+\*{0,2}\b(?!\s*\.)/.test(pl);
|
|
421
519
|
const inSection = sectionStart >= 0 && cursor2 >= sectionStart && cursor2 < sectionEnd;
|
|
422
520
|
if (isDef && !inSection) {
|
|
423
521
|
const tok = (/AM-(?:CP-)?\d+/.exec(pl) || ['AM-?'])[0];
|
|
@@ -479,7 +577,10 @@ else for (const t of acidTokens) if (!new RegExp(`\\b${t.replace(/[.*+?^${}()|[\
|
|
|
479
577
|
}
|
|
480
578
|
if (superseded) break;
|
|
481
579
|
}
|
|
482
|
-
|
|
580
|
+
// Сообщение называет ТУ ЖЕ форму, которую требует MARK выше. Прежняя редакция обещала
|
|
581
|
+
// простое `→ test <name>`, а проверка требовала имя и файл в обратных кавычках — человек,
|
|
582
|
+
// написавший ровно то, что просило сообщение, получал отказ снова и не понимал, почему.
|
|
583
|
+
if (!hasTest && !superseded) failures.push(`C6: ${defs[k].id} carries neither \`\u2192 test \`<имя>\` in \`<файл>\`\` (обратные кавычки обязательны, как и слово in) nor \`superseded by AM-N\` — an amendment without a confirmation is a wish, and a retracted one must say its successor`);
|
|
483
584
|
}
|
|
484
585
|
}
|
|
485
586
|
}
|
|
@@ -16,6 +16,49 @@ export const meta = {
|
|
|
16
16
|
],
|
|
17
17
|
}
|
|
18
18
|
|
|
19
|
+
// Self-contained mirror of harness-core/src/reqe.ts shouldEmitReqeDebt. TASK-3 executes both
|
|
20
|
+
// predicates against the shared cause table; keep branch order and decision text identical.
|
|
21
|
+
function reqeEmitDecision(input) {
|
|
22
|
+
const familyOf = function (spec) { return /codex|gpt|openai/i.test(String(spec ?? '')) ? 'openai' : 'claude' }
|
|
23
|
+
const coderFam = familyOf(input.coderUsed)
|
|
24
|
+
const qeFam = familyOf(input.qeReviewerUsed)
|
|
25
|
+
if (coderFam !== qeFam) {
|
|
26
|
+
return { emit: false, cause: null, degraded: false, reason: 'cross-family QE ran' }
|
|
27
|
+
}
|
|
28
|
+
if (input.routingRequested === false) {
|
|
29
|
+
return { emit: false, cause: null, degraded: false,
|
|
30
|
+
reason: 'same-family by configuration (cross-family never requested)' }
|
|
31
|
+
}
|
|
32
|
+
if (/\(usage-switched\)/.test(String(input.qeModelLabel ?? ''))) {
|
|
33
|
+
return {
|
|
34
|
+
emit: true, cause: 'usage-switched', degraded: false,
|
|
35
|
+
reason:
|
|
36
|
+
'usage-switched self-review: Step-8 QE ran on the coder’s own family (' + coderFam +
|
|
37
|
+
') under the limit override — the cross-model guard was suspended (FR-2.9)',
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
if (input.bridge?.rungState === 'probe-failed') {
|
|
41
|
+
return { emit: true, cause: 'probe-failed', degraded: false,
|
|
42
|
+
reason: 'codex probe found no usable id; review ran on the coder’s own family (' + coderFam + ')' }
|
|
43
|
+
}
|
|
44
|
+
if (input.bridge?.rungState === 'dispatched' || input.bridge?.rungState === 'refused-before-dispatch' ||
|
|
45
|
+
input.qeReviewerUsed === 'codex-fallback') {
|
|
46
|
+
return { emit: true, cause: 'same-family-fallback', degraded: false,
|
|
47
|
+
reason: 'same-family fallback: ' + (input.bridge?.decline ?? input.bridge?.rungReason ?? 'cross-family reviewer unavailable') +
|
|
48
|
+
'; review ran on the coder’s own family (' + coderFam + ')' }
|
|
49
|
+
}
|
|
50
|
+
if (input.bridge == null) {
|
|
51
|
+
return { emit: false, cause: null, degraded: true,
|
|
52
|
+
reason: "cause undeterminable (pre-change checkpoint) — re-run with resume:'never' to classify" }
|
|
53
|
+
}
|
|
54
|
+
if (input.routingRequested === true && input.bridge.rungState === 'pending') {
|
|
55
|
+
return { emit: true, cause: 'same-family-pinned', degraded: false,
|
|
56
|
+
reason: 'same-family by explicit qe pin (args.models.qe) while cross-family routing was requested' }
|
|
57
|
+
}
|
|
58
|
+
return { emit: false, cause: null, degraded: true,
|
|
59
|
+
reason: 'cause undeterminable (unknown rung state or routing request)' }
|
|
60
|
+
}
|
|
61
|
+
|
|
19
62
|
// Finalize on every normal return and on a caught runtime error; a killed host cannot finalize.
|
|
20
63
|
let finishRunRegistry = null
|
|
21
64
|
let registryOutcome = 'errored'
|
|
@@ -26,7 +69,6 @@ const A = typeof args === 'string' ? JSON.parse(args) : (args || {})
|
|
|
26
69
|
const SLUG = A.slug || 'feature'
|
|
27
70
|
const DESC = A.description || ''
|
|
28
71
|
const CODE_HINT = A.code || '(discover from the description)'
|
|
29
|
-
const MODE = A.mode || 'full-qe-extended'
|
|
30
72
|
const STOP_AFTER = A.stopAfter || null
|
|
31
73
|
// fa-phase-statusline (ADR-001 D2): tier holder for the ckpt-side phase-start fa-records. The real
|
|
32
74
|
// tier variable initializes only AFTER the router stage (TDZ — reading it from the router's own
|
|
@@ -207,6 +249,40 @@ function parseRoundCommandJson(raw) {
|
|
|
207
249
|
return null
|
|
208
250
|
}
|
|
209
251
|
|
|
252
|
+
// ablation-c-start (ADR-001 D6, fix-round-1 BLOCKER #2): the run's arm is NEVER accepted from a
|
|
253
|
+
// caller-supplied option — that let a run execute reference mode (or claim "direct" while actually
|
|
254
|
+
// running "reference") with zero journal entry, bypassing the lottery entirely. If args.experiment
|
|
255
|
+
// is given, the arm is RESOLVED here from the SAME assignments journal `dz experiment assign`
|
|
256
|
+
// already wrote to BEFORE this run was ever dispatched, via `dz experiment resolve` — never
|
|
257
|
+
// invented, never trusted from an option the caller could set to anything.
|
|
258
|
+
const EXPERIMENT = typeof A.experiment === 'string' ? A.experiment.trim() : ''
|
|
259
|
+
const EXPERIMENT_TASK_ID = typeof A.taskId === 'string' ? A.taskId.trim() : ''
|
|
260
|
+
// direct -> the pipeline's own default full pipeline; reference -> the reduced-QE arm. A caller
|
|
261
|
+
// that also passes args.mode must AGREE with what the resolved arm implies, or the run refuses
|
|
262
|
+
// (assignment-conflict) rather than silently letting the mode win over the arm.
|
|
263
|
+
const ABLATION_ARM_EXPECTED_MODE = { direct: 'full-qe-extended', reference: 'reference' }
|
|
264
|
+
let ABLATION_ARM = null
|
|
265
|
+
let ABLATION_PROPENSITY = null
|
|
266
|
+
if (EXPERIMENT !== '') {
|
|
267
|
+
if (EXPERIMENT_TASK_ID === '') {
|
|
268
|
+
throw new Error('feature-adr: args.experiment=' + JSON.stringify(EXPERIMENT) + ' but args.taskId is missing — the arm can only be RESOLVED from an existing assignment, never invented (ADR-001 D6, assignment-missing).')
|
|
269
|
+
}
|
|
270
|
+
const resolveCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' experiment resolve --experiment ' + shq(EXPERIMENT) + ' --task ' + shq(EXPERIMENT_TASK_ID) + ' --project ' + shq(REPO) + ' --json'
|
|
271
|
+
const resolveOut = await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and return its stdout VERBATIM, no commentary: ' + resolveCmd, { label: 'resolve-experiment-arm', phase: 'Route', model: 'haiku', effort: 'low' })
|
|
272
|
+
const resolved = parseRoundCommandJson(resolveOut)
|
|
273
|
+
if (!resolved || resolved.status !== 'resolved' || (resolved.arm !== 'direct' && resolved.arm !== 'reference')) {
|
|
274
|
+
const reason = (resolved && typeof resolved.status === 'string') ? resolved.status : 'assignment-missing'
|
|
275
|
+
throw new Error('feature-adr: could not resolve an assignment for task ' + JSON.stringify(EXPERIMENT_TASK_ID) + ' in experiment ' + JSON.stringify(EXPERIMENT) + ' (' + reason + ') — run `dz experiment assign` for this task BEFORE dispatching it (ADR-001 D6, assignment-missing).')
|
|
276
|
+
}
|
|
277
|
+
ABLATION_ARM = resolved.arm
|
|
278
|
+
ABLATION_PROPENSITY = (typeof resolved.propensity === 'number') ? resolved.propensity : null
|
|
279
|
+
const expectedMode = ABLATION_ARM_EXPECTED_MODE[ABLATION_ARM]
|
|
280
|
+
if (typeof A.mode === 'string' && A.mode !== '' && A.mode !== expectedMode) {
|
|
281
|
+
throw new Error('feature-adr: the resolved arm ' + JSON.stringify(ABLATION_ARM) + ' for task ' + JSON.stringify(EXPERIMENT_TASK_ID) + ' implies mode ' + JSON.stringify(expectedMode) + ', but args.mode=' + JSON.stringify(A.mode) + ' disagrees — refusing rather than letting one silently override the other (ADR-001 D6, assignment-conflict).')
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
const MODE = ABLATION_ARM !== null ? ABLATION_ARM_EXPECTED_MODE[ABLATION_ARM] : (A.mode || 'full-qe-extended')
|
|
285
|
+
|
|
210
286
|
// ── Durable checkpoints + resume (backlog 49e4a95b) — inline mirror of ──
|
|
211
287
|
// ── harness-core/src/feature-adr-checkpoints.ts (the workflow is self-contained, no imports) ──
|
|
212
288
|
// After each expensive stage a cheap effort-low agent appends {stage, inputHash, result} to
|
|
@@ -226,7 +302,7 @@ const CKPT_FILE = FDIR + '/.fa-state/checkpoints.jsonl'
|
|
|
226
302
|
// M10 Stage-A, feature loop-designer). This region is now a GENERATED BLOB (regen-diff-gated by
|
|
227
303
|
// loop-blobs-regen.test.ts): edit the canonical TS FIRST, run node scripts/gen-loop-blobs.mjs,
|
|
228
304
|
// then re-splice. The value-pinned wiring tests in feature-adr-checkpoints.test.ts stay the net.
|
|
229
|
-
// ── BEGIN BLOB checkpoints@1.2.0 sha256:
|
|
305
|
+
// ── BEGIN BLOB checkpoints@1.2.0 sha256:b602357649042d6b8aa68b5ca5593f7f5c483fdb36a287742e7b27bda8a2a6c7 src=packages/@dzhechkov/harness-core/src/feature-adr-checkpoints.ts ──
|
|
230
306
|
const CHECKPOINT_STAGES = ['router', 'design', 'plan', 'code', 'qe', 'fleet'];
|
|
231
307
|
const STAGE_ARTIFACTS = {
|
|
232
308
|
router: '00_complexity_assessment.md',
|
|
@@ -300,12 +376,12 @@ function decideCheckpointResume(opts) {
|
|
|
300
376
|
}
|
|
301
377
|
return { resume: true, reason: 'resumed' };
|
|
302
378
|
}
|
|
303
|
-
function serializeCheckpoint(stage, inputHash, result) {
|
|
379
|
+
function serializeCheckpoint(stage, inputHash, result, ts) {
|
|
304
380
|
if (result === null || result === undefined)
|
|
305
381
|
return null;
|
|
306
382
|
let line;
|
|
307
383
|
try {
|
|
308
|
-
line = JSON.stringify({ stage, inputHash, result });
|
|
384
|
+
line = JSON.stringify({ stage, inputHash, result, ...(ts === undefined ? {} : { ts }) });
|
|
309
385
|
}
|
|
310
386
|
catch {
|
|
311
387
|
return null;
|
|
@@ -315,6 +391,7 @@ function serializeCheckpoint(stage, inputHash, result) {
|
|
|
315
391
|
return line;
|
|
316
392
|
}
|
|
317
393
|
const CHECKPOINT_LS_SENTINEL = '---FA-CKPT-LS---';
|
|
394
|
+
const CHECKPOINT_TIMESTAMP = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{1,3})?Z$/;
|
|
318
395
|
function parseCheckpointRead(text) {
|
|
319
396
|
const out = { entries: {}, listing: new Set(), malformedLines: 0 };
|
|
320
397
|
const raw = String(text ?? '');
|
|
@@ -329,7 +406,8 @@ function parseCheckpointRead(text) {
|
|
|
329
406
|
try {
|
|
330
407
|
const e = JSON.parse(t);
|
|
331
408
|
if (e && typeof e === 'object' && typeof e.stage === 'string' && typeof e.inputHash === 'string' && 'result' in e && e.result !== null && e.result !== undefined) {
|
|
332
|
-
|
|
409
|
+
const { ts, ...entry } = e;
|
|
410
|
+
out.entries[e.stage] = typeof ts === 'string' && CHECKPOINT_TIMESTAMP.test(ts) ? { ...entry, ts } : entry;
|
|
333
411
|
}
|
|
334
412
|
else {
|
|
335
413
|
if (e && typeof e === 'object' && typeof e.stage === 'string')
|
|
@@ -360,7 +438,11 @@ function checkpointReadCmd(fdirAbs) {
|
|
|
360
438
|
function checkpointAppendCmd(fdirAbs, line) {
|
|
361
439
|
const dir = shellQuote(fdirAbs + '/.fa-state');
|
|
362
440
|
const file = shellQuote(fdirAbs + '/.fa-state/checkpoints.jsonl');
|
|
363
|
-
|
|
441
|
+
const plain = 'mkdir -p ' + dir + " && printf '%s\\n' " + shellQuote(line) + ' >> ' + file;
|
|
442
|
+
if (typeof line !== 'string' || !line.startsWith('{'))
|
|
443
|
+
return plain;
|
|
444
|
+
return 'ts=$(date -u +%Y-%m-%dT%H:%M:%SZ); mkdir -p ' + dir
|
|
445
|
+
+ ' && { printf \'{"ts":"%s",\' "$ts"; printf \'%s\\n\' ' + shellQuote(line.slice(1)) + '; } >> ' + file;
|
|
364
446
|
}
|
|
365
447
|
function parseArtifactProbe(opts) {
|
|
366
448
|
if (opts.stdout === null || opts.stdout === undefined)
|
|
@@ -406,7 +488,7 @@ let CKPT_LISTING = new Set()
|
|
|
406
488
|
const resumedStages = []
|
|
407
489
|
async function loadCheckpoints(phaseName) {
|
|
408
490
|
if (!CHECKPOINTS_ON) return
|
|
409
|
-
const readCmd =
|
|
491
|
+
const readCmd = checkpointReadCmd(FDIR)
|
|
410
492
|
const readOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and return its stdout VERBATIM (it may be empty) with NO code fences and NO commentary: ' + readCmd, { label: 'ckpt:read', phase: phaseName, effort: 'low' })
|
|
411
493
|
const raw = String(readOut == null ? '' : readOut)
|
|
412
494
|
// LINE-ANCHORED sentinel: a sentinel string INSIDE a recorded result shares its line with JSON
|
|
@@ -893,6 +975,70 @@ function qeFindingsSummary(gaps) {
|
|
|
893
975
|
}
|
|
894
976
|
return { findings: gaps.length, findingsBySeverity: bySeverity, findingsSource: 'gaps' }
|
|
895
977
|
}
|
|
978
|
+
// instrument-round-b T1/A1 (ADR-001 D1): a self-report can ONLY come from the QE agent's OWN return
|
|
979
|
+
// object, keyed to a NAMED source — never derived from MODE (mode is an intent the run was
|
|
980
|
+
// configured with, not an event that happened). A missing or non-boolean aqeInvoked is honestly
|
|
981
|
+
// 'not-reported', never guessed true/false from context. Pure, never throws.
|
|
982
|
+
// fix-round-1 (Codex r1 MEDIUM finding 6): `aqeEvidence` used to accept ANY nonempty string,
|
|
983
|
+
// including pure whitespace. Now TRIMMED; blank-after-trim keeps `aqeInvoked` (the agent DID answer
|
|
984
|
+
// the invoked/not-invoked question) but the evidence itself is honestly `null` with
|
|
985
|
+
// `aqeEvidenceStatus:'blank'`. A non-blank string is checked against a SOFT form (does it look like
|
|
986
|
+
// an MCP tool name or an `aqe ` command?) — a fabricated tool name is still indistinguishable from a
|
|
987
|
+
// truthful one (a named, honest limit, not a fixable gap: see the module README), so this can only
|
|
988
|
+
// catch the OBVIOUSLY wrong shape, never a convincing lie. The value is kept either way; only the
|
|
989
|
+
// status is marked 'unrecognized', never discarded.
|
|
990
|
+
function aqeSelfReport(qe) {
|
|
991
|
+
if (!qe || typeof qe !== 'object' || typeof qe.aqeInvoked !== 'boolean') {
|
|
992
|
+
return { aqeInvoked: null, aqeInvokedSource: 'not-reported', aqeEvidence: null }
|
|
993
|
+
}
|
|
994
|
+
const trimmed = typeof qe.aqeEvidence === 'string' ? qe.aqeEvidence.trim() : ''
|
|
995
|
+
if (trimmed === '') {
|
|
996
|
+
return { aqeInvoked: qe.aqeInvoked, aqeInvokedSource: 'qe-self-report', aqeEvidence: null, aqeEvidenceStatus: 'blank' }
|
|
997
|
+
}
|
|
998
|
+
// Lead delta after Codex r2 (#6 partial): a substring test called 'garbagemcp__agentic-qe__garbage'
|
|
999
|
+
// recognized evidence. The shape is now ANCHORED: the whole trimmed string is either an MCP tool
|
|
1000
|
+
// name (mcp__agentic-qe__<tool>) or an aqe command line starting with 'aqe '.
|
|
1001
|
+
const looksReal = /^mcp__agentic-qe__[a-z][a-z0-9_]*$/.test(trimmed) || /^aqe [\w-]/.test(trimmed)
|
|
1002
|
+
return {
|
|
1003
|
+
aqeInvoked: qe.aqeInvoked,
|
|
1004
|
+
aqeInvokedSource: 'qe-self-report',
|
|
1005
|
+
aqeEvidence: trimmed,
|
|
1006
|
+
aqeEvidenceStatus: looksReal ? 'recognized' : 'unrecognized',
|
|
1007
|
+
}
|
|
1008
|
+
}
|
|
1009
|
+
// instrument-round-b T2/A2/A7 (ADR-001 D2): `arm`/`propensity` are COPIES of
|
|
1010
|
+
// envelope.chosen.mode/envelope.policy.propensity, promoted to top-level ledger-row fields so a
|
|
1011
|
+
// consumer never has to parse the nested envelope object — copied at write time so the two can never
|
|
1012
|
+
// disagree by construction. No envelope (the router never completed) -> both null, armSource names
|
|
1013
|
+
// why. Pure, never throws.
|
|
1014
|
+
// fix-round-1 (Codex r1 HIGH finding 5): an envelope object present but carrying neither a valid
|
|
1015
|
+
// `chosen.mode` NOR a valid `policy.propensity` (e.g. `{}`) used to be labeled `armSource:'envelope'`
|
|
1016
|
+
// despite yielding two nulls — a validated-looking label on an unvalidated object. Now honestly
|
|
1017
|
+
// `'invalid-envelope'` unless AT LEAST one of the two fields actually resolved to a real value.
|
|
1018
|
+
function envelopeArmFields(envelope) {
|
|
1019
|
+
if (!envelope || typeof envelope !== 'object') return { arm: null, propensity: null, armSource: 'no-envelope' }
|
|
1020
|
+
const chosen = envelope.chosen
|
|
1021
|
+
const policy = envelope.policy
|
|
1022
|
+
const arm = (chosen && typeof chosen.mode === 'string' && chosen.mode !== '') ? chosen.mode : null
|
|
1023
|
+
const propensity = (policy && typeof policy.propensity === 'number') ? policy.propensity : null
|
|
1024
|
+
const armSource = (arm !== null || propensity !== null) ? 'envelope' : 'invalid-envelope'
|
|
1025
|
+
return { arm: arm, propensity: propensity, armSource: armSource }
|
|
1026
|
+
}
|
|
1027
|
+
// fix-round-1 (Codex r1 HIGH finding 5): keys a per-stage `extra` is FORBIDDEN from carrying —
|
|
1028
|
+
// each one is either derived from the SAME canonical ENVELOPE the row already stamps (arm,
|
|
1029
|
+
// propensity, armSource) or IS that canonical value (envelope) — stripped before the spread so an
|
|
1030
|
+
// `extra` that named one of these could never make the row's TOP-LEVEL fields disagree with its
|
|
1031
|
+
// NESTED envelope. Defense in depth: appendRunCostRow also restates `envelope: ENVELOPE` and
|
|
1032
|
+
// re-derives arm/propensity/armSource AFTER the (now-sanitized) spread.
|
|
1033
|
+
function sanitizeLedgerExtra(extra) {
|
|
1034
|
+
const RESERVED_LEDGER_EXTRA_KEYS = ['arm', 'propensity', 'armSource', 'envelope']
|
|
1035
|
+
if (!extra || typeof extra !== 'object') return {}
|
|
1036
|
+
const out = {}
|
|
1037
|
+
for (const k in extra) {
|
|
1038
|
+
if (Object.prototype.hasOwnProperty.call(extra, k) && RESERVED_LEDGER_EXTRA_KEYS.indexOf(k) === -1) out[k] = extra[k]
|
|
1039
|
+
}
|
|
1040
|
+
return out
|
|
1041
|
+
}
|
|
896
1042
|
function tpText(v) { if (typeof v === 'string') return v; if (v === null || v === undefined) return ''; try { const s = JSON.stringify(v); return typeof s === 'string' ? s : String(v) } catch (e) { return String(v) } }
|
|
897
1043
|
function tpBudget(raw) {
|
|
898
1044
|
try {
|
|
@@ -1139,7 +1285,18 @@ async function appendRunCostRow(stage, phaseName, outcome, extra) {
|
|
|
1139
1285
|
// the payload carries no runId), so passing it here is a strict reliability improvement.
|
|
1140
1286
|
runId: (typeof RUN_ID === 'string' && RUN_ID !== '') ? RUN_ID : null,
|
|
1141
1287
|
// aqe-ledger-row T1/NFR-1: additive-only — spreads nothing when `extra` is absent.
|
|
1142
|
-
|
|
1288
|
+
// fix-round-1 (Codex r1 HIGH finding 5): reserved keys are STRIPPED from `extra` first
|
|
1289
|
+
// (sanitizeLedgerExtra), which is what makes the single `envelope: ENVELOPE` above canonical —
|
|
1290
|
+
// nothing the spread carries can shadow it. Lead delta after Codex r2: the belt-and-braces
|
|
1291
|
+
// RESTATEMENT that used to sit here was removed — a second `envelope` key in the same literal
|
|
1292
|
+
// is invisible at runtime but visible to the key-order guard (feature-adr-training-pairs),
|
|
1293
|
+
// which reads the literal's keys, so the defense in depth read as a contract break.
|
|
1294
|
+
...sanitizeLedgerExtra(extra),
|
|
1295
|
+
// instrument-round-b T2/FR-2/A2/A7 (ADR-001 D2): arm/propensity/armSource land AFTER the
|
|
1296
|
+
// per-stage `extra` spread — the last fields on every row, exactly like NFR-1 requires. Derived
|
|
1297
|
+
// from the SAME `ENVELOPE` just restated above, so the row's nested envelope and its top-level
|
|
1298
|
+
// arm/propensity/armSource can never disagree (Codex r1 finding 5).
|
|
1299
|
+
...envelopeArmFields(ENVELOPE),
|
|
1143
1300
|
})
|
|
1144
1301
|
// WITNESSED WRITE (ADR-001): the subagent RUNS a command with data arguments; it is no longer
|
|
1145
1302
|
// handed a shell pipeline with the row baked in. The command refuses a malformed row, stamps the
|
|
@@ -1147,8 +1304,18 @@ async function appendRunCostRow(stage, phaseName, outcome, extra) {
|
|
|
1147
1304
|
// re-reading the tail. A courier could do none of those three.
|
|
1148
1305
|
// fix-round-1/F2: --auto is the TRUSTED CLI-level marker (harness-core run-records.ts) — the
|
|
1149
1306
|
// written row's auto:true no longer depends solely on the JSON payload remembering the field.
|
|
1307
|
+
// instrument-round-b FR-4/A5 (ADR-001 D4), fix-round-1 (Codex r1 HIGH finding 3): a NAMED,
|
|
1308
|
+
// SCOPED allowance — `--allow-incomplete tokens,minutes --incomplete-reason
|
|
1309
|
+
// sandbox-metrics-unavailable` — replaces the removed blanket `--no-strict`. This row's `tokens`/
|
|
1310
|
+
// `minutes` are LEGITIMATELY incomplete by construction, every single call: the sandbox has no
|
|
1311
|
+
// clock and budget.spent() exposes only a partial OUTPUT-token delta (tokensOut, a DIFFERENT
|
|
1312
|
+
// field than the `tokens`/`minutes` completeness checks), never the real total — that total is
|
|
1313
|
+
// visible only in the Workflow COMPLETION NOTIFICATION, which the running script cannot see (see
|
|
1314
|
+
// the HONESTY comment above the row literal). Naming exactly these two fields means if a THIRD
|
|
1315
|
+
// one (e.g. `envelope`) ever went missing too, the write would refuse rather than silently
|
|
1316
|
+
// widen what "legitimately incomplete" covers.
|
|
1150
1317
|
const cmd = DZ + ' feature-adr-record --kind ledger --stage ' + shq(stage) + ' --project ' + shq(REPO)
|
|
1151
|
-
+ ' --row ' + shq(line) + ' --auto --json'
|
|
1318
|
+
+ ' --row ' + shq(line) + ' --auto --allow-incomplete tokens,minutes --incomplete-reason sandbox-metrics-unavailable --json'
|
|
1152
1319
|
const out = await dispatchAgent(newRung(), 'Run this command via your Bash tool and reply with only its stdout: ' + cmd, { label: 'ledger:append', phase: phaseName, effort: 'low' })
|
|
1153
1320
|
const readback = String(out == null ? '' : out)
|
|
1154
1321
|
if (!/"verdict"\s*:\s*"written"/.test(readback)) {
|
|
@@ -1638,7 +1805,14 @@ function buildExperimentEnvelopeInline(input) {
|
|
|
1638
1805
|
// this mirror used to alias them directly, so a caller mutating its own arms.stages object
|
|
1639
1806
|
// after calling this function would silently mutate the built envelope too.
|
|
1640
1807
|
arms: { mode: input.arms.mode.slice(), stages: mergeOpts({}, input.arms.stages) },
|
|
1641
|
-
chosen: {
|
|
1808
|
+
chosen: {
|
|
1809
|
+
mode: input.chosen.mode,
|
|
1810
|
+
stages: mergeOpts({}, input.chosen.stages),
|
|
1811
|
+
overrides: mergeOpts({}, input.chosen.overrides === undefined ? {} : input.chosen.overrides),
|
|
1812
|
+
// ablation-c-start (ADR-001, T3): mirror of the core builder — carried through only
|
|
1813
|
+
// when the caller actually set it, so an unset qeMode never appears in the JSON.
|
|
1814
|
+
...(input.chosen.qeMode !== undefined ? { qeMode: input.chosen.qeMode } : {}),
|
|
1815
|
+
},
|
|
1642
1816
|
policy: { name: input.policy.name, version: input.policy.version, propensity: input.policy.propensity },
|
|
1643
1817
|
evaluator: { family: input.evaluator.family, model: input.evaluator.model, source: input.evaluator.source },
|
|
1644
1818
|
}
|
|
@@ -1748,6 +1922,12 @@ function validateExperimentEnvelopeInline(value) {
|
|
|
1748
1922
|
return { ok: false, reason: 'chosen.stages.' + stage + ': "' + spec + '" is not a member of arms.stages.' + stage + ' (' + offered.join('|') + ') and not declared in chosen.overrides' }
|
|
1749
1923
|
}
|
|
1750
1924
|
}
|
|
1925
|
+
// ablation-c-start (ADR-001, T3): qeMode is OPTIONAL — absent on every run this feature
|
|
1926
|
+
// does not touch — but when present must be a non-empty string, same shape rule every
|
|
1927
|
+
// other envelope field gets (mirror of the core validator).
|
|
1928
|
+
if (chosen.qeMode !== undefined && !isNonEmptyStringEnv(chosen.qeMode)) {
|
|
1929
|
+
return { ok: false, reason: 'chosen.qeMode: expected a non-empty string when present' }
|
|
1930
|
+
}
|
|
1751
1931
|
if (!isPlainObjectEnv(v.policy)) return { ok: false, reason: 'policy: expected an object' }
|
|
1752
1932
|
const policy = v.policy
|
|
1753
1933
|
if (!isNonEmptyStringEnv(policy.name)) return { ok: false, reason: 'policy.name: expected a non-empty string' }
|
|
@@ -3735,7 +3915,7 @@ const ARTIFACT = { type: 'object', additionalProperties: false, required: ['wrot
|
|
|
3735
3915
|
// (hasManifest + the who-injected report). The BIG per-stage guidance content is fetched by each stage
|
|
3736
3916
|
// agent directly from `dz project-skills` (never threaded through a model → fidelity preserved).
|
|
3737
3917
|
const PROJECT_SKILLS = { type: 'object', additionalProperties: false, required: ['hasManifest', 'report'], properties: { hasManifest: { type: 'boolean' }, report: { type: 'string' } } }
|
|
3738
|
-
const QE = { type: 'object', additionalProperties: false, required: ['grade', 'gaps', 'codeTestsAdequate', 'docTestsPresent'], properties: { grade: { type: 'string' }, codeTestsAdequate: { type: 'boolean' }, docTestsPresent: { type: 'boolean' }, gaps: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['sev', 'what'], properties: { sev: { type: 'string' }, what: { type: 'string' } } } }, claimCheck: { type: 'object', additionalProperties: false, properties: { findings: { type: 'number' }, high: { type: 'number' }, medium: { type: 'number' } } }, roundLessons: { type: 'array', items: { type: 'string' } }, roundNoNewKnowledge: { type: 'string' } } }
|
|
3918
|
+
const QE = { type: 'object', additionalProperties: false, required: ['grade', 'gaps', 'codeTestsAdequate', 'docTestsPresent'], properties: { grade: { type: 'string' }, codeTestsAdequate: { type: 'boolean' }, docTestsPresent: { type: 'boolean' }, gaps: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['sev', 'what'], properties: { sev: { type: 'string' }, what: { type: 'string' } } } }, claimCheck: { type: 'object', additionalProperties: false, properties: { findings: { type: 'number' }, high: { type: 'number' }, medium: { type: 'number' } } }, roundLessons: { type: 'array', items: { type: 'string' } }, roundNoNewKnowledge: { type: 'string' }, aqeInvoked: { type: 'boolean' }, aqeEvidence: { type: 'string' } } }
|
|
3739
3919
|
const CONFIRMATION_FILE_GATE = { type: 'object', additionalProperties: false, required: ['verdict', 'missing', 'checked', 'reason'], properties: { verdict: { type: 'string', enum: ['pass', 'fail', 'skipped', 'refused'] }, missing: { type: 'array', items: { type: 'string' } }, checked: { type: 'array', items: { type: 'string' } }, reason: { type: 'string' } } }
|
|
3740
3920
|
|
|
3741
3921
|
function normalizeConfirmationFileGate(raw) {
|
|
@@ -3783,7 +3963,7 @@ const MUTATION_GATE = 'MUTATION GATE (feature ha-mutation-gate — run alongside
|
|
|
3783
3963
|
// same scan runs mechanically at publish time as the SOFT `no-stubs` guard rule.
|
|
3784
3964
|
const STUB_RX = '(^|[^A-Za-z0-9_])(' + ['TO' + 'DO', 'FIX' + 'ME', 'HA' + 'CK', 'XX' + 'X', 'PLACE' + 'HOLDER'].join('|') + ')([^A-Za-z0-9_]|$)'
|
|
3785
3965
|
const STUB_PHRASE = 'imple' + 'ment later'
|
|
3786
|
-
const NO_STUBS_GATE = 'NO-STUBS GATE (backlog 0b403a0106103901 — layer 1 of the cost-of-detection ladder): over the files THIS RUN touched (the Step-7 change list; for a Codex coder, the landed-barrier file list), via Bash run EXACTLY `grep -nE \'' + STUB_RX + '\' <touched files>` (case-SENSITIVE — never add -i) plus `grep -niE \'' + STUB_PHRASE.replace(' ', '[[:space:]]+') + '\' <touched files>`. ANY match = the task shipped incomplete → HIGH gap naming file:line, UNLESS the line carries an inline `no-stubs: <reason>` waiver WITH a non-empty reason, or `.dz/guard.json` stubWaivers lists the path WITH a reason — a REASONLESS waiver is itself a HIGH gap, never an exemption. Cross-check mechanically: `dz guard check --op publish --json` runs the same scan as the SOFT `no-stubs` rule over the working-tree diff. When you QUOTE a marker in 08_qe_report.md, backtick it so the report itself scans clean (the same convention as the claim-check forbidden-phrase escape). Record the verdict in the 08_qe_report.md ADR Fitness section.'
|
|
3966
|
+
const NO_STUBS_GATE = 'NO-STUBS GATE (backlog 0b403a0106103901 — layer 1 of the cost-of-detection ladder): over the files THIS RUN touched (the Step-7 change list; for a Codex coder, the landed-barrier file list), via Bash run EXACTLY `grep -nE \'' + STUB_RX + '\' <touched files>` (case-SENSITIVE — never add -i) plus `grep -niE \'' + STUB_PHRASE.replace(' ', '[[:space:]]+') + '\' <touched files>`. ANY match = the task shipped incomplete → HIGH gap naming file:line, UNLESS the line carries an inline `no-stubs: <reason>` waiver WITH a non-empty reason, or `.dz/guard.json` stubWaivers lists the path WITH a reason — a REASONLESS waiver is itself a HIGH gap, never an exemption. Cross-check mechanically: `dz guard check --op publish --json` runs the same scan as the SOFT `no-stubs` rule over the working-tree diff. After the Step-7 code has landed, run `dz guard check --op code --json` and treat a HARD `block` verdict as a HIGH finding naming the drifted file. When you QUOTE a marker in 08_qe_report.md, backtick it so the report itself scans clean (the same convention as the claim-check forbidden-phrase escape). Record the verdict in the 08_qe_report.md ADR Fitness section.'
|
|
3787
3967
|
const DISCRIMINATION_GATE = '\u00a742 TEST-DISCRIMINATION GATE (run right after asserting the property has a test): the ADR Confirmation names `Required automated check: <test file>` for the load-bearing property. Prove that test DISCRIMINATES \u2014 via Bash run EXACTLY `' + DZ + ' discrimination-check --test <that test file> --base HEAD --json` (the PINNED workspace bin, never bare `dz` — the global install measurably lags the workspace) (the Step-7 feature diff is UNCOMMITTED, so HEAD is the pre-feature base). Parse the JSON: read `perTest[]` (each row carries verdict + reason), `findings[]` (ALL entries, not only the first), `measurementValid`, and `primaryAction` \u2014 the singular `finding` is a DEPRECATED alias; do not consume it. The SEVEN verdicts and the required QE action for each: `DISCRIMINATES` (assertion-red at base, execution-evidenced) = PASS. `DISCRIMINATES_VIA_ERROR` (evidenced load-error at base + evidenced pass at tip) = PASS \u2014 note the inference. `NON_DISCRIMINATING` (evidenced pass at base \u2014 a proven false green) \u2192 HIGH gap "property test does not discriminate: <file>"; advisory, not an automatic blocker. `TEST_FILE_ABSENT` (the named test is not a regular file) \u2192 HIGH gap; action create-missing-test; NEVER a pass. `LOAD_ERROR_AT_BOTH_REVS` (the instrument could not execute the test at either rev \u2014 zero signal) \u2192 HIGH gap; action fix-runner-invocation. `FAILS_AT_TIP` (the feature\'s own test is red WITH the feature present) \u2192 HIGH gap; action fix-red-feature-test \u2014 grade the feature code accordingly. `CANNOT_ISOLATE` (no established observation; the row\'s `reason` is one of no-execution-evidence | unrecognised-runner-output | no-tests-executed | inconsistent-evidence | tip-control-missing | tip-evidence-missing | timeout) \u2192 HIGH gap NAMING the reason; action per `primaryAction` (map-a-test or fix-runner-invocation). `measurementValid` false or \'partial\' means the instrument did not (fully) measure \u2014 report it verbatim; never convert a degraded reading into a pass. Record every verdict + reason in the 08_qe_report.md ADR Fitness section. If `discrimination-check` is unavailable at the pinned path, errors, or overruns its window \u2192 record a HIGH gap `discrimination gate INCONCLUSIVE: <unavailable|error|timeout>` (backlog 52d0ed08: an instrument that could not run is never a pass and never applicable-by-silence). Still never abort the run.'
|
|
3788
3968
|
// P2 (amendment-confirmation-discipline, fa-improvements 2026-07-18): amendments are where the SHARPEST design
|
|
3789
3969
|
// corrections land (challenge-panel/QCSD) and were the least-tested — prose deltas with no proving test. Every
|
|
@@ -4078,8 +4258,17 @@ ENVELOPE = buildExperimentEnvelopeInline({
|
|
|
4078
4258
|
treeSha: ENVELOPE_TREE_SHA,
|
|
4079
4259
|
treeShaReason: ENVELOPE_TREE_SHA_REASON,
|
|
4080
4260
|
arms: { mode: ['same-family', 'cross-family'], stages: envelopeArmsStages },
|
|
4081
|
-
|
|
4082
|
-
|
|
4261
|
+
// ablation-c-start (ADR-001, T3; fix-round-1 BLOCKER #2): a non-null ABLATION_ARM puts the arm
|
|
4262
|
+
// in chosen.qeMode (a field SEPARATE from chosen.mode's same-family/cross-family axis — mode
|
|
4263
|
+
// keeps its old meaning untouched) and renames the policy so a reader of the ledger's embedded
|
|
4264
|
+
// envelope can tell an ablation-c run from a routing-tables run at a glance. ABLATION_ARM and
|
|
4265
|
+
// ABLATION_PROPENSITY are RESOLVED above from the assignments journal via `dz experiment
|
|
4266
|
+
// resolve` — never taken from a caller-supplied option — so this line cannot be used to bypass
|
|
4267
|
+
// the lottery. ?? 0.5 covers only a missing/invalid propensity on the resolved record alongside
|
|
4268
|
+
// a present arm, never a missing arm. Without an arm, both lines stay byte-identical to before
|
|
4269
|
+
// this feature (NFR-2).
|
|
4270
|
+
chosen: { mode: ENVELOPE_CHOSEN_MODE, stages: envelopeChosenStages, overrides: envelopeChosenOverrides, ...(ABLATION_ARM ? { qeMode: ABLATION_ARM } : {}) },
|
|
4271
|
+
policy: ABLATION_ARM ? { name: 'ablation-c', version: POLICY_VERSION, propensity: (typeof ABLATION_PROPENSITY === 'number') ? ABLATION_PROPENSITY : 0.5 } : { name: 'routing-tables', version: POLICY_VERSION, propensity: null },
|
|
4083
4272
|
evaluator: { family: ENVELOPE_QE_FAMILY, model: ENVELOPE_QE_SPEC, source: 'planned' },
|
|
4084
4273
|
})
|
|
4085
4274
|
// fix-round-1/F1: the router's training pair, captured here — AFTER ENVELOPE is real — so it gets
|
|
@@ -4766,7 +4955,17 @@ if (stopHere) {
|
|
|
4766
4955
|
const rows = cpFindings.map((f, i) => '- AM-CP-' + (i + 1) + ' [' + f.severity + '] ' + String(f.title || '').replace(/[\r\n`]/g, ' ').slice(0, 160) + ' \u2192 test `названный кодером при реализации — заменить на имя реального теста` (panel ' + String(f.c || '') + ')').join('\n')
|
|
4767
4956
|
const marker = '<!-- challenge-panel amendments appended ' + fnv1a64(rows) + ' -->'
|
|
4768
4957
|
const planPath = FDIR + '/06_implementation_plan.md'
|
|
4769
|
-
|
|
4958
|
+
// The rows must land INSIDE `## Amendments`, not at EOF. MEASURED 2026-09-20 (backlog d48505cc,
|
|
4959
|
+
// which stood as an unverified inference until then): with `## Amendments` at line 13 and a later
|
|
4960
|
+
// `## Risks` at 17, the old `>> plan` put AM-CP-1 at line 21 and K2 C6 answered "AM-CP-1 is
|
|
4961
|
+
// DEFINED outside the `## Amendments` section". A plan whose Amendments section happens to be
|
|
4962
|
+
// last worked by luck, not by construction. The awk pass inserts before the NEXT `## ` heading
|
|
4963
|
+
// after the section, and falls back to EOF when the section IS last; the `a` flag is set AFTER
|
|
4964
|
+
// the heading test so the `## Amendments` line itself never triggers the insert. Idempotence is
|
|
4965
|
+
// unchanged: the marker grep still short-circuits to CP-DUP.
|
|
4966
|
+
const insertAwk = 'awk -v m=' + shq(marker) + ' -v r=' + shq(rows)
|
|
4967
|
+
+ ' \'BEGIN{a=0;d=0} { if(a&&!d&&/^## /){print m; print r; d=1; a=0} if($0 ~ /^## Amendments/)a=1; print } END{if(a&&!d){print m; print r}}\' '
|
|
4968
|
+
const appendCmd = 'cd ' + shq(REPO) + ' && grep -qF ' + shq(marker) + ' ' + shq(planPath) + ' && echo CP-DUP || { grep -q "^## Amendments" ' + shq(planPath) + ' || printf "\n## Amendments\n" >> ' + shq(planPath) + '; ' + insertAwk + shq(planPath) + ' > ' + shq(planPath) + '.cp-tmp && mv ' + shq(planPath) + '.cp-tmp ' + shq(planPath) + '; echo CP-APPENDED; }'
|
|
4770
4969
|
const cpOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and reply with ONLY its stdout: ' + appendCmd, { label: 'challenge:append-amendments', phase: 'Plan', effort: 'low' })
|
|
4771
4970
|
log('challenge panel \u2192 plan amendments: ' + (/CP-APPENDED/.test(String(cpOut || '')) ? cpFindings.length + ' AM-CP row(s) appended' : /CP-DUP/.test(String(cpOut || '')) ? 'already appended (idempotent)' : 'NOT appended (probe answered: ' + String(cpOut || '').slice(0, 80) + ')'))
|
|
4772
4971
|
}
|
|
@@ -5115,7 +5314,7 @@ const confirmationGateLine = confirmationFileGate.verdict === 'skipped'
|
|
|
5115
5314
|
log(confirmationGateLine)
|
|
5116
5315
|
const confirmationGateNote = ' MANDATORY CONFIRMATION FILE GATE RESULT: `' + confirmationGateLine + '`. Write that as a separate line in 08_qe_report.md. The independent QE review MUST still run. If the gate verdict is fail or refused, the final Step-8 grade cannot be A or B; the workflow also enforces that after the reviewer returns. This gate proves only existence/readability; all other ADR checklist items remain advisory.'
|
|
5117
5316
|
await usageProbe('QE')
|
|
5118
|
-
const qePrompt = 'Step 8 (QE - brutal-honesty review, agentic-qe) of /feature-adr for "' + DESC + '" (' + SLUG + '). Adversarially review the SHIPPED code (read it): correctness, edge cases, error handling, and the LOAD-BEARING property the ADR named (ASSERT it has a test that DISCRIMINATES - the recurring lesson: a test that would still pass with the protection deleted is documentation, not a gate). Run this ADR gate before final grading: ' + ADR_FITNESS_CHECKLIST + ' ' + DISCRIMINATION_GATE + ' ' + MUTATION_GATE + ' ' + NO_STUBS_GATE + ' ' + AMENDMENT_GATE + ' Grade A/B/C/D honestly. Assess code-test adequacy + doc-test presence. List CONFIRMED gaps with severity. Write ' + FDIR + '/08_qe_report.md with the primary findings under the exact heading `## Primary QE pass` and an ADR Fitness Checklist section showing PASS/FAIL per ADR and evidence for the Confirmation-linked test. MANDATORY SELF-LEARNING STORE (close the loop, never skip): compare every candidate lesson against the Step-0 recalled LEARNED patterns above. Teach ONLY lessons NOT covered by Step-0 recall. On overlap, run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate; if --reinforce is unavailable, skip the duplicate teach and report `reinforced existing pattern <id>` in the QE report. Store every genuinely new lesson in the CANONICAL BRAIN store at `' + BRAIN + '` so it is NOT lost to a target repo you may have cd`d into. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal) — the `cd ' + BRAIN + ' &&` prefix + `--project ' + BRAIN + '` pin guarantee the lesson lands in the brain regardless of your CWD. Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` (run it verbatim via Bash, do not skip). Do NOT teach trivia or invent gaps. In the return object set roundLessons to the teach:<id> receipts successfully written in this Step 8; when there were none, return roundLessons:[] and a non-empty roundNoNewKnowledge reason derived from this review/reinforcement decision. AUTHORING-TIME CLAIM-CHECK (Deliverable of claim-check-authoring-time): after writing ' + FDIR + '/08_qe_report.md, run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash, parse the {ok, findings, scanned} JSON, and report claimCheck: {findings: N, high: N, medium: N} (counts by severity) in your return object. TAG EVERY QUANTITATIVE CLAIM you write in the report using the convention the checker recognizes as honest — write "1131 tests pass (MEASURED — `npx vitest run`)", never a bare "1131 tests pass" — and where you QUOTE a forbidden phrase as an example (e.g. the retracted "100% passing" framing), backtick the literal so it reads as code, not an assertion, so your own compliant report scans clean. Return {grade, gaps, codeTestsAdequate, docTestsPresent, claimCheck, roundLessons, roundNoNewKnowledge}.' + ABSOLUTE_PATH_NOTE + FINDINGS_LEDGER_NOTE + landedNote + wqNote + confirmationGateNote + PS_GUIDANCE('qe')
|
|
5317
|
+
const qePrompt = 'Step 8 (QE - brutal-honesty review, agentic-qe) of /feature-adr for "' + DESC + '" (' + SLUG + '). Adversarially review the SHIPPED code (read it): correctness, edge cases, error handling, and the LOAD-BEARING property the ADR named (ASSERT it has a test that DISCRIMINATES - the recurring lesson: a test that would still pass with the protection deleted is documentation, not a gate). Run this ADR gate before final grading: ' + ADR_FITNESS_CHECKLIST + ' ' + DISCRIMINATION_GATE + ' ' + MUTATION_GATE + ' ' + NO_STUBS_GATE + ' ' + AMENDMENT_GATE + ' Grade A/B/C/D honestly. Assess code-test adequacy + doc-test presence. List CONFIRMED gaps with severity. Write ' + FDIR + '/08_qe_report.md with the primary findings under the exact heading `## Primary QE pass` and an ADR Fitness Checklist section showing PASS/FAIL per ADR and evidence for the Confirmation-linked test. MANDATORY SELF-LEARNING STORE (close the loop, never skip): compare every candidate lesson against the Step-0 recalled LEARNED patterns above. Teach ONLY lessons NOT covered by Step-0 recall. On overlap, run `dz teach --reinforce "<recalled pattern id or exact text>" --project ' + BRAIN + '` instead of minting a near-duplicate; if --reinforce is unavailable, skip the duplicate teach and report `reinforced existing pattern <id>` in the QE report. Store every genuinely new lesson in the CANONICAL BRAIN store at `' + BRAIN + '` so it is NOT lost to a target repo you may have cd`d into. Via Bash run EXACTLY `' + DZ_TEACH('<a durable reusable lesson from this feature - a rule/pattern/pitfall, NOT a checkpoint echo>', '<0.7-0.95>', '<area>') + '` for each genuine NEW lesson (1-3 max, high-signal) — the `cd ' + BRAIN + ' &&` prefix + `--project ' + BRAIN + '` pin guarantee the lesson lands in the brain regardless of your CWD. Then run `' + DZ + ' statusline --fa-record --slug ' + SLUG + ' --step "Step 8 QE" --recalled auto --run fa:' + SLUG + ' --count-project ' + BRAIN + ' --stored <count taught> --reinforced <count reinforced> --mode ' + MODE + ' --project ' + REPO + '` (run it verbatim via Bash, do not skip). Do NOT teach trivia or invent gaps. In the return object set roundLessons to the teach:<id> receipts successfully written in this Step 8; when there were none, return roundLessons:[] and a non-empty roundNoNewKnowledge reason derived from this review/reinforcement decision. AUTHORING-TIME CLAIM-CHECK (Deliverable of claim-check-authoring-time): after writing ' + FDIR + '/08_qe_report.md, run EXACTLY `dz claim-check ' + FDIR + '/08_qe_report.md --json --fail-on none` via Bash, parse the {ok, findings, scanned} JSON, and report claimCheck: {findings: N, high: N, medium: N} (counts by severity) in your return object. TAG EVERY QUANTITATIVE CLAIM you write in the report using the convention the checker recognizes as honest — write "1131 tests pass (MEASURED — `npx vitest run`)", never a bare "1131 tests pass" — and where you QUOTE a forbidden phrase as an example (e.g. the retracted "100% passing" framing), backtick the literal so it reads as code, not an assertion, so your own compliant report scans clean. AQE SELF-REPORT (instrument-round-b T1/A1, ADR-001 D1): state honestly in your return object whether you actually invoked LIVE agentic-qe in THIS QE pass — aqeInvoked: true or false — and name aqeEvidence: the exact MCP tool you called (e.g. mcp__agentic-qe__quality_assess) or the exact command you ran (e.g. aqe quality assess). This is a SELF-REPORT of what YOU did, never an inference from the run MODE (' + MODE + ') — the mode names an intent this run was configured with, not proof that live agentic-qe was actually called this pass. If you did not call live agentic-qe this pass, say aqeInvoked:false and name what you used instead in aqeEvidence. Return {grade, gaps, codeTestsAdequate, docTestsPresent, claimCheck, roundLessons, roundNoNewKnowledge, aqeInvoked, aqeEvidence}.' + ABSOLUTE_PATH_NOTE + FINDINGS_LEDGER_NOTE + landedNote + wqNote + confirmationGateNote + PS_GUIDANCE('qe')
|
|
5119
5318
|
// CROSS-MODEL QE (load-bearing): resolveStageModel('qe') derives the OTHER family than the resolved
|
|
5120
5319
|
// coder when args.models.qe is unset (coder-codex ⇒ opus; coder-Claude ⇒ codex, or opus if codex absent).
|
|
5121
5320
|
// An explicit args.models.qe wins. A Claude qe spec is merged onto the qe-code-reviewer base (role
|
|
@@ -5499,12 +5698,17 @@ if (qe !== null && qe2Spec !== null) {
|
|
|
5499
5698
|
}
|
|
5500
5699
|
}
|
|
5501
5700
|
if (qe === null) return null
|
|
5502
|
-
return { qe: qe, qeReviewerUsed: qeReviewerUsed, modelUsed: modelsUsed.qe, qe2: qe2, qe2ModelUsed: modelsUsed.qe2 || null
|
|
5503
|
-
|
|
5701
|
+
return { qe: qe, qeReviewerUsed: qeReviewerUsed, modelUsed: modelsUsed.qe, qe2: qe2, qe2ModelUsed: modelsUsed.qe2 || null,
|
|
5702
|
+
bridge: { happened: crossFamilyQeReport ? crossFamilyQeReport.happened : null, rungState: qeCodexRung ? qeCodexRung.state : null, rungReason: qeCodexRung ? qeCodexRung.reason : null, decline: lastCodexDecline } }
|
|
5703
|
+
// Missing bridge is valid for pre-change checkpoints; the debt decision logs the degradation.
|
|
5704
|
+
}, { validate: function (r) { return !!(r && typeof r === 'object' && r.qe && typeof r.qe === 'object' && typeof r.qeReviewerUsed === 'string' && (r.bridge == null || (typeof r.bridge === 'object' && !Array.isArray(r.bridge)))) } })
|
|
5504
5705
|
qe = qeStage ? qeStage.qe : null
|
|
5505
5706
|
let qeReviewerUsed = qeStage ? qeStage.qeReviewerUsed : 'claude'
|
|
5506
5707
|
if (qeStage && qeStage.modelUsed) modelsUsed.qe = qeStage.modelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
|
|
5507
5708
|
if (qeStage && qeStage.qe2ModelUsed) modelsUsed.qe2 = qeStage.qe2ModelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
|
|
5709
|
+
if (qeStage && qeStage.bridge == null && resumedStages.indexOf('qe') !== -1) {
|
|
5710
|
+
log('re-QE debt: resumed pre-change checkpoint without bridge — only usage-switched can be classified from the model label')
|
|
5711
|
+
}
|
|
5508
5712
|
|
|
5509
5713
|
// aqe-ledger-row T3/FR-1/FR-3/A1/A3/A4: the QE step's OWN autorow — who reviewed, how many
|
|
5510
5714
|
// findings, cross-family or not — so the instrument's own footprint in the ledger stops being
|
|
@@ -5536,6 +5740,10 @@ if (resumedStages.indexOf('qe') === -1) {
|
|
|
5536
5740
|
claimCheck: qeClaimCheck,
|
|
5537
5741
|
crossFamily: qeCrossFamily,
|
|
5538
5742
|
qeScope: qeScopeRow,
|
|
5743
|
+
// instrument-round-b T1/FR-1/A1 (ADR-001 D1): last fields on the qe row's own extra — the
|
|
5744
|
+
// agent's self-report of whether it actually invoked live agentic-qe this pass, never derived
|
|
5745
|
+
// from MODE.
|
|
5746
|
+
...aqeSelfReport(qe),
|
|
5539
5747
|
})
|
|
5540
5748
|
} else {
|
|
5541
5749
|
log('run-cost ledger: qe row skipped — stage resumed')
|
|
@@ -5576,19 +5784,17 @@ await capturePairs('code', 'QE', [{ input: codePrompt, output: codeStage, evalua
|
|
|
5576
5784
|
await capturePairs('qe', 'QE', [{ input: qePrompt, output: qeStage ? qeStage.qe : null, evaluation: { grade: qe ? qe.grade : null, gradedBy: qeReviewerUsed, lessonsInjected: [] }, provenance: { model: String(modelsUsed.qe || ''), family: tpFamily(qeReviewerUsed), role: 'reviewer' } }])
|
|
5577
5785
|
|
|
5578
5786
|
// ── re-QE debt emission (backlog 6b40e667) — mirror of harness-core/src/reqe.ts ──
|
|
5579
|
-
//
|
|
5580
|
-
//
|
|
5581
|
-
//
|
|
5582
|
-
//
|
|
5583
|
-
// switch that kept cross-family QE, or the codex-unavailable Claude belt (no override), creates no
|
|
5584
|
-
// debt. A RESUMED qe never re-emits (the original run emitted; a settlement must not be clobbered).
|
|
5787
|
+
// Record same-family reviews after usage switches, failed probes, fallbacks or explicit QE pins
|
|
5788
|
+
// as machine debt. Checkpoint-carried bridge facts preserve the cause on resume; routing OFF is
|
|
5789
|
+
// excluded explicitly. Missing old checkpoint facts degrade visibly. The guards below preserve
|
|
5790
|
+
// existing debt and this run's settlement when a resumed QE retries the write.
|
|
5585
5791
|
let reqeDue = false
|
|
5586
5792
|
{
|
|
5587
5793
|
const reqeFamOf = function (s) { return /codex|gpt|openai/i.test(String(s || '')) ? 'openai' : 'claude' }
|
|
5588
|
-
const
|
|
5589
|
-
if (
|
|
5794
|
+
const reqeDecision = reqeEmitDecision({ coderUsed: coderUsed, qeReviewerUsed: qeReviewerUsed, qeModelLabel: modelsUsed.qe, routingRequested: routingRequested, bridge: qeStage ? qeStage.bridge : undefined })
|
|
5795
|
+
if (reqeDecision.emit) {
|
|
5590
5796
|
reqeDue = true
|
|
5591
|
-
const reqeDebt = { schema: 'reqe-due-1', slug: SLUG, coderFamily: reqeFamOf(coderUsed), qeFamily: reqeFamOf(qeReviewerUsed), qeGrade: (qe && qe.grade) ? String(qe.grade) : null,
|
|
5797
|
+
const reqeDebt = { schema: 'reqe-due-1', slug: SLUG, coderFamily: reqeFamOf(coderUsed), qeFamily: reqeFamOf(qeReviewerUsed), qeGrade: (qe && qe.grade) ? String(qe.grade) : null, cause: reqeDecision.cause, bridge: qeStage ? qeStage.bridge : undefined, reason: reqeDecision.reason, emittedAt: null, runStamp: qeHash }
|
|
5592
5798
|
// IDEMPOTENT + VERIFIED (reqe QE #2 + r2 #2/#3): a RESUMED qe re-runs this block (the original
|
|
5593
5799
|
// run may have died between the qe checkpoint and this write), but: an existing due file is
|
|
5594
5800
|
// never clobbered; a settlement blocks re-emission ONLY when it carries THIS run's runStamp (an
|
|
@@ -5602,8 +5808,10 @@ let reqeDue = false
|
|
|
5602
5808
|
const emitText = String(emitOut || '')
|
|
5603
5809
|
if (/REQE-SETTLED-THIS-RUN/.test(emitText)) log('re-QE debt: THIS run’s debt was already settled — not re-opened')
|
|
5604
5810
|
else if (/REQE-EXISTS/.test(emitText)) log('re-QE debt: already recorded for ' + SLUG + ' — not overwritten')
|
|
5605
|
-
else if (emitText.indexOf('reqe-due-1') !== -1) log('re-QE DEBT recorded: Step-8 ran same-family
|
|
5811
|
+
else if (emitText.indexOf('reqe-due-1') !== -1) log('re-QE DEBT recorded: Step-8 ran same-family; cause=' + reqeDecision.cause + ' — run `dz reqe --slug ' + SLUG + '` for the independent cross-family pass')
|
|
5606
5812
|
else log('re-QE debt write UNVERIFIED (agent returned no readback) — the debt may be missing on disk; reqeDue=true is still reported, record it manually via features/' + SLUG + '/.fa-state/reqe-due.json')
|
|
5813
|
+
} else {
|
|
5814
|
+
log('re-QE debt: ' + (reqeDecision.degraded ? '' : 'none — ') + reqeDecision.reason)
|
|
5607
5815
|
}
|
|
5608
5816
|
}
|
|
5609
5817
|
|