@dzhechkov/skills-feature-adr 1.5.11 → 1.5.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +9 -9
- package/README.md +132 -2
- package/bin/cli.js +0 -0
- package/package.json +6 -5
- package/sbom.json +8 -8
- package/templates/.claude/skills/feature-adr/modules/00-complexity-router.md +21 -1
- package/templates/.claude/skills/feature-adr/modules/06-implementation-plan.md +13 -6
- package/templates/.claude/skills/feature-adr/modules/07-code.md +1 -0
- package/templates/.claude/skills/feature-adr/modules/08-qe.md +80 -0
- package/templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs +110 -2
- package/templates/.claude/workflows/feature-adr.js +1099 -44
package/.dz-manifest.json
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
},
|
|
14
14
|
{
|
|
15
15
|
"path": "README.md",
|
|
16
|
-
"sha256": "
|
|
16
|
+
"sha256": "619a231352a602c736ce6eec1e5fa79a056dc2f5f80f6b35eed236f9670f7337"
|
|
17
17
|
},
|
|
18
18
|
{
|
|
19
19
|
"path": "bin/cli.js",
|
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
},
|
|
26
26
|
{
|
|
27
27
|
"path": "package.json",
|
|
28
|
-
"sha256": "
|
|
28
|
+
"sha256": "07537ebbd424c668a37a42b670c033a4d3134a9e7e13803f426aa31149a85b58"
|
|
29
29
|
},
|
|
30
30
|
{
|
|
31
31
|
"path": "src/cli.js",
|
|
@@ -121,7 +121,7 @@
|
|
|
121
121
|
},
|
|
122
122
|
{
|
|
123
123
|
"path": "templates/.claude/skills/feature-adr/modules/00-complexity-router.md",
|
|
124
|
-
"sha256": "
|
|
124
|
+
"sha256": "6a98ca7af49aabcb2e14a081815f1167740ac4dba2e0cbb938c87aceac47bfa3"
|
|
125
125
|
},
|
|
126
126
|
{
|
|
127
127
|
"path": "templates/.claude/skills/feature-adr/modules/01-requirements.md",
|
|
@@ -149,15 +149,15 @@
|
|
|
149
149
|
},
|
|
150
150
|
{
|
|
151
151
|
"path": "templates/.claude/skills/feature-adr/modules/06-implementation-plan.md",
|
|
152
|
-
"sha256": "
|
|
152
|
+
"sha256": "5c8d4c79d5329702b8c75b9afb36ae335aa63c99e74045c0afa98e12658ee975"
|
|
153
153
|
},
|
|
154
154
|
{
|
|
155
155
|
"path": "templates/.claude/skills/feature-adr/modules/07-code.md",
|
|
156
|
-
"sha256": "
|
|
156
|
+
"sha256": "fb054591e86d55cee184eab96b5f0440053ada8fa9edf619e4a66ec5ac371587"
|
|
157
157
|
},
|
|
158
158
|
{
|
|
159
159
|
"path": "templates/.claude/skills/feature-adr/modules/08-qe.md",
|
|
160
|
-
"sha256": "
|
|
160
|
+
"sha256": "7ab9b7b1523289986a1f51b4cbbebf4410d96ad4a153f22fa8613e8a0f4f5fa7"
|
|
161
161
|
},
|
|
162
162
|
{
|
|
163
163
|
"path": "templates/.claude/skills/feature-adr/modules/09-fleet-qe.md",
|
|
@@ -249,7 +249,7 @@
|
|
|
249
249
|
},
|
|
250
250
|
{
|
|
251
251
|
"path": "templates/.claude/skills/feature-adr/scripts/check-plan-completeness.mjs",
|
|
252
|
-
"sha256": "
|
|
252
|
+
"sha256": "08e69d2e63350fe2dc64483226fcd30901b77a8e930ca6280ef71181c67675b4"
|
|
253
253
|
},
|
|
254
254
|
{
|
|
255
255
|
"path": "templates/.claude/skills/feature-adr/scripts/markdown-masker.mjs",
|
|
@@ -317,7 +317,7 @@
|
|
|
317
317
|
},
|
|
318
318
|
{
|
|
319
319
|
"path": "templates/.claude/workflows/feature-adr.js",
|
|
320
|
-
"sha256": "
|
|
320
|
+
"sha256": "e96c5280ad21b604036cc912418c9627b0ff0ae8ca1436c58dcff951ba9a034b"
|
|
321
321
|
},
|
|
322
322
|
{
|
|
323
323
|
"path": "templates/lib/memory-protocol.md",
|
|
@@ -329,5 +329,5 @@
|
|
|
329
329
|
}
|
|
330
330
|
]
|
|
331
331
|
},
|
|
332
|
-
"signature": "
|
|
332
|
+
"signature": "CfTJgsBbekrZE+IbzaQFLofuRDZ3B2CJTqPk6gjeOiLXhChXiPGkqxaWKCT+hdQTy6ORB7fu6vfMx37rwJf6Cg=="
|
|
333
333
|
}
|
package/README.md
CHANGED
|
@@ -208,6 +208,32 @@ run; it now survives it in `recordFailures`.
|
|
|
208
208
|
|
|
209
209
|
Requires `@dzhechkov/harness-core >= 0.6.1`.
|
|
210
210
|
|
|
211
|
+
### The QE instrument writes its own ledger row (aqe-ledger-row)
|
|
212
|
+
|
|
213
|
+
Before this, the run-cost ledger recorded `plan`, `full`, `design-gate` and friends automatically, but
|
|
214
|
+
the pass that actually reviews the code — Step 8 QE — left zero autorows: every `qe`/`impl` row in the
|
|
215
|
+
ledger was hand-entered, with no model, no findings count, no cross-family signal. The `Workflow`
|
|
216
|
+
pipeline now writes two additional autorows, both additive-only (existing rows are byte-identical to
|
|
217
|
+
before — `appendRunCostRow` gained an optional fourth `extra` argument, spread in only after every
|
|
218
|
+
pre-existing field):
|
|
219
|
+
|
|
220
|
+
- **`impl`** — written right after the Step 7.5 landing barrier settles: `coder`, `coderFamily`, and
|
|
221
|
+
`landed` (the barrier's verdict string, or `skipped-claude-sync` for a synchronous Claude coder that
|
|
222
|
+
never runs the barrier). Writing it here — before Step 8 starts — means the `qe` row's
|
|
223
|
+
`minutesSincePrev` measures the QE step alone, not QE-plus-code.
|
|
224
|
+
- **`qe`** — written right after the QE step resolves: `reviewer` (the model, or `null` if unknown),
|
|
225
|
+
`reviewerFamily`, `qeRole` (`qe-code-reviewer` for Claude; `codex-review`/`codex-exec` for Codex by
|
|
226
|
+
scope mode; never guessed), `grade`, `gradeSource` (if known), `findings` + `findingsBySeverity` +
|
|
227
|
+
`findingsSource` (normalized from the QE `gaps[]`; an unknown severity counts as `other`, never
|
|
228
|
+
dropped; no `gaps` array at all reads as `findings:0`, `findingsSource:'no-gaps-array'`), `claimCheck`
|
|
229
|
+
(if run), `crossFamily` (bool — reviewer family differs from coder family), and `qeScope` (if the
|
|
230
|
+
reviewer ran scoped, e.g. Codex's `mode`/`ref`/`files`).
|
|
231
|
+
|
|
232
|
+
Both rows are skipped — with a logged reason, never silently — when their stage resumed from a
|
|
233
|
+
checkpoint (`resumedStages`), so a resumed run never double-pays the ledger. Plain-mode runs (the
|
|
234
|
+
interactive SKILL, not the ultracode `Workflow`) carry the same fields as a manual step at the end of
|
|
235
|
+
`modules/08-qe.md` — see that module for the exact command.
|
|
236
|
+
|
|
211
237
|
### Step 0 writes the assessment down, and the acid check gets its input back (v1.5.0)
|
|
212
238
|
|
|
213
239
|
Step 0 classifies the feature and now **writes `00_complexity_assessment.md` before it returns** — the
|
|
@@ -416,10 +442,56 @@ cross-family QE is silently lost — on exactly the big features that need it mo
|
|
|
416
442
|
- **Every fallback names its cause.** The reason carried into
|
|
417
443
|
`opus (cross-family QE DID NOT happen — …)` comes from a locked taxonomy —
|
|
418
444
|
`timeout` (narrow the scope) · `no-verdict` · `tool-error` (fix the invocation) · `unusable-output` ·
|
|
419
|
-
`unavailable` (fix the account/model) · `over-ceiling
|
|
420
|
-
|
|
445
|
+
`unavailable` (fix the account/model) · `over-ceiling` · `scope-not-established` (mode A's scope
|
|
446
|
+
could not be built — see below) · `base-ref-not-established` (the scope's base-ref probe failed or
|
|
447
|
+
was unparseable — see below). A timeout and an unusable output can never render the same string,
|
|
448
|
+
because the operator's next move differs.
|
|
421
449
|
- **The pipeline still never blocks on Codex.** Both modes fail into the same Claude belt as before.
|
|
422
450
|
|
|
451
|
+
**Mode A's `--uncommitted` pass runs in an ISOLATED scope-repo, never on the shared tree.** MEASURED
|
|
452
|
+
2026-09-17 (run `wf_95211e0f`): on a hub with other dirty packages, `codex review --uncommitted`
|
|
453
|
+
wandered into unrelated files (`books/`, `features/clean-code-*`) and timed out at 600s having
|
|
454
|
+
reviewed nothing of the run's own feature — cross-family QE silently lost on exactly the runs that
|
|
455
|
+
need it most. Before mode A is attempted for scope `'uncommitted'` (the default), the pipeline now:
|
|
456
|
+
builds a throwaway git repo under `features/<slug>/.fa-state/review-scope/` containing ONLY the
|
|
457
|
+
ESTABLISHED change set (the same `modeBChanged` measurement mode B already uses — base versions via
|
|
458
|
+
`git show <BASE_REF>:<path>` in one commit, working versions copied on top); verifies the receipt
|
|
459
|
+
(`git status --porcelain` in that repo names EXACTLY the declared files, never "close enough"); and
|
|
460
|
+
runs `codex review --uncommitted` with `repo:` pointed at that isolated tree. Findings come back with
|
|
461
|
+
the scope-repo's own absolute path and are normalized to repo-relative before scoring. A change set
|
|
462
|
+
that is **not established** (`null` — no pre-code baseline) or **established but empty** (`[]` — no
|
|
463
|
+
files changed) refuses BEFORE any dispatch, under one decline kind `scope-not-established` (never a
|
|
464
|
+
silent fallback to `--uncommitted` on the shared tree); a failed scope-repo build or a receipt
|
|
465
|
+
mismatch refuse the same way. Knob: `args.qeIsolatedScope` (default `true`) — `false` restores the
|
|
466
|
+
prior `--uncommitted`-on-the-shared-tree behavior byte-for-byte, with a log line saying so. Scopes
|
|
467
|
+
`commit`/`base` are unaffected.
|
|
468
|
+
|
|
469
|
+
**Fix round 1 hardening (Codex r1 review, 2026-09-17), briefly:** the scope-repo path is validated by
|
|
470
|
+
PATH SEGMENT (`<repo>/features/<slug>/.fa-state/review-scope`, `<slug>` alphanumeric, no `.`/`..`
|
|
471
|
+
segment anywhere), not by a lexical prefix/substring check — a `..`-laced path can no longer walk the
|
|
472
|
+
one destructive `rm -rf` outside `features/`; the script itself repeats the check at runtime (symlink
|
|
473
|
+
+ `case` guard) as a second belt. A base-ref probe that fails or returns something unparseable now
|
|
474
|
+
REFUSES under `base-ref-not-established` instead of silently substituting `HEAD` — a bad ref used to
|
|
475
|
+
make Codex review a full-file addition instead of the real modification. The receipt is now a CONTENT
|
|
476
|
+
check, not just a pathname list: the scope-build script also emits a `sha256sum` of every working
|
|
477
|
+
file, compared against the same pre-measured hashes mode B already computes, so a `cp` that silently
|
|
478
|
+
degraded to a deletion (an unreadable file, one that vanished mid-copy) is caught even though the
|
|
479
|
+
pathname-only receipt would have passed; a genuine `cp` failure now aborts the build rather than being
|
|
480
|
+
read as an intentional deletion. The porcelain receipt parser reads a fixed two-character status
|
|
481
|
+
column (`--no-renames` on the git status call, so a rename can never arrive as the ambiguous
|
|
482
|
+
`old -> new` shape). A declared path is never trimmed and rejects only what the receipt genuinely
|
|
483
|
+
cannot express (control characters, `"`, `\`, backtick, `$`, non-ASCII, a leading `-`) — spaces and
|
|
484
|
+
shell metacharacters are accepted, because every value already passes through the same safe quoting
|
|
485
|
+
function used everywhere else in this file.
|
|
486
|
+
|
|
487
|
+
**Named limit of the isolated scope (owner-facing, so it reads as a boundary, not a defect):** the
|
|
488
|
+
scope-repo review is BLIND to everything outside the declared change set, by construction — that is
|
|
489
|
+
the whole point (it is why the shared-tree run above timed out reviewing unrelated dirty packages).
|
|
490
|
+
A finding phrased as *"file X does not exist"* or *"the twin file is missing"* from mode A is therefore
|
|
491
|
+
an artifact of that intentional narrowness, not a real defect: the twin/sibling file is simply not
|
|
492
|
+
copied into the scope repo. Context that spans beyond the declared files is covered by mode B (which
|
|
493
|
+
is told exactly which files it may open) and by the separate Claude QE pass, never by mode A alone.
|
|
494
|
+
|
|
423
495
|
After a Codex verdict a cheap Claude agent transcribes it into `08_qe_report.md` (mode A takes no
|
|
424
496
|
prompt, so the reviewer cannot be asked to write anything). It is a scribe, not a second reviewer: the
|
|
425
497
|
grade is Codex's and is stated as final.
|
|
@@ -1179,6 +1251,64 @@ pre-code probe that returns nothing no longer becomes an all-null baseline that
|
|
|
1179
1251
|
"every target changed".
|
|
1180
1252
|
|
|
1181
1253
|
|
|
1254
|
+
`next` — **the plan inherits requirements by contract, not by goodwill** (feature `plan-inherits-requirements`,
|
|
1255
|
+
staged: not yet versioned or published). Two swarms plus a cross-family check measured that the implementation
|
|
1256
|
+
plan referenced only 95 of 339 requirement ids across 8 M/L features (28 %), lost one requirement without a trace
|
|
1257
|
+
and introduced one contradiction, while the norm "every FR-N → a task" lived only in the module text — neither
|
|
1258
|
+
the planner prompt nor the K2 gate enforced it. Now:
|
|
1259
|
+
|
|
1260
|
+
- **C8 — requirement coverage.** The K2 gate reads `01_requirements.md`, extracts every id declared at line start
|
|
1261
|
+
in the four corpus shapes (heading / bold / list / table; `FR-N`, `NFR-N`, `AC-N`, `C-N`, with an optional
|
|
1262
|
+
letter group and dotted sub-number — measured over 394 requirement files) and checks each by word boundary in
|
|
1263
|
+
the plan. By default a WARN with the exact count; under `--require-requirements` (which the pipeline passes)
|
|
1264
|
+
every missing id is its own FAIL line. An absent `01_requirements.md` is a WARN naming the absence, never a
|
|
1265
|
+
skip; a file declaring no ids in the contract shapes says so. NAMED LIMIT, in the same form as C1's: C8 is a
|
|
1266
|
+
grep — a prose mention satisfies it; "mentioned but not tasked" is not caught.
|
|
1267
|
+
- **C1 counts decisions by heading, not by filename.** `# ADR-NNN` / `## ADR-NNN` headings inside each ADR file
|
|
1268
|
+
are the decisions the plan owes a task; a file holding four decisions now yields four checks, not one. No
|
|
1269
|
+
heading ⇒ filename prefix with a WARN.
|
|
1270
|
+
- **An unclosed code fence on the declaration side is NOT-ESTABLISHED.** Masking it silently dropped every id
|
|
1271
|
+
after it (measured); restoring it silently established a heading that was only example code. Neither silent
|
|
1272
|
+
reading is honest, so the gate refuses with the file named — close the fence, rerun. The corpus has 0 such
|
|
1273
|
+
files out of 831.
|
|
1274
|
+
- **The Step-6 planner is told the inputs by NAME** (`01_requirements.md`, every `03_adr/*.md`,
|
|
1275
|
+
`05_architecture.md`, `03.5_ideation_report.md` / `04_domain_model.md` when present) and the C8 contract in the
|
|
1276
|
+
same clause as C1/C2/C4 — the lesson that a gate whose contract is not named in the authoring prompt produces
|
|
1277
|
+
refusal after refusal.
|
|
1278
|
+
- **ONE automatic repair round.** A genuine script-verdict FAIL re-dispatches the planner with the FAIL lines
|
|
1279
|
+
("close EXACTLY these gaps, keep everything else"). Because that sentence is a prompt and not a guarantee, the
|
|
1280
|
+
round is bracketed: the plan is backed up first (no backup ⇒ no repair; a stale `.pre-repair` or a symlinked
|
|
1281
|
+
plan refuses), snapshots before/after compare byte length, every `EXPECTED_CODE_TARGETS` line and every task
|
|
1282
|
+
heading line WITH multiplicity, a snapshot that did not complete REJECTS (never fails open), the re-gate runs
|
|
1283
|
+
before the backup is archived, and a rejected or still-failing repair is restored from the backup with the
|
|
1284
|
+
restore PROVEN by POSIX `cksum` + length. `planGateAttempts` counts gate runs; `planRepair` carries the
|
|
1285
|
+
outcome; a `plan-repair` ledger row is written either way. NAMED LIMIT: task bodies are not proven preserved
|
|
1286
|
+
by any metric — a repair that keeps every heading, every target and 80 % of the bytes while gutting prose is
|
|
1287
|
+
undetectable by construction.
|
|
1288
|
+
- The pure halves (`shellQuote`, `planBackupCmd`, `planRestoreCmd`, `planArchiveBackupCmd`, `planSnapshotCmd`,
|
|
1289
|
+
`snapshotBlock`, `snapshotNumber`, `parsePlanSnapshot`) live in `@dzhechkov/harness-core` and are body-pinned
|
|
1290
|
+
against the inline copies by the drift guard. Three Codex review rounds (C, C, D) — every finding either fixed
|
|
1291
|
+
or named above; the full account is in `features/plan-inherits-requirements/08_qe_report.md`.
|
|
1292
|
+
|
|
1293
|
+
Also staged: **the coder now gets the same recall lane the planner already had** (feature
|
|
1294
|
+
`coder-reads-and-recall`). `1.5.9` wired advisory decision-point micro-recall into Step 3 (ADR
|
|
1295
|
+
selection) and Step 6 (plan routing) only — Step 7 (Code) never called `prepareDecisionRecall` at
|
|
1296
|
+
all, so any lesson reaching the coder was a night-shift human pasting it into the brief by hand.
|
|
1297
|
+
`drDecisionShape` now knows a third decision kind, `code-implementation` → `step-7` /
|
|
1298
|
+
`feature-adr-decision-code-implementation`, kept apart from the plan's own `step-6` bandit context.
|
|
1299
|
+
The call sits INSIDE the code stage's checkpoint, before any of the three places that read the
|
|
1300
|
+
coder's prompt (the Claude dispatch, the Codex dispatch, and the training-pair capture) — so a
|
|
1301
|
+
resumed stage neither re-spends the recall nor loses it, and all three see the identical
|
|
1302
|
+
recall-augmented text. Separately measured (Step 0 of this feature, instrument: host workflow
|
|
1303
|
+
records → the coding-stage agent's own tool-call transcript): a comment added 24.08 naming the plan,
|
|
1304
|
+
every ADR, the architecture doc, requirements and the domain model by file name moved how often
|
|
1305
|
+
Claude-family coders opened `01_requirements.md` from 39 % (11 of 28 runs) before the change to 71 %
|
|
1306
|
+
(5 of 7) after — a real jump, but on **n = 7**, not a controlled comparison, and it says nothing
|
|
1307
|
+
about Codex-family coders: they were 77 % of the post-change sample and are invisible to this
|
|
1308
|
+
instrument (a Codex coder's transcript carries exactly one entry, the dispatch itself). Full method,
|
|
1309
|
+
the two false reads caught before the number was trusted, and the raw counts are in
|
|
1310
|
+
`features/coder-reads-and-recall/00_complexity_assessment.md`.
|
|
1311
|
+
|
|
1182
1312
|
`1.5.3` — **the workflow stops crashing on the way into Step 7.** `1.5.2` shipped a workflow that
|
|
1183
1313
|
CALLED three helpers it never defined — `changeSetProbeCmd`, `parseHashProbe`, `changedFromHashes`
|
|
1184
1314
|
(5 call sites, 0 definitions). `QE_SCOPE` defaults to `uncommitted`, so the guarded branch was true
|
package/bin/cli.js
CHANGED
|
File without changes
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@dzhechkov/skills-feature-adr",
|
|
3
|
-
"version": "1.5.
|
|
3
|
+
"version": "1.5.13",
|
|
4
4
|
"description": "Adaptive Feature Development skill pack for Claude Code — 11-step pipeline with Complexity Router (S/M/L/XL), ADR-driven architecture, 15 agentic-qe skills, multi-agent fleet QE. Supports --full-qe, --full-qe-extended, --with-learning, and --knowledge-extractor modes.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"skills-feature-adr": "./bin/cli.js"
|
|
@@ -15,6 +15,10 @@
|
|
|
15
15
|
".dz-manifest.json",
|
|
16
16
|
"sbom.json"
|
|
17
17
|
],
|
|
18
|
+
"scripts": {
|
|
19
|
+
"test": "node --test \"test/**/*.test.js\"",
|
|
20
|
+
"prepack": "node -e \"const fs=require('fs');const bad=['.claude','.skills-feature-adr.json'].filter(p=>fs.existsSync(p));if(bad.length){console.error('prepack guard: stray init artifacts in package dir: '+bad.join(', ')+' — remove before packing');process.exit(1)}\""
|
|
21
|
+
},
|
|
18
22
|
"keywords": [
|
|
19
23
|
"claude",
|
|
20
24
|
"claude-code",
|
|
@@ -59,8 +63,5 @@
|
|
|
59
63
|
},
|
|
60
64
|
"publishConfig": {
|
|
61
65
|
"access": "public"
|
|
62
|
-
},
|
|
63
|
-
"scripts": {
|
|
64
|
-
"test": "node --test \"test/**/*.test.js\""
|
|
65
66
|
}
|
|
66
|
-
}
|
|
67
|
+
}
|
package/sbom.json
CHANGED
|
@@ -35,7 +35,7 @@
|
|
|
35
35
|
"hashes": [
|
|
36
36
|
{
|
|
37
37
|
"alg": "SHA-256",
|
|
38
|
-
"content": "
|
|
38
|
+
"content": "619a231352a602c736ce6eec1e5fa79a056dc2f5f80f6b35eed236f9670f7337"
|
|
39
39
|
}
|
|
40
40
|
]
|
|
41
41
|
},
|
|
@@ -69,7 +69,7 @@
|
|
|
69
69
|
},
|
|
70
70
|
{
|
|
71
71
|
"name": "dz:canonical-json-sha256-v2",
|
|
72
|
-
"value": "
|
|
72
|
+
"value": "07537ebbd424c668a37a42b670c033a4d3134a9e7e13803f426aa31149a85b58"
|
|
73
73
|
}
|
|
74
74
|
]
|
|
75
75
|
},
|
|
@@ -309,7 +309,7 @@
|
|
|
309
309
|
"hashes": [
|
|
310
310
|
{
|
|
311
311
|
"alg": "SHA-256",
|
|
312
|
-
"content": "
|
|
312
|
+
"content": "6a98ca7af49aabcb2e14a081815f1167740ac4dba2e0cbb938c87aceac47bfa3"
|
|
313
313
|
}
|
|
314
314
|
]
|
|
315
315
|
},
|
|
@@ -379,7 +379,7 @@
|
|
|
379
379
|
"hashes": [
|
|
380
380
|
{
|
|
381
381
|
"alg": "SHA-256",
|
|
382
|
-
"content": "
|
|
382
|
+
"content": "5c8d4c79d5329702b8c75b9afb36ae335aa63c99e74045c0afa98e12658ee975"
|
|
383
383
|
}
|
|
384
384
|
]
|
|
385
385
|
},
|
|
@@ -389,7 +389,7 @@
|
|
|
389
389
|
"hashes": [
|
|
390
390
|
{
|
|
391
391
|
"alg": "SHA-256",
|
|
392
|
-
"content": "
|
|
392
|
+
"content": "fb054591e86d55cee184eab96b5f0440053ada8fa9edf619e4a66ec5ac371587"
|
|
393
393
|
}
|
|
394
394
|
]
|
|
395
395
|
},
|
|
@@ -399,7 +399,7 @@
|
|
|
399
399
|
"hashes": [
|
|
400
400
|
{
|
|
401
401
|
"alg": "SHA-256",
|
|
402
|
-
"content": "
|
|
402
|
+
"content": "7ab9b7b1523289986a1f51b4cbbebf4410d96ad4a153f22fa8613e8a0f4f5fa7"
|
|
403
403
|
}
|
|
404
404
|
]
|
|
405
405
|
},
|
|
@@ -629,7 +629,7 @@
|
|
|
629
629
|
"hashes": [
|
|
630
630
|
{
|
|
631
631
|
"alg": "SHA-256",
|
|
632
|
-
"content": "
|
|
632
|
+
"content": "08e69d2e63350fe2dc64483226fcd30901b77a8e930ca6280ef71181c67675b4"
|
|
633
633
|
}
|
|
634
634
|
]
|
|
635
635
|
},
|
|
@@ -799,7 +799,7 @@
|
|
|
799
799
|
"hashes": [
|
|
800
800
|
{
|
|
801
801
|
"alg": "SHA-256",
|
|
802
|
-
"content": "
|
|
802
|
+
"content": "e96c5280ad21b604036cc912418c9627b0ff0ae8ca1436c58dcff951ba9a034b"
|
|
803
803
|
}
|
|
804
804
|
]
|
|
805
805
|
},
|
|
@@ -103,19 +103,39 @@ Runs after tier classification, only when `{AGENTIC_QE_MODE}` = `direct` | `dire
|
|
|
103
103
|
|
|
104
104
|
Step 1 folds `{LEARNED_PATTERNS}` into the requirements brief as "lessons from previous features" — advisory context only, never requirements themselves. Why here: Step 0 is the one point where recall can be keyed to the feature's phase and domain — per-prompt hook auto-injection cannot see which pipeline step is running.
|
|
105
105
|
|
|
106
|
+
### 7. Task Kind Classification (experiment-envelope, ADR-001)
|
|
107
|
+
|
|
108
|
+
Classify the feature as exactly ONE of the six task kinds below, and write `Task kind: <x>` as its
|
|
109
|
+
own line in the artifact. This feeds the experiment envelope the conveyor builds after this step —
|
|
110
|
+
downstream learning reads the kind, not a paraphrase, so the line must use one of the exact words.
|
|
111
|
+
|
|
112
|
+
| Task kind | Criterion |
|
|
113
|
+
|-----------|-----------|
|
|
114
|
+
| `feature` | A genuinely new capability, adapter, command, or skill did not exist before this run. |
|
|
115
|
+
| `bugfix` | The change corrects observed incorrect behavior in existing code — a defect, not an absence. |
|
|
116
|
+
| `refactor` | Behavior is unchanged; the change restructures, renames, or simplifies existing code. |
|
|
117
|
+
| `tooling` | The change is to build/CI/dev-tooling/scripts rather than to the product's own runtime behavior. |
|
|
118
|
+
| `docs` | The change is documentation-only (README, ADR prose, comments) with no code delta. |
|
|
119
|
+
| `research` | The deliverable is a finding or a design decision, not a shipped code change. |
|
|
120
|
+
|
|
121
|
+
If `args.taskKind` was passed explicitly by the caller, it OVERRIDES this classification — record
|
|
122
|
+
both: `Task kind: <forced value> (forced by the caller)` as the kind of record, and your own
|
|
123
|
+
classification separately if it differs, the same pattern Step 0 already uses for a forced tier.
|
|
124
|
+
|
|
106
125
|
## Output
|
|
107
126
|
|
|
108
127
|
Set the following variables:
|
|
109
128
|
|
|
110
129
|
```
|
|
111
130
|
{COMPLEXITY_TIER} = S | M | L | XL
|
|
131
|
+
{TASK_KIND} = feature | bugfix | refactor | tooling | docs | research
|
|
112
132
|
{ACTIVE_STEPS} = [0, 1, 6, 7, 8] # example for S
|
|
113
133
|
{TIME_BUDGET} = { requirements: 2, planning: 0, implementation: 10, qe: 3 }
|
|
114
134
|
{DIMENSION_SCORES} = { files: 1, domains: 1, integrations: 1, breaking: 1, models: 1, crosscutting: 1 }
|
|
115
135
|
{LEARNED_PATTERNS} = [ ... ] # top-3 by confidence from memory_query; [] in reference mode / no hits / error
|
|
116
136
|
```
|
|
117
137
|
|
|
118
|
-
Create artifact: `features/<slug>/00_complexity_assessment.md`
|
|
138
|
+
Create artifact: `features/<slug>/00_complexity_assessment.md`, including the `Task kind: <x>` line.
|
|
119
139
|
|
|
120
140
|
## Checkpoint 0 Format
|
|
121
141
|
|
|
@@ -127,7 +127,9 @@ For each group, identify:
|
|
|
127
127
|
|
|
128
128
|
Before finalizing the plan, validate completeness:
|
|
129
129
|
|
|
130
|
-
1. Cross-reference every `{REQUIREMENT}` (FR-N) → at least one TASK covers it
|
|
130
|
+
1. Cross-reference every `{REQUIREMENT}` (FR-N) → at least one TASK covers it (C8 — the gate checks
|
|
131
|
+
this by identifier: WARN with a count by default, FAIL under `--require-requirements`, which the
|
|
132
|
+
pipeline passes)
|
|
131
133
|
2. Cross-reference every `{ADR_DECISION}` → at least one TASK implements it
|
|
132
134
|
3. Cross-reference every critical risk from `{QUALITY_RISKS}` → mitigation in some TASK
|
|
133
135
|
4. Name every acid token `A<n>` from `00_complexity_assessment.md` VERBATIM in the plan (C4), each
|
|
@@ -233,11 +235,16 @@ C2 recognises JS/TS, pytest, Go, Rust, JVM and .NET test paths, extensible per p
|
|
|
233
235
|
|
|
234
236
|
Never proceed on a non-zero exit, and never treat empty output as a pass — the last line
|
|
235
237
|
(`K2 plan-completeness: PASS|FAIL|NOT-ESTABLISHED`) is the verdict, and its absence is not one.
|
|
236
|
-
What it checks: C1 every ADR
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
238
|
+
What it checks: C1 every ADR **decision** (a `# ADR-NNN` / `## ADR-NNN` heading INSIDE the file, not
|
|
239
|
+
just the filename prefix — a file with several headings owes several plan citations) has a plan task
|
|
240
|
+
citing it · C2 every ADR Confirmation test path is named in the plan · C3 the `EXPECTED_CODE_TARGETS:`
|
|
241
|
+
block parses line by line · C4 the feature's declared acid corpus is named · C5 (WARN) the
|
|
242
|
+
`Inputs read:` line · C8 every requirement id declared in `01_requirements.md` (`FR-N`, `NFR-N`,
|
|
243
|
+
`AC-N`, `C-N`) is cited by the plan by word boundary — WARN with a count by default, FAIL per missing
|
|
244
|
+
id under `--require-requirements` (the pipeline passes this flag; a bare interactive run of the
|
|
245
|
+
script does not). An S-tier run with no `03_adr/` skips C1/C2 with a note (it cannot be failed for
|
|
246
|
+
ADRs it never had) — unless the plan itself cites `ADR-<n>`, which is NOT-ESTABLISHED. C1 is a grep:
|
|
247
|
+
it catches "forgot entirely", not "mentioned but not tasked".
|
|
241
248
|
|
|
242
249
|
Pass the run's tier so the check cannot be dodged: `--tier=S|M|L|XL`. An M/L/XL feature with no
|
|
243
250
|
`03_adr/` FAILS C1/C2 (an M+ feature owes ADRs); only `--tier=S` — or no tier at all, and then the
|
|
@@ -20,6 +20,7 @@ opus (complex code generation)
|
|
|
20
20
|
- `{ADR_DECISIONS}` from Step 3 (M+)
|
|
21
21
|
- `{DOMAIN_MODEL}` from Step 4 (L/XL)
|
|
22
22
|
- Codebase context (existing patterns, conventions)
|
|
23
|
+
- {LEARNED_PATTERNS} for Step 7 — the decision-recall block (≤3 lessons), appended to the coder prompt by the pipeline
|
|
23
24
|
|
|
24
25
|
## Protocol
|
|
25
26
|
|
|
@@ -311,6 +311,86 @@ Compile all findings into a structured report:
|
|
|
311
311
|
✅ READY FOR MERGE | ❌ NEEDS FIXES | ⚠️ CONDITIONAL APPROVAL
|
|
312
312
|
```
|
|
313
313
|
|
|
314
|
+
### 7.1 Findings ledger (machine-readable)
|
|
315
|
+
|
|
316
|
+
Prose is for people; `dz score`/`dz recap` need a machine-readable surface too (qe-findings-record,
|
|
317
|
+
ADR-001). Write BOTH of these into `08_qe_report.md`, in addition to the prose report above:
|
|
318
|
+
|
|
319
|
+
1. Exactly ONE line, in the PROSE body of the file (never inside a fenced code block, an indented
|
|
320
|
+
code block, a `>` blockquote, or an HTML comment — the parser masks all four before it looks):
|
|
321
|
+
`QE-VERDICT: <A|A-|A+|B|B+|B-|C|C+|C-|D>` — the letter grade you gave above, machine-parseable,
|
|
322
|
+
ASCII hyphen for the sign (the parser also accepts U+2212 as the same sign; write ASCII). Two such
|
|
323
|
+
lines make the report `ambiguous`, never "last wins" — write it once. A line that starts with
|
|
324
|
+
`QE-VERDICT` but gets the grammar wrong (wrong dash, trailing prose, wrong case, no colon) is worse
|
|
325
|
+
than writing nothing: the parser reports `invalid` and REFUSES to fall back to guessing a grade
|
|
326
|
+
from prose — get the one line right rather than close.
|
|
327
|
+
2. A `## Findings ledger` section — this exact heading, occurring EXACTLY ONCE in the file, at the
|
|
328
|
+
top level (never inside a fence/indented block/blockquote/comment either) — with a table directly
|
|
329
|
+
under it, under this EXACT header (copy it verbatim):
|
|
330
|
+
|
|
331
|
+
```markdown
|
|
332
|
+
## Findings ledger
|
|
333
|
+
|
|
334
|
+
| Finding | Severity | Status | Round | Author | Title |
|
|
335
|
+
|---------|----------|--------|-------|--------|-------|
|
|
336
|
+
| F1 | HIGH | fixed | 1 | codex | escaping bug in the token scanner |
|
|
337
|
+
| F2 | MEDIUM | open | 1 | lead | style nit, not blocking |
|
|
338
|
+
```
|
|
339
|
+
|
|
340
|
+
A table found ANYWHERE else — before the heading, after the section ends at the next heading, with
|
|
341
|
+
no heading in the file, or with the heading duplicated — is REFUSED as `outside ledger section`,
|
|
342
|
+
never parsed as the real ledger; a second table found INSIDE the section is REFUSED as `duplicate
|
|
343
|
+
table`. Both refusals name how many of the rejected table's rows were ignored.
|
|
344
|
+
|
|
345
|
+
Closed dictionaries — a row using anything else is REFUSED by the parser, never coerced to the
|
|
346
|
+
nearest known value:
|
|
347
|
+
- **Severity**: `BLOCKER` `CRITICAL` `HIGH` `MEDIUM` `LOW` `INFO`
|
|
348
|
+
- **Status**: `confirmed` `fixed` `partial` `refuted` `named-limit` `open`
|
|
349
|
+
- **Round**: an integer ≥ 1 (which review round raised it)
|
|
350
|
+
- **Author**: `codex` `claude` `lead`
|
|
351
|
+
- **Finding**: a short id token with no spaces (`F1`, `R2-3`, …); **Title**: free text — a literal
|
|
352
|
+
`|` inside Title must be escaped as `\|`, or wrapped in inline code (`` `a|b` ``), or it fragments
|
|
353
|
+
the row into extra columns and the row is refused as malformed.
|
|
354
|
+
|
|
355
|
+
An empty (header-only) table is read as `hollow: true` — worse than no table at all, because it
|
|
356
|
+
claims a ledger exists and says nothing. If there are no findings, omit the section entirely rather
|
|
357
|
+
than writing an empty table.
|
|
358
|
+
|
|
359
|
+
### 7.2 Ledger row for this QE step (mandatory)
|
|
360
|
+
|
|
361
|
+
Plain mode (this SKILL, not the ultracode `Workflow`) has no in-process `appendRunCostRow` — without
|
|
362
|
+
this step the QE instrument's own pass leaves ZERO trace in the run-cost ledger (aqe-ledger-row,
|
|
363
|
+
owner audit 17.09, decision #3).
|
|
364
|
+
|
|
365
|
+
**Resume guard — check this FIRST, before writing anything.** If QE was RESTORED from a checkpoint
|
|
366
|
+
for THIS run (`features/<slug>/.fa-state/checkpoints.jsonl` already contains a line for stage `qe`
|
|
367
|
+
that matches this run) — the row for this review was already written by the run that actually did
|
|
368
|
+
it. Do **NOT** run the command below in that case; instead write exactly this line into
|
|
369
|
+
`08_qe_report.md`: `qe ledger row skipped — stage resumed`. Writing the row anyway would double-pay
|
|
370
|
+
the same review in the ledger (aqe-ledger-row fix-round-1/#3, Codex r1 HIGH #3 — plain mode had no
|
|
371
|
+
resume guard at all before this).
|
|
372
|
+
|
|
373
|
+
Otherwise — QE ran fresh in this pass — after `08_qe_report.md` is written and the grade is final, run:
|
|
374
|
+
|
|
375
|
+
```bash
|
|
376
|
+
dz feature-adr-record --kind ledger --stage qe --slug <slug> --row '<json>' --auto --json
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
`<json>` carries the same fields the ultracode pipeline writes for this row: `reviewer` (the model
|
|
380
|
+
that reviewed, or `null` if unknown — never guessed), `reviewerFamily` (`claude`|`codex`|`null` when
|
|
381
|
+
the reviewer identity is not one of the two known families — never guessed as `claude`), `qeRole`
|
|
382
|
+
(`qe-code-reviewer` for a Claude reviewer; `codex-review`/`codex-exec` for a codex reviewer by scope
|
|
383
|
+
mode; `null` otherwise), `grade`, `gradeSource` (if known), `findings` (gap count), `findingsBySeverity`
|
|
384
|
+
(normalized `sev` -> count, unknown severities counted as `other`, never dropped), `findingsSource`
|
|
385
|
+
(`gaps` or `no-gaps-array`), `claimCheck` (if run), `crossFamily` (bool: reviewer family != coder
|
|
386
|
+
family — `null` when either family is unknown, since a same/cross-family verdict cannot be asserted
|
|
387
|
+
without both), `qeScope` (if the reviewer ran scoped, e.g. codex `mode`/`ref`/`files`). Insert the
|
|
388
|
+
writer's verdict (`{verdict:"written"|...}`) verbatim into `08_qe_report.md` so the record is
|
|
389
|
+
auditable from the artifact itself. NAMED LIMIT (layer 4 of the cost-of-detection ladder, said
|
|
390
|
+
honestly): this step is a skill instruction, not a code gate — nothing forces the agent to run it or
|
|
391
|
+
to check the resume guard first; only its PRESENCE in this module is deterministically pinned (grep
|
|
392
|
+
for the command above and for the resume-guard sentence).
|
|
393
|
+
|
|
314
394
|
### 8. QE Pattern Store (Direct Mode only)
|
|
315
395
|
|
|
316
396
|
When `{AGENTIC_QE_MODE}` = `direct` | `direct-extended`, after the gap loop is closed and the verdict is set:
|