@dzhechkov/skills-feature-adr 1.5.12 → 1.5.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.dz-manifest.json CHANGED
@@ -13,7 +13,7 @@
13
13
  },
14
14
  {
15
15
  "path": "README.md",
16
- "sha256": "e72e7ae2a5adc046598feb5de0bcac114d5aa961571f57d2aa530186c1dbb4b3"
16
+ "sha256": "619a231352a602c736ce6eec1e5fa79a056dc2f5f80f6b35eed236f9670f7337"
17
17
  },
18
18
  {
19
19
  "path": "bin/cli.js",
@@ -25,7 +25,7 @@
25
25
  },
26
26
  {
27
27
  "path": "package.json",
28
- "sha256": "8b0f75ec75e5f2a4e121c65a625abe6c9a8fb4b12732349293b78232348b2234"
28
+ "sha256": "07537ebbd424c668a37a42b670c033a4d3134a9e7e13803f426aa31149a85b58"
29
29
  },
30
30
  {
31
31
  "path": "src/cli.js",
@@ -157,7 +157,7 @@
157
157
  },
158
158
  {
159
159
  "path": "templates/.claude/skills/feature-adr/modules/08-qe.md",
160
- "sha256": "7973676cc9a5a6d8429c0bcdbe163bb5dc16e2d2f28a72698e44ad72d234e199"
160
+ "sha256": "7ab9b7b1523289986a1f51b4cbbebf4410d96ad4a153f22fa8613e8a0f4f5fa7"
161
161
  },
162
162
  {
163
163
  "path": "templates/.claude/skills/feature-adr/modules/09-fleet-qe.md",
@@ -317,7 +317,7 @@
317
317
  },
318
318
  {
319
319
  "path": "templates/.claude/workflows/feature-adr.js",
320
- "sha256": "8900ff6bc777a078aa4e37d8e28a2ffbc612187354995f726daf93e6f87d3109"
320
+ "sha256": "e96c5280ad21b604036cc912418c9627b0ff0ae8ca1436c58dcff951ba9a034b"
321
321
  },
322
322
  {
323
323
  "path": "templates/lib/memory-protocol.md",
@@ -329,5 +329,5 @@
329
329
  }
330
330
  ]
331
331
  },
332
- "signature": "lPrCKcYVN87Vvd2+Xm6HFSVQdv8eu1AXQ1yGKm42Vad53oNafZvw3gF50Jm4pG1FkKd9vJ1SH/0zHJVNZi9bAQ=="
332
+ "signature": "CfTJgsBbekrZE+IbzaQFLofuRDZ3B2CJTqPk6gjeOiLXhChXiPGkqxaWKCT+hdQTy6ORB7fu6vfMx37rwJf6Cg=="
333
333
  }
package/README.md CHANGED
@@ -208,6 +208,32 @@ run; it now survives it in `recordFailures`.
208
208
 
209
209
  Requires `@dzhechkov/harness-core >= 0.6.1`.
210
210
 
211
+ ### The QE instrument writes its own ledger row (aqe-ledger-row)
212
+
213
+ Before this, the run-cost ledger recorded `plan`, `full`, `design-gate` and friends automatically, but
214
+ the pass that actually reviews the code — Step 8 QE — left zero autorows: every `qe`/`impl` row in the
215
+ ledger was hand-entered, with no model, no findings count, no cross-family signal. The `Workflow`
216
+ pipeline now writes two additional autorows, both additive-only (existing rows are byte-identical to
217
+ before — `appendRunCostRow` gained an optional fourth `extra` argument, spread in only after every
218
+ pre-existing field):
219
+
220
+ - **`impl`** — written right after the Step 7.5 landing barrier settles: `coder`, `coderFamily`, and
221
+ `landed` (the barrier's verdict string, or `skipped-claude-sync` for a synchronous Claude coder that
222
+ never runs the barrier). Writing it here — before Step 8 starts — means the `qe` row's
223
+ `minutesSincePrev` measures the QE step alone, not QE-plus-code.
224
+ - **`qe`** — written right after the QE step resolves: `reviewer` (the model, or `null` if unknown),
225
+ `reviewerFamily`, `qeRole` (`qe-code-reviewer` for Claude; `codex-review`/`codex-exec` for Codex by
226
+ scope mode; never guessed), `grade`, `gradeSource` (if known), `findings` + `findingsBySeverity` +
227
+ `findingsSource` (normalized from the QE `gaps[]`; an unknown severity counts as `other`, never
228
+ dropped; no `gaps` array at all reads as `findings:0`, `findingsSource:'no-gaps-array'`), `claimCheck`
229
+ (if run), `crossFamily` (bool — reviewer family differs from coder family), and `qeScope` (if the
230
+ reviewer ran scoped, e.g. Codex's `mode`/`ref`/`files`).
231
+
232
+ Both rows are skipped — with a logged reason, never silently — when their stage resumed from a
233
+ checkpoint (`resumedStages`), so a resumed run never double-pays the ledger. Plain-mode runs (the
234
+ interactive SKILL, not the ultracode `Workflow`) carry the same fields as a manual step at the end of
235
+ `modules/08-qe.md` — see that module for the exact command.
236
+
211
237
  ### Step 0 writes the assessment down, and the acid check gets its input back (v1.5.0)
212
238
 
213
239
  Step 0 classifies the feature and now **writes `00_complexity_assessment.md` before it returns** — the
@@ -416,10 +442,56 @@ cross-family QE is silently lost — on exactly the big features that need it mo
416
442
  - **Every fallback names its cause.** The reason carried into
417
443
  `opus (cross-family QE DID NOT happen — …)` comes from a locked taxonomy —
418
444
  `timeout` (narrow the scope) · `no-verdict` · `tool-error` (fix the invocation) · `unusable-output` ·
419
- `unavailable` (fix the account/model) · `over-ceiling`. A timeout and an unusable output can never
420
- render the same string, because the operator's next move differs.
445
+ `unavailable` (fix the account/model) · `over-ceiling` · `scope-not-established` (mode A's scope
446
+ could not be built — see below) · `base-ref-not-established` (the scope's base-ref probe failed or
447
+ was unparseable — see below). A timeout and an unusable output can never render the same string,
448
+ because the operator's next move differs.
421
449
  - **The pipeline still never blocks on Codex.** Both modes fail into the same Claude belt as before.
422
450
 
451
+ **Mode A's `--uncommitted` pass runs in an ISOLATED scope-repo, never on the shared tree.** MEASURED
452
+ 2026-09-17 (run `wf_95211e0f`): on a hub with other dirty packages, `codex review --uncommitted`
453
+ wandered into unrelated files (`books/`, `features/clean-code-*`) and timed out at 600s having
454
+ reviewed nothing of the run's own feature — cross-family QE silently lost on exactly the runs that
455
+ need it most. Before mode A is attempted for scope `'uncommitted'` (the default), the pipeline now:
456
+ builds a throwaway git repo under `features/<slug>/.fa-state/review-scope/` containing ONLY the
457
+ ESTABLISHED change set (the same `modeBChanged` measurement mode B already uses — base versions via
458
+ `git show <BASE_REF>:<path>` in one commit, working versions copied on top); verifies the receipt
459
+ (`git status --porcelain` in that repo names EXACTLY the declared files, never "close enough"); and
460
+ runs `codex review --uncommitted` with `repo:` pointed at that isolated tree. Findings come back with
461
+ the scope-repo's own absolute path and are normalized to repo-relative before scoring. A change set
462
+ that is **not established** (`null` — no pre-code baseline) or **established but empty** (`[]` — no
463
+ files changed) refuses BEFORE any dispatch, under one decline kind `scope-not-established` (never a
464
+ silent fallback to `--uncommitted` on the shared tree); a failed scope-repo build or a receipt
465
+ mismatch refuse the same way. Knob: `args.qeIsolatedScope` (default `true`) — `false` restores the
466
+ prior `--uncommitted`-on-the-shared-tree behavior byte-for-byte, with a log line saying so. Scopes
467
+ `commit`/`base` are unaffected.
468
+
469
+ **Fix round 1 hardening (Codex r1 review, 2026-09-17), briefly:** the scope-repo path is validated by
470
+ PATH SEGMENT (`<repo>/features/<slug>/.fa-state/review-scope`, `<slug>` alphanumeric, no `.`/`..`
471
+ segment anywhere), not by a lexical prefix/substring check — a `..`-laced path can no longer walk the
472
+ one destructive `rm -rf` outside `features/`; the script itself repeats the check at runtime (symlink
473
+ + `case` guard) as a second belt. A base-ref probe that fails or returns something unparseable now
474
+ REFUSES under `base-ref-not-established` instead of silently substituting `HEAD` — a bad ref used to
475
+ make Codex review a full-file addition instead of the real modification. The receipt is now a CONTENT
476
+ check, not just a pathname list: the scope-build script also emits a `sha256sum` of every working
477
+ file, compared against the same pre-measured hashes mode B already computes, so a `cp` that silently
478
+ degraded to a deletion (an unreadable file, one that vanished mid-copy) is caught even though the
479
+ pathname-only receipt would have passed; a genuine `cp` failure now aborts the build rather than being
480
+ read as an intentional deletion. The porcelain receipt parser reads a fixed two-character status
481
+ column (`--no-renames` on the git status call, so a rename can never arrive as the ambiguous
482
+ `old -> new` shape). A declared path is never trimmed and rejects only what the receipt genuinely
483
+ cannot express (control characters, `"`, `\`, backtick, `$`, non-ASCII, a leading `-`) — spaces and
484
+ shell metacharacters are accepted, because every value already passes through the same safe quoting
485
+ function used everywhere else in this file.
486
+
487
+ **Named limit of the isolated scope (owner-facing, so it reads as a boundary, not a defect):** the
488
+ scope-repo review is BLIND to everything outside the declared change set, by construction — that is
489
+ the whole point (it is why the shared-tree run above timed out reviewing unrelated dirty packages).
490
+ A finding phrased as *"file X does not exist"* or *"the twin file is missing"* from mode A is therefore
491
+ an artifact of that intentional narrowness, not a real defect: the twin/sibling file is simply not
492
+ copied into the scope repo. Context that spans beyond the declared files is covered by mode B (which
493
+ is told exactly which files it may open) and by the separate Claude QE pass, never by mode A alone.
494
+
423
495
  After a Codex verdict a cheap Claude agent transcribes it into `08_qe_report.md` (mode A takes no
424
496
  prompt, so the reviewer cannot be asked to write anything). It is a scribe, not a second reviewer: the
425
497
  grade is Codex's and is stated as final.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dzhechkov/skills-feature-adr",
3
- "version": "1.5.12",
3
+ "version": "1.5.13",
4
4
  "description": "Adaptive Feature Development skill pack for Claude Code — 11-step pipeline with Complexity Router (S/M/L/XL), ADR-driven architecture, 15 agentic-qe skills, multi-agent fleet QE. Supports --full-qe, --full-qe-extended, --with-learning, and --knowledge-extractor modes.",
5
5
  "bin": {
6
6
  "skills-feature-adr": "./bin/cli.js"
package/sbom.json CHANGED
@@ -35,7 +35,7 @@
35
35
  "hashes": [
36
36
  {
37
37
  "alg": "SHA-256",
38
- "content": "e72e7ae2a5adc046598feb5de0bcac114d5aa961571f57d2aa530186c1dbb4b3"
38
+ "content": "619a231352a602c736ce6eec1e5fa79a056dc2f5f80f6b35eed236f9670f7337"
39
39
  }
40
40
  ]
41
41
  },
@@ -69,7 +69,7 @@
69
69
  },
70
70
  {
71
71
  "name": "dz:canonical-json-sha256-v2",
72
- "value": "8b0f75ec75e5f2a4e121c65a625abe6c9a8fb4b12732349293b78232348b2234"
72
+ "value": "07537ebbd424c668a37a42b670c033a4d3134a9e7e13803f426aa31149a85b58"
73
73
  }
74
74
  ]
75
75
  },
@@ -399,7 +399,7 @@
399
399
  "hashes": [
400
400
  {
401
401
  "alg": "SHA-256",
402
- "content": "7973676cc9a5a6d8429c0bcdbe163bb5dc16e2d2f28a72698e44ad72d234e199"
402
+ "content": "7ab9b7b1523289986a1f51b4cbbebf4410d96ad4a153f22fa8613e8a0f4f5fa7"
403
403
  }
404
404
  ]
405
405
  },
@@ -799,7 +799,7 @@
799
799
  "hashes": [
800
800
  {
801
801
  "alg": "SHA-256",
802
- "content": "8900ff6bc777a078aa4e37d8e28a2ffbc612187354995f726daf93e6f87d3109"
802
+ "content": "e96c5280ad21b604036cc912418c9627b0ff0ae8ca1436c58dcff951ba9a034b"
803
803
  }
804
804
  ]
805
805
  },
@@ -356,6 +356,41 @@ An empty (header-only) table is read as `hollow: true` — worse than no table a
356
356
  claims a ledger exists and says nothing. If there are no findings, omit the section entirely rather
357
357
  than writing an empty table.
358
358
 
359
+ ### 7.2 Ledger row for this QE step (mandatory)
360
+
361
+ Plain mode (this SKILL, not the ultracode `Workflow`) has no in-process `appendRunCostRow` — without
362
+ this step the QE instrument's own pass leaves ZERO trace in the run-cost ledger (aqe-ledger-row,
363
+ owner audit 17.09, decision #3).
364
+
365
+ **Resume guard — check this FIRST, before writing anything.** If QE was RESTORED from a checkpoint
366
+ for THIS run (`features/<slug>/.fa-state/checkpoints.jsonl` already contains a line for stage `qe`
367
+ that matches this run) — the row for this review was already written by the run that actually did
368
+ it. Do **NOT** run the command below in that case; instead write exactly this line into
369
+ `08_qe_report.md`: `qe ledger row skipped — stage resumed`. Writing the row anyway would double-pay
370
+ the same review in the ledger (aqe-ledger-row fix-round-1/#3, Codex r1 HIGH #3 — plain mode had no
371
+ resume guard at all before this).
372
+
373
+ Otherwise — QE ran fresh in this pass — after `08_qe_report.md` is written and the grade is final, run:
374
+
375
+ ```bash
376
+ dz feature-adr-record --kind ledger --stage qe --slug <slug> --row '<json>' --auto --json
377
+ ```
378
+
379
+ `<json>` carries the same fields the ultracode pipeline writes for this row: `reviewer` (the model
380
+ that reviewed, or `null` if unknown — never guessed), `reviewerFamily` (`claude`|`codex`|`null` when
381
+ the reviewer identity is not one of the two known families — never guessed as `claude`), `qeRole`
382
+ (`qe-code-reviewer` for a Claude reviewer; `codex-review`/`codex-exec` for a codex reviewer by scope
383
+ mode; `null` otherwise), `grade`, `gradeSource` (if known), `findings` (gap count), `findingsBySeverity`
384
+ (normalized `sev` -> count, unknown severities counted as `other`, never dropped), `findingsSource`
385
+ (`gaps` or `no-gaps-array`), `claimCheck` (if run), `crossFamily` (bool: reviewer family != coder
386
+ family — `null` when either family is unknown, since a same/cross-family verdict cannot be asserted
387
+ without both), `qeScope` (if the reviewer ran scoped, e.g. codex `mode`/`ref`/`files`). Insert the
388
+ writer's verdict (`{verdict:"written"|...}`) verbatim into `08_qe_report.md` so the record is
389
+ auditable from the artifact itself. NAMED LIMIT (layer 4 of the cost-of-detection ladder, said
390
+ honestly): this step is a skill instruction, not a code gate — nothing forces the agent to run it or
391
+ to check the resume guard first; only its PRESENCE in this module is deterministically pinned (grep
392
+ for the command above and for the resume-guard sentence).
393
+
359
394
  ### 8. QE Pattern Store (Direct Mode only)
360
395
 
361
396
  When `{AGENTIC_QE_MODE}` = `direct` | `direct-extended`, after the gap loop is closed and the verdict is set:
@@ -851,6 +851,48 @@ function captureFailureRecord(stage, mode, reason, detail) {
851
851
  return { stage: normalizedStage, mode: normalizedMode, reason: normalizedReason, detail: normalizedDetail }
852
852
  }
853
853
  function tpFamily(spec) { return /codex|gpt|openai/i.test(String(spec == null ? '' : spec)) ? 'codex' : 'claude' }
854
+ // aqe-ledger-row fix-round-1/#1 (Codex r1 HIGH #1): a STRICT closed-set classifier, deliberately
855
+ // separate from tpFamily() above. tpFamily() maps every unrecognized string — including undefined,
856
+ // null and typos — to 'claude', because its ONE existing job (the cross-model-QE routing default,
857
+ // evaluatorActual/chosenActual) wants a safe default direction. That default is WRONG for a LEDGER
858
+ // IDENTITY field: an unrecognized reviewer must read as unknown (null), never be silently attributed
859
+ // to Claude. knownFamily() is that closed set — only the values this workflow itself ever assigns to
860
+ // coderUsed/qeReviewerUsed ('claude', 'codex', 'codex-fallback') resolve; anything else is null.
861
+ function knownFamily(spec) {
862
+ return spec === 'claude' ? 'claude' : (spec === 'codex' || spec === 'codex-fallback') ? 'codex' : null
863
+ }
864
+ // aqe-ledger-row fix-round-1/#1: reviewerIdentity(qeReviewerUsed, modelUsed, qeScopeMode) — pure,
865
+ // never guesses. `family` comes ONLY from knownFamily() (never tpFamily(), per the owner decision).
866
+ // `qeRole` is 'qe-code-reviewer' ONLY when family is 'claude'; 'codex-review'/'codex-exec' ONLY when
867
+ // family is 'codex' AND qeScopeMode is the matching 'A'/'B'; every other combination (including an
868
+ // unknown family, or a codex family with an unrecognized/missing scope mode) is null — never guessed.
869
+ function reviewerIdentity(qeReviewerUsed, modelUsed, qeScopeMode) {
870
+ const family = knownFamily(qeReviewerUsed)
871
+ const qeRole = family === 'claude'
872
+ ? 'qe-code-reviewer'
873
+ : (family === 'codex' && qeScopeMode === 'A') ? 'codex-review'
874
+ : (family === 'codex' && qeScopeMode === 'B') ? 'codex-exec'
875
+ : null
876
+ // r2-N1 (Codex r2 HIGH, lead delta): a model label is an identity claim too — with the FAMILY
877
+ // unknown, the label cannot be attributed (`reviewer:'gpt-x'` under an unrecognized route would
878
+ // read as a Codex review that never provably happened). Null outside the known set, all three.
879
+ const reviewer = family !== null && typeof modelUsed === 'string' && modelUsed !== '' ? modelUsed : null
880
+ return { reviewer: reviewer, reviewerFamily: family, qeRole: qeRole }
881
+ }
882
+ // aqe-ledger-row T2/A2/NFR-4: pure helper — {findings, findingsBySeverity, findingsSource} from a QE
883
+ // gaps[] array. Not an array (missing/malformed) -> findings:0, findingsSource:'no-gaps-array', never a
884
+ // guess. sev is normalized (String -> trim -> lowercase); anything outside the closed set counts as
885
+ // 'other' rather than being dropped, so a finding is never silently lost from the total.
886
+ function qeFindingsSummary(gaps) {
887
+ const bySeverity = { blocker: 0, critical: 0, high: 0, medium: 0, low: 0, info: 0, other: 0 }
888
+ if (!Array.isArray(gaps)) return { findings: 0, findingsBySeverity: bySeverity, findingsSource: 'no-gaps-array' }
889
+ for (const g of gaps) {
890
+ const sev = String((g && g.sev !== undefined && g.sev !== null) ? g.sev : '').trim().toLowerCase()
891
+ if (Object.prototype.hasOwnProperty.call(bySeverity, sev)) bySeverity[sev] += 1
892
+ else bySeverity.other += 1
893
+ }
894
+ return { findings: gaps.length, findingsBySeverity: bySeverity, findingsSource: 'gaps' }
895
+ }
854
896
  function tpText(v) { if (typeof v === 'string') return v; if (v === null || v === undefined) return ''; try { const s = JSON.stringify(v); return typeof s === 'string' ? s : String(v) } catch (e) { return String(v) } }
855
897
  function tpBudget(raw) {
856
898
  try {
@@ -1045,7 +1087,10 @@ async function capturePairs(stage, phaseName, records, resumeGuardStage) {
1045
1087
  }
1046
1088
  }
1047
1089
 
1048
- async function appendRunCostRow(stage, phaseName, outcome) {
1090
+ async function appendRunCostRow(stage, phaseName, outcome, extra) {
1091
+ // aqe-ledger-row T1/NFR-1: `extra` is OPTIONAL and ADDITIVE-ONLY. Its keys are spread into
1092
+ // the row literal AFTER every existing field (after `runId:`, the last pre-existing field), so
1093
+ // a call site that passes no fourth argument produces the exact byte-identical row it always did.
1049
1094
  // Like capturePairs, a ledger failure is a logged SECONDARY event that can NEVER fail the run;
1050
1095
  // the whole body therefore rides one best-effort try/catch and never rethrows.
1051
1096
  try {
@@ -1093,6 +1138,8 @@ async function appendRunCostRow(stage, phaseName, outcome) {
1093
1138
  // skips that guess entirely (run-records.ts / cli.ts: runIdForLookup is only computed when
1094
1139
  // the payload carries no runId), so passing it here is a strict reliability improvement.
1095
1140
  runId: (typeof RUN_ID === 'string' && RUN_ID !== '') ? RUN_ID : null,
1141
+ // aqe-ledger-row T1/NFR-1: additive-only — spreads nothing when `extra` is absent.
1142
+ ...(extra && typeof extra === 'object' ? extra : {}),
1096
1143
  })
1097
1144
  // WITNESSED WRITE (ADR-001): the subagent RUNS a command with data arguments; it is no longer
1098
1145
  // handed a shell pipeline with the row baked in. The command refuses a malformed row, stamps the
@@ -1157,6 +1204,11 @@ const CODEX_HINT = ' (If you are the Codex runtime, prefer the ' + CODEX_MODEL +
1157
1204
  // and --base HEAD review the identical tree while --uncommitted needs no ref at all.
1158
1205
  const QE_SCOPE = (A.qeScope === 'commit' || A.qeScope === 'base') ? A.qeScope : 'uncommitted'
1159
1206
  const QE_SCOPE_REF = (typeof A.qeScopeRef === 'string') ? A.qeScopeRef : ''
1207
+ // codex-review-scope (FR-5, A5): mode A's 'uncommitted' pass runs in an ISOLATED scope-repo by
1208
+ // default (MEASURED wf_95211e0f, 2026-09-17: --uncommitted on the shared tree wandered into other
1209
+ // dirty packages and timed out at 600s having reviewed nothing of the run's own feature).
1210
+ // args.qeIsolatedScope:false restores the prior --uncommitted-on-REPO behavior byte-for-byte.
1211
+ const QE_ISOLATED_SCOPE = A.qeIsolatedScope !== false
1160
1212
  // Mode-B questions: the ones --commit structurally forbids us from asking.
1161
1213
  const SCOPED_QE_QUESTIONS = [
1162
1214
  'Is the change correct — name any real defect with file and line, or say there is none.',
@@ -1982,7 +2034,16 @@ const CODEX_REVIEW_TIMEOUT_SECONDS = 600
1982
2034
  const CODEX_REVIEW_DEFAULT_EFFORT = 'high'
1983
2035
  const CODEX_TIMEOUT = 'CODEX_TIMEOUT'
1984
2036
  const CODEX_QE_SIGNAL_PREFIX = 'CODEX-QE-SIGNAL'
1985
- const CODEX_QE_DECLINE_KINDS = ['timeout', 'no-verdict', 'tool-error', 'unusable-output', 'unavailable', 'over-ceiling', 'wrong-tree']
2037
+ // 'scope-not-established' joined the set 2026-09-17 (feature codex-review-scope, FR-2/A2): mode A's
2038
+ // isolated scope-repo is built ONLY from an ESTABLISHED change set (modeBChanged an array >= 1
2039
+ // entry) — null (not established) or an established-but-empty set both refuse under this ONE kind,
2040
+ // distinguished by the reason TEXT (never a second kind), so the taxonomy grows by exactly one.
2041
+ // 'base-ref-not-established' joined the set 2026-09-17 (fix-round-1 #3, Codex r1 CRITICAL/HIGH): a
2042
+ // base-ref probe that fails or returns something unparseable used to fall back to 'HEAD' SILENTLY —
2043
+ // indistinguishable from a healthy default, and codex would then review a full-file addition
2044
+ // instead of the real modification. This is a DIFFERENT kind from 'scope-not-established' (not just
2045
+ // different reason text) because the failure is about the REF, not the change set.
2046
+ const CODEX_QE_DECLINE_KINDS = ['timeout', 'no-verdict', 'tool-error', 'unusable-output', 'unavailable', 'over-ceiling', 'wrong-tree', 'scope-not-established', 'base-ref-not-established']
1986
2047
  const SCOPED_QE_MAX_FILES = 3
1987
2048
  const SCOPED_QE_MAX_QUESTIONS = 4
1988
2049
  const SCOPED_QE_MAX_PATH_CHARS = 200
@@ -2042,6 +2103,172 @@ function codexReviewCommand(input) {
2042
2103
  return { cmd: cmd, carriesPrompt: false, scope: scope, reason: null }
2043
2104
  }
2044
2105
 
2106
+ // ── ISOLATED REVIEW SCOPE (feature codex-review-scope) ──────────────────────────────────────────
2107
+ // MEASURED 2026-09-17 (wf_95211e0f, 07:14-08:22): mode A's codex review --uncommitted on the
2108
+ // shared hub tree wandered into OTHER dirty packages (books/, features/clean-code-*) and timed out
2109
+ // at 600s having reviewed nothing of this run's own feature — the same class as teach:c551a2c3
2110
+ // (12.09: 816KB of log, no verdict). The fix is not a bigger timeout (that buys more reconnaissance,
2111
+ // per the qe-scoped-review ADR above); it is giving mode A a TREE that contains ONLY this run's
2112
+ // established changes.
2113
+ //
2114
+ // A live probe (14:25, codex-cli 0.154.0, gpt-5.6-terra medium) confirmed the mechanics: a throwaway
2115
+ // git repo with one committed base file plus one working-tree edit, codex review --uncommitted
2116
+ // exits 0 in 16.7s with a real finding whose location is the scope repo's ABSOLUTE path — hence
2117
+ // normalizeScopedFindings below.
2118
+ const SCOPE_REPO_GIT_EMAIL = 'dz-review-scope@localhost'
2119
+ const SCOPE_REPO_GIT_NAME = 'dz-review-scope'
2120
+
2121
+ // T1 (FR-3, A3, NFR-2). Pure builder: {repo, scopeDir, baseRef, files, quote} -> {script, reason}.
2122
+ // script is ONE shell command line (chained with && ; each per-file step is parenthesised so a
2123
+ // missing base file or a working-tree deletion never aborts the rest of the build); null + reason
2124
+ // on any refusal. Never invoked here — this file has no child_process; the caller dispatches script
2125
+ // through a shell agent (effort low), exactly like every other shell step in this pipeline.
2126
+ //
2127
+ // The emitted rm -rf is the ONLY destructive line this builder can produce, and it fires ONLY when
2128
+ // scopeDir is provably a child of <repo>/features/<slug>/.fa-state/review-scope — never a
2129
+ // caller-supplied path taken on faith (NFR-2).
2130
+ //
2131
+ // fix-round-1 #1 (BLOCKER, Codex r1): the old containment check was LEXICAL — scopeDir merely had
2132
+ // to START WITH '<repo>/features/' and CONTAIN '/.fa-state/review-scope' anywhere after that, so
2133
+ // '<repo>/features/../victim/.fa-state/review-scope' passed both tests while '..' walks the rm -rf
2134
+ // straight out of features/. The check below is SEGMENT-based (split('/'), never substring/indexOf
2135
+ // on the whole string): scopeDir must equal repo's own segments, followed by EXACTLY the four
2136
+ // segments ['features', <slug>, '.fa-state', 'review-scope'] with <slug> matching SCOPE_SLUG_RE —
2137
+ // and every segment anywhere in scopeDir is rejected if it is '', '.' or '..'.
2138
+ const SCOPE_SLUG_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,39}$/
2139
+ function reviewScopeRepoScript(input) {
2140
+ const o = input || {}
2141
+ const quote = o.quote
2142
+ if (typeof quote !== 'function') return { script: null, reason: 'no quote function supplied' }
2143
+ const repo = String(o.repo === undefined || o.repo === null ? '' : o.repo)
2144
+ const scopeDir = String(o.scopeDir === undefined || o.scopeDir === null ? '' : o.scopeDir)
2145
+ const baseRef = String(o.baseRef === undefined || o.baseRef === null ? '' : o.baseRef)
2146
+ if (repo === '' || scopeDir === '') return { script: null, reason: 'empty repo or scopeDir' }
2147
+ if (!isSafeCodexRef(baseRef)) return { script: null, reason: 'unsafe baseRef' }
2148
+ const repoSegs = repo.split('/')
2149
+ const dirSegs = scopeDir.split('/')
2150
+ for (const seg of dirSegs) {
2151
+ if (seg === '.' || seg === '..') return { script: null, reason: 'scopeDir contains an unsafe "." or ".." path segment' }
2152
+ }
2153
+ let repoMatches = dirSegs.length >= repoSegs.length
2154
+ if (repoMatches) for (let i = 0; i < repoSegs.length; i++) { if (dirSegs[i] !== repoSegs[i]) { repoMatches = false; break } }
2155
+ const rest = repoMatches ? dirSegs.slice(repoSegs.length) : null
2156
+ const slug = rest && rest.length === 4 ? rest[1] : ''
2157
+ const shapeOk = !!rest && rest.length === 4 && rest[0] === 'features' && rest[2] === '.fa-state' && rest[3] === 'review-scope' && SCOPE_SLUG_RE.test(slug)
2158
+ if (!shapeOk) {
2159
+ return { script: null, reason: 'scopeDir must be exactly ' + repo + '/features/<slug>/.fa-state/review-scope with a safe <slug>' }
2160
+ }
2161
+ const rawFiles = Array.isArray(o.files) ? o.files : []
2162
+ const files = []
2163
+ const seen = new Set()
2164
+ for (const raw of rawFiles) {
2165
+ // fix-round-1 #6 (MEDIUM, Codex r1): NEVER trim a declared path — trimming 'src/a.ts ' silently
2166
+ // retargets the shell command at 'src/a.ts', a DIFFERENT file than the one declared, and the
2167
+ // subsequent receipt mismatch then reads as a build failure rather than the truncation that
2168
+ // caused it. Reject only what the TRANSPORT genuinely cannot express: every character below is
2169
+ // safely embeddable through the injected 'quote' (single-quoting round-trips ANY byte, including
2170
+ // ';&|<>*?()[]{}!' and even a literal "'"), so the real constraint is the FR-3 receipt —
2171
+ // 'git status --porcelain' QUOTES a path containing a control character, '"', '\' or (by default)
2172
+ // non-ASCII, and our receipt parser reads that raw line without un-quoting it. A leading '-' is
2173
+ // refused so no downstream tool ever reads the path as a flag. Spaces are explicitly ALLOWED.
2174
+ const f = String(raw === undefined || raw === null ? '' : raw)
2175
+ if (f === '') continue
2176
+ if (seen.has(f)) continue
2177
+ if (f.charAt(0) === '/') return { script: null, reason: 'absolute path ' + JSON.stringify(f) }
2178
+ if (f === '..' || f.indexOf('../') === 0 || f.indexOf('/../') !== -1 || f.slice(-3) === '/..') {
2179
+ return { script: null, reason: 'path traversal ' + JSON.stringify(f) }
2180
+ }
2181
+ if (f.charAt(0) === '-') return { script: null, reason: 'path may not start with "-" ' + JSON.stringify(f) }
2182
+ if (f.charAt(f.length - 1) === '/') return { script: null, reason: 'path must not end with "/" ' + JSON.stringify(f) }
2183
+ if (/[^\x20-\x7e]/.test(f) || /["\\\x60$]/.test(f)) {
2184
+ return { script: null, reason: 'unsafe path (control/quote/backtick/$/non-ASCII character) ' + JSON.stringify(f) }
2185
+ }
2186
+ seen.add(f)
2187
+ files.push(f)
2188
+ }
2189
+ if (files.length === 0) return { script: null, reason: 'empty file list' }
2190
+ const parts = []
2191
+ // fix-round-1 #3 (HIGH, Codex r1) — script half: verify baseRef resolves to a real commit as the
2192
+ // FIRST command. Every later "git show <baseRef>:<path>" failure used to be swallowed identically
2193
+ // whether the FILE was absent at a valid commit (expected — a new file) or the REF itself was
2194
+ // bogus (a silent full-tree diff against nothing) — indistinguishable from outside. A bad ref now
2195
+ // aborts loudly (exit 4) before any per-file step runs.
2196
+ parts.push('git -C ' + quote(repo) + ' rev-parse --verify --quiet ' + quote(baseRef + '^{commit}') + ' > /dev/null || exit 4')
2197
+ // fix-round-1 #1 — runtime belt matching the JS-side segment check above: a symlink at scopeDir
2198
+ // (planted between the JS check and this script's execution) is refused rather than followed by
2199
+ // rm -rf, and a 'case' re-asserts the very prefix the JS validator just proved, so the two checks
2200
+ // can never silently drift apart.
2201
+ // Lead delta after the second manual e2e (2026-09-17 15:35): the form '[ -L dir ] && exit 3' inside a
2202
+ // '&&'-joined chain ABORTS the chain whenever dir is NOT a symlink (the test returns 1), so the
2203
+ // fix-round script exited 1 having built nothing — a dead feature that fails closed. A statement
2204
+ // form keeps the belt and lets the chain continue.
2205
+ parts.push('if [ -L ' + quote(scopeDir) + ' ]; then exit 3; fi')
2206
+ parts.push('case ' + quote(scopeDir) + ' in ' + quote(repo) + '/features/*/.fa-state/review-scope) : ;; *) exit 3 ;; esac')
2207
+ parts.push('rm -rf ' + quote(scopeDir))
2208
+ parts.push('mkdir -p ' + quote(scopeDir))
2209
+ parts.push('git -C ' + quote(scopeDir) + ' init -q')
2210
+ parts.push('git -C ' + quote(scopeDir) + ' config user.email ' + quote(SCOPE_REPO_GIT_EMAIL))
2211
+ parts.push('git -C ' + quote(scopeDir) + ' config user.name ' + quote(SCOPE_REPO_GIT_NAME))
2212
+ for (const f of files) {
2213
+ const dest = scopeDir + '/' + f
2214
+ const slash = dest.lastIndexOf('/')
2215
+ const destDir = dest.slice(0, slash)
2216
+ if (destDir !== scopeDir) parts.push('mkdir -p ' + quote(destDir))
2217
+ // fix-round-1 #3 (HIGH) — per file: distinguish "path absent at a valid commit" (expected for a
2218
+ // new file — degrade to rm -f) from every OTHER git-show failure (permissions, corrupt object,
2219
+ // …) which now ABORTS the build (exit 5) instead of silently degrading the same way. cat-file -e
2220
+ // is the existence check git itself uses; git show is only reached once existence is confirmed.
2221
+ // Lead delta (Codex r2 N1 HIGH): 'cat-file -e' folds "absent at the base" and "object error"
2222
+ // into one non-zero — a tree entry whose blob is missing/corrupt was silently turned into a
2223
+ // full-file ADDITION. Three-way probe instead: ls-tree FAILS → exit 7 (build failure, named);
2224
+ // empty listing → genuinely absent at the base → no base copy; non-empty → 'show' is MANDATORY
2225
+ // and its failure aborts (exit 5).
2226
+ parts.push('if ! lt=$(git -C ' + quote(repo) + ' ls-tree --name-only ' + quote(baseRef) + ' -- ' + quote(f) + '); then exit 7; fi; if [ -n "$lt" ]; then git -C ' + quote(repo) + ' show ' + quote(baseRef + ':' + f) + ' > ' + quote(dest) + ' || exit 5; else rm -f ' + quote(dest) + '; fi')
2227
+ }
2228
+ parts.push('git -C ' + quote(scopeDir) + ' add -A && git -C ' + quote(scopeDir) + ' commit -q --allow-empty -m base')
2229
+ for (const f of files) {
2230
+ const src = repo + '/' + f
2231
+ const dest = scopeDir + '/' + f
2232
+ // fix-round-1 #4 (HIGH, Codex r1): the old '[ -f src ] && cp src dest || rm -f dest' ran the
2233
+ // REMOVAL whenever 'cp' itself failed (permissions, disk full, the file vanishing mid-copy) —
2234
+ // indistinguishable from a genuine working-tree deletion, and the FR-3 pathname-only receipt
2235
+ // then accepted the wrong outcome as a pass. A real 'cp' failure now aborts the build (exit 6);
2236
+ // only a MISSING source file (a real deletion) degrades to rm -f.
2237
+ parts.push('if [ -f ' + quote(src) + ' ]; then cp ' + quote(src) + ' ' + quote(dest) + ' || exit 6; else rm -f ' + quote(dest) + '; fi')
2238
+ }
2239
+ // Lead delta (manual e2e 2026-09-17 14:51, BEFORE Codex r1): without --untracked-files=all a NEW
2240
+ // file of the change set is reported as its collapsed parent directory ('?? packages/x/'), the
2241
+ // receipt set never equals the declared set, and mode A is refused as scope-build-failed for
2242
+ // every feature that ADDS a file. Per-file listing makes the receipt compare paths with paths.
2243
+ // fix-round-1 #5 (HIGH, Codex r1): --no-renames forces git to report a rename as a plain
2244
+ // delete+add instead of the two-path 'R old -> new' line, which a fixed-width slice(3) parser
2245
+ // (the receipt side, below) cannot split back into two paths.
2246
+ parts.push('git -C ' + quote(scopeDir) + ' status --porcelain --untracked-files=all --no-renames')
2247
+ // fix-round-1 #4 (HIGH) — content receipt: a sha256sum line per declared working file, in the same
2248
+ // "<hash> <path>" shape changeSetProbeCmd's uncommitted probe already emits, so the workflow can
2249
+ // compare it against the ALREADY-MEASURED afterSnap for the same files. A cp that silently
2250
+ // degraded to rm -f shows up here as a hash MISMATCH even when the pathname-only porcelain receipt
2251
+ // above would have passed (a deletion and a failed-copy-then-deletion look identical by name alone).
2252
+ parts.push('( cd ' + quote(scopeDir) + ' && sha256sum -- ' + files.map(quote).join(' ') + ' 2>/dev/null || true )')
2253
+ return { script: parts.join(' && '), reason: null }
2254
+ }
2255
+
2256
+ // T3 (FR-4, A4). Mode-A findings from a scoped review carry the scope-repo's ABSOLUTE path (MEASURED
2257
+ // live probe above). Strip it back to repo-relative so partitionReviewFindings can match it against
2258
+ // modeBChanged. A finding whose location does not start with scopeDir is left UNTOUCHED — never
2259
+ // guessed as belonging to the scope.
2260
+ function normalizeScopedFindings(findings, scopeDir) {
2261
+ const list = Array.isArray(findings) ? findings : []
2262
+ const dir = String(scopeDir === undefined || scopeDir === null ? '' : scopeDir)
2263
+ if (dir === '') return list
2264
+ const prefix = dir.charAt(dir.length - 1) === '/' ? dir : dir + '/'
2265
+ return list.map(function (f) {
2266
+ const loc = (f && f.location) ? String(f.location) : ''
2267
+ if (loc.indexOf(prefix) !== 0) return f
2268
+ return { severity: f.severity, title: f.title, location: loc.slice(prefix.length) }
2269
+ })
2270
+ }
2271
+
2045
2272
  // Mode B. The "do NOT open any other file" clause is LOAD-BEARING TEXT — it is the difference
2046
2273
  // between the 41s graded run and the 280s ungraded one. An empty file list returns '' so an UNSCOPED
2047
2274
  // mode-B dispatch is not constructible at all.
@@ -2226,6 +2453,8 @@ function codexQeDeclineReason(kind, detail) {
2226
2453
  // tool-error right above rendered it — the asymmetry that made the field report unfixable blind.
2227
2454
  if (canonical === 'unavailable') return 'codex not used — ' + ((d.reason === undefined || d.reason === null || String(d.reason) === '') ? 'codex exec reported it could not run' : String(d.reason)) + (extra === 'no detail' ? '' : ' (' + extra + ')')
2228
2455
  if (canonical === 'over-ceiling') return 'prompt is ' + chars + ' chars / unscoped — refused before dispatch'
2456
+ if (canonical === 'scope-not-established') return 'review scope NOT ESTABLISHED for uncommitted QE — ' + ((d.reason === undefined || d.reason === null || String(d.reason) === '') ? 'the change set could not be measured' : String(d.reason)) + ' — refusing to fall back to --uncommitted on the shared tree'
2457
+ if (canonical === 'base-ref-not-established') return 'base ref for the isolated review scope NOT ESTABLISHED — ' + ((d.reason === undefined || d.reason === null || String(d.reason) === '') ? 'the base-ref probe failed or was unparseable' : String(d.reason)) + ' — refusing to silently fall back to HEAD'
2229
2458
  throw new Error('codexQeDeclineReason: unknown kind ' + k)
2230
2459
  }
2231
2460
 
@@ -2455,11 +2684,18 @@ function codexQeSignalCommand(inner, outPath) {
2455
2684
  // Shared tail of both dispatch modes: run the signal-wrapped command through a shell agent and
2456
2685
  // CLASSIFY what came back. signalExpected is true here — on the pipeline path a swallowed sentinel
2457
2686
  // means the command did not demonstrably run, which is a tool-error, never a pass.
2458
- async function runCodexQeCommand(stage, cmd, phaseName, label, probed, mode, scopeRef, files, allowStatedGrade, requestedReasoning, rung, announcementOpts) {
2687
+ async function runCodexQeCommand(stage, cmd, phaseName, label, probed, mode, scopeRef, files, allowStatedGrade, requestedReasoning, rung, announcementOpts, scopeDir) {
2459
2688
  const wrapped = 'Run EXACTLY this via Bash and reply with its stdout VERBATIM and nothing else, INCLUDING the final ' + CODEX_QE_SIGNAL_PREFIX + ' line (it is a machine signal, not prose — do not summarise, reformat or omit it). Only if you cannot run the command AT ALL (no shell, command not found) reply with exactly ' + CODEX_UNAVAILABLE + '; a timeout is NOT that case, it reports itself in the signal line.\n\n' + codexQeSignalCommand(cmd, '/tmp/dz-codex-qe-' + SLUG + '-' + stage + '-' + mode + '.out')
2460
2689
  const raw = await dispatchAgent(rung, wrapped, { label: stageLabel(label, { agentType: 'codex:codex-rescue', codexModel: probed, _reasoning: requestedReasoning || 'high' }), phase: phaseName, model: 'haiku', effort: 'low' }, announcementOpts)
2461
2690
  const sig = parseCodexReviewSignal(raw === null ? '' : String(raw))
2462
- const findings = parseCodexReviewFindings(sig.body)
2691
+ // fix-round-1 #2 (CRITICAL, Codex r1): normalize a scoped review's findings to repo-relative
2692
+ // IMMEDIATELY after parsing — before classifyCodexQeOutcome or gradeFromReviewFindings ever see
2693
+ // them, and before the caller does anything else with the return value. declaredFiles ('files')
2694
+ // are already repo-relative ('_scopeFiles'); leaving the PARSED findings on the scope-repo's
2695
+ // ABSOLUTE path for even one extra hop is the class of bug this closes — every consumer downstream
2696
+ // of this function now sees only repo-relative locations, never a mix of the two shapes.
2697
+ const rawFindings = parseCodexReviewFindings(sig.body)
2698
+ const findings = scopeDir ? normalizeScopedFindings(rawFindings, scopeDir) : rawFindings
2463
2699
  // Mode A NEVER asked for a letter (every scope flag rejects a prompt), so any "Grade: X" in its
2464
2700
  // output came from the CODE UNDER REVIEW, not from the reviewer. FOUND BY THE FIRST LIVE MODE-A RUN
2465
2701
  // (2026-08-21): this feature's own README and CHANGELOG quote "Grade: B", and the review of that
@@ -2480,19 +2716,42 @@ async function runCodexQeCommand(stage, cmd, phaseName, label, probed, mode, sco
2480
2716
  // and exit 124 without one). It cannot carry our questions: every scope flag refuses [PROMPT].
2481
2717
  async function codexReviewAgent(stage, scope, scopeRef, phaseName, requestedOpts, rung) {
2482
2718
  lastCodexDecline = null
2719
+ // codex-review-scope (A1/A2): the scope DECISION travels on requestedOpts as PRIVATE fields
2720
+ // (_scopeBlocked / _scopeRepo / _scopeFiles) rather than a new positional parameter —
2721
+ // codexReviewAgent's signature and its call site are BOTH pinned byte-for-byte
2722
+ // (cross-family-qe.test.ts, feature-adr-model-routing.test.ts), so this is the one channel that
2723
+ // extends behavior without touching either pin. A blocked scope refuses BEFORE any probe is spent —
2724
+ // mode A never dispatches against the shared tree for an unestablished or empty change set.
2725
+ if (requestedOpts && requestedOpts._scopeBlocked) {
2726
+ // fix-round-1 #3 (HIGH, Codex r1): a blocked scope carries its own taxonomy KIND when the
2727
+ // decision knows a more specific one (e.g. 'base-ref-not-established') — defaulting to
2728
+ // 'scope-not-established' keeps every EXISTING caller (none of which set _scopeBlockedKind)
2729
+ // byte-identical.
2730
+ const blockedKind = (requestedOpts._scopeBlockedKind && CODEX_QE_DECLINE_KINDS.indexOf(requestedOpts._scopeBlockedKind) !== -1) ? requestedOpts._scopeBlockedKind : 'scope-not-established'
2731
+ settleUndispatchedStage(requestedOpts, rung, 'refused-before-dispatch', blockedKind)
2732
+ return noteCodexDecline(stage, blockedKind, { reason: requestedOpts._scopeBlocked })
2733
+ }
2483
2734
  const requestedId = requestedOpts && requestedOpts.codexModel !== 'auto' ? requestedOpts.codexModel : null
2484
2735
  const requestedReasoning = (requestedOpts && requestedOpts._reasoning) || 'high'
2485
2736
  const probed = await probeCodexId(requestedId)
2486
2737
  if (!probed) { settleUndispatchedStage(requestedOpts, rung, 'probe-failed'); return noteCodexDecline(stage, 'unavailable', { reason: 'no codex model id answered the probe' }) }
2487
2738
  const announcementOpts = mergeOpts(requestedOpts || {}, { codexModel: probed, _stage: stage })
2488
- const built = codexReviewCommand({ scope: scope, ref: scopeRef, modelId: probed, reasoning: requestedReasoning, timeoutSeconds: CODEX_REVIEW_TIMEOUT_SECONDS, timeoutBin: await probeTimeoutBin(), repo: REPO })
2739
+ // A1: for scope 'uncommitted' this is REPO only when isolation was explicitly disabled
2740
+ // (args.qeIsolatedScope:false) or never applies (commit/base) — never a silent fallback.
2741
+ const scopeRepo = (requestedOpts && typeof requestedOpts._scopeRepo === 'string' && requestedOpts._scopeRepo !== '') ? requestedOpts._scopeRepo : REPO
2742
+ const declaredFiles = (requestedOpts && Array.isArray(requestedOpts._scopeFiles)) ? requestedOpts._scopeFiles : []
2743
+ const built = codexReviewCommand({ scope: scope, ref: scopeRef, modelId: probed, reasoning: requestedReasoning, timeoutSeconds: CODEX_REVIEW_TIMEOUT_SECONDS, timeoutBin: await probeTimeoutBin(), repo: scopeRepo })
2489
2744
  // R18: an id ANSWERED and this rung still dispatches nothing (a missing or unsafe scope ref).
2490
2745
  // The outcome is recorded as a REFUSAL so the belt below cannot report it as a rung that ran.
2491
2746
  if (built.cmd === null) { settleUndispatchedStage(announcementOpts, rung, 'refused-before-dispatch', built.reason); return noteCodexDecline(stage, 'tool-error', { exit: 2, detail: built.reason }) }
2492
2747
  // R4-F2a: AFTER the command exists. An unusable scope ref (commit/base with a missing or unsafe
2493
2748
  // ref) returns cmd:null and dispatches NOTHING — announcing above printed a line for a review that
2494
2749
  // never ran. The reason travels from requestedOpts so it is not silently defaulted either.
2495
- return await runCodexQeCommand(stage, built.cmd, phaseName, stage + ':codex-review', probed, 'A', built.scope + (scopeRef ? ' ' + scopeRef : ''), [], false, requestedReasoning, rung, announcementOpts)
2750
+ const result = await runCodexQeCommand(stage, built.cmd, phaseName, stage + ':codex-review', probed, 'A', built.scope + (scopeRef ? ' ' + scopeRef : ''), declaredFiles, false, requestedReasoning, rung, announcementOpts, scopeRepo !== REPO ? scopeRepo : undefined)
2751
+ // T3 (FR-4, A4): a scoped review's findings carry the scope-repo's ABSOLUTE path — normalize back
2752
+ // to repo-relative so partitionReviewFindings can match them against the declared change set.
2753
+ if (result && scopeRepo !== REPO) return mergeOpts(result, { findings: normalizeScopedFindings(result.findings, scopeRepo) })
2754
+ return result
2496
2755
  }
2497
2756
 
2498
2757
  // MODE B — the narrowed follow-up. Carries OUR questions over files we name, and is refused outright
@@ -3679,6 +3938,7 @@ let coderUsed = null
3679
3938
  let qe = null
3680
3939
  let pipelineRound = null
3681
3940
  let roundClosed = false
3941
+ let roundSkippedReason = null
3682
3942
  const LEARNED = router ? router.rationale : 'none recalled'
3683
3943
  const isMplus = tier === 'M' || tier === 'L' || tier === 'XL'
3684
3944
  const isLplus = tier === 'L' || tier === 'XL'
@@ -4774,6 +5034,31 @@ if (resumedStages.indexOf('code') !== -1 && landedNote !== '') {
4774
5034
  landedNote = '\n\n[RESUMED from checkpoint — the landing barrier below ran in the ORIGINAL run; the change-manifest artifact was re-verified present by the resume probe]' + landedNote
4775
5035
  }
4776
5036
 
5037
+ // aqe-ledger-row T4/FR-2/FR-3/A5: the `impl` row, written right here — AFTER the Step 7.5
5038
+ // landing barrier so its outcome/coder/landed fields are known, and BEFORE Step 8 QE runs —
5039
+ // so the `qe` row's minutesSincePrev measures the QE step itself, not QE+code combined
5040
+ // (00_complexity_assessment.md: today the row before `qe` is `plan`, which would fold code time in).
5041
+ // fix-round-1/#2 (Codex r1 HIGH #2): the OLD `landingStatus !== null ? landingStatus :
5042
+ // 'skipped-claude-sync'` guessed 'skipped-claude-sync' for ANY null landingStatus — including a
5043
+ // codex coder whose barrier composite never built (codeStage null). Now: a landingStatus from the
5044
+ // KNOWN barrier-verdict set (the same set codeStageResultShapeValid checks — landed /
5045
+ // genuinely-not-landed / inconclusive / 'synchronous', the last reachable only when the barrier was
5046
+ // never NEEDED, i.e. a Claude coder) is used AS-IS; 'skipped-claude-sync' is used ONLY when the
5047
+ // coder's family is genuinely 'claude' AND the barrier was never needed for it
5048
+ // (!needsCodeLandedBarrier); every other case (an unrecognized landingStatus paired with a codex — or
5049
+ // unknown — coder) is the honestly-named 'no-landing-status', which is ALWAYS a string, so — unlike
5050
+ // `undefined` — the field can never silently vanish from the JSON.stringify'd row.
5051
+ if (resumedStages.indexOf('code') === -1) {
5052
+ const knownLandingVerdicts = ['landed', 'genuinely-not-landed', 'inconclusive', 'synchronous']
5053
+ const coderFamilyForImpl = knownFamily(coderUsed)
5054
+ const implOutcome = knownLandingVerdicts.indexOf(landingStatus) !== -1
5055
+ ? landingStatus
5056
+ : (coderFamilyForImpl === 'claude' && !needsCodeLandedBarrier(coderUsed)) ? 'skipped-claude-sync' : 'no-landing-status'
5057
+ await appendRunCostRow('impl', 'Code', implOutcome, { coder: coderUsed, coderFamily: coderFamilyForImpl, landed: implOutcome })
5058
+ } else {
5059
+ log('run-cost ledger: impl row skipped — stage resumed')
5060
+ }
5061
+
4777
5062
  // Step 8: QE (brutal-honesty, agentic-qe) + MANDATORY teach
4778
5063
  phase('QE')
4779
5064
  await recordRegistryEvent('heartbeat', 'QE')
@@ -4861,7 +5146,11 @@ const qe2Spec = qePrecisionPassSpec(PRIMARY, BUDGET_MODE, tier)
4861
5146
  // re-teach (the original run already stored its lessons — replaying teach would double-store).
4862
5147
  // R6: the review SCOPE is part of what a QE verdict is about, so it enters the hash — a resume must
4863
5148
  // not present a verdict obtained over one scope as if it had been obtained over another.
4864
- const qeHash = ckptHash('qe', [fnv1a64(JSON.stringify(codeStage === undefined ? null : codeStage)), tier, DESC, QE_REVIEWER, MODELS.qe === undefined ? null : MODELS.qe, CODEX_MODEL, coderUsed, PRIMARY, BUDGET_MODE, qe2Spec, POLY.hasManifest, fnv1a64(String(POLY.report || '')), usageOverride, QE_SCOPE, QE_SCOPE_REF, confirmationFileGate])
5149
+ // L1 (fix-round-1, lead e2e 14:53): QE_ISOLATED_SCOPE must enter the hash — without it, flipping
5150
+ // args.qeIsolatedScope between runs of the SAME slug resumes a checkpointed verdict that was
5151
+ // obtained under the OTHER knob value (an --uncommitted shared-tree review standing in for an
5152
+ // isolated-scope one, or vice versa) instead of re-QEing under the new setting.
5153
+ const qeHash = ckptHash('qe', [fnv1a64(JSON.stringify(codeStage === undefined ? null : codeStage)), tier, DESC, QE_REVIEWER, MODELS.qe === undefined ? null : MODELS.qe, CODEX_MODEL, coderUsed, PRIMARY, BUDGET_MODE, qe2Spec, POLY.hasManifest, fnv1a64(String(POLY.report || '')), usageOverride, QE_SCOPE, QE_SCOPE_REF, confirmationFileGate, QE_ISOLATED_SCOPE])
4865
5154
  let crossFamilyQeReport = null
4866
5155
  const qeStage = await withCheckpoint('qe', 'QE', qeHash, async () => {
4867
5156
  let qe = null
@@ -4895,16 +5184,7 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
4895
5184
  // assert the very property that is being lost. The honest label for a rung after a dead rung.
4896
5185
  let qeCodexReason = stageReason('qe', qeDecision)
4897
5186
  if (!qeIsCodex) qeCodexReason = 'fallback-rung'
4898
- const qeCodexLabelOpts = mergeOpts(qeModel.agentType ? qeModel : specToOpts('codex:' + CODEX_MODEL + ':high'), { _stage: 'qe', _reason: qeCodexReason })
4899
- // R5-1: the id modelsUsed will report, resolved by the SAME memoized probe the dispatch uses.
4900
- const qeCodexResolved = await codexLabelOptsForDispatch(qeCodexLabelOpts)
4901
- qeCodexProbeFailed = !!qeCodexResolved._codexProbeFailed
4902
- const qeCodexDispatchLabel = modelLabel(qeCodexResolved)
4903
- // R16-1: the codex QE rung records its attempt HERE. The only dispatch-time write used to sit
4904
- // inside `if (!qeIsCodex)`, so a codex-routed QE whose modes and belt all returned null printed
4905
- // `▸ qe` lines and left modelsUsed.qe unassigned entirely. The marker is provisional: the success
4906
- // paths below overwrite it, and if nothing succeeds it stays and says so.
4907
- modelsUsed.qe = qeCodexDispatchLabel + (qeCodexProbeFailed ? ' (codex probe found no usable id)' : ' (no verdict)')
5187
+ let qeCodexLabelOpts = mergeOpts(qeModel.agentType ? qeModel : specToOpts('codex:' + CODEX_MODEL + ':high'), { _stage: 'qe', _reason: qeCodexReason })
4908
5188
  // ADR-001: QE's deliverable is its RETURN VALUE, so it dispatches SYNCHRONOUSLY, never through the
4909
5189
  // fire-and-forget wrapper, and the verdict is PARSED, never synthesised (the deleted
4910
5190
  // {grade:'codex-review', gaps: []} turned a stub into a clean review).
@@ -4927,6 +5207,10 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
4927
5207
  // commit/base committed work produced no status entry at all. Now the probe is built for the scope
4928
5208
  // in force, and for `uncommitted` it is a CONTENT comparison against the pre-code baseline.
4929
5209
  let modeBChanged = null
5210
+ // fix-round-1 #4 (HIGH): hoisted so the scope-decision block below (which runs BEFORE the codex
5211
+ // label resolution now) can compare the scope-repo's OWN content receipt against the SAME
5212
+ // measurement, with no second network round-trip.
5213
+ let modeBAfterSnap = null
4930
5214
  if (modeBPlanned.length > 0) {
4931
5215
  const probeCmd = changeSetProbeCmd({ scope: QE_SCOPE, ref: QE_SCOPE_REF, paths: modeBPlanned, quote: shq })
4932
5216
  if (probeCmd === null) {
@@ -4937,6 +5221,7 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
4937
5221
  if (QE_SCOPE === 'uncommitted') {
4938
5222
  // CONTENT, not dirtiness: only a hash that MOVED since the baseline is this run's doing.
4939
5223
  const afterSnap = parseHashProbe(String(chgOut), modeBPlanned)
5224
+ modeBAfterSnap = afterSnap
4940
5225
  modeBChanged = changedFromHashes(preCodeBaseline, afterSnap)
4941
5226
  if (modeBChanged === null) log('QE: no pre-code baseline — the delta is NOT ESTABLISHED (never treated as empty)')
4942
5227
  } else {
@@ -4945,6 +5230,109 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
4945
5230
  }
4946
5231
  }
4947
5232
  }
5233
+ // codex-review-scope (FR-1/FR-2/FR-5, A1/A2/A5): decide BEFORE mode A is ever attempted whether —
5234
+ // and how — it may run for scope 'uncommitted'. Never a bare fallback to --uncommitted on the
5235
+ // shared tree: the decision is recorded on qeCodexLabelOpts (private fields consumed by
5236
+ // codexReviewAgent above) so the call site immediately below stays byte-identical to before this
5237
+ // feature (pinned by cross-family-qe.test.ts / feature-adr-model-routing.test.ts).
5238
+ // fix-round-1 #8 (MEDIUM, Codex r1): this decision block now runs BEFORE codexLabelOptsForDispatch
5239
+ // (moved below) — a scope that is about to be BLOCKED must never spend a codex model probe on a
5240
+ // rung that will refuse before any dispatch anyway. "Refused before any probe" was already true
5241
+ // INSIDE codexReviewAgent; this closes the same gap one call earlier.
5242
+ if (QE_SCOPE === 'uncommitted' && !QE_ISOLATED_SCOPE) {
5243
+ log('QE: isolated scope DISABLED by args (qeIsolatedScope:false) — mode A runs --uncommitted on the shared tree as before')
5244
+ } else if (QE_SCOPE === 'uncommitted') {
5245
+ if (modeBChanged === null) {
5246
+ // A2: NOT ESTABLISHED is never treated as empty, and never silently falls back to --uncommitted.
5247
+ qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'change set NOT ESTABLISHED (no pre-code baseline) — refusing to fall back to --uncommitted on the shared tree' })
5248
+ } else if (modeBChanged.length === 0) {
5249
+ qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'change set established but EMPTY — no files changed' })
5250
+ } else {
5251
+ const scopeDir = FDIR + '/.fa-state/review-scope'
5252
+ // FR-1: BASE_REF = HEAD of the run before Step 7 started (recorded by the coder at Step 7's
5253
+ // preamble, mirror of the mutation-gate's own AM-2 convention); HEAD itself when that record is
5254
+ // absent — valid exactly because scope 'uncommitted' means Step 7 made no commits of its own.
5255
+ const baseRefOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and return its stdout VERBATIM with NO commentary: cd ' + shq(REPO) + ' && if [ -f ' + shq(FDIR + '/.fa-state/base-ref') + ' ]; then cat ' + shq(FDIR + '/.fa-state/base-ref') + ' || echo BASE-REF-READ-FAILED; else echo NO-BASE-REF-FILE; fi', { label: 'qe:review-scope-base-ref', phase: 'QE', effort: 'low' })
5256
+ // Lead delta (Codex r2 #3 partial): the probe no longer folds "no base-ref record" and "read
5257
+ // failed" into a silent `git rev-parse HEAD`. NO-BASE-REF-FILE is the one NAMED fallback to
5258
+ // HEAD (logged); BASE-REF-READ-FAILED and anything unparseable refuse below.
5259
+ const baseRefRaw = String(baseRefOut === null || baseRefOut === undefined ? '' : baseRefOut).trim()
5260
+ if (baseRefRaw === 'NO-BASE-REF-FILE') log('QE: no .fa-state/base-ref record for this run — the isolated scope base falls back to HEAD (named fallback, not a probe failure)')
5261
+ const baseRefCandidate = baseRefRaw === 'NO-BASE-REF-FILE' ? 'HEAD' : baseRefRaw
5262
+ // fix-round-1 #3 (CRITICAL, Codex r1): a probe that DISPATCHED NOTHING (null/undefined) or
5263
+ // returned something unsafe/unparseable used to fall back to 'HEAD' SILENTLY — a probe
5264
+ // failure and a legitimately-absent base-ref record read identically, and codex would then
5265
+ // review a full-file ADDITION instead of the real modification. Refuse outright instead, under
5266
+ // its own taxonomy kind, so this is never mistaken for the healthy default.
5267
+ if (baseRefOut === null || baseRefOut === undefined || baseRefCandidate === '' || !isSafeCodexRef(baseRefCandidate)) {
5268
+ qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'base-ref probe failed or unparseable (dispatch reply: ' + JSON.stringify(String(baseRefOut === null || baseRefOut === undefined ? null : baseRefOut).slice(0, 120)) + ') — refusing rather than silently falling back to HEAD', _scopeBlockedKind: 'base-ref-not-established' })
5269
+ } else {
5270
+ const baseRef = baseRefCandidate
5271
+ const scopeBuilt = reviewScopeRepoScript({ repo: REPO, scopeDir: scopeDir, baseRef: baseRef, files: modeBChanged, quote: shq })
5272
+ if (scopeBuilt.script === null) {
5273
+ qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'scope-build refused: ' + scopeBuilt.reason })
5274
+ } else {
5275
+ const scopeBuildOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and return its stdout VERBATIM with NO commentary: ' + scopeBuilt.script, { label: 'qe:review-scope', phase: 'QE', effort: 'low' })
5276
+ // FR-3 receipt: git status --porcelain in the scope-repo must name EXACTLY the declared
5277
+ // change set — an absent or partial receipt is a build failure, never "close enough"
5278
+ // (feedback-absence-of-receipt-is-not-success).
5279
+ // fix-round-1 #5 (HIGH): porcelain's status column is a FIXED two-character prefix plus one
5280
+ // space (line.slice(3)), never a `\S+\s+` regex — that regex reads a rename `R old -> new`
5281
+ // as the single path "old -> new", losing both real paths (the builder now also passes
5282
+ // --no-renames, so a rename never reaches this parser as a two-path line in the first place).
5283
+ // fix-round-1 #6 (MEDIUM): the path is read WITHOUT trimming — only a trailing '\n'/'\r' is
5284
+ // ever stripped — so a declared path containing a leading/trailing space still round-trips.
5285
+ // fix-round-1 #4 (HIGH): a second table of sha256 hashes is parsed out of the SAME reply
5286
+ // (the builder appends a `sha256sum` step after the porcelain line) and compared to
5287
+ // modeBAfterSnap — a receipt that matches on PATHNAME but not on CONTENT is still a
5288
+ // scope-build-failed, catching a cp that silently degraded to rm -f.
5289
+ const statusText = String(scopeBuildOut === null || scopeBuildOut === undefined ? '' : scopeBuildOut)
5290
+ const gotFiles = new Set()
5291
+ const gotHashes = new Map()
5292
+ for (const rawLine of statusText.split('\n')) {
5293
+ const line = rawLine.replace(/\r$/, '')
5294
+ if (line === '') continue
5295
+ // Lead delta (Codex r2 N2 MEDIUM): sha256sum's format is FIXED — 64 hex, one space, one
5296
+ // mode char (' ' text / '*' binary), then the path byte-for-byte. No trim: a declared
5297
+ // path with a trailing space must key the same way it was declared (r1 #6).
5298
+ const hm = /^([0-9a-f]{64}) [ *](.*)$/.exec(line)
5299
+ if (hm) { gotHashes.set(hm[2].replace(/^\.\//, ''), hm[1]); continue }
5300
+ if (line.length < 4 || line.charAt(2) !== ' ') continue
5301
+ const path = line.slice(3)
5302
+ if (path !== '') gotFiles.add(path)
5303
+ }
5304
+ const wantFiles = new Set(modeBChanged)
5305
+ const receiptOk = gotFiles.size === wantFiles.size && Array.from(wantFiles).every(function (f) { return gotFiles.has(f) })
5306
+ const hashMismatches = modeBAfterSnap === null ? [] : modeBChanged.filter(function (f) {
5307
+ const want = modeBAfterSnap.has(f) ? modeBAfterSnap.get(f) : null
5308
+ const got = gotHashes.has(f) ? gotHashes.get(f) : null
5309
+ return (want === null || want === undefined) ? (got !== null && got !== undefined) : got !== want
5310
+ })
5311
+ if (!receiptOk) {
5312
+ qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'scope-build-failed: git status --porcelain in the scope-repo did not return exactly the declared change set (got ' + gotFiles.size + ', wanted ' + wantFiles.size + ')' })
5313
+ } else if (hashMismatches.length > 0) {
5314
+ qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'scope-build-failed: content hash mismatch for ' + hashMismatches.join(', ') + ' — the scope-repo working copy does not match the measured content' })
5315
+ } else {
5316
+ qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeRepo: scopeDir, _scopeFiles: modeBChanged })
5317
+ }
5318
+ }
5319
+ }
5320
+ }
5321
+ }
5322
+ // fix-round-1 #8 (MEDIUM, Codex r1): resolve the dispatch LABEL after the scope decision — a
5323
+ // BLOCKED scope skips codexLabelOptsForDispatch entirely (no probe spent on a refusal).
5324
+ let qeCodexResolved = qeCodexLabelOpts
5325
+ if (!qeCodexLabelOpts._scopeBlocked) {
5326
+ // R5-1: the id modelsUsed will report, resolved by the SAME memoized probe the dispatch uses.
5327
+ qeCodexResolved = await codexLabelOptsForDispatch(qeCodexLabelOpts)
5328
+ }
5329
+ qeCodexProbeFailed = !!qeCodexResolved._codexProbeFailed
5330
+ const qeCodexDispatchLabel = modelLabel(qeCodexResolved)
5331
+ // R16-1: the codex QE rung records its attempt HERE. The only dispatch-time write used to sit
5332
+ // inside `if (!qeIsCodex)`, so a codex-routed QE whose modes and belt all returned null printed
5333
+ // `▸ qe` lines and left modelsUsed.qe unassigned entirely. The marker is provisional: the success
5334
+ // paths below overwrite it, and if nothing succeeds it stays and says so.
5335
+ modelsUsed.qe = qeCodexDispatchLabel + (qeCodexLabelOpts._scopeBlocked ? ' (scope not established)' : (qeCodexProbeFailed ? ' (codex probe found no usable id)' : ' (no verdict)'))
4948
5336
  const qeRungHolder = newRung()
4949
5337
  let qeLastRungHolder = qeRungHolder
4950
5338
  let codexQe = await codexReviewAgent('qe', QE_SCOPE, QE_SCOPE_REF, 'QE', qeCodexLabelOpts, qeRungHolder)
@@ -5118,19 +5506,55 @@ let qeReviewerUsed = qeStage ? qeStage.qeReviewerUsed : 'claude'
5118
5506
  if (qeStage && qeStage.modelUsed) modelsUsed.qe = qeStage.modelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
5119
5507
  if (qeStage && qeStage.qe2ModelUsed) modelsUsed.qe2 = qeStage.qe2ModelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
5120
5508
 
5509
+ // aqe-ledger-row T3/FR-1/FR-3/A1/A3/A4: the QE step's OWN autorow — who reviewed, how many
5510
+ // findings, cross-family or not — so the instrument's own footprint in the ledger stops being
5511
+ // zero (00_complexity_assessment.md: 0 of 355 rows before this). Guarded by the SAME
5512
+ // resumedStages check every other autorow uses: a resumed QE stage already has its row from the
5513
+ // run that actually did the review, so writing again would double-pay it.
5514
+ if (resumedStages.indexOf('qe') === -1) {
5515
+ // fix-round-1/#1 (Codex r1 HIGH #1): reviewer/reviewerFamily/qeRole now come from the STRICT
5516
+ // reviewerIdentity() classifier (knownFamily()-backed), not tpFamily() — an unrecognized
5517
+ // qeReviewerUsed (undefined, null, a typo) now reads as reviewerFamily:null/qeRole:null, never
5518
+ // guessed as 'claude'. crossFamily is null whenever EITHER family is unknown — comparing a known
5519
+ // family against an unknown one is not a fact, and reporting `false` for it (as the old
5520
+ // `qeReviewerFamily !== tpFamily(coderUsed)` did whenever tpFamily's default silently matched)
5521
+ // would assert a same-family review that was never established.
5522
+ const qeIdentity = reviewerIdentity(qeReviewerUsed, modelsUsed.qe, qe && qe.qeScope ? qe.qeScope.mode : null)
5523
+ const qeCoderFamily = knownFamily(coderUsed)
5524
+ const qeGrade = (qe && typeof qe.grade === 'string' && qe.grade !== '') ? qe.grade : null
5525
+ const qeGradeSource = (qe && typeof qe.gradeSource === 'string' && qe.gradeSource !== '') ? qe.gradeSource : null
5526
+ const qeClaimCheck = (qe && qe.claimCheck && typeof qe.claimCheck === 'object') ? qe.claimCheck : null
5527
+ const qeScopeRow = (qe && qe.qeScope && typeof qe.qeScope === 'object') ? qe.qeScope : null
5528
+ const qeCrossFamily = (qeIdentity.reviewerFamily === null || qeCoderFamily === null) ? null : (qeIdentity.reviewerFamily !== qeCoderFamily)
5529
+ await appendRunCostRow('qe', 'QE', qe ? 'reviewed' : 'no-deliverable', {
5530
+ reviewer: qeIdentity.reviewer,
5531
+ reviewerFamily: qeIdentity.reviewerFamily,
5532
+ qeRole: qeIdentity.qeRole,
5533
+ grade: qeGrade,
5534
+ gradeSource: qeGradeSource,
5535
+ ...qeFindingsSummary(qe && qe.gaps),
5536
+ claimCheck: qeClaimCheck,
5537
+ crossFamily: qeCrossFamily,
5538
+ qeScope: qeScopeRow,
5539
+ })
5540
+ } else {
5541
+ log('run-cost ledger: qe row skipped — stage resumed')
5542
+ }
5543
+
5121
5544
  // Step 8 has completed its teach/reinforce work and written 08_qe_report.md. Closing telemetry is
5122
5545
  // secondary: refusal is loud and reflected in roundClosed, but it never overturns the feature run.
5123
5546
  if (qe && pipelineRound !== null) {
5124
5547
  const roundGrade = String(qe.grade || '').trim().toUpperCase()
5125
5548
  const roundOutcome = ['A', 'A-', 'B+', 'B'].indexOf(roundGrade) !== -1 ? 'shipped' : (['C', 'D'].indexOf(roundGrade) !== -1 ? 'refuted' : null)
5126
5549
  if (roundOutcome === null) {
5127
- log('round close refused: unsupported Step 8 grade ' + JSON.stringify(roundGrade))
5550
+ log('round close skipped: non-terminal Step 8 grade ' + roundGrade + ' — the round stays open for the lead')
5551
+ roundSkippedReason = 'non-terminal-grade'
5128
5552
  } else {
5129
5553
  const roundLessonIds = Array.isArray(qe.roundLessons) ? qe.roundLessons.filter(function (id) { return /^teach:[a-z0-9]+$/i.test(String(id)) }) : []
5130
5554
  const roundLearningArg = roundLessonIds.length > 0
5131
5555
  ? roundLessonIds.map(function (id) { return ' --lesson ' + shq(String(id)) }).join('')
5132
5556
  : ' --no-new-knowledge ' + shq((typeof qe.roundNoNewKnowledge === 'string' && qe.roundNoNewKnowledge.trim() !== '') ? qe.roundNoNewKnowledge.trim() : 'Step 8 returned no new teach receipt for this round')
5133
- const roundCloseCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' round close --slug ' + shq(SLUG) + ' --round ' + pipelineRound + ' --outcome ' + roundOutcome + ' --reason ' + shq('grade ' + roundGrade) + roundLearningArg + ' --project ' + shq(BRAIN) + ' --no-cost --json'
5557
+ const roundCloseCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' round close --slug ' + shq(SLUG) + ' --round ' + pipelineRound + ' --outcome ' + roundOutcome + ' --reason ' + shq('grade ' + roundGrade) + roundLearningArg + ' --grade ' + shq(roundGrade) + ' --reviewer ' + shq(String(modelsUsed.qe || qeReviewerUsed)) + ' --project ' + shq(BRAIN) + ' --no-cost --json'
5134
5558
  const roundCloseOut = await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and return its stdout VERBATIM, nothing else: ' + roundCloseCmd, { label: 'round:close', phase: 'QE', effort: 'low' })
5135
5559
  const roundCloseReceipt = parseRoundCommandJson(roundCloseOut)
5136
5560
  roundClosed = !!(roundCloseReceipt && roundCloseReceipt.marker && roundCloseReceipt.row && roundCloseReceipt.row.stage === 'round')
@@ -5509,6 +5933,7 @@ return {
5509
5933
  codeWrote: code ? code.wrote : [],
5510
5934
  qeGrade: qe ? qe.grade : null,
5511
5935
  roundClosed: roundClosed,
5936
+ roundSkippedReason: roundSkippedReason,
5512
5937
  score: score,
5513
5938
  gaps: qe ? qe.gaps : [],
5514
5939
  codeTestsAdequate: qe ? qe.codeTestsAdequate : null,