@dzhechkov/skills-feature-adr 1.5.12 → 1.5.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.dz-manifest.json
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
},
|
|
14
14
|
{
|
|
15
15
|
"path": "README.md",
|
|
16
|
-
"sha256": "
|
|
16
|
+
"sha256": "619a231352a602c736ce6eec1e5fa79a056dc2f5f80f6b35eed236f9670f7337"
|
|
17
17
|
},
|
|
18
18
|
{
|
|
19
19
|
"path": "bin/cli.js",
|
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
},
|
|
26
26
|
{
|
|
27
27
|
"path": "package.json",
|
|
28
|
-
"sha256": "
|
|
28
|
+
"sha256": "07537ebbd424c668a37a42b670c033a4d3134a9e7e13803f426aa31149a85b58"
|
|
29
29
|
},
|
|
30
30
|
{
|
|
31
31
|
"path": "src/cli.js",
|
|
@@ -157,7 +157,7 @@
|
|
|
157
157
|
},
|
|
158
158
|
{
|
|
159
159
|
"path": "templates/.claude/skills/feature-adr/modules/08-qe.md",
|
|
160
|
-
"sha256": "
|
|
160
|
+
"sha256": "7ab9b7b1523289986a1f51b4cbbebf4410d96ad4a153f22fa8613e8a0f4f5fa7"
|
|
161
161
|
},
|
|
162
162
|
{
|
|
163
163
|
"path": "templates/.claude/skills/feature-adr/modules/09-fleet-qe.md",
|
|
@@ -317,7 +317,7 @@
|
|
|
317
317
|
},
|
|
318
318
|
{
|
|
319
319
|
"path": "templates/.claude/workflows/feature-adr.js",
|
|
320
|
-
"sha256": "
|
|
320
|
+
"sha256": "e96c5280ad21b604036cc912418c9627b0ff0ae8ca1436c58dcff951ba9a034b"
|
|
321
321
|
},
|
|
322
322
|
{
|
|
323
323
|
"path": "templates/lib/memory-protocol.md",
|
|
@@ -329,5 +329,5 @@
|
|
|
329
329
|
}
|
|
330
330
|
]
|
|
331
331
|
},
|
|
332
|
-
"signature": "
|
|
332
|
+
"signature": "CfTJgsBbekrZE+IbzaQFLofuRDZ3B2CJTqPk6gjeOiLXhChXiPGkqxaWKCT+hdQTy6ORB7fu6vfMx37rwJf6Cg=="
|
|
333
333
|
}
|
package/README.md
CHANGED
|
@@ -208,6 +208,32 @@ run; it now survives it in `recordFailures`.
|
|
|
208
208
|
|
|
209
209
|
Requires `@dzhechkov/harness-core >= 0.6.1`.
|
|
210
210
|
|
|
211
|
+
### The QE instrument writes its own ledger row (aqe-ledger-row)
|
|
212
|
+
|
|
213
|
+
Before this, the run-cost ledger recorded `plan`, `full`, `design-gate` and friends automatically, but
|
|
214
|
+
the pass that actually reviews the code — Step 8 QE — left zero autorows: every `qe`/`impl` row in the
|
|
215
|
+
ledger was hand-entered, with no model, no findings count, no cross-family signal. The `Workflow`
|
|
216
|
+
pipeline now writes two additional autorows, both additive-only (existing rows are byte-identical to
|
|
217
|
+
before — `appendRunCostRow` gained an optional fourth `extra` argument, spread in only after every
|
|
218
|
+
pre-existing field):
|
|
219
|
+
|
|
220
|
+
- **`impl`** — written right after the Step 7.5 landing barrier settles: `coder`, `coderFamily`, and
|
|
221
|
+
`landed` (the barrier's verdict string, or `skipped-claude-sync` for a synchronous Claude coder that
|
|
222
|
+
never runs the barrier). Writing it here — before Step 8 starts — means the `qe` row's
|
|
223
|
+
`minutesSincePrev` measures the QE step alone, not QE-plus-code.
|
|
224
|
+
- **`qe`** — written right after the QE step resolves: `reviewer` (the model, or `null` if unknown),
|
|
225
|
+
`reviewerFamily`, `qeRole` (`qe-code-reviewer` for Claude; `codex-review`/`codex-exec` for Codex by
|
|
226
|
+
scope mode; never guessed), `grade`, `gradeSource` (if known), `findings` + `findingsBySeverity` +
|
|
227
|
+
`findingsSource` (normalized from the QE `gaps[]`; an unknown severity counts as `other`, never
|
|
228
|
+
dropped; no `gaps` array at all reads as `findings:0`, `findingsSource:'no-gaps-array'`), `claimCheck`
|
|
229
|
+
(if run), `crossFamily` (bool — reviewer family differs from coder family), and `qeScope` (if the
|
|
230
|
+
reviewer ran scoped, e.g. Codex's `mode`/`ref`/`files`).
|
|
231
|
+
|
|
232
|
+
Both rows are skipped — with a logged reason, never silently — when their stage resumed from a
|
|
233
|
+
checkpoint (`resumedStages`), so a resumed run never double-pays the ledger. Plain-mode runs (the
|
|
234
|
+
interactive SKILL, not the ultracode `Workflow`) carry the same fields as a manual step at the end of
|
|
235
|
+
`modules/08-qe.md` — see that module for the exact command.
|
|
236
|
+
|
|
211
237
|
### Step 0 writes the assessment down, and the acid check gets its input back (v1.5.0)
|
|
212
238
|
|
|
213
239
|
Step 0 classifies the feature and now **writes `00_complexity_assessment.md` before it returns** — the
|
|
@@ -416,10 +442,56 @@ cross-family QE is silently lost — on exactly the big features that need it mo
|
|
|
416
442
|
- **Every fallback names its cause.** The reason carried into
|
|
417
443
|
`opus (cross-family QE DID NOT happen — …)` comes from a locked taxonomy —
|
|
418
444
|
`timeout` (narrow the scope) · `no-verdict` · `tool-error` (fix the invocation) · `unusable-output` ·
|
|
419
|
-
`unavailable` (fix the account/model) · `over-ceiling
|
|
420
|
-
|
|
445
|
+
`unavailable` (fix the account/model) · `over-ceiling` · `scope-not-established` (mode A's scope
|
|
446
|
+
could not be built — see below) · `base-ref-not-established` (the scope's base-ref probe failed or
|
|
447
|
+
was unparseable — see below). A timeout and an unusable output can never render the same string,
|
|
448
|
+
because the operator's next move differs.
|
|
421
449
|
- **The pipeline still never blocks on Codex.** Both modes fail into the same Claude belt as before.
|
|
422
450
|
|
|
451
|
+
**Mode A's `--uncommitted` pass runs in an ISOLATED scope-repo, never on the shared tree.** MEASURED
|
|
452
|
+
2026-09-17 (run `wf_95211e0f`): on a hub with other dirty packages, `codex review --uncommitted`
|
|
453
|
+
wandered into unrelated files (`books/`, `features/clean-code-*`) and timed out at 600s having
|
|
454
|
+
reviewed nothing of the run's own feature — cross-family QE silently lost on exactly the runs that
|
|
455
|
+
need it most. Before mode A is attempted for scope `'uncommitted'` (the default), the pipeline now:
|
|
456
|
+
builds a throwaway git repo under `features/<slug>/.fa-state/review-scope/` containing ONLY the
|
|
457
|
+
ESTABLISHED change set (the same `modeBChanged` measurement mode B already uses — base versions via
|
|
458
|
+
`git show <BASE_REF>:<path>` in one commit, working versions copied on top); verifies the receipt
|
|
459
|
+
(`git status --porcelain` in that repo names EXACTLY the declared files, never "close enough"); and
|
|
460
|
+
runs `codex review --uncommitted` with `repo:` pointed at that isolated tree. Findings come back with
|
|
461
|
+
the scope-repo's own absolute path and are normalized to repo-relative before scoring. A change set
|
|
462
|
+
that is **not established** (`null` — no pre-code baseline) or **established but empty** (`[]` — no
|
|
463
|
+
files changed) refuses BEFORE any dispatch, under one decline kind `scope-not-established` (never a
|
|
464
|
+
silent fallback to `--uncommitted` on the shared tree); a failed scope-repo build or a receipt
|
|
465
|
+
mismatch refuse the same way. Knob: `args.qeIsolatedScope` (default `true`) — `false` restores the
|
|
466
|
+
prior `--uncommitted`-on-the-shared-tree behavior byte-for-byte, with a log line saying so. Scopes
|
|
467
|
+
`commit`/`base` are unaffected.
|
|
468
|
+
|
|
469
|
+
**Fix round 1 hardening (Codex r1 review, 2026-09-17), briefly:** the scope-repo path is validated by
|
|
470
|
+
PATH SEGMENT (`<repo>/features/<slug>/.fa-state/review-scope`, `<slug>` alphanumeric, no `.`/`..`
|
|
471
|
+
segment anywhere), not by a lexical prefix/substring check — a `..`-laced path can no longer walk the
|
|
472
|
+
one destructive `rm -rf` outside `features/`; the script itself repeats the check at runtime (symlink
|
|
473
|
+
+ `case` guard) as a second belt. A base-ref probe that fails or returns something unparseable now
|
|
474
|
+
REFUSES under `base-ref-not-established` instead of silently substituting `HEAD` — a bad ref used to
|
|
475
|
+
make Codex review a full-file addition instead of the real modification. The receipt is now a CONTENT
|
|
476
|
+
check, not just a pathname list: the scope-build script also emits a `sha256sum` of every working
|
|
477
|
+
file, compared against the same pre-measured hashes mode B already computes, so a `cp` that silently
|
|
478
|
+
degraded to a deletion (an unreadable file, one that vanished mid-copy) is caught even though the
|
|
479
|
+
pathname-only receipt would have passed; a genuine `cp` failure now aborts the build rather than being
|
|
480
|
+
read as an intentional deletion. The porcelain receipt parser reads a fixed two-character status
|
|
481
|
+
column (`--no-renames` on the git status call, so a rename can never arrive as the ambiguous
|
|
482
|
+
`old -> new` shape). A declared path is never trimmed and rejects only what the receipt genuinely
|
|
483
|
+
cannot express (control characters, `"`, `\`, backtick, `$`, non-ASCII, a leading `-`) — spaces and
|
|
484
|
+
shell metacharacters are accepted, because every value already passes through the same safe quoting
|
|
485
|
+
function used everywhere else in this file.
|
|
486
|
+
|
|
487
|
+
**Named limit of the isolated scope (owner-facing, so it reads as a boundary, not a defect):** the
|
|
488
|
+
scope-repo review is BLIND to everything outside the declared change set, by construction — that is
|
|
489
|
+
the whole point (it is why the shared-tree run above timed out reviewing unrelated dirty packages).
|
|
490
|
+
A finding phrased as *"file X does not exist"* or *"the twin file is missing"* from mode A is therefore
|
|
491
|
+
an artifact of that intentional narrowness, not a real defect: the twin/sibling file is simply not
|
|
492
|
+
copied into the scope repo. Context that spans beyond the declared files is covered by mode B (which
|
|
493
|
+
is told exactly which files it may open) and by the separate Claude QE pass, never by mode A alone.
|
|
494
|
+
|
|
423
495
|
After a Codex verdict a cheap Claude agent transcribes it into `08_qe_report.md` (mode A takes no
|
|
424
496
|
prompt, so the reviewer cannot be asked to write anything). It is a scribe, not a second reviewer: the
|
|
425
497
|
grade is Codex's and is stated as final.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@dzhechkov/skills-feature-adr",
|
|
3
|
-
"version": "1.5.
|
|
3
|
+
"version": "1.5.13",
|
|
4
4
|
"description": "Adaptive Feature Development skill pack for Claude Code — 11-step pipeline with Complexity Router (S/M/L/XL), ADR-driven architecture, 15 agentic-qe skills, multi-agent fleet QE. Supports --full-qe, --full-qe-extended, --with-learning, and --knowledge-extractor modes.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"skills-feature-adr": "./bin/cli.js"
|
package/sbom.json
CHANGED
|
@@ -35,7 +35,7 @@
|
|
|
35
35
|
"hashes": [
|
|
36
36
|
{
|
|
37
37
|
"alg": "SHA-256",
|
|
38
|
-
"content": "
|
|
38
|
+
"content": "619a231352a602c736ce6eec1e5fa79a056dc2f5f80f6b35eed236f9670f7337"
|
|
39
39
|
}
|
|
40
40
|
]
|
|
41
41
|
},
|
|
@@ -69,7 +69,7 @@
|
|
|
69
69
|
},
|
|
70
70
|
{
|
|
71
71
|
"name": "dz:canonical-json-sha256-v2",
|
|
72
|
-
"value": "
|
|
72
|
+
"value": "07537ebbd424c668a37a42b670c033a4d3134a9e7e13803f426aa31149a85b58"
|
|
73
73
|
}
|
|
74
74
|
]
|
|
75
75
|
},
|
|
@@ -399,7 +399,7 @@
|
|
|
399
399
|
"hashes": [
|
|
400
400
|
{
|
|
401
401
|
"alg": "SHA-256",
|
|
402
|
-
"content": "
|
|
402
|
+
"content": "7ab9b7b1523289986a1f51b4cbbebf4410d96ad4a153f22fa8613e8a0f4f5fa7"
|
|
403
403
|
}
|
|
404
404
|
]
|
|
405
405
|
},
|
|
@@ -799,7 +799,7 @@
|
|
|
799
799
|
"hashes": [
|
|
800
800
|
{
|
|
801
801
|
"alg": "SHA-256",
|
|
802
|
-
"content": "
|
|
802
|
+
"content": "e96c5280ad21b604036cc912418c9627b0ff0ae8ca1436c58dcff951ba9a034b"
|
|
803
803
|
}
|
|
804
804
|
]
|
|
805
805
|
},
|
|
@@ -356,6 +356,41 @@ An empty (header-only) table is read as `hollow: true` — worse than no table a
|
|
|
356
356
|
claims a ledger exists and says nothing. If there are no findings, omit the section entirely rather
|
|
357
357
|
than writing an empty table.
|
|
358
358
|
|
|
359
|
+
### 7.2 Ledger row for this QE step (mandatory)
|
|
360
|
+
|
|
361
|
+
Plain mode (this SKILL, not the ultracode `Workflow`) has no in-process `appendRunCostRow` — without
|
|
362
|
+
this step the QE instrument's own pass leaves ZERO trace in the run-cost ledger (aqe-ledger-row,
|
|
363
|
+
owner audit 17.09, decision #3).
|
|
364
|
+
|
|
365
|
+
**Resume guard — check this FIRST, before writing anything.** If QE was RESTORED from a checkpoint
|
|
366
|
+
for THIS run (`features/<slug>/.fa-state/checkpoints.jsonl` already contains a line for stage `qe`
|
|
367
|
+
that matches this run) — the row for this review was already written by the run that actually did
|
|
368
|
+
it. Do **NOT** run the command below in that case; instead write exactly this line into
|
|
369
|
+
`08_qe_report.md`: `qe ledger row skipped — stage resumed`. Writing the row anyway would double-pay
|
|
370
|
+
the same review in the ledger (aqe-ledger-row fix-round-1/#3, Codex r1 HIGH #3 — plain mode had no
|
|
371
|
+
resume guard at all before this).
|
|
372
|
+
|
|
373
|
+
Otherwise — QE ran fresh in this pass — after `08_qe_report.md` is written and the grade is final, run:
|
|
374
|
+
|
|
375
|
+
```bash
|
|
376
|
+
dz feature-adr-record --kind ledger --stage qe --slug <slug> --row '<json>' --auto --json
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
`<json>` carries the same fields the ultracode pipeline writes for this row: `reviewer` (the model
|
|
380
|
+
that reviewed, or `null` if unknown — never guessed), `reviewerFamily` (`claude`|`codex`|`null` when
|
|
381
|
+
the reviewer identity is not one of the two known families — never guessed as `claude`), `qeRole`
|
|
382
|
+
(`qe-code-reviewer` for a Claude reviewer; `codex-review`/`codex-exec` for a codex reviewer by scope
|
|
383
|
+
mode; `null` otherwise), `grade`, `gradeSource` (if known), `findings` (gap count), `findingsBySeverity`
|
|
384
|
+
(normalized `sev` -> count, unknown severities counted as `other`, never dropped), `findingsSource`
|
|
385
|
+
(`gaps` or `no-gaps-array`), `claimCheck` (if run), `crossFamily` (bool: reviewer family != coder
|
|
386
|
+
family — `null` when either family is unknown, since a same/cross-family verdict cannot be asserted
|
|
387
|
+
without both), `qeScope` (if the reviewer ran scoped, e.g. codex `mode`/`ref`/`files`). Insert the
|
|
388
|
+
writer's verdict (`{verdict:"written"|...}`) verbatim into `08_qe_report.md` so the record is
|
|
389
|
+
auditable from the artifact itself. NAMED LIMIT (layer 4 of the cost-of-detection ladder, said
|
|
390
|
+
honestly): this step is a skill instruction, not a code gate — nothing forces the agent to run it or
|
|
391
|
+
to check the resume guard first; only its PRESENCE in this module is deterministically pinned (grep
|
|
392
|
+
for the command above and for the resume-guard sentence).
|
|
393
|
+
|
|
359
394
|
### 8. QE Pattern Store (Direct Mode only)
|
|
360
395
|
|
|
361
396
|
When `{AGENTIC_QE_MODE}` = `direct` | `direct-extended`, after the gap loop is closed and the verdict is set:
|
|
@@ -851,6 +851,48 @@ function captureFailureRecord(stage, mode, reason, detail) {
|
|
|
851
851
|
return { stage: normalizedStage, mode: normalizedMode, reason: normalizedReason, detail: normalizedDetail }
|
|
852
852
|
}
|
|
853
853
|
function tpFamily(spec) { return /codex|gpt|openai/i.test(String(spec == null ? '' : spec)) ? 'codex' : 'claude' }
|
|
854
|
+
// aqe-ledger-row fix-round-1/#1 (Codex r1 HIGH #1): a STRICT closed-set classifier, deliberately
|
|
855
|
+
// separate from tpFamily() above. tpFamily() maps every unrecognized string — including undefined,
|
|
856
|
+
// null and typos — to 'claude', because its ONE existing job (the cross-model-QE routing default,
|
|
857
|
+
// evaluatorActual/chosenActual) wants a safe default direction. That default is WRONG for a LEDGER
|
|
858
|
+
// IDENTITY field: an unrecognized reviewer must read as unknown (null), never be silently attributed
|
|
859
|
+
// to Claude. knownFamily() is that closed set — only the values this workflow itself ever assigns to
|
|
860
|
+
// coderUsed/qeReviewerUsed ('claude', 'codex', 'codex-fallback') resolve; anything else is null.
|
|
861
|
+
function knownFamily(spec) {
|
|
862
|
+
return spec === 'claude' ? 'claude' : (spec === 'codex' || spec === 'codex-fallback') ? 'codex' : null
|
|
863
|
+
}
|
|
864
|
+
// aqe-ledger-row fix-round-1/#1: reviewerIdentity(qeReviewerUsed, modelUsed, qeScopeMode) — pure,
|
|
865
|
+
// never guesses. `family` comes ONLY from knownFamily() (never tpFamily(), per the owner decision).
|
|
866
|
+
// `qeRole` is 'qe-code-reviewer' ONLY when family is 'claude'; 'codex-review'/'codex-exec' ONLY when
|
|
867
|
+
// family is 'codex' AND qeScopeMode is the matching 'A'/'B'; every other combination (including an
|
|
868
|
+
// unknown family, or a codex family with an unrecognized/missing scope mode) is null — never guessed.
|
|
869
|
+
function reviewerIdentity(qeReviewerUsed, modelUsed, qeScopeMode) {
|
|
870
|
+
const family = knownFamily(qeReviewerUsed)
|
|
871
|
+
const qeRole = family === 'claude'
|
|
872
|
+
? 'qe-code-reviewer'
|
|
873
|
+
: (family === 'codex' && qeScopeMode === 'A') ? 'codex-review'
|
|
874
|
+
: (family === 'codex' && qeScopeMode === 'B') ? 'codex-exec'
|
|
875
|
+
: null
|
|
876
|
+
// r2-N1 (Codex r2 HIGH, lead delta): a model label is an identity claim too — with the FAMILY
|
|
877
|
+
// unknown, the label cannot be attributed (`reviewer:'gpt-x'` under an unrecognized route would
|
|
878
|
+
// read as a Codex review that never provably happened). Null outside the known set, all three.
|
|
879
|
+
const reviewer = family !== null && typeof modelUsed === 'string' && modelUsed !== '' ? modelUsed : null
|
|
880
|
+
return { reviewer: reviewer, reviewerFamily: family, qeRole: qeRole }
|
|
881
|
+
}
|
|
882
|
+
// aqe-ledger-row T2/A2/NFR-4: pure helper — {findings, findingsBySeverity, findingsSource} from a QE
|
|
883
|
+
// gaps[] array. Not an array (missing/malformed) -> findings:0, findingsSource:'no-gaps-array', never a
|
|
884
|
+
// guess. sev is normalized (String -> trim -> lowercase); anything outside the closed set counts as
|
|
885
|
+
// 'other' rather than being dropped, so a finding is never silently lost from the total.
|
|
886
|
+
function qeFindingsSummary(gaps) {
|
|
887
|
+
const bySeverity = { blocker: 0, critical: 0, high: 0, medium: 0, low: 0, info: 0, other: 0 }
|
|
888
|
+
if (!Array.isArray(gaps)) return { findings: 0, findingsBySeverity: bySeverity, findingsSource: 'no-gaps-array' }
|
|
889
|
+
for (const g of gaps) {
|
|
890
|
+
const sev = String((g && g.sev !== undefined && g.sev !== null) ? g.sev : '').trim().toLowerCase()
|
|
891
|
+
if (Object.prototype.hasOwnProperty.call(bySeverity, sev)) bySeverity[sev] += 1
|
|
892
|
+
else bySeverity.other += 1
|
|
893
|
+
}
|
|
894
|
+
return { findings: gaps.length, findingsBySeverity: bySeverity, findingsSource: 'gaps' }
|
|
895
|
+
}
|
|
854
896
|
function tpText(v) { if (typeof v === 'string') return v; if (v === null || v === undefined) return ''; try { const s = JSON.stringify(v); return typeof s === 'string' ? s : String(v) } catch (e) { return String(v) } }
|
|
855
897
|
function tpBudget(raw) {
|
|
856
898
|
try {
|
|
@@ -1045,7 +1087,10 @@ async function capturePairs(stage, phaseName, records, resumeGuardStage) {
|
|
|
1045
1087
|
}
|
|
1046
1088
|
}
|
|
1047
1089
|
|
|
1048
|
-
async function appendRunCostRow(stage, phaseName, outcome) {
|
|
1090
|
+
async function appendRunCostRow(stage, phaseName, outcome, extra) {
|
|
1091
|
+
// aqe-ledger-row T1/NFR-1: `extra` is OPTIONAL and ADDITIVE-ONLY. Its keys are spread into
|
|
1092
|
+
// the row literal AFTER every existing field (after `runId:`, the last pre-existing field), so
|
|
1093
|
+
// a call site that passes no fourth argument produces the exact byte-identical row it always did.
|
|
1049
1094
|
// Like capturePairs, a ledger failure is a logged SECONDARY event that can NEVER fail the run;
|
|
1050
1095
|
// the whole body therefore rides one best-effort try/catch and never rethrows.
|
|
1051
1096
|
try {
|
|
@@ -1093,6 +1138,8 @@ async function appendRunCostRow(stage, phaseName, outcome) {
|
|
|
1093
1138
|
// skips that guess entirely (run-records.ts / cli.ts: runIdForLookup is only computed when
|
|
1094
1139
|
// the payload carries no runId), so passing it here is a strict reliability improvement.
|
|
1095
1140
|
runId: (typeof RUN_ID === 'string' && RUN_ID !== '') ? RUN_ID : null,
|
|
1141
|
+
// aqe-ledger-row T1/NFR-1: additive-only — spreads nothing when `extra` is absent.
|
|
1142
|
+
...(extra && typeof extra === 'object' ? extra : {}),
|
|
1096
1143
|
})
|
|
1097
1144
|
// WITNESSED WRITE (ADR-001): the subagent RUNS a command with data arguments; it is no longer
|
|
1098
1145
|
// handed a shell pipeline with the row baked in. The command refuses a malformed row, stamps the
|
|
@@ -1157,6 +1204,11 @@ const CODEX_HINT = ' (If you are the Codex runtime, prefer the ' + CODEX_MODEL +
|
|
|
1157
1204
|
// and --base HEAD review the identical tree while --uncommitted needs no ref at all.
|
|
1158
1205
|
const QE_SCOPE = (A.qeScope === 'commit' || A.qeScope === 'base') ? A.qeScope : 'uncommitted'
|
|
1159
1206
|
const QE_SCOPE_REF = (typeof A.qeScopeRef === 'string') ? A.qeScopeRef : ''
|
|
1207
|
+
// codex-review-scope (FR-5, A5): mode A's 'uncommitted' pass runs in an ISOLATED scope-repo by
|
|
1208
|
+
// default (MEASURED wf_95211e0f, 2026-09-17: --uncommitted on the shared tree wandered into other
|
|
1209
|
+
// dirty packages and timed out at 600s having reviewed nothing of the run's own feature).
|
|
1210
|
+
// args.qeIsolatedScope:false restores the prior --uncommitted-on-REPO behavior byte-for-byte.
|
|
1211
|
+
const QE_ISOLATED_SCOPE = A.qeIsolatedScope !== false
|
|
1160
1212
|
// Mode-B questions: the ones --commit structurally forbids us from asking.
|
|
1161
1213
|
const SCOPED_QE_QUESTIONS = [
|
|
1162
1214
|
'Is the change correct — name any real defect with file and line, or say there is none.',
|
|
@@ -1982,7 +2034,16 @@ const CODEX_REVIEW_TIMEOUT_SECONDS = 600
|
|
|
1982
2034
|
const CODEX_REVIEW_DEFAULT_EFFORT = 'high'
|
|
1983
2035
|
const CODEX_TIMEOUT = 'CODEX_TIMEOUT'
|
|
1984
2036
|
const CODEX_QE_SIGNAL_PREFIX = 'CODEX-QE-SIGNAL'
|
|
1985
|
-
|
|
2037
|
+
// 'scope-not-established' joined the set 2026-09-17 (feature codex-review-scope, FR-2/A2): mode A's
|
|
2038
|
+
// isolated scope-repo is built ONLY from an ESTABLISHED change set (modeBChanged an array >= 1
|
|
2039
|
+
// entry) — null (not established) or an established-but-empty set both refuse under this ONE kind,
|
|
2040
|
+
// distinguished by the reason TEXT (never a second kind), so the taxonomy grows by exactly one.
|
|
2041
|
+
// 'base-ref-not-established' joined the set 2026-09-17 (fix-round-1 #3, Codex r1 CRITICAL/HIGH): a
|
|
2042
|
+
// base-ref probe that fails or returns something unparseable used to fall back to 'HEAD' SILENTLY —
|
|
2043
|
+
// indistinguishable from a healthy default, and codex would then review a full-file addition
|
|
2044
|
+
// instead of the real modification. This is a DIFFERENT kind from 'scope-not-established' (not just
|
|
2045
|
+
// different reason text) because the failure is about the REF, not the change set.
|
|
2046
|
+
const CODEX_QE_DECLINE_KINDS = ['timeout', 'no-verdict', 'tool-error', 'unusable-output', 'unavailable', 'over-ceiling', 'wrong-tree', 'scope-not-established', 'base-ref-not-established']
|
|
1986
2047
|
const SCOPED_QE_MAX_FILES = 3
|
|
1987
2048
|
const SCOPED_QE_MAX_QUESTIONS = 4
|
|
1988
2049
|
const SCOPED_QE_MAX_PATH_CHARS = 200
|
|
@@ -2042,6 +2103,172 @@ function codexReviewCommand(input) {
|
|
|
2042
2103
|
return { cmd: cmd, carriesPrompt: false, scope: scope, reason: null }
|
|
2043
2104
|
}
|
|
2044
2105
|
|
|
2106
|
+
// ── ISOLATED REVIEW SCOPE (feature codex-review-scope) ──────────────────────────────────────────
|
|
2107
|
+
// MEASURED 2026-09-17 (wf_95211e0f, 07:14-08:22): mode A's codex review --uncommitted on the
|
|
2108
|
+
// shared hub tree wandered into OTHER dirty packages (books/, features/clean-code-*) and timed out
|
|
2109
|
+
// at 600s having reviewed nothing of this run's own feature — the same class as teach:c551a2c3
|
|
2110
|
+
// (12.09: 816KB of log, no verdict). The fix is not a bigger timeout (that buys more reconnaissance,
|
|
2111
|
+
// per the qe-scoped-review ADR above); it is giving mode A a TREE that contains ONLY this run's
|
|
2112
|
+
// established changes.
|
|
2113
|
+
//
|
|
2114
|
+
// A live probe (14:25, codex-cli 0.154.0, gpt-5.6-terra medium) confirmed the mechanics: a throwaway
|
|
2115
|
+
// git repo with one committed base file plus one working-tree edit, codex review --uncommitted
|
|
2116
|
+
// exits 0 in 16.7s with a real finding whose location is the scope repo's ABSOLUTE path — hence
|
|
2117
|
+
// normalizeScopedFindings below.
|
|
2118
|
+
const SCOPE_REPO_GIT_EMAIL = 'dz-review-scope@localhost'
|
|
2119
|
+
const SCOPE_REPO_GIT_NAME = 'dz-review-scope'
|
|
2120
|
+
|
|
2121
|
+
// T1 (FR-3, A3, NFR-2). Pure builder: {repo, scopeDir, baseRef, files, quote} -> {script, reason}.
|
|
2122
|
+
// script is ONE shell command line (chained with && ; each per-file step is parenthesised so a
|
|
2123
|
+
// missing base file or a working-tree deletion never aborts the rest of the build); null + reason
|
|
2124
|
+
// on any refusal. Never invoked here — this file has no child_process; the caller dispatches script
|
|
2125
|
+
// through a shell agent (effort low), exactly like every other shell step in this pipeline.
|
|
2126
|
+
//
|
|
2127
|
+
// The emitted rm -rf is the ONLY destructive line this builder can produce, and it fires ONLY when
|
|
2128
|
+
// scopeDir is provably a child of <repo>/features/<slug>/.fa-state/review-scope — never a
|
|
2129
|
+
// caller-supplied path taken on faith (NFR-2).
|
|
2130
|
+
//
|
|
2131
|
+
// fix-round-1 #1 (BLOCKER, Codex r1): the old containment check was LEXICAL — scopeDir merely had
|
|
2132
|
+
// to START WITH '<repo>/features/' and CONTAIN '/.fa-state/review-scope' anywhere after that, so
|
|
2133
|
+
// '<repo>/features/../victim/.fa-state/review-scope' passed both tests while '..' walks the rm -rf
|
|
2134
|
+
// straight out of features/. The check below is SEGMENT-based (split('/'), never substring/indexOf
|
|
2135
|
+
// on the whole string): scopeDir must equal repo's own segments, followed by EXACTLY the four
|
|
2136
|
+
// segments ['features', <slug>, '.fa-state', 'review-scope'] with <slug> matching SCOPE_SLUG_RE —
|
|
2137
|
+
// and every segment anywhere in scopeDir is rejected if it is '', '.' or '..'.
|
|
2138
|
+
const SCOPE_SLUG_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,39}$/
|
|
2139
|
+
function reviewScopeRepoScript(input) {
|
|
2140
|
+
const o = input || {}
|
|
2141
|
+
const quote = o.quote
|
|
2142
|
+
if (typeof quote !== 'function') return { script: null, reason: 'no quote function supplied' }
|
|
2143
|
+
const repo = String(o.repo === undefined || o.repo === null ? '' : o.repo)
|
|
2144
|
+
const scopeDir = String(o.scopeDir === undefined || o.scopeDir === null ? '' : o.scopeDir)
|
|
2145
|
+
const baseRef = String(o.baseRef === undefined || o.baseRef === null ? '' : o.baseRef)
|
|
2146
|
+
if (repo === '' || scopeDir === '') return { script: null, reason: 'empty repo or scopeDir' }
|
|
2147
|
+
if (!isSafeCodexRef(baseRef)) return { script: null, reason: 'unsafe baseRef' }
|
|
2148
|
+
const repoSegs = repo.split('/')
|
|
2149
|
+
const dirSegs = scopeDir.split('/')
|
|
2150
|
+
for (const seg of dirSegs) {
|
|
2151
|
+
if (seg === '.' || seg === '..') return { script: null, reason: 'scopeDir contains an unsafe "." or ".." path segment' }
|
|
2152
|
+
}
|
|
2153
|
+
let repoMatches = dirSegs.length >= repoSegs.length
|
|
2154
|
+
if (repoMatches) for (let i = 0; i < repoSegs.length; i++) { if (dirSegs[i] !== repoSegs[i]) { repoMatches = false; break } }
|
|
2155
|
+
const rest = repoMatches ? dirSegs.slice(repoSegs.length) : null
|
|
2156
|
+
const slug = rest && rest.length === 4 ? rest[1] : ''
|
|
2157
|
+
const shapeOk = !!rest && rest.length === 4 && rest[0] === 'features' && rest[2] === '.fa-state' && rest[3] === 'review-scope' && SCOPE_SLUG_RE.test(slug)
|
|
2158
|
+
if (!shapeOk) {
|
|
2159
|
+
return { script: null, reason: 'scopeDir must be exactly ' + repo + '/features/<slug>/.fa-state/review-scope with a safe <slug>' }
|
|
2160
|
+
}
|
|
2161
|
+
const rawFiles = Array.isArray(o.files) ? o.files : []
|
|
2162
|
+
const files = []
|
|
2163
|
+
const seen = new Set()
|
|
2164
|
+
for (const raw of rawFiles) {
|
|
2165
|
+
// fix-round-1 #6 (MEDIUM, Codex r1): NEVER trim a declared path — trimming 'src/a.ts ' silently
|
|
2166
|
+
// retargets the shell command at 'src/a.ts', a DIFFERENT file than the one declared, and the
|
|
2167
|
+
// subsequent receipt mismatch then reads as a build failure rather than the truncation that
|
|
2168
|
+
// caused it. Reject only what the TRANSPORT genuinely cannot express: every character below is
|
|
2169
|
+
// safely embeddable through the injected 'quote' (single-quoting round-trips ANY byte, including
|
|
2170
|
+
// ';&|<>*?()[]{}!' and even a literal "'"), so the real constraint is the FR-3 receipt —
|
|
2171
|
+
// 'git status --porcelain' QUOTES a path containing a control character, '"', '\' or (by default)
|
|
2172
|
+
// non-ASCII, and our receipt parser reads that raw line without un-quoting it. A leading '-' is
|
|
2173
|
+
// refused so no downstream tool ever reads the path as a flag. Spaces are explicitly ALLOWED.
|
|
2174
|
+
const f = String(raw === undefined || raw === null ? '' : raw)
|
|
2175
|
+
if (f === '') continue
|
|
2176
|
+
if (seen.has(f)) continue
|
|
2177
|
+
if (f.charAt(0) === '/') return { script: null, reason: 'absolute path ' + JSON.stringify(f) }
|
|
2178
|
+
if (f === '..' || f.indexOf('../') === 0 || f.indexOf('/../') !== -1 || f.slice(-3) === '/..') {
|
|
2179
|
+
return { script: null, reason: 'path traversal ' + JSON.stringify(f) }
|
|
2180
|
+
}
|
|
2181
|
+
if (f.charAt(0) === '-') return { script: null, reason: 'path may not start with "-" ' + JSON.stringify(f) }
|
|
2182
|
+
if (f.charAt(f.length - 1) === '/') return { script: null, reason: 'path must not end with "/" ' + JSON.stringify(f) }
|
|
2183
|
+
if (/[^\x20-\x7e]/.test(f) || /["\\\x60$]/.test(f)) {
|
|
2184
|
+
return { script: null, reason: 'unsafe path (control/quote/backtick/$/non-ASCII character) ' + JSON.stringify(f) }
|
|
2185
|
+
}
|
|
2186
|
+
seen.add(f)
|
|
2187
|
+
files.push(f)
|
|
2188
|
+
}
|
|
2189
|
+
if (files.length === 0) return { script: null, reason: 'empty file list' }
|
|
2190
|
+
const parts = []
|
|
2191
|
+
// fix-round-1 #3 (HIGH, Codex r1) — script half: verify baseRef resolves to a real commit as the
|
|
2192
|
+
// FIRST command. Every later "git show <baseRef>:<path>" failure used to be swallowed identically
|
|
2193
|
+
// whether the FILE was absent at a valid commit (expected — a new file) or the REF itself was
|
|
2194
|
+
// bogus (a silent full-tree diff against nothing) — indistinguishable from outside. A bad ref now
|
|
2195
|
+
// aborts loudly (exit 4) before any per-file step runs.
|
|
2196
|
+
parts.push('git -C ' + quote(repo) + ' rev-parse --verify --quiet ' + quote(baseRef + '^{commit}') + ' > /dev/null || exit 4')
|
|
2197
|
+
// fix-round-1 #1 — runtime belt matching the JS-side segment check above: a symlink at scopeDir
|
|
2198
|
+
// (planted between the JS check and this script's execution) is refused rather than followed by
|
|
2199
|
+
// rm -rf, and a 'case' re-asserts the very prefix the JS validator just proved, so the two checks
|
|
2200
|
+
// can never silently drift apart.
|
|
2201
|
+
// Lead delta after the second manual e2e (2026-09-17 15:35): the form '[ -L dir ] && exit 3' inside a
|
|
2202
|
+
// '&&'-joined chain ABORTS the chain whenever dir is NOT a symlink (the test returns 1), so the
|
|
2203
|
+
// fix-round script exited 1 having built nothing — a dead feature that fails closed. A statement
|
|
2204
|
+
// form keeps the belt and lets the chain continue.
|
|
2205
|
+
parts.push('if [ -L ' + quote(scopeDir) + ' ]; then exit 3; fi')
|
|
2206
|
+
parts.push('case ' + quote(scopeDir) + ' in ' + quote(repo) + '/features/*/.fa-state/review-scope) : ;; *) exit 3 ;; esac')
|
|
2207
|
+
parts.push('rm -rf ' + quote(scopeDir))
|
|
2208
|
+
parts.push('mkdir -p ' + quote(scopeDir))
|
|
2209
|
+
parts.push('git -C ' + quote(scopeDir) + ' init -q')
|
|
2210
|
+
parts.push('git -C ' + quote(scopeDir) + ' config user.email ' + quote(SCOPE_REPO_GIT_EMAIL))
|
|
2211
|
+
parts.push('git -C ' + quote(scopeDir) + ' config user.name ' + quote(SCOPE_REPO_GIT_NAME))
|
|
2212
|
+
for (const f of files) {
|
|
2213
|
+
const dest = scopeDir + '/' + f
|
|
2214
|
+
const slash = dest.lastIndexOf('/')
|
|
2215
|
+
const destDir = dest.slice(0, slash)
|
|
2216
|
+
if (destDir !== scopeDir) parts.push('mkdir -p ' + quote(destDir))
|
|
2217
|
+
// fix-round-1 #3 (HIGH) — per file: distinguish "path absent at a valid commit" (expected for a
|
|
2218
|
+
// new file — degrade to rm -f) from every OTHER git-show failure (permissions, corrupt object,
|
|
2219
|
+
// …) which now ABORTS the build (exit 5) instead of silently degrading the same way. cat-file -e
|
|
2220
|
+
// is the existence check git itself uses; git show is only reached once existence is confirmed.
|
|
2221
|
+
// Lead delta (Codex r2 N1 HIGH): 'cat-file -e' folds "absent at the base" and "object error"
|
|
2222
|
+
// into one non-zero — a tree entry whose blob is missing/corrupt was silently turned into a
|
|
2223
|
+
// full-file ADDITION. Three-way probe instead: ls-tree FAILS → exit 7 (build failure, named);
|
|
2224
|
+
// empty listing → genuinely absent at the base → no base copy; non-empty → 'show' is MANDATORY
|
|
2225
|
+
// and its failure aborts (exit 5).
|
|
2226
|
+
parts.push('if ! lt=$(git -C ' + quote(repo) + ' ls-tree --name-only ' + quote(baseRef) + ' -- ' + quote(f) + '); then exit 7; fi; if [ -n "$lt" ]; then git -C ' + quote(repo) + ' show ' + quote(baseRef + ':' + f) + ' > ' + quote(dest) + ' || exit 5; else rm -f ' + quote(dest) + '; fi')
|
|
2227
|
+
}
|
|
2228
|
+
parts.push('git -C ' + quote(scopeDir) + ' add -A && git -C ' + quote(scopeDir) + ' commit -q --allow-empty -m base')
|
|
2229
|
+
for (const f of files) {
|
|
2230
|
+
const src = repo + '/' + f
|
|
2231
|
+
const dest = scopeDir + '/' + f
|
|
2232
|
+
// fix-round-1 #4 (HIGH, Codex r1): the old '[ -f src ] && cp src dest || rm -f dest' ran the
|
|
2233
|
+
// REMOVAL whenever 'cp' itself failed (permissions, disk full, the file vanishing mid-copy) —
|
|
2234
|
+
// indistinguishable from a genuine working-tree deletion, and the FR-3 pathname-only receipt
|
|
2235
|
+
// then accepted the wrong outcome as a pass. A real 'cp' failure now aborts the build (exit 6);
|
|
2236
|
+
// only a MISSING source file (a real deletion) degrades to rm -f.
|
|
2237
|
+
parts.push('if [ -f ' + quote(src) + ' ]; then cp ' + quote(src) + ' ' + quote(dest) + ' || exit 6; else rm -f ' + quote(dest) + '; fi')
|
|
2238
|
+
}
|
|
2239
|
+
// Lead delta (manual e2e 2026-09-17 14:51, BEFORE Codex r1): without --untracked-files=all a NEW
|
|
2240
|
+
// file of the change set is reported as its collapsed parent directory ('?? packages/x/'), the
|
|
2241
|
+
// receipt set never equals the declared set, and mode A is refused as scope-build-failed for
|
|
2242
|
+
// every feature that ADDS a file. Per-file listing makes the receipt compare paths with paths.
|
|
2243
|
+
// fix-round-1 #5 (HIGH, Codex r1): --no-renames forces git to report a rename as a plain
|
|
2244
|
+
// delete+add instead of the two-path 'R old -> new' line, which a fixed-width slice(3) parser
|
|
2245
|
+
// (the receipt side, below) cannot split back into two paths.
|
|
2246
|
+
parts.push('git -C ' + quote(scopeDir) + ' status --porcelain --untracked-files=all --no-renames')
|
|
2247
|
+
// fix-round-1 #4 (HIGH) — content receipt: a sha256sum line per declared working file, in the same
|
|
2248
|
+
// "<hash> <path>" shape changeSetProbeCmd's uncommitted probe already emits, so the workflow can
|
|
2249
|
+
// compare it against the ALREADY-MEASURED afterSnap for the same files. A cp that silently
|
|
2250
|
+
// degraded to rm -f shows up here as a hash MISMATCH even when the pathname-only porcelain receipt
|
|
2251
|
+
// above would have passed (a deletion and a failed-copy-then-deletion look identical by name alone).
|
|
2252
|
+
parts.push('( cd ' + quote(scopeDir) + ' && sha256sum -- ' + files.map(quote).join(' ') + ' 2>/dev/null || true )')
|
|
2253
|
+
return { script: parts.join(' && '), reason: null }
|
|
2254
|
+
}
|
|
2255
|
+
|
|
2256
|
+
// T3 (FR-4, A4). Mode-A findings from a scoped review carry the scope-repo's ABSOLUTE path (MEASURED
|
|
2257
|
+
// live probe above). Strip it back to repo-relative so partitionReviewFindings can match it against
|
|
2258
|
+
// modeBChanged. A finding whose location does not start with scopeDir is left UNTOUCHED — never
|
|
2259
|
+
// guessed as belonging to the scope.
|
|
2260
|
+
function normalizeScopedFindings(findings, scopeDir) {
|
|
2261
|
+
const list = Array.isArray(findings) ? findings : []
|
|
2262
|
+
const dir = String(scopeDir === undefined || scopeDir === null ? '' : scopeDir)
|
|
2263
|
+
if (dir === '') return list
|
|
2264
|
+
const prefix = dir.charAt(dir.length - 1) === '/' ? dir : dir + '/'
|
|
2265
|
+
return list.map(function (f) {
|
|
2266
|
+
const loc = (f && f.location) ? String(f.location) : ''
|
|
2267
|
+
if (loc.indexOf(prefix) !== 0) return f
|
|
2268
|
+
return { severity: f.severity, title: f.title, location: loc.slice(prefix.length) }
|
|
2269
|
+
})
|
|
2270
|
+
}
|
|
2271
|
+
|
|
2045
2272
|
// Mode B. The "do NOT open any other file" clause is LOAD-BEARING TEXT — it is the difference
|
|
2046
2273
|
// between the 41s graded run and the 280s ungraded one. An empty file list returns '' so an UNSCOPED
|
|
2047
2274
|
// mode-B dispatch is not constructible at all.
|
|
@@ -2226,6 +2453,8 @@ function codexQeDeclineReason(kind, detail) {
|
|
|
2226
2453
|
// tool-error right above rendered it — the asymmetry that made the field report unfixable blind.
|
|
2227
2454
|
if (canonical === 'unavailable') return 'codex not used — ' + ((d.reason === undefined || d.reason === null || String(d.reason) === '') ? 'codex exec reported it could not run' : String(d.reason)) + (extra === 'no detail' ? '' : ' (' + extra + ')')
|
|
2228
2455
|
if (canonical === 'over-ceiling') return 'prompt is ' + chars + ' chars / unscoped — refused before dispatch'
|
|
2456
|
+
if (canonical === 'scope-not-established') return 'review scope NOT ESTABLISHED for uncommitted QE — ' + ((d.reason === undefined || d.reason === null || String(d.reason) === '') ? 'the change set could not be measured' : String(d.reason)) + ' — refusing to fall back to --uncommitted on the shared tree'
|
|
2457
|
+
if (canonical === 'base-ref-not-established') return 'base ref for the isolated review scope NOT ESTABLISHED — ' + ((d.reason === undefined || d.reason === null || String(d.reason) === '') ? 'the base-ref probe failed or was unparseable' : String(d.reason)) + ' — refusing to silently fall back to HEAD'
|
|
2229
2458
|
throw new Error('codexQeDeclineReason: unknown kind ' + k)
|
|
2230
2459
|
}
|
|
2231
2460
|
|
|
@@ -2455,11 +2684,18 @@ function codexQeSignalCommand(inner, outPath) {
|
|
|
2455
2684
|
// Shared tail of both dispatch modes: run the signal-wrapped command through a shell agent and
|
|
2456
2685
|
// CLASSIFY what came back. signalExpected is true here — on the pipeline path a swallowed sentinel
|
|
2457
2686
|
// means the command did not demonstrably run, which is a tool-error, never a pass.
|
|
2458
|
-
async function runCodexQeCommand(stage, cmd, phaseName, label, probed, mode, scopeRef, files, allowStatedGrade, requestedReasoning, rung, announcementOpts) {
|
|
2687
|
+
async function runCodexQeCommand(stage, cmd, phaseName, label, probed, mode, scopeRef, files, allowStatedGrade, requestedReasoning, rung, announcementOpts, scopeDir) {
|
|
2459
2688
|
const wrapped = 'Run EXACTLY this via Bash and reply with its stdout VERBATIM and nothing else, INCLUDING the final ' + CODEX_QE_SIGNAL_PREFIX + ' line (it is a machine signal, not prose — do not summarise, reformat or omit it). Only if you cannot run the command AT ALL (no shell, command not found) reply with exactly ' + CODEX_UNAVAILABLE + '; a timeout is NOT that case, it reports itself in the signal line.\n\n' + codexQeSignalCommand(cmd, '/tmp/dz-codex-qe-' + SLUG + '-' + stage + '-' + mode + '.out')
|
|
2460
2689
|
const raw = await dispatchAgent(rung, wrapped, { label: stageLabel(label, { agentType: 'codex:codex-rescue', codexModel: probed, _reasoning: requestedReasoning || 'high' }), phase: phaseName, model: 'haiku', effort: 'low' }, announcementOpts)
|
|
2461
2690
|
const sig = parseCodexReviewSignal(raw === null ? '' : String(raw))
|
|
2462
|
-
|
|
2691
|
+
// fix-round-1 #2 (CRITICAL, Codex r1): normalize a scoped review's findings to repo-relative
|
|
2692
|
+
// IMMEDIATELY after parsing — before classifyCodexQeOutcome or gradeFromReviewFindings ever see
|
|
2693
|
+
// them, and before the caller does anything else with the return value. declaredFiles ('files')
|
|
2694
|
+
// are already repo-relative ('_scopeFiles'); leaving the PARSED findings on the scope-repo's
|
|
2695
|
+
// ABSOLUTE path for even one extra hop is the class of bug this closes — every consumer downstream
|
|
2696
|
+
// of this function now sees only repo-relative locations, never a mix of the two shapes.
|
|
2697
|
+
const rawFindings = parseCodexReviewFindings(sig.body)
|
|
2698
|
+
const findings = scopeDir ? normalizeScopedFindings(rawFindings, scopeDir) : rawFindings
|
|
2463
2699
|
// Mode A NEVER asked for a letter (every scope flag rejects a prompt), so any "Grade: X" in its
|
|
2464
2700
|
// output came from the CODE UNDER REVIEW, not from the reviewer. FOUND BY THE FIRST LIVE MODE-A RUN
|
|
2465
2701
|
// (2026-08-21): this feature's own README and CHANGELOG quote "Grade: B", and the review of that
|
|
@@ -2480,19 +2716,42 @@ async function runCodexQeCommand(stage, cmd, phaseName, label, probed, mode, sco
|
|
|
2480
2716
|
// and exit 124 without one). It cannot carry our questions: every scope flag refuses [PROMPT].
|
|
2481
2717
|
async function codexReviewAgent(stage, scope, scopeRef, phaseName, requestedOpts, rung) {
|
|
2482
2718
|
lastCodexDecline = null
|
|
2719
|
+
// codex-review-scope (A1/A2): the scope DECISION travels on requestedOpts as PRIVATE fields
|
|
2720
|
+
// (_scopeBlocked / _scopeRepo / _scopeFiles) rather than a new positional parameter —
|
|
2721
|
+
// codexReviewAgent's signature and its call site are BOTH pinned byte-for-byte
|
|
2722
|
+
// (cross-family-qe.test.ts, feature-adr-model-routing.test.ts), so this is the one channel that
|
|
2723
|
+
// extends behavior without touching either pin. A blocked scope refuses BEFORE any probe is spent —
|
|
2724
|
+
// mode A never dispatches against the shared tree for an unestablished or empty change set.
|
|
2725
|
+
if (requestedOpts && requestedOpts._scopeBlocked) {
|
|
2726
|
+
// fix-round-1 #3 (HIGH, Codex r1): a blocked scope carries its own taxonomy KIND when the
|
|
2727
|
+
// decision knows a more specific one (e.g. 'base-ref-not-established') — defaulting to
|
|
2728
|
+
// 'scope-not-established' keeps every EXISTING caller (none of which set _scopeBlockedKind)
|
|
2729
|
+
// byte-identical.
|
|
2730
|
+
const blockedKind = (requestedOpts._scopeBlockedKind && CODEX_QE_DECLINE_KINDS.indexOf(requestedOpts._scopeBlockedKind) !== -1) ? requestedOpts._scopeBlockedKind : 'scope-not-established'
|
|
2731
|
+
settleUndispatchedStage(requestedOpts, rung, 'refused-before-dispatch', blockedKind)
|
|
2732
|
+
return noteCodexDecline(stage, blockedKind, { reason: requestedOpts._scopeBlocked })
|
|
2733
|
+
}
|
|
2483
2734
|
const requestedId = requestedOpts && requestedOpts.codexModel !== 'auto' ? requestedOpts.codexModel : null
|
|
2484
2735
|
const requestedReasoning = (requestedOpts && requestedOpts._reasoning) || 'high'
|
|
2485
2736
|
const probed = await probeCodexId(requestedId)
|
|
2486
2737
|
if (!probed) { settleUndispatchedStage(requestedOpts, rung, 'probe-failed'); return noteCodexDecline(stage, 'unavailable', { reason: 'no codex model id answered the probe' }) }
|
|
2487
2738
|
const announcementOpts = mergeOpts(requestedOpts || {}, { codexModel: probed, _stage: stage })
|
|
2488
|
-
|
|
2739
|
+
// A1: for scope 'uncommitted' this is REPO only when isolation was explicitly disabled
|
|
2740
|
+
// (args.qeIsolatedScope:false) or never applies (commit/base) — never a silent fallback.
|
|
2741
|
+
const scopeRepo = (requestedOpts && typeof requestedOpts._scopeRepo === 'string' && requestedOpts._scopeRepo !== '') ? requestedOpts._scopeRepo : REPO
|
|
2742
|
+
const declaredFiles = (requestedOpts && Array.isArray(requestedOpts._scopeFiles)) ? requestedOpts._scopeFiles : []
|
|
2743
|
+
const built = codexReviewCommand({ scope: scope, ref: scopeRef, modelId: probed, reasoning: requestedReasoning, timeoutSeconds: CODEX_REVIEW_TIMEOUT_SECONDS, timeoutBin: await probeTimeoutBin(), repo: scopeRepo })
|
|
2489
2744
|
// R18: an id ANSWERED and this rung still dispatches nothing (a missing or unsafe scope ref).
|
|
2490
2745
|
// The outcome is recorded as a REFUSAL so the belt below cannot report it as a rung that ran.
|
|
2491
2746
|
if (built.cmd === null) { settleUndispatchedStage(announcementOpts, rung, 'refused-before-dispatch', built.reason); return noteCodexDecline(stage, 'tool-error', { exit: 2, detail: built.reason }) }
|
|
2492
2747
|
// R4-F2a: AFTER the command exists. An unusable scope ref (commit/base with a missing or unsafe
|
|
2493
2748
|
// ref) returns cmd:null and dispatches NOTHING — announcing above printed a line for a review that
|
|
2494
2749
|
// never ran. The reason travels from requestedOpts so it is not silently defaulted either.
|
|
2495
|
-
|
|
2750
|
+
const result = await runCodexQeCommand(stage, built.cmd, phaseName, stage + ':codex-review', probed, 'A', built.scope + (scopeRef ? ' ' + scopeRef : ''), declaredFiles, false, requestedReasoning, rung, announcementOpts, scopeRepo !== REPO ? scopeRepo : undefined)
|
|
2751
|
+
// T3 (FR-4, A4): a scoped review's findings carry the scope-repo's ABSOLUTE path — normalize back
|
|
2752
|
+
// to repo-relative so partitionReviewFindings can match them against the declared change set.
|
|
2753
|
+
if (result && scopeRepo !== REPO) return mergeOpts(result, { findings: normalizeScopedFindings(result.findings, scopeRepo) })
|
|
2754
|
+
return result
|
|
2496
2755
|
}
|
|
2497
2756
|
|
|
2498
2757
|
// MODE B — the narrowed follow-up. Carries OUR questions over files we name, and is refused outright
|
|
@@ -3679,6 +3938,7 @@ let coderUsed = null
|
|
|
3679
3938
|
let qe = null
|
|
3680
3939
|
let pipelineRound = null
|
|
3681
3940
|
let roundClosed = false
|
|
3941
|
+
let roundSkippedReason = null
|
|
3682
3942
|
const LEARNED = router ? router.rationale : 'none recalled'
|
|
3683
3943
|
const isMplus = tier === 'M' || tier === 'L' || tier === 'XL'
|
|
3684
3944
|
const isLplus = tier === 'L' || tier === 'XL'
|
|
@@ -4774,6 +5034,31 @@ if (resumedStages.indexOf('code') !== -1 && landedNote !== '') {
|
|
|
4774
5034
|
landedNote = '\n\n[RESUMED from checkpoint — the landing barrier below ran in the ORIGINAL run; the change-manifest artifact was re-verified present by the resume probe]' + landedNote
|
|
4775
5035
|
}
|
|
4776
5036
|
|
|
5037
|
+
// aqe-ledger-row T4/FR-2/FR-3/A5: the `impl` row, written right here — AFTER the Step 7.5
|
|
5038
|
+
// landing barrier so its outcome/coder/landed fields are known, and BEFORE Step 8 QE runs —
|
|
5039
|
+
// so the `qe` row's minutesSincePrev measures the QE step itself, not QE+code combined
|
|
5040
|
+
// (00_complexity_assessment.md: today the row before `qe` is `plan`, which would fold code time in).
|
|
5041
|
+
// fix-round-1/#2 (Codex r1 HIGH #2): the OLD `landingStatus !== null ? landingStatus :
|
|
5042
|
+
// 'skipped-claude-sync'` guessed 'skipped-claude-sync' for ANY null landingStatus — including a
|
|
5043
|
+
// codex coder whose barrier composite never built (codeStage null). Now: a landingStatus from the
|
|
5044
|
+
// KNOWN barrier-verdict set (the same set codeStageResultShapeValid checks — landed /
|
|
5045
|
+
// genuinely-not-landed / inconclusive / 'synchronous', the last reachable only when the barrier was
|
|
5046
|
+
// never NEEDED, i.e. a Claude coder) is used AS-IS; 'skipped-claude-sync' is used ONLY when the
|
|
5047
|
+
// coder's family is genuinely 'claude' AND the barrier was never needed for it
|
|
5048
|
+
// (!needsCodeLandedBarrier); every other case (an unrecognized landingStatus paired with a codex — or
|
|
5049
|
+
// unknown — coder) is the honestly-named 'no-landing-status', which is ALWAYS a string, so — unlike
|
|
5050
|
+
// `undefined` — the field can never silently vanish from the JSON.stringify'd row.
|
|
5051
|
+
if (resumedStages.indexOf('code') === -1) {
|
|
5052
|
+
const knownLandingVerdicts = ['landed', 'genuinely-not-landed', 'inconclusive', 'synchronous']
|
|
5053
|
+
const coderFamilyForImpl = knownFamily(coderUsed)
|
|
5054
|
+
const implOutcome = knownLandingVerdicts.indexOf(landingStatus) !== -1
|
|
5055
|
+
? landingStatus
|
|
5056
|
+
: (coderFamilyForImpl === 'claude' && !needsCodeLandedBarrier(coderUsed)) ? 'skipped-claude-sync' : 'no-landing-status'
|
|
5057
|
+
await appendRunCostRow('impl', 'Code', implOutcome, { coder: coderUsed, coderFamily: coderFamilyForImpl, landed: implOutcome })
|
|
5058
|
+
} else {
|
|
5059
|
+
log('run-cost ledger: impl row skipped — stage resumed')
|
|
5060
|
+
}
|
|
5061
|
+
|
|
4777
5062
|
// Step 8: QE (brutal-honesty, agentic-qe) + MANDATORY teach
|
|
4778
5063
|
phase('QE')
|
|
4779
5064
|
await recordRegistryEvent('heartbeat', 'QE')
|
|
@@ -4861,7 +5146,11 @@ const qe2Spec = qePrecisionPassSpec(PRIMARY, BUDGET_MODE, tier)
|
|
|
4861
5146
|
// re-teach (the original run already stored its lessons — replaying teach would double-store).
|
|
4862
5147
|
// R6: the review SCOPE is part of what a QE verdict is about, so it enters the hash — a resume must
|
|
4863
5148
|
// not present a verdict obtained over one scope as if it had been obtained over another.
|
|
4864
|
-
|
|
5149
|
+
// L1 (fix-round-1, lead e2e 14:53): QE_ISOLATED_SCOPE must enter the hash — without it, flipping
|
|
5150
|
+
// args.qeIsolatedScope between runs of the SAME slug resumes a checkpointed verdict that was
|
|
5151
|
+
// obtained under the OTHER knob value (an --uncommitted shared-tree review standing in for an
|
|
5152
|
+
// isolated-scope one, or vice versa) instead of re-QEing under the new setting.
|
|
5153
|
+
const qeHash = ckptHash('qe', [fnv1a64(JSON.stringify(codeStage === undefined ? null : codeStage)), tier, DESC, QE_REVIEWER, MODELS.qe === undefined ? null : MODELS.qe, CODEX_MODEL, coderUsed, PRIMARY, BUDGET_MODE, qe2Spec, POLY.hasManifest, fnv1a64(String(POLY.report || '')), usageOverride, QE_SCOPE, QE_SCOPE_REF, confirmationFileGate, QE_ISOLATED_SCOPE])
|
|
4865
5154
|
let crossFamilyQeReport = null
|
|
4866
5155
|
const qeStage = await withCheckpoint('qe', 'QE', qeHash, async () => {
|
|
4867
5156
|
let qe = null
|
|
@@ -4895,16 +5184,7 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
|
|
|
4895
5184
|
// assert the very property that is being lost. The honest label for a rung after a dead rung.
|
|
4896
5185
|
let qeCodexReason = stageReason('qe', qeDecision)
|
|
4897
5186
|
if (!qeIsCodex) qeCodexReason = 'fallback-rung'
|
|
4898
|
-
|
|
4899
|
-
// R5-1: the id modelsUsed will report, resolved by the SAME memoized probe the dispatch uses.
|
|
4900
|
-
const qeCodexResolved = await codexLabelOptsForDispatch(qeCodexLabelOpts)
|
|
4901
|
-
qeCodexProbeFailed = !!qeCodexResolved._codexProbeFailed
|
|
4902
|
-
const qeCodexDispatchLabel = modelLabel(qeCodexResolved)
|
|
4903
|
-
// R16-1: the codex QE rung records its attempt HERE. The only dispatch-time write used to sit
|
|
4904
|
-
// inside `if (!qeIsCodex)`, so a codex-routed QE whose modes and belt all returned null printed
|
|
4905
|
-
// `▸ qe` lines and left modelsUsed.qe unassigned entirely. The marker is provisional: the success
|
|
4906
|
-
// paths below overwrite it, and if nothing succeeds it stays and says so.
|
|
4907
|
-
modelsUsed.qe = qeCodexDispatchLabel + (qeCodexProbeFailed ? ' (codex probe found no usable id)' : ' (no verdict)')
|
|
5187
|
+
let qeCodexLabelOpts = mergeOpts(qeModel.agentType ? qeModel : specToOpts('codex:' + CODEX_MODEL + ':high'), { _stage: 'qe', _reason: qeCodexReason })
|
|
4908
5188
|
// ADR-001: QE's deliverable is its RETURN VALUE, so it dispatches SYNCHRONOUSLY, never through the
|
|
4909
5189
|
// fire-and-forget wrapper, and the verdict is PARSED, never synthesised (the deleted
|
|
4910
5190
|
// {grade:'codex-review', gaps: []} turned a stub into a clean review).
|
|
@@ -4927,6 +5207,10 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
|
|
|
4927
5207
|
// commit/base committed work produced no status entry at all. Now the probe is built for the scope
|
|
4928
5208
|
// in force, and for `uncommitted` it is a CONTENT comparison against the pre-code baseline.
|
|
4929
5209
|
let modeBChanged = null
|
|
5210
|
+
// fix-round-1 #4 (HIGH): hoisted so the scope-decision block below (which runs BEFORE the codex
|
|
5211
|
+
// label resolution now) can compare the scope-repo's OWN content receipt against the SAME
|
|
5212
|
+
// measurement, with no second network round-trip.
|
|
5213
|
+
let modeBAfterSnap = null
|
|
4930
5214
|
if (modeBPlanned.length > 0) {
|
|
4931
5215
|
const probeCmd = changeSetProbeCmd({ scope: QE_SCOPE, ref: QE_SCOPE_REF, paths: modeBPlanned, quote: shq })
|
|
4932
5216
|
if (probeCmd === null) {
|
|
@@ -4937,6 +5221,7 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
|
|
|
4937
5221
|
if (QE_SCOPE === 'uncommitted') {
|
|
4938
5222
|
// CONTENT, not dirtiness: only a hash that MOVED since the baseline is this run's doing.
|
|
4939
5223
|
const afterSnap = parseHashProbe(String(chgOut), modeBPlanned)
|
|
5224
|
+
modeBAfterSnap = afterSnap
|
|
4940
5225
|
modeBChanged = changedFromHashes(preCodeBaseline, afterSnap)
|
|
4941
5226
|
if (modeBChanged === null) log('QE: no pre-code baseline — the delta is NOT ESTABLISHED (never treated as empty)')
|
|
4942
5227
|
} else {
|
|
@@ -4945,6 +5230,109 @@ if (qe === null && (qeIsCodex || QE_REVIEWER === 'codex-fallback')) {
|
|
|
4945
5230
|
}
|
|
4946
5231
|
}
|
|
4947
5232
|
}
|
|
5233
|
+
// codex-review-scope (FR-1/FR-2/FR-5, A1/A2/A5): decide BEFORE mode A is ever attempted whether —
|
|
5234
|
+
// and how — it may run for scope 'uncommitted'. Never a bare fallback to --uncommitted on the
|
|
5235
|
+
// shared tree: the decision is recorded on qeCodexLabelOpts (private fields consumed by
|
|
5236
|
+
// codexReviewAgent above) so the call site immediately below stays byte-identical to before this
|
|
5237
|
+
// feature (pinned by cross-family-qe.test.ts / feature-adr-model-routing.test.ts).
|
|
5238
|
+
// fix-round-1 #8 (MEDIUM, Codex r1): this decision block now runs BEFORE codexLabelOptsForDispatch
|
|
5239
|
+
// (moved below) — a scope that is about to be BLOCKED must never spend a codex model probe on a
|
|
5240
|
+
// rung that will refuse before any dispatch anyway. "Refused before any probe" was already true
|
|
5241
|
+
// INSIDE codexReviewAgent; this closes the same gap one call earlier.
|
|
5242
|
+
if (QE_SCOPE === 'uncommitted' && !QE_ISOLATED_SCOPE) {
|
|
5243
|
+
log('QE: isolated scope DISABLED by args (qeIsolatedScope:false) — mode A runs --uncommitted on the shared tree as before')
|
|
5244
|
+
} else if (QE_SCOPE === 'uncommitted') {
|
|
5245
|
+
if (modeBChanged === null) {
|
|
5246
|
+
// A2: NOT ESTABLISHED is never treated as empty, and never silently falls back to --uncommitted.
|
|
5247
|
+
qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'change set NOT ESTABLISHED (no pre-code baseline) — refusing to fall back to --uncommitted on the shared tree' })
|
|
5248
|
+
} else if (modeBChanged.length === 0) {
|
|
5249
|
+
qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'change set established but EMPTY — no files changed' })
|
|
5250
|
+
} else {
|
|
5251
|
+
const scopeDir = FDIR + '/.fa-state/review-scope'
|
|
5252
|
+
// FR-1: BASE_REF = HEAD of the run before Step 7 started (recorded by the coder at Step 7's
|
|
5253
|
+
// preamble, mirror of the mutation-gate's own AM-2 convention); HEAD itself when that record is
|
|
5254
|
+
// absent — valid exactly because scope 'uncommitted' means Step 7 made no commits of its own.
|
|
5255
|
+
const baseRefOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and return its stdout VERBATIM with NO commentary: cd ' + shq(REPO) + ' && if [ -f ' + shq(FDIR + '/.fa-state/base-ref') + ' ]; then cat ' + shq(FDIR + '/.fa-state/base-ref') + ' || echo BASE-REF-READ-FAILED; else echo NO-BASE-REF-FILE; fi', { label: 'qe:review-scope-base-ref', phase: 'QE', effort: 'low' })
|
|
5256
|
+
// Lead delta (Codex r2 #3 partial): the probe no longer folds "no base-ref record" and "read
|
|
5257
|
+
// failed" into a silent `git rev-parse HEAD`. NO-BASE-REF-FILE is the one NAMED fallback to
|
|
5258
|
+
// HEAD (logged); BASE-REF-READ-FAILED and anything unparseable refuse below.
|
|
5259
|
+
const baseRefRaw = String(baseRefOut === null || baseRefOut === undefined ? '' : baseRefOut).trim()
|
|
5260
|
+
if (baseRefRaw === 'NO-BASE-REF-FILE') log('QE: no .fa-state/base-ref record for this run — the isolated scope base falls back to HEAD (named fallback, not a probe failure)')
|
|
5261
|
+
const baseRefCandidate = baseRefRaw === 'NO-BASE-REF-FILE' ? 'HEAD' : baseRefRaw
|
|
5262
|
+
// fix-round-1 #3 (CRITICAL, Codex r1): a probe that DISPATCHED NOTHING (null/undefined) or
|
|
5263
|
+
// returned something unsafe/unparseable used to fall back to 'HEAD' SILENTLY — a probe
|
|
5264
|
+
// failure and a legitimately-absent base-ref record read identically, and codex would then
|
|
5265
|
+
// review a full-file ADDITION instead of the real modification. Refuse outright instead, under
|
|
5266
|
+
// its own taxonomy kind, so this is never mistaken for the healthy default.
|
|
5267
|
+
if (baseRefOut === null || baseRefOut === undefined || baseRefCandidate === '' || !isSafeCodexRef(baseRefCandidate)) {
|
|
5268
|
+
qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'base-ref probe failed or unparseable (dispatch reply: ' + JSON.stringify(String(baseRefOut === null || baseRefOut === undefined ? null : baseRefOut).slice(0, 120)) + ') — refusing rather than silently falling back to HEAD', _scopeBlockedKind: 'base-ref-not-established' })
|
|
5269
|
+
} else {
|
|
5270
|
+
const baseRef = baseRefCandidate
|
|
5271
|
+
const scopeBuilt = reviewScopeRepoScript({ repo: REPO, scopeDir: scopeDir, baseRef: baseRef, files: modeBChanged, quote: shq })
|
|
5272
|
+
if (scopeBuilt.script === null) {
|
|
5273
|
+
qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'scope-build refused: ' + scopeBuilt.reason })
|
|
5274
|
+
} else {
|
|
5275
|
+
const scopeBuildOut = await dispatchAgent(newRung(), 'Run EXACTLY this via Bash and return its stdout VERBATIM with NO commentary: ' + scopeBuilt.script, { label: 'qe:review-scope', phase: 'QE', effort: 'low' })
|
|
5276
|
+
// FR-3 receipt: git status --porcelain in the scope-repo must name EXACTLY the declared
|
|
5277
|
+
// change set — an absent or partial receipt is a build failure, never "close enough"
|
|
5278
|
+
// (feedback-absence-of-receipt-is-not-success).
|
|
5279
|
+
// fix-round-1 #5 (HIGH): porcelain's status column is a FIXED two-character prefix plus one
|
|
5280
|
+
// space (line.slice(3)), never a `\S+\s+` regex — that regex reads a rename `R old -> new`
|
|
5281
|
+
// as the single path "old -> new", losing both real paths (the builder now also passes
|
|
5282
|
+
// --no-renames, so a rename never reaches this parser as a two-path line in the first place).
|
|
5283
|
+
// fix-round-1 #6 (MEDIUM): the path is read WITHOUT trimming — only a trailing '\n'/'\r' is
|
|
5284
|
+
// ever stripped — so a declared path containing a leading/trailing space still round-trips.
|
|
5285
|
+
// fix-round-1 #4 (HIGH): a second table of sha256 hashes is parsed out of the SAME reply
|
|
5286
|
+
// (the builder appends a `sha256sum` step after the porcelain line) and compared to
|
|
5287
|
+
// modeBAfterSnap — a receipt that matches on PATHNAME but not on CONTENT is still a
|
|
5288
|
+
// scope-build-failed, catching a cp that silently degraded to rm -f.
|
|
5289
|
+
const statusText = String(scopeBuildOut === null || scopeBuildOut === undefined ? '' : scopeBuildOut)
|
|
5290
|
+
const gotFiles = new Set()
|
|
5291
|
+
const gotHashes = new Map()
|
|
5292
|
+
for (const rawLine of statusText.split('\n')) {
|
|
5293
|
+
const line = rawLine.replace(/\r$/, '')
|
|
5294
|
+
if (line === '') continue
|
|
5295
|
+
// Lead delta (Codex r2 N2 MEDIUM): sha256sum's format is FIXED — 64 hex, one space, one
|
|
5296
|
+
// mode char (' ' text / '*' binary), then the path byte-for-byte. No trim: a declared
|
|
5297
|
+
// path with a trailing space must key the same way it was declared (r1 #6).
|
|
5298
|
+
const hm = /^([0-9a-f]{64}) [ *](.*)$/.exec(line)
|
|
5299
|
+
if (hm) { gotHashes.set(hm[2].replace(/^\.\//, ''), hm[1]); continue }
|
|
5300
|
+
if (line.length < 4 || line.charAt(2) !== ' ') continue
|
|
5301
|
+
const path = line.slice(3)
|
|
5302
|
+
if (path !== '') gotFiles.add(path)
|
|
5303
|
+
}
|
|
5304
|
+
const wantFiles = new Set(modeBChanged)
|
|
5305
|
+
const receiptOk = gotFiles.size === wantFiles.size && Array.from(wantFiles).every(function (f) { return gotFiles.has(f) })
|
|
5306
|
+
const hashMismatches = modeBAfterSnap === null ? [] : modeBChanged.filter(function (f) {
|
|
5307
|
+
const want = modeBAfterSnap.has(f) ? modeBAfterSnap.get(f) : null
|
|
5308
|
+
const got = gotHashes.has(f) ? gotHashes.get(f) : null
|
|
5309
|
+
return (want === null || want === undefined) ? (got !== null && got !== undefined) : got !== want
|
|
5310
|
+
})
|
|
5311
|
+
if (!receiptOk) {
|
|
5312
|
+
qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'scope-build-failed: git status --porcelain in the scope-repo did not return exactly the declared change set (got ' + gotFiles.size + ', wanted ' + wantFiles.size + ')' })
|
|
5313
|
+
} else if (hashMismatches.length > 0) {
|
|
5314
|
+
qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeBlocked: 'scope-build-failed: content hash mismatch for ' + hashMismatches.join(', ') + ' — the scope-repo working copy does not match the measured content' })
|
|
5315
|
+
} else {
|
|
5316
|
+
qeCodexLabelOpts = mergeOpts(qeCodexLabelOpts, { _scopeRepo: scopeDir, _scopeFiles: modeBChanged })
|
|
5317
|
+
}
|
|
5318
|
+
}
|
|
5319
|
+
}
|
|
5320
|
+
}
|
|
5321
|
+
}
|
|
5322
|
+
// fix-round-1 #8 (MEDIUM, Codex r1): resolve the dispatch LABEL after the scope decision — a
|
|
5323
|
+
// BLOCKED scope skips codexLabelOptsForDispatch entirely (no probe spent on a refusal).
|
|
5324
|
+
let qeCodexResolved = qeCodexLabelOpts
|
|
5325
|
+
if (!qeCodexLabelOpts._scopeBlocked) {
|
|
5326
|
+
// R5-1: the id modelsUsed will report, resolved by the SAME memoized probe the dispatch uses.
|
|
5327
|
+
qeCodexResolved = await codexLabelOptsForDispatch(qeCodexLabelOpts)
|
|
5328
|
+
}
|
|
5329
|
+
qeCodexProbeFailed = !!qeCodexResolved._codexProbeFailed
|
|
5330
|
+
const qeCodexDispatchLabel = modelLabel(qeCodexResolved)
|
|
5331
|
+
// R16-1: the codex QE rung records its attempt HERE. The only dispatch-time write used to sit
|
|
5332
|
+
// inside `if (!qeIsCodex)`, so a codex-routed QE whose modes and belt all returned null printed
|
|
5333
|
+
// `▸ qe` lines and left modelsUsed.qe unassigned entirely. The marker is provisional: the success
|
|
5334
|
+
// paths below overwrite it, and if nothing succeeds it stays and says so.
|
|
5335
|
+
modelsUsed.qe = qeCodexDispatchLabel + (qeCodexLabelOpts._scopeBlocked ? ' (scope not established)' : (qeCodexProbeFailed ? ' (codex probe found no usable id)' : ' (no verdict)'))
|
|
4948
5336
|
const qeRungHolder = newRung()
|
|
4949
5337
|
let qeLastRungHolder = qeRungHolder
|
|
4950
5338
|
let codexQe = await codexReviewAgent('qe', QE_SCOPE, QE_SCOPE_REF, 'QE', qeCodexLabelOpts, qeRungHolder)
|
|
@@ -5118,19 +5506,55 @@ let qeReviewerUsed = qeStage ? qeStage.qeReviewerUsed : 'claude'
|
|
|
5118
5506
|
if (qeStage && qeStage.modelUsed) modelsUsed.qe = qeStage.modelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
|
|
5119
5507
|
if (qeStage && qeStage.qe2ModelUsed) modelsUsed.qe2 = qeStage.qe2ModelUsed + (resumedStages.indexOf('qe') !== -1 ? ' (resumed)' : '')
|
|
5120
5508
|
|
|
5509
|
+
// aqe-ledger-row T3/FR-1/FR-3/A1/A3/A4: the QE step's OWN autorow — who reviewed, how many
|
|
5510
|
+
// findings, cross-family or not — so the instrument's own footprint in the ledger stops being
|
|
5511
|
+
// zero (00_complexity_assessment.md: 0 of 355 rows before this). Guarded by the SAME
|
|
5512
|
+
// resumedStages check every other autorow uses: a resumed QE stage already has its row from the
|
|
5513
|
+
// run that actually did the review, so writing again would double-pay it.
|
|
5514
|
+
if (resumedStages.indexOf('qe') === -1) {
|
|
5515
|
+
// fix-round-1/#1 (Codex r1 HIGH #1): reviewer/reviewerFamily/qeRole now come from the STRICT
|
|
5516
|
+
// reviewerIdentity() classifier (knownFamily()-backed), not tpFamily() — an unrecognized
|
|
5517
|
+
// qeReviewerUsed (undefined, null, a typo) now reads as reviewerFamily:null/qeRole:null, never
|
|
5518
|
+
// guessed as 'claude'. crossFamily is null whenever EITHER family is unknown — comparing a known
|
|
5519
|
+
// family against an unknown one is not a fact, and reporting `false` for it (as the old
|
|
5520
|
+
// `qeReviewerFamily !== tpFamily(coderUsed)` did whenever tpFamily's default silently matched)
|
|
5521
|
+
// would assert a same-family review that was never established.
|
|
5522
|
+
const qeIdentity = reviewerIdentity(qeReviewerUsed, modelsUsed.qe, qe && qe.qeScope ? qe.qeScope.mode : null)
|
|
5523
|
+
const qeCoderFamily = knownFamily(coderUsed)
|
|
5524
|
+
const qeGrade = (qe && typeof qe.grade === 'string' && qe.grade !== '') ? qe.grade : null
|
|
5525
|
+
const qeGradeSource = (qe && typeof qe.gradeSource === 'string' && qe.gradeSource !== '') ? qe.gradeSource : null
|
|
5526
|
+
const qeClaimCheck = (qe && qe.claimCheck && typeof qe.claimCheck === 'object') ? qe.claimCheck : null
|
|
5527
|
+
const qeScopeRow = (qe && qe.qeScope && typeof qe.qeScope === 'object') ? qe.qeScope : null
|
|
5528
|
+
const qeCrossFamily = (qeIdentity.reviewerFamily === null || qeCoderFamily === null) ? null : (qeIdentity.reviewerFamily !== qeCoderFamily)
|
|
5529
|
+
await appendRunCostRow('qe', 'QE', qe ? 'reviewed' : 'no-deliverable', {
|
|
5530
|
+
reviewer: qeIdentity.reviewer,
|
|
5531
|
+
reviewerFamily: qeIdentity.reviewerFamily,
|
|
5532
|
+
qeRole: qeIdentity.qeRole,
|
|
5533
|
+
grade: qeGrade,
|
|
5534
|
+
gradeSource: qeGradeSource,
|
|
5535
|
+
...qeFindingsSummary(qe && qe.gaps),
|
|
5536
|
+
claimCheck: qeClaimCheck,
|
|
5537
|
+
crossFamily: qeCrossFamily,
|
|
5538
|
+
qeScope: qeScopeRow,
|
|
5539
|
+
})
|
|
5540
|
+
} else {
|
|
5541
|
+
log('run-cost ledger: qe row skipped — stage resumed')
|
|
5542
|
+
}
|
|
5543
|
+
|
|
5121
5544
|
// Step 8 has completed its teach/reinforce work and written 08_qe_report.md. Closing telemetry is
|
|
5122
5545
|
// secondary: refusal is loud and reflected in roundClosed, but it never overturns the feature run.
|
|
5123
5546
|
if (qe && pipelineRound !== null) {
|
|
5124
5547
|
const roundGrade = String(qe.grade || '').trim().toUpperCase()
|
|
5125
5548
|
const roundOutcome = ['A', 'A-', 'B+', 'B'].indexOf(roundGrade) !== -1 ? 'shipped' : (['C', 'D'].indexOf(roundGrade) !== -1 ? 'refuted' : null)
|
|
5126
5549
|
if (roundOutcome === null) {
|
|
5127
|
-
log('round close
|
|
5550
|
+
log('round close skipped: non-terminal Step 8 grade ' + roundGrade + ' — the round stays open for the lead')
|
|
5551
|
+
roundSkippedReason = 'non-terminal-grade'
|
|
5128
5552
|
} else {
|
|
5129
5553
|
const roundLessonIds = Array.isArray(qe.roundLessons) ? qe.roundLessons.filter(function (id) { return /^teach:[a-z0-9]+$/i.test(String(id)) }) : []
|
|
5130
5554
|
const roundLearningArg = roundLessonIds.length > 0
|
|
5131
5555
|
? roundLessonIds.map(function (id) { return ' --lesson ' + shq(String(id)) }).join('')
|
|
5132
5556
|
: ' --no-new-knowledge ' + shq((typeof qe.roundNoNewKnowledge === 'string' && qe.roundNoNewKnowledge.trim() !== '') ? qe.roundNoNewKnowledge.trim() : 'Step 8 returned no new teach receipt for this round')
|
|
5133
|
-
const roundCloseCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' round close --slug ' + shq(SLUG) + ' --round ' + pipelineRound + ' --outcome ' + roundOutcome + ' --reason ' + shq('grade ' + roundGrade) + roundLearningArg + ' --project ' + shq(BRAIN) + ' --no-cost --json'
|
|
5557
|
+
const roundCloseCmd = 'cd ' + shq(REPO) + ' && ' + DZ + ' round close --slug ' + shq(SLUG) + ' --round ' + pipelineRound + ' --outcome ' + roundOutcome + ' --reason ' + shq('grade ' + roundGrade) + roundLearningArg + ' --grade ' + shq(roundGrade) + ' --reviewer ' + shq(String(modelsUsed.qe || qeReviewerUsed)) + ' --project ' + shq(BRAIN) + ' --no-cost --json'
|
|
5134
5558
|
const roundCloseOut = await dispatchAgent(newRung(), 'Run EXACTLY this one shell command via your Bash tool and return its stdout VERBATIM, nothing else: ' + roundCloseCmd, { label: 'round:close', phase: 'QE', effort: 'low' })
|
|
5135
5559
|
const roundCloseReceipt = parseRoundCommandJson(roundCloseOut)
|
|
5136
5560
|
roundClosed = !!(roundCloseReceipt && roundCloseReceipt.marker && roundCloseReceipt.row && roundCloseReceipt.row.stage === 'round')
|
|
@@ -5509,6 +5933,7 @@ return {
|
|
|
5509
5933
|
codeWrote: code ? code.wrote : [],
|
|
5510
5934
|
qeGrade: qe ? qe.grade : null,
|
|
5511
5935
|
roundClosed: roundClosed,
|
|
5936
|
+
roundSkippedReason: roundSkippedReason,
|
|
5512
5937
|
score: score,
|
|
5513
5938
|
gaps: qe ? qe.gaps : [],
|
|
5514
5939
|
codeTestsAdequate: qe ? qe.codeTestsAdequate : null,
|