@ccoalm/ccl-skills 0.14.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +10 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +39 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +114 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +36 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +188 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +63 -5
- package/dist/assets/release.json +41 -36
- package/package.json +1 -1
|
@@ -108,6 +108,15 @@ run_gate() {
|
|
|
108
108
|
rc=$?
|
|
109
109
|
set -e
|
|
110
110
|
}
|
|
111
|
+
# Same gate, explicit base: for fixtures whose claim is about WHICH base the
|
|
112
|
+
# verdict was taken against, pinning it beats inferring it from the upstream.
|
|
113
|
+
run_gate_with_base() { # <base ref>
|
|
114
|
+
gate_runs=$((gate_runs + 1))
|
|
115
|
+
set +e
|
|
116
|
+
out="$(env -u ALIAS_AUDIT_CMD CCL_SKILL_BASE_REF="$1" ruby "$GATE_SCRIPT" "$REPO" 2>&1)"
|
|
117
|
+
rc=$?
|
|
118
|
+
set -e
|
|
119
|
+
}
|
|
111
120
|
|
|
112
121
|
# Case 1: an upstream owner's REFERENCE changed with NO impact-chain row -> block.
|
|
113
122
|
new_case case-ref-no-row
|
|
@@ -1194,12 +1203,13 @@ run_gate
|
|
|
1194
1203
|
assert_rc "$rc" 1 "a duplicated-then-dropped row must not vouch for new owner bytes"
|
|
1195
1204
|
assert_contains "impact_chain_gate_missing" "$out" "the net-zero ledger leaves the owner undeclared"
|
|
1196
1205
|
|
|
1197
|
-
# Round scoping 6: the partition's central claim is that
|
|
1198
|
-
#
|
|
1199
|
-
# so a linear-only fixture set proves nothing about the shape
|
|
1200
|
-
# on.
|
|
1201
|
-
#
|
|
1202
|
-
# cases pin that behaviour instead of asserting
|
|
1206
|
+
# Round scoping 6: the partition's central claim is that a MERGED worktree round is
|
|
1207
|
+
# judged exactly as its branch was — and this repo's integration branch is nothing
|
|
1208
|
+
# but merge commits, so a linear-only fixture set proves nothing about the shape
|
|
1209
|
+
# the gate actually runs on. A merge git rebuilds from its parents is expanded
|
|
1210
|
+
# into the branch's own rounds; work committed on the integration line after it
|
|
1211
|
+
# is a round of its own. These two cases pin that behaviour instead of asserting
|
|
1212
|
+
# it in a comment.
|
|
1203
1213
|
new_case case-round-scope-merged-worktree-round
|
|
1204
1214
|
git -C "$REPO" switch -q -c feature-round-merged
|
|
1205
1215
|
printf '\n- Never skip the fixture merged-round rule for this gate.\n' >> "$REPO/skills/platform-observability/SKILL.md"
|
|
@@ -1270,24 +1280,188 @@ run_gate
|
|
|
1270
1280
|
assert_rc "$rc" 0 "a pre-rename round declared against its own round head must pass"
|
|
1271
1281
|
assert_not_contains "impact_chain_evidence_missing_file" "$out" "a row citing the name that existed at its round head is valid evidence"
|
|
1272
1282
|
|
|
1273
|
-
# Round scoping 8: the
|
|
1274
|
-
# append lands BEFORE the owner work
|
|
1275
|
-
#
|
|
1276
|
-
#
|
|
1277
|
-
#
|
|
1278
|
-
#
|
|
1283
|
+
# Round scoping 8: the merged view IS the branch view. On the branch the ledger
|
|
1284
|
+
# append lands BEFORE the owner work, so as a pull request the branch is refused:
|
|
1285
|
+
# the work sits in a rowless later round. Collapsing the merged branch to one
|
|
1286
|
+
# boundary used to accept the same bytes — what a pull request could not land,
|
|
1287
|
+
# its merge could, and the verdict moved after it landed. A merge git rebuilds
|
|
1288
|
+
# from its parents is now expanded into the branch's own rounds, so the two views
|
|
1289
|
+
# agree; both are run on the same fixture so the equality is a checked fact.
|
|
1279
1290
|
new_case case-round-scope-merged-row-before-work
|
|
1280
1291
|
git -C "$REPO" switch -q -c feature-round-row-first
|
|
1292
|
+
git -C "$REPO" branch --set-upstream-to=fixture-base feature-round-row-first >/dev/null 2>&1
|
|
1281
1293
|
printf '| Fixture row-first merged round | `downstream-executor` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; bank-evidence: downscoped:REFSCRIPTS-FIXTURE-NO-BANK; firing-path: file:skills/platform-observability/SKILL.md#Never skip the fixture row-first rule | `updated` | `platform-observability/SKILL.md` merged round |\n' >> "$REGISTER"
|
|
1282
1294
|
routing_surface_downscope
|
|
1283
1295
|
commit_case "worktree round: ledger append first"
|
|
1284
1296
|
printf '\n- Never skip the fixture row-first rule for this gate.\n' >> "$REPO/skills/platform-observability/SKILL.md"
|
|
1285
1297
|
routing_surface_downscope
|
|
1286
1298
|
commit_case "worktree round: owner work after the append"
|
|
1299
|
+
run_gate
|
|
1300
|
+
branch_rc="$rc"
|
|
1301
|
+
assert_rc "$rc" 1 "as a pull request, owner work after the ledger append sits in a rowless round"
|
|
1302
|
+
assert_contains "impact_chain_gate_missing" "$out" "the branch view names the undeclared work"
|
|
1287
1303
|
git -C "$REPO" switch -q case-round-scope-merged-row-before-work
|
|
1288
1304
|
git -C "$REPO" merge -q --no-ff -m "Merge branch 'feature-round-row-first': one worktree round" feature-round-row-first
|
|
1289
1305
|
run_gate
|
|
1290
|
-
assert_rc "$rc"
|
|
1306
|
+
assert_rc "$rc" "$branch_rc" "the merge of that branch is judged exactly as the branch was"
|
|
1307
|
+
assert_contains "impact_chain_gate_missing" "$out" "the merged view names the same undeclared work"
|
|
1308
|
+
|
|
1309
|
+
# Round scoping 13: the observed failure. On the branch, round 1 adds a body rule
|
|
1310
|
+
# with its row and round 2 changes ONLY the description with a `#description`
|
|
1311
|
+
# row — as a pull request that is round scoping 2 and passes. Collapsed to one
|
|
1312
|
+
# boundary the owner's merged diff is body plus description, the routing-surface
|
|
1313
|
+
# class refuses the anchor, and a row that was valid when it landed reads
|
|
1314
|
+
# `impact_chain_firing_path_missing` on every post-merge evaluation (observed on
|
|
1315
|
+
# the integration branch's push build and on the promotion pull request, with
|
|
1316
|
+
# nothing about the row changed). Expanded, round 2 is judged on its own bytes.
|
|
1317
|
+
new_case case-round-scope-merged-description-round
|
|
1318
|
+
git -C "$REPO" switch -q -c feature-round-description
|
|
1319
|
+
git -C "$REPO" branch --set-upstream-to=fixture-base feature-round-description >/dev/null 2>&1
|
|
1320
|
+
printf '\n- Never skip the fixture merged-description body rule for this gate.\n' >> "$REPO/skills/product-rd-workflow/SKILL.md"
|
|
1321
|
+
printf '| Fixture merged body round | `downstream-executor` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; bank-evidence: downscoped:REFSCRIPTS-FIXTURE-NO-BANK; firing-path: file:skills/product-rd-workflow/SKILL.md#Never skip the fixture merged-description body rule | `updated` | `product-rd-workflow/SKILL.md` body round |\n' >> "$REGISTER"
|
|
1322
|
+
routing_surface_downscope
|
|
1323
|
+
commit_case "worktree round 1: body rule added to the entrypoint"
|
|
1324
|
+
perl -0pi -e 's/^(description: .+)$/$1 Fixture merged round-scope routing clause./m' "$REPO/skills/product-rd-workflow/SKILL.md"
|
|
1325
|
+
printf '| Fixture merged description round | `downstream-executor` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; bank-evidence: downscoped:REFSCRIPTS-FIXTURE-NO-BANK; firing-path: file:skills/product-rd-workflow/SKILL.md#description | `updated` | `product-rd-workflow/SKILL.md` description-only round |\n' >> "$REGISTER"
|
|
1326
|
+
routing_surface_downscope
|
|
1327
|
+
commit_case "worktree round 2: description-only edit to the same owner"
|
|
1328
|
+
run_gate
|
|
1329
|
+
branch_rc="$rc"
|
|
1330
|
+
assert_rc "$rc" 0 "as a pull request, the description-only round keeps its locator after the body round"
|
|
1331
|
+
git -C "$REPO" switch -q case-round-scope-merged-description-round
|
|
1332
|
+
git -C "$REPO" merge -q --no-ff -m "Merge branch 'feature-round-description': two worktree rounds" feature-round-description
|
|
1333
|
+
run_gate
|
|
1334
|
+
assert_rc "$rc" "$branch_rc" "the merge of that branch is judged exactly as the branch was"
|
|
1335
|
+
assert_not_contains "impact_chain_firing_path_missing" "$out" "the description anchor binds in its own round after the merge too"
|
|
1336
|
+
|
|
1337
|
+
# Round scoping 14: expansion is bounded by what git can rebuild. A merge whose
|
|
1338
|
+
# tree is not the automatic merge of its parents carries content that came from
|
|
1339
|
+
# neither branch; expanding it would leave that content in no round, so such a
|
|
1340
|
+
# merge keeps the single boundary at the merge. The row-before-work branch from
|
|
1341
|
+
# round scoping 8, merged with an extra owner edit inside the merge commit, is
|
|
1342
|
+
# therefore judged as one span — which accepts it, as it always did, because that
|
|
1343
|
+
# span holds both the row and the work. The review-ledger binder refuses a
|
|
1344
|
+
# non-automatic merge on a promotion chain, so the hand-carried content is still
|
|
1345
|
+
# caught where promotion is decided; this fixture pins only that the expansion
|
|
1346
|
+
# does not silently widen past git's own reconstruction.
|
|
1347
|
+
new_case case-round-scope-hand-resolved-merge
|
|
1348
|
+
git -C "$REPO" switch -q -c feature-round-hand-resolved
|
|
1349
|
+
printf '| Fixture hand-resolved merged round | `downstream-executor` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; bank-evidence: downscoped:REFSCRIPTS-FIXTURE-NO-BANK; firing-path: file:skills/platform-observability/SKILL.md#Never skip the fixture hand-resolved rule | `updated` | `platform-observability/SKILL.md` merged round |\n' >> "$REGISTER"
|
|
1350
|
+
routing_surface_downscope
|
|
1351
|
+
commit_case "worktree round: ledger append first"
|
|
1352
|
+
printf '\n- Never skip the fixture hand-resolved rule for this gate.\n' >> "$REPO/skills/platform-observability/SKILL.md"
|
|
1353
|
+
routing_surface_downscope
|
|
1354
|
+
commit_case "worktree round: owner work after the append"
|
|
1355
|
+
git -C "$REPO" switch -q case-round-scope-hand-resolved-merge
|
|
1356
|
+
git -C "$REPO" merge -q --no-ff --no-commit feature-round-hand-resolved >/dev/null 2>&1
|
|
1357
|
+
printf '\nFixture line carried by the merge commit itself.\n' >> "$REPO/skills/platform-observability/references/platform-observability-playbook.md"
|
|
1358
|
+
commit_case "Merge branch 'feature-round-hand-resolved' with a hand-carried edit"
|
|
1359
|
+
run_gate
|
|
1360
|
+
assert_rc "$rc" 0 "a merge git cannot rebuild from its parents keeps one boundary and is not expanded"
|
|
1361
|
+
|
|
1362
|
+
# Round scoping 15: the depth bound fails closed. Nine nested worktree rounds —
|
|
1363
|
+
# each branch cut from the previous one and merged back in turn, every merge the
|
|
1364
|
+
# automatic one — nest expansions past ROUND_WALK_MAX_DEPTH. The walk must stop
|
|
1365
|
+
# with its own diagnostic rather than fall back to a collapsed span, because a
|
|
1366
|
+
# fallback here would be the lenient verdict on exactly the history an author
|
|
1367
|
+
# could construct to reach it.
|
|
1368
|
+
new_case case-round-scope-nesting-depth
|
|
1369
|
+
prev=case-round-scope-nesting-depth
|
|
1370
|
+
for level in 1 2 3 4 5 6 7 8 9; do
|
|
1371
|
+
git -C "$REPO" switch -q -c "feature-nest-$level" "$prev"
|
|
1372
|
+
prev="feature-nest-$level"
|
|
1373
|
+
done
|
|
1374
|
+
printf '\n- Never skip the fixture nested rule for this gate.\n' >> "$REPO/skills/platform-observability/SKILL.md"
|
|
1375
|
+
printf '| Fixture nested round | `downstream-executor` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; bank-evidence: downscoped:REFSCRIPTS-FIXTURE-NO-BANK; firing-path: file:skills/platform-observability/SKILL.md#Never skip the fixture nested rule | `updated` | `platform-observability/SKILL.md` nested round |\n' >> "$REGISTER"
|
|
1376
|
+
routing_surface_downscope
|
|
1377
|
+
commit_case "deepest worktree round: owner work with its row"
|
|
1378
|
+
for level in 9 8 7 6 5 4 3 2 1; do
|
|
1379
|
+
if [ "$level" -gt 1 ]; then outer="feature-nest-$((level - 1))"; else outer=case-round-scope-nesting-depth; fi
|
|
1380
|
+
git -C "$REPO" switch -q "$outer"
|
|
1381
|
+
git -C "$REPO" merge -q --no-ff -m "Merge branch 'feature-nest-$level'" "feature-nest-$level"
|
|
1382
|
+
done
|
|
1383
|
+
run_gate
|
|
1384
|
+
assert_rc "$rc" 1 "merges nested past the depth bound must fail closed, not fall back to a collapsed span"
|
|
1385
|
+
assert_contains "impact_chain_round_walk_too_deep" "$out" "the depth refusal names itself"
|
|
1386
|
+
|
|
1387
|
+
# Round scoping 16 and 17: git failures inside the expansion decision fail closed.
|
|
1388
|
+
# A `git` shim on PATH delegates everything to the real git except the one
|
|
1389
|
+
# subcommand under test, which it makes fail with an operational status. If the
|
|
1390
|
+
# gate swallowed that status it would silently take the collapsed-span path — the
|
|
1391
|
+
# lenient verdict — so both lookups must abort with impact_chain_git_failed.
|
|
1392
|
+
GIT_REAL="$(command -v git)"
|
|
1393
|
+
SHIM_DIR="$(mktemp -d "${TMPDIR:-/tmp}/icshim.XXXXXX")"
|
|
1394
|
+
make_git_shim() { # <subcommand-to-break> <exit-status>
|
|
1395
|
+
cat > "$SHIM_DIR/git" <<SHIM
|
|
1396
|
+
#!/usr/bin/env bash
|
|
1397
|
+
real="$GIT_REAL"
|
|
1398
|
+
args=("\$@")
|
|
1399
|
+
sub=""
|
|
1400
|
+
i=0
|
|
1401
|
+
while [ \$i -lt \${#args[@]} ]; do
|
|
1402
|
+
case "\${args[\$i]}" in
|
|
1403
|
+
-C) i=\$((i + 2)); continue ;;
|
|
1404
|
+
-*) i=\$((i + 1)); continue ;;
|
|
1405
|
+
*) sub="\${args[\$i]}"; break ;;
|
|
1406
|
+
esac
|
|
1407
|
+
done
|
|
1408
|
+
if [ "\$sub" = "$1" ]; then
|
|
1409
|
+
case "$1" in
|
|
1410
|
+
merge-base) case " \${args[*]} " in *" --is-ancestor "*|*" HEAD "*) exec "\$real" "\$@" ;; esac ;;
|
|
1411
|
+
esac
|
|
1412
|
+
echo "shim: $1 unavailable" >&2
|
|
1413
|
+
exit $2
|
|
1414
|
+
fi
|
|
1415
|
+
exec "\$real" "\$@"
|
|
1416
|
+
SHIM
|
|
1417
|
+
chmod +x "$SHIM_DIR/git"
|
|
1418
|
+
}
|
|
1419
|
+
git -C "$REPO" switch -q case-round-scope-merged-description-round
|
|
1420
|
+
make_git_shim merge-tree 129
|
|
1421
|
+
PATH="$SHIM_DIR:$PATH" run_gate
|
|
1422
|
+
assert_rc "$rc" 1 "an unavailable merge-tree must abort the walk, not skip the expansion"
|
|
1423
|
+
assert_contains "impact_chain_git_failed: git merge-tree --write-tree" "$out" "the refusal names the merge-tree lookup"
|
|
1424
|
+
assert_contains "exited 129" "$out" "the refusal carries the shim status, so it is this lookup and not another git read"
|
|
1425
|
+
make_git_shim merge-base 128
|
|
1426
|
+
PATH="$SHIM_DIR:$PATH" run_gate
|
|
1427
|
+
assert_rc "$rc" 1 "a failing fork-point lookup must abort the walk, not skip the expansion"
|
|
1428
|
+
assert_contains "impact_chain_git_failed: git merge-base" "$out" "the refusal names the fork-point lookup"
|
|
1429
|
+
assert_contains "exited 128" "$out" "the refusal carries the shim status, so it is this lookup (the only two-commit merge-base in the gate) and not the base resolution, which passes HEAD"
|
|
1430
|
+
rm -rf "$SHIM_DIR"
|
|
1431
|
+
|
|
1432
|
+
# Round scoping 18: a sync of the TARGET inside a branch is a sync on that
|
|
1433
|
+
# branch's own line. The target advances under a feature that already carries
|
|
1434
|
+
# owner work; the feature merges the target (automatically) and only then
|
|
1435
|
+
# appends its row. As a pull request the branch passes: the sync brings nothing
|
|
1436
|
+
# new and work plus row share a round. Promoted and judged against the older
|
|
1437
|
+
# base, that sync's second parent is not below the outer base — expanding it
|
|
1438
|
+
# would cut the branch at the sync and strand the work in a rowless span, so the
|
|
1439
|
+
# sync test is made against the line being walked, not the outer base alone.
|
|
1440
|
+
new_case case-round-scope-branch-synced-target
|
|
1441
|
+
git -C "$REPO" switch -q -c feature-synced-target
|
|
1442
|
+
git -C "$REPO" branch --set-upstream-to=case-round-scope-branch-synced-target feature-synced-target >/dev/null 2>&1
|
|
1443
|
+
git -C "$REPO" switch -q case-round-scope-branch-synced-target
|
|
1444
|
+
mkdir -p "$REPO/specs/fixture-refscripts"
|
|
1445
|
+
printf 'Fixture target note unrelated to any owner.\n' >> "$REPO/specs/fixture-refscripts/target-note.md"
|
|
1446
|
+
commit_case "target advances with an unrelated change"
|
|
1447
|
+
SYNC_TARGET_T="$(git -C "$REPO" rev-parse HEAD)"
|
|
1448
|
+
SYNC_TARGET_O="$(git -C "$REPO" rev-parse fixture-base)"
|
|
1449
|
+
git -C "$REPO" switch -q feature-synced-target
|
|
1450
|
+
printf '\n- Never skip the fixture synced-target rule for this gate.\n' >> "$REPO/skills/platform-observability/SKILL.md"
|
|
1451
|
+
routing_surface_downscope
|
|
1452
|
+
commit_case "worktree round: owner work before the sync"
|
|
1453
|
+
git -C "$REPO" merge -q --no-ff -m "Merge branch 'case-round-scope-branch-synced-target' into feature-synced-target" case-round-scope-branch-synced-target
|
|
1454
|
+
printf '| Fixture synced-target round | `downstream-executor` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; bank-evidence: downscoped:REFSCRIPTS-FIXTURE-NO-BANK; firing-path: file:skills/platform-observability/SKILL.md#Never skip the fixture synced-target rule | `updated` | `platform-observability/SKILL.md` synced round |\n' >> "$REGISTER"
|
|
1455
|
+
routing_surface_downscope
|
|
1456
|
+
commit_case "worktree round: its ledger append after the sync"
|
|
1457
|
+
run_gate_with_base "$SYNC_TARGET_T"
|
|
1458
|
+
branch_rc="$rc"
|
|
1459
|
+
assert_rc "$rc" 0 "as a pull request (base: the advanced target), a branch that synced the target keeps its work and row in one round"
|
|
1460
|
+
git -C "$REPO" switch -q case-round-scope-branch-synced-target
|
|
1461
|
+
git -C "$REPO" merge -q --no-ff -m "Merge branch 'feature-synced-target': one worktree round" feature-synced-target
|
|
1462
|
+
run_gate_with_base "$SYNC_TARGET_O"
|
|
1463
|
+
assert_rc "$rc" "$branch_rc" "promoted against the older base (pinned: the commit the target advanced from), the target sync inside the branch is still a sync and the verdict is the branch's"
|
|
1464
|
+
assert_not_contains "impact_chain_gate_missing" "$out" "the branch's pre-sync work is not stranded by expanding the target sync"
|
|
1291
1465
|
|
|
1292
1466
|
# Round scoping 9: a TRANSIENT intermediate rename hop must not launder work.
|
|
1293
1467
|
# X renames to Y (declared), a later round substantively changes Y with no row,
|
|
@@ -1409,6 +1583,6 @@ assert_rc "$rc" 1 "a directory masquerading as SKILL.md must not read as a prese
|
|
|
1409
1583
|
assert_contains "platform-observability/SKILL.md" "$out" "the masqueraded owner must be named"
|
|
1410
1584
|
|
|
1411
1585
|
assert_rc "$full_check_runs" 1 "fixture suite must retain exactly one full checker wiring case"
|
|
1412
|
-
assert_rc "$gate_runs"
|
|
1586
|
+
assert_rc "$gate_runs" 97 "all remaining impact-chain fixtures must run the standalone gate"
|
|
1413
1587
|
|
|
1414
1588
|
echo "test_check_ccl_impact_chain_refscripts: ok"
|
|
@@ -24,6 +24,7 @@
|
|
|
24
24
|
# - test_check_ccl_register_pending_exclusion.sh
|
|
25
25
|
# - test_eval_routing_bank_grader_diagnostics.sh
|
|
26
26
|
# - test_eval_routing_bank_surface_binding.sh
|
|
27
|
+
# - test_eval_routing_bank_resolution.sh
|
|
27
28
|
# - test_eval_routing_prose_target.sh
|
|
28
29
|
# - test_validate_skill_credential_cwd.sh
|
|
29
30
|
# - test_validate_skill_root_depth.sh
|
|
@@ -145,6 +146,7 @@ fast_tests=(
|
|
|
145
146
|
test_impact_chain_round_attribution.sh
|
|
146
147
|
test_eval_routing_bank_grader_diagnostics.sh
|
|
147
148
|
test_eval_routing_bank_surface_binding.sh
|
|
149
|
+
test_eval_routing_bank_resolution.sh
|
|
148
150
|
test_eval_routing_prose_target.sh
|
|
149
151
|
test_validate_skill_credential_cwd.sh
|
|
150
152
|
test_validate_skill_root_depth.sh
|
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Regression test for the bank runner's screening-vs-action resolution signal.
|
|
3
|
+
#
|
|
4
|
+
# A report taken below the action floor locates candidates; it does not license
|
|
5
|
+
# a description edit. The runner says so on stdout and in the JSON report, and
|
|
6
|
+
# this test pins BOTH sides of the number — the reference that states the rule
|
|
7
|
+
# and the executable that enforces it — so they cannot drift apart the way the
|
|
8
|
+
# description-length thresholds once did.
|
|
9
|
+
#
|
|
10
|
+
# Uses a fake `claude` earlier in PATH: never invokes a real CLI or account.
|
|
11
|
+
set -euo pipefail
|
|
12
|
+
|
|
13
|
+
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
14
|
+
EVAL_SCRIPT="$SCRIPT_DIR/eval-routing-bank.rb"
|
|
15
|
+
DOC="$SCRIPT_DIR/../references/eval-routing.md"
|
|
16
|
+
[ -f "$EVAL_SCRIPT" ] || { echo "FAIL: eval script not found: $EVAL_SCRIPT" >&2; exit 1; }
|
|
17
|
+
[ -f "$DOC" ] || { echo "FAIL: reference not found: $DOC" >&2; exit 1; }
|
|
18
|
+
|
|
19
|
+
fail() { echo "FAIL: $*" >&2; exit 1; }
|
|
20
|
+
assert_contains() { case "$2" in *"$1"*) : ;; *) fail "expected output to contain: $1${3:+ ($3)}";; esac; }
|
|
21
|
+
assert_absent() { case "$2" in *"$1"*) fail "expected output NOT to contain: $1${3:+ ($3)}";; *) : ;; esac; }
|
|
22
|
+
|
|
23
|
+
TMP="$(mktemp -d "${TMPDIR:-/tmp}/eval-routing-bank-resolution.XXXXXX")"
|
|
24
|
+
trap 'rm -rf "$TMP"' EXIT
|
|
25
|
+
|
|
26
|
+
REPO="$TMP/repo"
|
|
27
|
+
FAKE_BIN="$TMP/bin"
|
|
28
|
+
mkdir -p "$REPO/skills/testing-strategy" "$REPO/skills/tighten-doc" "$REPO/eval" "$FAKE_BIN"
|
|
29
|
+
|
|
30
|
+
git -C "$TMP" init -q repo
|
|
31
|
+
git -C "$REPO" config user.email test@example.invalid
|
|
32
|
+
git -C "$REPO" config user.name "Test User"
|
|
33
|
+
|
|
34
|
+
cat > "$REPO/skills/testing-strategy/SKILL.md" <<'EOF'
|
|
35
|
+
---
|
|
36
|
+
description: Use when choosing test layers and regression evidence.
|
|
37
|
+
---
|
|
38
|
+
# Testing Strategy
|
|
39
|
+
EOF
|
|
40
|
+
|
|
41
|
+
# Keeps the outcome space larger than {expected, none} so the fixture is not
|
|
42
|
+
# vacuous: a grader that always answered the only skill would pass trivially.
|
|
43
|
+
cat > "$REPO/skills/tighten-doc/SKILL.md" <<'EOF'
|
|
44
|
+
---
|
|
45
|
+
description: Use when polishing document wording.
|
|
46
|
+
---
|
|
47
|
+
# Tighten Doc
|
|
48
|
+
EOF
|
|
49
|
+
|
|
50
|
+
# TWO cases, deliberately: with one case a per-case verdict and a report-wide
|
|
51
|
+
# conjunction are indistinguishable, so every assertion below would also pass
|
|
52
|
+
# against the defect they exist to catch. The degraded fixture fails only the
|
|
53
|
+
# first utterance, which leaves one case short of the floor and one clear of it.
|
|
54
|
+
cat > "$REPO/eval/routing-tasks.jsonl" <<'EOF'
|
|
55
|
+
{"id":"resolution-fixture","utterance":"补一个回归测试","expected_skill":"testing-strategy","frozen_at_sha":""}
|
|
56
|
+
{"id":"resolution-fixture-b","utterance":"帮我润色这段文档","expected_skill":"tighten-doc","frozen_at_sha":""}
|
|
57
|
+
EOF
|
|
58
|
+
|
|
59
|
+
git -C "$REPO" add -A
|
|
60
|
+
git -C "$REPO" commit -qm "fixture"
|
|
61
|
+
|
|
62
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
63
|
+
#!/usr/bin/env bash
|
|
64
|
+
prompt="$(cat)"
|
|
65
|
+
case "$prompt" in
|
|
66
|
+
*润色*) printf '{"selected_skill":"tighten-doc","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n' ;;
|
|
67
|
+
*) printf '{"selected_skill":"testing-strategy","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n' ;;
|
|
68
|
+
esac
|
|
69
|
+
EOF
|
|
70
|
+
chmod +x "$FAKE_BIN/claude"
|
|
71
|
+
export PATH="$FAKE_BIN:$PATH"
|
|
72
|
+
|
|
73
|
+
# --- (1) below the floor: banner printed, report says action_resolution false --
|
|
74
|
+
out_lo="$(ruby "$EVAL_SCRIPT" "$REPO" --replicas 3 --json "$TMP/lo.json" 2>&1)" \
|
|
75
|
+
|| fail "runner exited non-zero below the floor:\n$out_lo"
|
|
76
|
+
assert_contains "screening_resolution_only" "$out_lo" "sub-floor run must announce screening resolution"
|
|
77
|
+
assert_contains "replicas=3" "$out_lo" "banner must name the observed replica count"
|
|
78
|
+
assert_contains "floor 10" "$out_lo" "banner must name the required floor"
|
|
79
|
+
assert_contains "valid observations" "$out_lo" "banner must report the weakest task's valid-observation count, not the request alone"
|
|
80
|
+
grep -q '"action_resolution": false' "$TMP/lo.json" \
|
|
81
|
+
|| fail "sub-floor report must carry action_resolution:false — a consumer cannot read a printed banner"
|
|
82
|
+
|
|
83
|
+
# --- (2) at the floor: no banner, report says action_resolution true -----------
|
|
84
|
+
out_hi="$(ruby "$EVAL_SCRIPT" "$REPO" --replicas 10 --json "$TMP/hi.json" 2>&1)" \
|
|
85
|
+
|| fail "runner exited non-zero at the floor:\n$out_hi"
|
|
86
|
+
assert_absent "screening_resolution_only" "$out_hi" "a run at the floor must not be labelled screening-only"
|
|
87
|
+
grep -q '"action_resolution": true' "$TMP/hi.json" \
|
|
88
|
+
|| fail "at-floor report must carry action_resolution:true"
|
|
89
|
+
|
|
90
|
+
# --- (2b) a nominal at-floor run with an invalid observation is NOT actionable -
|
|
91
|
+
# The floor is on valid observations. A grader that fails one call leaves the
|
|
92
|
+
# task below the floor while `--replicas` still reads 10, and a report that
|
|
93
|
+
# trusted the request would license an edit its evidence cannot support.
|
|
94
|
+
# The runner gives each unparsable verdict ONE retry, treating it as a sampling
|
|
95
|
+
# accident, so a fixture that fails a single call is repaired and records no
|
|
96
|
+
# error. The failure has to persist across the retry to leave the task short of
|
|
97
|
+
# the floor -- which is exactly the real shape this guards: a grader that is
|
|
98
|
+
# reliably unable to answer one utterance, not a stray quote.
|
|
99
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
100
|
+
#!/usr/bin/env bash
|
|
101
|
+
prompt="$(cat)"
|
|
102
|
+
case "$prompt" in
|
|
103
|
+
*润色*) printf '{"selected_skill":"tighten-doc","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'; exit 0 ;;
|
|
104
|
+
esac
|
|
105
|
+
echo x >> "$RESOLUTION_COUNT_FILE"
|
|
106
|
+
n=$(wc -l < "$RESOLUTION_COUNT_FILE" | tr -d ' ')
|
|
107
|
+
if [ "$n" = "3" ] || [ "$n" = "4" ]; then printf 'not json at all\n'; exit 0; fi
|
|
108
|
+
printf '{"selected_skill":"testing-strategy","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'
|
|
109
|
+
EOF
|
|
110
|
+
chmod +x "$FAKE_BIN/claude"
|
|
111
|
+
: > "$TMP/count"
|
|
112
|
+
out_deg="$(RESOLUTION_COUNT_FILE="$TMP/count" ruby "$EVAL_SCRIPT" "$REPO" --replicas 10 --json "$TMP/deg.json" 2>&1)" \
|
|
113
|
+
|| fail "runner exited non-zero on the degraded run:\n$out_deg"
|
|
114
|
+
assert_contains "screening_resolution_only" "$out_deg" "a nominal 10-replica run with an invalid observation must not be actionable"
|
|
115
|
+
grep -q '"action_resolution": false' "$TMP/deg.json" \
|
|
116
|
+
|| fail "a 10-replica run with only 9 valid observations must report action_resolution:false"
|
|
117
|
+
grep -q '"min_valid_observations": 9' "$TMP/deg.json" \
|
|
118
|
+
|| fail "the report must expose the weakest task's valid-observation count"
|
|
119
|
+
|
|
120
|
+
# --- (2c) the verdict is per case, and says nothing about a case not measured --
|
|
121
|
+
# A report-wide boolean alone is unsound in the licensing direction: a subset run
|
|
122
|
+
# over one case would otherwise read as licence for an edit to a case the run
|
|
123
|
+
# never graded. Each result carries its own actionable and valid_observations.
|
|
124
|
+
grep -q '"actionable": true' "$TMP/hi.json" \
|
|
125
|
+
|| fail "an at-floor case must carry its own actionable:true"
|
|
126
|
+
grep -q '"valid_observations": 10' "$TMP/hi.json" \
|
|
127
|
+
|| fail "each case must expose its own valid-observation count"
|
|
128
|
+
grep -q '"action_resolution_scope"' "$TMP/hi.json" \
|
|
129
|
+
|| fail "the report must say its top-level verdict covers only the cases it measured"
|
|
130
|
+
grep -q '"actionable": false' "$TMP/deg.json" \
|
|
131
|
+
|| fail "a case left short by an invalid observation must carry actionable:false"
|
|
132
|
+
# The split is the point: the degraded case is short, the other one is not, and a
|
|
133
|
+
# report-wide conjunction substituted for the per-case field would mark both false.
|
|
134
|
+
python3 - "$TMP/deg.json" <<'PYEOF' || fail "the healthy case must stay actionable while the degraded one does not"
|
|
135
|
+
import json,sys
|
|
136
|
+
d=json.load(open(sys.argv[1]))
|
|
137
|
+
by={r["id"]:r for r in d["results"]}
|
|
138
|
+
ok = (by["resolution-fixture"]["actionable"] is False
|
|
139
|
+
and by["resolution-fixture-b"]["actionable"] is True
|
|
140
|
+
and by["resolution-fixture-b"]["valid_observations"] == 10
|
|
141
|
+
and d["action_resolution"] is False)
|
|
142
|
+
sys.exit(0 if ok else 1)
|
|
143
|
+
PYEOF
|
|
144
|
+
|
|
145
|
+
# --- (2d) shapes where a result could carry no usable verdict at all ----------
|
|
146
|
+
# The resolution loop reads every result's verdict list. Review could not check
|
|
147
|
+
# from the bounded packet that a list is always present, so the shapes that would
|
|
148
|
+
# expose an absent one are pinned here: a single-replica run, and a task whose
|
|
149
|
+
# every replica fails. Neither may crash, and neither may report itself
|
|
150
|
+
# actionable.
|
|
151
|
+
: > "$TMP/count"
|
|
152
|
+
out_one="$(RESOLUTION_COUNT_FILE="$TMP/count" ruby "$EVAL_SCRIPT" "$REPO" --replicas 1 --json "$TMP/one.json" 2>&1)" \
|
|
153
|
+
|| fail "runner exited non-zero on a single-replica run:\n$out_one"
|
|
154
|
+
grep -q '"action_resolution": false' "$TMP/one.json" \
|
|
155
|
+
|| fail "a single-replica run is below the floor and must not be actionable"
|
|
156
|
+
python3 - "$TMP/one.json" <<'PYEOF' || fail "every result of a single-replica run must carry its own verdict list and count"
|
|
157
|
+
import json,sys
|
|
158
|
+
d=json.load(open(sys.argv[1]))
|
|
159
|
+
ok = bool(d["results"]) and all(
|
|
160
|
+
isinstance(r.get("verdicts"), list) and len(r["verdicts"]) == 1
|
|
161
|
+
and r["valid_observations"] <= 1 and r["actionable"] is False
|
|
162
|
+
for r in d["results"])
|
|
163
|
+
sys.exit(0 if ok else 1)
|
|
164
|
+
PYEOF
|
|
165
|
+
|
|
166
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
167
|
+
#!/usr/bin/env bash
|
|
168
|
+
cat >/dev/null
|
|
169
|
+
printf 'never json\n'
|
|
170
|
+
EOF
|
|
171
|
+
chmod +x "$FAKE_BIN/claude"
|
|
172
|
+
# Every task fails every replica: the runner reports the wholly-unmeasured run and
|
|
173
|
+
# exits 3 by its documented contract, and the report must still be well formed.
|
|
174
|
+
# `set -e` would kill the suite on the non-zero exit before it could be read, so
|
|
175
|
+
# the status is captured in the same command that produces it.
|
|
176
|
+
rc=0
|
|
177
|
+
ruby "$EVAL_SCRIPT" "$REPO" --replicas 2 --json "$TMP/allfail.json" >/dev/null 2>&1 || rc=$?
|
|
178
|
+
[ "$rc" = "3" ] || fail "a wholly failed run must exit 3 by the runner's contract, got $rc"
|
|
179
|
+
python3 - "$TMP/allfail.json" <<'PYEOF' || fail "a wholly failed run must still report zero valid observations and no actionable case"
|
|
180
|
+
import json,sys
|
|
181
|
+
d=json.load(open(sys.argv[1]))
|
|
182
|
+
ok = bool(d["results"]) and d["action_resolution"] is False and d["min_valid_observations"] == 0 \
|
|
183
|
+
and all(r["valid_observations"] == 0 and r["actionable"] is False and isinstance(r["verdicts"], list)
|
|
184
|
+
for r in d["results"])
|
|
185
|
+
sys.exit(0 if ok else 1)
|
|
186
|
+
PYEOF
|
|
187
|
+
|
|
188
|
+
# --- (2e) a parseable answer naming no catalog skill is not an observation ------
|
|
189
|
+
# Review could not see from the packet whether a well-formed JSON verdict whose
|
|
190
|
+
# selected_skill is not in the catalog counts toward the floor. It must not: it
|
|
191
|
+
# is absence of evidence, not evidence against the route, so it is an ERROR
|
|
192
|
+
# verdict and the case it belongs to stays short of the floor.
|
|
193
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
194
|
+
#!/usr/bin/env bash
|
|
195
|
+
cat >/dev/null
|
|
196
|
+
printf '{"selected_skill":"skill-that-does-not-exist","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'
|
|
197
|
+
EOF
|
|
198
|
+
chmod +x "$FAKE_BIN/claude"
|
|
199
|
+
rc=0
|
|
200
|
+
ruby "$EVAL_SCRIPT" "$REPO" --replicas 10 --json "$TMP/badsel.json" >/dev/null 2>&1 || rc=$?
|
|
201
|
+
python3 - "$TMP/badsel.json" <<'PYEOF' || fail "a parseable verdict naming a non-catalog skill must count as an error, not a usable observation"
|
|
202
|
+
import json,sys
|
|
203
|
+
d=json.load(open(sys.argv[1]))
|
|
204
|
+
ok = d["action_resolution"] is False and d["min_valid_observations"] == 0 \
|
|
205
|
+
and all(r["valid_observations"] == 0 and r["actionable"] is False for r in d["results"]) \
|
|
206
|
+
and all(v["status"] == "ERROR" for r in d["results"] for v in r["verdicts"])
|
|
207
|
+
sys.exit(0 if ok else 1)
|
|
208
|
+
PYEOF
|
|
209
|
+
|
|
210
|
+
# --- (2f) a skill the prompt never offered is not selectable ---------------------
|
|
211
|
+
# A directory with a SKILL.md but no usable description is filtered out of the
|
|
212
|
+
# prompt catalog. If the selectable-name list were built separately it could still
|
|
213
|
+
# admit that name, and a verdict naming it would count. Both must come from one
|
|
214
|
+
# filtered list, so selecting the unoffered skill is an ERROR like any other
|
|
215
|
+
# non-catalog name.
|
|
216
|
+
mkdir -p "$REPO/skills/ghost-skill"
|
|
217
|
+
printf -- '---\ndescription: ""\n---\n# Ghost\n' > "$REPO/skills/ghost-skill/SKILL.md"
|
|
218
|
+
git -C "$REPO" add -A && git -C "$REPO" commit -qm "ghost skill with empty description"
|
|
219
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
220
|
+
#!/usr/bin/env bash
|
|
221
|
+
cat >/dev/null
|
|
222
|
+
printf '{"selected_skill":"ghost-skill","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'
|
|
223
|
+
EOF
|
|
224
|
+
chmod +x "$FAKE_BIN/claude"
|
|
225
|
+
rc=0
|
|
226
|
+
ruby "$EVAL_SCRIPT" "$REPO" --replicas 10 --json "$TMP/ghost.json" >/dev/null 2>&1 || rc=$?
|
|
227
|
+
python3 - "$TMP/ghost.json" <<'PYEOF' || fail "a verdict naming a skill the prompt never offered must be an error, not a usable observation"
|
|
228
|
+
import json,sys
|
|
229
|
+
d=json.load(open(sys.argv[1]))
|
|
230
|
+
ok = d["action_resolution"] is False and all(v["status"]=="ERROR" for r in d["results"] for v in r["verdicts"])
|
|
231
|
+
sys.exit(0 if ok else 1)
|
|
232
|
+
PYEOF
|
|
233
|
+
|
|
234
|
+
# --- (3) the two sides of the number must agree -------------------------------
|
|
235
|
+
# The rule is only as good as the agreement between the reference that states it
|
|
236
|
+
# and the executable that enforces it. Compare the numbers themselves rather
|
|
237
|
+
# than pinning one literal in two places.
|
|
238
|
+
exe_min="$(grep -oE 'ACTION_RESOLUTION_MIN_REPLICAS = [0-9]+' "$EVAL_SCRIPT" | grep -oE '[0-9]+$' | sort -u)"
|
|
239
|
+
[ -n "$exe_min" ] || fail "could not read ACTION_RESOLUTION_MIN_REPLICAS from $EVAL_SCRIPT"
|
|
240
|
+
[ "$(printf '%s\n' "$exe_min" | wc -l | tr -d ' ')" = "1" ] \
|
|
241
|
+
|| fail "ACTION_RESOLUTION_MIN_REPLICAS is defined with more than one value: $exe_min"
|
|
242
|
+
doc_min="$(grep -oE 'replicas < [0-9]+' "$DOC" | grep -oE '[0-9]+$' | sort -u)"
|
|
243
|
+
[ -n "$doc_min" ] || fail "$DOC does not state the replicas floor as \`replicas < N\`"
|
|
244
|
+
[ "$doc_min" = "$exe_min" ] \
|
|
245
|
+
|| fail "doc states floor $doc_min but the runner enforces $exe_min — they must be one number"
|
|
246
|
+
|
|
247
|
+
# The prose obligation (">=N valid observations before acting") must name the
|
|
248
|
+
# same number too, so a reader following the sentence and a consumer reading the
|
|
249
|
+
# report are held to one bar.
|
|
250
|
+
grep -qF "≥${exe_min} 个有效观测" "$DOC" \
|
|
251
|
+
|| fail "$DOC must state the per-case obligation with the same floor (≥${exe_min} 个有效观测)"
|
|
252
|
+
|
|
253
|
+
echo "PASS: eval-routing-bank resolution signal (banner, report field, doc/executor agreement)"
|
|
@@ -139,28 +139,52 @@ git -C "$REPO_ROOT" show "$BASELINE_GATE_COMMIT:$GATE_PATH" > "$BASELINE_GATE" 2
|
|
|
139
139
|
exit 1
|
|
140
140
|
}
|
|
141
141
|
|
|
142
|
-
# EXPECTED DIVERGENCES
|
|
143
|
-
# rewrite — the repair for a round that merged with this gate red — so the owner
|
|
144
|
-
# change sits below their base and the row cites an owner the range does not
|
|
145
|
-
# touch. The candidate refuses that shape by design and deliberately offers no
|
|
146
|
-
# author-declared escape, so these four historical ranges diverge. The gate is
|
|
147
|
-
# diff-scoped and never re-judges landed history, so nothing operational depends
|
|
148
|
-
# on them; this differential is the only thing that replays them.
|
|
142
|
+
# EXPECTED DIVERGENCES, two named classes, both "newly refused".
|
|
149
143
|
#
|
|
150
|
-
#
|
|
151
|
-
#
|
|
152
|
-
#
|
|
153
|
-
#
|
|
154
|
-
#
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
144
|
+
# `impact_chain_row_vouches_for_unchanged_owner` (four points): landings that
|
|
145
|
+
# back-filled a ledger row by corrective rewrite — the repair for a round that
|
|
146
|
+
# merged with this gate red — so the owner change sits below their base and the
|
|
147
|
+
# row cites an owner the range does not touch. The candidate refuses that shape by
|
|
148
|
+
# design and deliberately offers no author-declared escape.
|
|
149
|
+
#
|
|
150
|
+
# `impact_chain_gate_missing` (six points): merges from before CI judged the
|
|
151
|
+
# branch head (register row on the checkout ref binding, 2026-08-25). The gate now
|
|
152
|
+
# expands a merge git rebuilds from its parents into the branch's own rounds, so a
|
|
153
|
+
# merge is judged exactly as its branch was; these six branches carry owner work
|
|
154
|
+
# outside the round that declares it (a row appended before the work, or work
|
|
155
|
+
# after the last append) and the baseline accepted them only through the
|
|
156
|
+
# collapsed merged view, which no longer exists as a distinct verdict. Three of
|
|
157
|
+
# the six are refused on their own branch head by the baseline gate too; the other
|
|
158
|
+
# three are promotions or syncs whose second parent is the integration branch,
|
|
159
|
+
# where the same shapes sit one merge deeper.
|
|
160
|
+
#
|
|
161
|
+
# The gate is diff-scoped and never re-judges landed history, so nothing
|
|
162
|
+
# operational depends on these points; this differential is the only thing that
|
|
163
|
+
# replays them. Each entry is constrained to ONE direction and ONE diagnostic. A
|
|
164
|
+
# blanket "any mismatch at this SHA is fine" would also swallow the opposite
|
|
165
|
+
# direction — a loosening — which is the failure this whole suite exists to
|
|
166
|
+
# catch. Entries are named individually, never matched by pattern, and an entry
|
|
167
|
+
# that stops diverging is reported as stale rather than tolerated.
|
|
168
|
+
EXPECTED_DIVERGENCES="
|
|
169
|
+
f03b1140f:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
170
|
+
93d09c563:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
171
|
+
9f233728a:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
172
|
+
046612652:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
173
|
+
b9de13869:refused:impact_chain_gate_missing
|
|
174
|
+
8cea35e6d:refused:impact_chain_gate_missing
|
|
175
|
+
95f06b2e6:refused:impact_chain_gate_missing
|
|
176
|
+
c0561c74e:refused:impact_chain_gate_missing
|
|
177
|
+
fad480296:refused:impact_chain_gate_missing
|
|
178
|
+
90ec533e1:refused:impact_chain_gate_missing
|
|
179
|
+
"
|
|
180
|
+
expected_divergence() { # <full sha> <direction> <candidate output>; prints the matched token
|
|
181
|
+
local short="${1:0:9}" direction="$2" out="$3" entry sha dir token
|
|
182
|
+
case "$direction" in "newly refused") direction=refused ;; "newly accepted") direction=accepted ;; esac
|
|
183
|
+
for entry in $EXPECTED_DIVERGENCES; do
|
|
184
|
+
sha="${entry%%:*}"; token="${entry##*:}"; dir="${entry#*:}"; dir="${dir%%:*}"
|
|
185
|
+
[ "$sha" = "$short" ] || continue
|
|
186
|
+
[ "$dir" = "$direction" ] || return 1
|
|
187
|
+
case "$out" in *"$token"*) printf '%s' "$token"; return 0 ;; *) return 1 ;; esac
|
|
164
188
|
done
|
|
165
189
|
return 1
|
|
166
190
|
}
|
|
@@ -170,7 +194,7 @@ expected_divergence() { # <full sha> <direction> <candidate output>
|
|
|
170
194
|
# them, so the exemptions are stale and the run would pass while silently failing
|
|
171
195
|
# the staleness check it never reaches.
|
|
172
196
|
if cmp -s "$BASELINE_GATE" "$CANDIDATE_GATE"; then
|
|
173
|
-
if [ -n "$(printf '%s' "$
|
|
197
|
+
if [ -n "$(printf '%s' "$EXPECTED_DIVERGENCES" | tr -d '[:space:]')" ]; then
|
|
174
198
|
echo "FAIL: baseline and candidate are byte-identical, yet expected divergences are configured" >&2
|
|
175
199
|
echo " identical gates cannot diverge — the exemptions are stale and must be removed" >&2
|
|
176
200
|
exit 1
|
|
@@ -261,8 +285,8 @@ for point in $INTEGRATION_POINTS; do
|
|
|
261
285
|
direction="newly accepted"
|
|
262
286
|
fi
|
|
263
287
|
if [ -n "$direction" ]; then
|
|
264
|
-
if expected_divergence "$point" "$direction" "$candidate_out"; then
|
|
265
|
-
flag=" (expected divergence:
|
|
288
|
+
if matched_token="$(expected_divergence "$point" "$direction" "$candidate_out")"; then
|
|
289
|
+
flag=" (expected divergence: $matched_token, $direction)"
|
|
266
290
|
expected_seen=$((expected_seen + 1))
|
|
267
291
|
else
|
|
268
292
|
flag=" <== VERDICT MISMATCH: $direction"
|
|
@@ -412,7 +436,7 @@ if [ "$total_failures" -gt 0 ]; then
|
|
|
412
436
|
fi
|
|
413
437
|
# An expected divergence that stops diverging means the exemption is stale and
|
|
414
438
|
# should be removed, so it is reported rather than silently tolerated.
|
|
415
|
-
expected_total="$(for e in $
|
|
439
|
+
expected_total="$(for e in $EXPECTED_DIVERGENCES; do echo "$e"; done | wc -l | tr -d ' ')"
|
|
416
440
|
if [ "$expected_seen" != "$expected_total" ]; then
|
|
417
441
|
echo "FAIL: $expected_seen of $expected_total expected divergences actually diverged" >&2
|
|
418
442
|
echo " an exemption that no longer fires is stale — remove it" >&2
|