@ccoalm/ccl-skills 0.15.4 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +15 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +70 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +286 -40
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +177 -89
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +396 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +236 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +3 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +20 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +56 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +68 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +14 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/scenario-testing.md +1 -1
- package/dist/assets/release.json +17 -17
- package/package.json +1 -1
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh
CHANGED
|
@@ -1679,7 +1679,7 @@ diff_alternate = root / "alternate.patch"
|
|
|
1679
1679
|
diff_source.write_bytes(original_diff)
|
|
1680
1680
|
diff_alternate.write_bytes(alternate_diff)
|
|
1681
1681
|
with replace_after_symlink_check(diff_source, diff_alternate):
|
|
1682
|
-
packet_path, digest, _, _ = review_gate.freeze_packet(
|
|
1682
|
+
packet_path, digest, _candidate, _n, _, _ = review_gate.freeze_packet(
|
|
1683
1683
|
SimpleNamespace(
|
|
1684
1684
|
cwd=str(root), diff_file=str(diff_source), base=None, paths=[]
|
|
1685
1685
|
),
|
|
@@ -1774,6 +1774,241 @@ file_input_race_rc=$?
|
|
|
1774
1774
|
check "diff, prior, and completion inputs are read once from a bounded opened descriptor" \
|
|
1775
1775
|
'[ "$file_input_race_rc" = 0 ] && [ "$file_input_race_probe" = open_once_file_inputs_ok ]'
|
|
1776
1776
|
|
|
1777
|
+
# The reviewer's packet and the landing candidate are two objects. A widened
|
|
1778
|
+
# packet exists so a reviewer can judge a claim against code outside the diff;
|
|
1779
|
+
# the candidate exists so the merge-side binder can recompute what actually
|
|
1780
|
+
# lands. Aliasing them made the two mutually exclusive: widening produced a
|
|
1781
|
+
# receipt the binder could never match. These assert the split and the one
|
|
1782
|
+
# invariant that replaces the equality -- the candidate appears in the packet
|
|
1783
|
+
# verbatim, so nothing lands that its reviewer did not read.
|
|
1784
|
+
subject_packet_probe="$(
|
|
1785
|
+
PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY'
|
|
1786
|
+
import hashlib
|
|
1787
|
+
import subprocess
|
|
1788
|
+
import sys
|
|
1789
|
+
import time
|
|
1790
|
+
from pathlib import Path
|
|
1791
|
+
from types import SimpleNamespace
|
|
1792
|
+
|
|
1793
|
+
import review_gate
|
|
1794
|
+
|
|
1795
|
+
root = Path(sys.argv[1]) / "subject-vs-packet"
|
|
1796
|
+
root.mkdir()
|
|
1797
|
+
# Packet files live OUTSIDE the repository on purpose: the base-derived
|
|
1798
|
+
# candidate includes untracked files, so a packet written into the worktree
|
|
1799
|
+
# would become part of the very candidate it has to contain.
|
|
1800
|
+
outside = Path(sys.argv[1]) / "subject-vs-packet-packets"
|
|
1801
|
+
outside.mkdir()
|
|
1802
|
+
|
|
1803
|
+
|
|
1804
|
+
def git(*args):
|
|
1805
|
+
subprocess.run(
|
|
1806
|
+
["git", "-C", str(root), *args],
|
|
1807
|
+
check=True,
|
|
1808
|
+
stdout=subprocess.DEVNULL,
|
|
1809
|
+
stderr=subprocess.DEVNULL,
|
|
1810
|
+
)
|
|
1811
|
+
|
|
1812
|
+
|
|
1813
|
+
git("init", "-q")
|
|
1814
|
+
git("config", "user.email", "fixture@example.invalid")
|
|
1815
|
+
git("config", "user.name", "fixture")
|
|
1816
|
+
(root / "landing.txt").write_text("old\n", encoding="utf-8")
|
|
1817
|
+
(root / "context.txt").write_text("context-base\n", encoding="utf-8")
|
|
1818
|
+
git("add", "landing.txt", "context.txt")
|
|
1819
|
+
git("commit", "-qm", "base")
|
|
1820
|
+
base = subprocess.run(
|
|
1821
|
+
["git", "-C", str(root), "rev-parse", "HEAD"],
|
|
1822
|
+
check=True,
|
|
1823
|
+
capture_output=True,
|
|
1824
|
+
text=True,
|
|
1825
|
+
).stdout.strip()
|
|
1826
|
+
(root / "landing.txt").write_text("new\n", encoding="utf-8")
|
|
1827
|
+
|
|
1828
|
+
|
|
1829
|
+
def freeze(*, diff_file=None, paths=(), base_ref=base, wording=None):
|
|
1830
|
+
return review_gate.freeze_packet(
|
|
1831
|
+
SimpleNamespace(
|
|
1832
|
+
cwd=str(root),
|
|
1833
|
+
diff_file=str(diff_file) if diff_file else None,
|
|
1834
|
+
base=base_ref,
|
|
1835
|
+
paths=list(paths),
|
|
1836
|
+
wording_only_proof_file=wording,
|
|
1837
|
+
),
|
|
1838
|
+
time.monotonic() + 30,
|
|
1839
|
+
)
|
|
1840
|
+
|
|
1841
|
+
|
|
1842
|
+
def digest(value: bytes) -> str:
|
|
1843
|
+
return hashlib.sha256(value).hexdigest()
|
|
1844
|
+
|
|
1845
|
+
|
|
1846
|
+
def expect_refused(label, **kwargs):
|
|
1847
|
+
try:
|
|
1848
|
+
result = freeze(**kwargs)
|
|
1849
|
+
except review_gate.GateError as exc:
|
|
1850
|
+
return exc
|
|
1851
|
+
result[0].unlink()
|
|
1852
|
+
raise AssertionError(label)
|
|
1853
|
+
|
|
1854
|
+
|
|
1855
|
+
# The base-derived subject: exactly what the landing binder recomputes.
|
|
1856
|
+
subject_path, subject_hash, subject_candidate_hash, subject_n, subject_paths, _ = freeze()
|
|
1857
|
+
subject_bytes = subject_path.read_bytes()
|
|
1858
|
+
subject_path.unlink()
|
|
1859
|
+
assert subject_hash == digest(subject_bytes)
|
|
1860
|
+
assert subject_candidate_hash == subject_hash
|
|
1861
|
+
assert subject_n == len(subject_bytes)
|
|
1862
|
+
assert subject_paths == ["landing.txt"], subject_paths
|
|
1863
|
+
|
|
1864
|
+
# A7 -- with no --diff-file the two hashes are the same value, as they are today.
|
|
1865
|
+
plain_path, plain_packet_hash, plain_candidate_hash, _plain_n, plain_paths, _ = freeze()
|
|
1866
|
+
plain_path.unlink()
|
|
1867
|
+
assert plain_packet_hash == subject_hash
|
|
1868
|
+
assert plain_candidate_hash == subject_hash
|
|
1869
|
+
assert plain_paths == ["landing.txt"], plain_paths
|
|
1870
|
+
|
|
1871
|
+
# A1/A2/A3 -- a widened packet carries the whole subject plus context the
|
|
1872
|
+
# reviewer needs. The packet hash is the widened bytes; the candidate hash is
|
|
1873
|
+
# still the base-derived subject the binder will recompute.
|
|
1874
|
+
context = (
|
|
1875
|
+
b"\n--- context: skills/code-review/SKILL.md (unchanged, for judgment) ---\n"
|
|
1876
|
+
b"the sibling clause the changed lines must not contradict\n"
|
|
1877
|
+
)
|
|
1878
|
+
widened = outside / "widened.patch"
|
|
1879
|
+
widened.write_bytes(subject_bytes + context)
|
|
1880
|
+
wide_path, wide_packet_hash, wide_candidate_hash, wide_n, wide_paths, _ = freeze(
|
|
1881
|
+
diff_file=widened
|
|
1882
|
+
)
|
|
1883
|
+
try:
|
|
1884
|
+
assert wide_packet_hash == digest(subject_bytes + context)
|
|
1885
|
+
assert wide_candidate_hash == subject_hash
|
|
1886
|
+
assert wide_packet_hash != wide_candidate_hash
|
|
1887
|
+
# The reviewer is told where the candidate ends; without that, appended
|
|
1888
|
+
# hunks that continue or appear to revert the diff are indistinguishable
|
|
1889
|
+
# from candidate content in a packet-bounded read.
|
|
1890
|
+
assert wide_n == len(subject_bytes), wide_n
|
|
1891
|
+
# A10 -- candidate paths follow the subject, not the packet, so owner
|
|
1892
|
+
# selection and the wording-only changed-file comparison stay bound to what
|
|
1893
|
+
# lands rather than to whatever context was appended.
|
|
1894
|
+
assert wide_paths == ["landing.txt"], wide_paths
|
|
1895
|
+
finally:
|
|
1896
|
+
wide_path.unlink()
|
|
1897
|
+
|
|
1898
|
+
# A5 -- a packet missing part of the candidate is refused. This is the property
|
|
1899
|
+
# the equality used to provide for free.
|
|
1900
|
+
truncated = outside / "truncated.patch"
|
|
1901
|
+
truncated.write_bytes(subject_bytes[: len(subject_bytes) // 2] + context)
|
|
1902
|
+
exc = expect_refused(
|
|
1903
|
+
"a packet missing part of the candidate was accepted", diff_file=truncated
|
|
1904
|
+
)
|
|
1905
|
+
assert "BEGIN" in str(exc), str(exc)
|
|
1906
|
+
|
|
1907
|
+
# A6 -- context appended passes; context spliced into the middle of the
|
|
1908
|
+
# candidate does not, because then the candidate is no longer in the packet
|
|
1909
|
+
# verbatim and no cheap check can tell a splice from a silent edit.
|
|
1910
|
+
split = len(subject_bytes) // 2
|
|
1911
|
+
interleaved = outside / "interleaved.patch"
|
|
1912
|
+
interleaved.write_bytes(subject_bytes[:split] + context + subject_bytes[split:])
|
|
1913
|
+
expect_refused(
|
|
1914
|
+
"a packet interleaving context inside the candidate was accepted",
|
|
1915
|
+
diff_file=interleaved,
|
|
1916
|
+
)
|
|
1917
|
+
|
|
1918
|
+
# A11 -- context BEFORE the candidate is refused even though the candidate is
|
|
1919
|
+
# present verbatim. A bare containment test accepts this, and an adversarial
|
|
1920
|
+
# round showed what it buys: a sanitized decoy diff read as the change while the
|
|
1921
|
+
# real candidate reads as trailing context.
|
|
1922
|
+
prepended = outside / "prepended.patch"
|
|
1923
|
+
prepended.write_bytes(context + subject_bytes)
|
|
1924
|
+
exc = expect_refused(
|
|
1925
|
+
"a packet preceding the candidate with other content was accepted",
|
|
1926
|
+
diff_file=prepended,
|
|
1927
|
+
)
|
|
1928
|
+
assert "BEGIN" in str(exc), str(exc)
|
|
1929
|
+
|
|
1930
|
+
# A4 -- a packet with no relation to the candidate is refused.
|
|
1931
|
+
unrelated = outside / "unrelated.patch"
|
|
1932
|
+
unrelated.write_bytes(b"diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n")
|
|
1933
|
+
expect_refused("an unrelated packet was accepted", diff_file=unrelated)
|
|
1934
|
+
|
|
1935
|
+
# A8 -- --diff-file alone keeps today's meaning: no base, so no subject, and
|
|
1936
|
+
# the candidate hash stays the packet's own hash.
|
|
1937
|
+
alone_path, alone_packet_hash, alone_candidate_hash, _n, _, _ = freeze(
|
|
1938
|
+
diff_file=widened, base_ref=None
|
|
1939
|
+
)
|
|
1940
|
+
alone_path.unlink()
|
|
1941
|
+
assert alone_packet_hash == digest(subject_bytes + context)
|
|
1942
|
+
assert alone_candidate_hash == alone_packet_hash
|
|
1943
|
+
|
|
1944
|
+
# A9 -- the wording-only proof is a machine check over a full-context
|
|
1945
|
+
# base-derived diff and has no meaning over an author-assembled packet.
|
|
1946
|
+
proof = outside / "wording-only.json"
|
|
1947
|
+
proof.write_text("{}", encoding="utf-8")
|
|
1948
|
+
expect_refused(
|
|
1949
|
+
"a wording-only proof was accepted over an author-assembled packet",
|
|
1950
|
+
diff_file=widened,
|
|
1951
|
+
wording=str(proof),
|
|
1952
|
+
)
|
|
1953
|
+
# ... in the COMBINED form. Bare --diff-file with a wording-only proof stays
|
|
1954
|
+
# accepted, which the cases above this block exercise throughout; the boundary
|
|
1955
|
+
# is where a base-derived candidate and author-assembled bytes would both be in
|
|
1956
|
+
# play with nothing saying which one the proof's scope describes.
|
|
1957
|
+
result = freeze(diff_file=widened, base_ref=None, wording=str(proof))
|
|
1958
|
+
result[0].unlink()
|
|
1959
|
+
|
|
1960
|
+
# A12 -- an empty base-derived candidate is refused rather than trivially
|
|
1961
|
+
# satisfying the prefix check, which every packet does for empty bytes.
|
|
1962
|
+
empty_repo = Path(sys.argv[1]) / "empty-candidate"
|
|
1963
|
+
empty_repo.mkdir()
|
|
1964
|
+
subprocess.run(["git", "-C", str(empty_repo), "init", "-q"], check=True)
|
|
1965
|
+
subprocess.run(
|
|
1966
|
+
["git", "-C", str(empty_repo), "config", "user.email", "fixture@example.invalid"],
|
|
1967
|
+
check=True,
|
|
1968
|
+
)
|
|
1969
|
+
subprocess.run(
|
|
1970
|
+
["git", "-C", str(empty_repo), "config", "user.name", "fixture"], check=True
|
|
1971
|
+
)
|
|
1972
|
+
(empty_repo / "kept.txt").write_text("unchanged\n", encoding="utf-8")
|
|
1973
|
+
subprocess.run(
|
|
1974
|
+
["git", "-C", str(empty_repo), "add", "kept.txt"],
|
|
1975
|
+
check=True,
|
|
1976
|
+
stdout=subprocess.DEVNULL,
|
|
1977
|
+
)
|
|
1978
|
+
subprocess.run(
|
|
1979
|
+
["git", "-C", str(empty_repo), "commit", "-qm", "base"],
|
|
1980
|
+
check=True,
|
|
1981
|
+
stdout=subprocess.DEVNULL,
|
|
1982
|
+
)
|
|
1983
|
+
empty_base = subprocess.run(
|
|
1984
|
+
["git", "-C", str(empty_repo), "rev-parse", "HEAD"],
|
|
1985
|
+
check=True,
|
|
1986
|
+
capture_output=True,
|
|
1987
|
+
text=True,
|
|
1988
|
+
).stdout.strip()
|
|
1989
|
+
try:
|
|
1990
|
+
review_gate.freeze_packet(
|
|
1991
|
+
SimpleNamespace(
|
|
1992
|
+
cwd=str(empty_repo),
|
|
1993
|
+
diff_file=str(widened),
|
|
1994
|
+
base=empty_base,
|
|
1995
|
+
paths=[],
|
|
1996
|
+
wording_only_proof_file=None,
|
|
1997
|
+
),
|
|
1998
|
+
time.monotonic() + 30,
|
|
1999
|
+
)
|
|
2000
|
+
except review_gate.GateError as exc:
|
|
2001
|
+
assert exc.reason_code == "empty_diff", exc.reason_code
|
|
2002
|
+
else:
|
|
2003
|
+
raise AssertionError("an empty base-derived candidate was accepted")
|
|
2004
|
+
|
|
2005
|
+
print("subject_packet_split_ok")
|
|
2006
|
+
PY
|
|
2007
|
+
)"
|
|
2008
|
+
subject_packet_rc=$?
|
|
2009
|
+
check "a widened packet keeps the base-derived candidate and must contain it verbatim" \
|
|
2010
|
+
'[ "$subject_packet_rc" = 0 ] && [ "$subject_packet_probe" = subject_packet_split_ok ]'
|
|
2011
|
+
|
|
1777
2012
|
reset_case missing_coverage passed unavailable
|
|
1778
2013
|
out="$(run_gate --diff-file "$WORK/secret-diff.patch")"; rc=$?
|
|
1779
2014
|
check "missing coverage cannot widen egress for a secret-bearing diff without approval" \
|
|
@@ -194,7 +194,7 @@ Run this gate before finalizing a product R&D turn after any delivery slice land
|
|
|
194
194
|
- **Deferred-evidence continuation check (`DFE-CONT`).** When real/runtime evidence is due (named by an acceptance item, status source, landing-evidence row, required gate, user correction, or because it is the behavior's only meaningful proof) yet deferred, blocked after remediation, skipped at finalization, or replaced by local/mock verification. Report deferred real evidence as `interim`/outstanding; do NOT report the turn complete while it is outstanding. A local/mock substitution is terminal only when a cited **non-agent** anchor — **agent-authored or agent-co-edited status/router/gate/handoff text never satisfies this** — names the same evidence, declares the deferral terminal, and carries the outstanding command/source forward for the active slice/ref. Never add verifier/config/test hardening motivated only by missing deferred evidence; never auto-continue past the pending gate. **Load `references/pre-final-continuation-gate.md` before treating any deferral as terminal** — it owns the valid/invalid-anchor list and hardening boundary.
|
|
195
195
|
- **Affirmative-assent binding rule** lives in `references/pre-final-continuation-gate.md` §Assent binding — load it when recovering a short reply. Bind to the current explicit request or one recoverable concrete proposal, including an unmarked proposal; preserve its scope and existing authority. Ask only if action, scope, or required authority remains unresolved after recovery. A status remark or output marker cannot substitute for a proposal or permission; self-classifying the reply or marker away is never an exit from carrying out an already-clear request.
|
|
196
196
|
3. Continue automatically with a clearly owned, verifiable, low-risk next slice from an explicit task/status/acceptance source or active user continuation, within accepted scope and existing authority. Apply the eligibility and stop conditions in `references/pre-final-continuation-gate.md`. Necessary fixes, tests and review inherit task authorization; a reviewer-budget flag triggers a method checkpoint and cumulative-history record, not renewed permission. Explicit user limits still govern. Existing configured internal developer-self-use metered model/tool accounts aren't an external purchase here.
|
|
197
|
-
4. Stop only for an explicit stop/pause instruction, a user-requested status-only answer,
|
|
197
|
+
4. Stop only for an explicit stop/pause instruction, a user-requested status-only answer, a concrete blocker for the affected action, or no safe authorized work remains. Block materially differing viable approaches (none dominant-and-reversible) and a fix lacking evidenced cause; load `references/pre-final-continuation-gate.md` for the full stop conditions. **Scope each blocker to its dependent action or claim.** An unproven cause blocks the speculative patch, not available diagnosis; a pending gate blocks dependent landing/completion, not authorized remediation or independent work. Before ending, perform in-scope diagnosis, owner discovery, remediation or independent work, and poll any finite step you started to its result, never reporting it as running. Quality-gate failures require diagnosis and available related behavior-preserving cleanup before escalation; preserve readability and compatibility, never game counters (`references/refactoring-discipline.md`). Never bypass the blocked gate, invent a pass, widen scope, or substitute unrelated hardening. With one dominant reversible approach and no applicable stop condition, do not stop at a recommendation: deliver a tested reviewable draft.
|
|
198
198
|
5. If stopping, state the concrete stop reason and the exact evidence checked; an assent-triggered `blocked:` outcome uses the action/scope-plus-blocker form and classifies the turn `interim`. Ask one concise in-turn question when ambiguity or missing authority blocks; explicit stop/pause needs no reconfirmation. A `continuing:` outcome proceeds with the named slice before finalizing. A silent/completion stop is invalid. Do not send a completion-only, solved, fixed, or fully-closed final response after a merge/sync while a required review/challenge is pending or inconclusive; report interim or blocked with the next unblock step.
|
|
199
199
|
6. **Assent-outcome closeout check.** Every user reply immediately following an assistant message that states or implies a next action requires a visible `continuing:` or `blocked:` outcome before finalizing, even if the reply is not classified as assent; every explicit continuation request does too. Missing markers never waive it. Reconcile the current request, original proposal, scope/authority changes, tool/output evidence, and remaining blockers. Respect a current explicit stop or status-only request; name that reason in the blocked outcome without executing the prior proposal. Otherwise `continuing:` must be followed by execution in the same turn; a promised next step is not execution. If part remains blocked, report its pending state and independent work performed. A status-only handoff cannot discharge an unexecuted accepted action. Repair marker formatting; for short assent, if the original proposal cannot be recovered verbatim, select `blocked:` and ask. Formatting never requires clarification. Do not silently drop an accepted action or claim a pending gate passed.
|
|
200
200
|
|
|
@@ -101,12 +101,15 @@ An eligible next slice comes from an explicit status/task/acceptance source or a
|
|
|
101
101
|
|
|
102
102
|
Action-scoped stop conditions are: an explicit stop/pause instruction; a user-requested status-only answer; a failed, pending or inconclusive required gate; a dirty/conflicting worktree that cannot be isolated; a required environment unavailable after remediation; a high-impact product, architecture or compliance decision; a destructive action; an external purchase or financial commitment; unclear ownership; ambiguous assent; missing stricter authorization; materially different viable approaches with none dominant and reversible; a speculative fix without evidenced cause; or no low-risk slice. Apply each condition to the affected action. For a failed check, perform available authorized diagnosis and remediation before stopping the whole task: cite the failure output, repair attempts (or evidence that repair is unsafe or outside authority), and residual blocker. A failed verdict alone does not block diagnosis.
|
|
103
103
|
|
|
104
|
+
**Awaiting work you started yourself is not a stop condition.** A finite command, suite, gate, or review you launched, whose result only you consume, is in-flight work rather than a handoff: wait for it and continue in the same turn. A process meant to stay up — a dev server, a watch-mode runner, a tail — has no terminal result to wait for: take its readiness signal and proceed. Never poll it forever, and do not infer anything about its lifetime from this rule; whether it keeps running is the delivery's decision, and a service the user asked for is a deliverable, not a leftover. Ending the turn to report that it is running is a premature stop even when the report is accurate — the user gains nothing they can act on, and the next step was already authorized. Host behavior invites this: a backgrounded step returns control immediately, so the pause *looks* like a turn boundary. It is not one. Before ending any turn, name the next action; if you can perform it now, the turn is not over. The turn ends at the first action that genuinely needs the user — an unresolved decision, missing authority, an explicit stop — not at the nearest convenient pause. A user asking why you stopped is this defect's recurrence signal, not a request for a status update.
|
|
105
|
+
|
|
104
106
|
Check continuation on every user reply immediately following assistant prose that states or implies a next action, and on any explicit continuation request, regardless of landing status. Do not first require classifying the reply as assent; visibly report the continuing or blocked outcome even when the reply changes scope or stops the proposed action. Short replies include `ok`, `yes`, `可以`, `好`, `继续`, `proceed`, `do it`, `go ahead`, and `👍`; interpret them against the recovered action rather than formatting alone.
|
|
105
107
|
|
|
106
108
|
- Select `continuing: <action and scope>` when that action is clear and authorized, then execute it in the same turn. A tool call and its result or a produced artifact establish execution; the label alone does not.
|
|
107
109
|
- A blocked patch, review, or landing does not block every action. Keep that dependent action/claim pending while continuing available diagnosis, bounded remediation, monitoring of the existing live handle, or independent accepted work. These paths retain their own scope and permission checks; they cannot bypass the blocked gate or substitute unrelated hardening for missing evidence.
|
|
108
110
|
- A failed quality gate calls for a repair that preserves its purpose. Before asking the user to choose a workaround, inspect and perform a safe structural cleanup necessary for the authorized delivery when available, including baseline failures that block it, then rerun the gate and affected tests. Follow [refactoring discipline](refactoring-discipline.md#responding-to-quality-gates): preserve behavior, compatibility and readability; do not shrink identifiers or necessary comments, weaken a baseline or rewrite history solely to make the counter pass. If no safe in-scope repair remains, report the evidence and the actual decision needed.
|
|
109
111
|
- Independent work must neither depend on the pending verdict nor modify the candidate being evaluated. Name the pending gate and the independence basis when continuing. A candidate-changing fix is remediation, not independent work: let the existing run reach a terminal state, then refresh affected evidence and re-enter the owning gate. The deferred-evidence hardening prohibition still applies.
|
|
112
|
+
- A self-initiated step still running is `continuing:`, never `blocked:` and never a final response. Poll it to a terminal result, act on that result, and only then re-enter this gate. A step whose terminal result cannot be obtained after the normal remediation — it hangs, or its handle is lost — is the ordinary unavailable-environment case and blocks that dependent action, with the failure and the remediation attempted cited.
|
|
110
113
|
- Select `blocked: <action and scope> — <specific blocker>` when the remaining action needs an unresolved decision/authority or no safe authorized work remains after remediation. Cite the actual evidence; ask only for the missing decision or permission. An explicit stop/pause or status-only request blocks executing the prior proposal: name that reason in the outcome, answer the requested status, and do not reconfirm the stop.
|
|
111
114
|
- Apply landing-state proof to landing claims and derivation of post-landing work. For an authorized local investigation with no landed slice, record that landing checks do not apply and perform the investigation.
|
|
112
115
|
|
|
@@ -659,3 +659,23 @@ The pending classification above is superseded by the executed source comparison
|
|
|
659
659
|
| Required failures inherited from a baseline remain part of an authorized repair task | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/defect-diagnosis/SKILL.md#Required failures, including inherited debt | updated | Owner key `defect-diagnosis/SKILL.md`. The entry now requires diagnosis, safe repair and rerunning the original check, with a mandatory handoff reference. A synthetic deletion of the new entry makes its named retention assertion fail in the shared implementation-retention fixture; the unchanged control passes. Advisory cases F35-F38 in `eval/behavior-fixtures.jsonl` distinguish inherited blockers, unsafe repair, status-only and diagnosis scope. These checks establish text retention and reviewable scenarios, not a measured increase in autonomous delivery. |
|
|
660
660
|
| Delivery blockers require bounded repair evidence before an exception or blocked handoff | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/refactoring-discipline.md#Treat a required-check failure that blocks this delivery as work to resolve | updated | Owner key `product-rd-workflow/SKILL.md`. The quality-gate response and `skills/product-rd-workflow/references/pre-final-continuation-gate.md` preserve the check purpose, require repair attempts or evidence of an unsafe or unauthorized repair, and leave optional findings separate. Deleting each added retention predicate makes its own assertion fail in the shared implementation-retention fixture; controls pass. Earlier explicit-context task replay already chose repair, so the change makes the inherited-blocker and handoff rules explicit without claiming a demonstrated task-level improvement. |
|
|
661
661
|
| Retention checks must include every input surface when executed from an isolated fixture | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. New repair and goal-authorization assertions in `skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh` read the root contract and release documents. The prior isolated copy omitted those inputs and failed its clean control; copying them restores the 52-mutation controlled-escalation walk. Eleven new repair and authorization predicates also fail under individual deletion, with passing controls. Goal-authorized delivery remains bounded by the requested target, caller authority, explicit stop or narrow scope, and actual host enforcement. Advisory cases F39-F40 cover release continuation and unrelated protected actions; static pins do not implement a permission system or prove agent compliance. |
|
|
662
|
+
| A packet-only reviewer runs from a private home that never carried the user's MCP servers, rather than disabling them by name | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | updated | Owner key `code-review/SKILL.md`. Two faults on codex-cli 0.153.4. Every frozen-packet tool call was refused before it ran: `approval_policy="never"` with a sandbox lacking full disk write access leaves no auto-approve branch, so the packet server now declares `default_tools_approval_mode="approve"` for itself while the sandbox and policy stay unchanged. Separately, the preflight disabled every other server by name; a plugin contributes its server outside `mcp_servers`, so that override builds a transportless entry the CLI rejects outright, while leaving it enabled failed an exactly-one-server count -- the lane could not run at all on such a host. Removing the enumeration made the lane usable and made foreign servers reachable: measured, the host's own `node_repl` ran its `js` tool to completion during a packet-only review, and a stub was auto-approved purely by declaring `readOnlyHint`, which the CLI trusts from the server itself. Global approval-mode defaults did not override that hint and disabling plugins would disable the reviewer's own registry, so the run now gets a private `CODEX_HOME`: linked credential, carried model preference, copied owner skills, nothing else. Two draft claims are withdrawn rather than edited away -- that the parser's after-the-fact audit contained a foreign call, and that a hostile diff was a demonstrated path to one (two attempts did not reproduce it). `origin/dev` is not the unsafe baseline: its per-name disable works for config-declared servers and fails only for plugin-contributed ones. RED-baseline: seven wrapper assertions fail against `origin/dev`'s wrapper and the private-home assertion fails against the mid-round one; all pass on the final candidate. |
|
|
663
|
+
| Work the agent started itself and still awaits is not a stop condition: it polls that step to a terminal result and continues in the same turn | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/pre-final-continuation-gate.md#A self-initiated step still running is | updated | Owner key `product-rd-workflow/SKILL.md`. Observed twice in one session: after launching a test suite or gate whose result only it would consume, the agent ended the turn to report that the step was running; the user had to ask why it stopped, then named the stopping itself as the defect. The gate already said not to stop at a recommendation and to continue with an owned low-risk slice, so content was not the gap -- the stop-condition list simply did not name this shape, and the host returns control the moment a step is backgrounded, which makes the pause look like a turn boundary. Landed as a firing mechanism rather than a discipline reminder: the reference names awaiting a FINITE self-started step as a non-condition, requires naming the next action before any turn ends, and adds an outcome-contract line making a still-running self-initiated step `continuing:` rather than `blocked:` or a final response; the entrypoint carries the same clause so the rule fires without opening the reference. Independent review caught the first wording as an over-broad absolute -- a dev server or watch-mode runner has no terminal result, so the rule would have demanded indefinite polling; it now takes a readiness signal and says nothing about the process lifetime: a later challenge showed that shutting it down at closeout destroys a service that is itself the requested deliverable, so the clause stops prescribing what it does not own. RED-baseline (applied, differential): deleting each of the four clauses reds only its own assertion in the shared implementation-gates fixture (`test_ai_coding_implementation_gates.sh`) with no other assertion failing, and the unmutated control passes. |
|
|
664
|
+
| The shared implementation-gates fixture pins the continuation gate's non-stop clauses, so a later edit cannot silently delete them | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. The sibling row for `product-rd-workflow` records the failure itself; this row records why the fix cannot regress silently. Four assertions were added to the fixture -- the reference's non-stop clause, its turn-end firing check, its outcome-contract line, and the entrypoint's own clause. The fourth was added after independent review observed that the outcome-contract line could be deleted with every assertion still green, which is the same false-green shape the pins exist to prevent. RED-baseline (applied, differential): deleting each protected sentence reds only its owning assertion, with every other assertion passing and the unmutated control clean, so a partial deletion is attributable rather than lost in an aggregate failure. The fixture was chosen over a new suite because it already owns cross-owner rule-retention pins; no new registration surface is introduced. |
|
|
665
|
+
| The landing binder names the ordering cause at the failure point: evidence a round adds that stays inside the candidate is listed when nothing binds | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/review_ledger_binding.py | updated | Owner key `skill-extraction-workflow/SKILL.md`. Third occurrence of one class. The rule that bound evidence is committed before the review rounds already exists verbatim in the quickstart and already carries a register row marked observed twice in consecutive rounds; this round hit it again because the round was driven from the delivery and review owners and never opened that quickstart. Two prior landings answered the recurrence with more prose, so this one changes the mechanism instead: when nothing binds, the binder enumerates the added evidence that is NOT excluded -- the complement of the receipt exclusion it already computes -- and states that only added JSON carrying a candidate_sha256 is excluded, so committing a base attestation or excerpt after the rounds moves the candidate out from under their receipts. RED-baseline (applied): on this round's own failing candidate the pre-change binder reported only that nothing bound it, naming neither the file nor the ordering; the changed binder lists `landing-base.txt` and the round's markdown dispositions and states the ordering. The five binding suites pass unchanged. The diagnosis now reaches an agent at the moment it fails rather than requiring it to know which document to open. |
|
|
666
|
+
| A harness whose RECORDS are the evidence — an evaluation or benchmark runner, a conformance suite feeding a comparison, an A/B or regression rig — can be corrupted by the data it produces in three ways that all read green: absence stored as a bare null cannot separate confirmed-absent from never-observed, planned units and retries sharing one counter let a retry move the denominator, and a later attempt overwrites an earlier failure. Its own record layer is a high-risk failure class of the same standing as the canonical list, and is built against these before the happy path | `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/testing-strategy/references/ci-fixtures-and-flake-control.md#Absence carries a coded reason beside the value | updated | `testing-strategy/SKILL.md` is the owner key and is unchanged this round: the entrypoint is over its size budget and the growth gate blocks it, so the rule lands in `testing-strategy/references/ci-fixtures-and-flake-control.md` and is reached as a failure class from the canonical high-risk list in `testing-strategy/references/scenario-testing.md`, an enumeration the entrypoint already tells readers to walk. RED baseline: a paired walk over real artifacts — a held-out harness that contributed nothing to deriving the rules fails all three rows, each defect named by exactly one row while the other two do not mention it, while the control harness passes two and partially satisfies the first, so the check discriminates rather than accepting whatever is put to it. `observed-failure` is `no` deliberately: no malfunction of an existing repository rule was recorded this round, and the delta is measured against the held-out artifact rather than against a regression this repository observed; `result-class` is `failure` because that held-out artifact does exhibit all three defects the rule names. Sources read this round: the health-interchange data-absent-reason code system, a monitoring query language's absent-vector operators, and the controlled-trial reporting guidance for the flow diagram and per-group denominators. Known limit, stated in the landed text itself: the assembled rule has no located prior name, and measurement system analysis is the adjacent established field covering instrument accuracy and repeatability rather than record integrity. |
|
|
667
|
+
| The reviewer's packet and the landing candidate are two objects: `--diff-file` widens what the reviewer reads while `--base` keeps `candidate_sha256` the base-derived identity the merge-side binder recomputes, and the gate accepts the combination only when the packet BEGINS with that candidate, byte for byte | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/review_gate.py | `updated` | Owner key `code-review/SKILL.md`. Observed failure: the previous round's first review chain returned five findings of one class -- the code a containment claim depends on is not in the packet -- and answering them was impossible, because the skill tells the author that insufficient input is an input defect to be answered by widening the packet and rerunning the lane, while `freeze_packet` refused `--diff-file` together with `--base` and recorded `candidate_sha256` as the packet's own hash, so a widened rerun produced a receipt `review_ledger_binding.py` can never match. Following the contract produced evidence the merge side rejects. The controller now computes the candidate from the base independently of the packet, accepts the combination only when the packet begins with that candidate, and derives `candidate_paths` from the candidate so owner selection and the wording-only `changed_files` comparison cannot be widened by appended context; the in-chain checks that asserted the two hashes were equal now compare candidate to candidate, which is what lets a later round in one chain read more than an earlier one. Backward compatibility is byte-exact: with no `--diff-file` both hashes keep today's value, and `--diff-file` alone keeps today's meaning. The anchor is a prefix rather than a bare containment test because the round's adversarial challenge showed what containment alone buys: a packet PRECEDING the candidate with a sanitized decoy diff passes, and the reviewer then reads the decoy as the change and the real candidate as trailing context -- the repository's authoring rule already said context sits on top of the candidate, and until this round it was documented and unenforced. RED-baseline (applied, differential): collapsing the candidate identity back into the packet hash reds only the new acceptance case in `test_review_gate.sh` while its other 268 cases pass; removing the anchor check reds that same case alone; the unmutated control is green. Recorded because it cost a red rather than being reasoned out: the packet file must live outside the repository, because the base-derived candidate includes untracked files and a packet written into the worktree becomes part of the candidate it has to contain. |
|
|
668
|
+
| The binding gate's cross-side agreement is asserted rather than assumed: a widened packet's recorded candidate identity must equal the one the gate recomputes, and the base and paths for that comparison come from the gate's own scope resolution rather than being rebuilt by hand | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. `review_ledger_binding.py` has no behavior delta of its own -- it never passes `--diff-file`, so its candidate and its packet stay the same bytes -- and the honest record of that is a mutation which does NOT discriminate: making it return the packet hash instead of the candidate hash leaves all 85 binder cases green. So what lands here is the cross-side assertion, because the claim that motivated the change spans both sides and neither suite alone can hold it. The binder suite now freezes a widened packet through the controller and requires the identity recorded for it to equal the one this gate recomputes. It derives the base and the paths from `candidate_scope` instead of hand-building them, which is itself the finding: the base is a fork point and the paths carry the receipt exclusions the round has already added, so `--print-candidate` is the authority an author reads rather than a value an author reconstructs. RED-baseline (applied, differential): collapsing the controller's candidate identity into the packet hash reds this case alone while the other 85 pass, and the unmutated control is green. |
|
|
669
|
+
| A reviewer wrapper classifies a transport failure from the transport's own error channel, not from whichever stream is habitual: a CLI run under a structured-output flag reports supply and credential failures as events on stdout, so a classifier reading stderr alone reports a routine quota exhaustion as an unclassifiable client fault and stops the lane instead of cascading; and because only the transport's TOP-LEVEL error events count, model-authored text quoting the reviewed packet cannot steer that decision. Every transport failure additionally carries a bounded, redacted excerpt of what the transport said, because a failure whose captured streams are deleted leaves nothing that can contradict a wrong hypothesis | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Observed failure: the codex lane returned `codex_run_failed` / `unknown_client_failure` three rounds running, and the recorded suspicion was packet size. Both halves were wrong. A 20KB packet reproduces the failure identically, and the preserved run directory shows the account's usage limit announced on the event stream while stderr carried only an unrelated models-manager message -- the quota regex matches the first file and not the second, and the wrapper greps only the second. Invoking the CLI directly against the user's own home returns the same message, so the condition is account-level rather than wrapper-induced. The cost is not cosmetic: `review_gate.py` admits a cascade only for a reason code in `CANDIDATE_LOCAL_CODES` carrying `cascade_eligible` true, `quota` is in that set and `unknown_client_failure` is not, so the misclassification produced `stop_reviewer_lane` and the recovery was an operator reordering the clients by hand. Why the wrong hypothesis survived three rounds is the second half of the rule: the EXIT trap removes the run directory with both captured streams, and the receipt held only an exit code, so rounds 122 and 123 left six receipts with no codex record between them and nothing in the repository could contradict the size theory. RED-baseline (applied): eight assertions written against the unchanged wrapper each fail on the row they name; four single-predicate mutations then turn exactly their predicted rows red and no others -- removing the top-level restriction reds the two rows proving packet-derived text cannot classify, removing redaction reds the two redaction rows, removing the bound reds the truncation row, and restoring the stderr-only grep reds the two event-stream rows. Two rows are green on the unchanged baseline by construction and it is the mutations, not the rows, that establish their meaning. Recorded because each cost a red rather than being reasoned out: a heredoc nested in a command substitution is scanned by Bash 3.2 for shell quoting, so an apostrophe in a comment inside it ends the parse of the whole script; and a fixture carrying a literal credential-shaped value is refused by this repository's own credential scanner, which is the scanner behaving correctly. |
|
|
670
|
+
| A redactor guarding an evidence excerpt keys on the SHAPE of an assignment, not on a list of credential-sounding key names: the value of every `key=value` and quoted `"key": "value"` pair is removed whatever the key is called, while the key and any prose that is not assignment-shaped survive so the excerpt still says what failed | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Observed failure, twice in one round on the same predicate: the first review chain found that anchoring a word boundary before the keyword made `access_token=` and `client_secret=` unmatchable, because `_` is itself a word character, and the list was widened; the succeeding chain's challenge then found `session=`, `cookie=`, `auth=`, `code=` and `bearer=` still missing, plus quoted JSON forms. Widening a third time was the obvious move and is the one this row rejects: a list of credential-sounding names has no state in which it is finished, so the recurrence is evidence that the predicate is a proxy rather than the invariant. The list is deleted. What replaces it does not read the key at all. The cost is real and accepted: an informative `error=timeout` loses its value too, which is why the key is preserved rather than the whole pair, and why prose is left alone -- the message this round exists to classify contains no assignment and passes through whole, verified against the live condition rather than argued. RED-baseline (applied): a fixture carrying six unlisted key names and two quoted forms fails against the widened list and passes against the shape rule, and it is the only row that moves. Recorded because it cost a red rather than being reasoned out: this gate judges per commit, so a register row for an owner package must land in the same commit as the package change, not in a later one. |
|
|
671
|
+
| When a failure destroys its own evidence, the fix is to stop destroying it, not to copy an excerpt somewhere durable: the run directory holding the captured streams survives a failed run and the receipt names it, while the receipt itself carries only the transport's own error message. Arbitrary process output never reaches a committed artifact, so no filter over it has to be complete | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. This supersedes the excerpt-plus-redactor shape two rows above, and the reason is measured rather than argued: three review chains each found a different escape from the same filter -- a key name the list lacked (`access_token`), an assignment form the shape rule lacked (`session=`, quoted JSON), and URL userinfo, which is not an assignment at all. Replacing the name list with a shape rule was recorded as an invariant change and was not one: names and shapes are both enumerations of how a secret might look, so the recurrence continued. The predicate was never completable, because the input was arbitrary process output and the property wanted of it -- that nothing secret-shaped survives -- is not decidable. What ends it is removing the input: raw stderr has no path into the receipt. The original defect was also misread on the way in. `RUN_ROOT` is deleted by the EXIT trap, so the streams that named the cause were gone at the moment the cause became interesting; the minimal repair for that is to keep the directory on failure, which also leaves the whole streams for diagnosis rather than 600 redacted bytes. The directory is mode 0700 under TMPDIR and holds exactly what it held while the run was in flight, so nothing is exposed that was not already, and reclaiming it stays the platform's temp-directory lifetime. Stated limit, not claimed away: a CLI-authored error message can still echo a credential, so the redaction stays as defence in depth -- it is no longer the control the safety rests on. RED-baseline (applied): restoring the stderr fallback reds the row asserting stderr is not quoted; deleting the directory on failure reds the two preservation rows; dropping the userinfo strip or the quoted-value alternation reds the shape row; each mutation reds only its own rows. The redaction fixtures were moved onto the event stream in the same change, because on stderr they would have asserted nothing under the new design. |
|
|
672
|
+
| A test fixture that only LOOKS like a credential is still a credential to every scanner that guards a boundary, so a fixture exercising a redactor assembles its credential-shaped values at runtime rather than writing them into the repository | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Observed twice in one round against two different guards. A literal `sk-` token in the fixture was refused by this repository's own credential scanner in `validate-skill.sh`, turning `make test` red. A URL written with credentials in its userinfo component, in the same fixture, was later refused by the review gate's egress tripwire, which returned `egress_denied` and stopped the review lane before any reviewer ran -- and would have done so on every later review of this repository, because the fixture lives in the tracked tree that every packet carries. Both guards were behaving correctly; the fixture was the defect. Approving the egress by flag was available and rejected: it would spend a real control on synthetic data and teach the flag as routine. Assembling the value inside the fixture at runtime keeps the test at full strength -- the wrapper still sees the whole shape -- while the shape never exists in a tracked file. RED-baseline (applied): the suite is green with the assembled value and the redaction rows still red under their mutations, and the literal form is absent from the tree. |
|
|
673
|
+
| A test that reads a path out of a receipt expands it before touching the filesystem: a path recorded with the home directory elided is correct in the receipt and meaningless to a filesystem check, so the check false-REDs on exactly the hosts where the eliding fires | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Raised by the landing review and confirmed by hand rather than reasoned about: with TMPDIR under the home directory the wrapper records `transport_run_dir` as `~/.cache/.../codex-review.9YWG3E`, and the assertion's `[ -d ... ]` then tests a literal tilde path. The failure mode is a false RED, never a false GREEN, which is why it is a portability defect rather than a correctness one -- but a shared suite that reds on someone else's machine costs them the same time it would cost here. The suite's own TMPDIR is not under the home directory, so no row exercises the expansion; the tilde form was reproduced directly against the wrapper instead, and that is the evidence, not a suite row. |
|
|
674
|
+
| Eliding a home directory out of a persisted path compares PHYSICAL paths on both sides, not the literal environment variable: a home spelled with a trailing slash, or reached through a symlink, is the same directory, and a literal comparison leaves the username in the artifact on exactly the hosts that spell it differently | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Raised by the landing challenge against the eliding this round had just added, and it is the same class the round spent three chains on in a different place: matching one spelling of a thing is not matching the thing. `cd \| pwd -P` on both sides normalizes trailing slashes and symlinks in one move rather than enumerating the ways a path can be written, and the redactor gets the same treatment through a set of home spellings applied longest-first. RED-baseline (applied): a row exporting a home with a trailing slash and a TMPDIR beneath it reds against the literal comparison and greens against the physical one, and reverting only the physical resolution reds that row alone. Recorded because it cost a red: the suite's `run_codex` helper pins TMPDIR, so a row needing a TMPDIR under a specific home has to invoke the wrapper directly instead of through the helper -- the helper silently won, and the row failed for a reason that had nothing to do with the fix. |
|
|
675
|
+
| A guard's fixture must be inert to every OTHER rule in the same pipeline, or the guard cannot fail when its own rule is deleted: a secret shaped like a neighbouring rule's input is removed by that neighbour, and the row stays green over the hole it was written to watch | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Found by the landing challenge, not by the mutation pass that was supposed to catch exactly this: the row guarding the URL query strip used `?token=abc123`, which the assignment redactor rewrites on its own, so deleting the query strip left the row green and a bare non-assignment query secret would have reached a committed receipt with no red row anywhere. The earlier mutation run did remove the query strip and did report the row red -- but that run predated the assignment redactor, so the coverage it proved expired when a later rule was added and nothing re-established it. That is the reusable part: a mutation result is evidence about the pipeline as it stood, and adding a rule can silently subsume a neighbour's fixture. RED-baseline (applied, after the fixture was changed to a non-assignment shape): deleting the query strip reds that row alone, and the unmutated control is green. |
|
|
676
|
+
| A rule that strips a delimited region consumes to the LAST delimiter the region can legally contain, not the first: URL userinfo may itself contain the separator, so a non-greedy strip leaves the tail of the secret behind | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Fourth consecutive chain to find something in the same defence-in-depth redactor, and the count is the point rather than the individual defect: the class is bounded only because the redactor is no longer what the safety rests on -- raw process output has no path into a committed receipt, so what remains is a narrowing series of improvements to a secondary control rather than an open hole. This one: excluding the separator from the consumed class stopped the strip at the first one, so a password containing a literal separator left its tail. Consuming greedily to the last separator before the path boundary closes the whole shape rather than the one spelling that was reported. RED-baseline (applied): a fixture whose password contains the separator reds against the non-greedy rule and greens against the greedy one, and reverting only that character class reds that row alone. |
|
|
677
|
+
| A fixture built to exercise a redactor is chosen so that no PART of it reads as a different sensitive shape: a reviewer quoting the finding puts the fixture into a committed receipt, where every other content gate then reads it | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Third distinct guard this round has tripped on its own test data, after the credential scanner and the egress tripwire. The userinfo fixture needed a password containing the separator; with a dotted host, the tail of that password plus the host reads as an email address, and the public-sanitization gate refused the receipt a reviewer wrote it into. A host without a dot exercises the same wrapper behaviour and forms no such shape. The reusable part is the indirection: the fixture is not what the gate scans -- the receipt is, and its content is chosen by a reviewer quoting the fixture, so the fixture has to be clean under every gate rather than under the one it was written for. Also recorded here because it cost a full CI round: `check-public-sanitization.py` and `review_ledger_binding.py` run in CI's repository-gates job and are absent from `make test`, so a green local run is not evidence about those two. |
|
|
678
|
+
| A fixture that FAILS TO PRODUCE its input turns every assertion reading that input green, so a change to a fixture is verified by observing the input it emits, not by the suite's exit status: a green suite is exactly what a dead fixture produces | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Landed and caught by the next review round, which is the honest record: an edit to the fixture indented one statement inside an embedded interpreter block, the interpreter died before emitting anything, the wrapper fell through to its no-output placeholder, and five redaction assertions passed over a placeholder while the suite reported its success token. The mutation evidence recorded for those rows was true when taken and had silently expired. Two rules follow, and the second is the one that would have caught it: after changing a fixture, observe the input it now emits; and re-run the mutation for the rows that fixture feeds rather than citing the earlier run. RED-baseline (applied, after the indentation was corrected): removing the redactor reds all five rows and the unmutated control is green -- which is the check that was missing, because the same removal against the broken fixture left all five green. |
|
|
679
|
+
| A value used as a redaction NEEDLE is rejected when it is only structure: a home directory of `/` is a legitimate environment and a catastrophic needle, because replacing it rewrites every separator in the text and disables every rule that runs after it | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Raised by the landing review against the normalization the previous round added: gathering every spelling of the home directory is right, but a spelling that carries no content is not a path to elide. With `HOME=/` -- root, or an arbitrary-uid container -- the needle set contained `/`, the replacement ran before the URL rules, and the excerpt came out mangled with its credentials intact. The general shape is that a needle derived from the environment needs a content test, not only a presence test. RED-baseline (applied): a row invoking the wrapper with `HOME=/` reds without the content test and greens with it, and removing only that test reds that row alone. |
|
|
680
|
+
| A filter over free text in a persisted artifact is replaced by having no free text: "nothing secret-shaped survives" is not decidable over arbitrary text, so an adversarial reviewer can always spell one more escape, and the terminal state is a constant the input cannot influence rather than a filter that keeps growing | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. The measurement is the row: eight review chains after the input was already narrowed to CLI-authored error messages, each found a different escape -- an unlisted key name, an assignment form, URL userinfo, a password containing the separator, a fixture whose own shape tripped a neighbouring gate, an escaped quote closing a quoted value early, an uppercase scheme, a separator-only home used as a needle. Every one was real and none was derivable from the previous one. Two intermediate diagnoses were wrong on the way and are recorded above: swapping a key-name list for an assignment-shape rule was called an invariant change and was another enumeration, and narrowing the input was called sufficient when it only slowed the rate. What ends the class is that the receipt now carries a constant and the transport's output stays in the preserved run directory. The property is stated as equality with that constant, which a test can hold, instead of the absence of a list of shapes, which no test can. RED-baseline (applied): echoing the extracted message into the receipt reds the invariant row, and the unmutated control is green. Cost, recorded because the next round should be able to weigh it: each chain was roughly twelve minutes of wall clock, and the merge gate accepts no open challenge finding, so there was no landing state that carried the residue. |
|
|
681
|
+
| A fixture string is read by every scanner in the repository, not only by the suite it belongs to, so it is chosen to be inert under all of them: a host:port that exists only inside a quoted test payload still reads as a listening port to a lane-isolation scanner | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Fourth guard this round to reject the round's own test data, after the credential scanner, the egress tripwire and the public-sanitization gate. The userinfo fixture carried a port it never needed, and the parallel-lane isolation scanner reads any host:port in a lane member as evidence that concurrent suites could race on it. Dropping the port exercises the same wrapper behaviour. Recorded as one rule with the three before it: the cost of learning this one guard at a time was a full verification cycle each, and the cheaper order is to sweep every local gate after touching a fixture, before spending a review chain on the candidate. RED-baseline (applied): `test_lane_isolation.py` reds on the ported form and greens on the bare host, with the wrapper suite green either way -- which is why the suite alone was not evidence. |
|
|
@@ -299,12 +299,17 @@ def added_evidence_paths(repo_root: Path, base: str) -> list[str]:
|
|
|
299
299
|
"""
|
|
300
300
|
result = subprocess.run(
|
|
301
301
|
[
|
|
302
|
-
"git", "-C", str(repo_root), "diff", "--name-only",
|
|
302
|
+
"git", "-C", str(repo_root), "diff", "--name-only", "-z",
|
|
303
303
|
"--diff-filter=A", base, "HEAD", "--", EVIDENCE_ROOT,
|
|
304
304
|
],
|
|
305
305
|
stdout=subprocess.PIPE,
|
|
306
306
|
stderr=subprocess.PIPE,
|
|
307
|
+
# A pathname is bytes, and -z hands them over raw. Strict decoding would
|
|
308
|
+
# turn one undecodable filename into a crash inside a gate whose job is
|
|
309
|
+
# to fail cleanly, so undecodable bytes survive as surrogates and simply
|
|
310
|
+
# do not match the evidence pattern.
|
|
307
311
|
text=True,
|
|
312
|
+
errors="surrogateescape",
|
|
308
313
|
check=False,
|
|
309
314
|
)
|
|
310
315
|
if result.returncode != 0:
|
|
@@ -313,14 +318,47 @@ def added_evidence_paths(repo_root: Path, base: str) -> list[str]:
|
|
|
313
318
|
f"{result.stderr.strip()}"
|
|
314
319
|
)
|
|
315
320
|
excluded: list[str] = []
|
|
316
|
-
|
|
317
|
-
|
|
321
|
+
# -z output is NUL-separated and never C-quoted, so a path carrying a
|
|
322
|
+
# non-ASCII byte is enumerated as itself rather than as an escaped literal
|
|
323
|
+
# that no pattern here would match.
|
|
324
|
+
for line in result.stdout.split("\0"):
|
|
325
|
+
if not line or not EVIDENCE_MEMBER.match(line):
|
|
318
326
|
continue
|
|
319
327
|
if is_candidate_receipt(repo_root, line):
|
|
320
328
|
excluded.append(line)
|
|
321
329
|
return excluded
|
|
322
330
|
|
|
323
331
|
|
|
332
|
+
def bound_evidence_paths(repo_root: Path, base: str) -> list[str]:
|
|
333
|
+
"""Evidence this round ADDS that stays inside the candidate.
|
|
334
|
+
|
|
335
|
+
The complement of the exclusion, reported when nothing binds. Committing one
|
|
336
|
+
of these after the review rounds moves the candidate out from under their
|
|
337
|
+
receipts, and the failure that surfaces -- nothing binds -- names neither the
|
|
338
|
+
file nor the ordering. This class has now been observed three times; the
|
|
339
|
+
diagnosis belongs where the failure appears, not in a document the round has
|
|
340
|
+
to know to open.
|
|
341
|
+
"""
|
|
342
|
+
result = subprocess.run(
|
|
343
|
+
[
|
|
344
|
+
"git", "-C", str(repo_root), "diff", "--name-only", "-z",
|
|
345
|
+
"--diff-filter=A", base, "HEAD", "--", EVIDENCE_ROOT,
|
|
346
|
+
],
|
|
347
|
+
stdout=subprocess.PIPE,
|
|
348
|
+
stderr=subprocess.PIPE,
|
|
349
|
+
text=True,
|
|
350
|
+
errors="surrogateescape",
|
|
351
|
+
check=False,
|
|
352
|
+
)
|
|
353
|
+
if result.returncode != 0:
|
|
354
|
+
return []
|
|
355
|
+
return [
|
|
356
|
+
line
|
|
357
|
+
for line in result.stdout.split("\0")
|
|
358
|
+
if line and EVIDENCE_MEMBER.match(line) and not is_candidate_receipt(repo_root, line)
|
|
359
|
+
]
|
|
360
|
+
|
|
361
|
+
|
|
324
362
|
def is_candidate_receipt(repo_root: Path, path_value: str) -> bool:
|
|
325
363
|
"""Whether the committed blob at this path is a receipt about a candidate.
|
|
326
364
|
|
|
@@ -405,14 +443,14 @@ def candidate_hash(module: types.ModuleType, repo_root: Path, base: str, paths:
|
|
|
405
443
|
paths=list(paths),
|
|
406
444
|
wording_only_proof_file=None,
|
|
407
445
|
)
|
|
408
|
-
packet_path,
|
|
446
|
+
packet_path, _packet_sha256, candidate_sha256, _bytes, _paths, _secrets = module.freeze_packet(
|
|
409
447
|
args, time.monotonic() + 120
|
|
410
448
|
)
|
|
411
449
|
try:
|
|
412
450
|
Path(packet_path).unlink(missing_ok=True)
|
|
413
451
|
except OSError:
|
|
414
452
|
pass
|
|
415
|
-
return
|
|
453
|
+
return candidate_sha256
|
|
416
454
|
|
|
417
455
|
|
|
418
456
|
def canonical_digest(value: object) -> str:
|
|
@@ -1094,6 +1132,19 @@ def bind_candidate(
|
|
|
1094
1132
|
" no committed ledger records this candidate; run the extraction review "
|
|
1095
1133
|
"lane against the final, committed tree"
|
|
1096
1134
|
)
|
|
1135
|
+
inside = bound_evidence_paths(repo_root, fork)
|
|
1136
|
+
if inside:
|
|
1137
|
+
binding.failure.append(
|
|
1138
|
+
" this round added evidence that stays inside the candidate: "
|
|
1139
|
+
+ ", ".join(inside[:5])
|
|
1140
|
+
+ ("" if len(inside) <= 5 else f", and {len(inside) - 5} more")
|
|
1141
|
+
)
|
|
1142
|
+
binding.failure.append(
|
|
1143
|
+
" only added JSON carrying a candidate_sha256 is excluded, so bound "
|
|
1144
|
+
"evidence such as base attestations and excerpts must be committed "
|
|
1145
|
+
"BEFORE the review rounds; committing it after moves the candidate out "
|
|
1146
|
+
"from under their receipts"
|
|
1147
|
+
)
|
|
1097
1148
|
return binding
|
|
1098
1149
|
|
|
1099
1150
|
|
|
@@ -36,6 +36,12 @@ assert_contains "$PRODUCT_SKILL" 'gaps block `complete`' "product workflow gate
|
|
|
36
36
|
assert_contains "$PRODUCT_SKILL" "references/implementation-completeness-and-minimality.md" "product workflow pointer"
|
|
37
37
|
assert_contains "$PRODUCT_REF" "Requirement / acceptance point | Source decision | Implementation surface | Verification | Fresh evidence | Status" "acceptance closure matrix"
|
|
38
38
|
assert_contains "$PRODUCT_REF" "New concept | Current acceptance point or hard constraint | Simpler alternative | Decision" "concept delta matrix"
|
|
39
|
+
assert_contains "$PRE_FINAL_REF" "Awaiting work you started yourself is not a stop condition." "continuation gate (self-initiated in-flight work is not a stop)"
|
|
40
|
+
assert_contains "$PRE_FINAL_REF" "is in-flight work rather than a handoff: wait for it and continue in the same turn" "continuation gate (in-flight obligation)"
|
|
41
|
+
assert_contains "$PRE_FINAL_REF" "Before ending any turn, name the next action; if you can perform it now, the turn is not over." "continuation gate (turn-end firing check)"
|
|
42
|
+
assert_contains "$PRODUCT_SKILL" "poll any finite step you started to its result, never reporting it as running" "continuation gate (entry firing signal)"
|
|
43
|
+
assert_contains "$PRE_FINAL_REF" "is \`continuing:\`, never \`blocked:\` and never a final response. Poll it to a terminal result" "continuation gate (outcome-contract clause)"
|
|
44
|
+
assert_contains "$PRE_FINAL_REF" "has no terminal result to wait for: take its readiness signal and proceed" "continuation gate (persistent-process exception)"
|
|
39
45
|
assert_contains "$PRODUCT_REF" "Passing one question never compensates for failing the other." "independent axes"
|
|
40
46
|
assert_contains "$PRODUCT_REF" 'An implementer may not silently downscope a point' "no self-downscope"
|
|
41
47
|
assert_contains "$PRODUCT_REF" 'hypothetical reuse are not evidence' "no speculative concepts"
|
|
@@ -903,6 +903,74 @@ else
|
|
|
903
903
|
'{ [ "$real_rc" = 0 ] && { case "$real_out" in [0-9a-f]*) [ ${#real_out} = 64 ];; *) false;; esac || case "$real_out" in *review_ledger_binding_no_change*) true;; *) false;; esac; }; } || { [ "$real_rc" = 1 ] && case "$real_out" in *"cannot freeze the candidate packet"*"review packet exceeds 200000 bytes"*) true;; *) false;; esac; }'
|
|
904
904
|
fi
|
|
905
905
|
|
|
906
|
+
# The point of splitting the candidate from the packet is that a round which
|
|
907
|
+
# widened its packet to answer an evidence-gap finding still produces a receipt
|
|
908
|
+
# this gate can accept. That claim spans both sides, so it is asserted across
|
|
909
|
+
# both: the identity the controller records for a widened packet must be the
|
|
910
|
+
# identity this gate recomputes from the repository. Mutate the gate to return
|
|
911
|
+
# the packet hash instead and this goes red while nothing else does.
|
|
912
|
+
printf 'widened-candidate\n' >>"$REPO/skills/skill-extraction-workflow/SKILL.md"
|
|
913
|
+
git -C "$REPO" add -A
|
|
914
|
+
git -C "$REPO" commit -qm widened
|
|
915
|
+
WIDENED_BASE="$BASE"
|
|
916
|
+
GATE_CANDIDATE="$(run_gate --base "$WIDENED_BASE" --print-candidate)"
|
|
917
|
+
controller_candidate="$(
|
|
918
|
+
CONTROLLER_DIR="$(dirname "$CONTROLLER")" REPO="$REPO" BASE="$WIDENED_BASE" \
|
|
919
|
+
GATE="$GATE" WORK="$WORK" python3 - <<'PY' 2>&1
|
|
920
|
+
import importlib.util
|
|
921
|
+
import os
|
|
922
|
+
import sys
|
|
923
|
+
import time
|
|
924
|
+
from pathlib import Path
|
|
925
|
+
from types import SimpleNamespace
|
|
926
|
+
|
|
927
|
+
sys.path.insert(0, os.environ["CONTROLLER_DIR"])
|
|
928
|
+
import review_gate
|
|
929
|
+
|
|
930
|
+
spec = importlib.util.spec_from_file_location("binder", os.environ["GATE"])
|
|
931
|
+
binder = importlib.util.module_from_spec(spec)
|
|
932
|
+
spec.loader.exec_module(binder)
|
|
933
|
+
|
|
934
|
+
repo = Path(os.environ["REPO"])
|
|
935
|
+
# An author cannot hand-build these: the base is a fork point and the paths
|
|
936
|
+
# carry the receipt exclusions this round already added. Both come from the gate
|
|
937
|
+
# that will judge the receipt, which is what makes it one identity.
|
|
938
|
+
base, _excludes, paths, _changed = binder.candidate_scope(
|
|
939
|
+
repo, os.environ["BASE"], (".",)
|
|
940
|
+
)
|
|
941
|
+
|
|
942
|
+
|
|
943
|
+
def freeze(**overrides):
|
|
944
|
+
args = SimpleNamespace(
|
|
945
|
+
cwd=str(repo),
|
|
946
|
+
diff_file=None,
|
|
947
|
+
base=base,
|
|
948
|
+
paths=list(paths),
|
|
949
|
+
wording_only_proof_file=None,
|
|
950
|
+
)
|
|
951
|
+
for key, value in overrides.items():
|
|
952
|
+
setattr(args, key, value)
|
|
953
|
+
return review_gate.freeze_packet(args, time.monotonic() + 60)
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
narrow_path, _narrow_packet, narrow_candidate, _n, _p, _s = freeze()
|
|
957
|
+
subject = narrow_path.read_bytes()
|
|
958
|
+
narrow_path.unlink()
|
|
959
|
+
|
|
960
|
+
# Outside the repository: the candidate includes untracked files.
|
|
961
|
+
widened = Path(os.environ["WORK"]) / "widened-for-binder.patch"
|
|
962
|
+
widened.write_bytes(subject + b"\n--- appended context for the reviewer ---\n")
|
|
963
|
+
wide_path, wide_packet, wide_candidate, _wn, _p, _s = freeze(diff_file=str(widened))
|
|
964
|
+
wide_path.unlink()
|
|
965
|
+
|
|
966
|
+
assert wide_packet != wide_candidate, "the widened packet must not be its own candidate"
|
|
967
|
+
assert wide_candidate == narrow_candidate, "widening moved the candidate"
|
|
968
|
+
print(wide_candidate)
|
|
969
|
+
PY
|
|
970
|
+
)"
|
|
971
|
+
check "a widened packet records the candidate identity this gate recomputes" \
|
|
972
|
+
'[ ${#GATE_CANDIDATE} = 64 ] && [ "$controller_candidate" = "$GATE_CANDIDATE" ]'
|
|
973
|
+
|
|
906
974
|
if [ "$fails" -gt 0 ]; then
|
|
907
975
|
echo "test_review_ledger_binding: $fails failing case(s)" >&2
|
|
908
976
|
exit 1
|
|
@@ -89,6 +89,20 @@ When claiming tests pass, report:
|
|
|
89
89
|
|
|
90
90
|
Do not report "tests pass" from memory or from a previous turn. Verification must be fresh for the current change.
|
|
91
91
|
|
|
92
|
+
## Evidence-Record Integrity For Measurement Harnesses
|
|
93
|
+
|
|
94
|
+
Fires when the deliverable is a harness whose RECORDS are the evidence — an evaluation or benchmark runner, a conformance suite feeding a comparison, an A/B or regression measurement rig — rather than a suite whose deliverable is pass/fail. Reached as a failure class from `scenario-testing.md` (High-Risk Failure Classes). Such a harness can be corrupted by the data it produces, in three ways that all read green. Build the record layer against these before the happy path; retrofitting means re-judging records already collected. This is one layer upstream of the entry rule that assigns absence assertions to the producing layer: that rule says where absence can be PROVEN, these say whether the record can express WHICH absence occurred at all.
|
|
95
|
+
|
|
96
|
+
- **Absence carries a coded reason beside the value — never a bare null, never a new value-type.** One null cannot say whether the thing was confirmed not to exist or was never observed, and here those are opposite facts: the first is a result about the system under measurement, the second is a hole in the measurement. Put the reason in a sibling field from a closed vocabulary separating at least *confirmed absent*, *asked but unavailable*, and *not attempted* — they imply opposite retry decisions, so collapsing them also destroys the scheduling signal. Keep it beside the value, not inside its type: the relational model's own two-marker proposal (missing-but-applicable vs missing-but-inapplicable) needed four-valued logic and was never widely adopted, while the health-interchange standard's data-absent-reason coding works because it is an adjacent field only its readers pay for. Make the claim cost evidence — accept *confirmed absent* only with the retrievable observation that established it, or a writer clears a failure by asserting absence. And give absence its own assertion: a threshold over values cannot see a series that is not there, which is why a monitoring query language needs a dedicated absent-vector operator to alert on a series that stopped arriving.
|
|
97
|
+
- **Report the flow counts per compared arm, not only per run; caps are the secondary control.** Planned units, extra attempts, and blocked units must be separately counted and reported, so a reader can check the denominator instead of trusting a label — and broken out per arm or per analysis being compared, together with the mix of absence reasons. A run-level total hides the asymmetry that is precisely the bias below: ten blocked units on one side and none on the other reads green in the totals while the comparison is already spoiled. Controlled-trial reporting guidance is explicit that the denominator belongs to *each group* in *every* analysis, not to the study as a whole; it also dropped the requirement to *declare* an analysis intention-to-treat — because no label reliably says who was actually included — and replaced it with that required flow of numbers. Keep caps, but a cap nobody can audit against reported counts is a claim, not a control.
|
|
98
|
+
- **Retain the earliest decisive outcome; record why later attempts happened.** A retry must not replace a failure that already occurred. Keep every attempt and let a unit's recorded outcome be the earliest decisive one — including a failure that first appears on a later attempt after an inconclusive earlier one. Trial reporting is again the shape to copy: post-hoc change is not forbidden, it is required to be reported with its reason. An automatic retry that overwrites the first attempt produces exactly this corruption: the symptom is hidden, the run still reports itself complete, and the reported pass rate stops being a number a release decision can rest on.
|
|
99
|
+
|
|
100
|
+
Why this is not cosmetic: in a measurement harness missingness is rarely random — evidence is missing BECAUSE the run failed, the missing-not-at-random case — so dropping incomplete units biases the comparison toward whatever produced them.
|
|
101
|
+
|
|
102
|
+
Verify by building the harness's own negative cases first: one unit per absence reason, one that spends an extra attempt, and one that fails and then succeeds. A harness that cannot distinguish those three in its own output is not ready to measure anything else. Route the online-signal form of the absence rule — a metric that stopped arriving versus one reporting zero — to `platform-observability`.
|
|
103
|
+
|
|
104
|
+
Boundary: this is an assembled rule, not a named discipline. Measurement system analysis is the adjacent established field, and it covers instrument accuracy and repeatability, not record integrity.
|
|
105
|
+
|
|
92
106
|
## Conditional-Skip × Job-Selection Executed-Count Guards
|
|
93
107
|
|
|
94
108
|
Strongest form — a per-file invariant: the expected-file list derives from the job's own selection manifest, per job/environment (static; never from post-skip collection, which already lacks the silently-skipped file, and never shared across env-split jobs where different files legitimately run), and each expected file collects AND executes > 0 tests, with a missing-optional-dependency skip a hard failure in the job that exists to provide that dependency, never an "expected skip".
|
|
@@ -76,7 +76,7 @@ Use this matrix when a web or app surface contains many charts, grouped tables,
|
|
|
76
76
|
|
|
77
77
|
## High-Risk Failure Classes
|
|
78
78
|
|
|
79
|
-
The risk matrix for a high-risk workflow covers the triggered failure classes from this canonical list: duplicate submit/callback/message/job restart, permission service uncertainty, cross-tenant/user/resource mismatch, partial money/quota side effects, AI provider/model failure, unclear final status after refresh/offline, and missing trace/support identifier. Cover each triggered class at the lowest layer that can prove the invariant.
|
|
79
|
+
The risk matrix for a high-risk workflow covers the triggered failure classes from this canonical list: duplicate submit/callback/message/job restart, permission service uncertainty, cross-tenant/user/resource mismatch, partial money/quota side effects, AI provider/model failure, unclear final status after refresh/offline, and missing trace/support identifier. When the deliverable is itself a measurement harness — an evaluation or benchmark runner, a conformance suite feeding a comparison, an A/B or regression rig — its own record layer is a class of the same standing: absence reading as a pass, a retry moving the denominator, and a later attempt overwriting an earlier failure (`ci-fixtures-and-flake-control.md`, Evidence-Record Integrity For Measurement Harnesses). Cover each triggered class at the lowest layer that can prove the invariant.
|
|
80
80
|
|
|
81
81
|
Cross-reference: `non-functional-specialized-scenarios.md` (High-risk resilience boundaries) states the launch-gate counterpart — which classes require scenario tests or drills at release. That is a gate-criteria list; this is the test-matrix failure-class list. The two complement each other and neither replaces the other.
|
|
82
82
|
|