@ccoalm/ccl-skills 0.15.5 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1679,7 +1679,7 @@ diff_alternate = root / "alternate.patch"
1679
1679
  diff_source.write_bytes(original_diff)
1680
1680
  diff_alternate.write_bytes(alternate_diff)
1681
1681
  with replace_after_symlink_check(diff_source, diff_alternate):
1682
- packet_path, digest, _, _ = review_gate.freeze_packet(
1682
+ packet_path, digest, _candidate, _n, _, _ = review_gate.freeze_packet(
1683
1683
  SimpleNamespace(
1684
1684
  cwd=str(root), diff_file=str(diff_source), base=None, paths=[]
1685
1685
  ),
@@ -1774,6 +1774,241 @@ file_input_race_rc=$?
1774
1774
  check "diff, prior, and completion inputs are read once from a bounded opened descriptor" \
1775
1775
  '[ "$file_input_race_rc" = 0 ] && [ "$file_input_race_probe" = open_once_file_inputs_ok ]'
1776
1776
 
1777
+ # The reviewer's packet and the landing candidate are two objects. A widened
1778
+ # packet exists so a reviewer can judge a claim against code outside the diff;
1779
+ # the candidate exists so the merge-side binder can recompute what actually
1780
+ # lands. Aliasing them made the two mutually exclusive: widening produced a
1781
+ # receipt the binder could never match. These assert the split and the one
1782
+ # invariant that replaces the equality -- the candidate appears in the packet
1783
+ # verbatim, so nothing lands that its reviewer did not read.
1784
+ subject_packet_probe="$(
1785
+ PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY'
1786
+ import hashlib
1787
+ import subprocess
1788
+ import sys
1789
+ import time
1790
+ from pathlib import Path
1791
+ from types import SimpleNamespace
1792
+
1793
+ import review_gate
1794
+
1795
+ root = Path(sys.argv[1]) / "subject-vs-packet"
1796
+ root.mkdir()
1797
+ # Packet files live OUTSIDE the repository on purpose: the base-derived
1798
+ # candidate includes untracked files, so a packet written into the worktree
1799
+ # would become part of the very candidate it has to contain.
1800
+ outside = Path(sys.argv[1]) / "subject-vs-packet-packets"
1801
+ outside.mkdir()
1802
+
1803
+
1804
+ def git(*args):
1805
+ subprocess.run(
1806
+ ["git", "-C", str(root), *args],
1807
+ check=True,
1808
+ stdout=subprocess.DEVNULL,
1809
+ stderr=subprocess.DEVNULL,
1810
+ )
1811
+
1812
+
1813
+ git("init", "-q")
1814
+ git("config", "user.email", "fixture@example.invalid")
1815
+ git("config", "user.name", "fixture")
1816
+ (root / "landing.txt").write_text("old\n", encoding="utf-8")
1817
+ (root / "context.txt").write_text("context-base\n", encoding="utf-8")
1818
+ git("add", "landing.txt", "context.txt")
1819
+ git("commit", "-qm", "base")
1820
+ base = subprocess.run(
1821
+ ["git", "-C", str(root), "rev-parse", "HEAD"],
1822
+ check=True,
1823
+ capture_output=True,
1824
+ text=True,
1825
+ ).stdout.strip()
1826
+ (root / "landing.txt").write_text("new\n", encoding="utf-8")
1827
+
1828
+
1829
+ def freeze(*, diff_file=None, paths=(), base_ref=base, wording=None):
1830
+ return review_gate.freeze_packet(
1831
+ SimpleNamespace(
1832
+ cwd=str(root),
1833
+ diff_file=str(diff_file) if diff_file else None,
1834
+ base=base_ref,
1835
+ paths=list(paths),
1836
+ wording_only_proof_file=wording,
1837
+ ),
1838
+ time.monotonic() + 30,
1839
+ )
1840
+
1841
+
1842
+ def digest(value: bytes) -> str:
1843
+ return hashlib.sha256(value).hexdigest()
1844
+
1845
+
1846
+ def expect_refused(label, **kwargs):
1847
+ try:
1848
+ result = freeze(**kwargs)
1849
+ except review_gate.GateError as exc:
1850
+ return exc
1851
+ result[0].unlink()
1852
+ raise AssertionError(label)
1853
+
1854
+
1855
+ # The base-derived subject: exactly what the landing binder recomputes.
1856
+ subject_path, subject_hash, subject_candidate_hash, subject_n, subject_paths, _ = freeze()
1857
+ subject_bytes = subject_path.read_bytes()
1858
+ subject_path.unlink()
1859
+ assert subject_hash == digest(subject_bytes)
1860
+ assert subject_candidate_hash == subject_hash
1861
+ assert subject_n == len(subject_bytes)
1862
+ assert subject_paths == ["landing.txt"], subject_paths
1863
+
1864
+ # A7 -- with no --diff-file the two hashes are the same value, as they are today.
1865
+ plain_path, plain_packet_hash, plain_candidate_hash, _plain_n, plain_paths, _ = freeze()
1866
+ plain_path.unlink()
1867
+ assert plain_packet_hash == subject_hash
1868
+ assert plain_candidate_hash == subject_hash
1869
+ assert plain_paths == ["landing.txt"], plain_paths
1870
+
1871
+ # A1/A2/A3 -- a widened packet carries the whole subject plus context the
1872
+ # reviewer needs. The packet hash is the widened bytes; the candidate hash is
1873
+ # still the base-derived subject the binder will recompute.
1874
+ context = (
1875
+ b"\n--- context: skills/code-review/SKILL.md (unchanged, for judgment) ---\n"
1876
+ b"the sibling clause the changed lines must not contradict\n"
1877
+ )
1878
+ widened = outside / "widened.patch"
1879
+ widened.write_bytes(subject_bytes + context)
1880
+ wide_path, wide_packet_hash, wide_candidate_hash, wide_n, wide_paths, _ = freeze(
1881
+ diff_file=widened
1882
+ )
1883
+ try:
1884
+ assert wide_packet_hash == digest(subject_bytes + context)
1885
+ assert wide_candidate_hash == subject_hash
1886
+ assert wide_packet_hash != wide_candidate_hash
1887
+ # The reviewer is told where the candidate ends; without that, appended
1888
+ # hunks that continue or appear to revert the diff are indistinguishable
1889
+ # from candidate content in a packet-bounded read.
1890
+ assert wide_n == len(subject_bytes), wide_n
1891
+ # A10 -- candidate paths follow the subject, not the packet, so owner
1892
+ # selection and the wording-only changed-file comparison stay bound to what
1893
+ # lands rather than to whatever context was appended.
1894
+ assert wide_paths == ["landing.txt"], wide_paths
1895
+ finally:
1896
+ wide_path.unlink()
1897
+
1898
+ # A5 -- a packet missing part of the candidate is refused. This is the property
1899
+ # the equality used to provide for free.
1900
+ truncated = outside / "truncated.patch"
1901
+ truncated.write_bytes(subject_bytes[: len(subject_bytes) // 2] + context)
1902
+ exc = expect_refused(
1903
+ "a packet missing part of the candidate was accepted", diff_file=truncated
1904
+ )
1905
+ assert "BEGIN" in str(exc), str(exc)
1906
+
1907
+ # A6 -- context appended passes; context spliced into the middle of the
1908
+ # candidate does not, because then the candidate is no longer in the packet
1909
+ # verbatim and no cheap check can tell a splice from a silent edit.
1910
+ split = len(subject_bytes) // 2
1911
+ interleaved = outside / "interleaved.patch"
1912
+ interleaved.write_bytes(subject_bytes[:split] + context + subject_bytes[split:])
1913
+ expect_refused(
1914
+ "a packet interleaving context inside the candidate was accepted",
1915
+ diff_file=interleaved,
1916
+ )
1917
+
1918
+ # A11 -- context BEFORE the candidate is refused even though the candidate is
1919
+ # present verbatim. A bare containment test accepts this, and an adversarial
1920
+ # round showed what it buys: a sanitized decoy diff read as the change while the
1921
+ # real candidate reads as trailing context.
1922
+ prepended = outside / "prepended.patch"
1923
+ prepended.write_bytes(context + subject_bytes)
1924
+ exc = expect_refused(
1925
+ "a packet preceding the candidate with other content was accepted",
1926
+ diff_file=prepended,
1927
+ )
1928
+ assert "BEGIN" in str(exc), str(exc)
1929
+
1930
+ # A4 -- a packet with no relation to the candidate is refused.
1931
+ unrelated = outside / "unrelated.patch"
1932
+ unrelated.write_bytes(b"diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n")
1933
+ expect_refused("an unrelated packet was accepted", diff_file=unrelated)
1934
+
1935
+ # A8 -- --diff-file alone keeps today's meaning: no base, so no subject, and
1936
+ # the candidate hash stays the packet's own hash.
1937
+ alone_path, alone_packet_hash, alone_candidate_hash, _n, _, _ = freeze(
1938
+ diff_file=widened, base_ref=None
1939
+ )
1940
+ alone_path.unlink()
1941
+ assert alone_packet_hash == digest(subject_bytes + context)
1942
+ assert alone_candidate_hash == alone_packet_hash
1943
+
1944
+ # A9 -- the wording-only proof is a machine check over a full-context
1945
+ # base-derived diff and has no meaning over an author-assembled packet.
1946
+ proof = outside / "wording-only.json"
1947
+ proof.write_text("{}", encoding="utf-8")
1948
+ expect_refused(
1949
+ "a wording-only proof was accepted over an author-assembled packet",
1950
+ diff_file=widened,
1951
+ wording=str(proof),
1952
+ )
1953
+ # ... in the COMBINED form. Bare --diff-file with a wording-only proof stays
1954
+ # accepted, which the cases above this block exercise throughout; the boundary
1955
+ # is where a base-derived candidate and author-assembled bytes would both be in
1956
+ # play with nothing saying which one the proof's scope describes.
1957
+ result = freeze(diff_file=widened, base_ref=None, wording=str(proof))
1958
+ result[0].unlink()
1959
+
1960
+ # A12 -- an empty base-derived candidate is refused rather than trivially
1961
+ # satisfying the prefix check, which every packet does for empty bytes.
1962
+ empty_repo = Path(sys.argv[1]) / "empty-candidate"
1963
+ empty_repo.mkdir()
1964
+ subprocess.run(["git", "-C", str(empty_repo), "init", "-q"], check=True)
1965
+ subprocess.run(
1966
+ ["git", "-C", str(empty_repo), "config", "user.email", "fixture@example.invalid"],
1967
+ check=True,
1968
+ )
1969
+ subprocess.run(
1970
+ ["git", "-C", str(empty_repo), "config", "user.name", "fixture"], check=True
1971
+ )
1972
+ (empty_repo / "kept.txt").write_text("unchanged\n", encoding="utf-8")
1973
+ subprocess.run(
1974
+ ["git", "-C", str(empty_repo), "add", "kept.txt"],
1975
+ check=True,
1976
+ stdout=subprocess.DEVNULL,
1977
+ )
1978
+ subprocess.run(
1979
+ ["git", "-C", str(empty_repo), "commit", "-qm", "base"],
1980
+ check=True,
1981
+ stdout=subprocess.DEVNULL,
1982
+ )
1983
+ empty_base = subprocess.run(
1984
+ ["git", "-C", str(empty_repo), "rev-parse", "HEAD"],
1985
+ check=True,
1986
+ capture_output=True,
1987
+ text=True,
1988
+ ).stdout.strip()
1989
+ try:
1990
+ review_gate.freeze_packet(
1991
+ SimpleNamespace(
1992
+ cwd=str(empty_repo),
1993
+ diff_file=str(widened),
1994
+ base=empty_base,
1995
+ paths=[],
1996
+ wording_only_proof_file=None,
1997
+ ),
1998
+ time.monotonic() + 30,
1999
+ )
2000
+ except review_gate.GateError as exc:
2001
+ assert exc.reason_code == "empty_diff", exc.reason_code
2002
+ else:
2003
+ raise AssertionError("an empty base-derived candidate was accepted")
2004
+
2005
+ print("subject_packet_split_ok")
2006
+ PY
2007
+ )"
2008
+ subject_packet_rc=$?
2009
+ check "a widened packet keeps the base-derived candidate and must contain it verbatim" \
2010
+ '[ "$subject_packet_rc" = 0 ] && [ "$subject_packet_probe" = subject_packet_split_ok ]'
2011
+
1777
2012
  reset_case missing_coverage passed unavailable
1778
2013
  out="$(run_gate --diff-file "$WORK/secret-diff.patch")"; rc=$?
1779
2014
  check "missing coverage cannot widen egress for a secret-bearing diff without approval" \
@@ -664,3 +664,18 @@ The pending classification above is superseded by the executed source comparison
664
664
  | The shared implementation-gates fixture pins the continuation gate's non-stop clauses, so a later edit cannot silently delete them | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. The sibling row for `product-rd-workflow` records the failure itself; this row records why the fix cannot regress silently. Four assertions were added to the fixture -- the reference's non-stop clause, its turn-end firing check, its outcome-contract line, and the entrypoint's own clause. The fourth was added after independent review observed that the outcome-contract line could be deleted with every assertion still green, which is the same false-green shape the pins exist to prevent. RED-baseline (applied, differential): deleting each protected sentence reds only its owning assertion, with every other assertion passing and the unmutated control clean, so a partial deletion is attributable rather than lost in an aggregate failure. The fixture was chosen over a new suite because it already owns cross-owner rule-retention pins; no new registration surface is introduced. |
665
665
  | The landing binder names the ordering cause at the failure point: evidence a round adds that stays inside the candidate is listed when nothing binds | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/review_ledger_binding.py | updated | Owner key `skill-extraction-workflow/SKILL.md`. Third occurrence of one class. The rule that bound evidence is committed before the review rounds already exists verbatim in the quickstart and already carries a register row marked observed twice in consecutive rounds; this round hit it again because the round was driven from the delivery and review owners and never opened that quickstart. Two prior landings answered the recurrence with more prose, so this one changes the mechanism instead: when nothing binds, the binder enumerates the added evidence that is NOT excluded -- the complement of the receipt exclusion it already computes -- and states that only added JSON carrying a candidate_sha256 is excluded, so committing a base attestation or excerpt after the rounds moves the candidate out from under their receipts. RED-baseline (applied): on this round's own failing candidate the pre-change binder reported only that nothing bound it, naming neither the file nor the ordering; the changed binder lists `landing-base.txt` and the round's markdown dispositions and states the ordering. The five binding suites pass unchanged. The diagnosis now reaches an agent at the moment it fails rather than requiring it to know which document to open. |
666
666
  | A harness whose RECORDS are the evidence — an evaluation or benchmark runner, a conformance suite feeding a comparison, an A/B or regression rig — can be corrupted by the data it produces in three ways that all read green: absence stored as a bare null cannot separate confirmed-absent from never-observed, planned units and retries sharing one counter let a retry move the denominator, and a later attempt overwrites an earlier failure. Its own record layer is a high-risk failure class of the same standing as the canonical list, and is built against these before the happy path | `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/testing-strategy/references/ci-fixtures-and-flake-control.md#Absence carries a coded reason beside the value | updated | `testing-strategy/SKILL.md` is the owner key and is unchanged this round: the entrypoint is over its size budget and the growth gate blocks it, so the rule lands in `testing-strategy/references/ci-fixtures-and-flake-control.md` and is reached as a failure class from the canonical high-risk list in `testing-strategy/references/scenario-testing.md`, an enumeration the entrypoint already tells readers to walk. RED baseline: a paired walk over real artifacts — a held-out harness that contributed nothing to deriving the rules fails all three rows, each defect named by exactly one row while the other two do not mention it, while the control harness passes two and partially satisfies the first, so the check discriminates rather than accepting whatever is put to it. `observed-failure` is `no` deliberately: no malfunction of an existing repository rule was recorded this round, and the delta is measured against the held-out artifact rather than against a regression this repository observed; `result-class` is `failure` because that held-out artifact does exhibit all three defects the rule names. Sources read this round: the health-interchange data-absent-reason code system, a monitoring query language's absent-vector operators, and the controlled-trial reporting guidance for the flow diagram and per-group denominators. Known limit, stated in the landed text itself: the assembled rule has no located prior name, and measurement system analysis is the adjacent established field covering instrument accuracy and repeatability rather than record integrity. |
667
+ | The reviewer's packet and the landing candidate are two objects: `--diff-file` widens what the reviewer reads while `--base` keeps `candidate_sha256` the base-derived identity the merge-side binder recomputes, and the gate accepts the combination only when the packet BEGINS with that candidate, byte for byte | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/review_gate.py | `updated` | Owner key `code-review/SKILL.md`. Observed failure: the previous round's first review chain returned five findings of one class -- the code a containment claim depends on is not in the packet -- and answering them was impossible, because the skill tells the author that insufficient input is an input defect to be answered by widening the packet and rerunning the lane, while `freeze_packet` refused `--diff-file` together with `--base` and recorded `candidate_sha256` as the packet's own hash, so a widened rerun produced a receipt `review_ledger_binding.py` can never match. Following the contract produced evidence the merge side rejects. The controller now computes the candidate from the base independently of the packet, accepts the combination only when the packet begins with that candidate, and derives `candidate_paths` from the candidate so owner selection and the wording-only `changed_files` comparison cannot be widened by appended context; the in-chain checks that asserted the two hashes were equal now compare candidate to candidate, which is what lets a later round in one chain read more than an earlier one. Backward compatibility is byte-exact: with no `--diff-file` both hashes keep today's value, and `--diff-file` alone keeps today's meaning. The anchor is a prefix rather than a bare containment test because the round's adversarial challenge showed what containment alone buys: a packet PRECEDING the candidate with a sanitized decoy diff passes, and the reviewer then reads the decoy as the change and the real candidate as trailing context -- the repository's authoring rule already said context sits on top of the candidate, and until this round it was documented and unenforced. RED-baseline (applied, differential): collapsing the candidate identity back into the packet hash reds only the new acceptance case in `test_review_gate.sh` while its other 268 cases pass; removing the anchor check reds that same case alone; the unmutated control is green. Recorded because it cost a red rather than being reasoned out: the packet file must live outside the repository, because the base-derived candidate includes untracked files and a packet written into the worktree becomes part of the candidate it has to contain. |
668
+ | The binding gate's cross-side agreement is asserted rather than assumed: a widened packet's recorded candidate identity must equal the one the gate recomputes, and the base and paths for that comparison come from the gate's own scope resolution rather than being rebuilt by hand | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. `review_ledger_binding.py` has no behavior delta of its own -- it never passes `--diff-file`, so its candidate and its packet stay the same bytes -- and the honest record of that is a mutation which does NOT discriminate: making it return the packet hash instead of the candidate hash leaves all 85 binder cases green. So what lands here is the cross-side assertion, because the claim that motivated the change spans both sides and neither suite alone can hold it. The binder suite now freezes a widened packet through the controller and requires the identity recorded for it to equal the one this gate recomputes. It derives the base and the paths from `candidate_scope` instead of hand-building them, which is itself the finding: the base is a fork point and the paths carry the receipt exclusions the round has already added, so `--print-candidate` is the authority an author reads rather than a value an author reconstructs. RED-baseline (applied, differential): collapsing the controller's candidate identity into the packet hash reds this case alone while the other 85 pass, and the unmutated control is green. |
669
+ | A reviewer wrapper classifies a transport failure from the transport's own error channel, not from whichever stream is habitual: a CLI run under a structured-output flag reports supply and credential failures as events on stdout, so a classifier reading stderr alone reports a routine quota exhaustion as an unclassifiable client fault and stops the lane instead of cascading; and because only the transport's TOP-LEVEL error events count, model-authored text quoting the reviewed packet cannot steer that decision. Every transport failure additionally carries a bounded, redacted excerpt of what the transport said, because a failure whose captured streams are deleted leaves nothing that can contradict a wrong hypothesis | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Observed failure: the codex lane returned `codex_run_failed` / `unknown_client_failure` three rounds running, and the recorded suspicion was packet size. Both halves were wrong. A 20KB packet reproduces the failure identically, and the preserved run directory shows the account's usage limit announced on the event stream while stderr carried only an unrelated models-manager message -- the quota regex matches the first file and not the second, and the wrapper greps only the second. Invoking the CLI directly against the user's own home returns the same message, so the condition is account-level rather than wrapper-induced. The cost is not cosmetic: `review_gate.py` admits a cascade only for a reason code in `CANDIDATE_LOCAL_CODES` carrying `cascade_eligible` true, `quota` is in that set and `unknown_client_failure` is not, so the misclassification produced `stop_reviewer_lane` and the recovery was an operator reordering the clients by hand. Why the wrong hypothesis survived three rounds is the second half of the rule: the EXIT trap removes the run directory with both captured streams, and the receipt held only an exit code, so rounds 122 and 123 left six receipts with no codex record between them and nothing in the repository could contradict the size theory. RED-baseline (applied): eight assertions written against the unchanged wrapper each fail on the row they name; four single-predicate mutations then turn exactly their predicted rows red and no others -- removing the top-level restriction reds the two rows proving packet-derived text cannot classify, removing redaction reds the two redaction rows, removing the bound reds the truncation row, and restoring the stderr-only grep reds the two event-stream rows. Two rows are green on the unchanged baseline by construction and it is the mutations, not the rows, that establish their meaning. Recorded because each cost a red rather than being reasoned out: a heredoc nested in a command substitution is scanned by Bash 3.2 for shell quoting, so an apostrophe in a comment inside it ends the parse of the whole script; and a fixture carrying a literal credential-shaped value is refused by this repository's own credential scanner, which is the scanner behaving correctly. |
670
+ | A redactor guarding an evidence excerpt keys on the SHAPE of an assignment, not on a list of credential-sounding key names: the value of every `key=value` and quoted `"key": "value"` pair is removed whatever the key is called, while the key and any prose that is not assignment-shaped survive so the excerpt still says what failed | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Observed failure, twice in one round on the same predicate: the first review chain found that anchoring a word boundary before the keyword made `access_token=` and `client_secret=` unmatchable, because `_` is itself a word character, and the list was widened; the succeeding chain's challenge then found `session=`, `cookie=`, `auth=`, `code=` and `bearer=` still missing, plus quoted JSON forms. Widening a third time was the obvious move and is the one this row rejects: a list of credential-sounding names has no state in which it is finished, so the recurrence is evidence that the predicate is a proxy rather than the invariant. The list is deleted. What replaces it does not read the key at all. The cost is real and accepted: an informative `error=timeout` loses its value too, which is why the key is preserved rather than the whole pair, and why prose is left alone -- the message this round exists to classify contains no assignment and passes through whole, verified against the live condition rather than argued. RED-baseline (applied): a fixture carrying six unlisted key names and two quoted forms fails against the widened list and passes against the shape rule, and it is the only row that moves. Recorded because it cost a red rather than being reasoned out: this gate judges per commit, so a register row for an owner package must land in the same commit as the package change, not in a later one. |
671
+ | When a failure destroys its own evidence, the fix is to stop destroying it, not to copy an excerpt somewhere durable: the run directory holding the captured streams survives a failed run and the receipt names it, while the receipt itself carries only the transport's own error message. Arbitrary process output never reaches a committed artifact, so no filter over it has to be complete | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. This supersedes the excerpt-plus-redactor shape two rows above, and the reason is measured rather than argued: three review chains each found a different escape from the same filter -- a key name the list lacked (`access_token`), an assignment form the shape rule lacked (`session=`, quoted JSON), and URL userinfo, which is not an assignment at all. Replacing the name list with a shape rule was recorded as an invariant change and was not one: names and shapes are both enumerations of how a secret might look, so the recurrence continued. The predicate was never completable, because the input was arbitrary process output and the property wanted of it -- that nothing secret-shaped survives -- is not decidable. What ends it is removing the input: raw stderr has no path into the receipt. The original defect was also misread on the way in. `RUN_ROOT` is deleted by the EXIT trap, so the streams that named the cause were gone at the moment the cause became interesting; the minimal repair for that is to keep the directory on failure, which also leaves the whole streams for diagnosis rather than 600 redacted bytes. The directory is mode 0700 under TMPDIR and holds exactly what it held while the run was in flight, so nothing is exposed that was not already, and reclaiming it stays the platform's temp-directory lifetime. Stated limit, not claimed away: a CLI-authored error message can still echo a credential, so the redaction stays as defence in depth -- it is no longer the control the safety rests on. RED-baseline (applied): restoring the stderr fallback reds the row asserting stderr is not quoted; deleting the directory on failure reds the two preservation rows; dropping the userinfo strip or the quoted-value alternation reds the shape row; each mutation reds only its own rows. The redaction fixtures were moved onto the event stream in the same change, because on stderr they would have asserted nothing under the new design. |
672
+ | A test fixture that only LOOKS like a credential is still a credential to every scanner that guards a boundary, so a fixture exercising a redactor assembles its credential-shaped values at runtime rather than writing them into the repository | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Observed twice in one round against two different guards. A literal `sk-` token in the fixture was refused by this repository's own credential scanner in `validate-skill.sh`, turning `make test` red. A URL written with credentials in its userinfo component, in the same fixture, was later refused by the review gate's egress tripwire, which returned `egress_denied` and stopped the review lane before any reviewer ran -- and would have done so on every later review of this repository, because the fixture lives in the tracked tree that every packet carries. Both guards were behaving correctly; the fixture was the defect. Approving the egress by flag was available and rejected: it would spend a real control on synthetic data and teach the flag as routine. Assembling the value inside the fixture at runtime keeps the test at full strength -- the wrapper still sees the whole shape -- while the shape never exists in a tracked file. RED-baseline (applied): the suite is green with the assembled value and the redaction rows still red under their mutations, and the literal form is absent from the tree. |
673
+ | A test that reads a path out of a receipt expands it before touching the filesystem: a path recorded with the home directory elided is correct in the receipt and meaningless to a filesystem check, so the check false-REDs on exactly the hosts where the eliding fires | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Raised by the landing review and confirmed by hand rather than reasoned about: with TMPDIR under the home directory the wrapper records `transport_run_dir` as `~/.cache/.../codex-review.9YWG3E`, and the assertion's `[ -d ... ]` then tests a literal tilde path. The failure mode is a false RED, never a false GREEN, which is why it is a portability defect rather than a correctness one -- but a shared suite that reds on someone else's machine costs them the same time it would cost here. The suite's own TMPDIR is not under the home directory, so no row exercises the expansion; the tilde form was reproduced directly against the wrapper instead, and that is the evidence, not a suite row. |
674
+ | Eliding a home directory out of a persisted path compares PHYSICAL paths on both sides, not the literal environment variable: a home spelled with a trailing slash, or reached through a symlink, is the same directory, and a literal comparison leaves the username in the artifact on exactly the hosts that spell it differently | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Raised by the landing challenge against the eliding this round had just added, and it is the same class the round spent three chains on in a different place: matching one spelling of a thing is not matching the thing. `cd \| pwd -P` on both sides normalizes trailing slashes and symlinks in one move rather than enumerating the ways a path can be written, and the redactor gets the same treatment through a set of home spellings applied longest-first. RED-baseline (applied): a row exporting a home with a trailing slash and a TMPDIR beneath it reds against the literal comparison and greens against the physical one, and reverting only the physical resolution reds that row alone. Recorded because it cost a red: the suite's `run_codex` helper pins TMPDIR, so a row needing a TMPDIR under a specific home has to invoke the wrapper directly instead of through the helper -- the helper silently won, and the row failed for a reason that had nothing to do with the fix. |
675
+ | A guard's fixture must be inert to every OTHER rule in the same pipeline, or the guard cannot fail when its own rule is deleted: a secret shaped like a neighbouring rule's input is removed by that neighbour, and the row stays green over the hole it was written to watch | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Found by the landing challenge, not by the mutation pass that was supposed to catch exactly this: the row guarding the URL query strip used `?token=abc123`, which the assignment redactor rewrites on its own, so deleting the query strip left the row green and a bare non-assignment query secret would have reached a committed receipt with no red row anywhere. The earlier mutation run did remove the query strip and did report the row red -- but that run predated the assignment redactor, so the coverage it proved expired when a later rule was added and nothing re-established it. That is the reusable part: a mutation result is evidence about the pipeline as it stood, and adding a rule can silently subsume a neighbour's fixture. RED-baseline (applied, after the fixture was changed to a non-assignment shape): deleting the query strip reds that row alone, and the unmutated control is green. |
676
+ | A rule that strips a delimited region consumes to the LAST delimiter the region can legally contain, not the first: URL userinfo may itself contain the separator, so a non-greedy strip leaves the tail of the secret behind | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Fourth consecutive chain to find something in the same defence-in-depth redactor, and the count is the point rather than the individual defect: the class is bounded only because the redactor is no longer what the safety rests on -- raw process output has no path into a committed receipt, so what remains is a narrowing series of improvements to a secondary control rather than an open hole. This one: excluding the separator from the consumed class stopped the strip at the first one, so a password containing a literal separator left its tail. Consuming greedily to the last separator before the path boundary closes the whole shape rather than the one spelling that was reported. RED-baseline (applied): a fixture whose password contains the separator reds against the non-greedy rule and greens against the greedy one, and reverting only that character class reds that row alone. |
677
+ | A fixture built to exercise a redactor is chosen so that no PART of it reads as a different sensitive shape: a reviewer quoting the finding puts the fixture into a committed receipt, where every other content gate then reads it | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Third distinct guard this round has tripped on its own test data, after the credential scanner and the egress tripwire. The userinfo fixture needed a password containing the separator; with a dotted host, the tail of that password plus the host reads as an email address, and the public-sanitization gate refused the receipt a reviewer wrote it into. A host without a dot exercises the same wrapper behaviour and forms no such shape. The reusable part is the indirection: the fixture is not what the gate scans -- the receipt is, and its content is chosen by a reviewer quoting the fixture, so the fixture has to be clean under every gate rather than under the one it was written for. Also recorded here because it cost a full CI round: `check-public-sanitization.py` and `review_ledger_binding.py` run in CI's repository-gates job and are absent from `make test`, so a green local run is not evidence about those two. |
678
+ | A fixture that FAILS TO PRODUCE its input turns every assertion reading that input green, so a change to a fixture is verified by observing the input it emits, not by the suite's exit status: a green suite is exactly what a dead fixture produces | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Landed and caught by the next review round, which is the honest record: an edit to the fixture indented one statement inside an embedded interpreter block, the interpreter died before emitting anything, the wrapper fell through to its no-output placeholder, and five redaction assertions passed over a placeholder while the suite reported its success token. The mutation evidence recorded for those rows was true when taken and had silently expired. Two rules follow, and the second is the one that would have caught it: after changing a fixture, observe the input it now emits; and re-run the mutation for the rows that fixture feeds rather than citing the earlier run. RED-baseline (applied, after the indentation was corrected): removing the redactor reds all five rows and the unmutated control is green -- which is the check that was missing, because the same removal against the broken fixture left all five green. |
679
+ | A value used as a redaction NEEDLE is rejected when it is only structure: a home directory of `/` is a legitimate environment and a catastrophic needle, because replacing it rewrites every separator in the text and disables every rule that runs after it | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Raised by the landing review against the normalization the previous round added: gathering every spelling of the home directory is right, but a spelling that carries no content is not a path to elide. With `HOME=/` -- root, or an arbitrary-uid container -- the needle set contained `/`, the replacement ran before the URL rules, and the excerpt came out mangled with its credentials intact. The general shape is that a needle derived from the environment needs a content test, not only a presence test. RED-baseline (applied): a row invoking the wrapper with `HOME=/` reds without the content test and greens with it, and removing only that test reds that row alone. |
680
+ | A filter over free text in a persisted artifact is replaced by having no free text: "nothing secret-shaped survives" is not decidable over arbitrary text, so an adversarial reviewer can always spell one more escape, and the terminal state is a constant the input cannot influence rather than a filter that keeps growing | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. The measurement is the row: eight review chains after the input was already narrowed to CLI-authored error messages, each found a different escape -- an unlisted key name, an assignment form, URL userinfo, a password containing the separator, a fixture whose own shape tripped a neighbouring gate, an escaped quote closing a quoted value early, an uppercase scheme, a separator-only home used as a needle. Every one was real and none was derivable from the previous one. Two intermediate diagnoses were wrong on the way and are recorded above: swapping a key-name list for an assignment-shape rule was called an invariant change and was another enumeration, and narrowing the input was called sufficient when it only slowed the rate. What ends the class is that the receipt now carries a constant and the transport's output stays in the preserved run directory. The property is stated as equality with that constant, which a test can hold, instead of the absence of a list of shapes, which no test can. RED-baseline (applied): echoing the extracted message into the receipt reds the invariant row, and the unmutated control is green. Cost, recorded because the next round should be able to weigh it: each chain was roughly twelve minutes of wall clock, and the merge gate accepts no open challenge finding, so there was no landing state that carried the residue. |
681
+ | A fixture string is read by every scanner in the repository, not only by the suite it belongs to, so it is chosen to be inert under all of them: a host:port that exists only inside a quoted test payload still reads as a listening port to a lane-isolation scanner | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Fourth guard this round to reject the round's own test data, after the credential scanner, the egress tripwire and the public-sanitization gate. The userinfo fixture carried a port it never needed, and the parallel-lane isolation scanner reads any host:port in a lane member as evidence that concurrent suites could race on it. Dropping the port exercises the same wrapper behaviour. Recorded as one rule with the three before it: the cost of learning this one guard at a time was a full verification cycle each, and the cheaper order is to sweep every local gate after touching a fixture, before spending a review chain on the candidate. RED-baseline (applied): `test_lane_isolation.py` reds on the ported form and greens on the bare host, with the wrapper suite green either way -- which is why the suite alone was not evidence. |
@@ -443,14 +443,14 @@ def candidate_hash(module: types.ModuleType, repo_root: Path, base: str, paths:
443
443
  paths=list(paths),
444
444
  wording_only_proof_file=None,
445
445
  )
446
- packet_path, packet_sha256, _paths, _secrets = module.freeze_packet(
446
+ packet_path, _packet_sha256, candidate_sha256, _bytes, _paths, _secrets = module.freeze_packet(
447
447
  args, time.monotonic() + 120
448
448
  )
449
449
  try:
450
450
  Path(packet_path).unlink(missing_ok=True)
451
451
  except OSError:
452
452
  pass
453
- return packet_sha256
453
+ return candidate_sha256
454
454
 
455
455
 
456
456
  def canonical_digest(value: object) -> str:
@@ -903,6 +903,74 @@ else
903
903
  '{ [ "$real_rc" = 0 ] && { case "$real_out" in [0-9a-f]*) [ ${#real_out} = 64 ];; *) false;; esac || case "$real_out" in *review_ledger_binding_no_change*) true;; *) false;; esac; }; } || { [ "$real_rc" = 1 ] && case "$real_out" in *"cannot freeze the candidate packet"*"review packet exceeds 200000 bytes"*) true;; *) false;; esac; }'
904
904
  fi
905
905
 
906
+ # The point of splitting the candidate from the packet is that a round which
907
+ # widened its packet to answer an evidence-gap finding still produces a receipt
908
+ # this gate can accept. That claim spans both sides, so it is asserted across
909
+ # both: the identity the controller records for a widened packet must be the
910
+ # identity this gate recomputes from the repository. Mutate the gate to return
911
+ # the packet hash instead and this goes red while nothing else does.
912
+ printf 'widened-candidate\n' >>"$REPO/skills/skill-extraction-workflow/SKILL.md"
913
+ git -C "$REPO" add -A
914
+ git -C "$REPO" commit -qm widened
915
+ WIDENED_BASE="$BASE"
916
+ GATE_CANDIDATE="$(run_gate --base "$WIDENED_BASE" --print-candidate)"
917
+ controller_candidate="$(
918
+ CONTROLLER_DIR="$(dirname "$CONTROLLER")" REPO="$REPO" BASE="$WIDENED_BASE" \
919
+ GATE="$GATE" WORK="$WORK" python3 - <<'PY' 2>&1
920
+ import importlib.util
921
+ import os
922
+ import sys
923
+ import time
924
+ from pathlib import Path
925
+ from types import SimpleNamespace
926
+
927
+ sys.path.insert(0, os.environ["CONTROLLER_DIR"])
928
+ import review_gate
929
+
930
+ spec = importlib.util.spec_from_file_location("binder", os.environ["GATE"])
931
+ binder = importlib.util.module_from_spec(spec)
932
+ spec.loader.exec_module(binder)
933
+
934
+ repo = Path(os.environ["REPO"])
935
+ # An author cannot hand-build these: the base is a fork point and the paths
936
+ # carry the receipt exclusions this round already added. Both come from the gate
937
+ # that will judge the receipt, which is what makes it one identity.
938
+ base, _excludes, paths, _changed = binder.candidate_scope(
939
+ repo, os.environ["BASE"], (".",)
940
+ )
941
+
942
+
943
+ def freeze(**overrides):
944
+ args = SimpleNamespace(
945
+ cwd=str(repo),
946
+ diff_file=None,
947
+ base=base,
948
+ paths=list(paths),
949
+ wording_only_proof_file=None,
950
+ )
951
+ for key, value in overrides.items():
952
+ setattr(args, key, value)
953
+ return review_gate.freeze_packet(args, time.monotonic() + 60)
954
+
955
+
956
+ narrow_path, _narrow_packet, narrow_candidate, _n, _p, _s = freeze()
957
+ subject = narrow_path.read_bytes()
958
+ narrow_path.unlink()
959
+
960
+ # Outside the repository: the candidate includes untracked files.
961
+ widened = Path(os.environ["WORK"]) / "widened-for-binder.patch"
962
+ widened.write_bytes(subject + b"\n--- appended context for the reviewer ---\n")
963
+ wide_path, wide_packet, wide_candidate, _wn, _p, _s = freeze(diff_file=str(widened))
964
+ wide_path.unlink()
965
+
966
+ assert wide_packet != wide_candidate, "the widened packet must not be its own candidate"
967
+ assert wide_candidate == narrow_candidate, "widening moved the candidate"
968
+ print(wide_candidate)
969
+ PY
970
+ )"
971
+ check "a widened packet records the candidate identity this gate recomputes" \
972
+ '[ ${#GATE_CANDIDATE} = 64 ] && [ "$controller_candidate" = "$GATE_CANDIDATE" ]'
973
+
906
974
  if [ "$fails" -gt 0 ]; then
907
975
  echo "test_review_ledger_binding: $fails failing case(s)" >&2
908
976
  exit 1
@@ -1,8 +1,8 @@
1
1
  {
2
2
  "schema": 1,
3
3
  "npmPackage": "@ccoalm/ccl-skills",
4
- "version": "0.15.5",
5
- "sourceCommit": "88e842ffb4ab944dde3179fa5a8e43755fbba27a",
4
+ "version": "0.16.0",
5
+ "sourceCommit": "200cffbcfb46eb6499d4bd827174268e5ae9359b",
6
6
  "sourceState": "clean",
7
7
  "files": [
8
8
  {
@@ -287,7 +287,7 @@
287
287
  },
288
288
  {
289
289
  "path": "marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md",
290
- "sha256": "900f09bfc21bdd78c9bf85794a361ae35dbf5b1ce0ae1910baa1991615b51093",
290
+ "sha256": "1d3e7d63670364fd6f3ff3ad30722f42d95971da26b5634dceb5f275b32ca1d1",
291
291
  "mode": 420
292
292
  },
293
293
  {
@@ -317,7 +317,7 @@
317
317
  },
318
318
  {
319
319
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh",
320
- "sha256": "9b98ba819ed9baee557c43aa735ab006e0dfb4e02e0e337cdca5e4867845c105",
320
+ "sha256": "23f5ccd8a703d4fa4ab376aea7c0b56a66a01377fadbe812d222af0538496b99",
321
321
  "mode": 493
322
322
  },
323
323
  {
@@ -377,7 +377,7 @@
377
377
  },
378
378
  {
379
379
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py",
380
- "sha256": "b87410280137ace6340c8ee147c89999069e6d72d8913c8d4dc78e6f150e7152",
380
+ "sha256": "079f756554727a90da3636a1c6e98ef77d5515caa15fb1a9f89471779ba6e9f8",
381
381
  "mode": 493
382
382
  },
383
383
  {
@@ -412,7 +412,7 @@
412
412
  },
413
413
  {
414
414
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh",
415
- "sha256": "bd67dcc75343071913411c43fe727d4e4361318318fe85c7f350cd9f000f2c79",
415
+ "sha256": "9c9d588cc991a6634d6e1e80e4c466537c1fe9ac611900d82d171e8fa59e9187",
416
416
  "mode": 493
417
417
  },
418
418
  {
@@ -482,7 +482,7 @@
482
482
  },
483
483
  {
484
484
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh",
485
- "sha256": "55366dc96415e175c1b9a23f6fa6e118679bb656ed96c818f4a534ca14826ef0",
485
+ "sha256": "2affc2bffbb2de7eb0379b349b65c9450d1fe4c35e34111a26669c733b1ae80b",
486
486
  "mode": 493
487
487
  },
488
488
  {
@@ -502,7 +502,7 @@
502
502
  },
503
503
  {
504
504
  "path": "marketplace/plugins/ccl-skills/skills/code-review/SKILL.md",
505
- "sha256": "b057a175a34dbff8b4352429548c38cbcaf16c0b3f2be59b0c6216471ce28d9a",
505
+ "sha256": "adaba6d52d3a4417dbfce54ef3d39e9d04e5c7018c9a5a3d2e8d749382041db0",
506
506
  "mode": 420
507
507
  },
508
508
  {
@@ -2132,7 +2132,7 @@
2132
2132
  },
2133
2133
  {
2134
2134
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md",
2135
- "sha256": "a1f297987a1c91f283d83ede2bd1994b727316a45e978b3480dec6a556644e3a",
2135
+ "sha256": "03da4744e93b396b7983aea966c14378c960e14655d1ac0829240aae929ddc01",
2136
2136
  "mode": 420
2137
2137
  },
2138
2138
  {
@@ -2272,7 +2272,7 @@
2272
2272
  },
2273
2273
  {
2274
2274
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py",
2275
- "sha256": "a75b043aa63e278a45e3c31f418b042a0ffbef3497ff4c1e139f777bd72e6d29",
2275
+ "sha256": "b0b8205f47e263fcd8ad0bcff8392be903db3a2a12089f5f0c7be7d2f101bdc5",
2276
2276
  "mode": 493
2277
2277
  },
2278
2278
  {
@@ -2522,7 +2522,7 @@
2522
2522
  },
2523
2523
  {
2524
2524
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh",
2525
- "sha256": "a299ded873de1796f8e377e832d59faa1909bcceb5504347b072186139362f36",
2525
+ "sha256": "ee702f4b842d93617f060d7f85626c3965b13c0ec7114783566a929f4b84d07f",
2526
2526
  "mode": 493
2527
2527
  },
2528
2528
  {
@@ -3433,5 +3433,5 @@
3433
3433
  "mode": 420
3434
3434
  }
3435
3435
  ],
3436
- "snapshotHash": "098f8ae71c1a51738986e49cf1667aaa62e714b3c4b10d4ff96b0a08f2655d96"
3436
+ "snapshotHash": "609330c0a6b06dd2e7dfd57a083fe049c8d374a57506de067b9c9bd89ea001a0"
3437
3437
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ccoalm/ccl-skills",
3
- "version": "0.15.5",
3
+ "version": "0.16.0",
4
4
  "description": "Reusable workflows that help coding agents plan, build, test, review, and release software — for Claude Code, Codex, and OpenCode",
5
5
  "keywords": ["skills", "agent-skills", "claude", "claude-code", "codex", "opencode", "agent", "ai", "ai-agents", "cli", "anthropic", "developer-tools"],
6
6
  "type": "module",