@ccoalm/ccl-skills 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +10 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +14 -41
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/correction-routing-map.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/coverage-exhaustion-traps.md +7 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/incident-postmortem-extraction.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +73 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/entrypoint_form_census.py +169 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +114 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/reference-access-census.sh +157 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +483 -109
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +188 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_form_census.sh +174 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_reference_access_census.sh +209 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +394 -5
- package/dist/assets/release.json +77 -47
- package/package.json +1 -1
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Regression test for the bank runner's screening-vs-action resolution signal.
|
|
3
|
+
#
|
|
4
|
+
# A report taken below the action floor locates candidates; it does not license
|
|
5
|
+
# a description edit. The runner says so on stdout and in the JSON report, and
|
|
6
|
+
# this test pins BOTH sides of the number — the reference that states the rule
|
|
7
|
+
# and the executable that enforces it — so they cannot drift apart the way the
|
|
8
|
+
# description-length thresholds once did.
|
|
9
|
+
#
|
|
10
|
+
# Uses a fake `claude` earlier in PATH: never invokes a real CLI or account.
|
|
11
|
+
set -euo pipefail
|
|
12
|
+
|
|
13
|
+
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
14
|
+
EVAL_SCRIPT="$SCRIPT_DIR/eval-routing-bank.rb"
|
|
15
|
+
DOC="$SCRIPT_DIR/../references/eval-routing.md"
|
|
16
|
+
[ -f "$EVAL_SCRIPT" ] || { echo "FAIL: eval script not found: $EVAL_SCRIPT" >&2; exit 1; }
|
|
17
|
+
[ -f "$DOC" ] || { echo "FAIL: reference not found: $DOC" >&2; exit 1; }
|
|
18
|
+
|
|
19
|
+
fail() { echo "FAIL: $*" >&2; exit 1; }
|
|
20
|
+
assert_contains() { case "$2" in *"$1"*) : ;; *) fail "expected output to contain: $1${3:+ ($3)}";; esac; }
|
|
21
|
+
assert_absent() { case "$2" in *"$1"*) fail "expected output NOT to contain: $1${3:+ ($3)}";; *) : ;; esac; }
|
|
22
|
+
|
|
23
|
+
TMP="$(mktemp -d "${TMPDIR:-/tmp}/eval-routing-bank-resolution.XXXXXX")"
|
|
24
|
+
trap 'rm -rf "$TMP"' EXIT
|
|
25
|
+
|
|
26
|
+
REPO="$TMP/repo"
|
|
27
|
+
FAKE_BIN="$TMP/bin"
|
|
28
|
+
mkdir -p "$REPO/skills/testing-strategy" "$REPO/skills/tighten-doc" "$REPO/eval" "$FAKE_BIN"
|
|
29
|
+
|
|
30
|
+
git -C "$TMP" init -q repo
|
|
31
|
+
git -C "$REPO" config user.email test@example.invalid
|
|
32
|
+
git -C "$REPO" config user.name "Test User"
|
|
33
|
+
|
|
34
|
+
cat > "$REPO/skills/testing-strategy/SKILL.md" <<'EOF'
|
|
35
|
+
---
|
|
36
|
+
description: Use when choosing test layers and regression evidence.
|
|
37
|
+
---
|
|
38
|
+
# Testing Strategy
|
|
39
|
+
EOF
|
|
40
|
+
|
|
41
|
+
# Keeps the outcome space larger than {expected, none} so the fixture is not
|
|
42
|
+
# vacuous: a grader that always answered the only skill would pass trivially.
|
|
43
|
+
cat > "$REPO/skills/tighten-doc/SKILL.md" <<'EOF'
|
|
44
|
+
---
|
|
45
|
+
description: Use when polishing document wording.
|
|
46
|
+
---
|
|
47
|
+
# Tighten Doc
|
|
48
|
+
EOF
|
|
49
|
+
|
|
50
|
+
# TWO cases, deliberately: with one case a per-case verdict and a report-wide
|
|
51
|
+
# conjunction are indistinguishable, so every assertion below would also pass
|
|
52
|
+
# against the defect they exist to catch. The degraded fixture fails only the
|
|
53
|
+
# first utterance, which leaves one case short of the floor and one clear of it.
|
|
54
|
+
cat > "$REPO/eval/routing-tasks.jsonl" <<'EOF'
|
|
55
|
+
{"id":"resolution-fixture","utterance":"补一个回归测试","expected_skill":"testing-strategy","frozen_at_sha":""}
|
|
56
|
+
{"id":"resolution-fixture-b","utterance":"帮我润色这段文档","expected_skill":"tighten-doc","frozen_at_sha":""}
|
|
57
|
+
EOF
|
|
58
|
+
|
|
59
|
+
git -C "$REPO" add -A
|
|
60
|
+
git -C "$REPO" commit -qm "fixture"
|
|
61
|
+
|
|
62
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
63
|
+
#!/usr/bin/env bash
|
|
64
|
+
prompt="$(cat)"
|
|
65
|
+
case "$prompt" in
|
|
66
|
+
*润色*) printf '{"selected_skill":"tighten-doc","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n' ;;
|
|
67
|
+
*) printf '{"selected_skill":"testing-strategy","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n' ;;
|
|
68
|
+
esac
|
|
69
|
+
EOF
|
|
70
|
+
chmod +x "$FAKE_BIN/claude"
|
|
71
|
+
export PATH="$FAKE_BIN:$PATH"
|
|
72
|
+
|
|
73
|
+
# --- (1) below the floor: banner printed, report says action_resolution false --
|
|
74
|
+
out_lo="$(ruby "$EVAL_SCRIPT" "$REPO" --replicas 3 --json "$TMP/lo.json" 2>&1)" \
|
|
75
|
+
|| fail "runner exited non-zero below the floor:\n$out_lo"
|
|
76
|
+
assert_contains "screening_resolution_only" "$out_lo" "sub-floor run must announce screening resolution"
|
|
77
|
+
assert_contains "replicas=3" "$out_lo" "banner must name the observed replica count"
|
|
78
|
+
assert_contains "floor 10" "$out_lo" "banner must name the required floor"
|
|
79
|
+
assert_contains "valid observations" "$out_lo" "banner must report the weakest task's valid-observation count, not the request alone"
|
|
80
|
+
grep -q '"action_resolution": false' "$TMP/lo.json" \
|
|
81
|
+
|| fail "sub-floor report must carry action_resolution:false — a consumer cannot read a printed banner"
|
|
82
|
+
|
|
83
|
+
# --- (2) at the floor: no banner, report says action_resolution true -----------
|
|
84
|
+
out_hi="$(ruby "$EVAL_SCRIPT" "$REPO" --replicas 10 --json "$TMP/hi.json" 2>&1)" \
|
|
85
|
+
|| fail "runner exited non-zero at the floor:\n$out_hi"
|
|
86
|
+
assert_absent "screening_resolution_only" "$out_hi" "a run at the floor must not be labelled screening-only"
|
|
87
|
+
grep -q '"action_resolution": true' "$TMP/hi.json" \
|
|
88
|
+
|| fail "at-floor report must carry action_resolution:true"
|
|
89
|
+
|
|
90
|
+
# --- (2b) a nominal at-floor run with an invalid observation is NOT actionable -
|
|
91
|
+
# The floor is on valid observations. A grader that fails one call leaves the
|
|
92
|
+
# task below the floor while `--replicas` still reads 10, and a report that
|
|
93
|
+
# trusted the request would license an edit its evidence cannot support.
|
|
94
|
+
# The runner gives each unparsable verdict ONE retry, treating it as a sampling
|
|
95
|
+
# accident, so a fixture that fails a single call is repaired and records no
|
|
96
|
+
# error. The failure has to persist across the retry to leave the task short of
|
|
97
|
+
# the floor -- which is exactly the real shape this guards: a grader that is
|
|
98
|
+
# reliably unable to answer one utterance, not a stray quote.
|
|
99
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
100
|
+
#!/usr/bin/env bash
|
|
101
|
+
prompt="$(cat)"
|
|
102
|
+
case "$prompt" in
|
|
103
|
+
*润色*) printf '{"selected_skill":"tighten-doc","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'; exit 0 ;;
|
|
104
|
+
esac
|
|
105
|
+
echo x >> "$RESOLUTION_COUNT_FILE"
|
|
106
|
+
n=$(wc -l < "$RESOLUTION_COUNT_FILE" | tr -d ' ')
|
|
107
|
+
if [ "$n" = "3" ] || [ "$n" = "4" ]; then printf 'not json at all\n'; exit 0; fi
|
|
108
|
+
printf '{"selected_skill":"testing-strategy","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'
|
|
109
|
+
EOF
|
|
110
|
+
chmod +x "$FAKE_BIN/claude"
|
|
111
|
+
: > "$TMP/count"
|
|
112
|
+
out_deg="$(RESOLUTION_COUNT_FILE="$TMP/count" ruby "$EVAL_SCRIPT" "$REPO" --replicas 10 --json "$TMP/deg.json" 2>&1)" \
|
|
113
|
+
|| fail "runner exited non-zero on the degraded run:\n$out_deg"
|
|
114
|
+
assert_contains "screening_resolution_only" "$out_deg" "a nominal 10-replica run with an invalid observation must not be actionable"
|
|
115
|
+
grep -q '"action_resolution": false' "$TMP/deg.json" \
|
|
116
|
+
|| fail "a 10-replica run with only 9 valid observations must report action_resolution:false"
|
|
117
|
+
grep -q '"min_valid_observations": 9' "$TMP/deg.json" \
|
|
118
|
+
|| fail "the report must expose the weakest task's valid-observation count"
|
|
119
|
+
|
|
120
|
+
# --- (2c) the verdict is per case, and says nothing about a case not measured --
|
|
121
|
+
# A report-wide boolean alone is unsound in the licensing direction: a subset run
|
|
122
|
+
# over one case would otherwise read as licence for an edit to a case the run
|
|
123
|
+
# never graded. Each result carries its own actionable and valid_observations.
|
|
124
|
+
grep -q '"actionable": true' "$TMP/hi.json" \
|
|
125
|
+
|| fail "an at-floor case must carry its own actionable:true"
|
|
126
|
+
grep -q '"valid_observations": 10' "$TMP/hi.json" \
|
|
127
|
+
|| fail "each case must expose its own valid-observation count"
|
|
128
|
+
grep -q '"action_resolution_scope"' "$TMP/hi.json" \
|
|
129
|
+
|| fail "the report must say its top-level verdict covers only the cases it measured"
|
|
130
|
+
grep -q '"actionable": false' "$TMP/deg.json" \
|
|
131
|
+
|| fail "a case left short by an invalid observation must carry actionable:false"
|
|
132
|
+
# The split is the point: the degraded case is short, the other one is not, and a
|
|
133
|
+
# report-wide conjunction substituted for the per-case field would mark both false.
|
|
134
|
+
python3 - "$TMP/deg.json" <<'PYEOF' || fail "the healthy case must stay actionable while the degraded one does not"
|
|
135
|
+
import json,sys
|
|
136
|
+
d=json.load(open(sys.argv[1]))
|
|
137
|
+
by={r["id"]:r for r in d["results"]}
|
|
138
|
+
ok = (by["resolution-fixture"]["actionable"] is False
|
|
139
|
+
and by["resolution-fixture-b"]["actionable"] is True
|
|
140
|
+
and by["resolution-fixture-b"]["valid_observations"] == 10
|
|
141
|
+
and d["action_resolution"] is False)
|
|
142
|
+
sys.exit(0 if ok else 1)
|
|
143
|
+
PYEOF
|
|
144
|
+
|
|
145
|
+
# --- (2d) shapes where a result could carry no usable verdict at all ----------
|
|
146
|
+
# The resolution loop reads every result's verdict list. Review could not check
|
|
147
|
+
# from the bounded packet that a list is always present, so the shapes that would
|
|
148
|
+
# expose an absent one are pinned here: a single-replica run, and a task whose
|
|
149
|
+
# every replica fails. Neither may crash, and neither may report itself
|
|
150
|
+
# actionable.
|
|
151
|
+
: > "$TMP/count"
|
|
152
|
+
out_one="$(RESOLUTION_COUNT_FILE="$TMP/count" ruby "$EVAL_SCRIPT" "$REPO" --replicas 1 --json "$TMP/one.json" 2>&1)" \
|
|
153
|
+
|| fail "runner exited non-zero on a single-replica run:\n$out_one"
|
|
154
|
+
grep -q '"action_resolution": false' "$TMP/one.json" \
|
|
155
|
+
|| fail "a single-replica run is below the floor and must not be actionable"
|
|
156
|
+
python3 - "$TMP/one.json" <<'PYEOF' || fail "every result of a single-replica run must carry its own verdict list and count"
|
|
157
|
+
import json,sys
|
|
158
|
+
d=json.load(open(sys.argv[1]))
|
|
159
|
+
ok = bool(d["results"]) and all(
|
|
160
|
+
isinstance(r.get("verdicts"), list) and len(r["verdicts"]) == 1
|
|
161
|
+
and r["valid_observations"] <= 1 and r["actionable"] is False
|
|
162
|
+
for r in d["results"])
|
|
163
|
+
sys.exit(0 if ok else 1)
|
|
164
|
+
PYEOF
|
|
165
|
+
|
|
166
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
167
|
+
#!/usr/bin/env bash
|
|
168
|
+
cat >/dev/null
|
|
169
|
+
printf 'never json\n'
|
|
170
|
+
EOF
|
|
171
|
+
chmod +x "$FAKE_BIN/claude"
|
|
172
|
+
# Every task fails every replica: the runner reports the wholly-unmeasured run and
|
|
173
|
+
# exits 3 by its documented contract, and the report must still be well formed.
|
|
174
|
+
# `set -e` would kill the suite on the non-zero exit before it could be read, so
|
|
175
|
+
# the status is captured in the same command that produces it.
|
|
176
|
+
rc=0
|
|
177
|
+
ruby "$EVAL_SCRIPT" "$REPO" --replicas 2 --json "$TMP/allfail.json" >/dev/null 2>&1 || rc=$?
|
|
178
|
+
[ "$rc" = "3" ] || fail "a wholly failed run must exit 3 by the runner's contract, got $rc"
|
|
179
|
+
python3 - "$TMP/allfail.json" <<'PYEOF' || fail "a wholly failed run must still report zero valid observations and no actionable case"
|
|
180
|
+
import json,sys
|
|
181
|
+
d=json.load(open(sys.argv[1]))
|
|
182
|
+
ok = bool(d["results"]) and d["action_resolution"] is False and d["min_valid_observations"] == 0 \
|
|
183
|
+
and all(r["valid_observations"] == 0 and r["actionable"] is False and isinstance(r["verdicts"], list)
|
|
184
|
+
for r in d["results"])
|
|
185
|
+
sys.exit(0 if ok else 1)
|
|
186
|
+
PYEOF
|
|
187
|
+
|
|
188
|
+
# --- (2e) a parseable answer naming no catalog skill is not an observation ------
|
|
189
|
+
# Review could not see from the packet whether a well-formed JSON verdict whose
|
|
190
|
+
# selected_skill is not in the catalog counts toward the floor. It must not: it
|
|
191
|
+
# is absence of evidence, not evidence against the route, so it is an ERROR
|
|
192
|
+
# verdict and the case it belongs to stays short of the floor.
|
|
193
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
194
|
+
#!/usr/bin/env bash
|
|
195
|
+
cat >/dev/null
|
|
196
|
+
printf '{"selected_skill":"skill-that-does-not-exist","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'
|
|
197
|
+
EOF
|
|
198
|
+
chmod +x "$FAKE_BIN/claude"
|
|
199
|
+
rc=0
|
|
200
|
+
ruby "$EVAL_SCRIPT" "$REPO" --replicas 10 --json "$TMP/badsel.json" >/dev/null 2>&1 || rc=$?
|
|
201
|
+
python3 - "$TMP/badsel.json" <<'PYEOF' || fail "a parseable verdict naming a non-catalog skill must count as an error, not a usable observation"
|
|
202
|
+
import json,sys
|
|
203
|
+
d=json.load(open(sys.argv[1]))
|
|
204
|
+
ok = d["action_resolution"] is False and d["min_valid_observations"] == 0 \
|
|
205
|
+
and all(r["valid_observations"] == 0 and r["actionable"] is False for r in d["results"]) \
|
|
206
|
+
and all(v["status"] == "ERROR" for r in d["results"] for v in r["verdicts"])
|
|
207
|
+
sys.exit(0 if ok else 1)
|
|
208
|
+
PYEOF
|
|
209
|
+
|
|
210
|
+
# --- (2f) a skill the prompt never offered is not selectable ---------------------
|
|
211
|
+
# A directory with a SKILL.md but no usable description is filtered out of the
|
|
212
|
+
# prompt catalog. If the selectable-name list were built separately it could still
|
|
213
|
+
# admit that name, and a verdict naming it would count. Both must come from one
|
|
214
|
+
# filtered list, so selecting the unoffered skill is an ERROR like any other
|
|
215
|
+
# non-catalog name.
|
|
216
|
+
mkdir -p "$REPO/skills/ghost-skill"
|
|
217
|
+
printf -- '---\ndescription: ""\n---\n# Ghost\n' > "$REPO/skills/ghost-skill/SKILL.md"
|
|
218
|
+
git -C "$REPO" add -A && git -C "$REPO" commit -qm "ghost skill with empty description"
|
|
219
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
220
|
+
#!/usr/bin/env bash
|
|
221
|
+
cat >/dev/null
|
|
222
|
+
printf '{"selected_skill":"ghost-skill","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'
|
|
223
|
+
EOF
|
|
224
|
+
chmod +x "$FAKE_BIN/claude"
|
|
225
|
+
rc=0
|
|
226
|
+
ruby "$EVAL_SCRIPT" "$REPO" --replicas 10 --json "$TMP/ghost.json" >/dev/null 2>&1 || rc=$?
|
|
227
|
+
python3 - "$TMP/ghost.json" <<'PYEOF' || fail "a verdict naming a skill the prompt never offered must be an error, not a usable observation"
|
|
228
|
+
import json,sys
|
|
229
|
+
d=json.load(open(sys.argv[1]))
|
|
230
|
+
ok = d["action_resolution"] is False and all(v["status"]=="ERROR" for r in d["results"] for v in r["verdicts"])
|
|
231
|
+
sys.exit(0 if ok else 1)
|
|
232
|
+
PYEOF
|
|
233
|
+
|
|
234
|
+
# --- (3) the two sides of the number must agree -------------------------------
|
|
235
|
+
# The rule is only as good as the agreement between the reference that states it
|
|
236
|
+
# and the executable that enforces it. Compare the numbers themselves rather
|
|
237
|
+
# than pinning one literal in two places.
|
|
238
|
+
exe_min="$(grep -oE 'ACTION_RESOLUTION_MIN_REPLICAS = [0-9]+' "$EVAL_SCRIPT" | grep -oE '[0-9]+$' | sort -u)"
|
|
239
|
+
[ -n "$exe_min" ] || fail "could not read ACTION_RESOLUTION_MIN_REPLICAS from $EVAL_SCRIPT"
|
|
240
|
+
[ "$(printf '%s\n' "$exe_min" | wc -l | tr -d ' ')" = "1" ] \
|
|
241
|
+
|| fail "ACTION_RESOLUTION_MIN_REPLICAS is defined with more than one value: $exe_min"
|
|
242
|
+
doc_min="$(grep -oE 'replicas < [0-9]+' "$DOC" | grep -oE '[0-9]+$' | sort -u)"
|
|
243
|
+
[ -n "$doc_min" ] || fail "$DOC does not state the replicas floor as \`replicas < N\`"
|
|
244
|
+
[ "$doc_min" = "$exe_min" ] \
|
|
245
|
+
|| fail "doc states floor $doc_min but the runner enforces $exe_min — they must be one number"
|
|
246
|
+
|
|
247
|
+
# The prose obligation (">=N valid observations before acting") must name the
|
|
248
|
+
# same number too, so a reader following the sentence and a consumer reading the
|
|
249
|
+
# report are held to one bar.
|
|
250
|
+
grep -qF "≥${exe_min} 个有效观测" "$DOC" \
|
|
251
|
+
|| fail "$DOC must state the per-case obligation with the same floor (≥${exe_min} 个有效观测)"
|
|
252
|
+
|
|
253
|
+
echo "PASS: eval-routing-bank resolution signal (banner, report field, doc/executor agreement)"
|
|
@@ -139,28 +139,52 @@ git -C "$REPO_ROOT" show "$BASELINE_GATE_COMMIT:$GATE_PATH" > "$BASELINE_GATE" 2
|
|
|
139
139
|
exit 1
|
|
140
140
|
}
|
|
141
141
|
|
|
142
|
-
# EXPECTED DIVERGENCES
|
|
143
|
-
# rewrite — the repair for a round that merged with this gate red — so the owner
|
|
144
|
-
# change sits below their base and the row cites an owner the range does not
|
|
145
|
-
# touch. The candidate refuses that shape by design and deliberately offers no
|
|
146
|
-
# author-declared escape, so these four historical ranges diverge. The gate is
|
|
147
|
-
# diff-scoped and never re-judges landed history, so nothing operational depends
|
|
148
|
-
# on them; this differential is the only thing that replays them.
|
|
142
|
+
# EXPECTED DIVERGENCES, two named classes, both "newly refused".
|
|
149
143
|
#
|
|
150
|
-
#
|
|
151
|
-
#
|
|
152
|
-
#
|
|
153
|
-
#
|
|
154
|
-
#
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
144
|
+
# `impact_chain_row_vouches_for_unchanged_owner` (four points): landings that
|
|
145
|
+
# back-filled a ledger row by corrective rewrite — the repair for a round that
|
|
146
|
+
# merged with this gate red — so the owner change sits below their base and the
|
|
147
|
+
# row cites an owner the range does not touch. The candidate refuses that shape by
|
|
148
|
+
# design and deliberately offers no author-declared escape.
|
|
149
|
+
#
|
|
150
|
+
# `impact_chain_gate_missing` (six points): merges from before CI judged the
|
|
151
|
+
# branch head (register row on the checkout ref binding, 2026-08-25). The gate now
|
|
152
|
+
# expands a merge git rebuilds from its parents into the branch's own rounds, so a
|
|
153
|
+
# merge is judged exactly as its branch was; these six branches carry owner work
|
|
154
|
+
# outside the round that declares it (a row appended before the work, or work
|
|
155
|
+
# after the last append) and the baseline accepted them only through the
|
|
156
|
+
# collapsed merged view, which no longer exists as a distinct verdict. Three of
|
|
157
|
+
# the six are refused on their own branch head by the baseline gate too; the other
|
|
158
|
+
# three are promotions or syncs whose second parent is the integration branch,
|
|
159
|
+
# where the same shapes sit one merge deeper.
|
|
160
|
+
#
|
|
161
|
+
# The gate is diff-scoped and never re-judges landed history, so nothing
|
|
162
|
+
# operational depends on these points; this differential is the only thing that
|
|
163
|
+
# replays them. Each entry is constrained to ONE direction and ONE diagnostic. A
|
|
164
|
+
# blanket "any mismatch at this SHA is fine" would also swallow the opposite
|
|
165
|
+
# direction — a loosening — which is the failure this whole suite exists to
|
|
166
|
+
# catch. Entries are named individually, never matched by pattern, and an entry
|
|
167
|
+
# that stops diverging is reported as stale rather than tolerated.
|
|
168
|
+
EXPECTED_DIVERGENCES="
|
|
169
|
+
f03b1140f:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
170
|
+
93d09c563:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
171
|
+
9f233728a:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
172
|
+
046612652:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
173
|
+
b9de13869:refused:impact_chain_gate_missing
|
|
174
|
+
8cea35e6d:refused:impact_chain_gate_missing
|
|
175
|
+
95f06b2e6:refused:impact_chain_gate_missing
|
|
176
|
+
c0561c74e:refused:impact_chain_gate_missing
|
|
177
|
+
fad480296:refused:impact_chain_gate_missing
|
|
178
|
+
90ec533e1:refused:impact_chain_gate_missing
|
|
179
|
+
"
|
|
180
|
+
expected_divergence() { # <full sha> <direction> <candidate output>; prints the matched token
|
|
181
|
+
local short="${1:0:9}" direction="$2" out="$3" entry sha dir token
|
|
182
|
+
case "$direction" in "newly refused") direction=refused ;; "newly accepted") direction=accepted ;; esac
|
|
183
|
+
for entry in $EXPECTED_DIVERGENCES; do
|
|
184
|
+
sha="${entry%%:*}"; token="${entry##*:}"; dir="${entry#*:}"; dir="${dir%%:*}"
|
|
185
|
+
[ "$sha" = "$short" ] || continue
|
|
186
|
+
[ "$dir" = "$direction" ] || return 1
|
|
187
|
+
case "$out" in *"$token"*) printf '%s' "$token"; return 0 ;; *) return 1 ;; esac
|
|
164
188
|
done
|
|
165
189
|
return 1
|
|
166
190
|
}
|
|
@@ -170,7 +194,7 @@ expected_divergence() { # <full sha> <direction> <candidate output>
|
|
|
170
194
|
# them, so the exemptions are stale and the run would pass while silently failing
|
|
171
195
|
# the staleness check it never reaches.
|
|
172
196
|
if cmp -s "$BASELINE_GATE" "$CANDIDATE_GATE"; then
|
|
173
|
-
if [ -n "$(printf '%s' "$
|
|
197
|
+
if [ -n "$(printf '%s' "$EXPECTED_DIVERGENCES" | tr -d '[:space:]')" ]; then
|
|
174
198
|
echo "FAIL: baseline and candidate are byte-identical, yet expected divergences are configured" >&2
|
|
175
199
|
echo " identical gates cannot diverge — the exemptions are stale and must be removed" >&2
|
|
176
200
|
exit 1
|
|
@@ -261,8 +285,8 @@ for point in $INTEGRATION_POINTS; do
|
|
|
261
285
|
direction="newly accepted"
|
|
262
286
|
fi
|
|
263
287
|
if [ -n "$direction" ]; then
|
|
264
|
-
if expected_divergence "$point" "$direction" "$candidate_out"; then
|
|
265
|
-
flag=" (expected divergence:
|
|
288
|
+
if matched_token="$(expected_divergence "$point" "$direction" "$candidate_out")"; then
|
|
289
|
+
flag=" (expected divergence: $matched_token, $direction)"
|
|
266
290
|
expected_seen=$((expected_seen + 1))
|
|
267
291
|
else
|
|
268
292
|
flag=" <== VERDICT MISMATCH: $direction"
|
|
@@ -412,7 +436,7 @@ if [ "$total_failures" -gt 0 ]; then
|
|
|
412
436
|
fi
|
|
413
437
|
# An expected divergence that stops diverging means the exemption is stale and
|
|
414
438
|
# should be removed, so it is reported rather than silently tolerated.
|
|
415
|
-
expected_total="$(for e in $
|
|
439
|
+
expected_total="$(for e in $EXPECTED_DIVERGENCES; do echo "$e"; done | wc -l | tr -d ' ')"
|
|
416
440
|
if [ "$expected_seen" != "$expected_total" ]; then
|
|
417
441
|
echo "FAIL: $expected_seen of $expected_total expected divergences actually diverged" >&2
|
|
418
442
|
echo " an exemption that no longer fires is stale — remove it" >&2
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Regression for reference-access-census.sh: synthetic transcripts only, no
|
|
3
|
+
# host logs are read. Proves (1) per-file session counts and the touching
|
|
4
|
+
# denominator, (2) the unevaluated sentinel when no transcript exists (never a
|
|
5
|
+
# zero table), (3) the privacy contract — output carries no transcript text,
|
|
6
|
+
# no absolute log path, no session id, (2b) input errors (an unreadable
|
|
7
|
+
# transcript) withhold the table with exit 2 instead of printing zeros or the ok
|
|
8
|
+
# token, (4) usage errors exit 2, and (5) the
|
|
9
|
+
# ARG_MAX regression: a candidate set larger than one xargs batch still counts
|
|
10
|
+
# (the first version silently reported 0 on a large window).
|
|
11
|
+
set -euo pipefail
|
|
12
|
+
|
|
13
|
+
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
14
|
+
CENSUS="$HERE/reference-access-census.sh"
|
|
15
|
+
TMP="$(mktemp -d)"; trap 'rm -rf "$TMP"' EXIT
|
|
16
|
+
fail() { echo "FAIL: $*" >&2; exit 1; }
|
|
17
|
+
|
|
18
|
+
# --- fixture repo: one skill package with an entrypoint and three references ---
|
|
19
|
+
REPO="$TMP/repo"; mkdir -p "$REPO/skills/demo-skill/references"
|
|
20
|
+
git -C "$TMP" init -q repo 2>/dev/null || git init -q "$REPO"
|
|
21
|
+
printf '# demo\n' > "$REPO/skills/demo-skill/SKILL.md"
|
|
22
|
+
for r in alpha beta gamma; do printf '# %s\n' "$r" > "$REPO/skills/demo-skill/references/$r.md"; done
|
|
23
|
+
git -C "$REPO" add -A && git -C "$REPO" -c user.email=t@t -c user.name=t commit -qm fixture
|
|
24
|
+
|
|
25
|
+
# --- fixture transcripts: two log roots, distinct mention shapes ---
|
|
26
|
+
LOGS="$TMP/logs-claude"; mkdir -p "$LOGS/proj-a" "$TMP/logs-codex/2026/09"
|
|
27
|
+
SECRET_PROMPT="user-private-prompt-text-7f3a"
|
|
28
|
+
SESSION_ID="session-9c1e2d3b-secret"
|
|
29
|
+
# session 1 (claude shape): reads alpha and beta via a Read tool call
|
|
30
|
+
printf '{"type":"tool_use","name":"Read","input":{"file_path":"/abs/checkout/skills/demo-skill/references/alpha.md"},"prompt":"%s","session":"%s"}\n' "$SECRET_PROMPT" "$SESSION_ID" > "$LOGS/proj-a/one.jsonl"
|
|
31
|
+
printf '{"type":"tool_use","name":"Read","input":{"file_path":"/abs/checkout/skills/demo-skill/references/beta.md"}}\n' >> "$LOGS/proj-a/one.jsonl"
|
|
32
|
+
# session 2 (codex shape): shell command mentioning alpha twice (must count once)
|
|
33
|
+
printf '{"cmd":"sed -n 1,40p skills/demo-skill/references/alpha.md && cat skills/demo-skill/references/alpha.md"}\n' > "$TMP/logs-codex/2026/09/rollout-two.jsonl"
|
|
34
|
+
# session 3: touches the package (SKILL.md) but no reference
|
|
35
|
+
printf '{"cmd":"cat skills/demo-skill/SKILL.md"}\n' > "$TMP/logs-codex/2026/09/rollout-three.jsonl"
|
|
36
|
+
# session 4: unrelated transcript (must not count as touching)
|
|
37
|
+
printf '{"cmd":"ls skills/other-skill/"}\n' > "$LOGS/proj-a/four.jsonl"
|
|
38
|
+
|
|
39
|
+
# deterministic, differing mtimes so the last_touched column can be asserted exactly (newest wins)
|
|
40
|
+
touch -t "$(date -v-3d +%Y%m%d%H%M 2>/dev/null || date -d '3 days ago' +%Y%m%d%H%M)" "$LOGS/proj-a/one.jsonl"
|
|
41
|
+
touch -t "$(date -v-1d +%Y%m%d%H%M 2>/dev/null || date -d '1 day ago' +%Y%m%d%H%M)" "$TMP/logs-codex/2026/09/rollout-two.jsonl"
|
|
42
|
+
NEWEST="$(date -v-1d +%Y-%m-%d 2>/dev/null || date -d '1 day ago' +%Y-%m-%d)"
|
|
43
|
+
OLDER="$(date -v-3d +%Y-%m-%d 2>/dev/null || date -d '3 days ago' +%Y-%m-%d)"
|
|
44
|
+
# a stale transcript (older than the window) that mentions tracked files must not move any count, share, or date
|
|
45
|
+
printf '{"cmd":"cat skills/demo-skill/references/alpha.md skills/demo-skill/references/gamma.md"}\n' > "$LOGS/proj-a/stale.jsonl"
|
|
46
|
+
touch -t "$(date -v-45d +%Y%m%d%H%M 2>/dev/null || date -d '45 days ago' +%Y%m%d%H%M)" "$LOGS/proj-a/stale.jsonl"
|
|
47
|
+
run() { bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS,$TMP/logs-codex"; }
|
|
48
|
+
out="$(run)" || fail "census exited non-zero on a valid fixture"
|
|
49
|
+
|
|
50
|
+
# (1) counts and denominator
|
|
51
|
+
printf '%s\n' "$out" | grep -qF 'transcripts=4 sessions_touching_package=3' || fail "denominator wrong:\n$out"
|
|
52
|
+
printf '%s\n' "$out" | grep -qF "references/alpha.md | 2 | $NEWEST | 66%" || fail "alpha should count 2 sessions with the newest touching date $NEWEST:\n$out"
|
|
53
|
+
printf '%s\n' "$out" | grep -qF "references/beta.md | 1 | $OLDER | 33%" || fail "beta should count 1 session dated $OLDER:\n$out"
|
|
54
|
+
printf '%s\n' "$out" | grep -qE '^references/gamma\.md \| 0 \| - \| 0%' || fail "gamma should count 0 with no date:\n$out"
|
|
55
|
+
printf '%s\n' "$out" | grep -qE '^SKILL\.md \| 1 \|' || fail "SKILL.md should count 1 session:\n$out"
|
|
56
|
+
[ "$(printf '%s\n' "$out" | tail -1)" = "reference_access_census_ok" ] || fail "last token must be the ok marker"
|
|
57
|
+
|
|
58
|
+
# (3) privacy contract
|
|
59
|
+
printf '%s\n' "$out" | grep -qF "$SECRET_PROMPT" && fail "transcript text leaked into census output"
|
|
60
|
+
printf '%s\n' "$out" | grep -qF "$SESSION_ID" && fail "session id leaked into census output"
|
|
61
|
+
printf '%s\n' "$out" | grep -qF "$LOGS" && fail "absolute log path leaked into census output"
|
|
62
|
+
printf '%s\n' "$out" | grep -qF "/abs/checkout" && fail "absolute checkout path leaked into census output"
|
|
63
|
+
|
|
64
|
+
# (2) unevaluated sentinel: no transcripts at all
|
|
65
|
+
EMPTY="$TMP/empty-logs"; mkdir -p "$EMPTY"
|
|
66
|
+
eout="$(bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$EMPTY")" || fail "empty-log run must exit 0"
|
|
67
|
+
[ "$(printf '%s\n' "$eout" | tail -1)" = "reference_access_census_unevaluated: no transcripts within 30d under 1 log root(s); counts withheld" ] || fail "missing unevaluated sentinel:\n$eout"
|
|
68
|
+
printf '%s\n' "$eout" | grep -qE '\| 0 \|' && fail "an unevaluated run must not print a zero table"
|
|
69
|
+
printf '%s\n' "$eout" | grep -qF "$EMPTY" && fail "unevaluated sentinel leaked the log root path"
|
|
70
|
+
printf '%s\n' "$eout" | grep -qF "$TMP" && fail "unevaluated sentinel leaked an absolute path"
|
|
71
|
+
|
|
72
|
+
# (2b) input errors withhold the table: an unreadable transcript is exit 2 + unevaluated, never zeros or ok
|
|
73
|
+
UNREAD="$TMP/logs-unreadable"; mkdir -p "$UNREAD"
|
|
74
|
+
printf '{"cmd":"cat skills/demo-skill/references/alpha.md"}\n' > "$UNREAD/readable.jsonl"
|
|
75
|
+
printf '{"cmd":"cat skills/demo-skill/references/beta.md"}\n' > "$UNREAD/locked.jsonl"
|
|
76
|
+
chmod 000 "$UNREAD/locked.jsonl"
|
|
77
|
+
if [ -r "$UNREAD/locked.jsonl" ]; then
|
|
78
|
+
echo "note: chmod 000 is readable here (root); skipping the unreadable-input leg"
|
|
79
|
+
else
|
|
80
|
+
uout="$(bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$UNREAD" 2>"$TMP/unread.stderr")" && fail "unreadable transcript must exit non-zero"
|
|
81
|
+
urc=$?
|
|
82
|
+
uout="$(bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$UNREAD" 2>/dev/null || true)"
|
|
83
|
+
bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$UNREAD" >/dev/null 2>&1 || [ $? -eq 2 ] || fail "unreadable transcript must exit 2"
|
|
84
|
+
printf '%s\n' "$uout" | grep -qE 'reference_access_census_unevaluated: [0-9]+ input error\(s\)' || fail "unreadable transcript must withhold counts as unevaluated:\n$uout"
|
|
85
|
+
printf '%s\n' "$uout" | grep -qF 'reference_access_census_ok' && fail "ok token printed after an input error"
|
|
86
|
+
printf '%s\n' "$uout" | grep -qE '\| [0-9]+ \|' && fail "a table was printed after an input error"
|
|
87
|
+
printf '%s\n' "$uout" | grep -qF "$UNREAD" && fail "input-error sentinel leaked the log path"
|
|
88
|
+
fi
|
|
89
|
+
chmod 644 "$UNREAD/locked.jsonl"
|
|
90
|
+
|
|
91
|
+
# (4) usage errors exit 2
|
|
92
|
+
if bash "$CENSUS" --repo-root "$REPO" --skill no-such-skill --logs "$LOGS" >/dev/null 2>&1; then fail "unknown skill must exit 2"; fi
|
|
93
|
+
bash "$CENSUS" --repo-root "$REPO" --skill no-such-skill --logs "$LOGS" >/dev/null 2>&1 || [ $? -eq 2 ] || fail "unknown skill must exit 2 exactly"
|
|
94
|
+
bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days abc --logs "$LOGS" >/dev/null 2>&1 || [ $? -eq 2 ] || fail "non-integer --days must exit 2"
|
|
95
|
+
|
|
96
|
+
# (5) ARG_MAX regression: the candidate path list must exceed THIS host's exec limit
|
|
97
|
+
# (getconf ARG_MAX) so that a `$(cat candidates)` expansion — the first version's shape,
|
|
98
|
+
# which silently reported 0 on a real 60-day window — cannot fit in one exec. The fixture
|
|
99
|
+
# size is derived from the limit, never a fixed byte figure; a host whose limit exceeds the
|
|
100
|
+
# fixture bound runs only the counting leg and says so. Verified RED by applying that exact
|
|
101
|
+
# mutation to a disposable copy of the script.
|
|
102
|
+
DPAD=""; i=0; while [ $i -lt 150 ]; do DPAD="${DPAD}d"; i=$((i+1)); done
|
|
103
|
+
BIG="$TMP/logs-big/$DPAD"; mkdir -p "$BIG"
|
|
104
|
+
PAD=""; i=0; while [ $i -lt 40 ]; do PAD="${PAD}x"; i=$((i+1)); done
|
|
105
|
+
arg_max="$(getconf ARG_MAX 2>/dev/null || echo 2097152)"
|
|
106
|
+
per_path=$(( ${#BIG} + 60 ))
|
|
107
|
+
need=$(( (arg_max + 262144) / per_path + 1 ))
|
|
108
|
+
[ "$need" -lt 11000 ] && need=11000
|
|
109
|
+
probe=1
|
|
110
|
+
if [ "$need" -gt 60000 ]; then
|
|
111
|
+
echo "note: ARG_MAX=$arg_max would need $need fixture files; the E2BIG probe is skipped on this host and only the counting leg runs"
|
|
112
|
+
need=11000; probe=0
|
|
113
|
+
fi
|
|
114
|
+
( cd "$BIG" && i=1; while [ $i -le $need ]; do : > "s$i-$PAD.jsonl"; i=$((i+1)); done )
|
|
115
|
+
find "$BIG" -type f -name '*.jsonl' > "$TMP/big-list"
|
|
116
|
+
inv_bytes=$(wc -c < "$TMP/big-list" | tr -d ' ')
|
|
117
|
+
if [ "$probe" -eq 1 ]; then
|
|
118
|
+
erc=0; bash -c 'grep -lF "skills/demo-skill/" $(cat "$1") >/dev/null 2>&1' _ "$TMP/big-list" 2>"$TMP/e2big.err" || erc=$?
|
|
119
|
+
[ "$erc" -eq 126 ] || grep -qi "argument list too long" "$TMP/e2big.err" || fail "ARG_MAX fixture ($inv_bytes bytes, ARG_MAX=$arg_max) does not exceed this host's exec limit (rc=$erc), so the leg cannot kill the single-exec mutation"
|
|
120
|
+
fi
|
|
121
|
+
for i in 7 5208 10999; do printf '{"cmd":"cat skills/demo-skill/references/gamma.md"}\n' > "$BIG/s$i-$PAD.jsonl"; done
|
|
122
|
+
bout="$(bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$BIG")" || fail "big run exited non-zero"
|
|
123
|
+
printf '%s\n' "$bout" | grep -qF "transcripts=$need sessions_touching_package=3" || fail "large candidate set miscounted (ARG_MAX regression):\n$bout"
|
|
124
|
+
printf '%s\n' "$bout" | grep -qE '^references/gamma\.md \| 3 \|' || fail "gamma should count 3 across xargs batches:\n$bout"
|
|
125
|
+
|
|
126
|
+
# (6) a flag without its value is a usage error (exit 2), not an unbound-variable abort
|
|
127
|
+
bash "$CENSUS" --repo-root "$REPO" --skill >/dev/null 2>&1 || [ $? -eq 2 ] || fail "flag without value must exit 2"
|
|
128
|
+
|
|
129
|
+
# (7) an unknown argument never echoes its value (an absolute path passed by mistake stays private)
|
|
130
|
+
uerr="$(bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --logs "$LOGS" "/private/mistyped/path" 2>&1 >/dev/null || true)"
|
|
131
|
+
bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --logs "$LOGS" "/private/mistyped/path" >/dev/null 2>&1 || [ $? -eq 2 ] || fail "unknown argument must exit 2"
|
|
132
|
+
printf '%s\n' "$uerr" | grep -qF "/private/mistyped/path" && fail "unknown-argument error echoed the caller's value"
|
|
133
|
+
printf '%s\n' "$uerr" | grep -qF "unknown argument" || fail "unknown-argument error must still say what went wrong"
|
|
134
|
+
|
|
135
|
+
# (8) an explicitly supplied log root that does not exist is an input error — alone or mixed with a valid root
|
|
136
|
+
for roots in "$TMP/no-such-root" "$TMP/no-such-root,$LOGS"; do
|
|
137
|
+
mout="$(bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$roots" 2>/dev/null || true)"
|
|
138
|
+
bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$roots" >/dev/null 2>&1 || [ $? -eq 2 ] || fail "missing explicit log root must exit 2 ($roots)"
|
|
139
|
+
printf '%s\n' "$mout" | grep -qF 'reference_access_census_unevaluated: 1 input error(s)' || fail "missing explicit log root must withhold counts:\n$mout"
|
|
140
|
+
printf '%s\n' "$mout" | grep -qF 'reference_access_census_ok' && fail "ok token printed with a missing explicit log root"
|
|
141
|
+
printf '%s\n' "$mout" | grep -qE '\| [0-9]+ \|' && fail "a table was printed with a missing explicit log root"
|
|
142
|
+
printf '%s\n' "$mout" | grep -qF "$TMP" && fail "missing-root sentinel leaked a path"
|
|
143
|
+
done
|
|
144
|
+
# a missing DEFAULT root stays normal: with no --logs and an empty HOME the result is the no-transcript sentinel, exit 0
|
|
145
|
+
dout="$(HOME="$TMP/empty-home" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30)" || fail "absent default roots must not be an input error"
|
|
146
|
+
[ "$(printf '%s\n' "$dout" | tail -1)" = "reference_access_census_unevaluated: no transcripts within 30d under 2 log root(s); counts withheld" ] || fail "absent default roots must yield the no-transcript sentinel:\n$dout"
|
|
147
|
+
|
|
148
|
+
# (10) overlapping log roots (a root and its own subtree, or a root listed twice) count each transcript once
|
|
149
|
+
oout="$(bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$TMP/logs-codex,$TMP/logs-codex/2026,$TMP/logs-codex")" || fail "overlapping roots run exited non-zero"
|
|
150
|
+
printf '%s\n' "$oout" | grep -qF 'transcripts=2 sessions_touching_package=2' || fail "overlapping roots must not double-count transcripts:\n$oout"
|
|
151
|
+
printf '%s\n' "$oout" | grep -qE '^references/alpha\.md \| 1 \|' || fail "alpha must count once under overlapping roots:\n$oout"
|
|
152
|
+
|
|
153
|
+
# (11) an invalid --skill value is never echoed back (a mistyped private identifier stays private)
|
|
154
|
+
kerr="$(bash "$CENSUS" --repo-root "$REPO" --skill "acme-internal-secret-project" --logs "$LOGS" 2>&1 >/dev/null || true)"
|
|
155
|
+
printf '%s\n' "$kerr" | grep -qF "acme-internal-secret-project" && fail "unknown --skill error echoed the caller's value"
|
|
156
|
+
printf '%s\n' "$kerr" | grep -qF "usage_error" || fail "unknown --skill must still be reported as a usage error"
|
|
157
|
+
|
|
158
|
+
# (12) a transcript whose name contains a newline is one candidate, not two broken ones
|
|
159
|
+
NL="$TMP/logs-newline"; mkdir -p "$NL"; printf '{"cmd":"cat skills/demo-skill/references/beta.md"}\n' > "$NL/odd
|
|
160
|
+
name.jsonl"
|
|
161
|
+
nout="$(bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$NL")" || fail "newline-named transcript run exited non-zero"
|
|
162
|
+
printf '%s\n' "$nout" | grep -qF 'transcripts=1 sessions_touching_package=1' || fail "newline in a transcript name must not split the candidate:\n$nout"
|
|
163
|
+
|
|
164
|
+
# (13) HOME unset and no --logs: no default roots, the no-transcript sentinel, exit 0 (never an unbound-variable abort)
|
|
165
|
+
hout="$(env -u HOME bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30)" || fail "unset HOME must not abort the census"
|
|
166
|
+
[ "$(printf '%s\n' "$hout" | tail -1)" = "reference_access_census_unevaluated: no transcripts within 30d under 0 log root(s); counts withheld" ] || fail "unset HOME must yield the no-transcript sentinel:\n$hout"
|
|
167
|
+
|
|
168
|
+
# (14) a transcript-read error forced by a command shim (root-independent twin of the chmod leg)
|
|
169
|
+
GSHIM="$TMP/gshim"; mkdir -p "$GSHIM"
|
|
170
|
+
printf '#!/usr/bin/env bash\necho "grep: transcript: Input/output error" >&2\nexit 2\n' > "$GSHIM/grep"; chmod +x "$GSHIM/grep"
|
|
171
|
+
gout="$(PATH="$GSHIM:$PATH" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS" 2>/dev/null || true)"
|
|
172
|
+
PATH="$GSHIM:$PATH" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS" >/dev/null 2>&1 || [ $? -eq 2 ] || fail "a grep read error must exit 2"
|
|
173
|
+
printf '%s\n' "$gout" | grep -qF 'reference_access_census_unevaluated:' || fail "a grep read error must withhold counts:\n$gout"
|
|
174
|
+
printf '%s\n' "$gout" | grep -qF 'reference_access_census_ok' && fail "ok token printed after a grep read error"
|
|
175
|
+
|
|
176
|
+
# (15) a grep that fails silently (exit 2, no stderr) is still an input error: withhold, exit 2, no ok token
|
|
177
|
+
QSHIM="$TMP/qshim"; mkdir -p "$QSHIM"
|
|
178
|
+
printf '#!/usr/bin/env bash\nexit 2\n' > "$QSHIM/grep"; chmod +x "$QSHIM/grep"
|
|
179
|
+
qout="$(PATH="$QSHIM:$PATH" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS" 2>/dev/null || true)"
|
|
180
|
+
PATH="$QSHIM:$PATH" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS" >/dev/null 2>&1 || [ $? -eq 2 ] || fail "a silent grep failure must exit 2"
|
|
181
|
+
printf '%s\n' "$qout" | grep -qF 'reference_access_census_unevaluated:' || fail "a silent grep failure must withhold counts:\n$qout"
|
|
182
|
+
printf '%s\n' "$qout" | grep -qF 'reference_access_census_ok' && fail "ok token printed after a silent grep failure"
|
|
183
|
+
|
|
184
|
+
# (16) an inherited errexit (caller exported SHELLOPTS=errexit) must not turn an all-no-match batch into an input error
|
|
185
|
+
NM="$TMP/logs-nomatch"; mkdir -p "$NM"; printf '{"cmd":"ls skills/other-skill/"}\n' > "$NM/quiet.jsonl"
|
|
186
|
+
mout2="$(env SHELLOPTS=errexit bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$NM")" || fail "inherited errexit must not abort a valid census"
|
|
187
|
+
printf '%s\n' "$mout2" | grep -qF 'transcripts=1 sessions_touching_package=0' || fail "all-no-match batch under inherited errexit must be a valid zero census:\n$mout2"
|
|
188
|
+
[ "$(printf '%s\n' "$mout2" | tail -1)" = "reference_access_census_ok" ] || fail "inherited errexit must not withhold a valid census"
|
|
189
|
+
|
|
190
|
+
# (17) a find that fails silently (exit 2, no stderr) is an input error: withhold, exit 2, no ok token
|
|
191
|
+
FSHIM="$TMP/fshim"; mkdir -p "$FSHIM"
|
|
192
|
+
printf '#!/usr/bin/env bash\nexit 2\n' > "$FSHIM/find"; chmod +x "$FSHIM/find"
|
|
193
|
+
fout="$(PATH="$FSHIM:$PATH" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS" 2>/dev/null || true)"
|
|
194
|
+
frc=0; PATH="$FSHIM:$PATH" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS" >/dev/null 2>&1 || frc=$?
|
|
195
|
+
[ "$frc" -eq 2 ] || fail "a silent find failure must exit 2 (got $frc)"
|
|
196
|
+
printf '%s\n' "$fout" | grep -qE 'reference_access_census_unevaluated: [0-9]+ input error' || fail "a silent find failure must withhold counts as an input error, not the no-transcript sentinel:\n$fout"
|
|
197
|
+
printf '%s\n' "$fout" | grep -qF 'reference_access_census_ok' && fail "ok token printed after a silent find failure"
|
|
198
|
+
|
|
199
|
+
# (9) a failing stat (a transcript vanishing before the timestamp read) withholds the table with exit 2 —
|
|
200
|
+
# deterministic via a PATH shim, independent of filesystem permissions or root
|
|
201
|
+
SHIM="$TMP/shim"; mkdir -p "$SHIM"
|
|
202
|
+
printf '#!/usr/bin/env bash\necho "stat: cannot stat: No such file or directory" >&2\nexit 1\n' > "$SHIM/stat"; chmod +x "$SHIM/stat"
|
|
203
|
+
sout="$(PATH="$SHIM:$PATH" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS,$TMP/logs-codex" 2>/dev/null || true)"
|
|
204
|
+
PATH="$SHIM:$PATH" bash "$CENSUS" --repo-root "$REPO" --skill demo-skill --days 30 --logs "$LOGS,$TMP/logs-codex" >/dev/null 2>&1 || [ $? -eq 2 ] || fail "failing stat must exit 2, not the pipeline status"
|
|
205
|
+
printf '%s\n' "$sout" | grep -qF 'reference_access_census_unevaluated:' || fail "failing stat must print the unevaluated sentinel:\n$sout"
|
|
206
|
+
printf '%s\n' "$sout" | grep -qF 'reference_access_census_ok' && fail "ok token printed after a failing stat"
|
|
207
|
+
printf '%s\n' "$sout" | grep -qE '\| [0-9]+ \|' && fail "a table was printed after a failing stat"
|
|
208
|
+
|
|
209
|
+
echo "test_reference_access_census_ok"
|