@ccoalm/ccl-skills 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/classify_envelope.py +34 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_abort_leak_state_helpers.sh +148 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_classify_envelope.sh +28 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_review_json.sh +7 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate_abort_leak.sh +271 -34
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +8 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +7 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +7 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/recurring-anti-patterns-checklist.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +29 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +28 -29
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +153 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-health.rb +23 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +303 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +222 -91
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_dateless_host.sh +6 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_round_attribution.sh +12 -12
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh +455 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_source_refuted.sh +20 -20
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_liveness_predicate_gate.sh +288 -0
- package/dist/assets/release.json +39 -24
- package/package.json +1 -1
|
@@ -7,10 +7,10 @@ Keep source-specific provenance outside the distributed repository. Public skill
|
|
|
7
7
|
| Field | Required answer |
|
|
8
8
|
| --- | --- |
|
|
9
9
|
| Task or extraction name | Name the reusable workflow or source family with a source-neutral label. |
|
|
10
|
-
| Purpose | State the future failure or
|
|
10
|
+
| Purpose | State the future failure or drift this extraction prevents, or the evidenced success mechanism it preserves. |
|
|
11
11
|
| Scope | List included source classes, target skills, sibling boundaries, and exclusions. |
|
|
12
12
|
| Depth | Record wording cleanup, targeted check, file refresh, artifact inventory, full workflow extraction, or tooling change. |
|
|
13
|
-
|
|
|
13
|
+
| Result analysis | Classify failure/correction, stable success, or unstable/insufficient evidence, then record the matching RCA, success attribution, or observation-only boundary. |
|
|
14
14
|
| Lifecycle impact | Name the affected intake, design, implementation, testing, launch, iteration, onboarding, and documentation stages. |
|
|
15
15
|
| Evidence plan | List source categories and how each is inspected, routed, excluded, or marked unavailable. |
|
|
16
16
|
| Completion standard | Name the scenario, command, review, and source-map evidence required to finish. |
|
|
@@ -22,8 +22,28 @@ Target-output map:
|
|
|
22
22
|
|
|
23
23
|
Required for upstream-owner skill changes:
|
|
24
24
|
|
|
25
|
+
**Declaration fragments (045).** A row's behavior cell is semicolon-delimited. Alongside
|
|
26
|
+
`behavioral-evidence:` / `observed-failure:` / `firing-path:`, a row for a changed upstream owner
|
|
27
|
+
carries `result-class: failure|stable-success|insufficient-evidence`, and — when the round changes
|
|
28
|
+
a routing surface (a SKILL.md frontmatter `description` entry, or `eval/routing-tasks.jsonl`) —
|
|
29
|
+
`bank-evidence: <locator>` or `bank-evidence: downscoped:<token>` with that token recorded in the
|
|
30
|
+
round's spec. `impact-chain-gate.rb` enforces both. The value stays the author's call; its
|
|
31
|
+
**absence** does not, which is the whole point: an omitted claim is one a reviewer cannot refuse.
|
|
32
|
+
A `bank-evidence` locator must not point back into the owner's own package — the change is not
|
|
33
|
+
evidence about itself.
|
|
34
|
+
|
|
35
|
+
**A round is judged by the grammar its own head declares.** The gate looks for this paragraph's
|
|
36
|
+
`result-class:` definition in the ledger at the round's head; a round that predates it is not held
|
|
37
|
+
to it. That is deliberate and is not a grandfather clause keyed on dates or commit ids: adding a
|
|
38
|
+
required field would otherwise retroactively refuse every historical round on replay, which the
|
|
39
|
+
verdict-differential suite correctly reports as a regression. Rounds landing after this paragraph
|
|
40
|
+
carry the obligation; rounds before it were never told.
|
|
41
|
+
|
|
25
42
|
| Upstream rule | Downstream owner | Expected executable behavior | Status (updated, unchanged, routed, or not-applicable) | Evidence |
|
|
26
43
|
| --- | --- | --- | --- | --- |
|
|
44
|
+
| For model-controlled result text, `line start` and `whole-string start` are different trust boundaries: multiline matching lets a later payload line select a transport-auth decision, so the textual fallback permits only leading whitespace from offset zero and any textual preamble fails closed unless a structured transport status independently classifies it | `code-review` | behavioral-evidence: semantic-control; observed-failure: no; firing-path: command:skills/code-review/scripts/test_classify_envelope.sh | updated | A review proposed `re.MULTILINE` for leading-preamble tolerance, but `result_text()` returns only `env["result"]` and does not concatenate transport fields. The candidate therefore keeps whole-string anchoring: leading spaces and line breaks are accepted, while `review preamble` followed by the exact 401 phrase remains a generic error. Scratch mutations separately prove removing the whitespace allowance reds its regression and enabling `re.M` reds the preamble control. Structured `api_error_status=401` remains the vocabulary-independent auth arm. Disposition: `specs/044-review-auth-fallback/plan.md`. |
|
|
45
|
+
| A transport-error text fallback is safe only when it pins the observed transport shape and carries benign near-miss cases; searching generic authentication phrases inside model-controlled result text can turn review content into an auth decision and unnecessarily widen packet egress | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_classify_envelope.sh | updated | Independent review found the broad `authentication failed` / expired-token search could classify an errored review whose payload merely discussed authentication as `auth`. The expression is now anchored to the observed line-start `Failed to authenticate. API Error: 401` transport shape, while structured `api_error_status=401` stays the primary vocabulary-independent arm. RED/GREEN adds two benign near-miss fixtures that remain generic errors; a scratch mutation restoring the broad phrase search turns the named precision fixture RED. Plan and disposition: `specs/044-review-auth-fallback/plan.md`. |
|
|
46
|
+
| A Claude result envelope that explicitly reports authentication failure must enter the existing bounded auth-path remediation contract even when its subtype is `success`; otherwise a logged-in-but-unreadable or expired OAuth path is mislabeled as a generic local tool failure, creating contradictory fallback metadata that the controller correctly refuses | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_classify_envelope.sh | updated | Artifact classification and frozen decision table: `specs/044-review-auth-fallback/plan.md`. Observed baseline: an exact-candidate review returned an explicit 401/OAuth-expired result while local auth status still reported logged in; the classifier emitted generic `error:success`, the wrapper paired `reason_code=local_tool_failure` with fallback metadata, and the controller stopped before another client. Minimal correction: `skills/code-review/scripts/classify_envelope.py` maps structured `api_error_status=401` and the exact errored-envelope 401 transport phrase, with optional leading whitespace, to the existing `auth` token; it does not read credentials or alter the gate's terminal non-auth, concern, packet, binding, or tool-boundary paths. RED/GREEN: exact, whitespace-prefixed, and structured-status fixtures fail on the relevant unmodified branch and pass on the candidate; separately removing each classifier arm in scratch makes its named fixture RED. The structured comparison normalizes through a local `api_status_int`, mirroring the sibling normalizer in `parse_probe_result.py`, because a bare `== 401` matches only a Python int and a `"401"` serialization would miss both arms and re-create the generic `error:success` token this row exists to remove; string `401`/`429` and a non-numeric status are pinned, and reverting the normalization reds the string fixture. An earlier draft of this row claimed wrapper parse precedence protects a completed verdict from status 401 — that description was wrong and is corrected: `parse_review_json.py` refuses ANY envelope carrying a non-null `api_error_status`, so such an envelope is never a clean verdict and this change only renames the reason it is refused under; a `run_fail` twin of the existing 429 row now pins 401 plus a valid `structured_output` as refused instead of leaving it to prose. Integration evidence: `test_claude_review_probe.sh`, `test_review_gate.sh`, and the complete `make -j3 test-code-review` family pass, including auth host-retry/fallback and abort-leak contracts. |
|
|
27
47
|
| Requirement records stay with the product workflow while generic Wiki/Base mechanics remain resource-specific | `lark-wiki` / `lark-base` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#Create or reuse only the requested resource | routed | `product-rd-workflow/SKILL.md`; `bootstrap.md` |
|
|
28
48
|
| Structured testcase delivery includes its testcase Base lifecycle | `test-artifact-management` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/test-artifact-management/SKILL.md#but it is optional and must not trigger creation | updated | `test-artifact-management/SKILL.md`; `test-artifact-management/references/bitable-setup.md` |
|
|
29
49
|
| Review-client compatibility is capability-based and leaves model selection to host configuration | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/kimi_review.sh | updated | `code-review/SKILL.md`; `code-review/scripts/kimi_review.sh`; `code-review/scripts/opencode_review.sh`; `code-review/scripts/test_review_client_compat.py` |
|
|
@@ -295,3 +315,10 @@ and inverted the sense (production, not product), and the coordinator now shares
|
|
|
295
315
|
| When every successive bar built to make a mechanical exemption safe is broken by independent review, the signal is that the exempting capability should not exist rather than that another bar is owed; an evidence class may still force an honest label and a real obligation map while leaving the unverifiable judgement to a named risk owner | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/impact-chain-gate.rb | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key and is unchanged this round; implementation and regressions in `skill-extraction-workflow/scripts/impact-chain-gate.rb` and `skill-extraction-workflow/scripts/test_impact_chain_source_refuted.sh`. RED baseline: restoring the automatic floor lift turns the suite red on the compliant-withdrawal case, which must now stay blocked; the other fifteen cases and the repo's four existing impact-chain tests are unchanged either way. Seven review rounds and ten findings are recorded in `specs/043-evidence-class-source-refuted/frozen-acceptance.md` |
|
|
296
316
|
| A rule may fire perfectly and still be false: an entrypoint claim asserting that only the single most-salient prose rule applies while co-resident ones stay dormant is unsupported by either major vendor's published guidance, and the corollary it generated — that appending a clear rule does not add compliance — is contradicted outright, so it must be withdrawn rather than re-explained. Withdrawing an unsupported rationale is a `semantic-control` change, not a behaviour delta: every obligation the bullet carried is preserved verbatim and the entrypoint shrinks. **The `observed-failure: no` classification is a judgement worth challenging**: the withdrawn claim did mislead a reader into building a redirection on it, but this field asks whether the owning rule failed to *fire*, and it fired — it was simply wrong. The repo's evidence taxonomy has no class for 「一条正确触发但内容为假的规则」; recording it here rather than mislabelling it as a RED-baseline | `skill-extraction-workflow` | behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#Descriptive, not permissive | updated | `skill-extraction-workflow/SKILL.md` 的机制 bullet(1128 → 825 字节,入口净 −304);出处、证据分级与被撤回的原文保留在 `skill-extraction-workflow/references/external-practice-controls.md#instruction-following-mechanisms`;一手源为 Anthropic 的 context-engineering 文与 OpenAI 的 GPT-4.1 / GPT-5.1 prompting guide;本轮对该撤回做过的三次行为测量全部作废并记账于 `specs/042-skill-corpus-optimization/batch-1-result.md` 与 `evidence/AGENTS.md` |
|
|
297
317
|
| A suite that adds a sibling test without registering it in a lane ships a false green: the file exists, reads as covered, and never runs, so the registration self-audit must be treated as a merge-blocking check rather than a lint nicety — and a clone-based multi-case suite belongs in the heavy lane, not the pre-commit one | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key and is unchanged this round; the lane registry is in `skill-extraction-workflow/scripts/test_check_ccl_regressions.sh` and the audit in `skill-extraction-workflow/scripts/test_regression_runner_registration.sh`. RED baseline: CI demonstrated it — `regression-fast` and `regression-heavy` both failed with `test_*.sh not registered in fast_tests/heavy_tests` naming `test_impact_chain_source_refuted.sh`, which the previous round had added and never registered; both lanes pass once the suite is registered in the heavy lane, and reverting the registration turns them red again |
|
|
318
|
+
| A learning workflow that requires a failure RCA for every result forces stable success through an invented bad-outcome story, so the workflow first classifies failure/correction, stable success, or insufficient evidence; only failures run RCA, while stable success lands only with mechanism, non-luck evidence, reuse conditions, firing point, and owner | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/source-to-skill-extraction.md#is a claim the **independent review must accept** | updated | The owner rule and canonical templates are synchronized in `skill-extraction-workflow/SKILL.md`, `references/source-to-skill-extraction.md`, `references/extraction-quickstart.md`, and this register template. The entrypoint still names task/session summaries and lessons-learned requests as triggers and keeps missing classification or matching analysis at `interim`; the change broadens the analysis, not the trigger. Controlled A/B with Claude Code 2.1.235, hooks disabled and tools empty, used the same three-times-successful documentation scenario and varied only the quoted governing rule: the base failure-only rule produced `analysis_type=pre_extraction_rca_on_success`, invented a future bad outcome and proposed a new merge-blocking prevention gate; the candidate classified mechanism attribution, left `future_bad_outcome=null`, and returned the evidenced mechanism, non-luck evidence, reuse boundary, firing point, and owner. The existing LARGE-retrospective SUSTAIN rule is the semantic control: it already required mechanism + non-luck evidence + owner, so this round generalizes that owned invariant rather than creating a second success-learning method. Implementer self-review frozen before independent review — acceptance: (1) failure, stable success, and insufficient evidence select different analyses without a parallel 0→1 flow; (2) F4 keeps T2/T3 advisory and lets only an existing owner, risk, or review gate adopt their evidence as a per-change acceptance condition; the roll-up remains navigation, never a universal score or warn→block path; (3) delegation docs preserve pre-dispatch confidentiality, executable leaf containment, and scoped reviewer verdicts; (4) testing/self-review additions remain owner-linked. Candidate scope: `Makefile`, `README.md`, `docs/ARCHITECTURE.md`, `docs/f4-skill-effectiveness-harness.md`, `docs/feature-delivery-handbook.md`, `docs/multi-agent-delegation-handbook.md`, `docs/skill-extraction-handbook.md`, `docs/skills-theory-foundations.md`, `docs/technical-review-handbook.md`, `docs/testing-handbook.md`, `skills/skill-extraction-workflow/SKILL.md`, `references/eval-routing.md`, `references/extraction-quickstart.md`, `references/harness-patterns-and-eval.md`, `references/source-register.md`, `references/source-to-skill-extraction.md`, `scripts/eval-health.rb`. The same frozen candidate also carries three review-classifier and test files belonging to the co-resident review-auth-fallback slice; that slice is adjudicated by its own rows above plus `specs/044-review-auth-fallback/plan.md`, not by this one. Saying so is deliberate — an earlier draft of this row said "no excluded candidate file", which was false for the frozen diff and would have told a reader those files were outside the candidate and separately landed. They are named without package paths on purpose: a path here would read as this row declaring a second changed upstream owner, which is the ambiguity the impact-chain gate refuses. Edge/failure paths checked: success with no non-luck evidence remains observation; a T2/T3 miss never changes runner exit and becomes a landing condition only when an existing gate adopts it; skipped dashboard dimensions stay visible; prompt redaction leaves affected review obligations controller-side. Known residuals: vendor behavior remains time-bound to the linked official pages; the Wiki industry page already carried the current distinction and needed read-back rather than another rewrite; no dedicated F4 Wiki page exists, so 06 carries the reader-facing boundary. |
|
|
319
|
+
| A probe's baseline assertions must state facts the run LEAVES BEHIND, not what was true at the instant it looked: an existence test answers yes about an exited-but-unreaped process, so a precondition spelled that way accepts a corpse as a live orphan and the probe goes on to assert about a scenario it never built, while the verdict scan in the same file excludes zombies and reports the same pid gone — one process state, three answers, and a red that names the code under test for something it did not do. Give the whole probe ONE live/zombie/absent vocabulary, require `live` where the scenario is constructed, and prove WHY a process ended from an artifact of the code path under test (a work dir the cleanup would have deleted, a marker the fixture writes only past its bound) rather than from a liveness sample. A scenario the probe fails to BUILD is retried, never reported as a failed assertion | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate_abort_leak.sh | updated | `code-review/SKILL.md` is the owner key and is unchanged this round; the change lands in `skills/code-review/scripts/test_review_gate_abort_leak.sh` (a `wrapper_state()` helper used by every liveness question, a `live`-requiring reparent check, a work-dir-survives-SIGKILL assertion replacing "the suite exited within 30s", a bound-marker assertion replacing "still alive right after the kill", an arming step that freezes the wrapper's process group across the abort and READS the marker rather than deleting it — while the group is stopped nothing can write that file, so the ordering is enforced instead of sampled between racing events, and no evidence is destroyed — bounded setup retry, and its own deadline for the reparent wait, which previously reused the controller wait's countdown) and `skills/code-review/scripts/test_review_gate.sh` (both hang stubs record `<client>_hang_bound_reached` past their countdown). Round plan: `specs/046-abort-leak-baseline-invariants/plan.md`. Observed failure: the CI run identified in that plan red on `leg2: no reaper ran, so the wrapper is still alive right after the kill` while the leg's own behavioural assertion passed 13ms later on its first loop iteration; three reds total, each on a different leg-2 assertion, each asserting the probe's environment. Verified mechanism, reproduced rather than inferred: a real zombie answers `kill -0`, reports a ppid, and is excluded by both `$stat !~ /Z/` scans — so the reparent check passes, the instantaneous liveness check fails, and the verdict check reports gone, with no timing coincidence needed to explain the 19ms between them. Three candidate causes for the wrapper's early exit were rejected on evidence and recorded as such: the fixture's 25s bound (measured `since_detect=0s` across five runs, including under load average 54 on 18 cores), the suite-group SIGKILL (`review_gate.py:545` is the only spawn path and passes `start_new_session=True`, so the wrapper holds its own pgid), and the controller's own 5-12s wrapper timeout (a forced 15s delay left the wrapper `Ss` at `etime=00:15`). The trigger on that CI run remains unidentified; the bound marker makes the next occurrence name its own cause. RED-baseline (applied, differential): reverting the candidate stub's bound to unbounded reds leg2/fallback on both verdicts with `state=live bound_marker=absent`; reverting the claude stub's bound reds leg2/claude; removing only the marker write reds exactly one assertion — the bound verdict — while the residue verdict stays green, which is the discrimination the deleted timing proxy could not make. Controls green after each revert: `make test-code-review-abort-leak-1`, `-2`, and the full `test_review_gate.sh` |
|
|
320
|
+
| A rule that is already written and already followed elsewhere in the same repository can still not fire on new code, and "the owner skill states it" is enforcement absent, not enforcement present: the fix is the trigger, not another restatement. Promote the class to a mechanical gate scoped to where the predicate is a TEST VERDICT, pick the one spelling whose meaning is unambiguous so precision stays high, and state the recall limits rather than broadening the match | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_liveness_predicate_gate.sh | updated | `skill-extraction-workflow/SKILL.md` is the owner key and is unchanged this round; the change lands in `skills/skill-extraction-workflow/scripts/check-ccl-skills.sh` (`liveness_predicate_scan`, Anti-pattern 28, same shape as the Anti-pattern 27 gate it sits beside), `references/recurring-anti-patterns-checklist.md` (the Anti-pattern 28 row the gate cites), the new `scripts/test_liveness_predicate_gate.sh`, and its registration in `scripts/test_check_ccl_regressions.sh`. Observed failure: the rule forbidding an instantaneous liveness sample already existed in `testing-strategy` (its CI-fixtures and flake-control reference, named without a package path because that owner is unchanged this round and a path there reads as a second upstream claim) and the review-gate suite's own hang cases followed it, yet the probe written in the SAME round violated it in two of its four liveness sites and red CI three times. The predicate is the orphan-oracle shape only — a `ps -o ppid=` read compared against init's pid on the same line — because that spelling has exactly one meaning, while a ppid read used to identify a parent is the common correct use; a `stat=` consult within the window, or a call to a state helper the same file defines, clears the line, so the documented fix is also the way out. Precision on the current corpus is 100% (one hit, the real defect); five recall limits are named in the gate comment and the checklist row rather than papered over, per the checklist's own promotion guidance that a false positive costs more than a recall gap on a BLOCKING gate. RED-baseline (applied, differential): restoring the pre-fix probe takes the checker to `liveness_predicate_scan_failed` naming line 285 — the verified defect site — while the fixed tree reports `liveness_predicate_scan_ok` and `ccl_skill_check_clean_ok`; attribution is differential in that the failure is raised by this gate and no other. The behaviour suite's fifteen probes pin both directions, and building it under seven review rounds caught five real defects in the gate itself: the sorted hit list lost its trailing newline so the last hit and the diagnosis ran together, and the `trap 'rm -f "${var:-/dev/null}"' EXIT` idiom would have run `rm -f /dev/null` when the variable was unset — harmless unprivileged, a deleted device node in a root container — now a guarded cleanup function that names no fallback path; a whole-line comment naming the helper cleared a real hit; any `*_state` token counted as remediation; and an unrestricted `.*=` swallowed the `!` of `!=`, flagging the negated assertion that does the right thing |
|
|
321
|
+
| Stopping a process GROUP is not an atomic state transition, so reading the LEADER's state after the signal is a proxy for the condition rather than the condition: the leader can report stopped while a sibling has not yet been scheduled to handle it, and when that sibling is the countdown itself it can still reach the write the freeze exists to prevent. Wait for every member that is neither stopped nor already gone, bounded, and treat "still running" as a lost scenario rather than a verdict | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate_abort_leak.sh | updated | `code-review/SKILL.md` is the owner key and is unchanged this round; the change lands in `skills/code-review/scripts/test_review_gate_abort_leak.sh` (`wrapper_group_unstopped`, and an arming step that waits for the whole group). Supersedes the group-freeze half of the earlier row in this round, which checked only the leader. Observed failure: raised as P1 by the independent review lane against the pushed candidate, on the leg whose stub runs its countdown in a backgrounded child sharing the wrapper's group — the half of the mechanism the leader check cannot see. Evidence: with the fixture bound cut to 3s against the CLAUDE stub specifically (the one with the child), all five leg-2 assertions still pass, which they could not if the child were still counting during the abort window; both abort-leak targets and leg 1 stay green |
|
|
322
|
+
| A test harness owns a pid only while that pid is outstanding: once it has been reaped the number is the OS's to reissue, so a cleanup list that still carries it aims its signals at a stranger — and a suite that runs eight-way parallel makes that stranger a sibling lane. Drop ownership at the moment of reaping, which is the same prove-it-now rule the code under test applies before it signals anything | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_abort_leak_state_helpers.sh | updated | `code-review/SKILL.md` is the owner key and is unchanged this round; the change lands in `skills/code-review/scripts/test_abort_leak_state_helpers.sh` (`drop_kid`/`reap_kid`). Observed failure: raised as P1 by the independent review lane against this round's own new test — the harness written to check the probe's ownership discipline violated it, accumulating reaped pids in its cleanup list. RED-baseline: reverting the drop-at-reap change leaves reaped pids in the trap's signal list, which the suite's own accounting shows as entries no longer owned; the hazard is structural rather than timing-reproducible, so the recorded evidence is that accounting rather than a raced kill |
|
|
323
|
+
| A mechanical gate's DOCUMENTED limit belongs in its behaviour suite as a probe asserting the gate does NOT fire, not only in prose: pinned that way, any later tightening turns the probe red and forces the documentation to be corrected, whereas a limit described only in text goes stale silently and is then re-discovered as a finding | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_liveness_predicate_gate.sh | updated | `skill-extraction-workflow/SKILL.md` is the owner key and is unchanged this round; the change lands in `skills/skill-extraction-workflow/scripts/test_liveness_predicate_gate.sh` (P13/P14) with the limit paragraph in `skills/skill-extraction-workflow/references/recurring-anti-patterns-checklist.md` pointing at them. Observed failure: the adversarial challenge re-raised the different-pid / discarded-result waiver hole and correctly noted the suite never exercised it, so the limit existed only as prose. RED-baseline: P13/P14 pass against the current predicate and go red against a stricter one — that inversion is the signal they exist to raise |
|
|
324
|
+
| An obligation whose skip leaves NO artifact is not enforced, however normative its wording: the party it constrains states the entry condition, and a reviewer cannot refuse a claim that was never made. Demoting an overclaim must not demote the obligation riding on it — separate the two, and give the surviving obligation a trigger keyed on a fact of the DIFF (which file changed, which key a row carries) rather than on prose. A round is held only to the grammar its own head declares, or adding a required field retroactively refuses every historical round on replay | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: command:skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key and is unchanged this round; the change lands in `scripts/impact-chain-gate.rb` (two refusals: `impact_chain_result_class_missing`, `impact_chain_bank_evidence_missing`, both scoped to owners the round changed and both gated on the head-declared grammar), `references/source-register.md` (the declaration-fragment paragraph that dates them), and the new `scripts/test_impact_chain_self_adjudication.sh` registered in the heavy lane. Observed failure: round 044 withdrew several overclaims and demoted their obligations in the same move; five consecutive challenge rounds returned one shape, each naming an executable bypass — change a description and never run the bank (no absence to detect), omit the result-class cell (static checks pass, nothing emits interim), or write the class the author prefers (the reviewer has no field to refuse). RED-baseline (applied, differential, re-measured on the final suite — an earlier draft of this row froze a twelve-leg table and an `exactly A2/A3/A5 / B2/B3` partition and went stale as legs were added; independent review caught the drift, and the numbers below are the measured ones): the 28-leg decision table is red on the legs each refusal exists for before the gate change. Disabling `impact_chain_bank_evidence_missing` reds exactly A2 A3 A5 A7 A8 A9 A10 A11 A13 A14 A15 A16 A17 A18 A19; disabling `impact_chain_result_class_missing` reds exactly B2 B3 B5; disabling `impact_chain_grammar_withdrawn` reds exactly G2 G3 G4. Clean partition, no overlap, control green either side of every mutation. Bootstrap: removing `result-class` from this row itself, committed, reds the gate naming this row. Scope limit recorded rather than overclaimed: these close OMISSION, not MISCLASSIFICATION — the value stays the author's, and a locator is not proof the measurement ran. Retroactivity was measured, not assumed: naked, the two triggers newly refuse 34 of the 64 replayed historical integration points (5 for the bank trigger alone); gated on the head-declared grammar the differential returns 64/64 with zero new refusals, and leg G1 pins that property. Round plan: `specs/045-self-adjudicated-obligation-trigger/plan.md` |
|
|
@@ -46,12 +46,12 @@ Use this before reading deeply or editing any skill. The charter is the guardrai
|
|
|
46
46
|
|
|
47
47
|
| Field | Required answer |
|
|
48
48
|
| --- | --- |
|
|
49
|
-
| Purpose | What future failure
|
|
49
|
+
| Purpose | What future failure or drift should this extraction prevent, or what evidenced success mechanism should it preserve and reuse? |
|
|
50
50
|
| Scope | Which skill(s), source classes, users/tasks, and sibling boundaries are in scope? What is out of scope? For an iterative-program source (multi-round research/writing/delivery), also answer: does a project-local `covered-through` watermark exist, and what does this round cover above it? (see the watermark rule under Task Retrospective Extraction) |
|
|
51
51
|
| Depth | Is this wording cleanup/no new source read, targeted check, file-level refresh, node/artifact inventory, full workflow extraction, or generator/tooling change? |
|
|
52
|
-
|
|
|
53
|
-
|
|
|
54
|
-
| Failure mode
|
|
52
|
+
| Result classification | Is the observed result a failure/correction, stable success, or unstable/insufficient evidence? What observation supports that classification? |
|
|
53
|
+
| Matching analysis | Failure/correction: scaled RCA from observed issue to controllable prevention. Stable success: reusable mechanism, non-luck evidence, reuse conditions, firing point, and owner. Unstable/insufficient evidence: what remains unknown and why no executable rule lands yet. |
|
|
54
|
+
| Failure mode or success boundary | What bad output would weak extraction permit, or under which conditions would the success mechanism stop transferring? |
|
|
55
55
|
| Lifecycle impact | Which stages are affected: product intent, design/UX, implementation, debugging, testing, launch acceptance, iteration feedback, team onboarding, and use without source access? |
|
|
56
56
|
| Evidence plan | Which source categories must be inspected, routed, discarded, or marked unavailable? For a task/session retrospective over a session that produced artifacts, the FIRST source class listed MUST be those produced artifacts (deliverables, reports, scripts, datasets — the a0 enumeration, owed at charter time, not only before an exhaustion claim); a session with genuinely no produced artifacts records an explicit `produced artifacts: not-applicable` entry carrying the reason, the minimum checked surfaces (deliverable directories, script/output locations, dataset paths), and a resolvable inventory-check locator — the command or listing that establishes absence — instead of the class. Either way, correction turns and the agent's own summaries are friction-biased digests — they record only what rubbed, so what went RIGHT is structurally invisible in them — and cannot substitute for the artifact class or excuse skipping the check. |
|
|
57
57
|
| Completion standard | What pressure scenario, independent review, command, install check, or source-map evidence proves done? |
|
|
@@ -82,30 +82,30 @@ If the user asks for complete, deep, full, or repeat extraction, or challenges s
|
|
|
82
82
|
|
|
83
83
|
Before a source portfolio can confirm or contradict a reusable rule, classify it as `stable` (in production, not slated for replacement), `evolving` (actively iterating, design not frozen), `legacy-deprecating` (scheduled for retirement), or `mixed`. Only `stable` portfolios can be used as confirmation/contradiction baseline. `Evolving`, `legacy-deprecating`, and `mixed` portfolios are audit/anti-pattern signals unless the extraction is explicitly downscoped to that status; state the long-lived caveat reason instead of implying future verification will upgrade it automatically.
|
|
84
84
|
|
|
85
|
-
## Baseline
|
|
85
|
+
## Result-Learning Baseline For Every Extraction
|
|
86
86
|
|
|
87
|
-
|
|
87
|
+
Classify the result before choosing an analysis method:
|
|
88
88
|
|
|
89
|
-
|
|
|
90
|
-
| --- | --- |
|
|
91
|
-
| Future
|
|
92
|
-
|
|
|
93
|
-
|
|
|
94
|
-
|
|
95
|
-
|
|
89
|
+
| Result class | Required analysis | Landing condition |
|
|
90
|
+
| --- | --- | --- |
|
|
91
|
+
| Failure or correction | Future bad outcome, contributing factors, counterfactual ranking, controllable prevention, firing path, and owner | Evidence shows the control addresses the failure class; known failures use correction RCA |
|
|
92
|
+
| Stable success | Mechanism that produced the result, evidence it recurs and is not luck, reuse conditions and transfer boundary, firing point, and owner | The mechanism is observable and reusable; praise or a single good run is insufficient |
|
|
93
|
+
| Unstable or insufficient evidence | What was observed, competing explanations, and missing evidence | Observation only; no executable rule until the classification becomes supportable |
|
|
94
|
+
|
|
95
|
+
**The classification is not self-elective.** Two of the three classes skip RCA, and the agent choosing the class is the same agent whose work the RCA would examine — so left to the author the cheap classes are always available. Therefore:
|
|
96
96
|
|
|
97
|
-
|
|
97
|
+
- An extraction triggered by a correction, a review finding, a failed run, a regression, or a user pointing out a miss is **`Failure or correction` by default**, whatever the author's own reading of it.
|
|
98
|
+
- Relabelling such an extraction into `Stable success` or `Unstable or insufficient evidence` is a claim the **independent review must accept**; the author recording the relabel is not the adjudication, and neither is a reason written into the round artifact.
|
|
99
|
+
- Every classification carries the observation that would **disconfirm** it — for `Stable success`, what would show the result was luck; for `Unstable`, what evidence would settle it. A class with no disconfirming observation named is unrecorded, not recorded-and-passed.
|
|
100
|
+
- Missing classification, an unaccepted relabel, or missing matching analysis leaves the extraction `interim`, and `interim` is a state the closeout reports rather than a label the author clears.
|
|
98
101
|
|
|
99
|
-
|
|
100
|
-
- Targeted check: future failure, source boundary, and proof.
|
|
101
|
-
- File-level or broader extraction: full table above, plus lifecycle impact, run through the Deep RCA five moves below (the table's single "enabling cause" row becomes the *set* of contributing factors).
|
|
102
|
-
- Broad or multi-skill extraction: full table above, source register, target-output map, independent review, and the Deep RCA five moves below.
|
|
102
|
+
Scale depth to the task. Wording cleanup records one concise classification and owner. A small non-wording failure gets a widen-check plus control; a non-trivial failure uses Deep RCA. A stable-success extraction increases evidence depth with the breadth of the reuse claim. Broad or multi-skill extraction also requires the source register, target-output map, and independent review.
|
|
103
103
|
|
|
104
|
-
If the
|
|
104
|
+
If the analysis reveals a product decision, architecture decision, test strategy, design readiness issue, source-access problem, or sibling-skill update, route it before editing the target skill.
|
|
105
105
|
|
|
106
106
|
### Deep RCA For Extraction
|
|
107
107
|
|
|
108
|
-
|
|
108
|
+
For a failure or correction, a causal account is required; 5 Why is only the **entry technique** to get past a visible symptom. Used alone it has a documented failure mode: it traces ONE linear chain to ONE "root cause", is bounded by the investigator's current knowledge, is non-reproducible (different agents reach different ends), and the word "why" drifts toward "who" (blame) and toward hindsight. Most process/agent failures are not single-cause — overt failure requires several contributing causes to coincide — so for any non-trivial failure extraction run the fuller method below, not just a why-chain. Stable success uses the Result-Learning baseline above instead of inventing a failure. (For pure wording cleanup — the strict wording-only test, no trigger/scope/routing/validation/owner-meaning change — one concise classification is enough.)
|
|
109
109
|
|
|
110
110
|
Do not force exactly five questions, and do not accept a single straight chain. Ask enough "why" to leave the symptom; ask "how/what conditions" to widen; stop a branch once its next action is concrete and owned.
|
|
111
111
|
|
|
@@ -170,18 +170,17 @@ Source-specific prompts (where each branch's RCA should resolve):
|
|
|
170
170
|
|
|
171
171
|
## Task Retrospective Extraction
|
|
172
172
|
|
|
173
|
-
Use this when the user asks to summarize this task, summarize lessons learned, review what went wrong, or turn the current session into reusable team practice.
|
|
173
|
+
Use this when the user asks to summarize this task, summarize lessons learned, review what went wrong or right, or turn the current session into reusable team practice.
|
|
174
174
|
|
|
175
|
-
The current task is a source, but it is not automatically a skill rule. Treat task history as evidence and
|
|
175
|
+
The current task is a source, but it is not automatically a skill rule. Treat task history as evidence and classify each result before choosing RCA, success-mechanism attribution, or observation-only treatment.
|
|
176
176
|
|
|
177
177
|
Required flow:
|
|
178
178
|
|
|
179
179
|
1. Define the task boundary: which user request, implementation slice, review, bug, correction, or validation result is being summarized.
|
|
180
|
-
2.
|
|
181
|
-
-
|
|
182
|
-
-
|
|
183
|
-
-
|
|
184
|
-
- Which skill, validator, shared project doc, memory note, repo doc, or final-response rule owns the prevention?
|
|
180
|
+
2. Classify the result and run the matching analysis:
|
|
181
|
+
- Failure/correction: what bad outcome would repeat, which factors enabled it, which control should have caught it, and which owner must carry the prevention?
|
|
182
|
+
- Stable success: what mechanism produced the result, what proves it was not luck, under which conditions it transfers, where it should fire again, and who owns it?
|
|
183
|
+
- Unstable/insufficient evidence: which explanations remain open and what evidence is missing? Keep it as an observation.
|
|
185
184
|
- For delivery-chain failures, ask why the requirement/contract was not defined correctly, why implementation could proceed by inference, why unit/contract/integration/E2E tests or review/MR readiness did not block it, and why any earlier retrospective missed the deeper cause; land prevention at every failed owning layer, not only one target skill.
|
|
186
185
|
3. Classify each lesson:
|
|
187
186
|
- `skill`: reusable agent behavior that belongs in an existing or new skill.
|
|
@@ -195,9 +194,9 @@ Required flow:
|
|
|
195
194
|
|
|
196
195
|
Minimum retrospective table:
|
|
197
196
|
|
|
198
|
-
| Task
|
|
197
|
+
| Task result | Result analysis | Lesson classification | Durable owner | Verification |
|
|
199
198
|
| --- | --- | --- | --- | --- |
|
|
200
|
-
| What happened
|
|
199
|
+
| What happened; failure, stable success, or insufficient evidence | RCA; or success mechanism + non-luck evidence + reuse boundary; or observation-only reason | skill / validator / project artifact / memory / final response only | File, skill, script, shared artifact, memory note for local preference only, or no-skill reason | Diff, command, review, or explicit non-skill reason |
|
|
201
200
|
|
|
202
201
|
### LARGE-Session Lesson Axes And The Delivery-State Axis
|
|
203
202
|
|
|
@@ -514,6 +514,159 @@ if [[ -n "${gid_hits//[$'\n']/}" ]]; then
|
|
|
514
514
|
fi
|
|
515
515
|
echo "git_identity_predicate_scan_ok"
|
|
516
516
|
|
|
517
|
+
# Anti-pattern 28 — process liveness decided by an EXISTENCE test that a corpse answers
|
|
518
|
+
# (see references/recurring-anti-patterns-checklist.md). A pid whose process has exited
|
|
519
|
+
# but has not been reaped is still in the process table: it answers `kill -0`, and its
|
|
520
|
+
# ppid still reads — as 1 once it is reparented. A check that concludes "alive" or
|
|
521
|
+
# "orphaned and alive" from existence alone therefore says yes about a corpse, while the
|
|
522
|
+
# verdict scans in the same suite exclude zombies and say gone — one process state, two
|
|
523
|
+
# answers, and a probe that reds while the code under test is behaving.
|
|
524
|
+
# Promoted from checklist to gate because the class recurred three times in CI on the
|
|
525
|
+
# abort-leak probe, each time on a different assertion, each time asserting the probe's
|
|
526
|
+
# environment rather than the suite.
|
|
527
|
+
# Scope: `test_*.sh` / `test.sh` under the repo — the surface where such a predicate is a
|
|
528
|
+
# TEST VERDICT rather than a signalling guard. Excluded on purpose: non-test shell (a
|
|
529
|
+
# `kill -0` before signalling asks about existence, which is the right question there),
|
|
530
|
+
# and WHOLE-LINE comments (this comment block names the banned spelling).
|
|
531
|
+
# The predicate is the orphan-oracle shape specifically: a `ps -o ppid=` read compared
|
|
532
|
+
# against init's pid on the same line. That shape has exactly one meaning — "has this
|
|
533
|
+
# been reparented, i.e. is it a live orphan" — so precision is high; a ppid read used to
|
|
534
|
+
# IDENTIFY a parent (the common use) is untouched. A process-state consult (`stat=` or a
|
|
535
|
+
# helper whose name ends in `_state`) within the window clears the line, which is the fix.
|
|
536
|
+
# Recall limits — stated so the contract is not overclaimed, per the checklist's
|
|
537
|
+
# promotion guidance (high precision plus a documented recall limit beats broad matching
|
|
538
|
+
# for a BLOCKING gate):
|
|
539
|
+
# 1. `while kill -0 "$pid"` watchdog loops are NOT flagged. Two exist here; both wait on
|
|
540
|
+
# a direct child and `wait` for it immediately after, so the shell reaps it and the
|
|
541
|
+
# loop ends — a narrower risk than the shape above, and forcing churn on them would
|
|
542
|
+
# trade a real false-positive cost for an unproven gain.
|
|
543
|
+
# 2. a liveness branch spelled with a bare `kill -0` inside a loop BODY.
|
|
544
|
+
# 3. any dynamic or indirect spelling.
|
|
545
|
+
# Those stay checklist plus adversarial-challenge checks.
|
|
546
|
+
liveness_hits=""
|
|
547
|
+
liveness_list=$(mktemp "${TMPDIR:-/tmp}/ccl-skills-liveness.XXXXXX") || {
|
|
548
|
+
echo "liveness_predicate_scan_error: mktemp failed" >&2; exit 1; }
|
|
549
|
+
# Guarded rather than the `${var:-/dev/null}` idiom: this trap is installed
|
|
550
|
+
# unconditionally, so an unset var would make the cleanup `rm -f /dev/null` — a no-op for
|
|
551
|
+
# an unprivileged user and a deleted device node for a root container. Never name a path
|
|
552
|
+
# in a removal that a fallback can turn into someone else's file.
|
|
553
|
+
liveness_scan_cleanup() {
|
|
554
|
+
[ -n "${liveness_list:-}" ] && [ -e "${liveness_list:-}" ] && rm -f "$liveness_list"
|
|
555
|
+
[ -n "${gid_list:-}" ] && [ -e "${gid_list:-}" ] && rm -f "$gid_list"
|
|
556
|
+
return 0
|
|
557
|
+
}
|
|
558
|
+
trap liveness_scan_cleanup EXIT
|
|
559
|
+
# Same traversal hardening as Anti-pattern 27, and for the same reason: every leg here
|
|
560
|
+
# exists because its absence makes the gate print "ok" for a scan that did not happen.
|
|
561
|
+
if find -L "$root" -name '*.sh' -type f -print0 > "$liveness_list" 2>/dev/null; then
|
|
562
|
+
liveness_find_rc=0
|
|
563
|
+
else
|
|
564
|
+
liveness_find_rc=$?
|
|
565
|
+
fi
|
|
566
|
+
if [[ "$liveness_find_rc" -ne 0 ]]; then
|
|
567
|
+
rm -f "$liveness_list"
|
|
568
|
+
echo "liveness_predicate_scan_error: find exited $liveness_find_rc — the walk is INCOMPLETE, which is not the same as clean" >&2
|
|
569
|
+
exit 1
|
|
570
|
+
fi
|
|
571
|
+
while IFS= read -r -d '' liveness_file; do
|
|
572
|
+
[[ -n "$liveness_file" ]] || continue
|
|
573
|
+
case "${liveness_file##*/}" in test_*.sh|test.sh) : ;; *) continue ;; esac
|
|
574
|
+
# `[^=]*[^!=]=` rather than `.*=`: an unrestricted `.*` swallows the `!` of a `!=`, so
|
|
575
|
+
# the negated assertion `[ "$(ps -o ppid= -p "$pid")" != "1" ]` — which checks a process
|
|
576
|
+
# is NOT reparented, the opposite of the banned oracle — was reported as a violation.
|
|
577
|
+
# A false positive on a blocking gate is the failure this checklist's promotion
|
|
578
|
+
# guidance weighs heaviest, because it is what gets a gate loosened until it catches
|
|
579
|
+
# nothing.
|
|
580
|
+
if liveness_out="$(grep -nE 'ps[[:space:]]+-o[[:space:]]+ppid=[^=]*[^!=]=[[:space:]]*"?1"?[[:space:]]*\]' "$liveness_file" 2>/dev/null)"; then
|
|
581
|
+
liveness_rc=0
|
|
582
|
+
else
|
|
583
|
+
liveness_rc=$?
|
|
584
|
+
fi
|
|
585
|
+
if [[ "$liveness_rc" -gt 1 ]]; then
|
|
586
|
+
rm -f "$liveness_list"
|
|
587
|
+
echo "liveness_predicate_scan_error: grep exited $liveness_rc while scanning ${liveness_file#"$root"/} — unreadable or I/O failure means the result is UNKNOWN, not clean" >&2
|
|
588
|
+
exit 1
|
|
589
|
+
fi
|
|
590
|
+
[[ "$liveness_rc" -eq 0 ]] || continue
|
|
591
|
+
while IFS= read -r liveness_line; do
|
|
592
|
+
[[ -n "$liveness_line" ]] || continue
|
|
593
|
+
liveness_no="${liveness_line%%:*}"
|
|
594
|
+
liveness_text="${liveness_line#*:}"
|
|
595
|
+
[[ "$liveness_text" =~ ^[[:space:]]*# ]] && continue
|
|
596
|
+
# Window clear: a process-state consult within two lines either side is the fix, so
|
|
597
|
+
# a corrected site stops being reported without needing a waiver marker.
|
|
598
|
+
liveness_lo=$(( liveness_no > 2 ? liveness_no - 2 : 1 ))
|
|
599
|
+
liveness_hi=$(( liveness_no + 2 ))
|
|
600
|
+
# Whole-line comments are dropped before the window is inspected: a prose mention
|
|
601
|
+
# such as a TODO naming the helper would otherwise clear a real violation beside it,
|
|
602
|
+
# and the gate would print ok for a line it had actually found. The waiver must be
|
|
603
|
+
# CODE that consults process state, not a note saying someone should.
|
|
604
|
+
# Only WHOLE-LINE comments are dropped, matching how the hit line itself is filtered.
|
|
605
|
+
# Stripping trailing comments would need to decide whether a `#` opens a comment or
|
|
606
|
+
# belongs to `${var#prefix}`, which needs a shell parser — the same call Anti-pattern
|
|
607
|
+
# 27 makes above. Recall limit: a mention in a TRAILING comment can still clear the
|
|
608
|
+
# window; the checklist row records it.
|
|
609
|
+
liveness_win="$(sed -n "${liveness_lo},${liveness_hi}p" "$liveness_file" 2>/dev/null |
|
|
610
|
+
grep -vE '^[[:space:]]*#' || true)"
|
|
611
|
+
# A direct `ps -o stat=` in the window is the consult itself. Anchored on `-o stat=`
|
|
612
|
+
# rather than bare `stat=`: an unrelated assignment such as `stat=unknown` is not a
|
|
613
|
+
# process-state read, and accepting it was the fourth way this waiver was found to
|
|
614
|
+
# clear a real hit.
|
|
615
|
+
if printf '%s' "$liveness_win" | grep -qE '\-o[[:space:]]*stat='; then
|
|
616
|
+
continue
|
|
617
|
+
fi
|
|
618
|
+
# Otherwise the window may CALL a state helper — but only one this file actually
|
|
619
|
+
# defines and which itself consults process state. Accepting any `*_state` token
|
|
620
|
+
# made an unrelated `record_state "$pid"` read as remediation and skipped a real
|
|
621
|
+
# hit, which is a hole rather than a stated recall limit.
|
|
622
|
+
# The file must also actually consult process state somewhere, so a file with no
|
|
623
|
+
# `stat=` at all cannot be cleared by a bookkeeping helper that merely ends in
|
|
624
|
+
# `_state`. Recall limit: a file that separately defines such a helper AND reads
|
|
625
|
+
# process state elsewhere clears the window without proving the two are connected —
|
|
626
|
+
# tying a call to its definition needs a shell parser, the same call made above.
|
|
627
|
+
# Deliberately POSIX-portable: `\b` and BSD-sed `\?` are not, and a silently failing
|
|
628
|
+
# match here would loosen the gate on exactly the host that runs it most.
|
|
629
|
+
# The named helper's OWN BODY must reach a process-state read. Checking "the file
|
|
630
|
+
# defines it" and "the file reads state somewhere" as two independent facts let a
|
|
631
|
+
# hollow helper plus an unrelated `ps -o stat=` elsewhere clear a real hit — the two
|
|
632
|
+
# conditions were never tied to each other. awk rather than sed: BSD sed lacks `\?`
|
|
633
|
+
# and a silently failing match here loosens the gate on the host that runs it most.
|
|
634
|
+
liveness_helper_ok=0
|
|
635
|
+
while IFS= read -r liveness_helper; do
|
|
636
|
+
[[ -n "$liveness_helper" ]] || continue
|
|
637
|
+
if awk -v fn="$liveness_helper" '
|
|
638
|
+
$0 ~ "^[[:space:]]*(function[[:space:]]+)?" fn "[[:space:]]*\\(\\)" {
|
|
639
|
+
# The declaration line is part of the body: a one-line definition both opens
|
|
640
|
+
# and closes here. Skipping it with a bare `next` left `inbody` set through
|
|
641
|
+
# `fn() { echo live; }` and credited the NEXT function\047s state read to this
|
|
642
|
+
# hollow one.
|
|
643
|
+
if ($0 ~ /-o[[:space:]]*stat=/) { found = 1; exit }
|
|
644
|
+
if ($0 ~ /}/) { exit }
|
|
645
|
+
inbody = 1
|
|
646
|
+
next
|
|
647
|
+
}
|
|
648
|
+
inbody && /-o[[:space:]]*stat=/ { found = 1; exit }
|
|
649
|
+
inbody && /^[[:space:]]*}/ { exit }
|
|
650
|
+
END { exit !found }
|
|
651
|
+
' "$liveness_file" 2>/dev/null; then
|
|
652
|
+
liveness_helper_ok=1
|
|
653
|
+
break
|
|
654
|
+
fi
|
|
655
|
+
done < <(printf '%s\n' "$liveness_win" | grep -oE '[A-Za-z_][A-Za-z0-9_]*_state' | sort -u)
|
|
656
|
+
[[ "$liveness_helper_ok" -eq 1 ]] && continue
|
|
657
|
+
liveness_hits+="${liveness_file#"$root"/}:$liveness_line"$'\n'
|
|
658
|
+
done <<< "$liveness_out"
|
|
659
|
+
done < "$liveness_list"
|
|
660
|
+
rm -f "$liveness_list"
|
|
661
|
+
if [[ -n "${liveness_hits//[$'\n']/}" ]]; then
|
|
662
|
+
# Trailing newline restored explicitly: command substitution strips it, which ran the
|
|
663
|
+
# last hit and the diagnosis together on one line.
|
|
664
|
+
printf '%s\n' "$(printf '%s' "$liveness_hits" | sort)"
|
|
665
|
+
echo "liveness_predicate_scan_failed: the lines above decide that a process is a LIVE orphan from a ppid read alone, which an unreaped corpse answers the same way — the check says alive while the suite's own verdict scans exclude zombies and say gone. Consult process STATE (\`ps -o stat=\`, or a helper that returns live/zombie/absent) before concluding liveness. See skill-extraction-workflow/references/recurring-anti-patterns-checklist.md (Anti-pattern 28)." >&2
|
|
666
|
+
exit 1
|
|
667
|
+
fi
|
|
668
|
+
echo "liveness_predicate_scan_ok"
|
|
669
|
+
|
|
517
670
|
# Evidence-card leak PREFLIGHT (best-effort, NOT the gate of record): catches obvious
|
|
518
671
|
# card-shaped markdown accidentally committed under skills/** (L0 storage rule, see
|
|
519
672
|
# l0-l1-l2-routing.md). Delegated to a dedicated, self-tested detector. The gate runs the
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
#!/usr/bin/env ruby
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
|
-
# F4
|
|
4
|
+
# F4 signal dashboard — advisory display roll-up (0-10) + same-stick change.
|
|
5
5
|
#
|
|
6
|
-
#
|
|
7
|
-
#
|
|
8
|
-
#
|
|
9
|
-
#
|
|
10
|
-
#
|
|
6
|
+
# Shows the F4 dimensions together and retains a weighted 0-10 display value plus
|
|
7
|
+
# a same-corpus/dimensions delta for compatibility with existing reports. The
|
|
8
|
+
# shape resembles OpenSSF Scorecard, but the value is not a verdict that the
|
|
9
|
+
# skill repository is globally better or worse: these dimensions have different
|
|
10
|
+
# semantics, and a quality or safety failure is never averaged away.
|
|
11
11
|
#
|
|
12
12
|
# Dimensions (each scored 0-10, risk-weighted OpenSSF-style):
|
|
13
13
|
# structural weight 10 (Critical) — validate-skill.sh: skills load / no leakage
|
|
@@ -29,7 +29,8 @@
|
|
|
29
29
|
# (structural validation + Tier-1 blocking findings) remain the source of truth
|
|
30
30
|
# and block independently of this score. Per Goodhart's law ("when a measure
|
|
31
31
|
# becomes a target, it ceases to be a good measure"), a composite that became a
|
|
32
|
-
# merge gate would just get gamed; it stays a lens, not a gate
|
|
32
|
+
# merge gate would just get gamed; it stays a navigation lens, not a gate or an
|
|
33
|
+
# acceptance grade. The history file
|
|
33
34
|
# is git-ignored for the same reason — a committed number invites tuning the
|
|
34
35
|
# number instead of the repo.
|
|
35
36
|
#
|
|
@@ -289,7 +290,14 @@ unless no_write
|
|
|
289
290
|
end
|
|
290
291
|
|
|
291
292
|
report = entry.merge(
|
|
293
|
+
# `band` keeps its name and values for existing report consumers, but the
|
|
294
|
+
# machine surface must not be the one place the withdrawal does not reach: a
|
|
295
|
+
# consumer reads this JSON, not the header line, and CLEAN / WARNING / NEEDS
|
|
296
|
+
# WORK / CRITICAL is grade vocabulary that gets quoted as an acceptance verdict.
|
|
297
|
+
# The qualifier therefore travels as its own field rather than only in `puts`.
|
|
292
298
|
"band" => band(composite),
|
|
299
|
+
"band_semantics" => "navigation-only display label; not an acceptance grade or an overall-quality verdict",
|
|
300
|
+
"score_semantics" => "weighted display roll-up over the dims present in this run; comparable only against the same corpus and dims",
|
|
293
301
|
"weights_present" => present.to_h { |k, _| [k.to_s, RISK[k]] },
|
|
294
302
|
"details" => dims.to_h { |k, d| [k.to_s, d[:detail]] },
|
|
295
303
|
"trend" => trend,
|
|
@@ -306,13 +314,18 @@ if out_json
|
|
|
306
314
|
end
|
|
307
315
|
|
|
308
316
|
unless quiet
|
|
309
|
-
puts "F4
|
|
317
|
+
puts "F4 SIGNAL DASHBOARD (advisory — not a gate or overall-quality verdict)"
|
|
310
318
|
puts " repo=#{repo_sha} corpus=#{corpus} dims=#{present_dims.join(',')}"
|
|
311
319
|
dims.each do |k, d|
|
|
312
320
|
s = d[:score].nil? ? " - " : format("%2d/10", d[:score])
|
|
313
321
|
puts " #{k.to_s.ljust(15)} #{s} #{d[:detail]}"
|
|
314
322
|
end
|
|
315
|
-
|
|
323
|
+
# The band label carries grade vocabulary (CLEAN / WARNING / NEEDS WORK /
|
|
324
|
+
# CRITICAL) that reads as an acceptance verdict and gets quoted downstream as
|
|
325
|
+
# one. The header withdrew that claim; printing the bare label next to the
|
|
326
|
+
# number would have left the executable surface contradicting the withdrawal,
|
|
327
|
+
# so the qualifier travels with the label rather than only with the header.
|
|
328
|
+
puts " DISPLAY ROLL-UP #{format('%.1f', composite)}/10 #{band(composite)} (navigation only, not an acceptance grade)"
|
|
316
329
|
if prior_comparable
|
|
317
330
|
arrow = trend.positive? ? "IMPROVING +#{trend}" : (trend.negative? ? "DECLINING #{trend}" : "FLAT")
|
|
318
331
|
puts " trend vs #{prior_comparable['repo_sha']} (same corpus+dims): #{arrow}"
|
|
@@ -321,7 +334,7 @@ unless quiet
|
|
|
321
334
|
else
|
|
322
335
|
puts " trend: first run — no history yet"
|
|
323
336
|
end
|
|
324
|
-
puts " (
|
|
337
|
+
puts " (inspect dimensions and task evidence; gates block independently; history=#{history}, git-ignored)"
|
|
325
338
|
end
|
|
326
339
|
|
|
327
340
|
exit 0
|