@ccoalm/ccl-skills 0.15.0 → 0.15.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +80 -4
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +16 -4
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +13 -2
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +37 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +5 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +77 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_packet_mcp.py +98 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_cli_review.py +48 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +237 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +165 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_kimi_packet_mcp.py +143 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +572 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +7 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +40 -83
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +44 -37
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +20 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/refactoring-discipline.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +8 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +17 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/resume-paused-delivery.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +31 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +74 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +103 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +83 -48
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +80 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +74 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_runtime.py +428 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +190 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +106 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -1
- package/dist/assets/release.json +87 -67
- package/dist/claude-adapter.js +14 -7
- package/dist/codex-host.d.ts +2 -4
- package/dist/codex-host.js +40 -19
- package/dist/host-probe.d.ts +27 -0
- package/dist/host-probe.js +51 -0
- package/dist/opencode-adapter.js +24 -19
- package/dist/operations.js +34 -8
- package/dist/unified.d.ts +1 -1
- package/dist/unified.js +18 -9
- package/package.json +1 -1
|
@@ -522,16 +522,16 @@ A **scope-cut / out-of-phase** finding (the scope-direction signal in `SKILL.md`
|
|
|
522
522
|
|
|
523
523
|
**A convergence or closure declaration must be written falsifiably.** Name the exact candidate identity it covers, each lane's terminal evidence, the axes/dimensions the closing self-audit actually crossed, and every standing open item by name (e.g. "the final challenge's own fix has not itself been re-challenged") — an aggregate "converged / all axes closed" whose axes are unnamed cannot be checked false and is inconclusive, and any "full X" adjective is scoped to the named axes, never wider. The named enumeration is what lets a fresh challenge falsify the claim by pointing at an un-crossed axis (observed both ways in one program: a self-audit that named its five walked axes was caught exactly one axis short by the final challenge — the naming is why the gap was findable — and the honest handoff that named its open item let the human choose between one fresh pass and explicit risk acceptance instead of inheriting a false "done").
|
|
524
524
|
|
|
525
|
-
The initial independent review plus Agent-initiated challenges share
|
|
525
|
+
The initial independent review plus Agent-initiated challenges share a **bounded external-review sequence of at most five rounds**. The initial review consumes round 1, so `challenge_budget` is `0..4`. An authorized task includes its necessary fixes, tests and review by default; a sequence limit triggers the checkpoint below, not a new permission request. Candidate edits, commits, rebases, amended plans or renamed slices never erase cumulative spending or broaden that task authority. A stateless local controller cannot prove omitted history against a caller that controls its files, so the consuming workflow must preserve the complete review ledger and treat an Agent-created reset as a contract violation.
|
|
526
526
|
|
|
527
|
-
Five is the generic `code-review` transport ceiling, not this extraction lane's spend. Non-wording Agent-autonomous extraction calls go through `scripts/extraction_review_gate.sh`, which fixes `challenge_budget=1` per chain: one review plus one challenge. **
|
|
527
|
+
Five is the generic `code-review` transport ceiling, not this extraction lane's spend. Non-wording Agent-autonomous extraction calls go through `scripts/extraction_review_gate.sh`, which fixes `challenge_budget=1` per chain: one review plus one challenge. **Each extraction receipt sequence spans at most two chains and three rounds; the third exists only because a fix batch moved the candidate.** Holding fixes keeps the challenge on the frozen round-1 candidate, so the batch that lands is unreviewed until a succeeding chain challenges it — and a fix touching a selected owner's `SKILL.md` or `references/**.md` moves that owner digest and ends the first chain anyway. The trigger is the candidate, never a disposition label the author writes: **landing hash equal to the challenged hash owes nothing; different owes one succession challenge bound to what lands.** There that receipt sequence ends. Necessary further review follows the checkpoint and original-task authority rule below in another bounded sequence, with complete cumulative history retained. Record a later round as human-requested only when a human actually requested that round. Unused generic capacity alone never justifies another call. The closeout validator rejects referenced receipts whose recorded budget is not the wrapper-fixed value, rejects any post-chain round that is not a succession, and checks budget and ordering consistency within the caller-supplied set. `scripts/review_ledger_binding.py` is its merge-side half: it recomputes the candidate with the controller's own packet freeze and refuses a landing whose evidence binds a different one. Evidence lives outside the reviewed paths, so committing the ledger cannot move the hash it records. A candidate larger than one packet is not split as a pull request but as a review: `--print-manifest --partition <paths> [--partition <paths> ...]` renders a landing partition manifest whose path partitions cover every changed file exactly once, each partition hashing to what `--print-candidate --paths <partition>` answers; commit the manifest with one validated closeout ledger per partition, and the gate recomputes every partition and refuses a manifest whose parts do not add up to the whole (an uncovered or overlapping file, a partition that no longer reproduces, a base other than the fork point, or an aggregate hash that does not reproduce its partitions). An integration branch that accumulated several reviewed rounds is promoted as one pull request without a new ledger: when neither a single ledger nor a manifest binds the promotion, the gate walks HEAD's first-parent chain down to the first commit already on the target and rebinds each round merge in a detached checkout of its second parent against its first parent, judged with the landing tree's own controller and validator rather than the round's (a round could carry a hollowed validator that a later round restores); a merge whose second parent is already on the target is a sync merge and owes nothing; every step must be exactly the automatic merge of its parents (a hand resolution or an extra file in the merge commit is refused as unreviewed), a non-merge commit on the chain is refused, and the chain is consulted only for the default path set. Rounds that appended to the same register therefore no longer force the promotion to be split by round. It cannot authenticate that the wrapper produced those receipts or that the caller retained every earlier chain or receipt. The wrapper does not mint or persist `review_chain_id` or `autonomous_review_index`: the caller still supplies both, and could start a fresh-looking chain after the final round. The validator detects bad order inside the referenced set but cannot detect a prior chain the caller omitted, so complete caller-owned ledger retention—and treating an Agent reset as a contract violation—remains part of the boundary rather than a property the local scripts prove.
|
|
528
528
|
|
|
529
|
-
**Self-hosted chains break on
|
|
529
|
+
**Self-hosted chains break on owner edits; sum rounds across chains within each sequence and retain cumulative spending across sequences.** In a skill repository the candidate edits its own owner package by construction, so the chain's stable bindings make the dead-end the norm, not an edge case: the selected-owner digest hashes each owner package's current working tree and owners derive from the candidate's own paths, so a fix that touches any selected-owner tree ends the tracked chain (`review_chain_invalid`) — in an extraction round that is nearly every fix, while a fix confined to files outside every selected owner drifts only the candidate hash and continues in-chain — and a plan edit that changes the normalized review scope (intent, acceptance, stage/depth, risk tags, budget) ends it as `review_scope_changed` — a self-review- or evidence-only plan refresh keeps the scope digest and the chain (binding mechanics are owned by the staged review contract in `code-review`). A chain restarted at index 1 after such a break still consumes its sequence's budget; a later sequence requires the checkpoint below and retains every earlier round. Treating each restarted chain as a procedurally required fresh review loop is the observed way the budget hollows out: two consecutive extraction rounds ran 20+ reviewer rounds and then 12 restarted chains — 21 reviewer invocations to land a three-line diff — each restart looking locally mandatory. When a round returns findings, walk this enumeration before any further external call:
|
|
530
530
|
|
|
531
|
-
1. **Batch dispositions; never re-chain per finding — and hold every fix until the round-2 challenge has run.** Triage the whole batch through the disposition bar and deep-self-review once, then hold, never deciding by the urge to fix now: applying any fix to a selected-owner tree ends the tracked chain, and round 2 binds the round-1 candidate, so a fix applied between the two forfeits the double-receipt terminal and
|
|
532
|
-
2. **Sum spent rounds across all chains before opening one more;
|
|
531
|
+
1. **Batch dispositions; never re-chain per finding — and hold every fix until the round-2 challenge has run.** Triage the whole batch through the disposition bar and deep-self-review once, then hold, never deciding by the urge to fix now: applying any fix to a selected-owner tree ends the tracked chain, and round 2 binds the round-1 candidate, so a fix applied between the two forfeits the double-receipt terminal and requires a recorded recovery checkpoint before a fresh bounded sequence. So the rule through round 2 is unconditional: accumulate every fix unapplied, run the challenge on the frozen, unchanged round-1 candidate, then apply the held batch, MR/PR-listed, and let round 3's succession challenge — owed exactly when the batch moved the candidate — be what inspects it.
|
|
532
|
+
2. **Sum spent rounds across all chains before opening one more; each sequence's cap is three rounds across two chains.** Record every prior external round — every sequence and chain, finished or broken — and the cumulative count in the caller-owned task artifact. Within one sequence, the only restart is the single succession challenge opened with `--predecessor-chain-result-file`; never append an over-budget receipt or clear earlier spending. At the cap, or when remaining rounds cannot fund that sequence's closeout floor, run the checkpoint before starting another bounded sequence under existing task authority.
|
|
533
533
|
3. **Front-load packet quality in chain 1.** The first chain's packet must already be the full-context diff (`--unified` wide enough to carry whole files, e.g. `-U200`) with the plan frozen alongside the candidate; narrow packets breed packet-boundary pseudo-findings whose fixes break chains and burn rounds on artifacts of the packet itself.
|
|
534
|
-
4. **At the cap — or at effective exhaustion — the
|
|
534
|
+
4. **At the cap — or at effective exhaustion — finish disposition and reassess the method before another sequence.** On the unchanged candidate, when review and challenge are conclusive and every finding occurrence is source-refuted by first-hand evidence, run the local `complete --finding-dispositions-file` path in the [staged contract](../../code-review/references/staged-review-contract.md#mechanical-self-review-gate). Preserve original findings and the full history; an empty model verdict is not required after evidenced refutation. Otherwise, apply or disposition the final batch, name every post-review fix and unreviewed delta in the MR/PR description, and retain `continuation_authorization_required` or an honest interim record. Unreviewed changes and unresolved risks still need their applicable review or human decision. First trace findings to source, run targeted tests and deep self-review, then fix the evidenced cause, improve missing packet context or change the failed review/diagnostic method. Name the specific remaining verification before starting another necessary bounded sequence under the rule below. A reviewer cap alone never asks the user to renew task authority. Never repeat calls solely to obtain zero findings, report an unreviewed batch as reviewed, or reset cumulative spending.
|
|
535
535
|
|
|
536
536
|
A strictly proven wording-only change has no convergence loop: it uses one
|
|
537
537
|
generic `code-review` pass, records the independent-review row and the
|
|
@@ -539,13 +539,13 @@ challenge-not-required proof, and does not create a schema-v3 multi-round
|
|
|
539
539
|
terminal ledger. This exception does not apply to frontmatter, routing,
|
|
540
540
|
validation, acceptance, example, owner or behavior changes.
|
|
541
541
|
|
|
542
|
-
This budget
|
|
542
|
+
This budget bounds each reviewer sequence. It does **not** stop implementation, tests, debugging or deep self-review, and reaching it requires a method checkpoint before necessary review continues under the existing task scope. Explicit user cost, round-count and stop limits still govern:
|
|
543
543
|
|
|
544
544
|
- A human may request another review or self-review, stop a live review or the overall iteration, commit, or merge. Record human-requested review separately from Agent-autonomous rounds.
|
|
545
545
|
- A human merge/risk decision must come from platform-authenticated authority outside the candidate diff, such as a protected maintainer approval. A repository file, branch flag, CLI argument, environment variable, model statement, or Agent-written note is not human authentication.
|
|
546
546
|
- A narrow authenticated `review_waiver` clears only the review-process gate for the exact candidate and records decision-maker, time, reason, residual findings, and accepted risk.
|
|
547
547
|
- A distinct authenticated `merge_authorization` is the human's final decision for the exact candidate. CI still runs and reports review/build/test/security/compliance failures, but none remains merge-blocking after that decision. Report `merge_authorized_by_human` / `failed_but_human_overridden`; never rewrite any underlying result as `passed` or discard residual findings.
|
|
548
|
-
-
|
|
548
|
+
- **`continuation_authorization`** must first be checked against the original task authorization: necessary in-scope fixes, tests and review are already authorized by default. At a sequence checkpoint, record `continuation_basis=existing-task-scope` in the caller-owned task artifact, with the original authorization reference and scope, the reason another bounded sequence is needed, changed method or added evidence, cumulative rounds, and links between the old sequence's terminal evidence and the new sequence. Each new sequence uses fresh current-candidate bindings and preserves every prior receipt, focus, finding and disposition; no CLI flag or runtime receipt field is added. The existing per-sequence format, timeout and validation bounds remain unchanged. This is inherited task authority, not a new human request for each round: never relabel these calls as newly human-requested or erase earlier spending. Ask only for scope or authority the original task lacks, an explicit user limit that prevents the next action, or a genuine unresolved product/design/risk decision; continue independent authorized work. Continuation waives no review, test or evidence obligation and grants no merge, publication or risk-acceptance authority. Never infer a lane waiver from silence or from authorization to continue.
|
|
549
549
|
|
|
550
550
|
When a round returns findings, hand them to the implementer before another autonomous review. The implementer verifies each failure path, classifies it as a local fix, false positive, deferred risk, or human decision, and records targeted self-review plus tests. Do not blindly apply every suggestion and do not use the reviewer as the primary defect finder.
|
|
551
551
|
|
|
@@ -553,12 +553,12 @@ The mechanical reminder is `self_review_gate`, not prose alone. It records outst
|
|
|
553
553
|
|
|
554
554
|
In this gate, `stop`, `terminal`, `abort`, or `revert` applies to the current reviewer lane, readiness claim, or defective dependent slice unless an authenticated human explicitly stops the overall iteration. Repeated root cause, two no-progress attempts, or recurring findings trigger a method change, narrower reproduction, redesign, validation switch, or parked decision item; they never auto-stop unrelated runnable work.
|
|
555
555
|
|
|
556
|
-
At the final
|
|
556
|
+
At the final round of a bounded sequence, perform the checkpoint before another necessary sequence. If findings remain:
|
|
557
557
|
|
|
558
|
-
- keep fixing local bugs, testing, and self-reviewing under `post_review_budget / human_decision_required`;
|
|
558
|
+
- keep fixing local bugs, testing, and self-reviewing under `post_review_budget / human_decision_required`; these legacy fields describe the spent sequence, so check existing task authority before asking for permission;
|
|
559
559
|
- record the last externally reviewed candidate and every later candidate delta; stale review evidence never certifies changed content;
|
|
560
560
|
- mark findings that need product/design/risk authority as `needs_human_decision`, freeze only dependent work, and continue independent runnable slices;
|
|
561
|
-
- enter `awaiting_human` only
|
|
561
|
+
- enter `awaiting_human` only for an actual missing decision or authority after available authorized work and checkpoint recovery are exhausted. A sequence cap alone is not that blocker.
|
|
562
562
|
|
|
563
563
|
The terminal checkpoint is an extraction closeout record, not a state emitted by
|
|
564
564
|
`review_gate.py`, and its schema-v3 state is derived from evidence rather than
|
|
@@ -574,8 +574,8 @@ stop as race immediately after round 1 rather than spending an illegal challenge
|
|
|
574
574
|
after the terminal predicate already fired.
|
|
575
575
|
It ends in exactly one state:
|
|
576
576
|
|
|
577
|
-
- `ready_for_human_decision`: a real `complete` receipt is `passed / self_reviewed`, binds the final external receipt and exact current candidate, and there is no unresolved finding occurrence, unreviewed delta, or unmatched sweep instance.
|
|
578
|
-
- `continuation_authorization_required`: the final round itself returned `findings / post_review_budget`; a passed/unknown/inconclusive state cannot be relabelled continuation.
|
|
577
|
+
- `ready_for_human_decision`: a real `complete` receipt is `passed / self_reviewed`, binds the final external receipt and exact current candidate, and there is no unresolved finding occurrence, unreviewed delta, or unmatched sweep instance. For `completion_basis=source_refuted_findings`, add the same-directory `finding_dispositions: {file, sha256}` reference. Its digest must match the completion receipt; every original occurrence must also retain its separately bound `source_refuted` class evidence with exactly the same evidence array. This proves consistency and coverage, not the truth of the reasoning or permission to accept risk.
|
|
578
|
+
- `continuation_authorization_required`: the final round itself returned `findings / post_review_budget`; a passed/unknown/inconclusive state cannot be relabelled continuation. Preserve this legacy schema value and first check the original task scope; it does not unconditionally require another user grant.
|
|
579
579
|
- `baseline_race`: the referenced ordered base rows contain a second SHA change, including A→B→A; there is no completion receipt and the unreviewed delta is non-empty. Open findings and unmatched sweep instances remain visible and do not prevent this stop state.
|
|
580
580
|
|
|
581
581
|
Run `scripts/validate_extraction_review_state.py <closeout.json>` before reporting
|
|
@@ -591,16 +591,16 @@ convergence.
|
|
|
591
591
|
For a focused single-skill change:
|
|
592
592
|
- **Round 1 — independent review**: inspect the self-reviewed candidate broadly.
|
|
593
593
|
- **Round 2 — challenge**: after implementer triage — fixes stay HELD: applying any fix before this round breaks the chain, so the challenge runs on the frozen round-1 candidate (self-hosted-chain rule; enumeration item 1 above) — attack the highest-risk unresolved surface with an unprimed prompt.
|
|
594
|
-
- **Round 3 — succession challenge, owed only when the fix batch moved the candidate**: apply the held batch, commit it, then ask `scripts/review_ledger_binding.py --print-candidate` what the landing candidate now hashes to. Unchanged (every finding accepted, pre-existing, or source-refuted) ⇒ the lane ends at round 2 and owes nothing. Changed ⇒ open ONE succeeding chain with `--predecessor-chain-result-file <round-2 receipt>` and challenge the landing candidate on a focus distinct from round 2's. This is the final
|
|
594
|
+
- **Round 3 — succession challenge, owed only when the fix batch moved the candidate**: apply the held batch, commit it, then ask `scripts/review_ledger_binding.py --print-candidate` what the landing candidate now hashes to. Unchanged (every finding accepted, pre-existing, or source-refuted) ⇒ the lane ends at round 2 and owes nothing. Changed ⇒ open ONE succeeding chain with `--predecessor-chain-result-file <round-2 receipt>` and challenge the landing candidate on a focus distinct from round 2's. This is the sequence's final external round; its findings feed the method/authority checkpoint before any necessary next bounded sequence, and the batch lands MR/PR-listed.
|
|
595
595
|
|
|
596
|
-
Broad extractions use the same
|
|
596
|
+
Broad extractions use the same per-sequence bounds, round 3 included on the same condition. Continue necessary implementation and checkpoint-qualified review within the original task scope; retain cumulative history rather than resetting the task.
|
|
597
597
|
|
|
598
598
|
### Anti-patterns
|
|
599
599
|
|
|
600
600
|
- **Landing a fix batch no round ever saw**. The round-1 fix-up itself may introduce bugs, so a batch that moved the candidate owes the succession challenge of round 3 above — the earlier absolute ("always re-challenge after a non-trivial fix-up") was unreachable while the budget was two rounds, and an unreachable obligation reads as satisfied. A candidate the batch did not move owes nothing: the condition is the candidate hash, not the author's sense of how big the fix was.
|
|
601
|
-
- **Iterating external review until zero findings**. Stop
|
|
601
|
+
- **Iterating external review until zero findings**. Stop blind repetition at the configured sequence limit and run the checkpoint. Stabilized or repeated findings require source disposition and a method/design or evidence change before necessary review continues under existing task authority; a genuine decision blocks only its dependent slice.
|
|
602
602
|
- **Treating "no new high-severity findings" as "ready to ship" without recording the deferred items**. Deferred findings still need a written reason in the validation log.
|
|
603
|
-
- **Treating every tiny edit as an automatic new external round**. Re-run deep self-review at the required checkpoint; consume another
|
|
603
|
+
- **Treating every tiny edit as an automatic new external round**. Re-run deep self-review at the required checkpoint; consume another review round only when recorded verification needs and current risk call for it, or when a human explicitly requests one. A fresh bounded sequence never resets task history or cumulative spending — retain both and apply the self-hosted-chain checkpoint above.
|
|
604
604
|
- **Re-running with a softer prompt after fixes**. Use the same adversarial framing every round; weakening the prompt to make later rounds "pass" defeats the purpose.
|
|
605
605
|
|
|
606
606
|
### Recording the loop
|
|
@@ -44,9 +44,9 @@ Primary and official sources:
|
|
|
44
44
|
Disposition:
|
|
45
45
|
|
|
46
46
|
- Supported: a rule's existence is not proof it executed; use behavioral assertions and coverage/firing evidence. The digest-binding practices these sources describe (complete subject sets, provenance, raw-result digests) are sound for supply-chain trust boundaries where authors and verifiers are distinct parties; the local policy below explains why this repository adopts the firing/coverage principle but not the digest binding.
|
|
47
|
-
- Local evidence policy: the impact-chain gate machine-verifies what is cheap and deterministic — an owner-scoped firing path that resolves to its round's added lines (a unique anchor on a changed normative numbered/list rule, or a changed owner executable), the letters/digits-free wording-only classification computed from that round's owner diff, and the owner-level floor that a non-wording package carries at least one `RED-baseline` row (a `semantic-control` label may supplement but never close a package alone, because an author-selected stable label cannot vouch for a different hidden delta). The `behavioral-evidence` and `observed-failure` fields themselves are required author declarations. A digest-bound attestation apparatus for these rows (in-toto-style subject digests, same-prompt model result pairs, command-result envelopes) was built, evaluated against real iteration, and deliberately removed: under the unsigned-repository-local trust model the author can regenerate every hash, so the apparatus only detected stale records — while costing a full-suite rerun and whole-evidence regeneration whenever any owner script changed by a single byte. That cost defeated normal multi-commit iteration (it broke its own author's branch twice), so behavior claims rest on the firing-path gate, honest authorship, and the mandatory independent review/challenge instead of hashes.
|
|
48
|
-
- Local trust model: register rows remain honest-but-fallible workflow evidence, not a hostile-author security boundary. The gate proves that a changed
|
|
49
|
-
- Machine format (relocated from the `SKILL.md` firing-mechanism rule; the local evidence policy above carries the rationale): every added source-register row must carry `behavioral-evidence: RED-baseline` (any observed delta — `observed-failure: yes` requires it) or `semantic-control` (only with `observed-failure: no`), an `observed-failure: yes/no` state, and an owner-scoped `firing-path` — each declaration in its own semicolon-delimited fragment of the cell (`…prose; behavioral-evidence: …; observed-failure: …; firing-path: …`), so a key embedded mid-prose never parses as a declaration. The firing-path anchor is at least 16 characters, occurs once in the file and once in its round's added lines, and lands on a numbered/list Markdown rule with a normative action. A row that survives at HEAD must also resolve to an owner this range actually changes — an owner reverted to its base bytes by a rebase or a base-side conflict resolution leaves the changed set while its row stays behind, and the row then vouches for a change the delivered diff does not contain. There is no author-declared escape from this: a corrective rewrite that back-fills a row for a round which merged red produces the same shape, and it is a deliberate, person-adjudicated repair that can adjudicate this refusal too.
|
|
47
|
+
- Local evidence policy: the impact-chain gate machine-verifies what is cheap and deterministic — an owner-scoped firing path that resolves to its round's added lines (a unique anchor on a changed normative numbered/list rule or table data cell, or a changed owner executable), the letters/digits-free wording-only classification computed from that round's owner diff, and the owner-level floor that a non-wording package carries at least one `RED-baseline` row (a `semantic-control` label may supplement but never close a package alone, because an author-selected stable label cannot vouch for a different hidden delta). The `behavioral-evidence` and `observed-failure` fields themselves are required author declarations. A digest-bound attestation apparatus for these rows (in-toto-style subject digests, same-prompt model result pairs, command-result envelopes) was built, evaluated against real iteration, and deliberately removed: under the unsigned-repository-local trust model the author can regenerate every hash, so the apparatus only detected stale records — while costing a full-suite rerun and whole-evidence regeneration whenever any owner script changed by a single byte. That cost defeated normal multi-commit iteration (it broke its own author's branch twice), so behavior claims rest on the firing-path gate, honest authorship, and the mandatory independent review/challenge instead of hashes.
|
|
48
|
+
- Local trust model: register rows remain honest-but-fallible workflow evidence, not a hostile-author security boundary. The gate proves that a changed owner-scoped rule or table data line (or changed owner executable) exists for every claimed firing path; it does not prove a model run occurred, that a named executable implements the claimed enforcement (a shebang stub passes the static check), that a mangled or ambiguous ledger row was honest (those are warned, not blocked, to avoid false positives on other table shapes), author identity, or non-tampering by an authorized contributor. Independent review/challenge and the fixed checker remain the assurance case.
|
|
49
|
+
- Machine format (relocated from the `SKILL.md` firing-mechanism rule; the local evidence policy above carries the rationale): every added source-register row must carry `behavioral-evidence: RED-baseline` (any observed delta — `observed-failure: yes` requires it) or `semantic-control` (only with `observed-failure: no`), an `observed-failure: yes/no` state, and an owner-scoped `firing-path` — each declaration in its own semicolon-delimited fragment of the cell (`…prose; behavioral-evidence: …; observed-failure: …; firing-path: …`), so a key embedded mid-prose never parses as a declaration. The firing-path anchor is at least 16 characters, occurs once in the file and once in its round's added lines, and lands on a numbered/list Markdown rule with a normative action or a data cell in an explicit Markdown table with a header and delimiter. A table anchor must also be absent from the round's base file; changing only a source link cannot reuse the old definition as evidence. Table headers, delimiters, comments, code examples, and HTML blocks are not definition anchors. The source register itself cannot be a file firing path: evidence does not certify itself. The table check establishes the changed location, not the definition's factual accuracy. A row that survives at HEAD must also resolve to an owner this range actually changes — an owner reverted to its base bytes by a rebase or a base-side conflict resolution leaves the changed set while its row stays behind, and the row then vouches for a change the delivered diff does not contain. There is no author-declared escape from this: a corrective rewrite that back-fills a row for a round which merged red produces the same shape, and it is a deliberate, person-adjudicated repair that can adjudicate this refusal too.
|
|
50
50
|
|
|
51
51
|
- **Round scoping — a row is judged against the round it landed in, never the accumulating range.** A row is authored against one round's diff, so reading the whole `base..HEAD` range to classify it judges the row against work it never described. That mismatch produced both directions of the same defect: an already-gated row turned red once a LATER round touched the same owner (which is what the ledger's superseded-row notes were absorbing), and a description-only round lost its routing-surface locator because an EARLIER round had edited that owner's body. The gate cuts rounds at the commits that touch the ledger along a first-parent line, and each round spans from the previous boundary so work commits sit in the round whose ledger append describes them. A merge that git rebuilds from its two parents is expanded into its branch's own rounds, so a merged worktree round is judged exactly as its pull request was — the same history must not partition differently after it lands; a merge git cannot rebuild (a hand resolution, a conflict) keeps a single boundary at the merge, so content that came from neither parent is never left in no round. The partition is derived from git alone — an author cannot nominate, widen, or move their own scope.
|
|
52
52
|
- Both obligations move together, in opposite directions. **Classification** narrows to the round: whether a diff is wording-only, an identifier retarget, or description-only is asked of that round's bytes, which is what makes a verdict stable once it lands. **Presence** narrows to the round too: the round that changed an owner is the round that owes the row, so owner work committed after a ledger append can no longer ride on an earlier round's row. Narrowing classification without narrowing presence would have opened exactly that laundering route.
|
|
@@ -209,15 +209,15 @@ skill 改动后,让 agent 重跑这条 trace,**结构性偏离 = 回归信
|
|
|
209
209
|
| **副作用边界触达** | 任何 destructive op(rm -rf / force-push / drop table / cross-team shared-doc overwrite)| stop + 等用户确认 |
|
|
210
210
|
| **不可解决依赖** | 等外部服务 / 等人审批 / 等数据到 | stop + 报当前状态 + 等依赖解除 |
|
|
211
211
|
|
|
212
|
-
**Warning
|
|
212
|
+
**Warning(命中即报告并自查 / 调整方法;原任务授权覆盖必要续行,显式用户限制仍优先)**:
|
|
213
213
|
|
|
214
214
|
| Trigger | 启发阈值 | 决策 |
|
|
215
215
|
|---|---|---|
|
|
216
|
-
| **同一失败重复 N 次** | N = 3(同 error 第 4 次出现)| 报告 +
|
|
217
|
-
| **预算 warning** | tool call > 100 / token 紧张 / wallclock > 30 分钟 |
|
|
216
|
+
| **同一失败重复 N 次** | N = 3(同 error 第 4 次出现)| 报告 + 停止相同重试,核对失败证据后调整方法或补上下文;**严格指"identical retry"**,不是 challenge 多轮发现新问题 |
|
|
217
|
+
| **预算 warning** | tool call > 100 / token 紧张 / wallclock > 30 分钟 | 报中间状态、累计用量和下一步依据,按原授权继续必要工作;仅缺权限、超出范围、真实取舍或用户显式限制阻断该行动时等人,不因启发阈值重新请批 |
|
|
218
218
|
|
|
219
219
|
**与 convergence standard 的区别**(重要):本节阈值针对 **same-error retry**(重复尝试同一失败方案),不替代 [convergence standard](../SKILL.md)(针对 challenge round — 只要每轮还有 P1 required 就继续,不按 round 计数停)。区分:
|
|
220
|
-
- **Same-error retry**:每次尝试本质相同方案 →
|
|
220
|
+
- **Same-error retry**:每次尝试本质相同方案 → 命中上表阈值后停止相同重试并调整方法,不能只换名称继续重试
|
|
221
221
|
- **Challenge convergence round**:每轮发现不同新层 / 新 P1 → 继续直到 0 P1 required(recursive self-validation 类 extraction 可能走 5-6 轮,每轮抓不同层的新 P1,不该按 round count 停)
|
|
222
222
|
|
|
223
223
|
**Escalation 必含 5 字段**(避免"escalate"沦为"stop"):
|
|
@@ -6,11 +6,11 @@ Companion to the **auto-trigger durable learning** Core Rule in `SKILL.md` (unde
|
|
|
6
6
|
|
|
7
7
|
Bind recovery when either:
|
|
8
8
|
|
|
9
|
-
- **(a)**
|
|
9
|
+
- **(a)** one concrete original proposal, its scope, and the assent are preserved verbatim in the visible conversation or read back verbatim from trusted host-owned session state; a `proposed-next:` marker is not required; or
|
|
10
10
|
- **(b)** the current user's premature-stop correction itself literally names the paused action and scope — on path (b), quote those exact user words in the visible `continuing:` line before proceeding.
|
|
11
11
|
|
|
12
|
-
A semantic compaction paraphrase
|
|
12
|
+
A semantic compaction paraphrase supplies neither binding path; recover the original proposal and assent before deciding path (a) is unavailable. A bare "why did you stop" complaint does not itself name path (b)'s action and scope. Never copy real conversation text into a shared repository record, reconstruct, broaden, or substitute it. The user's challenge reactivates that exact slice. Restate and proceed when either path binds; ask only when the action, scope, or required authority remains unresolved. A new user message or a changed gate requires reassessment, not automatic reconfirmation.
|
|
13
13
|
|
|
14
14
|
## Invalid `blocked:` recovery
|
|
15
15
|
|
|
16
|
-
A `blocked:` recovery without
|
|
16
|
+
A `blocked:` recovery without applicable state evidence and a specific remaining blocker is invalid: recover intent and rerun the owning gate. If a decision or permission remains unresolved, ask in the same turn and block that dependent action. Continue available authorized diagnosis, bounded remediation, or independent work; do not let stale assent bypass a newly pending or inconclusive gate. Do not let correction RCA or extraction delay recovery of a still-authorized delivery.
|
|
@@ -631,3 +631,27 @@ Supersede note (round 115, sixth ledger correction with no rule change, two rows
|
|
|
631
631
|
Supersede note (round 115, seventh ledger correction with no rule change): the observation-validity row above says all 4,905 non-error verdicts in the round's archived reports name a catalog skill or `none`. That total is counted over the archived runner reports in the maintainer's scratch directory, which are not committed and cannot be recounted from this repository. What the repository does carry is `replica-verdicts.tsv`, whose 2430 non-error verdicts all resolve to a catalog skill or `none`; the archive-wide figure stands as the author's count, not as committed evidence.
|
|
632
632
|
| A round partition that depends on which ref is judged moves verdicts after they land: a branch judged per ledger commit as a pull request collapsed to one round once merged, so a row valid at pull-request time (a routing-surface `#description` anchor in a commit that changed only the description) was refused on every post-merge evaluation with nothing about it changed; a merge git rebuilds from its two parents must be expanded into the branch's own rounds so the same history partitions identically before and after it lands, and a merge git cannot rebuild keeps one boundary so content from neither parent is never left in no round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/impact-chain-gate.rb, its round-scoping fixtures, the verdict differential's named divergences, references/external-practice-controls.md and the .github/workflows/ci.yml checkout comments). Observed failure: the integration branch's promotion pull request reported `impact_chain_firing_path_missing` for a row that was green on its own pull request (15 rounds on the branch head, one after the merge), and the integration branch's push build had been red since that merge. RED baseline: round scoping 8 now runs the branch view and the merged view on one fixture and asserts them equal; on the previous gate it fails with `expected rc=1 got rc=0`, and the real promotion shape (integration head against the target) goes from rc=1 to rc=0 with no other change. Verdict differential: 64 integration points, six newly refused, each named by sha with `impact_chain_gate_missing` — all merges from before CI checked out the branch head, whose branches carry owner work outside the round that declares it; none newly accepted. Round scoping 13 pins the observed shape (body round then description-only round, merged) green and equal to its branch view; round scoping 14 pins that a merge whose tree is not the automatic merge keeps a single boundary. Refines the row above beginning "确定性闸对历史形态有前提": the checkout ref binding stays, and the partition no longer depends on it. |
|
|
633
633
|
| A landing chain that looks for a round's review evidence only inside that round's own checkout can never bind a round that merged without its ledger, however honestly the same bytes are reviewed later: evidence is a validator-accepted closeout whose candidate hash equals the round's packet, so the chain reads the landing tree's committed evidence, and a later review of exactly those bytes, landed as a round of its own, binds the earlier round — a closeout for any other digest binds nothing, wherever it sits | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Observed failure: the same promotion pull request's chain refused at a two-file round that had merged with the binder red (`no accepted review evidence binds the landing candidate`), and no forward path existed because the rebind enumerated evidence from the round's detached checkout. RED baseline: the chain case "a validator-accepted closeout for the round's own candidate, committed on the integration branch after the merge, binds it through the chain" fails on the previous binder and passes on this one; the companion case with a closeout for a different digest is refused on both. The round's packet is still frozen from its own checkout at its own base with the landing tree's controller, and its excludes still come from its own added receipts, so a later ledger's absence in the round checkout leaves the round's hash unchanged. The retrospective review of that round's bytes is committed in this round's evidence directory (its retro-round folder) and binds its candidate hash. |
|
|
634
|
+
| Self-review uniqueness is per owner and concern while required owner and concern coverage remain independent | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`; implementation in `code-review/scripts/review_gate.py`. The same nine-owner synthetic harness failed its review and completion assertions before the change while eleven controls passed; afterward all thirteen assertions passed. Missing owners, missing concerns, duplicate pairs, and implicit-versus-explicit default-owner duplicates remain rejected before provider execution. |
|
|
635
|
+
| Actionable capacity warnings can use internal measurements while SLO paging uses the corresponding SLIs | `platform-observability` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-observability/references/alerting-and-on-call.md#diagnostic counters do not become availability | updated | Owner key `platform-observability/SKILL.md`; detail in `platform-observability/references/alerting-and-on-call.md`. The source-contract check alerts_not_exclusively_sli failed against the prior blanket exclusion of raw measurements and passed after correction. Actionability, response ownership, severity, and runbook requirements remain. This is source-contract evidence, not a model-task or production-monitoring trial. |
|
|
636
|
+
| HTTP retry configuration follows route ownership and every replay remains within the caller budget and replay-safety boundary | `platform-service-connectivity` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-service-connectivity/SKILL.md#a 5xx or missing response alone never proves replay safe | updated | Owner key `platform-service-connectivity/SKILL.md`; detail in `platform-service-connectivity/references/retry-timeout-circuit-breaker.md`. Paired source-contract checks rejected the former DestinationRule retry ownership, impossible timeout ordering, and retry-count arithmetic. Corrected arithmetic yields a 500 ms caller budget and sixteen attempts for two layers each allowing three retries. These checks establish instruction and arithmetic corrections, not a live mesh trial. |
|
|
637
|
+
| Evaluation exceptions require bounded process cleanup and preserve the original failure rather than producing successful samples | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/skill-behavior-eval.py | updated | Owner key `skill-extraction-workflow/SKILL.md`; regression cases in `skill-extraction-workflow/scripts/test_eval_runtime.py`. Before the exception-path correction, two focused tests reported four assertion failures and one drain error; afterward all thirteen runtime tests passed. Cleanup uncertainty remains explicit and stops subsequent evaluation work; non-timeout exceptions retain their original identity. |
|
|
638
|
+
|
|
639
|
+
Pending evidence classification: the Deployment Rework Rate table in `product-rd-workflow/references/delivery-lifecycle.md` now defines unplanned deployments caused by production incidents as a proportion of all deployments, with the DORA source linked in that table. This is a factual source comparison. The current impact-chain gate requires a changed normative list rule or executable as the firing path and does not accept this table cell. No behavioral RED baseline or gate acceptance is claimed for that correction.
|
|
640
|
+
|
|
641
|
+
The pending classification above is superseded by the executed source comparison and table-locator regression below. These checks establish source-contract and gate behavior, not model-task improvement.
|
|
642
|
+
|
|
643
|
+
| Deployment rework rate counts unplanned deployments caused by production incidents as a share of all deployments | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/delivery-lifecycle.md#由生产事故引发的非计划部署占全部部署的比例 | updated | Owner key `product-rd-workflow/SKILL.md`. The executed before/after source-contract comparison checked the production-incident cause, unplanned-deployment numerator, and all-deployments denominator against [DORA's metric definition](https://dora.dev/guides/dora-metrics/). All three were absent from the old table definition and present in the corrected definition. This is a factual source comparison, not a model-task or production trial. |
|
|
644
|
+
| Changed definition tables can provide a firing path without inventing a normative list rule | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. The accepted-definition fixture failed before the gate change with expected rc=0 and actual rc=1. Afterward the complete suite passed: one full-checker wiring case and 112 standalone-gate cases. Fifteen table cases cover acceptance plus stale or unchanged anchors, foreign owners, duplicate anchors, comments, fences, raw HTML, indented code, missing headers, header anchors, and missing RED evidence. Existing owner, round, normative-list, and executable checks remain. |
|
|
645
|
+
| An impact-chain evidence row cannot use the source register itself as its firing path | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. A synthetic owner bookkeeping edit and a self-citing table row passed the prior gate with actual rc=0 where refusal rc=1 was required. The anchor occurred only once, inside its own locator, so uniqueness alone did not prevent self-certification. Rejecting the source register as a file firing path made that case pass; the full focused suite passed one checker wiring case and 113 standalone-gate cases. Ordinary changed definition tables still pass. |
|
|
646
|
+
| Table firing anchors exclude raw HTML processing instructions, declarations, CDATA and multiline raw-tag openers | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. An executed paired check of the current table predicate accepted plain data and also accepted data inside processing-instruction, declaration and CDATA blocks. The full gate fixture additionally returned rc=0 for a raw script opener where refusal rc=1 was required. Explicit block terminators now exclude these non-table surfaces; a completed CDATA block followed by a real table remains accepted. The full focused suite passed one checker wiring case and 118 standalone-gate cases. |
|
|
647
|
+
| Diagnosed code and test repairs require implementer self-checks and automatic independent review before completion | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/defect-diagnosis/SKILL.md#Code/test changes require self-checks | updated | Owner key `defect-diagnosis/SKILL.md`. `test_ai_coding_implementation_gates.sh` checks this owner's automatic-review trigger. An applied removal of that trigger failed its owning assertion; unchanged and restored controls passed. The current focused suite also passes with the trigger expressed as a normative list rule and the original introduction retained. This proves source-contract coverage, not automatic invocation in every agent runtime. |
|
|
648
|
+
| LLM and inference code or test changes require implementer self-checks and automatic independent review before completion | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/SKILL.md#Code/test changes require self-checks | updated | Owner key `llm-inference-integration/SKILL.md`. `test_ai_coding_implementation_gates.sh` checks this owner's automatic-review trigger. An applied removal of that trigger failed its owning assertion; unchanged and restored controls passed. The current focused suite also passes with the trigger expressed as a normative list rule and the original introduction retained. This proves source-contract coverage, not automatic invocation in every agent runtime. |
|
|
649
|
+
| Node.js service code and test changes require implementer self-checks and automatic independent review before completion | `nodejs-service-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/nodejs-service-dev/SKILL.md#Code/test changes require self-checks | updated | Owner key `nodejs-service-dev/SKILL.md`. `test_ai_coding_implementation_gates.sh` checks this owner's automatic-review trigger. An applied removal of that trigger failed its owning assertion; unchanged and restored controls passed. The current focused suite also passes with the trigger expressed as a normative list rule and the original introduction retained. This proves source-contract coverage, not automatic invocation in every agent runtime. |
|
|
650
|
+
| Observability code and test changes require implementer self-checks and automatic independent review before completion | `platform-observability` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-observability/SKILL.md#Code/test changes require self-checks | updated | Owner key `platform-observability/SKILL.md`. `test_ai_coding_implementation_gates.sh` checks this owner's automatic-review trigger. An applied removal of that trigger failed its owning assertion; unchanged and restored controls passed. The current focused suite also passes with the trigger expressed as a normative list rule and the original introduction retained. This proves source-contract coverage, not automatic invocation in every agent runtime. |
|
|
651
|
+
| Release-engineering code and test changes require implementer self-checks and automatic independent review before completion | `platform-release-engineering` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-release-engineering/SKILL.md#Code/test changes require self-checks | updated | Owner key `platform-release-engineering/SKILL.md`. `test_ai_coding_implementation_gates.sh` checks this owner's automatic-review trigger. An applied removal of that trigger failed its owning assertion; unchanged and restored controls passed. The current focused suite also passes with the trigger expressed as a normative list rule and the original introduction retained. This proves source-contract coverage, not automatic invocation in every agent runtime. |
|
|
652
|
+
| Service-connectivity code and test changes require implementer self-checks and automatic independent review before completion | `platform-service-connectivity` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-service-connectivity/SKILL.md#Code/test changes require self-checks | updated | Owner key `platform-service-connectivity/SKILL.md`. `test_ai_coding_implementation_gates.sh` checks this owner's automatic-review trigger. An applied removal of that trigger failed its owning assertion; unchanged and restored controls passed. The current focused suite also passes with the trigger expressed as a normative list rule and the original introduction retained. This proves source-contract coverage, not automatic invocation in every agent runtime. |
|
|
653
|
+
| Executable test changes require implementer self-checks and automatic independent review before completion | `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/testing-strategy/SKILL.md#Code/test changes require self-checks | updated | Owner key `testing-strategy/SKILL.md`. `test_ai_coding_implementation_gates.sh` checks this owner's automatic-review trigger. An applied removal of that trigger failed its owning assertion; unchanged and restored controls passed. The current focused suite also passes with the trigger expressed as a normative list rule and the original introduction retained. This proves source-contract coverage, not automatic invocation in every agent runtime. |
|
|
654
|
+
| Repeated identical verification failures require a reported method checkpoint; existing task authority continues to cover necessary work while genuine missing decisions remain blockers | `multi-agent-delegation` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/multi-agent-delegation/SKILL.md#Repeated identical verification failures | updated | Owner key `multi-agent-delegation/SKILL.md`. The warning-family checks in `test_ai_coding_implementation_gates.sh` reject deletion of inherited task authority, real-blocker handling or explicit user limits. Applied mutations failed their owning assertions and both controls passed; the current focused suite passes. Unknown completion, unavailable dependencies, unclear direction and missing authority remain actionable blockers after bounded remediation. Evidence is source-contract validation, not a host-enforced authorization mechanism. |
|
|
655
|
+
| Continuation recovers the current request and original authorized proposal, scopes blockers to dependent work, and uses a method checkpoint instead of renewing permission for necessary review | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#do not stop at a recommendation | updated | Owner key `product-rd-workflow/SKILL.md`. `test_ai_coding_implementation_gates.sh` binds the entry rules to `product-rd-workflow/references/pre-final-continuation-gate.md`; applied trigger and boundary removals fail their owning assertions and restored controls pass. The current suite passes. Classification fixtures cover inherited review authority, explicit review limits and out-of-scope review; their labels are not proof of tool execution or universal runtime improvement. |
|
|
656
|
+
| A complete checkpoint may bind source-refuted findings without rewriting external receipts or refreshing review authority | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/review_gate.py; bank-evidence: file:specs/continuation-control/routing-evidence.md#The code-review routing comparison must preserve | updated | Owner key `code-review/SKILL.md`. `code-review/scripts/test_review_client_compat.py` exercises `CompletionFindingDispositionTest`: the former passed-only predicate rejected complete same-candidate refutation evidence; the current 17 focused tests pass. Original ordered receipt hashes, canonical occurrence coverage, disposition evidence and candidate bindings remain checked; omitted or altered evidence, duplicate dispositions and unresolved findings are rejected. Validation establishes binding and coverage, not the truth of source reasoning. |
|
|
657
|
+
| Extraction reviewer limits bound each receipt sequence; source disposition, method changes and complete cumulative history govern necessary continuation under existing task authority | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. The warning-family source check failed against the former mandatory-human-warning clause. Seven applied warning and delegation mutations failed their owning assertions with unchanged and restored controls passing; the current implementation-gate suite passes. `skill-extraction-workflow/references/dual-track-review-gate.md` preserves per-sequence bounds, source findings, cumulative spending and genuine decision boundaries. The new owner rows also repair a reproduced impact-chain failure for missing owner evidence; they do not turn source checks into runtime or external-review passes. |
|
|
@@ -120,24 +120,47 @@ rescue Errno::ENOENT
|
|
|
120
120
|
[nil, "claude_not_found"]
|
|
121
121
|
end
|
|
122
122
|
|
|
123
|
-
# Parse a stream-json transcript:
|
|
123
|
+
# Parse a stream-json transcript: observed tools plus a validated terminal result.
|
|
124
124
|
def parse_transcript(stream)
|
|
125
125
|
skills = []
|
|
126
126
|
commands = []
|
|
127
|
+
results = []
|
|
127
128
|
stream.each_line do |line|
|
|
128
|
-
|
|
129
|
+
next if line.strip.empty?
|
|
130
|
+
ev = JSON.parse(line)
|
|
131
|
+
return [skills.uniq, commands, "invalid_stream_event"] unless ev.is_a?(Hash)
|
|
132
|
+
results << ev if ev["type"] == "result"
|
|
129
133
|
next unless ev["type"] == "assistant"
|
|
130
|
-
|
|
134
|
+
message = ev["message"]
|
|
135
|
+
return [skills.uniq, commands, "invalid_stream_event"] unless message.is_a?(Hash) && message["content"].is_a?(Array)
|
|
136
|
+
message["content"].each do |c|
|
|
137
|
+
return [skills.uniq, commands, "invalid_stream_event"] unless c.is_a?(Hash)
|
|
131
138
|
next unless c["type"] == "tool_use"
|
|
139
|
+
input = c["input"]
|
|
140
|
+
return [skills.uniq, commands, "invalid_stream_event"] unless input.is_a?(Hash)
|
|
132
141
|
if c["name"] == "Skill"
|
|
133
|
-
s = (
|
|
142
|
+
s = (input["skill"] || input["command"]).to_s
|
|
134
143
|
skills << s.split(":").last unless s.empty?
|
|
135
144
|
elsif c["name"] == "Bash"
|
|
136
|
-
commands <<
|
|
145
|
+
commands << input["command"].to_s
|
|
137
146
|
end
|
|
138
147
|
end
|
|
139
148
|
end
|
|
140
|
-
[skills.uniq, commands]
|
|
149
|
+
return [skills.uniq, commands, "missing_success_result"] if results.empty?
|
|
150
|
+
return [skills.uniq, commands, "invalid_terminal_result"] unless results.size == 1
|
|
151
|
+
terminal = results.first
|
|
152
|
+
unless terminal["subtype"] == "success"
|
|
153
|
+
return [skills.uniq, commands, "result_#{terminal['subtype']}"]
|
|
154
|
+
end
|
|
155
|
+
unless [nil, false].include?(terminal["is_error"]) &&
|
|
156
|
+
[nil, [], {}].include?(terminal["permission_denials"]) &&
|
|
157
|
+
[nil, 0, "0"].include?(terminal["api_error_status"]) &&
|
|
158
|
+
[nil, "completed"].include?(terminal["terminal_reason"])
|
|
159
|
+
return [skills.uniq, commands, "invalid_terminal_result"]
|
|
160
|
+
end
|
|
161
|
+
[skills.uniq, commands, nil]
|
|
162
|
+
rescue JSON::ParserError, JSON::NestingError
|
|
163
|
+
[skills.uniq, commands, "invalid_stream_event"]
|
|
141
164
|
end
|
|
142
165
|
|
|
143
166
|
if dry_run
|
|
@@ -154,7 +177,8 @@ results = traces.map do |t|
|
|
|
154
177
|
frozen_ref = t["frozen_at_sha"] == "root" ? `git -C #{Shellwords.escape(root)} rev-list --max-parents=0 HEAD`.lines.first.to_s.strip : t["frozen_at_sha"]
|
|
155
178
|
frozen_ok = ancestor?(root, frozen_ref)
|
|
156
179
|
stream, error = run_agent(root, max_turns, timeout_s, t["trigger_prompt"])
|
|
157
|
-
invoked, commands = error ? [[], []] : parse_transcript(stream)
|
|
180
|
+
invoked, commands, stream_error = error ? [[], [], nil] : parse_transcript(stream)
|
|
181
|
+
error ||= stream_error
|
|
158
182
|
a = t["assert"] || {}
|
|
159
183
|
missing = (a["must_invoke_skill"] || []) - invoked
|
|
160
184
|
forbidden_hit = (a["must_not_invoke_skill"] || []) & invoked
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
# upstream-owner skill must declare a behavioral-evidence status and an
|
|
6
6
|
# observed-failure state, and (for non-wording changes) name an owner-scoped
|
|
7
7
|
# FIRING PATH that resolves to this diff — an anchor on a changed normative
|
|
8
|
-
# rule line, or a changed owner executable. The statuses are required author
|
|
8
|
+
# rule or table data line, or a changed owner executable. The statuses are required author
|
|
9
9
|
# declarations; the firing path and the wording-only classification are the
|
|
10
10
|
# machine-verified core. Extracted from the former inline `ruby -e` block in
|
|
11
11
|
# check-ccl-skills.sh so the program gets normal Ruby tooling and no
|
|
@@ -1141,9 +1141,80 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
|
|
|
1141
1141
|
blob[:content].scan(Regexp.new(Regexp.escape(anchor))).length == 1
|
|
1142
1142
|
end
|
|
1143
1143
|
end
|
|
1144
|
+
# Definition/decision tables are firing surfaces too. Recognize explicit
|
|
1145
|
+
# Markdown tables with a header and delimiter; comments and code examples do
|
|
1146
|
+
# not count. This proves location/shape, not the truth of the definition.
|
|
1147
|
+
table_data_anchor_valid = lambda do |scope, parts|
|
|
1148
|
+
prior = blob_at.call(scope.base, parts[:path])
|
|
1149
|
+
next false if prior && prior[:content].include?(parts[:anchor])
|
|
1150
|
+
previous_cells = nil
|
|
1151
|
+
columns = nil
|
|
1152
|
+
fence = nil
|
|
1153
|
+
comment = false
|
|
1154
|
+
raw_html = nil
|
|
1155
|
+
html_block = false
|
|
1156
|
+
head_blob.call(scope, parts[:path])[:content].each_line do |raw_line|
|
|
1157
|
+
line = raw_line.chomp
|
|
1158
|
+
if fence
|
|
1159
|
+
fence = nil if line.match?(/\A {0,3}#{Regexp.escape(fence[0])}{#{fence.length},}\s*\z/)
|
|
1160
|
+
next
|
|
1161
|
+
end
|
|
1162
|
+
if raw_html
|
|
1163
|
+
raw_html = nil if line.match?(raw_html)
|
|
1164
|
+
next
|
|
1165
|
+
end
|
|
1166
|
+
if html_block
|
|
1167
|
+
html_block = false if line.strip.empty?
|
|
1168
|
+
next
|
|
1169
|
+
end
|
|
1170
|
+
hidden = comment || line.include?("<!--") || line.include?("-->")
|
|
1171
|
+
line.scan(/<!--|-->/).each { |marker| comment = marker == "<!--" }
|
|
1172
|
+
if hidden
|
|
1173
|
+
previous_cells = columns = nil
|
|
1174
|
+
next
|
|
1175
|
+
end
|
|
1176
|
+
if (opening = line.match(/\A {0,3}(`{3,}|~{3,})/))
|
|
1177
|
+
fence = opening[1]
|
|
1178
|
+
previous_cells = columns = nil
|
|
1179
|
+
next
|
|
1180
|
+
end
|
|
1181
|
+
terminator = case line
|
|
1182
|
+
when /\A {0,3}<\?/ then /\?>/
|
|
1183
|
+
when /\A {0,3}<!\[CDATA\[/ then /\]\]>/
|
|
1184
|
+
when /\A {0,3}<![A-Z]/ then />/
|
|
1185
|
+
end
|
|
1186
|
+
if terminator
|
|
1187
|
+
raw_html = terminator unless line.match?(terminator)
|
|
1188
|
+
previous_cells = columns = nil
|
|
1189
|
+
next
|
|
1190
|
+
end
|
|
1191
|
+
if line.match?(%r{\A {0,3}</?[A-Za-z][\w-]*(?:\s|>|/|\z)})
|
|
1192
|
+
tag = line[/\A {0,3}<(script|pre|style|textarea)(?:\s|>|\z)/i, 1]
|
|
1193
|
+
raw_html = %r{</#{tag}\s*>}i if tag && !line.match?(%r{</#{tag}\s*>}i)
|
|
1194
|
+
html_block = !tag
|
|
1195
|
+
previous_cells = columns = nil
|
|
1196
|
+
next
|
|
1197
|
+
end
|
|
1198
|
+
unless line.match?(/\A {0,3}\|.*\|\s*\z/)
|
|
1199
|
+
previous_cells = columns = nil
|
|
1200
|
+
next
|
|
1201
|
+
end
|
|
1202
|
+
cells = line.strip[1...-1].split(/(?<!\\)\|/, -1).map(&:strip)
|
|
1203
|
+
delimiter = cells.length >= 2 && cells.all? { |cell| cell.match?(/\A:?-{3,}:?\z/) }
|
|
1204
|
+
if delimiter
|
|
1205
|
+
columns = previous_cells && previous_cells.length == cells.length ? cells.length : nil
|
|
1206
|
+
elsif columns && columns == cells.length
|
|
1207
|
+
break true if cells.any? { |cell| cell.include?(parts[:anchor]) }
|
|
1208
|
+
else
|
|
1209
|
+
columns = nil
|
|
1210
|
+
end
|
|
1211
|
+
previous_cells = cells
|
|
1212
|
+
end == true
|
|
1213
|
+
end
|
|
1144
1214
|
enforcing_file_locator_valid = lambda do |scope, parts|
|
|
1145
1215
|
next false unless parts && parts[:kind] == "file"
|
|
1146
1216
|
next false unless parts[:path].end_with?(".md")
|
|
1217
|
+
next false if parts[:path] == LEDGER_PATH # Evidence cannot certify itself.
|
|
1147
1218
|
next false unless locator_valid.call(scope, "file:#{parts[:path]}##{parts[:anchor]}")
|
|
1148
1219
|
line = added_lines_for.call(scope, parts[:path]).find { |added| added.include?(parts[:anchor]) }
|
|
1149
1220
|
next false unless line
|
|
@@ -1161,7 +1232,7 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
|
|
|
1161
1232
|
# "the adapter does not support"), and single characters with broad
|
|
1162
1233
|
# compounds (应/只/别 — 应用/只是/区别).
|
|
1163
1234
|
normative = line.match?(/(?:\b(?:must|shall|never|do\s+not|don'?t|required?|requires?|block(?:s|ed)?|reject(?:s|ed)?|deny|denied|invalidates?|forbid(?:s|den)?|cannot|enforcement)\b|必须|不得|禁止|拒绝|作废|仅限|只能|应当|应该|务必|不能|不允许|不可)/i)
|
|
1164
|
-
list_rule && normative
|
|
1235
|
+
(list_rule && normative) || table_data_anchor_valid.call(scope, parts)
|
|
1165
1236
|
end
|
|
1166
1237
|
# A routing-surface-only owner has no changed rule line to anchor on: its whole
|
|
1167
1238
|
# change is one YAML scalar. The answer is NOT to exempt it — a description edit
|
|
@@ -1360,7 +1431,7 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
|
|
|
1360
1431
|
firing_parts = locator_parts.call(firing_path)
|
|
1361
1432
|
firing_path_valid = firing_locator_valid.call(row_scope, firing_parts, owner)
|
|
1362
1433
|
# The machine-checked core is the FIRING PATH (an owner-scoped anchor on a
|
|
1363
|
-
# changed normative rule, or a changed owner executable) plus the
|
|
1434
|
+
# changed normative rule/table data, or a changed owner executable) plus the
|
|
1364
1435
|
# deterministic wording-only classification. The behavioral-evidence
|
|
1365
1436
|
# status and observed-failure fields are required author declarations —
|
|
1366
1437
|
# honest labels, not digest-verified artifacts: a digest-bound evidence
|