@ccoalm/ccl-skills 0.18.1 → 0.18.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/README.md +2 -0
  2. package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-policy.md +50 -0
  3. package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +30 -40
  4. package/dist/assets/marketplace/plugins/ccl-skills/hooks/guard-delegation-owner.sh +9 -122
  5. package/dist/assets/marketplace/plugins/ccl-skills/hooks/guard-edit-isolation.sh +43 -9
  6. package/dist/assets/marketplace/plugins/ccl-skills/hooks/guard-merge-authorization.sh +95 -2
  7. package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +42 -1
  8. package/dist/assets/marketplace/plugins/ccl-skills/hooks/host-input.py +560 -0
  9. package/dist/assets/marketplace/plugins/ccl-skills/hooks/merge-authorization-prompt.sh +103 -26
  10. package/dist/assets/marketplace/plugins/ccl-skills/hooks/owner-dispatch-guard.sh +17 -8
  11. package/dist/assets/marketplace/plugins/ccl-skills/hooks/proposed-next-stop.sh +14 -0
  12. package/dist/assets/marketplace/plugins/ccl-skills/hooks/session-start.sh +31 -2
  13. package/dist/assets/marketplace/plugins/ccl-skills/hooks/skill-context-compact.sh +10 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/hooks/skill-extraction-gate-stop.sh +17 -4
  15. package/dist/assets/marketplace/plugins/ccl-skills/hooks/skill-loading.py +277 -0
  16. package/dist/assets/marketplace/plugins/ccl-skills/hooks/task-entry.sh +46 -0
  17. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_guard_delegation_owner.sh +61 -55
  18. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_guard_merge_authorization.sh +140 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_host_input.py +630 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_merge_authorization_prompt.sh +56 -3
  21. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_proposed_next.py +292 -0
  22. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_session_start.sh +85 -3
  23. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_skill_loading.py +278 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_task_entry.py +103 -0
  25. package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +36 -6
  26. package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/README.md +23 -15
  27. package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +41 -2
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +5 -5
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +12 -1
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +26 -1
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +9 -1
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +11 -0
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +2 -2
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh +5 -0
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +59 -5
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/SKILL.md +1 -1
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/hook-authorization.md +13 -0
  38. package/dist/assets/release.json +95 -30
  39. package/dist/codex-hooks.d.ts +15 -0
  40. package/dist/codex-hooks.js +186 -0
  41. package/dist/opencode-adapter.js +8 -3
  42. package/dist/operations.js +13 -9
  43. package/package.json +1 -1
@@ -20,8 +20,8 @@
20
20
  # - FAIL-OPEN: any internal error (missing jq/git, parse failure, unwritable/unsafe
21
21
  # state dir, missing session id) ALLOWS the action. A broken gate must never brick
22
22
  # editing, and no error path may fail-CLOSED.
23
- # - DEFAULT `ask`, NOT `deny`: hard `deny` is the explicit `strict:true` opt-in, is
24
- # Claude-Code-only (Codex ignores the decision), applies ONLY to precise
23
+ # - DEFAULT `ask`, NOT `deny`: hard `deny` is the explicit `strict:true` opt-in
24
+ # on compatible native Claude/Codex hooks. It applies ONLY to precise
25
25
  # Edit/Write/MultiEdit/NotebookEdit file paths (never the heuristic Bash match),
26
26
  # and is downgraded to `ask` when the boundary state dir is not safely writable
27
27
  # (so strict can never brick a repo whose state dir is unavailable).
@@ -54,10 +54,17 @@ AIDKEY="" # filename-safe, INJECTIVE key for agent_id, computed by jq @b
54
54
  # collide on the same marker). Empty iff no agent_id. Scopes the activity
55
55
  # marker / block-cap / waiver to ONE subagent.
56
56
 
57
+ HOST_INPUT="$(cd "$(dirname "$SELF")/../.." && pwd)/hooks/host-input.py"
58
+ HOST_TOOL=""
59
+
57
60
  # ----- fail-open helpers ------------------------------------------------------
58
61
  allow_pretool() { exit 0; } # no output => allow
59
62
  allow_stop() { exit 0; } # no output => allow stop
60
63
  emit_pretool_deny() { # $1 reason $2 decision(deny|ask)
64
+ if [ "$2" = ask ] && [ "$HOST_TOOL" = apply_patch ]; then
65
+ jq -nc --arg r "$1" '{hookSpecificOutput:{hookEventName:"PreToolUse",additionalContext:("Advisory owner routing: " + $r)}}'
66
+ exit 0
67
+ fi
61
68
  jq -nc --arg r "$1" --arg d "$2" \
62
69
  '{hookSpecificOutput:{hookEventName:"PreToolUse",permissionDecision:$d,permissionDecisionReason:$r}}'
63
70
  exit 0
@@ -393,6 +400,10 @@ boundary_state() { # $1 root -> echo valid|expired|discontinuous|absent
393
400
  # Emit the base skill names invoked in a transcript, ONE PER LINE (strip any "prefix:").
394
401
  invoked_skills() { # $1 transcript-path
395
402
  [ -n "$1" ] && [ -r "$1" ] || return 0
403
+ if have python3 && [ -r "$HOST_INPUT" ]; then
404
+ python3 "$HOST_INPUT" transcript "$1" 2>/dev/null | jq -r '.requested_skills[] | if startswith("ccl-skills:") then ltrimstr("ccl-skills:") else . end' 2>/dev/null
405
+ return 0
406
+ fi
396
407
  # Per-line jq (JSONL); tolerate malformed lines. A line's message.content[] may
397
408
  # carry tool_use blocks. input.skill is "ccl-skills:foo" or bare "foo".
398
409
  jq -rR '
@@ -415,6 +426,10 @@ invoked_skills() { # $1 transcript-path
415
426
  # transcript or Skill-event shape drifted" without turning valid-JSON drift into a trap.
416
427
  transcript_has_verifiable_invocation_shape() { # $1 transcript-path
417
428
  [ -n "$1" ] && [ -r "$1" ] || return 1
429
+ if have python3 && [ -r "$HOST_INPUT" ]; then
430
+ python3 "$HOST_INPUT" transcript "$1" 2>/dev/null | jq -e '.verifiable' >/dev/null 2>&1
431
+ return $?
432
+ fi
418
433
  jq -seR '
419
434
  [ split("\n")[]
420
435
  | try fromjson catch empty
@@ -666,6 +681,30 @@ cmd_pretool() {
666
681
  AIDKEY=$(printf '%s' "$input" | aid_key_from) # injective key (jq, raw JSON)
667
682
  tool=$(printf '%s' "$input" | jq -r '.tool_name // empty' 2>/dev/null)
668
683
 
684
+ HOST_TOOL="${1:-$tool}"
685
+ if [ "$tool" = apply_patch ]; then
686
+ # Inspect every target, including moves and deletions. A safe first path
687
+ # must not short-circuit checks of the remaining paths in the same patch.
688
+ have python3 && [ -r "$HOST_INPUT" ] || emit_pretool_deny "owner-dispatch cannot inspect apply_patch: Python input normalizer unavailable." deny
689
+ local normalized target converted response deny_response="" advisory_response=""
690
+ normalized=$(printf '%s' "$input" | python3 "$HOST_INPUT" paths 2>/dev/null) || emit_pretool_deny "owner-dispatch cannot inspect apply_patch input." deny
691
+ [ "$(printf '%s' "$normalized" | jq -r '.malformed_patch')" = false ] || emit_pretool_deny "owner-dispatch cannot inspect malformed apply_patch targets." deny
692
+ while IFS= read -r -d '' target; do
693
+ converted=$(printf '%s' "$input" | jq -c --arg p "$target" '.tool_name="Edit" | .tool_input={file_path:$p}')
694
+ response=$(printf '%s' "$converted" | cmd_pretool apply_patch)
695
+ [ -n "$response" ] || continue
696
+ if printf '%s' "$response" | jq -e '.hookSpecificOutput.permissionDecision == "deny"' >/dev/null 2>&1; then
697
+ [ -n "$deny_response" ] || deny_response="$response"
698
+ else
699
+ [ -n "$advisory_response" ] || advisory_response="$response"
700
+ fi
701
+ done < <(printf '%s' "$normalized" | jq -j '.paths[] | . + "\u0000"')
702
+ # An advisory first target must not hide a later strict denial. Inspect all
703
+ # targets before emitting one response so activity evidence is complete too.
704
+ [ -n "$deny_response" ] && { printf '%s\n' "$deny_response"; exit 0; }
705
+ [ -n "$advisory_response" ] && { printf '%s\n' "$advisory_response"; exit 0; }
706
+ allow_pretool
707
+ fi
669
708
  case "$tool" in
670
709
  Edit|Write|MultiEdit|NotebookEdit)
671
710
  fp=$(printf '%s' "$input" | jq -r '.tool_input.file_path // .tool_input.notebook_path // .tool_input.path // empty' 2>/dev/null)
@@ -66,15 +66,15 @@ credentials only; broad semantic confidentiality stays operator-owned per the
66
66
  Diff Confidentiality section. Consult stays on `claude_review.sh`. Load the
67
67
  staged-contract and client-routing references below for details.
68
68
 
69
- Current contract: review/challenge may use a stamped
69
+ Staged contract: review/challenge may use a stamped
70
70
  `review_plan_source=derived-default`; `complete` requires a plan and output uses
71
71
  schema 3. Automation retains one chain (one review, at most four challenges) plus one
72
72
  succession.
73
73
  Positive challenge capacity opens it at index 1; budget zero is untracked.
74
- The sole release/high-risk budget-zero exception is a controller-proved
75
- `markdown-punctuation-only` review: it requires `wording_only_boundary`, permits
76
- no `complete`, and rejects an author assertion alone (recipe:
77
- `references/wording-only-review.md`). After a clean/source-refuted tracked
74
+ Staged release/high-risk budget zero requires controller-proved
75
+ `markdown-punctuation-only`, `wording_only_boundary` and no `complete`; author
76
+ assertions cannot qualify (`references/wording-only-review.md`). Extraction uses
77
+ separate passes (`references/staged-review-contract.md`). After a clean/source-refuted tracked
78
78
  challenge, `complete` may close early and preserve unused rounds. Every result
79
79
  exposes controller-owned `self_review_gate`; an outstanding checkpoint blocks
80
80
  only external review or completion, not implementation or tests. Even a passed
@@ -6,7 +6,7 @@ The controller has three modes:
6
6
  - `challenge`: a focused adversarial external round;
7
7
  - `complete`: a local deep-self-review checkpoint that calls no reviewer.
8
8
 
9
- Explore/build may configure `challenge_budget=0..4`; release/high-risk requires
9
+ In the default staged lane, explore/build may configure `challenge_budget=0..4`; release/high-risk requires
10
10
  at least one challenge unless the exact candidate qualifies for the
11
11
  proof-bound wording-only single-review exception below. The initial review consumes chain round 1;
12
12
  each bounded chain uses at most five rounds. Necessary task-scoped review after a
@@ -48,6 +48,17 @@ Two mechanics that cost time when they are discovered by experiment:
48
48
 
49
49
  ## Plan and owner binding
50
50
 
51
+ The extraction-owned wrapper selects `--review-lane extraction` for its separate
52
+ single-shot review and challenge. This lane retains the full stage and risk
53
+ concerns and records the lane in the bound review scope. The actual candidate
54
+ must derive `skill-extraction-workflow` ownership; a declared owner alone is
55
+ insufficient. Extraction register changes provide that evidence for this
56
+ repository's plugin runtime deliveries. The lane cannot use tracked-chain,
57
+ wording-waiver or completion inputs.
58
+ Its initial review does not carry challenge capacity because the extraction
59
+ owner requires a separate challenge receipt; it never satisfies that challenge.
60
+ The default `staged` lane retains its existing high-risk challenge requirement.
61
+
51
62
  The plan is optional for `review` and `challenge` and required for `complete`.
52
63
  When supplied, the bounded UTF-8 plan contains exactly intent, acceptance,
53
64
  self-review, and evidence. When omitted for review/challenge, the controller
@@ -1981,6 +1981,8 @@ def _canonical_review_scope(profile: dict[str, Any]) -> dict[str, Any]:
1981
1981
  else None
1982
1982
  ),
1983
1983
  }
1984
+ if profile.get("review_lane") == "extraction":
1985
+ scope["review_lane"] = "extraction"
1984
1986
  if _review_scope_digest(scope) != profile["review_scope_sha256"]:
1985
1987
  raise GateError(
1986
1988
  "review scope reconstruction does not reproduce its recorded digest",
@@ -3303,6 +3305,19 @@ def freeze_review_profile(
3303
3305
  raise GateError(
3304
3306
  "--challenge-budget must be between 0 and 4 so the initial review plus challenges never exceeds five Agent-autonomous external rounds"
3305
3307
  )
3308
+ extraction_pass = args.review_lane == "extraction"
3309
+ if extraction_pass and (
3310
+ args.mode not in {"review", "challenge"}
3311
+ or challenge_budget != (0 if args.mode == "review" else 1)
3312
+ or args.review_chain_id is not None
3313
+ or args.autonomous_review_index is not None
3314
+ or args.prior_review_result_file
3315
+ or args.predecessor_chain_result_file
3316
+ or args.completion_review_result_file
3317
+ or args.finding_dispositions_file
3318
+ or args.wording_only_proof_file
3319
+ ):
3320
+ raise GateError("extraction runs separate single-shot review and challenge passes")
3306
3321
  wording_only_proof_sha256: str | None = None
3307
3322
  wording_only_scope: dict[str, Any] | None = None
3308
3323
  if args.wording_only_proof_file:
@@ -3324,7 +3339,7 @@ def freeze_review_profile(
3324
3339
  candidate_paths,
3325
3340
  Path(args.cwd),
3326
3341
  )
3327
- if review_depth == "release" and challenge_budget == 0:
3342
+ if review_depth == "release" and challenge_budget == 0 and not extraction_pass:
3328
3343
  if wording_only_scope is None:
3329
3344
  raise GateError("release and high-risk review require at least one challenge")
3330
3345
  if wording_only_scope["check_kind"] != "markdown-punctuation-only":
@@ -3358,6 +3373,10 @@ def freeze_review_profile(
3358
3373
  owner_selection_evidence = derive_owner_selection(candidate_paths, registry_root)
3359
3374
  derived_skill_names = {item["skill"] for item in owner_selection_evidence}
3360
3375
  declared_skill_names = {item["skill"] for item in self_review}
3376
+ if extraction_pass and "skill-extraction-workflow" not in derived_skill_names:
3377
+ raise GateError(
3378
+ "extraction lane requires controller-derived skill-extraction-workflow ownership"
3379
+ )
3361
3380
  missing_self_review_owners = sorted(
3362
3381
  derived_skill_names - declared_skill_names - {"code-review"}
3363
3382
  )
@@ -3508,6 +3527,9 @@ def freeze_review_profile(
3508
3527
  else None
3509
3528
  ),
3510
3529
  }
3530
+ if extraction_pass:
3531
+ # The lane shape never lowers risk concerns or supplies a challenge receipt.
3532
+ review_scope["review_lane"] = "extraction"
3511
3533
  review_scope_sha256 = _review_scope_digest(review_scope)
3512
3534
  if review_scope_sha256 is None:
3513
3535
  raise GateError("review scope is not representable", "invalid_input")
@@ -3814,6 +3836,8 @@ def freeze_review_profile(
3814
3836
  "self_review": self_review,
3815
3837
  "evidence": evidence,
3816
3838
  }
3839
+ if extraction_pass:
3840
+ profile["review_lane"] = "extraction"
3817
3841
  encoded = json.dumps(
3818
3842
  profile,
3819
3843
  ensure_ascii=False,
@@ -4548,6 +4572,7 @@ def build_parser() -> argparse.ArgumentParser:
4548
4572
  parser.add_argument("--wording-only-proof-file")
4549
4573
  parser.add_argument("--stage", choices=tuple(STAGE_CONCERNS), default="build")
4550
4574
  parser.add_argument("--risk-tag", action="append", default=[])
4575
+ parser.add_argument("--review-lane", choices=("staged", "extraction"), default="staged")
4551
4576
  parser.add_argument("--challenge-budget", type=int)
4552
4577
  # No argparse default: 0 is illegal in challenge mode and required outside
4553
4578
  # it, so a single static default is wrong for one of the two. main() derives
@@ -93,7 +93,15 @@ At the start of the next turn, recover intent in this order:
93
93
  2. For short assent, read back the original wording of the most recent still-active concrete proposal and check that later messages or task state have not withdrawn or superseded its action, scope, or authority. Quote that original proposal when stating the recovered action and scope; a summary or paraphrase alone cannot bind short assent. If recovery adds an action or broadens that quoted scope, select `blocked:` and ask. One recoverable action can bind with or without a marker; a stale, repeated, or conflicting marker is an assistant formatting defect to repair.
94
94
  3. If materially different proposals remain unresolved, or scope/authority is still unclear, ask one targeted question about that uncertainty. Do not ask the user to repair a marker or repeat a clear instruction. A marker alone never supplies missing authority.
95
95
 
96
- Use the active owner's entry and safety gates for the recovered action. An authorized task includes necessary fixes, tests and review by default; neither a router nor a dispatched owner may discard that authority by relabeling its turn or exhausting an internal review sequence. Apply the owning review checkpoint and record `continuation_basis=existing-task-scope` with cumulative history in the caller-owned task artifact, not runtime JSON. Legacy `human_decision_required` / `continuation_authorization_required` values first require checking existing authority, not asking again. Explicit user cost, round-count and stop limits prevail; new scope, missing authority or real tradeoffs need a decision. Continuation grants no merge, publication or waiver authority. This is a prose contract, not a host-enforced hook; it cannot prove compliance by a task that never loads it.
96
+ - Use the active owner's entry and safety gates for the recovered action. An authorized task includes necessary fixes, tests and review by default; neither a router nor a dispatched owner may discard that authority by relabeling its turn or exhausting an internal review sequence. Apply the owning review checkpoint and record `continuation_basis=existing-task-scope` with cumulative history in the caller-owned task artifact, not runtime JSON. Legacy `human_decision_required` / `continuation_authorization_required` values first require checking existing authority, not asking again. Explicit user cost, round-count and stop limits prevail; new scope, missing authority or real tradeoffs need a decision. Continuation grants no merge, publication or waiver authority. Intent recovery and authorization remain prose obligations. The optional `proposed-next-stop.sh` backstop checks missing handoff labels and declared next actions as described below; a label or hook receipt never proves the action is correct, authorized or complete.
97
+
98
+ ### Stop-time continuation reminder
99
+
100
+ On hosts providing a current final message, `proposed-next-stop.sh` returns one bounded Stop reminder when the assistant declares a non-status `proposed-next:` action. Recheck the active request: execute a runnable, already-authorized action in the same turn; otherwise preserve explicit stop, planning-only and status-only scope, or state the concrete decision/resource/authority blocker. Missing labels with observable delivery evidence retain their formatting reminder. A status-only marker without another action declaration, quoted example, complete machine artifact, unsupported payload or host `stop_hook_active` retry does not trigger a continuation reminder.
101
+
102
+ - Do not request continuation for `blocked:` with a concrete explanation or `none` with a dash-separated status explanation. Mixed status/action markers still require reconciliation.
103
+
104
+ The hook recognizes declarations, not authorization or actual task completion, and cannot force the model to follow through. OpenCode idle does not expose the required final-message evidence; its Stop behavior remains unverified.
97
105
 
98
106
  ## Gate triggers and outcome contract
99
107
 
@@ -697,3 +697,14 @@ The pending classification above is superseded by the executed source comparison
697
697
  | 评审模式把被评审仓库自己入库的约定文件引在候选之后交给评审员:从仓库根到每个改动路径的每层目录取 `AGENTS.override.md`(否则 `AGENTS.md`)、`CLAUDE.md`、`.claude/CLAUDE.md`,根在前,32 KiB 以内;未入库和 `CLAUDE.local.md` 一律不读(不得外发给别家评审模型),链接、非 UTF-8、超预算的整份省略并记原因,永不因约定文件让评审失败;有引用时加 `repository_contract` concern,要求评审员报改动违反的规则和规则本身的缺陷(`Contract defect:`),规则只作数据、不能为缺陷开脱 | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md`(入口未改);改动在 `skills/code-review/scripts/review_gate.py`(`repository_contract_section`、profile 与结果的 `repository_contract` 字段、条件 concern、trust boundary 一句)、`skills/code-review/references/staged-review-contract.md`、`skills/code-review/references/development-completion.md`(偏离约定要在 plan intent 里声明、`Contract defect:` 的处置、不许悄悄放松约定文件),`skills/code-review/scripts/test_review_client_compat.py` 的 provider mock 改为放行控制器自己的 git 读取;共享读取函数 `read_bounded_regular_file` 逐级打开目录中途失败时不再泄漏已持有的目录描述符(修复前 100 次失败读取后打开的描述符由 4 个涨到 254 个,约定文件被逐个省略时会反复触发)。发现规则按两家宿主一手文档核对(Codex:每层目录 override 优先、根到深拼接、默认 32 KiB;Claude Code:`CLAUDE.md` 或 `.claude/CLAUDE.md`,`CLAUDE.local.md` 是个人文件)。`candidate_sha256` 在追加前算定,引用内容不能增加候选路径;challenge、complete、wording-only 不附加。不读 `@path` 导入、`.claude/rules`、宿主配置的备用文件名(结果里 `sources_not_read` 列明)。RED-baseline(applied,differential):在副本上分别去掉「只收已入库文件」、去掉 override 优先、不加 concern,三条约定用例里恰好两条变红、其余用例全绿;base 版控制器配新套件时三条约定用例全红;中间目录被换成链接、带第二个硬链接的约定文件都被省略且内容不进评审包。 |
698
698
  | 产品研发验证门里「评审后有实质改动才全量重跑」的「实质」一词删除:评审后的任何改动(含测试 / 文档)都触发全量重跑,作者不能自行判定改动无关紧要 | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#any post-review change, tests/docs included | `updated` | Owner key `product-rd-workflow/SKILL.md`(改一句,字数与字节都不增)。Observed failure 同上一条生产会话:它把评审后的提交归为「只有测试和文档」而没有重审,原句的「实质」正好给了这个归类余地。RED-baseline 同上:红的一半是生产失效,隔离探针在改前文本上未复现,改动效果未被证明。 |
699
699
  | 提炼流程的增量复查上限由两次改为五次(用户裁决),同步 `skills/skill-extraction-workflow/SKILL.md`、`dual-track-review-gate.md`、`extraction-quickstart.md` 与被测试钉住的原文;本行按指针取代 127 轮评审线那一行里的「最多两次」 | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#After five delta passes a still-open P0/P1 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`(同一行内省 6 字节,入口不增长)。delta 复查改为 `--base <已评审提交>` 取增量,使它绑定工作区并留下 PR hook 读取的本机回执。RED-baseline(applied):`test_extraction_review_gate.sh` 钉住的短语由两次改为五次后,对 base 文本变红、对 head 文本变绿。 |
700
+ | Host boundary normalization preserves shared isolation and authorization rules while native inventory separates installation from trust | `worktree-isolation` / `multi-agent-delegation` / `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:hooks/guard-edit-isolation.sh | updated | `specs/129-host-hook-compatibility/plan.md` binds synthetic baseline failures, patch/delegation transcript normalization, bounded target authorization, startup context budgets and native trust diagnostics. Tests cover malformed/multiple/moved paths, incomplete skill reads, revoked/expired/wrong-target grants and incomplete or wrong-plugin hook inventories. Trust receipts remain distinct from runtime effect. |
701
+ | Missing delivery handoffs receive a bounded current-message formatting reminder | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/pre-final-continuation-gate.md#checks missing handoff labels and declared next actions | updated | `product-rd-workflow/SKILL.md` owns the shared handoff rule; `specs/129-host-hook-compatibility/plan.md` records the absent-hook baseline and 18 native-shaped cases. The startup cue and Stop backstop preserve current user scope, pure artifacts and host loop prevention. A marker proves neither authorization nor completion; unsupported host final-message data remains unverifiable. |
702
+ | Catalog fixtures pair candidate contracts with candidate hook executables | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh | updated | `skill-extraction-workflow/SKILL.md` owns the fixture contract. The pristine case failed when a candidate ledger referenced a new hook absent from the committed clone. Copying the candidate hook tree with its skill contracts restores the complete fixture; the existing catalog suite passes without changing gate predicates. |
703
+ | Source editing and delegation receive one bounded skill-loading replan per actor and current context | `product-rd-workflow` / `multi-agent-delegation` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:hooks/test_skill_loading.py | updated | `agent-context/session-start.md` supplies the transition rule and `hooks/skill-loading.py` implements its bounded checkpoint; both named owner skill entries are unchanged in this round. `specs/129-host-hook-compatibility/plan.md` records the missing default first-edit checkpoint, stale post-compaction evidence and incomplete-read false positive. The checkpoint defers one precise edit or cold dispatch attempt, uses the canonical transition rule, never asks the user to approve skill loading, and preserves configured decisions. PreCompact snapshots the previous native boundary; PostCompact invalidates old visibility, and only a confirmed new boundary permits fresh full-read evidence. Guidance is delivered on subsequent PreToolUse. A capped retry or parallel sibling can proceed, so a reminder is neither correct-owner proof nor a substitute for opt-in enforcement. Native execution and model adherence are separate validation claims. |
704
+ | Task entry delivers owner-loading guidance before investigation and analysis | `product-rd-workflow` / `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:hooks/test_task_entry.py | updated | `hooks/task-entry.sh` reuses the canonical startup routing table at native UserPromptSubmit and the OpenCode system transform. `specs/130-first-turn-skill-routing/plan.md` covers the absent prompt-time registration and existing-bootstrap suppression cases, and reinforces continued investigation, safe repair and retesting for unfinished verification. Skill entrypoints and owner boundaries are unchanged. Tests verify early context delivery, current-context reuse instructions, prompt independence, bounded output and fail-soft invalid sources. Context delivery is not a guarantee of model adherence; existing source-edit, delegation and permission checks remain independent. |
705
+ | Declared next actions receive a bounded continuation recheck before stopping | `product-rd-workflow` / `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:hooks/test_proposed_next.py | updated | `specs/130-first-turn-skill-routing/plan.md` covers the non-status handoff bypass in the Stop hook. The hook now asks the agent to execute already-authorized runnable work or preserve the current scope and report a concrete blocker. Status-only markers, quoted actions, machine artifacts and native loop prevention retain their boundaries. The reminder never derives authority from a marker and supplies no task-completion verdict. |
706
+ | Extraction review retains release and high-risk profiles in its separate-pass lane | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. The real wrapper failed before reviewer selection because its zero challenge capacity collided with the staged high-risk guard. The wrapper selects the explicit extraction lane and preserves the separate review/challenge obligation. The regression covers build, release and shared-gate profiles, default staged refusal and same-family exclusion without invoking a model. |
707
+ | Review scope distinguishes extraction passes from staged chains without lowering risk concerns | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/review_gate.py | updated | Owner key `code-review/SKILL.md`. The controller binds extraction lane identity in both profile construction and scope reconstruction, selects the extraction owner and rejects chain/completion/waiver inputs. Default staged high-risk review still requires challenge capacity. The real-wrapper regression and five disposable mutations verify the entry, rejection and scope-binding boundaries. |
708
+ | Extraction cadence requires candidate-derived ownership | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/review_gate.py | updated | Owner key `code-review/SKILL.md`. An ordinary source candidate could select the extraction lane by flag while declaring the owner in its plan. The controller now requires its existing path-derived owner evidence before reviewer selection; it never adds the owner solely because the caller selected the lane. |
709
+ | Ordinary candidates cannot select extraction review or challenge | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. A real-controller negative fixture fails before the ownership repair and passes afterward for both extraction modes, while ordinary staged review still reaches reviewer selection. The declared extraction owner in the fixture plan cannot substitute for candidate-derived evidence. |
710
+ | Blocker and waiting-status handoffs do not request continuation | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/pre-final-continuation-gate.md#Do not request continuation for | updated | Owner key `product-rd-workflow/SKILL.md`. Native-shaped fixtures expose the extra reminder for a concrete blocked handoff or a none marker with a waiting explanation. These statuses now remain non-actionable; a separate action marker and action words that merely start with none still receive the bounded recheck. |
@@ -32,7 +32,7 @@ for arg in "$@"; do
32
32
  # full spellings, so a shortened flag cannot reopen a chain.
33
33
  --challenge-b*|--challenge-i*)
34
34
  fail "the extraction lane fixes the challenge budget and index; do not pass $arg" ;;
35
- --review-c*|--au*|--prio*|--pre*|--com*)
35
+ --review-c*|--review-l*|--au*|--prio*|--pre*|--com*)
36
36
  fail "the extraction lane is single-shot; review-chain option $arg is not accepted" ;;
37
37
  esac
38
38
  done
@@ -48,4 +48,4 @@ if [[ ! -x "$CONTROLLER" ]]; then
48
48
  fail "code-review controller is unavailable"
49
49
  fi
50
50
 
51
- exec bash "$CONTROLLER" "${fixed[@]}" "$@"
51
+ exec bash "$CONTROLLER" --review-lane extraction "${fixed[@]}" "$@"
@@ -60,6 +60,11 @@ new_case() {
60
60
  rm -rf "$CASE_DIR/skills"
61
61
  mkdir -p "$CASE_DIR/skills"
62
62
  cp -R "$candidate_skill_root/skills/." "$CASE_DIR/skills/"
63
+ # Candidate ledger firing paths may name a newly added shipped hook. Keep the
64
+ # executable surface with its candidate contracts instead of committed HEAD.
65
+ rm -rf "$CASE_DIR/hooks"
66
+ mkdir -p "$CASE_DIR/hooks"
67
+ cp -R "$REPO_ROOT/hooks/." "$CASE_DIR/hooks/"
63
68
  cp "$REPO_ROOT/docs/SKILLS.md" "$CASE_DIR/docs/SKILLS.md"
64
69
  cp "$REPO_ROOT/agent-context/session-start.md" "$CASE_DIR/agent-context/session-start.md"
65
70
  cp "$candidate_eval_root/eval/routing-tasks.jsonl" "$CASE_DIR/eval/routing-tasks.jsonl"
@@ -38,6 +38,7 @@ PY
38
38
 
39
39
  CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode review --cwd /synthetic --implementer-family openai
40
40
  assert_contains "--challenge-budget 0 --mode review" "$(captured)" "review is single-shot with no challenge capacity"
41
+ assert_contains "--review-lane extraction" "$(captured)" "wrapper selects its owner lane"
41
42
 
42
43
  CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode challenge --cwd /synthetic --implementer-family openai --focus f
43
44
  assert_contains "--challenge-budget 1 --challenge-index 1 --mode challenge" "$(captured)" "challenge is the single untracked challenge"
@@ -61,6 +62,7 @@ done
61
62
  # Every chain option, full or abbreviated, with or without =VALUE, is refused.
62
63
  for spelling in \
63
64
  --review-chain-id --review-chain-id=x --review-c \
65
+ --review-lane --review-lane=staged --review-l \
64
66
  --autonomous-review-index --autonomous-review-index=2 --au \
65
67
  --prior-review-result-file --prior-review-result-file=/r.json --prio \
66
68
  --predecessor-chain-result-file --pre \
@@ -114,7 +116,7 @@ diff_path.write_text(
114
116
  # keeping a copy that drifts when the set changes.
115
117
  required = subprocess.run(
116
118
  [sys.executable, str(root / "skills/code-review/scripts/review_gate.py"),
117
- "--print-required-concerns", "--stage", "build"],
119
+ "--print-required-concerns", "--stage", "release", "--risk-tag", "shared-gate"],
118
120
  capture_output=True, text=True, check=True,
119
121
  ).stdout.split()
120
122
  assert required, "the controller printed no required concerns"
@@ -144,24 +146,76 @@ PY
144
146
  mkdir -p "$TMP/fake-bin"
145
147
  printf '%s\n' '#!/usr/bin/env bash' 'printf invoked >"$CODEX_MARKER"' 'exit 99' >"$TMP/fake-bin/codex"
146
148
  chmod +x "$TMP/fake-bin/codex"
149
+ run_real_controller() {
150
+ CODEX_MARKER="$TMP/codex-invoked" PATH="$TMP/fake-bin:$PATH" CODE_REVIEW_CLIENT_ORDER=codex \
151
+ "$REAL_CONTROLLER" "$@"
152
+ }
147
153
  real_args=(
148
154
  --cwd "$ROOT" --diff-file "$TMP/real.diff" --implementer-family openai
149
- --review-plan-file "$TMP/real-plan.json" --stage build --review-harness
155
+ --review-plan-file "$TMP/real-plan.json" --review-harness
150
156
  --timeout 5 --total-timeout 5
151
157
  )
152
- for pass in review challenge; do
158
+ for profile in build release shared-gate; do
159
+ risk_args=(--stage "$profile")
160
+ [ "$profile" = shared-gate ] && risk_args=(--stage build --risk-tag shared-gate)
161
+ for pass in review challenge; do
153
162
  extra=()
154
163
  [ "$pass" = challenge ] && extra=(--focus "single-shot probe")
155
164
  set +e
156
165
  out="$(CODEX_MARKER="$TMP/codex-invoked" PATH="$TMP/fake-bin:$PATH" CODE_REVIEW_CLIENT_ORDER=codex \
157
- "$WRAPPER" --mode "$pass" "${real_args[@]}" "${extra[@]}" 2>&1)"
166
+ "$WRAPPER" --mode "$pass" "${real_args[@]}" "${risk_args[@]}" "${extra[@]}" 2>&1)"
158
167
  rc=$?
159
168
  set -e
160
- assert_rc "$rc" 2 "real $pass must stop before model inference"
169
+ assert_rc "$rc" 2 "real $profile $pass must stop before model inference: $out"
161
170
  assert_contains '"reason_code":"no_independent_reviewer_available"' "$out" "real $pass reaches reviewer selection"
162
171
  assert_not_contains 'review_chain_required' "$out" "real $pass needs no chain"
163
172
  assert_not_contains 'review_chain_invalid' "$out" "real $pass needs no chain"
173
+ assert_contains '"review_lane":"extraction"' "$out" "$profile $pass binds the extraction lane"
174
+ done
175
+ if [ "$profile" != build ]; then
176
+ set +e
177
+ out="$(run_real_controller --mode review --challenge-budget 0 "${real_args[@]}" "${risk_args[@]}" 2>&1)"
178
+ rc=$?
179
+ set -e
180
+ assert_rc "$rc" 2 "staged $profile review still requires challenge capacity"
181
+ assert_contains 'release and high-risk review require at least one challenge' "$out" "staged risk guard remains active"
182
+ fi
183
+ done
184
+
185
+ # The explicit owner lane must never turn into a chain or claim completion.
186
+ for incompatible in '--review-chain-id probe' '--challenge-budget 1' '--wording-only-proof-file /missing'; do
187
+ set +e
188
+ # shellcheck disable=SC2086
189
+ out="$(run_real_controller --review-lane extraction --mode review --challenge-budget 0 \
190
+ "${real_args[@]}" --stage release $incompatible 2>&1)"
191
+ rc=$?
192
+ set -e
193
+ assert_rc "$rc" 2 "extraction rejects $incompatible"
194
+ assert_contains 'extraction runs separate single-shot review and challenge passes' "$out" "extraction configuration fails closed"
195
+ done
196
+ [ ! -e "$TMP/codex-invoked" ] || fail "same-family Codex executable was invoked"
197
+
198
+ # Declaring the extraction owner in the plan cannot route an ordinary candidate
199
+ # around staged high-risk requirements. The real candidate must derive the owner.
200
+ sed 's@skills/skill-extraction-workflow/scripts/extraction_review_gate.sh@src/example.py@g' \
201
+ "$TMP/real.diff" >"$TMP/ordinary.diff"
202
+ for pass in review challenge; do
203
+ extra=(--challenge-budget 0)
204
+ [ "$pass" = challenge ] && extra=(--challenge-budget 1 --focus "ordinary candidate")
205
+ set +e
206
+ out="$(run_real_controller "${real_args[@]}" --diff-file "$TMP/ordinary.diff" \
207
+ --review-lane extraction --mode "$pass" --stage release "${extra[@]}" 2>&1)"
208
+ rc=$?
209
+ set -e
210
+ assert_rc "$rc" 2 "ordinary candidate cannot use extraction $pass"
211
+ assert_contains 'extraction lane requires controller-derived skill-extraction-workflow ownership' "$out" "candidate ownership is required"
164
212
  done
213
+ set +e
214
+ out="$(run_real_controller "${real_args[@]}" --diff-file "$TMP/ordinary.diff" --mode review --stage build 2>&1)"
215
+ rc=$?
216
+ set -e
217
+ assert_rc "$rc" 2 "ordinary staged review reaches reviewer selection"
218
+ assert_contains '"reason_code":"no_independent_reviewer_available"' "$out" "ordinary staged scope stays valid"
165
219
  [ ! -e "$TMP/codex-invoked" ] || fail "same-family Codex executable was invoked"
166
220
 
167
221
  # The owner documents must route non-wording work through this wrapper and must
@@ -147,7 +147,7 @@ worktree 的活一旦**集成进目标分支**就完了,立刻清理(唯一
147
147
  **合并执行协议(canonical——always-on 层「硬纪律 1」指向本节,两面同步修改;执行配方只放这里,不进 always-on 层)**:
148
148
  1. **按目标判断授权**:用户已要求“做完并合并”“发布这个版本”等端到端结果时,必需的提交、推送、创建/更新 MR、平台合并和既定发布步骤默认已授权;不要求等 MR 创建后再说一次“合并”。授权限于当前目标,持续至完成、撤回或范围变更;普通补充消息和范围内修复不撤销目标授权。agent 先展示已核对的范围、源→目标、MR 链接、head SHA、CI/验证状态和执行顺序;展示是执行义务,不新增审批。只要求单项、准备或待审时不得扩展成发布。单个“合并”仍指当前唯一 MR;显式“批量合并 N”仍只覆盖已展示计划内至多 N 次合并(该计数授权 4 小时有效,用户新消息清除剩余额度)。目标不明、混入无关变更或额外高风险动作时,只暂停对应动作并确认。
149
149
  2. **变化先核验**:目标/批量授权内由 agent 完成的修复、新提交或新建 MR,先刷新检查、评审与状态,不重复请求权限。单个对象授权后 head 改变、混入第三方或目标外内容、或多个 MR 指向不明时再确认。CI 从运行中变为通过本身不是权限失效;失败和冲突先诊断修复,不能绕过门禁。
150
- **宿主机械放行阀**:Claude Code 的 `hooks/merge-authorization-prompt.sh` 只识别单独“合并/merge”或“批量合并 N”等锚定指令,`hooks/guard-merge-authorization.sh` 消费一次/计数额度;它不能从自然语言目标推导授权,目标授权也不会自动生成哨兵。若真实宿主拒绝且没有已获授权的正常审批路径,说明宿主限制并请求最小放行,不得自行写哨兵、关闸或换工具绕过。直推默认分支、auto-merge/排队/`--admin` 及一条命令内多个合并仍不放行;其他宿主按实际权限机制和上述目标边界执行。
150
+ **宿主机械放行阀**:平台合并前读 `references/hook-authorization.md`,核对锚定指令、仓库/编号、额度期限及暂停/撤销。它不推导发布目标或未来 PR 归属;机械额度缺失、暂停或过期不等于目标授权不存在。若真实宿主拒绝且没有已获授权的正常审批路径,说明宿主限制并请求最小放行,不得自行写哨兵、关闸或换工具绕过。直推默认分支、auto-merge/排队/`--admin` 及一条命令内多个合并仍不放行;其他宿主按实际权限机制和上述目标边界执行。
151
151
  3. **执行建议(agent 防呆,不增加用户负担)**:获授权后的执行一次性立即合并、不转 auto-merge/排队;显式点名目标 MR/PR(glab/gh 缺省都解析"当前分支",同分支多 MR/PR 时会合错对象);建议把自己已知的 head SHA 作为守卫传给命令:`glab mr merge <iid> --sha <head SHA> --auto-merge=false --yes` / `gh pr merge <PR号|URL> --merge --match-head-commit <head SHA>`(合并策略显式给 `--merge`/`--squash`/`--rebase`,缺省会进交互)。守卫被平台拒绝时重新读取目标并按第 2 条核验授权范围。**一次性合并授权按「命令被放行」消耗,不按「合并成功」消耗**:命令因你自己的参数错误而失败(自造不存在的 flag、SHA 用前缀而非平台现读的完整值、点错 MR 号)同样烧掉这次授权,该机械额度需重新放行;没有此宿主限制的目标授权不因参数错误失效,确认前次未合并后修正重试。所以执行前把 flag 与取值当成不可凭记忆的东西核一遍——**flag 拼写以本机该 CLI 的 `--help` 为准**(同名工具跨版本/跨平台差异很大,"我记得有这个 flag" 是最常见的烧授权方式),**SHA 一律从平台 API 现读完整值**(前缀补全会被守卫拒成 409)。已实测两次:一次前缀补全 409,一次自造 `--merge`(该版本 glab 无此 flag,合并策略缺省即 merge commit)——守卫两次都按设计挡住了错误合并,代价都是让用户重新授权一次。
152
152
  4. **仓库策略例外**:仓库强制 merge queue / auto-merge、或只能直推默认分支时,停下把该仓的合并语义摆给用户裁决,不得套用立即合并流程近似执行。
153
153
  5. **合并后自查**:合并后核对实际合入内容与本次交付预期一致,发现超出如实报告用户裁决(回滚/接受),不得静默带过。
@@ -0,0 +1,13 @@
1
+ # Hook 合并授权
2
+
3
+ 以下执行额度规则受 [合并执行协议](../SKILL.md) 的目标授权、验证和安全边界约束。`hooks/merge-authorization-prompt.sh` 从用户指令生成额度,`hooks/guard-merge-authorization.sh` 在平台合并命令放行时消费额度。
4
+
5
+ ## 指令与有效期
6
+
7
+ 支持原有单独“合并/merge”和“批量合并 N”,另支持完整单行“完成并合并 PR #123 / MR !123”(英文 `finish and merge PR #123`)。原有单次/计数额度仍被任何新消息清除。
8
+
9
+ 新形式只绑定当前 `origin` 仓库和指定编号,原始 60 分钟内消费一次;单独“继续/继续吧/进度/状态/continue/status/progress”保留原额度和到期时间,“停止/停一下/不要合并/撤销合并授权/stop/pause/cancel merge”撤销,其他消息暂停机械额度,之后“继续”不能恢复。
10
+
11
+ ## 执行命令
12
+
13
+ 新形式的执行命令必须是单条直接 `gh pr merge` 或 `glab mr merge`,显式编号;gh 使用 `--repo host/owner/repo` 并指定策略,glab 使用 `--repo https://host/namespace/repo` 并指定 `--auto-merge=false --yes`,可附完整 head SHA,其他参数和 API 形式保持未核验。