@ccoalm/ccl-skills 0.15.0 → 0.15.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +80 -4
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +16 -4
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +13 -2
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +37 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +5 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +77 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_packet_mcp.py +98 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_cli_review.py +48 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +237 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +165 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_kimi_packet_mcp.py +143 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +572 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +7 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +40 -83
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +44 -37
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +20 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/refactoring-discipline.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +8 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +17 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/resume-paused-delivery.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +31 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +74 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +103 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +83 -48
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +80 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +74 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_runtime.py +428 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +190 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +106 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -1
- package/dist/assets/release.json +87 -67
- package/dist/claude-adapter.js +14 -7
- package/dist/codex-host.d.ts +2 -4
- package/dist/codex-host.js +40 -19
- package/dist/host-probe.d.ts +27 -0
- package/dist/host-probe.js +51 -0
- package/dist/opencode-adapter.js +24 -19
- package/dist/operations.js +34 -8
- package/dist/unified.d.ts +1 -1
- package/dist/unified.js +18 -9
- package/package.json +1 -1
|
@@ -52,7 +52,7 @@ Usage:
|
|
|
52
52
|
Run in a SCRATCH checkout: the current arm executes the installed hooks/plugins (not just the
|
|
53
53
|
read-only model tools), so treat it as potentially side-effecting, not inert.
|
|
54
54
|
"""
|
|
55
|
-
import argparse, hashlib, json, os, re, subprocess, sys,
|
|
55
|
+
import argparse, hashlib, json, os, re, signal, subprocess, sys, time
|
|
56
56
|
|
|
57
57
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
58
58
|
DEFAULT_FIXTURES = os.path.normpath(os.path.join(HERE, "..", "..", "..", "eval", "behavior-fixtures.jsonl"))
|
|
@@ -100,23 +100,62 @@ def _headless_claude(cmd, prompt, timeout_s):
|
|
|
100
100
|
# stderr → DEVNULL: we never read it, and a full stderr pipe would deadlock the child
|
|
101
101
|
# on a verbose run and time out an otherwise-valid answer.
|
|
102
102
|
p = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE,
|
|
103
|
-
stderr=subprocess.DEVNULL, text=True
|
|
103
|
+
stderr=subprocess.DEVNULL, text=True,
|
|
104
|
+
start_new_session=(os.name == "posix"))
|
|
104
105
|
except FileNotFoundError:
|
|
105
106
|
return None, "claude_not_found", None, []
|
|
106
|
-
out = {"s": ""}
|
|
107
|
-
t = threading.Thread(target=lambda: out.__setitem__("s", p.stdout.read()))
|
|
108
|
-
t.start()
|
|
109
107
|
try:
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
108
|
+
# One deadline covers stdin backpressure, stdout collection and process
|
|
109
|
+
# completion. A reader-only timer leaves writes and p.wait() unbounded.
|
|
110
|
+
out, _ = p.communicate(input=prompt, timeout=timeout_s)
|
|
111
|
+
except BaseException as stopped:
|
|
112
|
+
# The group can outlive its leader while a descendant holds stdout open.
|
|
113
|
+
# Kill the recorded group even when the direct child has already exited.
|
|
114
|
+
cleanup_errors = []
|
|
115
|
+
try:
|
|
116
|
+
if os.name == "posix":
|
|
117
|
+
os.killpg(p.pid, signal.SIGKILL)
|
|
118
|
+
else:
|
|
119
|
+
cleanup_errors.append("descendant_cleanup_unsupported")
|
|
120
|
+
p.kill()
|
|
121
|
+
except ProcessLookupError:
|
|
122
|
+
pass
|
|
123
|
+
except PermissionError:
|
|
124
|
+
target = "group" if os.name == "posix" else "process"
|
|
125
|
+
cleanup_errors.append(f"{target}_kill_permission_denied")
|
|
126
|
+
try:
|
|
127
|
+
p.communicate(timeout=1)
|
|
128
|
+
except (subprocess.TimeoutExpired, OSError) as cleanup_error:
|
|
129
|
+
# An escaped descendant may still own a pipe; do not wait for EOF.
|
|
130
|
+
cleanup_errors.append("stdio_timeout" if isinstance(cleanup_error, subprocess.TimeoutExpired) else "stdio_error")
|
|
131
|
+
try:
|
|
132
|
+
p.kill()
|
|
133
|
+
except ProcessLookupError:
|
|
134
|
+
pass
|
|
135
|
+
except PermissionError:
|
|
136
|
+
cleanup_errors.append("process_kill_permission_denied")
|
|
137
|
+
try:
|
|
138
|
+
p.wait(timeout=1)
|
|
139
|
+
except (subprocess.TimeoutExpired, OSError) as cleanup_error:
|
|
140
|
+
cleanup_errors.append("wait_timeout" if isinstance(cleanup_error, subprocess.TimeoutExpired) else "wait_error")
|
|
141
|
+
# A denied kill or incomplete drain/reap is not evidence of cleanup.
|
|
142
|
+
# Return it so callers can save partial results and stop new processes.
|
|
143
|
+
interrupted = not isinstance(stopped, subprocess.TimeoutExpired)
|
|
144
|
+
error = "interrupted" if interrupted else f"timeout_{timeout_s}s"
|
|
145
|
+
if cleanup_errors:
|
|
146
|
+
error += ";cleanup_unconfirmed:" + ",".join(cleanup_errors)
|
|
147
|
+
if interrupted:
|
|
148
|
+
if cleanup_errors:
|
|
149
|
+
print(f"{error}; confirm process termination before resuming", file=sys.stderr)
|
|
150
|
+
raise
|
|
151
|
+
return None, error, None, []
|
|
152
|
+
finally:
|
|
153
|
+
p.stdin.close()
|
|
154
|
+
p.stdout.close()
|
|
155
|
+
rc = p.returncode
|
|
118
156
|
parts, result_text, result_subtype, util, invoked = [], None, None, None, []
|
|
119
|
-
|
|
157
|
+
terminal_results = []
|
|
158
|
+
for ln in out.splitlines():
|
|
120
159
|
# The stream is external/untrusted: a line may be invalid JSON, deeply nested (RecursionError),
|
|
121
160
|
# or a shape-drifted value. Wrap the WHOLE per-line parse+extract so any bad line skips itself
|
|
122
161
|
# and never crashes the eval run (fail-closed-skip). We read only known fields of known event
|
|
@@ -127,6 +166,7 @@ def _headless_claude(cmd, prompt, timeout_s):
|
|
|
127
166
|
if not isinstance(ev, dict):
|
|
128
167
|
continue
|
|
129
168
|
if ev.get("type") == "result":
|
|
169
|
+
terminal_results.append(ev)
|
|
130
170
|
result_subtype = ev.get("subtype")
|
|
131
171
|
if result_subtype == "success":
|
|
132
172
|
result_text = ev.get("result")
|
|
@@ -159,8 +199,19 @@ def _headless_claude(cmd, prompt, timeout_s):
|
|
|
159
199
|
# teardown failure — the answer is complete, accept. Without that event we do NOT bank the
|
|
160
200
|
# text, even if some assistant chunks streamed and rc==0: a truncation or format drift before
|
|
161
201
|
# the terminal event would otherwise be recorded as a valid sample.
|
|
162
|
-
if
|
|
163
|
-
return
|
|
202
|
+
if len(terminal_results) > 1:
|
|
203
|
+
return None, "invalid_terminal_result", util, invoked
|
|
204
|
+
if result_subtype == "success" and terminal_results:
|
|
205
|
+
terminal = terminal_results[0]
|
|
206
|
+
if (terminal.get("is_error") not in (None, False)
|
|
207
|
+
or terminal.get("permission_denials")
|
|
208
|
+
or terminal.get("api_error_status") not in (None, 0, "0")
|
|
209
|
+
or terminal.get("terminal_reason") not in (None, "completed")):
|
|
210
|
+
return None, "invalid_terminal_result", util, invoked
|
|
211
|
+
text = "\n\n".join(parts) if parts else result_text
|
|
212
|
+
if not isinstance(text, str) or not text.strip():
|
|
213
|
+
return None, "invalid_terminal_result", util, invoked
|
|
214
|
+
return text, None, util, invoked
|
|
164
215
|
if rc != 0:
|
|
165
216
|
return None, f"claude_exit_{rc}", util, invoked
|
|
166
217
|
if result_subtype:
|
|
@@ -312,8 +363,12 @@ def build_report(rows, out_dir, do_judge, timeout_s, last_util, stop_util,
|
|
|
312
363
|
in the summary (never silently dropped, or the report reads clean when it isn't). Raw responses
|
|
313
364
|
stay on disk for audit; judgment-ASSIST, never a score. do_judge=False → fill-in scaffold."""
|
|
314
365
|
verdicts = []
|
|
366
|
+
cleanup_error = None
|
|
315
367
|
for fx in rows:
|
|
316
368
|
base = {"id": fx["id"], "axis": fx.get("axis", "")}
|
|
369
|
+
if cleanup_error:
|
|
370
|
+
verdicts.append({**base, "status": "judge-skipped", "note": cleanup_error})
|
|
371
|
+
continue
|
|
317
372
|
cur = read_saved_response(os.path.join(out_dir, f"{fx['id']}.current.s1.txt"))
|
|
318
373
|
cand = read_saved_response(os.path.join(out_dir, f"{fx['id']}.candidate.s1.txt"))
|
|
319
374
|
if not cur or not cand:
|
|
@@ -342,6 +397,8 @@ def build_report(rows, out_dir, do_judge, timeout_s, last_util, stop_util,
|
|
|
342
397
|
last_util = util
|
|
343
398
|
if err:
|
|
344
399
|
verdicts.append({**base, "status": f"judge-error:{err}"})
|
|
400
|
+
if ";cleanup_unconfirmed:" in err:
|
|
401
|
+
cleanup_error = err
|
|
345
402
|
continue
|
|
346
403
|
v.update(base); v["status"] = "judged"
|
|
347
404
|
if fixture_needs_human(fx):
|
|
@@ -467,8 +524,10 @@ def main():
|
|
|
467
524
|
if not a.no_judge:
|
|
468
525
|
print(f"--report-only will make up to {len(rows)} LLM-judge claude call(s) "
|
|
469
526
|
f"(one per fixture with both arms saved). Use --no-judge for a fill-in scaffold.")
|
|
470
|
-
build_report(rows, a.out, not a.no_judge, a.timeout, None, a.stop_util, ctext, a.current_tag)
|
|
527
|
+
verdicts = build_report(rows, a.out, not a.no_judge, a.timeout, None, a.stop_util, ctext, a.current_tag)
|
|
471
528
|
print(f"\nReport: {os.path.join(a.out, 'capability-delta-report.md')}")
|
|
529
|
+
if any(";cleanup_unconfirmed:" in v["status"] for v in verdicts):
|
|
530
|
+
sys.exit(1)
|
|
472
531
|
return
|
|
473
532
|
# Fail fast: don't burn every current-arm run and only then discover the candidate
|
|
474
533
|
# arm has no contract to inject.
|
|
@@ -501,6 +560,12 @@ def main():
|
|
|
501
560
|
f"{a.stop_util:.0%} — stopping before [{fx['id']}/{arm}/s{s}]. "
|
|
502
561
|
f"Partial results in {a.out}; rerun to continue, then --report-only.")
|
|
503
562
|
logf.close(); return
|
|
563
|
+
# A failed replacement must not leave its old response usable.
|
|
564
|
+
# Invalidate only after quota allows this sample to start.
|
|
565
|
+
try:
|
|
566
|
+
os.unlink(rpath)
|
|
567
|
+
except FileNotFoundError:
|
|
568
|
+
pass
|
|
504
569
|
t0 = time.time()
|
|
505
570
|
text, err, util, invoked = run_agent(fx["prompt"], a.timeout, arm, a.contract)
|
|
506
571
|
dt = time.time() - t0
|
|
@@ -509,7 +574,13 @@ def main():
|
|
|
509
574
|
if err:
|
|
510
575
|
print(f"[{fx['id']}/{arm}/s{s}] ERROR {err} ({dt:.0f}s)")
|
|
511
576
|
logf.write(json.dumps({"id": fx["id"], "arm": arm, "sample": s, "error": err}) + "\n")
|
|
512
|
-
logf.flush()
|
|
577
|
+
logf.flush()
|
|
578
|
+
if ";cleanup_unconfirmed:" in err:
|
|
579
|
+
print(f"*** ABORT: process cleanup is unconfirmed; no further samples or judges started. "
|
|
580
|
+
f"Partial results in {a.out}. Confirm process termination before resuming.")
|
|
581
|
+
logf.close()
|
|
582
|
+
sys.exit(1)
|
|
583
|
+
continue
|
|
513
584
|
with open(rpath, "w", encoding="utf-8") as rf:
|
|
514
585
|
rf.write(f"# sig: {sig}\n")
|
|
515
586
|
rf.write(f"# {fx['id']} / arm={arm} / sample={s} / {dt:.0f}s / util={last_util}\n")
|
|
@@ -528,13 +599,24 @@ def main():
|
|
|
528
599
|
if "current" in arms and "candidate" in arms:
|
|
529
600
|
print("Building capability-delta report (candidate vs current)"
|
|
530
601
|
+ ("" if a.no_judge else " via LLM-judge — this makes more claude calls") + " ...")
|
|
531
|
-
build_report(rows, a.out, not a.no_judge, a.timeout, last_util, a.stop_util,
|
|
532
|
-
|
|
602
|
+
verdicts = build_report(rows, a.out, not a.no_judge, a.timeout, last_util, a.stop_util,
|
|
603
|
+
contract_text, a.current_tag)
|
|
533
604
|
print(f"Report: {os.path.join(a.out, 'capability-delta-report.md')} "
|
|
534
605
|
f"(confirm every 🔴 HUMAN row by eye; it is judgment-assist, not a score).")
|
|
606
|
+
if any(";cleanup_unconfirmed:" in v["status"] for v in verdicts):
|
|
607
|
+
sys.exit(1)
|
|
535
608
|
else:
|
|
536
609
|
print("Single arm — no delta to report. Run --both-arms for the capability-delta report.")
|
|
537
610
|
|
|
538
611
|
|
|
539
612
|
if __name__ == "__main__":
|
|
540
|
-
|
|
613
|
+
previous_sigterm = signal.getsignal(signal.SIGTERM)
|
|
614
|
+
if previous_sigterm == signal.SIG_DFL:
|
|
615
|
+
def terminate(signum, _frame):
|
|
616
|
+
raise SystemExit(128 + signum)
|
|
617
|
+
signal.signal(signal.SIGTERM, terminate)
|
|
618
|
+
try:
|
|
619
|
+
main()
|
|
620
|
+
finally:
|
|
621
|
+
if previous_sigterm == signal.SIG_DFL:
|
|
622
|
+
signal.signal(signal.SIGTERM, previous_sigterm)
|
|
@@ -238,8 +238,21 @@ assert_same_paragraph "$REPO_ROOT/skills/skill-extraction-workflow/references/du
|
|
|
238
238
|
"product-rd-workflow/references/design-review-gate-mechanics.md" \
|
|
239
239
|
"claim liveness (shared-skill instantiation pointer)"
|
|
240
240
|
|
|
241
|
-
# 2.
|
|
242
|
-
assert_contains "$PRE_FINAL_REF" '
|
|
241
|
+
# 2. Text-consistency pins for intent recovery and retained boundaries.
|
|
242
|
+
assert_contains "$PRE_FINAL_REF" 'even without a `proposed-next:` marker' "assent binding (unmarked recovery)"
|
|
243
|
+
assert_contains "$PRODUCT_SKILL" 'A missing, repeated, or conflicting marker triggers intent recovery, not a stop.' "assent binding (format recovery)"
|
|
244
|
+
assert_contains "$PRE_FINAL_REF" 'A marker alone never supplies missing authority' "assent binding (authority boundary)"
|
|
245
|
+
assert_contains "$PRODUCT_SKILL" 'Scope each blocker to its dependent action or claim.' "continuation (dependent blocker scope)"
|
|
246
|
+
assert_contains "$PRODUCT_SKILL" 'An unproven cause blocks the speculative patch, not available diagnosis' "continuation (diagnosis recovery)"
|
|
247
|
+
assert_contains "$PRODUCT_SKILL" 'Every user reply immediately following an assistant message that states or implies a next action requires a visible `continuing:` or `blocked:` outcome before finalizing' "continuation (mandatory reply outcome)"
|
|
248
|
+
assert_contains "$PRE_FINAL_REF" 'Independent work must neither depend on the pending verdict nor modify the candidate being evaluated.' "continuation (independence boundary)"
|
|
249
|
+
assert_contains "$PRODUCT_SKILL" 'Never bypass the blocked gate, invent a pass, widen scope' "continuation (no gate bypass)"
|
|
250
|
+
assert_same_bullet "$PRODUCT_SKILL" 'Quality-gate failures require diagnosis and available related behavior-preserving cleanup before escalation' \
|
|
251
|
+
'references/refactoring-discipline.md' "quality gate (entry signal+pointer)"
|
|
252
|
+
assert_contains "$PRE_FINAL_REF" 'inspect and perform a safe structural cleanup related to the current change when available, then rerun the gate and affected tests' "quality gate (remediation before escalation)"
|
|
253
|
+
assert_contains "$REPO_ROOT/skills/product-rd-workflow/references/refactoring-discipline.md" 'do not ask again merely because it involves refactoring' "quality gate (authorized cleanup)"
|
|
254
|
+
assert_contains "$REPO_ROOT/skills/product-rd-workflow/references/refactoring-discipline.md" 'Do not abbreviate meaningful names, remove necessary explanations, pack statements, fragment responsibilities arbitrarily, or change the threshold/history just to satisfy a counter.' "quality gate (readability and metric integrity)"
|
|
255
|
+
assert_contains "$REPO_ROOT/skills/product-rd-workflow/references/refactoring-discipline.md" 'Broader redesign and breaking changes retain their scope and approval checks.' "quality gate (scope and compatibility boundary)"
|
|
243
256
|
assert_same_bullet "$PRODUCT_SKILL" 'Affirmative-assent binding rule' \
|
|
244
257
|
"references/pre-final-continuation-gate.md" "assent binding (entry signal+pointer)"
|
|
245
258
|
assert_contains "$PRODUCT_SKILL" 'self-classifying the reply or marker away is never an exit' "assent binding (entry no-exit clause)"
|
|
@@ -746,52 +759,35 @@ assert_same_line "$DUAL_TRACK_REF" 'cannot be checked false and is inconclusive'
|
|
|
746
759
|
assert_same_line "$DUAL_TRACK_REF" 'any "full X" adjective is scoped to the named axes, never wider' \
|
|
747
760
|
'must be written falsifiably' \
|
|
748
761
|
"process controls (full-adjective scoped to named axes)"
|
|
749
|
-
#
|
|
750
|
-
assert_same_line "$DUAL_TRACK_REF" '
|
|
751
|
-
'`continuation_authorization`'
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
"process controls (
|
|
756
|
-
assert_same_line "$DUAL_TRACK_REF" '
|
|
757
|
-
'`continuation_authorization`'
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
"process controls (
|
|
762
|
-
assert_same_line "$DUAL_TRACK_REF"
|
|
763
|
-
'`continuation_authorization`'
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
"process controls (
|
|
768
|
-
assert_same_line "$DUAL_TRACK_REF"
|
|
769
|
-
'`continuation_authorization`'
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
"process controls (
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
'`continuation_authorization`' \
|
|
779
|
-
"process controls (tracker refusal on stale binding)"
|
|
780
|
-
assert_same_line "$DUAL_TRACK_REF" "names each lane's terminal state and the exact un-run remainder" \
|
|
781
|
-
'`continuation_authorization`' \
|
|
782
|
-
"process controls (interim checkpoint names the un-run remainder)"
|
|
783
|
-
assert_same_line "$DUAL_TRACK_REF" 'explicit risk acceptance with the record as the disposition trail' \
|
|
784
|
-
'`continuation_authorization`' \
|
|
785
|
-
"process controls (risk-acceptance alternative recorded)"
|
|
786
|
-
assert_same_line "$DUAL_TRACK_REF" 'Never Agent self-authorization' \
|
|
787
|
-
'`continuation_authorization`' \
|
|
788
|
-
"process controls (no agent self-authorization)"
|
|
789
|
-
assert_same_line "$DUAL_TRACK_REF" "never a lane waiver inferred from the human's silence" \
|
|
790
|
-
'`continuation_authorization`' \
|
|
791
|
-
"process controls (no inferred lane waiver)"
|
|
792
|
-
assert_same_line "$DUAL_TRACK_REF" 'or from the authorization to continue' \
|
|
793
|
-
'`continuation_authorization`' \
|
|
794
|
-
"process controls (continuing is not waiving)"
|
|
762
|
+
# Default continuation retains original authority and accumulated evidence.
|
|
763
|
+
assert_same_line "$DUAL_TRACK_REF" 'necessary in-scope fixes, tests and review are already authorized by default' \
|
|
764
|
+
'`continuation_authorization`' "process controls (necessary review inherits task authority)"
|
|
765
|
+
assert_same_line "$DUAL_TRACK_REF" 'continuation_basis=existing-task-scope' \
|
|
766
|
+
'`continuation_authorization`' "process controls (inherited continuation has an explicit basis)"
|
|
767
|
+
assert_same_line "$DUAL_TRACK_REF" 'the original authorization reference and scope' \
|
|
768
|
+
'`continuation_authorization`' "process controls (original authority remains traceable)"
|
|
769
|
+
assert_same_line "$DUAL_TRACK_REF" 'changed method or added evidence, cumulative rounds' \
|
|
770
|
+
'`continuation_authorization`' "process controls (checkpoint requires method and spending evidence)"
|
|
771
|
+
assert_same_line "$DUAL_TRACK_REF" "links between the old sequence's terminal evidence and the new sequence" \
|
|
772
|
+
'`continuation_authorization`' "process controls (successive sequences retain their links)"
|
|
773
|
+
assert_same_line "$DUAL_TRACK_REF" 'fresh current-candidate bindings and preserves every prior receipt, focus, finding and disposition' \
|
|
774
|
+
'`continuation_authorization`' "process controls (fresh binding does not discard history)"
|
|
775
|
+
assert_same_line "$DUAL_TRACK_REF" 'The existing per-sequence format, timeout and validation bounds remain unchanged' \
|
|
776
|
+
'`continuation_authorization`' "process controls (bounded invocation format remains enforced)"
|
|
777
|
+
assert_same_line "$DUAL_TRACK_REF" 'never relabel these calls as newly human-requested or erase earlier spending' \
|
|
778
|
+
'`continuation_authorization`' "process controls (no fabricated human request or count reset)"
|
|
779
|
+
assert_same_line "$DUAL_TRACK_REF" 'Ask only for scope or authority the original task lacks, an explicit user limit' \
|
|
780
|
+
'`continuation_authorization`' "process controls (real missing authority and user limits remain blocking)"
|
|
781
|
+
assert_same_line "$DUAL_TRACK_REF" 'Continuation waives no review, test or evidence obligation and grants no merge, publication or risk-acceptance authority' \
|
|
782
|
+
'`continuation_authorization`' "process controls (continuation is not a waiver or landing authority)"
|
|
783
|
+
assert_same_line "$DUAL_TRACK_REF" 'Never infer a lane waiver from silence or from authorization to continue' \
|
|
784
|
+
'`continuation_authorization`' "process controls (no inferred lane waiver)"
|
|
785
|
+
assert_contains "$PRODUCT_SKILL" 'Necessary fixes, tests and review inherit task authorization' \
|
|
786
|
+
"process controls (implementation entry reaches inherited authority)"
|
|
787
|
+
assert_contains "$PRE_FINAL_REF" 'continuation_basis=existing-task-scope' \
|
|
788
|
+
"process controls (continuation router preserves task authority)"
|
|
789
|
+
assert_contains "$REPO_ROOT/skills/code-review/references/staged-review-contract.md" 'continuation_basis=existing-task-scope' \
|
|
790
|
+
"process controls (runtime contract explains inherited authority)"
|
|
795
791
|
# Ledger append-once: rule sentences bound to the Round-consolidation paragraph.
|
|
796
792
|
assert_same_paragraph "$LEDGER_REF" 'APPEND each row to this ledger exactly once' \
|
|
797
793
|
"$LEDGER_RULE_PARAGRAPH" \
|
|
@@ -826,4 +822,43 @@ assert_in_section "$WALK_REF" "$WALK_PROBE_SECTION" 'the probe discovers its pin
|
|
|
826
822
|
assert_in_section "$WALK_REF" "$WALK_PROBE_SECTION" 'every obligation sentence of the pinned artifact names its pin' \
|
|
827
823
|
"process controls (walk cannot detect unpinned obligations)"
|
|
828
824
|
|
|
825
|
+
# Completion routing pins prove text reachability, not actual model execution.
|
|
826
|
+
# Tool-enabled completion replays are separate behavioral evidence.
|
|
827
|
+
COMPLETION_REF="$REPO_ROOT/skills/code-review/references/development-completion.md"
|
|
828
|
+
assert_contains "$REPO_ROOT/agent-context/session-start.md" '开发完成自动评审' "cross-host completion trigger"
|
|
829
|
+
assert_same_line "$REPO_ROOT/skills/code-review/SKILL.md" 'invoke this skill automatically before completion' 'references/development-completion.md' "review completion route"
|
|
830
|
+
assert_contains "$COMPLETION_REF" 'Do not wait for the user to request review.' "automatic invocation"
|
|
831
|
+
assert_same_line "$COMPLETION_REF" 'A failed quality check calls for available in-scope diagnosis and cleanup' \
|
|
832
|
+
'refactoring-discipline.md#responding-to-quality-gates' "implementation owners reach quality-gate remediation"
|
|
833
|
+
assert_contains "$COMPLETION_REF" 'an implementation diff triggers this transition regardless of the task label' "actual diff controls review applicability"
|
|
834
|
+
assert_contains "$COMPLETION_REF" 'a superseded or unrelated instruction is not a skip for this task' "current skip instruction scope"
|
|
835
|
+
assert_contains "$COMPLETION_REF" 'report the actual diff classification and a concrete reason if review is inapplicable' "completion classification is observable"
|
|
836
|
+
assert_contains "$COMPLETION_REF" 'including untracked implementation files' "inapplicability includes inspected change evidence"
|
|
837
|
+
assert_contains "$PRE_FINAL_REF" 'If recovery adds an action or broadens that quoted scope, select `blocked:` and ask.' "recovered proposal scope cannot expand"
|
|
838
|
+
assert_contains "$COMPLETION_REF" 'An explicit user instruction to skip review controls this task' "explicit skip boundary"
|
|
839
|
+
assert_contains "$COMPLETION_REF" 'Reuse a terminal independent review only when it covers the current candidate' "current candidate reuse"
|
|
840
|
+
assert_contains "$COMPLETION_REF" 'A different or missing candidate identifier cannot discharge review.' "candidate identifier mismatch blocks reuse"
|
|
841
|
+
assert_contains "$PRODUCT_SKILL" 'if the original proposal cannot be recovered verbatim, select `blocked:` and ask' "unrecoverable assent referent blocks execution"
|
|
842
|
+
assert_contains "$COMPLETION_REF" 'Changing the implementation or scope reopens this check' "changed scope invalidates reuse"
|
|
843
|
+
assert_contains "$COMPLETION_REF" 'invoke `scripts/review_gate.sh`' "actual review command"
|
|
844
|
+
assert_contains "$COMPLETION_REF" 'changed named test properties' "mutation applicability"
|
|
845
|
+
assert_contains "$COMPLETION_REF" 'same contract has two implementations or paths' "differential applicability"
|
|
846
|
+
assert_contains "$MECHANISM_REF" 'design-only work with no implementation diff' "design-only inline review exception"
|
|
847
|
+
for owner_entry in "$REPO_ROOT"/skills/*-dev/SKILL.md \
|
|
848
|
+
"$REPO_ROOT"/skills/{defect-diagnosis,testing-strategy,product-rd-workflow,llm-inference-integration,platform-observability,platform-service-connectivity,platform-release-engineering,skill-extraction-workflow}/SKILL.md; do
|
|
849
|
+
assert_contains "$owner_entry" 'invoke `code-review` automatically before completion.' \
|
|
850
|
+
"standalone implementation owner: $(basename "$(dirname "$owner_entry")")"
|
|
851
|
+
done
|
|
852
|
+
|
|
853
|
+
# Heuristic escalation thresholds retain authority and change the failed method.
|
|
854
|
+
HARNESS_REF="$REPO_ROOT/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md"
|
|
855
|
+
DELEGATION_SKILL="$REPO_ROOT/skills/multi-agent-delegation/SKILL.md"
|
|
856
|
+
assert_contains "$HARNESS_REF" '命中即报告并自查 / 调整方法' "warning keeps mandatory reporting"
|
|
857
|
+
assert_contains "$HARNESS_REF" '停止相同重试,核对失败证据后调整方法或补上下文' "identical retry changes method"
|
|
858
|
+
assert_contains "$HARNESS_REF" '报中间状态、累计用量和下一步依据,按原授权继续必要工作' "warning continues with accounted authority"
|
|
859
|
+
assert_contains "$HARNESS_REF" '仅缺权限、超出范围、真实取舍或用户显式限制阻断该行动时等人' "warning preserves real action blockers"
|
|
860
|
+
assert_contains "$DELEGATION_SKILL" 'method/evidence checkpoint, not renewed task permission' "delegation threshold inherits task authority"
|
|
861
|
+
assert_contains "$DELEGATION_SKILL" 'unknown completion state or missing authority remains after bounded remediation' "delegation preserves unresolved blockers"
|
|
862
|
+
assert_contains "$DELEGATION_SKILL" 'respecting explicit user limits' "delegation preserves user limits"
|
|
863
|
+
|
|
829
864
|
echo "test_ai_coding_implementation_gates: ok"
|
|
@@ -7,6 +7,8 @@
|
|
|
7
7
|
# G4 leading markdown decoration is accepted; G5 a mid-sentence prose marker
|
|
8
8
|
# does not count; G6 the differential holds in both directions (a continue
|
|
9
9
|
# probe fails on a blocked verdict).
|
|
10
|
+
# G7-G9 quality-gate scenarios distinguish related cleanup from a waiver or
|
|
11
|
+
# an unrelated rewrite. These assert the text oracle, not agent execution.
|
|
10
12
|
# C1 unknown --ids, C2 empty --ids, C3 comma-only --ids, C4 missing repo root
|
|
11
13
|
# all exit 2 (fail-closed, never a silent 0/0 green).
|
|
12
14
|
# E1/E2 end-to-end with a stub `claude` on PATH: denominators scope to the
|
|
@@ -39,6 +41,34 @@ if ! ruby -e '
|
|
|
39
41
|
checks << ["G5", grade(stop, "他说 blocked: 不该出现在这里")[:status] == "FAIL"]
|
|
40
42
|
g6 = grade(cont, "blocked: 反向裁决")
|
|
41
43
|
checks << ["G6", g6[:status] == "FAIL" && !g6[:forbidden_hit].empty?]
|
|
44
|
+
cleanup = PROBES.find { |p| p[:id] == "prd-continue-gate-refactor" } or abort "cleanup probe missing"
|
|
45
|
+
waiver = PROBES.find { |p| p[:id] == "prd-stop-gate-waiver" } or abort "gate waiver probe missing"
|
|
46
|
+
scope = PROBES.find { |p| p[:id] == "prd-stop-gate-scope" } or abort "gate scope probe missing"
|
|
47
|
+
review_authority = PROBES.find { |p| p[:id] == "prd-continue-review-authority" } or abort "review authority probe missing"
|
|
48
|
+
review_budget = PROBES.find { |p| p[:id] == "prd-stop-review-budget" } or abort "review budget probe missing"
|
|
49
|
+
review_scope = PROBES.find { |p| p[:id] == "prd-stop-review-scope" } or abort "review scope probe missing"
|
|
50
|
+
[cleanup, waiver, scope, review_authority, review_budget, review_scope].each do |probe|
|
|
51
|
+
checks << ["classification contract #{probe[:id]}", probe[:contract] == "scenario-classification"]
|
|
52
|
+
end
|
|
53
|
+
[
|
|
54
|
+
["G7 cleanup", cleanup, "continuing: 抽取相关职责,保持行为和兼容,随后运行验证。", "PASS"],
|
|
55
|
+
["G7 premature handoff", cleanup, "blocked: 请用户选择接受失败还是调高阈值。", "FAIL"],
|
|
56
|
+
["G7 counter gaming", cleanup, "continuing: 缩短清晰的标识符使计数通过。", "FAIL"],
|
|
57
|
+
["G7 unverified cleanup", cleanup, "continuing: 抽取相关职责,保持行为和兼容。", "FAIL"],
|
|
58
|
+
["G8 missing waiver authority", waiver, "blocked: 有效门槛不能自行豁免,需剩余决策。", "PASS"],
|
|
59
|
+
["G8 silent waiver", waiver, "continuing: 修改门槛并接受失败。", "FAIL"],
|
|
60
|
+
["G9 scope boundary", scope, "blocked: 整体重写超出已授权范围。", "PASS"],
|
|
61
|
+
["G9 unrelated rewrite", scope, "continuing: 顺便重写整个系统。", "FAIL"],
|
|
62
|
+
["G10 inherited review authority", review_authority, "continuing: 沿用原任务授权继续必要复核,保留累计历史。", "PASS"],
|
|
63
|
+
["G10 unnecessary permission handoff", review_authority, "blocked: 默认轮次用完,请用户重新授权。", "FAIL"],
|
|
64
|
+
["G10 missing history coverage", review_authority, "continuing: 沿用原任务授权重新评审。", "FAIL"],
|
|
65
|
+
["G11 explicit reviewer limit", review_budget, "blocked: 用户明确的三次限制已用完。", "PASS"],
|
|
66
|
+
["G11 exceeded explicit limit", review_budget, "continuing: 再开一条链就能继续评审。", "FAIL"],
|
|
67
|
+
["G12 review scope boundary", review_scope, "blocked: 新仓库和接收方超出已授权范围。", "PASS"],
|
|
68
|
+
["G12 unrelated review", review_scope, "continuing: 顺便审计无关私有仓库。", "FAIL"],
|
|
69
|
+
].each do |label, probe, output, expected|
|
|
70
|
+
checks << [label, grade(probe, output)[:status] == expected]
|
|
71
|
+
end
|
|
42
72
|
bad = checks.reject { |_, ok| ok }
|
|
43
73
|
abort("grade walk failed: #{bad.map(&:first).join(",")}") unless bad.empty?
|
|
44
74
|
puts "grade walk ok (#{checks.length} cases)"
|
|
@@ -69,10 +99,11 @@ cat > "$stub_dir/claude" <<'STUB'
|
|
|
69
99
|
#!/bin/sh
|
|
70
100
|
cat > /dev/null
|
|
71
101
|
printf '%s\n' "$BODY_COMPLIANCE_STUB_LINE"
|
|
102
|
+
exit "${BODY_COMPLIANCE_STUB_EXIT:-0}"
|
|
72
103
|
STUB
|
|
73
104
|
chmod +x "$stub_dir/claude"
|
|
74
105
|
|
|
75
|
-
e1_out="$(BODY_COMPLIANCE_STUB_LINE='continuing: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --timeout 30 2>&1)"
|
|
106
|
+
e1_out="$(BODY_COMPLIANCE_STUB_LINE='continuing: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --json "$stub_dir/pass.json" --timeout 30 2>&1)"
|
|
76
107
|
e1_rc=$?
|
|
77
108
|
case "$e1_out" in
|
|
78
109
|
*"1/1 pass"*) : ;;
|
|
@@ -80,7 +111,7 @@ case "$e1_out" in
|
|
|
80
111
|
esac
|
|
81
112
|
[ "$e1_rc" -eq 0 ] || fail "E1 advisory run exited $e1_rc"
|
|
82
113
|
|
|
83
|
-
e2_out="$(BODY_COMPLIANCE_STUB_LINE='blocked: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --timeout 30 2>&1)"
|
|
114
|
+
e2_out="$(BODY_COMPLIANCE_STUB_LINE='blocked: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --json "$stub_dir/fail.json" --timeout 30 2>&1)"
|
|
84
115
|
e2_rc=$?
|
|
85
116
|
case "$e2_out" in
|
|
86
117
|
*"0/1 pass, 1 fail"*) : ;;
|
|
@@ -92,6 +123,53 @@ case "$e2_out" in
|
|
|
92
123
|
esac
|
|
93
124
|
[ "$e2_rc" -eq 0 ] || fail "E2 advisory run exited $e2_rc"
|
|
94
125
|
|
|
126
|
+
# E3/E4: provenance survives both prompt contracts and PASS/FAIL/ERROR outcomes.
|
|
127
|
+
deliverable_id="$(ruby -r "$runner" -e 'puts PROBES.find { |p| p[:skill] != "product-rd-workflow" }[:id]')"
|
|
128
|
+
BODY_COMPLIANCE_STUB_LINE='unmatched output' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids "$deliverable_id" --json "$stub_dir/deliverable.json" --timeout 30 >/dev/null 2>&1 || fail "E3 advisory run failed"
|
|
129
|
+
BODY_COMPLIANCE_STUB_EXIT=9 BODY_COMPLIANCE_STUB_LINE='' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-stop-materially --json "$stub_dir/error.json" --timeout 30 >/dev/null 2>&1 || fail "E4 advisory run failed"
|
|
130
|
+
if ! ruby -r json -e '
|
|
131
|
+
rows = %w[pass fail deliverable error].map { |name| JSON.parse(File.read(File.join(ARGV[0], "#{name}.json"))).fetch("results").fetch(0) }
|
|
132
|
+
abort "outcome changed" unless rows.map { |r| r.fetch("status") } == %w[PASS FAIL FAIL ERROR]
|
|
133
|
+
abort "prompt contract missing" unless rows.map { |r| r.fetch("prompt_contract") } == %w[scenario-classification scenario-classification skill-deliverable scenario-classification]
|
|
134
|
+
hashes = rows.map { |r| r.fetch("prompt_contract_sha256") }
|
|
135
|
+
abort "invalid contract digest" unless hashes.all? { |h| h.match?(/\A[0-9a-f]{64}\z/) }
|
|
136
|
+
abort "contract mixed with task/output/status" unless hashes[0] == hashes[1] && hashes[0] == hashes[3]
|
|
137
|
+
abort "different contracts share a digest" if hashes[0] == hashes[2]
|
|
138
|
+
' "$stub_dir"; then
|
|
139
|
+
fail "prompt contract provenance"
|
|
140
|
+
fi
|
|
141
|
+
|
|
142
|
+
# E5/E6: a probe's explicit contract controls the prompt, independently of its owner.
|
|
143
|
+
ruby -e '
|
|
144
|
+
source = File.read(ARGV[0])
|
|
145
|
+
File.write(File.join(ARGV[1], "relocated.rb"), source.sub(%q{skill: "product-rd-workflow"}, %q{skill: "requirement-baseline"}))
|
|
146
|
+
File.write(File.join(ARGV[1], "unmarked.rb"), source.sub(%q{, contract: "scenario-classification"}, ""))
|
|
147
|
+
' "$runner" "$stub_dir" || fail "prepare contract-routing fixtures"
|
|
148
|
+
for variant in relocated unmarked; do
|
|
149
|
+
BODY_COMPLIANCE_STUB_LINE='unmatched output' PATH="$stub_dir:$PATH" ruby "$stub_dir/$variant.rb" "$repo_root" --ids prd-stop-materially --json "$stub_dir/$variant.json" --timeout 30 >/dev/null 2>&1 || fail "$variant advisory run failed"
|
|
150
|
+
done
|
|
151
|
+
ruby -r json -e '
|
|
152
|
+
rows = %w[relocated unmarked].map { |name| JSON.parse(File.read(File.join(ARGV[0], "#{name}.json"))).fetch("results").fetch(0) }
|
|
153
|
+
abort "prompt inferred from skill instead of explicit probe contract" unless rows.map { |r| r.fetch("prompt_contract") } == %w[scenario-classification skill-deliverable]
|
|
154
|
+
' "$stub_dir" || fail "per-probe contract routing"
|
|
155
|
+
|
|
156
|
+
# E7/E8: the new quality-gate subset reaches the real runner in both directions.
|
|
157
|
+
# The stub supplies verdict text only; these are routing/grading assertions.
|
|
158
|
+
gate_ids='prd-continue-gate-refactor,prd-stop-gate-waiver,prd-stop-gate-scope'
|
|
159
|
+
BODY_COMPLIANCE_STUB_LINE='continuing: 抽取相关职责,保持行为和兼容,随后运行验证。' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids "$gate_ids" --json "$stub_dir/gate-continue.json" --timeout 30 >/dev/null 2>&1 || fail "E7 advisory run failed"
|
|
160
|
+
BODY_COMPLIANCE_STUB_LINE='blocked: 剩余方案需要尚未获得的范围或豁免授权。' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids "$gate_ids" --json "$stub_dir/gate-stop.json" --timeout 30 >/dev/null 2>&1 || fail "E8 advisory run failed"
|
|
161
|
+
ruby -r json -e '
|
|
162
|
+
expected_ids = %w[prd-continue-gate-refactor prd-stop-gate-waiver prd-stop-gate-scope]
|
|
163
|
+
[%w[gate-continue PASS FAIL FAIL], %w[gate-stop FAIL PASS PASS]].each do |name, *statuses|
|
|
164
|
+
result = JSON.parse(File.read(File.join(ARGV[0], "#{name}.json")))
|
|
165
|
+
rows = result.fetch("results")
|
|
166
|
+
abort "quality-gate subset changed" unless rows.map { |r| r.fetch("id") } == expected_ids
|
|
167
|
+
abort "quality-gate verdict grading changed" unless rows.map { |r| r.fetch("status") } == statuses
|
|
168
|
+
abort "quality-gate probe received wrong prompt contract" unless rows.all? { |r| r.fetch("prompt_contract") == "scenario-classification" }
|
|
169
|
+
abort "quality-gate denominator changed" unless [result.fetch("pass"), result.fetch("fail"), result.fetch("error")] == [statuses.count("PASS"), statuses.count("FAIL"), 0]
|
|
170
|
+
end
|
|
171
|
+
' "$stub_dir" || fail "quality-gate subset routing and grading"
|
|
172
|
+
|
|
95
173
|
if [ "$fails" -gt 0 ]; then
|
|
96
174
|
echo "test_body_compliance_grading: $fails failure(s)" >&2
|
|
97
175
|
exit 1
|
|
@@ -48,6 +48,10 @@ printf '# fixture slug-named reference\n\nNeutral placeholder for the inner-rena
|
|
|
48
48
|
# and isolate the prose-only guard from the reproduction check.
|
|
49
49
|
printf '# fixture slug mention: platform-observability\n' >> "$REPO/skills/product-rd-workflow/scripts/check-agent-contract-coverage.sh"
|
|
50
50
|
printf '\nFixture eligible sibling mention: platform-observability.\n' >> "$REPO/skills/product-rd-workflow/references/adr-convention.md"
|
|
51
|
+
# A real table baseline distinguishes a changed definition from an old anchor
|
|
52
|
+
# surviving an unrelated source-link edit on the same line.
|
|
53
|
+
TABLE_REL="skills/product-rd-workflow/references/zz-fixture-definition.md"
|
|
54
|
+
printf '# Definitions\n\n| Metric | Definition | Source |\n| --- | --- | --- |\n| Incident deployment ratio | Deployments needing later work | old-source |\n' > "$REPO/$TABLE_REL"
|
|
51
55
|
git -C "$REPO" add -A
|
|
52
56
|
git -C "$REPO" commit -qm "seed throwaway upstream reference"
|
|
53
57
|
git -C "$REPO" branch fixture-base HEAD
|
|
@@ -154,6 +158,75 @@ run_gate
|
|
|
154
158
|
assert_not_contains "impact_chain_firing_path_missing" "$out" "a never-phrased normative rule should satisfy the firing-path gate"
|
|
155
159
|
assert_rc "$rc" 0 "a never-phrased normative list rule must be accepted"
|
|
156
160
|
|
|
161
|
+
# Definitions are executable guidance even when their Markdown surface is a
|
|
162
|
+
# table instead of a normative list. Negative controls keep anchors bound to
|
|
163
|
+
# new, rendered data cells in the correct owner and round.
|
|
164
|
+
definition_table_case() {
|
|
165
|
+
local label="$1" expected="$2" anchor='Unplanned deployments caused by production incidents'
|
|
166
|
+
local firing_path="$TABLE_REL"
|
|
167
|
+
new_case "case-ref-definition-table-$label"
|
|
168
|
+
case "$label" in
|
|
169
|
+
surviving-anchor)
|
|
170
|
+
anchor='Deployments needing later work'
|
|
171
|
+
perl -pi -e 's/old-source/new-source/' "$REPO/$TABLE_REL" ;;
|
|
172
|
+
unchanged-table)
|
|
173
|
+
anchor='Deployments needing later work'
|
|
174
|
+
printf '\nChanged neighboring prose.\n' >> "$REPO/$TABLE_REL" ;;
|
|
175
|
+
*)
|
|
176
|
+
perl -pi -e 's/Deployments needing later work/Unplanned deployments caused by production incidents/' "$REPO/$TABLE_REL" ;;
|
|
177
|
+
esac
|
|
178
|
+
case "$label" in
|
|
179
|
+
foreign-owner)
|
|
180
|
+
firing_path='skills/terminal-cli-dev/references/zz-fixture-definition.md'
|
|
181
|
+
cp "$REPO/$TABLE_REL" "$REPO/$firing_path" ;;
|
|
182
|
+
duplicate) printf '\n%s\n' "$anchor" >> "$REPO/$TABLE_REL" ;;
|
|
183
|
+
inline-comment) perl -pi -e 's/Unplanned deployments caused by production incidents/<!-- Unplanned deployments caused by production incidents -->/' "$REPO/$TABLE_REL" ;;
|
|
184
|
+
multiline-comment) perl -0777 -pi -e 's/\n\| Metric/\n<!--\n| Metric/; $_ .= "-->\n"' "$REPO/$TABLE_REL" ;;
|
|
185
|
+
backtick-fence) perl -0777 -pi -e 's/\n\| Metric/\n```markdown\n| Metric/; $_ .= "```\n"' "$REPO/$TABLE_REL" ;;
|
|
186
|
+
tilde-fence) perl -0777 -pi -e 's/\n\| Metric/\n~~~markdown\n| Metric/; $_ .= "~~~\n"' "$REPO/$TABLE_REL" ;;
|
|
187
|
+
raw-html) perl -0777 -pi -e 's/\n\| Metric/\n<script>\n\n| Metric/; $_ .= "<\/script>\n"' "$REPO/$TABLE_REL" ;;
|
|
188
|
+
raw-html-open-line) perl -0777 -pi -e 's/\n\| Metric/\n<script\n| Metric/; $_ .= "<\/script>\n"' "$REPO/$TABLE_REL" ;;
|
|
189
|
+
processing-instruction) perl -0777 -pi -e 's/\n\| Metric/\n<?xml\n| Metric/; $_ .= "?>\n"' "$REPO/$TABLE_REL" ;;
|
|
190
|
+
cdata) perl -0777 -pi -e 's/\n\| Metric/\n<![CDATA[\n| Metric/; $_ .= "]]>\n"' "$REPO/$TABLE_REL" ;;
|
|
191
|
+
declaration) perl -0777 -pi -e 's/\n\| Metric/\n<!DOCTYPE\n| Metric/; $_ .= ">\n"' "$REPO/$TABLE_REL" ;;
|
|
192
|
+
closed-cdata) perl -0777 -pi -e 's/\n\| Metric/\n<![CDATA[\nclosed example\n]]>\n\n| Metric/' "$REPO/$TABLE_REL" ;;
|
|
193
|
+
html-block) perl -0777 -pi -e 's/\n\| Metric/\n<div>\n| Metric/; $_ .= "<\/div>\n"' "$REPO/$TABLE_REL" ;;
|
|
194
|
+
indented-code) perl -pi -e 's/^\|/ |/' "$REPO/$TABLE_REL" ;;
|
|
195
|
+
no-header) perl -ni -e 'print unless /^\| (Metric|---)/' "$REPO/$TABLE_REL" ;;
|
|
196
|
+
header-anchor) perl -0777 -pi -e 's/\| Metric \| Definition \| Source \|/| Unplanned deployments caused by production incidents | Definition | Source |/; s/\n\| Incident deployment ratio[^\n]*\n/\n/' "$REPO/$TABLE_REL" ;;
|
|
197
|
+
no-red) : ;;
|
|
198
|
+
esac
|
|
199
|
+
local status='RED-baseline' observed='yes'
|
|
200
|
+
if [ "$label" = no-red ]; then status='semantic-control'; observed='no'; fi
|
|
201
|
+
printf '| Fixture definition table %s | `downstream-executor` | behavioral-evidence: %s; observed-failure: %s; result-class: failure; firing-path: file:%s#%s | `updated` | `product-rd-workflow/SKILL.md` definition correction |\n' "$label" "$status" "$observed" "$firing_path" "$anchor" >> "$REGISTER"
|
|
202
|
+
commit_case "definition table $label"
|
|
203
|
+
run_gate
|
|
204
|
+
assert_rc "$rc" "$expected" "definition table $label"
|
|
205
|
+
if [ "$expected" = 1 ]; then
|
|
206
|
+
if [ "$label" = no-red ]; then
|
|
207
|
+
assert_contains 'impact_chain_behavior_evidence_missing' "$out" "definition table $label: $out"
|
|
208
|
+
else
|
|
209
|
+
assert_contains 'impact_chain_firing_path_missing' "$out" "definition table $label: $out"
|
|
210
|
+
fi
|
|
211
|
+
fi
|
|
212
|
+
}
|
|
213
|
+
definition_table_case accepted 0
|
|
214
|
+
definition_table_case closed-cdata 0
|
|
215
|
+
for definition_control in surviving-anchor unchanged-table foreign-owner duplicate inline-comment multiline-comment backtick-fence tilde-fence raw-html raw-html-open-line processing-instruction cdata declaration html-block indented-code no-header header-anchor no-red; do
|
|
216
|
+
definition_table_case "$definition_control" 1
|
|
217
|
+
done
|
|
218
|
+
|
|
219
|
+
# Evidence is not its own firing surface. Keep the anchor only in the locator
|
|
220
|
+
# itself so uniqueness cannot accidentally reject the self-certifying row.
|
|
221
|
+
new_case case-ref-definition-table-self-register
|
|
222
|
+
printf '\nFixture bookkeeping note, with no new enforcing rule.\n' >> "$REPO/skills/skill-extraction-workflow/references/validation-and-landing.md"
|
|
223
|
+
printf '\n| Rule | Owner | Behavior | Status | Evidence |\n| --- | --- | --- | --- | --- |\n' >> "$REGISTER"
|
|
224
|
+
printf '| Fixture self-citing table | `downstream-executor` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: file:skills/skill-extraction-workflow/references/source-register.md#SELF-REGISTER-DEFINITION-ANCHOR | updated | `skill-extraction-workflow/SKILL.md` bookkeeping edit |\n' >> "$REGISTER"
|
|
225
|
+
commit_case "source register cannot certify its own firing path"
|
|
226
|
+
run_gate
|
|
227
|
+
assert_rc "$rc" 1 "source register must not be its own firing surface"
|
|
228
|
+
assert_contains 'impact_chain_firing_path_missing' "$out" "self-register refusal must name the firing-path boundary"
|
|
229
|
+
|
|
157
230
|
# A purely DESCRIPTIVE list line ("always exposes" — no imperative/prohibitive
|
|
158
231
|
# verb) is not an enforcing rule; the widened verb list must not admit it.
|
|
159
232
|
new_case case-ref-descriptive-always-firing-path
|
|
@@ -1583,6 +1656,6 @@ assert_rc "$rc" 1 "a directory masquerading as SKILL.md must not read as a prese
|
|
|
1583
1656
|
assert_contains "platform-observability/SKILL.md" "$out" "the masqueraded owner must be named"
|
|
1584
1657
|
|
|
1585
1658
|
assert_rc "$full_check_runs" 1 "fixture suite must retain exactly one full checker wiring case"
|
|
1586
|
-
assert_rc "$gate_runs"
|
|
1659
|
+
assert_rc "$gate_runs" 118 "all remaining impact-chain fixtures must run the standalone gate"
|
|
1587
1660
|
|
|
1588
1661
|
echo "test_check_ccl_impact_chain_refscripts: ok"
|
|
@@ -21,9 +21,10 @@ fail() { printf 'FAIL: %s\n' "$1" >&2; exit 1; }
|
|
|
21
21
|
|
|
22
22
|
tmp_root="$(mktemp -d "${TMPDIR:-/tmp}/controlled-escalation-pins.XXXXXX")"
|
|
23
23
|
trap 'rm -rf "$tmp_root"' EXIT
|
|
24
|
-
# The fixture
|
|
25
|
-
# from its own location three levels up
|
|
24
|
+
# The fixture reads skills/ and the cross-host bootstrap, deriving its repo
|
|
25
|
+
# root from its own location three levels up. Preserve both input trees.
|
|
26
26
|
cp -R "$repo_root/skills" "$tmp_root/skills"
|
|
27
|
+
cp -R "$repo_root/agent-context" "$tmp_root/agent-context"
|
|
27
28
|
copy_fixture="$tmp_root/$fixture_rel"
|
|
28
29
|
copy_ref="$tmp_root/$ref_rel"
|
|
29
30
|
[[ -f "$copy_fixture" && -f "$copy_ref" ]] || fail "copy is missing the fixture or the reference"
|