@ccoalm/ccl-skills 0.15.0 → 0.15.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +80 -4
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +16 -4
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +13 -2
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +14 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +40 -83
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +44 -37
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +8 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +31 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +74 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +103 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +74 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_runtime.py +428 -0
- package/dist/assets/release.json +45 -30
- package/dist/claude-adapter.js +14 -7
- package/dist/codex-host.d.ts +1 -3
- package/dist/codex-host.js +6 -9
- package/dist/host-probe.d.ts +11 -0
- package/dist/host-probe.js +27 -0
- package/dist/opencode-adapter.js +24 -19
- package/dist/unified.js +11 -9
- package/package.json +1 -1
|
@@ -1,90 +1,47 @@
|
|
|
1
|
-
# gRPC
|
|
2
|
-
|
|
3
|
-
##
|
|
4
|
-
|
|
5
|
-
HTTP/2
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
- Enforce at CI / lint when defining new services.
|
|
22
|
-
- Migrate existing names by renaming + parallel registration during a deprecation window.
|
|
23
|
-
|
|
24
|
-
### Fix 2 (live-system workaround): mesh-level rewrite
|
|
25
|
-
|
|
26
|
-
When you can't break existing names, an Envoy Lua filter rewrites `:authority` on the fly:
|
|
27
|
-
|
|
28
|
-
```yaml
|
|
29
|
-
apiVersion: networking.istio.io/v1alpha3
|
|
30
|
-
kind: EnvoyFilter
|
|
31
|
-
metadata:
|
|
32
|
-
name: modify-grpc-authority
|
|
33
|
-
namespace: istio-system
|
|
34
|
-
spec:
|
|
35
|
-
configPatches:
|
|
36
|
-
- applyTo: HTTP_FILTER
|
|
37
|
-
match:
|
|
38
|
-
context: ANY
|
|
39
|
-
listener:
|
|
40
|
-
filterChain:
|
|
41
|
-
filter:
|
|
42
|
-
name: "envoy.filters.network.http_connection_manager"
|
|
43
|
-
patch:
|
|
44
|
-
operation: INSERT_BEFORE
|
|
45
|
-
value:
|
|
46
|
-
name: envoy.filters.http.lua
|
|
47
|
-
typed_config:
|
|
48
|
-
"@type": type.googleapis.com/envoy.extensions.filters.http.lua.v3.Lua
|
|
49
|
-
inlineCode: |
|
|
50
|
-
function envoy_on_request(request_handle)
|
|
51
|
-
local authority = request_handle:headers():get(":authority")
|
|
52
|
-
local content_type = request_handle:headers():get("content-type")
|
|
53
|
-
if authority and content_type and string.find(content_type, "application/grpc") then
|
|
54
|
-
local modified_authority = string.gsub(authority, "_", "-")
|
|
55
|
-
request_handle:headers():replace(":authority", modified_authority)
|
|
56
|
-
end
|
|
57
|
-
end
|
|
58
|
-
```
|
|
59
|
-
|
|
60
|
-
Effects:
|
|
61
|
-
- Applies to gRPC traffic only (content-type check).
|
|
62
|
-
- Rewrites `_` to `-` in `:authority`.
|
|
63
|
-
- Callee's registry instance name must also use the `-` form so routing matches.
|
|
64
|
-
|
|
65
|
-
## When to use which
|
|
66
|
-
|
|
67
|
-
| Situation | Fix |
|
|
1
|
+
# gRPC Authority Compatibility
|
|
2
|
+
|
|
3
|
+
## Separate the names and constraints
|
|
4
|
+
|
|
5
|
+
HTTP/2 `:authority` conveys the target URI's authority, as defined by [RFC 9113 §8.3.1](https://www.rfc-editor.org/rfc/rfc9113.html#section-8.3.1). [RFC 3986 §3.2.2](https://www.rfc-editor.org/rfc/rfc3986.html#section-3.2.2) permits a registered name containing unreserved characters, including `_`. This syntax does not guarantee DNS resolution, certificate identity matching, or acceptance by every SDK and proxy version.
|
|
6
|
+
|
|
7
|
+
Keep these values distinct when diagnosing a request:
|
|
8
|
+
|
|
9
|
+
- Service-registry identifier and resolved network endpoint.
|
|
10
|
+
- HTTP/2 authority used for virtual-host routing.
|
|
11
|
+
- TLS server name and the certificate identity expected on each TLS hop.
|
|
12
|
+
- Platform naming, routing, and authorization policies.
|
|
13
|
+
|
|
14
|
+
A platform can require DNS-compatible service names and enforce that choice at registration and CI. Existing identifiers do not require migration merely because they contain an underscore; first establish which constraint the actual path violates.
|
|
15
|
+
|
|
16
|
+
## Locate the rejection before choosing a fix
|
|
17
|
+
|
|
18
|
+
Record the runtime/SDK, proxy versions and relevant configuration, exact authority, and the failing run's error or trace. Use a synthetic payload and redact credentials from captured evidence.
|
|
19
|
+
|
|
20
|
+
| Observed boundary | Next action |
|
|
68
21
|
|---|---|
|
|
69
|
-
|
|
|
70
|
-
|
|
|
71
|
-
|
|
|
72
|
-
|
|
|
22
|
+
| Authority syntax is malformed | Validate URI authority syntax, including brackets around an IPv6 literal, before changing service registration or routing. |
|
|
23
|
+
| Client rejects before transmitting HTTP/2 headers | Check that client's authority validation and supported configuration. Fix the client-side mapping or naming contract; a downstream proxy cannot rewrite a request it never receives. |
|
|
24
|
+
| DNS resolution fails | Check the resolved hostname and resolver's naming rules. Changing a later HTTP header does not repair failed resolution. |
|
|
25
|
+
| TLS handshake or certificate identity check fails | Check that hop's endpoint, server name, trust chain, and certificate identities. Retain verification; changing authority is not evidence that TLS is fixed. |
|
|
26
|
+
| Proxy/parser rejects before the HTTP filter runs | Fix the supported parser/input contract or an earlier owned mapping. A Lua filter after the rejection cannot intervene. |
|
|
27
|
+
| Request reaches HTTP filters, then the intended virtual-host route does not match | Compare the received authority with the generated route configuration. A supported authority mapping may be appropriate after proving the mismatch. |
|
|
28
|
+
| Existing path accepts the name and reaches the intended service | Preserve it unless a separate platform naming-policy migration is required. |
|
|
73
29
|
|
|
74
|
-
|
|
30
|
+
`RST_STREAM` or `INTERNAL_ERROR` alone does not identify an underscore problem. Confirm the first rejecting layer rather than treating every transport failure as the same naming defect.
|
|
75
31
|
|
|
76
|
-
|
|
77
|
-
- Registry rejects registration with `_` in name.
|
|
78
|
-
- Linter / CI rejects PR adding such a name.
|
|
32
|
+
## Choose a bounded compatibility change
|
|
79
33
|
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
34
|
+
**Naming policy or client mapping.** Where a DNS-compatible name is required, define the allowed form and the mapping from registry identity to endpoint/authority. Check uniqueness before migration: replacing `_` with `-` can collapse two distinct names. Migrate registrations, routes, and callers together with a compatibility window and rollback path. Use supported client options; do not bypass certificate or authorization checks to make a name work.
|
|
35
|
+
|
|
36
|
+
**Proxy mapping.** Use only if the request reaches the chosen filter and the mapping addresses a reproduced failure. Prefer the platform's supported routing mechanism. If an EnvoyFilter is necessary, verify the installed Istio/Envoy API and generated configuration, and scope it to the affected workloads, listener/direction, route, and explicit old-to-new authority mapping. Do not install an all-workload, all-direction underscore replacement. The mapping must preserve the intended destination, tenant/lane routing, authorization, and TLS identity on each hop; rewriting a header does not itself update those contracts.
|
|
37
|
+
|
|
38
|
+
## Verification
|
|
83
39
|
|
|
84
|
-
|
|
40
|
+
Retain the original failing case and expected rejecting layer. After the change, verify:
|
|
85
41
|
|
|
86
|
-
|
|
87
|
-
-
|
|
88
|
-
|
|
42
|
+
1. The same request succeeds through the intended path and reaches the intended service; observe authority before/after any mapping and the selected route.
|
|
43
|
+
2. Unrelated authorities and non-target traffic retain their behavior. Include potentially colliding names and unknown authorities as negative controls.
|
|
44
|
+
3. Certificate identity and authorization failures still reject requests; the compatibility change has not disabled those checks.
|
|
45
|
+
4. Registration/client/route changes can be rolled back together without sending traffic to another service.
|
|
89
46
|
|
|
90
|
-
|
|
47
|
+
Static configuration validation proves only configuration properties. Claims about a deployed SDK/proxy path require execution evidence from that path.
|
|
@@ -33,7 +33,7 @@ Reference topology and component rules. Istio is the recurring example; rules ge
|
|
|
33
33
|
|
|
34
34
|
## Istio Ambient Mode (GA Nov 2024, v1.24+)
|
|
35
35
|
|
|
36
|
-
- **Istio Ambient Mode reached General Availability in Istio v1.24 (announced November 2024)** per `istio.io/latest/blog/2024/ambient-reaches-ga/` — `ztunnel` + `waypoint` architecture is the stable sidecar alternative for new production mesh deployments. **Two-layer split**: `ztunnel` (Rust-based DaemonSet, one per node, L4-only — mTLS + simple L4 authz + telemetry) handles every pod's transport-layer mesh participation without per-pod sidecar injection; `waypoint` proxies (Envoy-based, scaled independently from app workloads) handle L7 features when needed (rich authz, traffic routing, resilience). Per Istio's own reported numbers, the architecture can save 90%+ memory/CPU vs the sidecar model in dense-pod-per-node workloads (Istio's claim — re-measure on the team's actual workload before quoting savings). **Architecture choice for new mesh**: (a) **choose ambient** when memory/CPU per pod is the binding constraint, sidecar injection causes restart cycles the team wants to avoid, or a subset of namespaces don't need L7 features at all; (b) **stay on sidecar mode** when the team has deep sidecar-specific tooling (custom `EnvoyFilter`s injected per pod, sidecar-aware debugging recipes, app code that expects `localhost:15001` proxy conventions), or when ambient's narrower L7-feature coverage hits a gap the team relies on. **Mixed-mode in one cluster is officially supported** per the Istio docs: sidecar-mode namespaces and ambient-mode namespaces coexist; migration can be incremental namespace-by-namespace. **CRITICAL migration block:
|
|
36
|
+
- **Istio Ambient Mode reached General Availability in Istio v1.24 (announced November 2024)** per `istio.io/latest/blog/2024/ambient-reaches-ga/` — `ztunnel` + `waypoint` architecture is the stable sidecar alternative for new production mesh deployments. **Two-layer split**: `ztunnel` (Rust-based DaemonSet, one per node, L4-only — mTLS + simple L4 authz + telemetry) handles every pod's transport-layer mesh participation without per-pod sidecar injection; `waypoint` proxies (Envoy-based, scaled independently from app workloads) handle L7 features when needed (rich authz, traffic routing, resilience). Per Istio's own reported numbers, the architecture can save 90%+ memory/CPU vs the sidecar model in dense-pod-per-node workloads (Istio's claim — re-measure on the team's actual workload before quoting savings). **Architecture choice for new mesh**: (a) **choose ambient** when memory/CPU per pod is the binding constraint, sidecar injection causes restart cycles the team wants to avoid, or a subset of namespaces don't need L7 features at all; (b) **stay on sidecar mode** when the team has deep sidecar-specific tooling (custom `EnvoyFilter`s injected per pod, sidecar-aware debugging recipes, app code that expects `localhost:15001` proxy conventions), or when ambient's narrower L7-feature coverage hits a gap the team relies on. **Mixed-mode in one cluster is officially supported** per the Istio docs: sidecar-mode namespaces and ambient-mode namespaces coexist; migration can be incremental namespace-by-namespace. **CRITICAL migration block: L7 policy enforcement changes when moving from sidecar to ambient.** Ztunnel enforces only L4 policy. A workload-selector policy containing L7 conditions that is picked up by ztunnel [fails safe as a DENY policy](https://istio.io/latest/docs/ambient/usage/l4-policy/#policies-with-layer-7-conditions), potentially blocking legitimate traffic. L7 enforcement requires a waypoint and `targetRefs` bound to the intended Service or waypoint Gateway; deploying a waypoint alone does not migrate selector-based policies. **Pre-flip gate**: inventory affected authorization policies, prepare the waypoint and correctly scoped bindings, and plan the old-policy/pod-restart transition using the deployed version's [migration guide](https://istio.io/latest/docs/ambient/migrate/migrate-policies/). Preserve mTLS and L4 authorization; do not remove a rejecting policy merely to restore connectivity. Verify allowed, denied, and waypoint-bypass paths before completing migration. If the transition cannot maintain required L7 protection, hold migration or use an approved maintenance window. **Observability shift**: ztunnel emits L4 metrics; waypoint emits L7 metrics; existing dashboards keyed on sidecar `istio_requests_total` need a ambient-mode equivalent for the L7-routed traffic — route this update through `platform-observability` rather than redesigning here.
|
|
37
37
|
|
|
38
38
|
## Sidecar injection rules
|
|
39
39
|
|
|
@@ -94,7 +94,7 @@ Egress gateway is OFF by default. Turn ON only when a workload needs controlled
|
|
|
94
94
|
EnvoyFilter is the escape hatch. Use sparingly; each filter is hard to test and easy to break across Envoy upgrades.
|
|
95
95
|
|
|
96
96
|
Acceptable uses:
|
|
97
|
-
- Header
|
|
97
|
+
- Header mapping for a reproduced compatibility failure, scoped to the affected workloads and explicit names while preserving routing, TLS, and authorization (see `grpc-authority-workaround.md`).
|
|
98
98
|
- Adding a Lua filter for one-off business handling that doesn't yet have a first-class WASM filter.
|
|
99
99
|
- Custom rate-limit before the canonical RLS is rolled out.
|
|
100
100
|
|
|
@@ -1,48 +1,48 @@
|
|
|
1
1
|
# Retry, Timeout, Circuit Breaker
|
|
2
2
|
|
|
3
|
-
Where each policy lives, how
|
|
3
|
+
Where each policy lives, how timeouts compose, and how to bound retry amplification. Istio field names below apply to Istio HTTP/gRPC paths; other transports keep their platform-owned equivalents.
|
|
4
4
|
|
|
5
5
|
## Layered policy
|
|
6
6
|
|
|
7
|
-
For a single call A → B,
|
|
7
|
+
For a single call A → B, several independent timers can end the call:
|
|
8
8
|
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
TCP/HTTP2 connect + idle timeouts ← Envoy defaults; rarely tuned
|
|
17
|
-
```
|
|
9
|
+
| Timer | Owner | Example |
|
|
10
|
+
|---|---|---|
|
|
11
|
+
| Caller deadline | App context | 800ms remaining |
|
|
12
|
+
| Total client-call budget | Framework client | 500ms, including retries and backoff |
|
|
13
|
+
| Mesh request timeout | VirtualService HTTP route `timeout` | 2s platform cap |
|
|
14
|
+
| Per-attempt timeout | VirtualService HTTP route `retries.perTryTimeout`, or the retry-owning SDK | Must fit the remaining call budget |
|
|
15
|
+
| Connection / idle timeout | DestinationRule connection pool / transport | Bounds connection setup or inactivity, not total business work |
|
|
18
16
|
|
|
19
|
-
|
|
17
|
+
The earliest applicable expiry wins. With the example values measured from call start, the client ends the call at 500ms; the 2s mesh cap can remain a platform backstop. These are not nested durations that must decrease from client to mesh. Propagate cancellation so downstream work stops when the caller's budget expires; a longer transport cap does not extend the caller's deadline.
|
|
18
|
+
|
|
19
|
+
Istio's [HTTPRoute and HTTPRetry fields](https://istio.io/latest/docs/reference/config/networking/virtual-service/) own request/attempt timeout and request retries. [DestinationRule connection-pool settings](https://istio.io/latest/docs/reference/config/networking/destination-rule/) own connection timeouts and concurrent-retry limits; its traffic policy owns outlier detection.
|
|
20
20
|
|
|
21
21
|
## Timeout budget rule
|
|
22
22
|
|
|
23
23
|
For a chain A → B → C:
|
|
24
24
|
|
|
25
25
|
```
|
|
26
|
-
A.ctx.deadline
|
|
27
|
-
A's client-to-B budget
|
|
28
|
-
B's
|
|
29
|
-
|
|
26
|
+
A's remaining duration = A.ctx.deadline - now
|
|
27
|
+
A's client-to-B budget ≤ remaining duration - margin (e.g. 50-100ms for serialization)
|
|
28
|
+
B's client-to-C budget ≤ B's remaining duration - margin
|
|
29
|
+
All sequential work, attempts, and backoff fit within that remaining budget
|
|
30
30
|
```
|
|
31
31
|
|
|
32
|
-
A framework helper
|
|
32
|
+
A framework helper computes `min(remaining_duration - margin, configured_call_budget, applicable_platform_cap)`. If no usable duration remains, fail without dispatching another attempt. Pass the resulting deadline downstream and recompute the remainder after work or backoff; do not compare an absolute deadline with a duration or restart the full budget at each hop.
|
|
33
33
|
|
|
34
34
|
## Retry placement
|
|
35
35
|
|
|
36
36
|
| Layer | Retries WHAT | When |
|
|
37
37
|
|---|---|---|
|
|
38
|
-
| Mesh (
|
|
38
|
+
| Mesh (VirtualService HTTP route) | Explicitly selected network errors or response statuses | Declared idempotent calls, or failures proven to precede server receipt |
|
|
39
39
|
| Framework client | idempotent business RPCs | Per-method opt-in |
|
|
40
40
|
| App handler | nothing | App level retry usually wrong |
|
|
41
41
|
| App business logic | high-level workflows | Saga / orchestration patterns, not "I'll retry the call once" |
|
|
42
42
|
|
|
43
|
-
|
|
43
|
+
Retry layers multiply total attempts. If mesh and framework each make at most two total attempts, one logical call can reach the backend four times. Istio `retries.attempts` counts retries after the initial request: a value of 2 permits up to 3 total attempts, subject to time and retry budgets.
|
|
44
44
|
|
|
45
|
-
**Rule**: when enabling framework-level retry,
|
|
45
|
+
**Rule**: when enabling framework-level retry, set `retries.attempts: 0` on every matching VirtualService HTTP route used by that call and verify the effective generated route configuration. Keep DestinationRule connection limits and outlier detection; they do not replace the route-level retry switch.
|
|
46
46
|
|
|
47
47
|
Mesh retries back off automatically (Istio/Envoy: jittered exponential backoff with a default 25ms *base* interval — fully jittered, so an actual delay can be shorter than the base; it is not a guaranteed minimum gap); framework-level retry gets no such freebie — it must implement its own jittered backoff that fits inside the caller's remaining deadline.
|
|
48
48
|
|
|
@@ -51,24 +51,24 @@ Mesh retries back off automatically (Istio/Envoy: jittered exponential backoff w
|
|
|
51
51
|
Per-call retry counts bound retries *per request*; they do not bound a caller's total retry share during a partial outage — at high QPS, "2 retries each" is up to a 3× load multiplier at the exact moment the upstream is sickest. Envoy's cluster circuit breakers cap this per proxy:
|
|
52
52
|
|
|
53
53
|
- `max_retries` — max **concurrent** retries to the cluster, per priority. Retries beyond it overflow (fail fast, counted in `upstream_rq_retry_overflow`). The raw Envoy default is 3, but the control plane above Envoy may override it: Istio's `connectionPool.http.maxRetries` defaults to **2^32-1 — effectively unlimited** — so in an Istio mesh "leave it unset and rely on the default cap" is a trap. Set the limit explicitly and verify the *generated* Envoy cluster config, not the assumption.
|
|
54
|
-
- `retry_budget` — replaces the fixed cap with a load-proportional one: concurrent retries ≤ `budget_percent` (default 20%) of active + pending requests, with a `min_retry_concurrency` floor so low-traffic clusters can still retry. When set,
|
|
54
|
+
- `retry_budget` — replaces the fixed cap with a load-proportional one: in the default instantaneous mode, concurrent retries ≤ `budget_percent` (default 20%) of active + pending requests, with a `min_retry_concurrency` floor so low-traffic clusters can still retry. Versions exposing a non-zero `budget_interval` can count requests over that interval instead; verify the configured version and mode. When set, the budget overrides `max_retries`. Reachability caveat: Istio's DestinationRule API exposes only `connectionPool.http.maxRetries`, NOT `retry_budget` — on plain Istio, set a finite `maxRetries` first; adopting `retry_budget` there means an EnvoyFilter, acceptable only with the *generated* cluster config verified.
|
|
55
55
|
- Know exactly what the budget bounds — and what it doesn't. It bounds **Envoy-originated, concurrent** retries, per proxy. It does NOT bound: retry attempt *rate*; **framework-level retries** (each framework attempt arrives at Envoy as a fresh request and bypasses `max_retries`/`retry_budget` entirely — a platform running framework retries needs a framework-side budget or strict per-call caps); or the **fleet aggregate** (circuit breaking is distributed, not coordinated — each sidecar enforces its own budget and floor, so aggregate retry load still scales with caller replica count). A true service-wide load bound requires callee-side protection (admission control / load shedding, owned by the service-architecture skills) on top.
|
|
56
56
|
- When tuning for a flaky dependency, set a retry budget rather than raising per-call retry counts — but pick `budget_percent` AND `min_retry_concurrency` deliberately against the callee's capacity: on a very high-QPS caller, an unexamined 20% of active requests is far looser than `max_retries: 3`, and with many low-traffic sidecars the aggregate floor (≈ replicas × `min_retry_concurrency`) dominates instead. Alert on the overflow counter: a growing overflow stat means callers are shedding retries, which is the budget doing its job; do not "fix" it by raising the cap.
|
|
57
57
|
|
|
58
58
|
## Idempotency awareness
|
|
59
59
|
|
|
60
|
-
|
|
60
|
+
Both mesh and framework client retries MUST consider idempotency:
|
|
61
61
|
|
|
62
62
|
```
|
|
63
63
|
RPC method declares "idempotent: true" in IDL or annotation
|
|
64
64
|
↓
|
|
65
|
-
|
|
65
|
+
The retry-owning layer applies a policy only to methods covered by that declaration
|
|
66
66
|
↓
|
|
67
67
|
Non-idempotent retry happens only if the network error proves the request didn't reach the server
|
|
68
68
|
(connect refused, TLS handshake failure — yes; "request sent, no response" — no)
|
|
69
69
|
```
|
|
70
70
|
|
|
71
|
-
Don't trust HTTP method (POST can be idempotent; GET can have side effects). Trust the declaration.
|
|
71
|
+
Don't trust HTTP method (POST can be idempotent; GET can have side effects). Trust the declaration. A 5xx, gRPC `UNAVAILABLE`, timeout, or missing response alone does not prove that a write was not applied. If the mesh cannot distinguish safe methods, keep its request retries disabled and let the method-aware SDK own the policy.
|
|
72
72
|
|
|
73
73
|
## Circuit breaker / outlier detection
|
|
74
74
|
|
|
@@ -78,12 +78,12 @@ Mesh outlier-detection (Envoy):
|
|
|
78
78
|
trafficPolicy:
|
|
79
79
|
outlierDetection:
|
|
80
80
|
consecutive5xxErrors: 5 # 5 consecutive 5xx → eject this pod
|
|
81
|
-
interval: 10s #
|
|
81
|
+
interval: 10s # periodic ejection analysis / recovery sweep
|
|
82
82
|
baseEjectionTime: 30s # eject for 30s minimum
|
|
83
83
|
maxEjectionPercent: 50 # never eject more than 50% of pool
|
|
84
84
|
```
|
|
85
85
|
|
|
86
|
-
|
|
86
|
+
For a meshed path, this provides per-host ejection observable through Envoy stats. [Envoy's ejection algorithm](https://www.envoyproxy.io/docs/envoy/latest/intro/arch_overview/upstream/outlier) checks consecutive-error thresholds inline; periodic analyses use `interval`. An ejection also depends on the configured threshold/enforcement and pool limits. Killing a pod once does not prove those conditions fired.
|
|
87
87
|
|
|
88
88
|
Framework SDK circuit breakers (e.g. hystrix-style) are a fallback for environments without mesh, or for business-logic-driven breaking (e.g. "this dependency's error rate hit 10%, switch to degraded mode").
|
|
89
89
|
|
|
@@ -91,36 +91,40 @@ Don't run mesh outlier-detection AND SDK circuit breaker simultaneously without
|
|
|
91
91
|
|
|
92
92
|
## Cascading cancel
|
|
93
93
|
|
|
94
|
-
When a caller's ctx is cancelled (deadline, client disconnect, user back-button),
|
|
94
|
+
When a caller's ctx is cancelled (deadline, client disconnect, user back-button), propagate it to in-flight downstream calls. [gRPC cancellation](https://grpc.io/docs/guides/cancellation/) requires application handlers to cooperate; outgoing-call propagation also depends on the language/runtime. Verify cancellation reaches the handler, stops its work and child calls, and releases resources within the service's documented cancellation bound. A transport cancellation signal alone does not prove application work stopped.
|
|
95
95
|
|
|
96
96
|
If a service swallows ctx cancel, downstream load amplifies during user disconnects (every abandoned tab continues hammering the DB).
|
|
97
97
|
|
|
98
98
|
## Hedging
|
|
99
99
|
|
|
100
|
-
Hedging
|
|
100
|
+
Hedging sends an additional request while the first is still in flight and returns the first acceptable response. It can reduce tail latency for idempotent reads, at the cost of concurrent upstream work.
|
|
101
101
|
|
|
102
102
|
Risks:
|
|
103
103
|
- Doubles load if T is too short.
|
|
104
104
|
- Not safe for non-idempotent calls.
|
|
105
|
-
-
|
|
105
|
+
- Bound concurrent attempts, total deadline and admitted load; cancel losing attempts after choosing a response and propagate caller cancellation.
|
|
106
|
+
- Envoy's [HedgePolicy](https://www.envoyproxy.io/docs/envoy/latest/api-v3/config/route/v3/route_components.proto#config-route-v3-hedgepolicy) uses `hedge_policy.hedge_on_per_try_timeout` with a finite per-try timeout and a retry policy containing retry conditions and a positive retry limit. `retry_priority` selects upstream priorities; it does not enable hedging.
|
|
107
|
+
- Istio's HTTPRetry API does not expose a hedging field. Use an explicitly supported platform mechanism (such as an EnvoyFilter with generated-route and runtime verification), or a method-aware SDK; do not infer hedging from `perTryTimeout` alone. Keep a single retry/hedge owner.
|
|
106
108
|
|
|
107
109
|
Default: off. Enable per-method after measuring p99 latency distribution.
|
|
108
110
|
|
|
109
111
|
## Common mistakes
|
|
110
112
|
|
|
111
|
-
-
|
|
112
|
-
- Retrying
|
|
113
|
+
- Treating a shorter mesh timeout as invisible to the client → the client can receive a mesh timeout response before its own deadline. Verify error mapping and the retry-owning layer; do not turn that response into an unsafe replay.
|
|
114
|
+
- Retrying every 4xx → permanent client errors do not fix themselves. Only a documented transient condition, such as a rate-limit response with a bounded retry delay, may be eligible under the idempotency and remaining-budget rules.
|
|
113
115
|
- Retrying with exponential backoff while caller deadline is 200ms → backoff exceeds deadline, retry never fires, you wasted code.
|
|
114
|
-
- Mesh
|
|
116
|
+
- Mesh and SDK each allow 3 retries after the initial request → up to 16 backend attempts per logical call, before deadline/budget limits.
|
|
115
117
|
- Circuit breaker tripped but no metric → debugging blind.
|
|
116
118
|
|
|
117
119
|
## Tuning starting points
|
|
118
120
|
|
|
119
|
-
|
|
121
|
+
These are example starting points for eligible traffic, not vendor defaults or a mandate to enable retries. Apply the idempotency and single-owner rules first.
|
|
122
|
+
|
|
123
|
+
| Policy | Starting point |
|
|
120
124
|
|---|---|
|
|
121
125
|
| Mesh request timeout | 5s (HTTP), 10s (RPC); per-callee override |
|
|
122
|
-
| Mesh retry attempts | 2 |
|
|
123
|
-
| Mesh retry per-try timeout |
|
|
126
|
+
| Mesh retry attempts | 0 unless mesh owns safe retries; then up to 2 retries within the call budget |
|
|
127
|
+
| Mesh retry per-try timeout | Fit within the remaining call budget; reserve time for backoff and any later attempt |
|
|
124
128
|
| Mesh outlier detection | 5 consecutive 5xx, 30s eject |
|
|
125
129
|
| Framework client per-call budget | 500ms or shrunk from ctx deadline |
|
|
126
130
|
| Framework client retry | OFF by default; opt-in per idempotent method |
|
|
@@ -130,6 +134,9 @@ Tune from observed p99 + error rate, not vibes.
|
|
|
130
134
|
## Verification
|
|
131
135
|
|
|
132
136
|
- Trigger downstream 5xx storm → mesh outlier-detection metrics show ejection events; client-side error rate spikes then recovers.
|
|
133
|
-
- Force a connect refusal →
|
|
137
|
+
- Force a connect refusal → only the configured retry owner produces attempts, counts fit its budget, and the final caller sees one error. With SDK-owned retries, confirm effective route retries are disabled.
|
|
134
138
|
- Set per-call timeout below mesh ceiling → confirm caller sees its own deadline, not mesh's.
|
|
135
|
-
-
|
|
139
|
+
- Set mesh request timeout below the client budget → confirm the mapped mesh error is visible and does not trigger a second, unintended retry layer.
|
|
140
|
+
- For a non-idempotent write with a lost response, confirm neither layer blindly replays it.
|
|
141
|
+
- With hedging enabled, delay an eligible call past the per-try timeout → confirm bounded concurrent attempts, first acceptable response selection, loser cancellation, and no second retry/hedge owner.
|
|
142
|
+
- Cancel a request mid-flight → confirm handler work and downstream calls stop and resources release within the declared cancellation bound.
|
|
@@ -79,7 +79,7 @@ The framework client resolver:
|
|
|
79
79
|
## Cross-language registration
|
|
80
80
|
|
|
81
81
|
If services span Go, Python, Java, Node: every language SDK must agree on:
|
|
82
|
-
- Service name shape
|
|
82
|
+
- Service name shape and its mapping to endpoint, authority, and TLS identity. Apply the agreed platform naming policy and actual DNS/SDK/proxy constraints; gRPC alone does not require renaming an existing working identifier (see `grpc-authority-workaround.md`).
|
|
83
83
|
- Tag key names (`lane`, not `env`; pick one).
|
|
84
84
|
- Heartbeat interval and TTL.
|
|
85
85
|
- Health state semantics.
|
|
@@ -180,7 +180,7 @@ Before release, confirm:
|
|
|
180
180
|
| **Change Lead Time** | commit 到 prod 的时长 | — |
|
|
181
181
|
| **Change Failure Rate (CFR)** | release 中需要 hotfix / rollback / fail forward 的比例 | — |
|
|
182
182
|
| **Failed Deployment Recovery Time** | failed deployment 恢复时长 | 2023 年 DORA 重命名(原 MTTR)|
|
|
183
|
-
| **Deployment Rework Rate** |
|
|
183
|
+
| **Deployment Rework Rate** | 由生产事故引发的非计划部署占全部部署的比例 | 2024 年新增;[当前 DORA 口径](https://dora.dev/guides/dora-metrics/) |
|
|
184
184
|
|
|
185
185
|
Elite / High / Medium / Low 具体阈值**按当年 DORA Annual State of DevOps Report 取**(不同年份数字略有变化,本 ref 不固定数字以免过时)。
|
|
186
186
|
|
|
@@ -12,7 +12,10 @@ description: 改动范围 / 影响范围 / scope / 需求拆分 / MVP 边界 /
|
|
|
12
12
|
1. **P0 核心 in/out 由 `human-decision` 关闭。** agent 不得自行决定,也不得决定产品目标、核心路径、成本级别或验收承诺。
|
|
13
13
|
2. **as-is 证据不裁决 should-be 范围。** 代码、数据、架构等描述性证据不能自行决定本轮改不改。
|
|
14
14
|
3. **边界要可审查。** in/out 写成行为边界,不用「优化体验」这类不可判定的模糊标签。
|
|
15
|
-
4.
|
|
15
|
+
4. **先判 appetite 是否适用,再记录投入与取舍。** [Shape Up 的 appetite](https://basecamp.com/shapeup/1.2-chapter-03) 用于在固定投入下调整范围;它不是每张影响范围表的必填前提。
|
|
16
|
+
- 任务要求投入决策,或已有适用的投入上限时,记录已确认上限和超出时可调整的范围。取舍未定就保留已知上限,把未知子项标 `open`,继续交付其余字段;不得自行砍掉核心范围、降低既有验收要求或承诺按期完成。
|
|
17
|
+
- 人工兜底只在业务连续性、降级或已批准决策确实需要时填写;不需要时写明依据,不为填表虚构人工接管。
|
|
18
|
+
- 仅做影响范围盘点、未涉及投入决策时,可按任务边界记 `not-applicable`。用户要求省略展示细节时遵从该要求,已确认约束和未决事项保留在关闭表中;省略展示不等于撤销既有约束或关闭未决承诺。
|
|
16
19
|
5. **关闭表只补自己那部分。** 唯一 canonical 是 `requirement-doc-writer/references/requirement-closure-contract.md`。本技能按复合字段子项补版本、范围、appetite、依赖和验收,逐项记录推导、决策权和决策证据;只能自动填写单个低风险、可逆的非核心展示/表达细节(decision_authority 记 `bounded-agent-policy`,且须有可引用的已批准 policy),**任一适用子项 open 时复合字段和整行保持 open/blocked**。
|
|
17
20
|
6. **「非目标」在这里是变更级**——本轮明确不改、延后、保持兼容、无需迁移的对象。意图级的「本轮不追求什么目标」属 `requirement-intent`。同理「验收」在这里是每个切片的验收边界与不验收项,不是功能点 pass/fail。
|
|
18
21
|
|
|
@@ -30,7 +33,7 @@ description: 改动范围 / 影响范围 / scope / 需求拆分 / MVP 边界 /
|
|
|
30
33
|
| Out of scope(变更级非目标) | 明确不改、延后、保持兼容、无需迁移的部分 |
|
|
31
34
|
| 受影响对象 | 角色、页面、入口、API、数据、运营规则、通知、报表、权限、文档 |
|
|
32
35
|
| 版本切片 | MVP、后续版本、迁移/兼容切片、回滚/降级边界 |
|
|
33
|
-
| Appetite / timebox |
|
|
36
|
+
| Appetite / timebox | 适用性与依据;适用时记录已确认投入上限、超限取舍和必要的兜底 |
|
|
34
37
|
| 依赖 | 上游决策、外部系统、数据准备、设计、法务/运营/支持动作 |
|
|
35
38
|
| 风险点 | 权限、隔离、计费/配额、删除/覆盖、数据迁移、发布复杂度 |
|
|
36
39
|
| 验收范围 | 每个切片的可观察验收边界和不验收项 |
|
|
@@ -41,7 +44,7 @@ description: 改动范围 / 影响范围 / scope / 需求拆分 / MVP 边界 /
|
|
|
41
44
|
1. 锁定目标与输入:引用已澄清需求或盘点事实。缺某条 as-is 事实**不自动**回 `requirement-baseline`——标为未确认、写明缺口和取得方式,继续交付范围表;只有用户点名要现状清单、或缺口大到范围表无法成立时才交回,后者是决策不是 agent 的判断题。
|
|
42
45
|
2. 列受影响对象:用户、流程、界面/API、数据、权限、运营、文档逐项过一遍。
|
|
43
46
|
3. 写 in/out scope:按硬约束 2、3。
|
|
44
|
-
4.
|
|
47
|
+
4. 切版本:P0、后续、迁移、兼容、回滚/降级分别列清;按硬约束 4 判断并填写 appetite。
|
|
45
48
|
5. 标依赖和风险:把安全 4 问命中项、跨团队/系统依赖、数据风险拉出来。
|
|
46
49
|
6. 更新 `需求点关闭表`(按硬约束 5)。
|
|
47
50
|
7. 写验收范围:每个切片对应 pass/fail 条件,明确不验收项;安全命中项的负向用例写进验收范围。
|
|
@@ -63,7 +66,7 @@ description: 改动范围 / 影响范围 / scope / 需求拆分 / MVP 边界 /
|
|
|
63
66
|
- in/out 每一项都是可审查的行为边界,没有模糊标签。
|
|
64
67
|
- 受影响对象十类(角色/页面/入口/API/数据/运营规则/通知/报表/权限/文档)逐类过了一遍,不适用的显式写「无」。
|
|
65
68
|
- 每个切片都有可观察的验收边界**和**不验收项。
|
|
66
|
-
- appetite
|
|
69
|
+
- appetite 按硬约束 4 判定适用性;适用子项有决定或明确的未决状态,不适用有依据。未知项不被抹掉,也不阻断其余范围材料。
|
|
67
70
|
- P0 核心 in/out 标了 `human-decision`,没有被 agent 自行关闭。
|
|
68
71
|
- 关闭表里任一适用子项 open 时,复合字段和整行保持 open/blocked。
|
|
69
72
|
- 安全 4 问:记了「无命中」,或四项逐条答案 + 写明所依据的 canonical 文件名;命中项的负向用例已在验收范围里。
|
|
@@ -77,7 +80,7 @@ description: 改动范围 / 影响范围 / scope / 需求拆分 / MVP 边界 /
|
|
|
77
80
|
- Out of scope(变更级非目标):
|
|
78
81
|
- 受影响对象:
|
|
79
82
|
- 版本切片:
|
|
80
|
-
- Appetite / timebox
|
|
83
|
+
- Appetite / timebox(适用性与依据;适用时写投入、取舍与必要兜底):
|
|
81
84
|
- 依赖:
|
|
82
85
|
- 风险点:
|
|
83
86
|
- 验收范围(含不验收项):
|
|
@@ -44,9 +44,9 @@ Primary and official sources:
|
|
|
44
44
|
Disposition:
|
|
45
45
|
|
|
46
46
|
- Supported: a rule's existence is not proof it executed; use behavioral assertions and coverage/firing evidence. The digest-binding practices these sources describe (complete subject sets, provenance, raw-result digests) are sound for supply-chain trust boundaries where authors and verifiers are distinct parties; the local policy below explains why this repository adopts the firing/coverage principle but not the digest binding.
|
|
47
|
-
- Local evidence policy: the impact-chain gate machine-verifies what is cheap and deterministic — an owner-scoped firing path that resolves to its round's added lines (a unique anchor on a changed normative numbered/list rule, or a changed owner executable), the letters/digits-free wording-only classification computed from that round's owner diff, and the owner-level floor that a non-wording package carries at least one `RED-baseline` row (a `semantic-control` label may supplement but never close a package alone, because an author-selected stable label cannot vouch for a different hidden delta). The `behavioral-evidence` and `observed-failure` fields themselves are required author declarations. A digest-bound attestation apparatus for these rows (in-toto-style subject digests, same-prompt model result pairs, command-result envelopes) was built, evaluated against real iteration, and deliberately removed: under the unsigned-repository-local trust model the author can regenerate every hash, so the apparatus only detected stale records — while costing a full-suite rerun and whole-evidence regeneration whenever any owner script changed by a single byte. That cost defeated normal multi-commit iteration (it broke its own author's branch twice), so behavior claims rest on the firing-path gate, honest authorship, and the mandatory independent review/challenge instead of hashes.
|
|
48
|
-
- Local trust model: register rows remain honest-but-fallible workflow evidence, not a hostile-author security boundary. The gate proves that a changed
|
|
49
|
-
- Machine format (relocated from the `SKILL.md` firing-mechanism rule; the local evidence policy above carries the rationale): every added source-register row must carry `behavioral-evidence: RED-baseline` (any observed delta — `observed-failure: yes` requires it) or `semantic-control` (only with `observed-failure: no`), an `observed-failure: yes/no` state, and an owner-scoped `firing-path` — each declaration in its own semicolon-delimited fragment of the cell (`…prose; behavioral-evidence: …; observed-failure: …; firing-path: …`), so a key embedded mid-prose never parses as a declaration. The firing-path anchor is at least 16 characters, occurs once in the file and once in its round's added lines, and lands on a numbered/list Markdown rule with a normative action. A row that survives at HEAD must also resolve to an owner this range actually changes — an owner reverted to its base bytes by a rebase or a base-side conflict resolution leaves the changed set while its row stays behind, and the row then vouches for a change the delivered diff does not contain. There is no author-declared escape from this: a corrective rewrite that back-fills a row for a round which merged red produces the same shape, and it is a deliberate, person-adjudicated repair that can adjudicate this refusal too.
|
|
47
|
+
- Local evidence policy: the impact-chain gate machine-verifies what is cheap and deterministic — an owner-scoped firing path that resolves to its round's added lines (a unique anchor on a changed normative numbered/list rule or table data cell, or a changed owner executable), the letters/digits-free wording-only classification computed from that round's owner diff, and the owner-level floor that a non-wording package carries at least one `RED-baseline` row (a `semantic-control` label may supplement but never close a package alone, because an author-selected stable label cannot vouch for a different hidden delta). The `behavioral-evidence` and `observed-failure` fields themselves are required author declarations. A digest-bound attestation apparatus for these rows (in-toto-style subject digests, same-prompt model result pairs, command-result envelopes) was built, evaluated against real iteration, and deliberately removed: under the unsigned-repository-local trust model the author can regenerate every hash, so the apparatus only detected stale records — while costing a full-suite rerun and whole-evidence regeneration whenever any owner script changed by a single byte. That cost defeated normal multi-commit iteration (it broke its own author's branch twice), so behavior claims rest on the firing-path gate, honest authorship, and the mandatory independent review/challenge instead of hashes.
|
|
48
|
+
- Local trust model: register rows remain honest-but-fallible workflow evidence, not a hostile-author security boundary. The gate proves that a changed owner-scoped rule or table data line (or changed owner executable) exists for every claimed firing path; it does not prove a model run occurred, that a named executable implements the claimed enforcement (a shebang stub passes the static check), that a mangled or ambiguous ledger row was honest (those are warned, not blocked, to avoid false positives on other table shapes), author identity, or non-tampering by an authorized contributor. Independent review/challenge and the fixed checker remain the assurance case.
|
|
49
|
+
- Machine format (relocated from the `SKILL.md` firing-mechanism rule; the local evidence policy above carries the rationale): every added source-register row must carry `behavioral-evidence: RED-baseline` (any observed delta — `observed-failure: yes` requires it) or `semantic-control` (only with `observed-failure: no`), an `observed-failure: yes/no` state, and an owner-scoped `firing-path` — each declaration in its own semicolon-delimited fragment of the cell (`…prose; behavioral-evidence: …; observed-failure: …; firing-path: …`), so a key embedded mid-prose never parses as a declaration. The firing-path anchor is at least 16 characters, occurs once in the file and once in its round's added lines, and lands on a numbered/list Markdown rule with a normative action or a data cell in an explicit Markdown table with a header and delimiter. A table anchor must also be absent from the round's base file; changing only a source link cannot reuse the old definition as evidence. Table headers, delimiters, comments, code examples, and HTML blocks are not definition anchors. The source register itself cannot be a file firing path: evidence does not certify itself. The table check establishes the changed location, not the definition's factual accuracy. A row that survives at HEAD must also resolve to an owner this range actually changes — an owner reverted to its base bytes by a rebase or a base-side conflict resolution leaves the changed set while its row stays behind, and the row then vouches for a change the delivered diff does not contain. There is no author-declared escape from this: a corrective rewrite that back-fills a row for a round which merged red produces the same shape, and it is a deliberate, person-adjudicated repair that can adjudicate this refusal too.
|
|
50
50
|
|
|
51
51
|
- **Round scoping — a row is judged against the round it landed in, never the accumulating range.** A row is authored against one round's diff, so reading the whole `base..HEAD` range to classify it judges the row against work it never described. That mismatch produced both directions of the same defect: an already-gated row turned red once a LATER round touched the same owner (which is what the ledger's superseded-row notes were absorbing), and a description-only round lost its routing-surface locator because an EARLIER round had edited that owner's body. The gate cuts rounds at the commits that touch the ledger along a first-parent line, and each round spans from the previous boundary so work commits sit in the round whose ledger append describes them. A merge that git rebuilds from its two parents is expanded into its branch's own rounds, so a merged worktree round is judged exactly as its pull request was — the same history must not partition differently after it lands; a merge git cannot rebuild (a hand resolution, a conflict) keeps a single boundary at the merge, so content that came from neither parent is never left in no round. The partition is derived from git alone — an author cannot nominate, widen, or move their own scope.
|
|
52
52
|
- Both obligations move together, in opposite directions. **Classification** narrows to the round: whether a diff is wording-only, an identifier retarget, or description-only is asked of that round's bytes, which is what makes a verdict stable once it lands. **Presence** narrows to the round too: the round that changed an owner is the round that owes the row, so owner work committed after a ledger append can no longer ride on an earlier round's row. Narrowing classification without narrowing presence would have opened exactly that laundering route.
|
|
@@ -631,3 +631,16 @@ Supersede note (round 115, sixth ledger correction with no rule change, two rows
|
|
|
631
631
|
Supersede note (round 115, seventh ledger correction with no rule change): the observation-validity row above says all 4,905 non-error verdicts in the round's archived reports name a catalog skill or `none`. That total is counted over the archived runner reports in the maintainer's scratch directory, which are not committed and cannot be recounted from this repository. What the repository does carry is `replica-verdicts.tsv`, whose 2430 non-error verdicts all resolve to a catalog skill or `none`; the archive-wide figure stands as the author's count, not as committed evidence.
|
|
632
632
|
| A round partition that depends on which ref is judged moves verdicts after they land: a branch judged per ledger commit as a pull request collapsed to one round once merged, so a row valid at pull-request time (a routing-surface `#description` anchor in a commit that changed only the description) was refused on every post-merge evaluation with nothing about it changed; a merge git rebuilds from its two parents must be expanded into the branch's own rounds so the same history partitions identically before and after it lands, and a merge git cannot rebuild keeps one boundary so content from neither parent is never left in no round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/impact-chain-gate.rb, its round-scoping fixtures, the verdict differential's named divergences, references/external-practice-controls.md and the .github/workflows/ci.yml checkout comments). Observed failure: the integration branch's promotion pull request reported `impact_chain_firing_path_missing` for a row that was green on its own pull request (15 rounds on the branch head, one after the merge), and the integration branch's push build had been red since that merge. RED baseline: round scoping 8 now runs the branch view and the merged view on one fixture and asserts them equal; on the previous gate it fails with `expected rc=1 got rc=0`, and the real promotion shape (integration head against the target) goes from rc=1 to rc=0 with no other change. Verdict differential: 64 integration points, six newly refused, each named by sha with `impact_chain_gate_missing` — all merges from before CI checked out the branch head, whose branches carry owner work outside the round that declares it; none newly accepted. Round scoping 13 pins the observed shape (body round then description-only round, merged) green and equal to its branch view; round scoping 14 pins that a merge whose tree is not the automatic merge keeps a single boundary. Refines the row above beginning "确定性闸对历史形态有前提": the checkout ref binding stays, and the partition no longer depends on it. |
|
|
633
633
|
| A landing chain that looks for a round's review evidence only inside that round's own checkout can never bind a round that merged without its ledger, however honestly the same bytes are reviewed later: evidence is a validator-accepted closeout whose candidate hash equals the round's packet, so the chain reads the landing tree's committed evidence, and a later review of exactly those bytes, landed as a round of its own, binds the earlier round — a closeout for any other digest binds nothing, wherever it sits | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Observed failure: the same promotion pull request's chain refused at a two-file round that had merged with the binder red (`no accepted review evidence binds the landing candidate`), and no forward path existed because the rebind enumerated evidence from the round's detached checkout. RED baseline: the chain case "a validator-accepted closeout for the round's own candidate, committed on the integration branch after the merge, binds it through the chain" fails on the previous binder and passes on this one; the companion case with a closeout for a different digest is refused on both. The round's packet is still frozen from its own checkout at its own base with the landing tree's controller, and its excludes still come from its own added receipts, so a later ledger's absence in the round checkout leaves the round's hash unchanged. The retrospective review of that round's bytes is committed in this round's evidence directory (its retro-round folder) and binds its candidate hash. |
|
|
634
|
+
| Self-review uniqueness is per owner and concern while required owner and concern coverage remain independent | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`; implementation in `code-review/scripts/review_gate.py`. The same nine-owner synthetic harness failed its review and completion assertions before the change while eleven controls passed; afterward all thirteen assertions passed. Missing owners, missing concerns, duplicate pairs, and implicit-versus-explicit default-owner duplicates remain rejected before provider execution. |
|
|
635
|
+
| Actionable capacity warnings can use internal measurements while SLO paging uses the corresponding SLIs | `platform-observability` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-observability/references/alerting-and-on-call.md#diagnostic counters do not become availability | updated | Owner key `platform-observability/SKILL.md`; detail in `platform-observability/references/alerting-and-on-call.md`. The source-contract check alerts_not_exclusively_sli failed against the prior blanket exclusion of raw measurements and passed after correction. Actionability, response ownership, severity, and runbook requirements remain. This is source-contract evidence, not a model-task or production-monitoring trial. |
|
|
636
|
+
| HTTP retry configuration follows route ownership and every replay remains within the caller budget and replay-safety boundary | `platform-service-connectivity` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-service-connectivity/SKILL.md#a 5xx or missing response alone never proves replay safe | updated | Owner key `platform-service-connectivity/SKILL.md`; detail in `platform-service-connectivity/references/retry-timeout-circuit-breaker.md`. Paired source-contract checks rejected the former DestinationRule retry ownership, impossible timeout ordering, and retry-count arithmetic. Corrected arithmetic yields a 500 ms caller budget and sixteen attempts for two layers each allowing three retries. These checks establish instruction and arithmetic corrections, not a live mesh trial. |
|
|
637
|
+
| Evaluation exceptions require bounded process cleanup and preserve the original failure rather than producing successful samples | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/skill-behavior-eval.py | updated | Owner key `skill-extraction-workflow/SKILL.md`; regression cases in `skill-extraction-workflow/scripts/test_eval_runtime.py`. Before the exception-path correction, two focused tests reported four assertion failures and one drain error; afterward all thirteen runtime tests passed. Cleanup uncertainty remains explicit and stops subsequent evaluation work; non-timeout exceptions retain their original identity. |
|
|
638
|
+
|
|
639
|
+
Pending evidence classification: the Deployment Rework Rate table in `product-rd-workflow/references/delivery-lifecycle.md` now defines unplanned deployments caused by production incidents as a proportion of all deployments, with the DORA source linked in that table. This is a factual source comparison. The current impact-chain gate requires a changed normative list rule or executable as the firing path and does not accept this table cell. No behavioral RED baseline or gate acceptance is claimed for that correction.
|
|
640
|
+
|
|
641
|
+
The pending classification above is superseded by the executed source comparison and table-locator regression below. These checks establish source-contract and gate behavior, not model-task improvement.
|
|
642
|
+
|
|
643
|
+
| Deployment rework rate counts unplanned deployments caused by production incidents as a share of all deployments | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/delivery-lifecycle.md#由生产事故引发的非计划部署占全部部署的比例 | updated | Owner key `product-rd-workflow/SKILL.md`. The executed before/after source-contract comparison checked the production-incident cause, unplanned-deployment numerator, and all-deployments denominator against [DORA's metric definition](https://dora.dev/guides/dora-metrics/). All three were absent from the old table definition and present in the corrected definition. This is a factual source comparison, not a model-task or production trial. |
|
|
644
|
+
| Changed definition tables can provide a firing path without inventing a normative list rule | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. The accepted-definition fixture failed before the gate change with expected rc=0 and actual rc=1. Afterward the complete suite passed: one full-checker wiring case and 112 standalone-gate cases. Fifteen table cases cover acceptance plus stale or unchanged anchors, foreign owners, duplicate anchors, comments, fences, raw HTML, indented code, missing headers, header anchors, and missing RED evidence. Existing owner, round, normative-list, and executable checks remain. |
|
|
645
|
+
| An impact-chain evidence row cannot use the source register itself as its firing path | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. A synthetic owner bookkeeping edit and a self-citing table row passed the prior gate with actual rc=0 where refusal rc=1 was required. The anchor occurred only once, inside its own locator, so uniqueness alone did not prevent self-certification. Rejecting the source register as a file firing path made that case pass; the full focused suite passed one checker wiring case and 113 standalone-gate cases. Ordinary changed definition tables still pass. |
|
|
646
|
+
| Table firing anchors exclude raw HTML processing instructions, declarations, CDATA and multiline raw-tag openers | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. An executed paired check of the current table predicate accepted plain data and also accepted data inside processing-instruction, declaration and CDATA blocks. The full gate fixture additionally returned rc=0 for a raw script opener where refusal rc=1 was required. Explicit block terminators now exclude these non-table surfaces; a completed CDATA block followed by a real table remains accepted. The full focused suite passed one checker wiring case and 118 standalone-gate cases. |
|
|
@@ -120,24 +120,47 @@ rescue Errno::ENOENT
|
|
|
120
120
|
[nil, "claude_not_found"]
|
|
121
121
|
end
|
|
122
122
|
|
|
123
|
-
# Parse a stream-json transcript:
|
|
123
|
+
# Parse a stream-json transcript: observed tools plus a validated terminal result.
|
|
124
124
|
def parse_transcript(stream)
|
|
125
125
|
skills = []
|
|
126
126
|
commands = []
|
|
127
|
+
results = []
|
|
127
128
|
stream.each_line do |line|
|
|
128
|
-
|
|
129
|
+
next if line.strip.empty?
|
|
130
|
+
ev = JSON.parse(line)
|
|
131
|
+
return [skills.uniq, commands, "invalid_stream_event"] unless ev.is_a?(Hash)
|
|
132
|
+
results << ev if ev["type"] == "result"
|
|
129
133
|
next unless ev["type"] == "assistant"
|
|
130
|
-
|
|
134
|
+
message = ev["message"]
|
|
135
|
+
return [skills.uniq, commands, "invalid_stream_event"] unless message.is_a?(Hash) && message["content"].is_a?(Array)
|
|
136
|
+
message["content"].each do |c|
|
|
137
|
+
return [skills.uniq, commands, "invalid_stream_event"] unless c.is_a?(Hash)
|
|
131
138
|
next unless c["type"] == "tool_use"
|
|
139
|
+
input = c["input"]
|
|
140
|
+
return [skills.uniq, commands, "invalid_stream_event"] unless input.is_a?(Hash)
|
|
132
141
|
if c["name"] == "Skill"
|
|
133
|
-
s = (
|
|
142
|
+
s = (input["skill"] || input["command"]).to_s
|
|
134
143
|
skills << s.split(":").last unless s.empty?
|
|
135
144
|
elsif c["name"] == "Bash"
|
|
136
|
-
commands <<
|
|
145
|
+
commands << input["command"].to_s
|
|
137
146
|
end
|
|
138
147
|
end
|
|
139
148
|
end
|
|
140
|
-
[skills.uniq, commands]
|
|
149
|
+
return [skills.uniq, commands, "missing_success_result"] if results.empty?
|
|
150
|
+
return [skills.uniq, commands, "invalid_terminal_result"] unless results.size == 1
|
|
151
|
+
terminal = results.first
|
|
152
|
+
unless terminal["subtype"] == "success"
|
|
153
|
+
return [skills.uniq, commands, "result_#{terminal['subtype']}"]
|
|
154
|
+
end
|
|
155
|
+
unless [nil, false].include?(terminal["is_error"]) &&
|
|
156
|
+
[nil, [], {}].include?(terminal["permission_denials"]) &&
|
|
157
|
+
[nil, 0, "0"].include?(terminal["api_error_status"]) &&
|
|
158
|
+
[nil, "completed"].include?(terminal["terminal_reason"])
|
|
159
|
+
return [skills.uniq, commands, "invalid_terminal_result"]
|
|
160
|
+
end
|
|
161
|
+
[skills.uniq, commands, nil]
|
|
162
|
+
rescue JSON::ParserError, JSON::NestingError
|
|
163
|
+
[skills.uniq, commands, "invalid_stream_event"]
|
|
141
164
|
end
|
|
142
165
|
|
|
143
166
|
if dry_run
|
|
@@ -154,7 +177,8 @@ results = traces.map do |t|
|
|
|
154
177
|
frozen_ref = t["frozen_at_sha"] == "root" ? `git -C #{Shellwords.escape(root)} rev-list --max-parents=0 HEAD`.lines.first.to_s.strip : t["frozen_at_sha"]
|
|
155
178
|
frozen_ok = ancestor?(root, frozen_ref)
|
|
156
179
|
stream, error = run_agent(root, max_turns, timeout_s, t["trigger_prompt"])
|
|
157
|
-
invoked, commands = error ? [[], []] : parse_transcript(stream)
|
|
180
|
+
invoked, commands, stream_error = error ? [[], [], nil] : parse_transcript(stream)
|
|
181
|
+
error ||= stream_error
|
|
158
182
|
a = t["assert"] || {}
|
|
159
183
|
missing = (a["must_invoke_skill"] || []) - invoked
|
|
160
184
|
forbidden_hit = (a["must_not_invoke_skill"] || []) & invoked
|