universal-dev-standards 6.4.0 → 6.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/bundled/ai/standards/agent-dispatch.ai.yaml +162 -0
  2. package/bundled/ai/standards/ai-friendly-architecture.ai.yaml +1 -1
  3. package/bundled/ai/standards/ai-instruction-standards.ai.yaml +190 -15
  4. package/bundled/ai/standards/ai-response-navigation.ai.yaml +43 -3
  5. package/bundled/ai/standards/class-level-fix.ai.yaml +38 -3
  6. package/bundled/ai/standards/commit-message.ai.yaml +2 -0
  7. package/bundled/ai/standards/model-selection.ai.yaml +370 -72
  8. package/bundled/ai/standards/mutation-testing.ai.yaml +105 -2
  9. package/bundled/ai/standards/project-structure.ai.yaml +1 -1
  10. package/bundled/ai/standards/security-standards.ai.yaml +22 -1
  11. package/bundled/ai/standards/spec-driven-development.ai.yaml +86 -2
  12. package/bundled/ai/standards/test-governance.ai.yaml +49 -2
  13. package/bundled/ai/standards/testing.ai.yaml +49 -3
  14. package/bundled/ai/standards/translation-lifecycle-standards.ai.yaml +4 -4
  15. package/bundled/ai/standards/verification-evidence.ai.yaml +48 -4
  16. package/bundled/core/ai-response-navigation.md +75 -2
  17. package/bundled/core/class-level-fix.md +26 -3
  18. package/bundled/core/model-selection.md +383 -125
  19. package/bundled/core/mutation-testing.md +41 -2
  20. package/bundled/core/spec-driven-development.md +57 -2
  21. package/bundled/core/test-governance.md +22 -2
  22. package/bundled/core/translation-lifecycle-standards.md +6 -6
  23. package/bundled/core/verification-evidence.md +42 -3
  24. package/bundled/locales/zh-CN/CHANGELOG.md +24 -3
  25. package/bundled/locales/zh-CN/README.md +1 -1
  26. package/bundled/locales/zh-CN/SECURITY.md +1 -1
  27. package/bundled/locales/zh-CN/core/ai-response-navigation.md +69 -5
  28. package/bundled/locales/zh-CN/core/model-selection.md +375 -60
  29. package/bundled/locales/zh-CN/core/mutation-testing.md +1 -1
  30. package/bundled/locales/zh-CN/core/spec-driven-development.md +1 -1
  31. package/bundled/locales/zh-CN/core/test-governance.md +1 -1
  32. package/bundled/locales/zh-CN/core/translation-lifecycle-standards.md +1 -1
  33. package/bundled/locales/zh-CN/core/verification-evidence.md +1 -1
  34. package/bundled/locales/zh-CN/docs/CHEATSHEET.md +7 -12
  35. package/bundled/locales/zh-CN/docs/FEATURE-REFERENCE.md +10 -15
  36. package/bundled/locales/zh-TW/CHANGELOG.md +49 -3
  37. package/bundled/locales/zh-TW/README.md +1 -1
  38. package/bundled/locales/zh-TW/SECURITY.md +1 -1
  39. package/bundled/locales/zh-TW/core/ai-response-navigation.md +69 -5
  40. package/bundled/locales/zh-TW/core/class-level-fix.md +22 -7
  41. package/bundled/locales/zh-TW/core/model-selection.md +385 -47
  42. package/bundled/locales/zh-TW/core/mutation-testing.md +45 -6
  43. package/bundled/locales/zh-TW/core/spec-driven-development.md +1 -1
  44. package/bundled/locales/zh-TW/core/test-governance.md +22 -3
  45. package/bundled/locales/zh-TW/core/translation-lifecycle-standards.md +1 -1
  46. package/bundled/locales/zh-TW/core/verification-evidence.md +33 -6
  47. package/bundled/locales/zh-TW/docs/CHEATSHEET.md +7 -12
  48. package/bundled/locales/zh-TW/docs/FEATURE-REFERENCE.md +10 -15
  49. package/bundled/locales/zh-TW/integrations/claude-code/README.md +31 -5
  50. package/package.json +1 -1
  51. package/src/utils/reference-sync.js +83 -16
  52. package/standards-registry.json +20 -8
@@ -3,12 +3,19 @@
3
3
 
4
4
  id: mutation-testing
5
5
  meta:
6
- version: "1.0.0"
7
- updated: "2026-05-04"
6
+ version: "1.1.0"
7
+ updated: "2026-08-14"
8
8
  source: core/mutation-testing.md
9
9
  description: >
10
10
  Mutation testing methodology to evaluate test suite effectiveness.
11
11
  Answers "do my tests actually catch bugs?" beyond line coverage.
12
+ changelog:
13
+ - version: "1.1.0"
14
+ date: "2026-08-14"
15
+ change: >
16
+ Added attribution scope (kill is credited to the first failing test,
17
+ not to any one level), one-sided-invariant pairing requirement, and
18
+ equivalent-mutant classification requirement.
12
19
 
13
20
  # ─────────────────────────────────────────────────────────
14
21
  # Core Concepts
@@ -40,6 +47,67 @@ core_concepts:
40
47
  - category: Boolean literal
41
48
  examples: ["true → false", "false → true"]
42
49
 
50
+ # ─────────────────────────────────────────────────────────
51
+ # Attribution: kill is credited to the first failing test
52
+ # ─────────────────────────────────────────────────────────
53
+ attribution:
54
+ description: >
55
+ "Killed" means some test in the run failed against the mutant — most
56
+ tools do not record which test killed it, and none records which test
57
+ level. A 7/7 kill score therefore verifies the whole suite that ran
58
+ against the mutant, not any single test, and not any single test level
59
+ (unit vs. integration vs. property).
60
+ consequence: >
61
+ "The property suite verifies X" is not a claim an aggregate mutation run
62
+ supports. To support it, re-run mutation testing with only the property
63
+ suite active. If the isolated run kills fewer mutants than the aggregate
64
+ run, the gap is exactly what unit/integration tests were quietly covering.
65
+ isolated_rerun_example: "npx stryker run --mutate 'src/module/**' -- --project=property"
66
+ rule: >
67
+ High-risk modules (the same set the 80% threshold applies to —
68
+ auth/license/payment/security) must re-run mutants against the property
69
+ suite in isolation before any claim of "property-verified" is made.
70
+
71
+ # ─────────────────────────────────────────────────────────
72
+ # One-sided invariants miss fail-closed defects
73
+ # ─────────────────────────────────────────────────────────
74
+ one_sided_invariants:
75
+ description: >
76
+ A property like "output never exceeds the limit" is one-sided: it cannot
77
+ detect a mutant that makes the code fail closed (reject everything,
78
+ including valid input), because a fail-closed mutant never produces an
79
+ over-limit output. The mutation score looks unaffected while an entire
80
+ class of defect (denial of service, wrongly rejected requests) is
81
+ invisible to the suite.
82
+ fix: >
83
+ Pair every one-sided invariant with its opposite boundary — "never
84
+ exceeds the limit" needs a companion property such as "accepts everything
85
+ at or below the limit" — so both over-permissive and over-restrictive
86
+ mutants have a path to detection.
87
+
88
+ # ─────────────────────────────────────────────────────────
89
+ # Equivalent mutants are not survivors to chase
90
+ # ─────────────────────────────────────────────────────────
91
+ equivalent_mutants:
92
+ description: >
93
+ A surviving mutant is not automatically a test gap. Some mutants are
94
+ semantically equivalent to the original — no input can distinguish their
95
+ behavior — and no test can kill them, regardless of how it's written.
96
+ Forcing a kill with an assertion that exists only to move the score
97
+ (e.g. toBeDefined() on an incidental value) produces a hollow test
98
+ without closing any real gap.
99
+ required_classification:
100
+ - outcome: genuine_gap
101
+ action: Write a test that exercises the distinguishing behavior.
102
+ - outcome: equivalent
103
+ action: >
104
+ Record as "equivalent, because <reason>", stating the input(s)
105
+ checked to reach that conclusion.
106
+ note: >
107
+ An unclassified survivor is not the same as a classified-equivalent one.
108
+ Only a classified-equivalent mutant may be excluded from the score's
109
+ denominator.
110
+
43
111
  # ─────────────────────────────────────────────────────────
44
112
  # Tools
45
113
  # ─────────────────────────────────────────────────────────
@@ -175,12 +243,44 @@ rules:
175
243
  priority: required
176
244
  note: Reserve for pre-release gate and on-demand runs
177
245
 
246
+ - id: mutation-property-isolation-required
247
+ trigger: claiming a property-based test suite verifies specific behavior, based on an aggregate mutation run
248
+ instruction: >
249
+ Re-run mutation testing with only the property suite active before making
250
+ the claim. An aggregate kill score (property + unit + integration
251
+ together) verifies the whole suite, not the property suite alone.
252
+ Required for high-risk modules (auth/license/payment/security) before
253
+ any "property-verified" claim.
254
+ priority: required
255
+
256
+ - id: mutation-one-sided-invariant-pairing
257
+ trigger: writing or reviewing a property/invariant used as a mutation-testing target
258
+ instruction: >
259
+ A one-sided invariant (e.g. "never exceeds the limit") cannot detect a
260
+ fail-closed mutant. Pair it with an opposite-boundary property (e.g.
261
+ "accepts everything at or below the limit") so both over-permissive and
262
+ over-restrictive mutants have a path to detection.
263
+ priority: required
264
+
265
+ - id: mutation-equivalent-classification-required
266
+ trigger: reviewing a surviving mutant
267
+ instruction: >
268
+ Classify every reviewed survivor as either a genuine gap (write a test
269
+ exercising the distinguishing behavior) or "equivalent, because
270
+ <reason>" (record the input(s) checked). Do not force a kill with an
271
+ assertion that exists only to move the score. Only a
272
+ classified-equivalent mutant may be excluded from the denominator.
273
+ priority: required
274
+
178
275
  anti_patterns:
179
276
  - Treating 100% line coverage as sufficient (lines covered ≠ mutations killed)
180
277
  - Adding mutation testing to pre-commit hooks (makes commits 10-60 minutes long)
181
278
  - Accepting AI-generated tests without mutation score validation
182
279
  - Killing mutations by adding trivial assertions (expect(x).toBeDefined())
183
280
  - Targeting only happy paths in mutation testing (branches and boundaries are key)
281
+ - Claiming "the property suite verifies X" from an aggregate (not isolated) mutation run
282
+ - Using only a one-sided invariant for a property that has a fail-closed failure mode
283
+ - Forcing a kill on a semantically equivalent mutant instead of classifying it "equivalent, because <reason>"
184
284
 
185
285
  quick_reference:
186
286
  mutation_testing_checklist: |
@@ -190,3 +290,6 @@ quick_reference:
190
290
  □ Pre-release: run full mutation suite before tagging version
191
291
  □ AI-generated tests: validate with mutation score before accepting
192
292
  □ NOT in commit hooks (too slow)
293
+ □ High-risk modules: property suite re-run in isolation before "property-verified" claims
294
+ □ One-sided invariants paired with their opposite boundary
295
+ □ Every reviewed survivor classified: genuine gap, or "equivalent, because <reason>"
@@ -110,7 +110,7 @@ physical_spec:
110
110
  schema:
111
111
  root:
112
112
  required: ["src", "tests", "docs"]
113
- ignored: ["dist", "build", "node_modules", "out", "bin"]
113
+ ignored: ["dist", "build", "node_modules", "out", "bin", "__pycache__", ".venv", "venv", "target", "vendor", ".gradle", "pkg", ".dart_tool"]
114
114
  validator:
115
115
  command: "find . -maxdepth 1 -not -path '*/.*'"
116
116
  rule: "check_required_directories"
@@ -126,13 +126,34 @@ checklist:
126
126
  physical_spec:
127
127
  type: custom_script
128
128
  validator:
129
- command: "npm audit --dry-run > /dev/null 2>&1 || test -f .trivyignore || test -f .snyk"
129
+ command: >
130
+ if [ -f uds.project.yaml ] && grep -q 'security:' uds.project.yaml; then
131
+ exit 0;
132
+ elif [ -f Makefile ] && grep -q '^security:' Makefile; then
133
+ exit 0;
134
+ elif [ -f package.json ]; then
135
+ npm audit --dry-run > /dev/null 2>&1 || test -f .trivyignore || test -f .snyk;
136
+ elif [ -f requirements.txt ] || [ -f pyproject.toml ] || [ -f setup.py ]; then
137
+ pip-audit --dry-run > /dev/null 2>&1 || python -m safety check --dry-run > /dev/null 2>&1 || test -f .trivyignore || test -f .snyk;
138
+ elif [ -f go.mod ]; then
139
+ govulncheck ./... > /dev/null 2>&1 || test -f .trivyignore || test -f .snyk;
140
+ elif [ -f pom.xml ] || [ -f build.gradle ] || [ -f build.gradle.kts ]; then
141
+ test -f dependency-check-report.xml || test -f .trivyignore || test -f .snyk;
142
+ elif [ -f Cargo.toml ]; then
143
+ cargo audit > /dev/null 2>&1 || test -f .trivyignore || test -f .snyk;
144
+ elif [ -f Gemfile ]; then
145
+ bundle audit > /dev/null 2>&1 || test -f .trivyignore || test -f .snyk;
146
+ else
147
+ test -f .trivyignore || test -f .snyk;
148
+ fi
130
149
  rule: "security_scan_configured"
131
150
 
132
151
  enforcement:
133
152
  hook_type: PreToolUse
153
+ trigger: PreToolUse
134
154
  matcher:
135
155
  tool: Bash
136
156
  script_ref: "scripts/hooks/check-dangerous-cmd.js"
157
+ hook_script: "scripts/hooks/check-dangerous-cmd.js"
137
158
  severity: error
138
159
  timeout_ms: 500
@@ -3,8 +3,8 @@
3
3
 
4
4
  id: spec-driven-development
5
5
  meta:
6
- version: "1.2.0"
7
- updated: "2025-12-30"
6
+ version: "1.4.0"
7
+ updated: "2026-08-12"
8
8
  source: methodologies/guides/sdd-guide.md
9
9
  description: Spec-Driven Development workflow where documentation precedes implementation
10
10
 
@@ -53,6 +53,63 @@ principles:
53
53
  - Critical hotfixes (restore service immediately, document later)
54
54
  - Trivial changes (typos, comments, formatting)
55
55
 
56
+ spec_types:
57
+ description: SDD supports multiple granularity levels via spec-type field
58
+ default: feature
59
+ values:
60
+ - type: feature
61
+ description: Describes a single functional change (default, backward compatible)
62
+ granularity: function/feature level
63
+ template: standard SDD spec template
64
+
65
+ - type: agent
66
+ description: Describes an Agent role definition spanning multiple features
67
+ granularity: role level
68
+ template: agent_spec_template
69
+ required_sections:
70
+ - role_definition
71
+ - capability_scope
72
+ - interface_contract
73
+ - agent_interactions
74
+ - related_feature_specs
75
+
76
+ - type: infrastructure
77
+ description: Describes cross-cutting concerns or infrastructure changes
78
+ granularity: system level
79
+ template: standard SDD spec template with infrastructure focus
80
+
81
+ agent_spec_template:
82
+ sections:
83
+ - name: Role Definition
84
+ fields: [role_name, role_responsibility, autonomy_level]
85
+ autonomy_levels: "L1-L5 (per DEC-065)"
86
+
87
+ - name: Capability Scope
88
+ fields: [capabilities_list, out_of_scope]
89
+
90
+ - name: Interface Contract
91
+ subsections:
92
+ - name: Input
93
+ fields: [accepted_message_types, required_fields, optional_fields]
94
+ - name: Output
95
+ fields: [artifact_types, success_exit_condition, failure_exit_condition]
96
+
97
+ - name: Agent Interactions
98
+ fields: [upstream_agents, downstream_agents, parallel_agents]
99
+
100
+ - name: Related Feature SPECs
101
+ description: List of feature SPECs this agent participates in
102
+ format: "- [SPEC-NNN] description of this agent's role in that spec"
103
+
104
+ traceability:
105
+ feature_to_agent:
106
+ field: agent-id
107
+ description: Optional field in feature SPEC header referencing the owning Agent SPEC
108
+ example: "agent-id: SPEC-090"
109
+ agent_to_features:
110
+ section: Related Feature SPECs
111
+ description: Agent SPEC lists all feature SPECs it participates in
112
+
56
113
  spec_template:
57
114
  sections:
58
115
  - name: Summary
@@ -141,6 +198,33 @@ rules:
141
198
  instruction: Archive spec with links to commits/PRs
142
199
  priority: required
143
200
 
201
+ - id: SDD-AC-VERIFIED
202
+ name: An AC with no verification item is not an AC
203
+ name_zh: 沒有驗證項的 AC 不是 AC
204
+ severity: high
205
+ rule: >
206
+ Every acceptance criterion MUST have a verification item pointing at it —
207
+ a test, a check, a gate, or an explicitly recorded manual step. An AC that
208
+ no verification item references MUST be demoted to a design intent rather
209
+ than carried as an AC.
210
+ rule_zh: >
211
+ 每一條 AC 必須有指向它的驗證項(測試/檢查/閘門/明確記錄的手動步驟)。
212
+ 沒有任何驗證項引用的 AC 必須降級為設計意圖,不得繼續掛在 AC 欄。
213
+ rationale: >
214
+ An unverified AC does not fail loudly — it stops being true while the spec
215
+ continues to assert it. Measured instance (XSPEC-380): a spec's AC-7 required
216
+ a component be "fully preserved, no regression"; its Test Plan had seven items
217
+ and none pointed at AC-7. The component stopped running on the day the AC was
218
+ written and was found three months later by accident.
219
+ rationale_zh: >
220
+ 未經驗證的 AC 不會大聲失敗——它安靜地停止成立,而規格繼續宣稱它為真。
221
+ 實測案例:某規格的 AC-7 要求「完整保留、無回歸」,Test Plan 七項無一指向它;
222
+ 它保護的東西死在該 AC 被寫下的同一天,三個月後偶然才被發現。
223
+ anti_pattern: >
224
+ Treating "the reviewer will notice" as a verification item. A reviewer reads
225
+ the spec, and the spec says the AC holds. An AC is a claim about the world;
226
+ only something that touches the world can falsify it.
227
+
144
228
  best_practices:
145
229
  do:
146
230
  - Keep specs focused and atomic (one change per spec)
@@ -8,8 +8,16 @@ standard:
8
8
  description: Test policy, completion criteria, and environment management standards
9
9
 
10
10
  meta:
11
- version: "1.0.0"
12
- updated: "2026-03-11"
11
+ version: "1.2.0"
12
+ updated: "2026-08-14"
13
+ changelog:
14
+ - version: "1.2.0"
15
+ date: "2026-08-14"
16
+ change: >
17
+ Added fail-closed-threshold-gate rule: coverage/lint/mutation/any
18
+ bounded-metric check must exit non-zero when below threshold, using
19
+ the tool's own enforcement flag rather than a wrapper that only
20
+ prints the number and always exits 0.
13
21
  references:
14
22
  - "ISO/IEC/IEEE 29119-2 (Test Processes)"
15
23
  - "ISO/IEC/IEEE 29119-3 (Test Documentation)"
@@ -58,6 +66,36 @@ standard:
58
66
  - "eslint . (TypeScript/JavaScript)"
59
67
  - "mypy . (Python type checking)"
60
68
 
69
+ threshold_gates:
70
+ description: >
71
+ A measurement layer that prints a percentage but exits 0 regardless of
72
+ whether it cleared the threshold is a report, not a gate. It stays
73
+ green while the number it prints drifts downward across commits, and
74
+ nothing stops the next merge.
75
+ requirement: >
76
+ Every check with a pass/fail threshold (coverage, lint, mutation
77
+ score, or any other bounded metric) must translate "below threshold"
78
+ into a non-zero exit code via the tool's own enforcement flag — not a
79
+ wrapper script that re-parses printed output after the fact.
80
+ fail_closed_flags:
81
+ - tool: pytest-cov
82
+ flag: "--cov-fail-under=<N>"
83
+ - tool: coverage.py
84
+ flag: "coverage report --fail-under=<N>"
85
+ - tool: diff-cover
86
+ flag: "diff-cover coverage.xml --fail-under=<N>"
87
+ - tool: "nyc / Istanbul"
88
+ flag: "--check-coverage --lines <N>"
89
+ - tool: Stryker Mutator
90
+ flag: "thresholds.break in stryker.config.json"
91
+ - tool: ESLint
92
+ flag: "--max-warnings 0"
93
+ note: >
94
+ A wrapper that computes the number, prints it, and always exit 0
95
+ satisfies neither this rule nor verification-evidence's Evidence
96
+ Validity rule 1 — the tool's exit code no longer carries any
97
+ information about the artefact it measured.
98
+
61
99
  completion_criteria:
62
100
  description: >
63
101
  Test completion criteria define when testing activities can be considered done.
@@ -159,6 +197,15 @@ standard:
159
197
  BUG-A08 post-mortem (2026-04-20): 22 tests existed in UDS but were never
160
198
  executed by any CI gate, passing silently and masking real failures.
161
199
 
200
+ - id: fail-closed-threshold-gate
201
+ trigger: configuring or reviewing any coverage/lint/mutation/other threshold check
202
+ instruction: >
203
+ The check must use the tool's fail-under (or equivalent) enforcement
204
+ flag so it exits non-zero when the threshold is not met. A script that
205
+ only prints the number and always exits 0 is a report, not a gate, and
206
+ does not satisfy this rule.
207
+ priority: required
208
+
162
209
  - id: gate-wiring-required
163
210
  trigger: adding any quality detection script to the repository
164
211
  instruction: |
@@ -12,8 +12,8 @@ standard:
12
12
  - "See full-coverage-testing.ai.yaml for coverage policy (XSPEC-178)"
13
13
 
14
14
  meta:
15
- version: "2.1.0"
16
- updated: "2026-05-06"
15
+ version: "2.2.0"
16
+ updated: "2026-08-12"
17
17
  source: core/testing-standards.md
18
18
  guide: skills/testing-guide/testing-theory.md
19
19
  description: Testing structure and principles. Coverage policy moved to full-coverage-testing (XSPEC-178).
@@ -134,9 +134,55 @@ standard:
134
134
  feature page returned 500 silently — full E2E suite passed with false
135
135
  confidence, masking a production crash.
136
136
 
137
+ migration_testing:
138
+ schema_parity:
139
+ description: For any project migrating a database schema, include automated schema parity verification rather than relying on manually maintained documents.
140
+ pattern:
141
+ step_1: Query source schema metadata at test time (e.g., SQLite PRAGMA table_info, MySQL information_schema)
142
+ step_2: Query target schema metadata at test time (e.g., EF Core DbContext.Model.GetProperties(), direct information_schema query)
143
+ step_3: Assert every expected source column exists in target with compatible type, OR is documented as an intentional change (intentional_removals list)
144
+ ci_gate:
145
+ trigger: PR touches entity definitions or migration scripts
146
+ enforcement: required (blocking)
147
+ catches:
148
+ - Column renames not propagated to mapper
149
+ - Type changes (e.g. TEXT → NVARCHAR(50)) that silently truncate data
150
+ - Source columns added but never mapped to target
151
+ - Target columns removed but still referenced in source code
152
+ reference_implementation:
153
+ language: csharp
154
+ framework: EF Core vs SQLite
155
+ pattern: |
156
+ var properties = dbContext.Model.FindEntityType(typeof(Entity))!
157
+ .GetProperties().Select(p => p.GetColumnName()).ToHashSet();
158
+ var sourceColumns = conn.Query<string>(
159
+ "SELECT name FROM pragma_table_info('tablename')").ToHashSet();
160
+ var unmapped = sourceColumns.Except(knownRemovals).Except(properties);
161
+ Assert.Empty(unmapped);
162
+
137
163
  physical_spec:
138
164
  type: custom_script
139
165
  validator:
140
166
  command: >
141
- test -f vitest.config.ts || test -f vitest.config.js || test -f jest.config.js || grep -q '"test":' package.json
167
+ if [ -f uds.project.yaml ] && grep -q 'commands:' uds.project.yaml; then
168
+ exit 0;
169
+ elif [ -f Makefile ] && grep -q '^test:' Makefile; then
170
+ exit 0;
171
+ elif [ -f justfile ] && grep -q '^test:' justfile; then
172
+ exit 0;
173
+ elif [ -f package.json ]; then
174
+ test -f vitest.config.ts || test -f vitest.config.js || test -f jest.config.js || grep -q '"test":' package.json;
175
+ elif [ -f requirements.txt ] || [ -f pyproject.toml ] || [ -f setup.py ] || [ -f setup.cfg ]; then
176
+ python -m pytest --collect-only -q > /dev/null 2>&1 || test -f pytest.ini || test -f pyproject.toml;
177
+ elif [ -f go.mod ]; then
178
+ test -f go.sum || go list ./... > /dev/null 2>&1;
179
+ elif [ -f pom.xml ] || [ -f build.gradle ] || [ -f build.gradle.kts ]; then
180
+ test -f pom.xml || test -f build.gradle || test -f build.gradle.kts;
181
+ elif [ -f Cargo.toml ]; then
182
+ grep -q '\[\[test\]\]\|test = ' Cargo.toml || test -d tests;
183
+ elif [ -f Gemfile ]; then
184
+ test -f spec/spec_helper.rb || test -f test/test_helper.rb;
185
+ else
186
+ echo "⚠️ 請執行 uds configure 建立 uds.project.yaml" && exit 1;
187
+ fi
142
188
  rule: "test_runner_configured"
@@ -3,8 +3,8 @@
3
3
 
4
4
  id: translation-lifecycle-standards
5
5
  meta:
6
- version: "1.0.0"
7
- updated: "2026-04-20"
6
+ version: "1.0.1"
7
+ updated: "2026-08-12"
8
8
  status: trial
9
9
  since: "2026-04-20"
10
10
  expires: "2026-10-20"
@@ -69,7 +69,7 @@ automation:
69
69
  file: .githooks/pre-commit
70
70
  trigger: core/*.md files staged
71
71
  behavior: warn on OUTDATED, never block commit
72
- setup: ./scripts/install-hooks.sh
72
+ setup: node scripts/install-hooks.mjs
73
73
  release_gate:
74
74
  command: bash scripts/check-translation-sync.sh
75
75
  exit_1_conditions: [MISSING, MAJOR]
@@ -139,7 +139,7 @@ integration_points:
139
139
  role: primary automation; enforces severity levels
140
140
  - system: .githooks/pre-commit
141
141
  role: commit-time reminder when core/ files staged
142
- - system: bump-version.sh
142
+ - system: bump-version.mjs
143
143
  role: release-time advisory snapshot
144
144
  - system: pre-release-check.sh
145
145
  role: final gate before npm publish (calls check-translation-sync.sh)
@@ -8,14 +8,18 @@ standard:
8
8
  description: 驗證證據標準,強化 anti-hallucination
9
9
 
10
10
  meta:
11
- version: "1.2.0"
12
- updated: "2026-07-17"
11
+ version: "1.3.0"
12
+ updated: "2026-08-14"
13
13
  source: core/verification-evidence.md
14
14
  description: >
15
15
  驗證證據標準 — Iron Law: 無驗證證據不可聲稱完成。
16
16
  v1.1.0: Evidence must specify which environment layer it was collected from (XSPEC-204).
17
17
  v1.2.0: Evidence itself must be validated — a tool can fail silently and its
18
18
  output is indistinguishable from a real result (XSPEC-340).
19
+ v1.3.0: Evidence must postdate the last edit to what it verifies (VE-011,
20
+ stale evidence); a documented coverage gap must be registered in a dated
21
+ exception inventory, not left as prose alone (VE-012, narrow-coverage
22
+ registration — shared with class-level-fix).
19
23
  inspired_by: superpowers/verification-before-completion
20
24
 
21
25
  guidelines:
@@ -27,6 +31,8 @@ standard:
27
31
  - "代理報告 success ≠ 實際 success,需獨立驗證"
28
32
  - "**工具回報 success ≠ 實際 success**——驗證指令本身可能沒跑起來,而其輸出與真結果無法區分"
29
33
  - "驗證輸出截斷至合理長度(2000 字元)但保留關鍵資訊"
34
+ - "**證據必須晚於它所驗證對象的最後一次編輯**——最近一次執行剛好通過,不代表它驗證的是現狀(VE-011)"
35
+ - "**窄涵蓋必須登記,不能只揭露**——文件化的證據/涵蓋缺口須登記到帶到期日的例外清冊,單純寫下來不算滿足(VE-012)"
30
36
 
31
37
  # 以下不構成驗證證據。原本只存在於 zh-TW 譯文裡(英文來源與本檔皆無)——
32
38
  # v1.2.0 一併升上來源與本檔(XSPEC-340 R4:譯文比來源更完整,且無 gate 會報)。
@@ -67,12 +73,22 @@ standard:
67
73
  detail: >
68
74
  `set -o pipefail` 下 `producer | grep -q pattern` 會繼承 producer 的非 0,
69
75
  grep 中不中都無關。需要依內容決策時:**先接住輸出,再判斷**。
76
+ - name: "證據必須晚於它所驗證對象的最後一次編輯"
77
+ detail: >
78
+ 在受測程式碼、提示詞或設定的最近一次變更之前擷取的驗證執行,證明的是別的東西——
79
+ 不是正在被主張的那件事。陷阱是靜默的:完整套件先跑過且全綠,之後才編輯了某個東西
80
+ (提示詞、設定檔、門檻),編輯之後只跑了範圍更窄的檢查(lint、部分套件),
81
+ 結果卻拿更早的那次綠燈當作現狀的證據。**證據必須是最後一次編輯之後的單一次新鮮執行,
82
+ 不是「剛好通過的最近一次執行」。**(VE-011)
70
83
  provenance: >
71
- 證據來自 2026-07-17 單一 agent(Claude Opus 4.8)單日的十次實例,
84
+ 規則 1–4 的證據來自 2026-07-17 單一 agent(Claude Opus 4.8)單日的十次實例,
72
85
  記錄於 AsiaOstrich XSPEC-340;其中 exit_code=0 四次為假、exit_code≠0 兩次為真
73
86
  且依 VE-002 行動摧毀了健康的產物。樣本密集但來源單一——然而每個失敗都源自
74
87
  **工具本身的語意**(sudo / gpg / pipefail / POSIX exit code)而非模型的性質,
75
- 故任何驅動同一批工具的 agent 都暴露在同樣的陷阱下。
88
+ 故任何驅動同一批工具的 agent 都暴露在同樣的陷阱下。規則 5(VE-011)為獨立補強,
89
+ 非同一批樣本:其失效樣態為「完整套件跑過通過 → 之後改了提示詞/設定 →
90
+ 只跑範圍更窄的檢查(如 lint)就提交 → CI 紅在釘住新內容雜湊的測試」——
91
+ 通過的是「最後一次變更之前」的套件。
76
92
 
77
93
  evidence_format:
78
94
  fields:
@@ -117,6 +133,7 @@ standard:
117
133
  - "exit_code ≠ 0 → 標記為驗證失敗——**除非**該工具在受測狀態下設計上就回非 0(見 VE-007),此時改依輸出內容判斷"
118
134
  - "exit_code = 0 但該指令不可能量到它宣稱的東西 → 標記為未驗證"
119
135
  - "證據斷言「不存在」(0/空/查無)→ 未證明查詢工具執行成功前,標記為未驗證"
136
+ - "證據的時間戳早於它所驗證對象的最後一次編輯 → 標記為過期(stale),須在編輯後重跑,不得引用更早那次通過"
120
137
  - "多個驗證步驟 → 所有步驟都必須通過"
121
138
 
122
139
  rules:
@@ -174,6 +191,31 @@ standard:
174
191
  trigger: "證據的 exit_code 來自管線(尤其在 `set -o pipefail` 下)"
175
192
  action: "該 exit code 不歸屬於任何單一階段。先接住輸出,再依內容判斷。"
176
193
  priority: medium
194
+ - id: VE-011
195
+ trigger: "證據的執行時間戳早於它所宣稱驗證之產物的最後一次編輯"
196
+ action: >
197
+ 標記為過期(stale)——對現狀不成立為證據;要求在編輯後補一次單一次新鮮執行。
198
+ priority: high
199
+
200
+ # ── v1.3.0 窄涵蓋登記規則(B-01 借鑒)──
201
+ # 與 class-level-fix 的同名規則共用同一條要求:文件化的缺口若沒有登記到
202
+ # 帶到期日的例外清冊,揭露就會變成永久豁免。
203
+ narrow_coverage_registration:
204
+ description: "文件化的證據/涵蓋缺口(environment-stratification ⚠️/❌、或任何窄於其主張的檢查)必須登記,不能只揭露"
205
+ requirement: >
206
+ 任何這一類已記錄的缺口,須同時登記到一份帶到期日的例外清冊——獨立於標準本文或
207
+ matrix 條目本身之外——指名缺口、其成因、並附上覆核或到期日期。只有一則註腳而沒有
208
+ 對應清冊條目,不滿足此規則。
209
+ falsifiable_condition: >
210
+ 一則清冊條目連續兩期審查都未變動,代表該揭露已經變成逃生口,本條對該條目失效——
211
+ 不是「部分滿足」,是失效。清冊放在哪裡、什麼格式、多久審一次,留給採用專案自行決定;
212
+ 本標準只要求「有這麼一份東西存在,且條目會動」。
213
+ shared_with: class-level-fix
214
+ rules:
215
+ - id: VE-012
216
+ trigger: "文件化的證據/涵蓋缺口(environment-stratification ⚠️/❌,或任何窄於其主張的檢查)沒有登記到帶到期日的例外清冊"
217
+ action: "揭露本身不滿足此規則——要求一則登記過、帶到期日的清冊條目,指名缺口與其覆核/到期日期"
218
+ priority: required
177
219
 
178
220
  physical_spec:
179
221
  type: checklist
@@ -189,5 +231,7 @@ physical_spec:
189
231
  - "證據若斷言「不存在」,是否已證明查詢工具執行成功(VE-008)"
190
232
  - "判斷存在與否的指令是否抑制了 stderr(VE-009)"
191
233
  - "exit_code 是否來自管線而非受測指令本身(VE-010)"
234
+ - "證據的執行時間戳是否晚於它所驗證對象的最後一次編輯(VE-011)"
235
+ - "文件化的證據/涵蓋缺口是否登記於帶到期日的例外清冊,而非只有揭露文字(VE-012)"
192
236
  - "Bug fix 是否有 RED → GREEN 循環證據"
193
237
  - "有外部服務依賴的 AC 是否標明 environment_layer"