tribunal-kit 4.5.0 → 4.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agent/.shared/ui-ux-pro-max/README.md +4 -4
- package/.agent/ARCHITECTURE.md +279 -277
- package/.agent/GEMINI.md +127 -121
- package/.agent/agents/accessibility-reviewer.md +187 -187
- package/.agent/agents/ai-code-reviewer.md +199 -199
- package/.agent/agents/api-architect.md +71 -66
- package/.agent/agents/backend-specialist.md +219 -215
- package/.agent/agents/cloud-engineer.md +98 -0
- package/.agent/agents/code-archaeologist.md +168 -161
- package/.agent/agents/database-architect.md +184 -184
- package/.agent/agents/db-latency-auditor.md +213 -216
- package/.agent/agents/debugger.md +198 -191
- package/.agent/agents/dependency-reviewer.md +106 -103
- package/.agent/agents/devops-engineer.md +218 -218
- package/.agent/agents/documentation-writer.md +209 -201
- package/.agent/agents/explorer-agent.md +167 -160
- package/.agent/agents/frontend-reviewer.md +162 -160
- package/.agent/agents/frontend-specialist.md +257 -248
- package/.agent/agents/game-developer.md +48 -48
- package/.agent/agents/logic-reviewer.md +118 -116
- package/.agent/agents/mobile-developer.md +197 -200
- package/.agent/agents/mobile-reviewer.md +159 -162
- package/.agent/agents/orchestrator.md +187 -181
- package/.agent/agents/penetration-tester.md +160 -157
- package/.agent/agents/performance-optimizer.md +183 -183
- package/.agent/agents/performance-reviewer.md +178 -178
- package/.agent/agents/precedence-reviewer.md +251 -250
- package/.agent/agents/product-manager.md +149 -142
- package/.agent/agents/product-owner.md +81 -80
- package/.agent/agents/project-planner.md +152 -142
- package/.agent/agents/qa-automation-engineer.md +216 -225
- package/.agent/agents/resilience-reviewer.md +88 -88
- package/.agent/agents/schema-reviewer.md +67 -67
- package/.agent/agents/security-auditor.md +180 -174
- package/.agent/agents/seo-specialist.md +188 -193
- package/.agent/agents/sql-reviewer.md +159 -161
- package/.agent/agents/supervisor-agent.md +173 -184
- package/.agent/agents/swarm-worker-contracts.md +170 -166
- package/.agent/agents/swarm-worker-registry.md +92 -92
- package/.agent/agents/system-architect.md +85 -0
- package/.agent/agents/test-coverage-reviewer.md +158 -160
- package/.agent/agents/test-engineer.md +118 -118
- package/.agent/agents/throughput-optimizer.md +291 -299
- package/.agent/agents/type-safety-reviewer.md +182 -175
- package/.agent/agents/ui-ux-auditor.md +300 -292
- package/.agent/agents/vitals-reviewer.md +223 -223
- package/.agent/mcp_config.json +37 -40
- package/.agent/patterns/generator.md +11 -9
- package/.agent/patterns/inversion.md +14 -12
- package/.agent/patterns/pipeline.md +11 -9
- package/.agent/patterns/reviewer.md +15 -13
- package/.agent/patterns/tool-wrapper.md +11 -9
- package/.agent/routing_index.json +654 -0
- package/.agent/rules/GEMINI.md +358 -352
- package/.agent/scripts/compile_router.py +112 -0
- package/.agent/scripts/migrate_skills_frontmatter.py +64 -0
- package/.agent/scripts/strengthen_skills.js +1 -1
- package/.agent/skills/advanced-rag-pipelines/SKILL.md +56 -0
- package/.agent/skills/agent-organizer/SKILL.md +156 -150
- package/.agent/skills/agentic-patterns/SKILL.md +313 -315
- package/.agent/skills/ai-prompt-injection-defense/SKILL.md +190 -184
- package/.agent/skills/api-patterns/SKILL.md +253 -247
- package/.agent/skills/api-security-auditor/SKILL.md +195 -193
- package/.agent/skills/app-builder/SKILL.md +573 -572
- package/.agent/skills/app-builder/templates/SKILL.md +108 -115
- package/.agent/skills/app-builder/templates/astro-static/TEMPLATE.md +76 -76
- package/.agent/skills/app-builder/templates/chrome-extension/TEMPLATE.md +92 -92
- package/.agent/skills/app-builder/templates/cli-tool/TEMPLATE.md +88 -88
- package/.agent/skills/app-builder/templates/electron-desktop/TEMPLATE.md +88 -88
- package/.agent/skills/app-builder/templates/express-api/TEMPLATE.md +83 -83
- package/.agent/skills/app-builder/templates/flutter-app/TEMPLATE.md +90 -90
- package/.agent/skills/app-builder/templates/monorepo-turborepo/TEMPLATE.md +90 -90
- package/.agent/skills/app-builder/templates/nextjs-fullstack/TEMPLATE.md +126 -122
- package/.agent/skills/app-builder/templates/nextjs-saas/TEMPLATE.md +127 -122
- package/.agent/skills/app-builder/templates/nextjs-static/TEMPLATE.md +172 -169
- package/.agent/skills/app-builder/templates/nuxt-app/TEMPLATE.md +139 -134
- package/.agent/skills/app-builder/templates/python-fastapi/TEMPLATE.md +83 -83
- package/.agent/skills/app-builder/templates/react-native-app/TEMPLATE.md +122 -119
- package/.agent/skills/appflow-wireframe/SKILL.md +146 -145
- package/.agent/skills/architecture/SKILL.md +226 -219
- package/.agent/skills/authentication-best-practices/SKILL.md +197 -189
- package/.agent/skills/backend-security-expert/SKILL.md +16 -2
- package/.agent/skills/bash-linux/SKILL.md +179 -179
- package/.agent/skills/behavioral-modes/SKILL.md +239 -223
- package/.agent/skills/brainstorming/SKILL.md +498 -486
- package/.agent/skills/browser-native-ai/SKILL.md +57 -4
- package/.agent/skills/building-native-ui/SKILL.md +202 -202
- package/.agent/skills/cicd-pro/SKILL.md +442 -0
- package/.agent/skills/clean-code/SKILL.md +400 -381
- package/.agent/skills/cloud-architect/SKILL.md +439 -0
- package/.agent/skills/code-review-checklist/SKILL.md +203 -194
- package/.agent/skills/config-validator/SKILL.md +165 -165
- package/.agent/skills/containerization-pro/SKILL.md +452 -0
- package/.agent/skills/csharp-developer/SKILL.md +518 -518
- package/.agent/skills/data-validation-schemas/SKILL.md +333 -328
- package/.agent/skills/database-design/SKILL.md +247 -240
- package/.agent/skills/deployment-procedures/SKILL.md +172 -169
- package/.agent/skills/devops-engineer/SKILL.md +345 -345
- package/.agent/skills/devops-incident-responder/SKILL.md +143 -137
- package/.agent/skills/doc.md +209 -177
- package/.agent/skills/documentation-templates/SKILL.md +291 -279
- package/.agent/skills/edge-computing/SKILL.md +183 -181
- package/.agent/skills/error-resilience/SKILL.md +411 -428
- package/.agent/skills/extract-design-system/SKILL.md +160 -158
- package/.agent/skills/framer-motion-expert/SKILL.md +253 -244
- package/.agent/skills/frontend-design/SKILL.md +208 -201
- package/.agent/skills/frontend-security-expert/SKILL.md +16 -3
- package/.agent/skills/game-design-expert/SKILL.md +132 -129
- package/.agent/skills/game-engineering-expert/SKILL.md +148 -146
- package/.agent/skills/generative-ui-expert/SKILL.md +57 -1
- package/.agent/skills/geo-fundamentals/SKILL.md +148 -147
- package/.agent/skills/git-pro/SKILL.md +435 -0
- package/.agent/skills/github-operations/SKILL.md +335 -329
- package/.agent/skills/gsap-core/SKILL.md +319 -308
- package/.agent/skills/gsap-frameworks/SKILL.md +213 -207
- package/.agent/skills/gsap-performance/SKILL.md +139 -133
- package/.agent/skills/gsap-plugins/SKILL.md +486 -480
- package/.agent/skills/gsap-react/SKILL.md +202 -189
- package/.agent/skills/gsap-scrolltrigger/SKILL.md +357 -350
- package/.agent/skills/gsap-timeline/SKILL.md +165 -161
- package/.agent/skills/gsap-utils/SKILL.md +344 -338
- package/.agent/skills/harness-protocol/SKILL.md +48 -0
- package/.agent/skills/i18n-localization/SKILL.md +174 -163
- package/.agent/skills/intelligent-routing/SKILL.md +202 -246
- package/.agent/skills/knowledge-graph/SKILL.md +60 -52
- package/.agent/skills/lint-and-validate/SKILL.md +261 -261
- package/.agent/skills/llm-engineering/SKILL.md +400 -394
- package/.agent/skills/local-first/SKILL.md +178 -178
- package/.agent/skills/mcp-builder/SKILL.md +143 -142
- package/.agent/skills/mobile-design/SKILL.md +272 -263
- package/.agent/skills/monorepo-management/SKILL.md +335 -334
- package/.agent/skills/motion-engineering/SKILL.md +266 -234
- package/.agent/skills/nextjs-react-expert/SKILL.md +236 -234
- package/.agent/skills/nodejs-best-practices/SKILL.md +547 -548
- package/.agent/skills/observability/SKILL.md +343 -343
- package/.agent/skills/parallel-agents/SKILL.md +143 -146
- package/.agent/skills/performance-profiling/SKILL.md +259 -267
- package/.agent/skills/plan-writing/SKILL.md +150 -142
- package/.agent/skills/platform-engineer/SKILL.md +148 -147
- package/.agent/skills/playwright-best-practices/SKILL.md +188 -187
- package/.agent/skills/powershell-windows/SKILL.md +162 -162
- package/.agent/skills/project-idioms/SKILL.md +137 -137
- package/.agent/skills/python-patterns/SKILL.md +260 -259
- package/.agent/skills/python-pro/SKILL.md +324 -323
- package/.agent/skills/react-specialist/SKILL.md +305 -277
- package/.agent/skills/readme-builder/SKILL.md +310 -300
- package/.agent/skills/realtime-patterns/SKILL.md +323 -319
- package/.agent/skills/red-team-tactics/SKILL.md +231 -218
- package/.agent/skills/rust-pro/SKILL.md +671 -673
- package/.agent/skills/seo-fundamentals/SKILL.md +179 -179
- package/.agent/skills/server-management/SKILL.md +218 -214
- package/.agent/skills/shadcn-ui-expert/SKILL.md +231 -231
- package/.agent/skills/skill-creator/SKILL.md +87 -86
- package/.agent/skills/sql-pro/SKILL.md +629 -629
- package/.agent/skills/supabase-postgres-best-practices/SKILL.md +97 -97
- package/.agent/skills/swiftui-expert/SKILL.md +204 -201
- package/.agent/skills/system-design-pro/SKILL.md +345 -0
- package/.agent/skills/systematic-debugging/SKILL.md +153 -142
- package/.agent/skills/tailwind-patterns/SKILL.md +610 -566
- package/.agent/skills/tdd-workflow/SKILL.md +169 -161
- package/.agent/skills/test-result-analyzer/SKILL.md +313 -309
- package/.agent/skills/testing-patterns/SKILL.md +566 -579
- package/.agent/skills/trend-researcher/SKILL.md +243 -237
- package/.agent/skills/typescript-advanced/SKILL.md +336 -335
- package/.agent/skills/ui-ux-pro-max/SKILL.md +590 -562
- package/.agent/skills/ui-ux-researcher/SKILL.md +244 -244
- package/.agent/skills/vue-expert/SKILL.md +294 -275
- package/.agent/skills/vulnerability-scanner/SKILL.md +416 -404
- package/.agent/skills/web-accessibility-auditor/SKILL.md +219 -218
- package/.agent/skills/web-design-guidelines/SKILL.md +192 -186
- package/.agent/skills/webapp-testing/SKILL.md +167 -169
- package/.agent/skills/webgpu-performance/SKILL.md +56 -2
- package/.agent/skills/whimsy-injector/SKILL.md +346 -325
- package/.agent/skills/workflow-optimizer/SKILL.md +231 -229
- package/.agent/workflows/acf.md +141 -0
- package/.agent/workflows/api-tester.md +176 -151
- package/.agent/workflows/audit.md +150 -127
- package/.agent/workflows/brainstorm.md +134 -110
- package/.agent/workflows/changelog.md +140 -112
- package/.agent/workflows/create.md +168 -124
- package/.agent/workflows/debug.md +190 -165
- package/.agent/workflows/deploy.md +201 -180
- package/.agent/workflows/enhance.md +154 -128
- package/.agent/workflows/fix.md +136 -114
- package/.agent/workflows/generate.md +198 -183
- package/.agent/workflows/marathon.md +37 -11
- package/.agent/workflows/migrate.md +184 -160
- package/.agent/workflows/orchestrate.md +192 -168
- package/.agent/workflows/performance-benchmarker.md +135 -114
- package/.agent/workflows/plan.md +196 -173
- package/.agent/workflows/preview.md +103 -80
- package/.agent/workflows/refactor.md +192 -161
- package/.agent/workflows/review-ai.md +125 -101
- package/.agent/workflows/review.md +141 -116
- package/.agent/workflows/session.md +122 -94
- package/.agent/workflows/status.md +101 -79
- package/.agent/workflows/strengthen-skills.md +164 -138
- package/.agent/workflows/super-prompt.md +24 -0
- package/.agent/workflows/swarm.md +193 -179
- package/.agent/workflows/test.md +211 -189
- package/.agent/workflows/tribunal-backend.md +136 -105
- package/.agent/workflows/tribunal-database.md +122 -95
- package/.agent/workflows/tribunal-frontend.md +221 -96
- package/.agent/workflows/tribunal-full.md +129 -100
- package/.agent/workflows/tribunal-mobile.md +122 -95
- package/.agent/workflows/tribunal-performance.md +136 -110
- package/.agent/workflows/tribunal-speed.md +209 -183
- package/.agent/workflows/ui-ux-pro-max.md +145 -122
- package/README.md +107 -55
- package/bin/mcp-server.js +159 -0
- package/bin/tribunal-kit.js +105 -29
- package/bin/wrapper.js +16 -7
- package/mcp_config.json +9 -0
- package/package.json +94 -86
- package/scripts/changelog.js +4 -3
- package/scripts/validate-payload.js +6 -1
- package/scripts/postinstall.js +0 -127
|
@@ -1,313 +1,315 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: test-result-analyzer
|
|
3
|
-
description: Ingests test logs and identifies root causes across multiple failing test files. Provides actionable fix recommendations.
|
|
4
|
-
skills:
|
|
5
|
-
- systematic-debugging
|
|
6
|
-
- testing-patterns
|
|
7
|
-
version: 1.0.0
|
|
8
|
-
last-updated: 2026-03-12
|
|
9
|
-
applies-to-model: gemini-2.5-pro, claude-3-7-sonnet
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
-
|
|
22
|
-
- When
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
│
|
|
32
|
-
▼
|
|
33
|
-
|
|
34
|
-
│
|
|
35
|
-
▼
|
|
36
|
-
|
|
37
|
-
│
|
|
38
|
-
▼
|
|
39
|
-
|
|
40
|
-
│
|
|
41
|
-
▼
|
|
42
|
-
|
|
43
|
-
│
|
|
44
|
-
▼
|
|
45
|
-
|
|
46
|
-
│
|
|
47
|
-
▼
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
|
59
|
-
|
|
|
60
|
-
|
|
|
61
|
-
|
|
|
62
|
-
|
|
|
63
|
-
|
|
|
64
|
-
|
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
```
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
→
|
|
203
|
-
→
|
|
204
|
-
|
|
205
|
-
Step
|
|
206
|
-
→ Expected:
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
1. Check if dev server / database is running
|
|
220
|
-
2. Check .env.test for missing variables
|
|
221
|
-
3. Check node_modules exists (run npm install)
|
|
222
|
-
4. Check for breaking dependency upgrade in recent commits
|
|
223
|
-
```
|
|
224
|
-
|
|
225
|
-
### Flaky Tests
|
|
226
|
-
```
|
|
227
|
-
If same test passes on retry → flaky:
|
|
228
|
-
1. Check for shared mutable state between tests
|
|
229
|
-
2. Check for time-dependent assertions
|
|
230
|
-
3. Check for unresolved promises / async leaks
|
|
231
|
-
4. Check for network-dependent tests without mocks
|
|
232
|
-
```
|
|
233
|
-
|
|
234
|
-
### Only Snapshot Tests Fail
|
|
235
|
-
```
|
|
236
|
-
If only snapshot tests fail → likely intentional UI change:
|
|
237
|
-
1. Review snapshot diffs
|
|
238
|
-
2. If changes are expected: run with --updateSnapshot
|
|
239
|
-
3. If changes are unexpected: check for unintended CSS/component changes
|
|
240
|
-
```
|
|
241
|
-
|
|
242
|
-
## Cross-Skill Integration
|
|
243
|
-
|
|
244
|
-
|Paired Skill|Integration Point|
|
|
245
|
-
|---|---|
|
|
246
|
-
|`systematic-debugging`|Escalate when FPF is unclear → 4-phase debug methodology|
|
|
247
|
-
|`testing-patterns`|Reference when recommending test structure improvements|
|
|
248
|
-
|`workflow-optimizer`|Flag inefficient test-debug-retest loops|
|
|
249
|
-
|
|
250
|
-
## Anti-Hallucination Guard
|
|
251
|
-
|
|
252
|
-
- **Only analyze test output that was actually produced** — never generate fake test results.
|
|
253
|
-
- **Never invent file paths or line numbers** — only reference what appears in the stack trace.
|
|
254
|
-
- **Verify source files exist** before suggesting fixes — use `view_file` or `find_by_name`.
|
|
255
|
-
- **Mark uncertainty**: `// UNCERTAIN: log format not fully recognized, manual review recommended`.
|
|
256
|
-
- **Never guess at assertion values** — quote exactly what "Expected" and "Received" say in the output.
|
|
257
|
-
- **Don't assume test runner** — auto-detect from output format, don't assume Jest.
|
|
258
|
-
|
|
259
|
-
---
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
---
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
AI coding assistants often fall into specific bad habits when dealing with this domain. These are strictly forbidden:
|
|
267
|
-
|
|
268
|
-
1. **Over-engineering:** Proposing complex abstractions or distributed systems when a simpler approach suffices.
|
|
269
|
-
2. **Hallucinated Libraries/Methods:** Using non-existent methods or packages. Always `// VERIFY` or check `package.json` / `requirements.txt`.
|
|
270
|
-
3. **Skipping Edge Cases:** Writing the "happy path" and ignoring error handling, timeouts, or data validation.
|
|
271
|
-
4. **Context Amnesia:** Forgetting the user's constraints and offering generic advice instead of tailored solutions.
|
|
272
|
-
5. **Silent Degradation:** Catching and suppressing errors without logging or re-raising.
|
|
273
|
-
|
|
274
|
-
---
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
**Slash command: `/review` or `/tribunal-full`**
|
|
279
|
-
**Active reviewers: `logic-reviewer` · `security-auditor`**
|
|
280
|
-
|
|
281
|
-
### ❌ Forbidden AI Tropes
|
|
282
|
-
|
|
283
|
-
1. **Blind Assumptions:** Never make an assumption without documenting it clearly with `// VERIFY: [reason]`.
|
|
284
|
-
2. **Silent Degradation:** Catching and suppressing errors without logging or handling.
|
|
285
|
-
3. **Context Amnesia:** Forgetting the user's constraints and offering generic advice instead of tailored solutions.
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
Review these questions before confirming output:
|
|
290
|
-
```
|
|
291
|
-
✅ Did I rely ONLY on real, verified tools and methods?
|
|
292
|
-
✅ Is this solution appropriately scoped to the user's constraints?
|
|
293
|
-
✅ Did I handle potential failure modes and edge cases?
|
|
294
|
-
✅ Have I avoided generic boilerplate that doesn't add value?
|
|
295
|
-
```
|
|
296
|
-
|
|
297
|
-
### 🛑 Verification-Before-Completion (VBC) Protocol
|
|
298
|
-
|
|
299
|
-
**CRITICAL:** You must follow a strict "evidence-based closeout" state machine.
|
|
300
|
-
- ❌ **Forbidden:** Declaring a task complete because the output "looks correct."
|
|
301
|
-
- ✅ **Required:** You are explicitly forbidden from finalizing any task without providing **concrete evidence** (terminal output, passing tests, compile success, or equivalent proof) that your output works as intended.
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
## Pre-Flight Checklist
|
|
305
|
-
- [ ] Have I reviewed the user's specific constraints and requests?
|
|
306
|
-
- [ ] Have I checked the environment for relevant existing implementations?
|
|
307
|
-
|
|
308
|
-
## VBC Protocol (Verification-Before-Completion)
|
|
309
|
-
You MUST verify existing code signatures and variables before attempting to modify or call them. No hallucination is permitted.
|
|
1
|
+
---
|
|
2
|
+
name: test-result-analyzer
|
|
3
|
+
description: Ingests test logs and identifies root causes across multiple failing test files. Provides actionable fix recommendations.
|
|
4
|
+
skills:
|
|
5
|
+
- systematic-debugging
|
|
6
|
+
- testing-patterns
|
|
7
|
+
version: 1.0.0
|
|
8
|
+
last-updated: 2026-03-12
|
|
9
|
+
applies-to-model: gemini-2.5-pro, claude-3-7-sonnet
|
|
10
|
+
routing:
|
|
11
|
+
domain: general
|
|
12
|
+
tier: basic
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
# Test Result Analyzer Skill
|
|
16
|
+
|
|
17
|
+
You are a specialist in analyzing test output — not writing tests, but _understanding why tests fail_. You turn walls of red error text into a prioritized action plan.
|
|
18
|
+
|
|
19
|
+
## When to Activate
|
|
20
|
+
|
|
21
|
+
- After a test run with multiple failures.
|
|
22
|
+
- When the user says "tests are failing", "analyze test results", "what broke?", or "test failed".
|
|
23
|
+
- During CI/CD pipeline debugging.
|
|
24
|
+
- When `test_runner.py` or any test command exits with failures.
|
|
25
|
+
- When paired with `systematic-debugging` for deep root-cause investigation.
|
|
26
|
+
|
|
27
|
+
## Analysis Pipeline
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
Test output (terminal or log file)
|
|
31
|
+
│
|
|
32
|
+
▼
|
|
33
|
+
Runner detection — identify test framework from output format
|
|
34
|
+
│
|
|
35
|
+
▼
|
|
36
|
+
Failure extraction — parse each FAIL block into structured data
|
|
37
|
+
│
|
|
38
|
+
▼
|
|
39
|
+
Clustering — group failures by root module, error type, shared dependency
|
|
40
|
+
│
|
|
41
|
+
▼
|
|
42
|
+
FPF detection — find the First Point of Failure
|
|
43
|
+
│
|
|
44
|
+
▼
|
|
45
|
+
Dependency graph — map cascade relationships
|
|
46
|
+
│
|
|
47
|
+
▼
|
|
48
|
+
Fix recommendations — ordered by impact (most failures resolved first)
|
|
49
|
+
│
|
|
50
|
+
▼
|
|
51
|
+
Report — structured output with confidence levels
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Step 1: Runner Detection
|
|
55
|
+
|
|
56
|
+
Auto-detect the test framework from output patterns:
|
|
57
|
+
|
|
58
|
+
| Framework | Detection Pattern | Failure Marker |
|
|
59
|
+
| ----------- | ------------------------------------------------- | ----------------------------- |
|
|
60
|
+
| Jest | `PASS`/`FAIL` with file paths, `●` for test names | `FAIL src/...` |
|
|
61
|
+
| Vitest | `✓`/`×` markers, `FAIL` blocks | `❯ FAIL` or `× test name` |
|
|
62
|
+
| pytest | `PASSED`/`FAILED` with `::` separator | `FAILED tests/...::test_name` |
|
|
63
|
+
| Go test | `ok`/`FAIL` with package paths | `--- FAIL: TestName` |
|
|
64
|
+
| Mocha | `passing`/`failing` counts, indented suites | `N failing` section |
|
|
65
|
+
| JUnit (XML) | `<testsuite>` XML structure | `<failure>` elements |
|
|
66
|
+
| RSpec | `.F` markers, `Failures:` section | `Failure/Error:` |
|
|
67
|
+
| Cargo test | `test result: FAILED` | `---- test_name stdout ----` |
|
|
68
|
+
|
|
69
|
+
## Step 2: Failure Extraction
|
|
70
|
+
|
|
71
|
+
For each failure, extract a structured record:
|
|
72
|
+
|
|
73
|
+
```
|
|
74
|
+
{
|
|
75
|
+
test_name: "should return 401 for unauthenticated requests"
|
|
76
|
+
test_file: "src/api/auth.test.ts"
|
|
77
|
+
test_line: 42
|
|
78
|
+
error_type: "AssertionError"
|
|
79
|
+
expected: "401"
|
|
80
|
+
received: "200"
|
|
81
|
+
stack_trace: ["auth.test.ts:42", "auth.middleware.ts:18", "express/router.ts:..."]
|
|
82
|
+
source_files: ["auth.middleware.ts:18"] // files from YOUR codebase in the stack
|
|
83
|
+
}
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## Step 3: Failure Clustering
|
|
87
|
+
|
|
88
|
+
Group failures into clusters based on shared characteristics:
|
|
89
|
+
|
|
90
|
+
### Cluster Types
|
|
91
|
+
|
|
92
|
+
| Cluster Type | How to Detect | Typical Root Cause |
|
|
93
|
+
| ------------------- | ----------------------------------------------------- | ---------------------------------------- |
|
|
94
|
+
| **Shared Module** | Multiple tests import from the same file that changed | Missing export, type change, API change |
|
|
95
|
+
| **Same Error Type** | All failures throw `TypeError` or `ConnectionError` | Broken dependency, env issue |
|
|
96
|
+
| **Shared Fixture** | Tests using same `beforeEach`/setup fail together | Fixture setup failure cascading |
|
|
97
|
+
| **Import Chain** | Failures follow the import graph | Dependency that fails to resolve |
|
|
98
|
+
| **Environment** | All tests fail with connection/config errors | Missing env var, DB not running |
|
|
99
|
+
| **Timing** | Tests pass individually, fail together | Race condition, shared state |
|
|
100
|
+
| **Snapshot** | Multiple `toMatchSnapshot` failures | Intentional UI change (update snapshots) |
|
|
101
|
+
|
|
102
|
+
### Cascade Detection Algorithm
|
|
103
|
+
|
|
104
|
+
```
|
|
105
|
+
1. Sort failures by file path and execution order.
|
|
106
|
+
2. Find the FIRST failure in execution order → candidate FPF.
|
|
107
|
+
3. Check if the FPF's source file appears in other failures' import chains.
|
|
108
|
+
4. If yes → FPF is the root cause, other failures are cascades.
|
|
109
|
+
5. If no → failures are independent (multiple root causes).
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Step 4: First Point of Failure (FPF) Detection
|
|
113
|
+
|
|
114
|
+
The FPF is the most valuable finding — fix it first, and cascading failures resolve automatically.
|
|
115
|
+
|
|
116
|
+
```
|
|
117
|
+
Example:
|
|
118
|
+
12 test files fail.
|
|
119
|
+
11 of them import from `utils/auth.ts`.
|
|
120
|
+
The first failure is in `utils/auth.test.ts` at line 42.
|
|
121
|
+
Error: `generateToken is not exported from './auth'`
|
|
122
|
+
|
|
123
|
+
FPF: utils/auth.ts:42 — missing export
|
|
124
|
+
Cascade: 11 other test files fail because they can't import generateToken
|
|
125
|
+
Fix: Add `export { generateToken }` to utils/auth.ts
|
|
126
|
+
Expected resolution: 12 of 12 failures (100%)
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
**FPF Confidence Levels:**
|
|
130
|
+
|
|
131
|
+
| Confidence | Criteria |
|
|
132
|
+
| ---------- | -------------------------------------------------------- |
|
|
133
|
+
| **HIGH** | Same source file in >50% of failure stack traces |
|
|
134
|
+
| **MEDIUM** | Same error type across multiple test files |
|
|
135
|
+
| **LOW** | Failures appear independent, multiple root causes likely |
|
|
136
|
+
|
|
137
|
+
## Step 5: Fix Recommendations
|
|
138
|
+
|
|
139
|
+
For each cluster, provide actionable fixes:
|
|
140
|
+
|
|
141
|
+
| Fix Type | Example | How to Verify |
|
|
142
|
+
| --------------------- | ----------------------------------------------- | ------------------------------------- |
|
|
143
|
+
| **Missing Export** | `export { fn }` added to module | Re-run failing tests |
|
|
144
|
+
| **Type Mismatch** | Function signature changed, callers need update | Check callers with `grep_search` |
|
|
145
|
+
| **Stale Mock** | Mock doesn't match new interface | Compare mock to actual implementation |
|
|
146
|
+
| **Env Variable** | `.env.test` missing `DATABASE_URL` | Check `.env.example` vs `.env.test` |
|
|
147
|
+
| **Snapshot Update** | Intentional UI change | Run with `--updateSnapshot` flag |
|
|
148
|
+
| **Race Condition** | Tests share global state | Add isolation or `beforeEach` reset |
|
|
149
|
+
| **Dependency Update** | Package API changed after upgrade | Check changelog of updated package |
|
|
150
|
+
|
|
151
|
+
### Fix Priority Formula
|
|
152
|
+
|
|
153
|
+
```
|
|
154
|
+
Priority = (Tests_Resolved × 10) + (Confidence_Score × 5) - (Estimated_Fix_Time_Minutes)
|
|
155
|
+
|
|
156
|
+
Fix in this order:
|
|
157
|
+
1. Highest priority score first
|
|
158
|
+
2. If tied, prefer HIGH confidence
|
|
159
|
+
3. If still tied, prefer fewer files to change
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
## Report Format
|
|
163
|
+
|
|
164
|
+
```
|
|
165
|
+
━━━ Test Result Analysis ━━━━━━━━━━━━━━━━
|
|
166
|
+
|
|
167
|
+
Runner: [Jest / Vitest / pytest / Go / auto-detected]
|
|
168
|
+
Total: 48 tests across 12 files
|
|
169
|
+
Result: 36 passed | 12 failed | 0 skipped
|
|
170
|
+
Duration: 4.2s
|
|
171
|
+
Coverage: 78% statements (if available)
|
|
172
|
+
|
|
173
|
+
━━━ First Point of Failure ━━━━━━━━━━━━━━
|
|
174
|
+
|
|
175
|
+
📍 utils/auth.test.ts → line 42
|
|
176
|
+
Error: `generateToken` is not exported from `./auth`
|
|
177
|
+
Type: ImportError
|
|
178
|
+
Impact: Cascades to 11 other test files
|
|
179
|
+
|
|
180
|
+
This is the root cause. Fix this first.
|
|
181
|
+
|
|
182
|
+
━━━ Failure Clusters ━━━━━━━━━━━━━━━━━━━━
|
|
183
|
+
|
|
184
|
+
Cluster 1: Missing Export (11 tests, HIGH confidence)
|
|
185
|
+
Root: utils/auth.ts:42
|
|
186
|
+
Cascade: auth.test.ts, users.test.ts, sessions.test.ts, ...
|
|
187
|
+
Fix: Add `export { generateToken }` to auth.ts
|
|
188
|
+
Resolution: 11 of 12 failures (92%)
|
|
189
|
+
Priority: ★★★★★ (115 pts)
|
|
190
|
+
|
|
191
|
+
Cluster 2: Stale Mock (1 test, MEDIUM confidence)
|
|
192
|
+
Root: api/users.test.ts:98
|
|
193
|
+
Error: Expected { name, email, role } but received { name, email }
|
|
194
|
+
Fix: Add `role: "user"` to mock at line 15
|
|
195
|
+
Resolution: 1 of 12 failures (8%)
|
|
196
|
+
Priority: ★★☆☆☆ (20 pts)
|
|
197
|
+
|
|
198
|
+
━━━ Fix Plan ━━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
199
|
+
|
|
200
|
+
Step 1: Fix utils/auth.ts export
|
|
201
|
+
→ Expected: 11 failures resolved
|
|
202
|
+
→ Time: ~2 minutes
|
|
203
|
+
→ Run: npx jest utils/auth.test.ts (verify FPF fix)
|
|
204
|
+
|
|
205
|
+
Step 2: Update mock in api/users.test.ts:15
|
|
206
|
+
→ Expected: 1 failure resolved
|
|
207
|
+
→ Time: ~1 minute
|
|
208
|
+
|
|
209
|
+
Step 3: Re-run full suite
|
|
210
|
+
→ Expected: all 12 failures resolved (0 remaining)
|
|
211
|
+
|
|
212
|
+
━━━ Warnings ━━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
213
|
+
|
|
214
|
+
⚠️ No test coverage report detected. Consider adding --coverage flag.
|
|
215
|
+
⚠️ 3 test files have no assertions (test names end in `.todo`).
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
## Edge Cases
|
|
310
219
|
|
|
220
|
+
### All Tests Fail
|
|
221
|
+
|
|
222
|
+
```
|
|
223
|
+
If 100% of tests fail → likely environment issue, not code:
|
|
224
|
+
1. Check if dev server / database is running
|
|
225
|
+
2. Check .env.test for missing variables
|
|
226
|
+
3. Check node_modules exists (run npm install)
|
|
227
|
+
4. Check for breaking dependency upgrade in recent commits
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
### Flaky Tests
|
|
231
|
+
|
|
232
|
+
```
|
|
233
|
+
If same test passes on retry → flaky:
|
|
234
|
+
1. Check for shared mutable state between tests
|
|
235
|
+
2. Check for time-dependent assertions
|
|
236
|
+
3. Check for unresolved promises / async leaks
|
|
237
|
+
4. Check for network-dependent tests without mocks
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
### Only Snapshot Tests Fail
|
|
241
|
+
|
|
242
|
+
```
|
|
243
|
+
If only snapshot tests fail → likely intentional UI change:
|
|
244
|
+
1. Review snapshot diffs
|
|
245
|
+
2. If changes are expected: run with --updateSnapshot
|
|
246
|
+
3. If changes are unexpected: check for unintended CSS/component changes
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
## Cross-Skill Integration
|
|
250
|
+
|
|
251
|
+
| Paired Skill | Integration Point |
|
|
252
|
+
| ---------------------- | -------------------------------------------------------- |
|
|
253
|
+
| `systematic-debugging` | Escalate when FPF is unclear → 4-phase debug methodology |
|
|
254
|
+
| `testing-patterns` | Reference when recommending test structure improvements |
|
|
255
|
+
| `workflow-optimizer` | Flag inefficient test-debug-retest loops |
|
|
256
|
+
|
|
257
|
+
## Anti-Hallucination Guard
|
|
258
|
+
|
|
259
|
+
- **Only analyze test output that was actually produced** — never generate fake test results.
|
|
260
|
+
- **Never invent file paths or line numbers** — only reference what appears in the stack trace.
|
|
261
|
+
- **Verify source files exist** before suggesting fixes — use `view_file` or `find_by_name`.
|
|
262
|
+
- **Mark uncertainty**: `// UNCERTAIN: log format not fully recognized, manual review recommended`.
|
|
263
|
+
- **Never guess at assertion values** — quote exactly what "Expected" and "Received" say in the output.
|
|
264
|
+
- **Don't assume test runner** — auto-detect from output format, don't assume Jest.
|
|
265
|
+
|
|
266
|
+
---
|
|
267
|
+
|
|
268
|
+
---
|
|
269
|
+
|
|
270
|
+
AI coding assistants often fall into specific bad habits when dealing with this domain. These are strictly forbidden:
|
|
271
|
+
|
|
272
|
+
1. **Over-engineering:** Proposing complex abstractions or distributed systems when a simpler approach suffices.
|
|
273
|
+
2. **Hallucinated Libraries/Methods:** Using non-existent methods or packages. Always `// VERIFY` or check `package.json` / `requirements.txt`.
|
|
274
|
+
3. **Skipping Edge Cases:** Writing the "happy path" and ignoring error handling, timeouts, or data validation.
|
|
275
|
+
4. **Context Amnesia:** Forgetting the user's constraints and offering generic advice instead of tailored solutions.
|
|
276
|
+
5. **Silent Degradation:** Catching and suppressing errors without logging or re-raising.
|
|
277
|
+
|
|
278
|
+
---
|
|
279
|
+
|
|
280
|
+
**Slash command: `/review` or `/tribunal-full`**
|
|
281
|
+
**Active reviewers: `logic-reviewer` · `security-auditor`**
|
|
282
|
+
|
|
283
|
+
### ❌ Forbidden AI Tropes
|
|
284
|
+
|
|
285
|
+
1. **Blind Assumptions:** Never make an assumption without documenting it clearly with `// VERIFY: [reason]`.
|
|
286
|
+
2. **Silent Degradation:** Catching and suppressing errors without logging or handling.
|
|
287
|
+
3. **Context Amnesia:** Forgetting the user's constraints and offering generic advice instead of tailored solutions.
|
|
288
|
+
|
|
289
|
+
Review these questions before confirming output:
|
|
290
|
+
|
|
291
|
+
```
|
|
292
|
+
✅ Did I rely ONLY on real, verified tools and methods?
|
|
293
|
+
✅ Is this solution appropriately scoped to the user's constraints?
|
|
294
|
+
✅ Did I handle potential failure modes and edge cases?
|
|
295
|
+
✅ Have I avoided generic boilerplate that doesn't add value?
|
|
296
|
+
```
|
|
297
|
+
|
|
298
|
+
### 🛑 Verification-Before-Completion (VBC) Protocol
|
|
299
|
+
|
|
300
|
+
**CRITICAL:** You must follow a strict "evidence-based closeout" state machine.
|
|
301
|
+
|
|
302
|
+
- ❌ **Forbidden:** Declaring a task complete because the output "looks correct."
|
|
303
|
+
- ✅ **Required:** You are explicitly forbidden from finalizing any task without providing **concrete evidence** (terminal output, passing tests, compile success, or equivalent proof) that your output works as intended.
|
|
304
|
+
|
|
305
|
+
## Pre-Flight Checklist
|
|
306
|
+
|
|
307
|
+
- [ ] Have I reviewed the user's specific constraints and requests?
|
|
308
|
+
- [ ] Have I checked the environment for relevant existing implementations?
|
|
309
|
+
|
|
310
|
+
## VBC Protocol (Verification-Before-Completion)
|
|
311
|
+
|
|
312
|
+
You MUST verify existing code signatures and variables before attempting to modify or call them. No hallucination is permitted.
|
|
311
313
|
|
|
312
314
|
---
|
|
313
315
|
|
|
@@ -337,6 +339,7 @@ AI coding assistants often fall into specific bad habits when dealing with this
|
|
|
337
339
|
### ✅ Pre-Flight Self-Audit
|
|
338
340
|
|
|
339
341
|
Review these questions before confirming output:
|
|
342
|
+
|
|
340
343
|
```
|
|
341
344
|
✅ Did I rely ONLY on real, verified tools and methods?
|
|
342
345
|
✅ Is this solution appropriately scoped to the user's constraints?
|
|
@@ -347,5 +350,6 @@ Review these questions before confirming output:
|
|
|
347
350
|
### 🛑 Verification-Before-Completion (VBC) Protocol
|
|
348
351
|
|
|
349
352
|
**CRITICAL:** You must follow a strict "evidence-based closeout" state machine.
|
|
353
|
+
|
|
350
354
|
- ❌ **Forbidden:** Declaring a task complete because the output "looks correct."
|
|
351
355
|
- ✅ **Required:** You are explicitly forbidden from finalizing any task without providing **concrete evidence** (terminal output, passing tests, compile success, or equivalent proof) that your output works as intended.
|