@phuc1403/musketeer 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/README.md +49 -49
  2. package/manifest.json +333 -301
  3. package/package.json +1 -1
  4. package/template/.claude/agents/code-reviewer.md +182 -166
  5. package/template/.claude/hooks/git-skill-reminder.cjs +53 -0
  6. package/template/.claude/hooks/lib/colors.cjs +180 -122
  7. package/template/.claude/hooks/lib/transcript-parser.cjs +300 -277
  8. package/template/.claude/skills/code-review/SKILL.md +201 -54
  9. package/template/.claude/skills/code-review/references/checklist-workflow.md +96 -0
  10. package/template/.claude/skills/code-review/references/checklists/api.md +52 -52
  11. package/template/.claude/skills/code-review/references/checklists/base.md +100 -100
  12. package/template/.claude/skills/code-review/references/checklists/web-app.md +54 -54
  13. package/template/.claude/skills/code-review/references/code-review-reception.md +113 -0
  14. package/template/.claude/skills/code-review/references/codebase-scan-workflow.md +30 -0
  15. package/template/.claude/skills/code-review/references/edge-case-scouting.md +119 -0
  16. package/template/.claude/skills/code-review/references/input-mode-resolution.md +135 -0
  17. package/template/.claude/skills/code-review/references/parallel-review-workflow.md +76 -0
  18. package/template/.claude/skills/code-review/references/requesting-code-review.md +116 -0
  19. package/template/.claude/skills/code-review/references/spec-compliance-review.md +43 -0
  20. package/template/.claude/skills/code-review/references/task-management-reviews.md +140 -0
  21. package/template/.claude/skills/code-review/references/verification-before-completion.md +139 -0
  22. package/template/.claude/skills/git/SKILL.md +131 -115
  23. package/template/.claude/skills/git/references/branch-management.md +88 -88
  24. package/template/.claude/skills/git/references/commit-standards.md +46 -46
  25. package/template/.claude/skills/git/references/context-efficiency.md +54 -0
  26. package/template/.claude/skills/git/references/gh-cli-guide.md +109 -109
  27. package/template/.claude/skills/git/references/safety-protocols.md +69 -69
  28. package/template/.claude/skills/git/references/workflow-commit.md +58 -58
  29. package/template/.claude/skills/git/references/workflow-merge-pr.md +136 -0
  30. package/template/.claude/skills/git/references/workflow-merge.md +48 -48
  31. package/template/.claude/skills/git/references/workflow-pr.md +58 -58
  32. package/template/.claude/skills/git/references/workflow-push.md +52 -52
  33. package/template/.claude/skills/skill-creator/LICENSE.txt +201 -201
  34. package/template/.claude/skills/skill-creator/SKILL.md +154 -149
  35. package/template/.claude/skills/skill-creator/agents/analyzer.md +274 -274
  36. package/template/.claude/skills/skill-creator/agents/comparator.md +202 -202
  37. package/template/.claude/skills/skill-creator/agents/grader.md +223 -223
  38. package/template/.claude/skills/skill-creator/assets/eval_review.html +146 -146
  39. package/template/.claude/skills/skill-creator/eval-viewer/generate_review.py +471 -471
  40. package/template/.claude/skills/skill-creator/eval-viewer/viewer.html +1325 -1325
  41. package/template/.claude/skills/skill-creator/references/benchmark-optimization-guide.md +86 -86
  42. package/template/.claude/skills/skill-creator/references/distribution-guide.md +79 -79
  43. package/template/.claude/skills/skill-creator/references/eval-infrastructure-guide.md +129 -129
  44. package/template/.claude/skills/skill-creator/references/eval-schemas.md +121 -121
  45. package/template/.claude/skills/skill-creator/references/mcp-skills-integration.md +71 -71
  46. package/template/.claude/skills/skill-creator/references/metadata-quality-criteria.md +94 -94
  47. package/template/.claude/skills/skill-creator/references/plugin-marketplace-hosting.md +104 -104
  48. package/template/.claude/skills/skill-creator/references/plugin-marketplace-overview.md +89 -89
  49. package/template/.claude/skills/skill-creator/references/plugin-marketplace-schema.md +93 -93
  50. package/template/.claude/skills/skill-creator/references/plugin-marketplace-sources.md +103 -103
  51. package/template/.claude/skills/skill-creator/references/plugin-marketplace-troubleshooting.md +76 -76
  52. package/template/.claude/skills/skill-creator/references/script-quality-criteria.md +106 -106
  53. package/template/.claude/skills/skill-creator/references/skill-anatomy-and-requirements.md +77 -77
  54. package/template/.claude/skills/skill-creator/references/skill-creation-workflow.md +152 -151
  55. package/template/.claude/skills/skill-creator/references/skill-design-patterns.md +75 -75
  56. package/template/.claude/skills/skill-creator/references/skillmark-benchmark-criteria.md +102 -102
  57. package/template/.claude/skills/skill-creator/references/structure-organization-criteria.md +114 -114
  58. package/template/.claude/skills/skill-creator/references/testing-and-iteration.md +78 -78
  59. package/template/.claude/skills/skill-creator/references/token-efficiency-criteria.md +74 -74
  60. package/template/.claude/skills/skill-creator/references/troubleshooting-guide.md +81 -81
  61. package/template/.claude/skills/skill-creator/references/validation-checklist.md +83 -83
  62. package/template/.claude/skills/skill-creator/references/writing-effective-instructions.md +88 -88
  63. package/template/.claude/skills/skill-creator/references/yaml-frontmatter-reference.md +92 -92
  64. package/template/.claude/skills/skill-creator/scripts/aggregate_benchmark.py +401 -401
  65. package/template/.claude/skills/skill-creator/scripts/encoding_utils.py +36 -36
  66. package/template/.claude/skills/skill-creator/scripts/generate_report.py +326 -326
  67. package/template/.claude/skills/skill-creator/scripts/improve_description.py +248 -248
  68. package/template/.claude/skills/skill-creator/scripts/init_skill.py +360 -360
  69. package/template/.claude/skills/skill-creator/scripts/package_skill.py +143 -143
  70. package/template/.claude/skills/skill-creator/scripts/quick_validate.py +110 -110
  71. package/template/.claude/skills/skill-creator/scripts/run_eval.py +310 -310
  72. package/template/.claude/skills/skill-creator/scripts/run_loop.py +332 -332
  73. package/template/.claude/skills/skill-creator/scripts/utils.py +47 -47
  74. package/template/.claude/statusline.cjs +0 -0
  75. package/template/.claude/skills/code-review/references/adversarial-review.md +0 -223
  76. /package/template/.claude/hooks/{usage-context-awareness.cjs → usage-quota-cache-refresh.cjs} +0 -0
@@ -1,54 +1,201 @@
1
- ---
2
- name: code-review
3
- description: "Review code quality with adversarial rigor. Supports input modes: pending changes, PR number, commit hash, codebase scan. Always-on red-team analysis finds security holes, false assumptions, and failure modes."
4
- argument-hint: "[#PR | COMMIT | --pending | codebase [parallel]]"
5
- metadata:
6
- author: claudekit
7
- version: "3.0.0"
8
- ---
9
-
10
- # Code Review
11
-
12
- Adversarial, evidence-based code review. Two stages: **Quality** then **Adversarial** (red-team). Be honest, brutal, concise. **YAGNI / KISS / DRY.** Verify before claiming; evidence before assertions.
13
-
14
- ## 1. Resolve the input
15
-
16
- Parse arguments, first match wins, then get the diff:
17
-
18
- | Argument | Mode | Diff command |
19
- |----------|------|--------------|
20
- | `#123` / PR URL | PR | `gh pr diff <n>` (+ `gh pr view <n> --json title,body,baseRefName` for intent) |
21
- | `[0-9a-f]{7,40}` | Commit | `git show <sha>` |
22
- | `--pending` | Pending | `git diff HEAD` (staged + unstaged) — ask user for intent |
23
- | *(none, changes in context)* | Default | recent changes already in context |
24
- | `codebase` / `codebase parallel` | Codebase | full-codebase scan (see below) |
25
- | *(none, no context)* | Prompt | `AskUserQuestion`: pending / PR / commit / codebase |
26
-
27
- Errors: PR not found → "PR #N not found"; bad SHA → "commit not found — is it pushed?"; empty `git diff HEAD` → "no pending changes". Ambiguous PR-vs-commit → prefer PR, note the assumption.
28
-
29
- Always review **added/modified lines** (`+` in diff). Pre-existing code is out of scope unless the change makes it newly broken.
30
-
31
- ## 2. Stage 1 — Quality review
32
-
33
- If the work implemented a plan/spec, first confirm compliance: list each requirement, mark PASS / MISSING / EXTRA. Missing requirements fail the review before quality matters — well-written code that doesn't match the spec is still wrong.
34
-
35
- Then review the diff for: correctness, standards, edge cases (null/empty/boundary/error paths), performance, and reuse/simplification. Dispatch a `code-reviewer` subagent for non-trivial diffs; review inline for small ones. Each finding: `file:line`, problem, fix.
36
-
37
- **Checklists (optional, for pre-landing / security audits):** detect project type — `package.json` with react/vue/next/etc → load `references/checklists/web-app.md`; `src/routes|api|controllers` → `references/checklists/api.md`; always load `references/checklists/base.md`. Run critical categories first (blocking), informational second. Honor the suppressions list at the bottom of `base.md`.
38
-
39
- ## 3. Stage 2 — Adversarial review (always-on)
40
-
41
- **Skip only when ALL true:** ≤2 files changed, ≤30 lines, no security-sensitive files (auth/crypto/input-parsing/SQL/env), no new dependencies. When skipped, note `Adversarial: skipped (below threshold)`.
42
-
43
- **Never skip** when auth/middleware/security/crypto, a lockfile, env vars, DB schema, or an API route changed.
44
-
45
- Run the red-team pass per **`references/adversarial-review.md`** — spawn an adversarial `code-reviewer` whose only job is to break the code (security holes, false assumptions, failure modes, races, data corruption, supply chain, observability gaps). Then adjudicate each finding **Accept / Reject / Defer** with a reason — no silent dismissals; benefit of the doubt goes to the adversary. Critical accepted findings block merge. On re-review, pass only the fix diff.
46
-
47
- ## Codebase modes
48
-
49
- - **`codebase`** — explore relevant files, dispatch parallel `code-reviewer` subagents by area, then run the always-on adversarial pass over the full scope. Report combined findings.
50
- - **`codebase parallel`** — first enumerate edge cases exhaustively (null, boundary, error, race, validation, security, leaks, untested paths), group into ≤6 categories, dispatch one `code-reviewer` per category to **verify** them, aggregate handled/unhandled, then run the adversarial pass.
51
-
52
- ## Bottom line
53
-
54
- Resolve what you're reviewing → quality pass → adversarial pass (scope-gated) → adjudicate → fix Critical, then claim only with verification evidence.
1
+ ---
2
+ name: code-review
3
+ description: "Review code quality with evidence-based rigor. Supports input modes: pending changes, PR number, commit hash, and codebase scan. Focuses on bugs, regressions, maintainability, reliability, and verification gaps."
4
+ user-invocable: true
5
+ when_to_use: "Invoke to review diffs, PRs, commits, or full codebases."
6
+ category: utilities
7
+ keywords: [review, quality, verification, reliability]
8
+ argument-hint: "[#PR | COMMIT | --pending | codebase [parallel]]"
9
+ metadata:
10
+ author: claudekit
11
+ version: "4.0.0"
12
+ ---
13
+
14
+ # Code Review
15
+
16
+ Production-readiness code review with technical rigor, evidence-based claims, and verification over performative responses. Reviews focus on production risks, regression paths, and whether the implementation matches the requested change.
17
+
18
+ ## Input Modes
19
+
20
+ Auto-detect from arguments. If ambiguous or no arguments, prompt via `AskUserQuestion`.
21
+
22
+ | Input | Mode | What Gets Reviewed |
23
+ |-------|------|--------------------|
24
+ | `#123` or PR URL | **PR** | Full PR diff fetched via `gh pr diff` |
25
+ | `abc1234` (7+ hex chars) | **Commit** | Single commit diff via `git show` |
26
+ | `--pending` | **Pending** | Staged + unstaged changes via `git diff` |
27
+ | *(no args, recent changes)* | **Default** | Recent changes in context |
28
+ | `codebase` | **Codebase** | Full codebase scan |
29
+ | `codebase parallel` | **Codebase+** | Parallel multi-reviewer audit |
30
+
31
+ **Resolution details:** `references/input-mode-resolution.md`
32
+
33
+ ### No Arguments
34
+
35
+ If invoked WITHOUT arguments and no recent changes in context, use `AskUserQuestion` with header "Review Target", question "What would you like to review?":
36
+
37
+ | Option | Description |
38
+ |--------|-------------|
39
+ | Pending changes | Review staged/unstaged git diff |
40
+ | Enter PR number | Fetch and review a specific PR |
41
+ | Enter commit hash | Review a specific commit |
42
+ | Full codebase scan | Deep codebase analysis |
43
+ | Parallel codebase audit | Multi-reviewer codebase scan |
44
+
45
+ ## Core Principle
46
+
47
+ **YAGNI**, **KISS**, **DRY** always. Technical correctness over social comfort.
48
+ **Be honest, be brutal, straight to the point, and be concise.**
49
+
50
+ Default assumption: reviewed code may be AI-assisted. Do not trust polished shape, confident comments, or happy-path tests. Verify behavior, project-rule compliance, and scope discipline from evidence.
51
+
52
+ No rubber-stamp reviews. The reviewer is not trying to please the author or preserve momentum; the reviewer enforces the rulebook and blocks defects, regressions, hidden scope drift, and AI-slop patterns.
53
+
54
+ Verify before implementing. Ask before assuming. Evidence before claims.
55
+
56
+ ## Practices
57
+
58
+ | Practice | When | Reference |
59
+ |----------|------|-----------|
60
+ | **Spec compliance** | After implementing from plan/spec, BEFORE quality review | `references/spec-compliance-review.md` |
61
+ | Receiving feedback | Unclear feedback, external reviewers, needs prioritization | `references/code-review-reception.md` |
62
+ | Requesting review | After tasks, before merge, stuck on problem | `references/requesting-code-review.md` |
63
+ | Verification gates | Before any completion claim, commit, PR | `references/verification-before-completion.md` |
64
+ | Edge case scouting | After implementation, before review | `references/edge-case-scouting.md` |
65
+ | **Checklist review** | Pre-landing, pre-merge, security audit | `references/checklist-workflow.md` |
66
+ | **Task-managed reviews** | Multi-file features (3+ files), parallel reviewers, fix cycles | `references/task-management-reviews.md` |
67
+
68
+ ## Quick Decision Tree
69
+
70
+ ```
71
+ SITUATION?
72
+ │
73
+ ├─ Input mode? → Resolve diff (references/input-mode-resolution.md)
74
+ │ ├─ #PR / URL → fetch PR diff
75
+ │ ├─ commit hash → git show
76
+ │ ├─ --pending → git diff (staged + unstaged)
77
+ │ ├─ codebase → full scan (references/codebase-scan-workflow.md)
78
+ │ ├─ codebase parallel → parallel audit (references/parallel-review-workflow.md)
79
+ │ └─ default → recent changes in context
80
+ │
81
+ ├─ Received feedback → STOP if unclear, verify if external, implement if human partner
82
+ ├─ Completed work from plan/spec:
83
+ │ ├─ Stage 1: Spec compliance review (references/spec-compliance-review.md)
84
+ │ │ └─ PASS? → Stage 2 │ FAIL? → Fix → Re-review Stage 1
85
+ │ ├─ Stage 2: Code quality review (code-reviewer subagent)
86
+ │ │ └─ Scout edge cases → Review standards, performance
87
+ │ └─ Verification gate → Run required tests/builds before claims
88
+ ├─ Completed work (no plan) → Scout → Code quality → Verification
89
+ ├─ Pre-landing / ship → Load checklists → Two-pass review → Verification
90
+ ├─ Multi-file feature (3+ files) → Create review pipeline tasks (scout→review→fix→verify)
91
+ └─ About to claim status → RUN verification command FIRST
92
+ ```
93
+
94
+ ### Review Protocol
95
+
96
+ **Stage 1 — Spec Compliance** (load `references/spec-compliance-review.md`)
97
+ - Does code match what was requested?
98
+ - Any missing requirements? Any unjustified extras?
99
+ - MUST pass before Stage 2
100
+
101
+ **Stage 2 — Code Quality** (code-reviewer subagent)
102
+ - Only runs AFTER spec compliance passes
103
+ - Standards, security, performance, edge cases
104
+
105
+ **Final Verification**
106
+ - Runs AFTER Stage 2 passes
107
+ - Re-run the relevant tests, build, lint, or manual reproduction
108
+ - Verify accepted findings are fixed and no new regression is introduced
109
+ - Critical findings block merge until fixed and re-verified
110
+
111
+ ## Receiving Feedback
112
+
113
+ **Pattern:** READ → UNDERSTAND → VERIFY → EVALUATE → RESPOND → IMPLEMENT
114
+ No performative agreement. Verify before implementing. Push back if wrong.
115
+
116
+ **Full protocol:** `references/code-review-reception.md`
117
+
118
+ ## Requesting Review
119
+
120
+ **When:** After each task, major features, before merge
121
+
122
+ **Process:**
123
+ 1. **Scout edge cases first** (see below)
124
+ 2. Get SHAs: `BASE_SHA=$(git rev-parse HEAD~1)` and `HEAD_SHA=$(git rev-parse HEAD)`
125
+ 3. Dispatch code-reviewer subagent with: WHAT, PLAN, BASE_SHA, HEAD_SHA, DESCRIPTION
126
+ 4. Fix Critical immediately, Important before proceeding
127
+
128
+ **Full protocol:** `references/requesting-code-review.md`
129
+
130
+ ## Edge Case Scouting
131
+
132
+ **When:** After implementation, before requesting code-reviewer
133
+
134
+ **Process:**
135
+ 1. Dispatch an `Explore` subagent with an edge-case-focused prompt
136
+ 2. It analyzes: affected files, data flows, error paths, boundary conditions
137
+ 3. Review the returned findings for potential issues
138
+ 4. Address critical gaps before code review
139
+
140
+ **Full protocol:** `references/edge-case-scouting.md`
141
+
142
+ ## Task-Managed Review Pipeline
143
+
144
+ **When:** Multi-file features (3+ changed files), parallel code-reviewer scopes, review cycles with Critical fix iterations.
145
+
146
+ **Fallback:** Task tools (`TaskCreate`/`TaskUpdate`/`TaskGet`/`TaskList`) are CLI-only — unavailable in VSCode extension. If they error, use `TodoWrite` for tracking and run pipeline sequentially. Review quality is identical.
147
+
148
+ **Pipeline:** scout → review → fix → verify (each a Task with dependency chain)
149
+
150
+ ```
151
+ TaskCreate: "Scout edge cases" → pending
152
+ TaskCreate: "Review implementation" → pending, blockedBy: [scout]
153
+ TaskCreate: "Fix critical issues" → pending, blockedBy: [review]
154
+ TaskCreate: "Verify fixes pass" → pending, blockedBy: [fix]
155
+ ```
156
+
157
+ **Parallel reviews:** Spawn scoped code-reviewer subagents for independent file groups (e.g., backend + frontend). Fix task blocks on all reviewers completing.
158
+
159
+ **Re-review cycles:** If fixes introduce new issues, create cycle-2 review task. Limit 3 cycles, escalate to user after.
160
+
161
+ **Full protocol:** `references/task-management-reviews.md`
162
+
163
+ ## Verification Gates
164
+
165
+ **Iron Law:** NO COMPLETION CLAIMS WITHOUT FRESH VERIFICATION EVIDENCE
166
+
167
+ **Gate:** IDENTIFY command → RUN full → READ output → VERIFY confirms → THEN claim
168
+
169
+ **Requirements:**
170
+ - Tests pass: Output shows 0 failures
171
+ - Build succeeds: Exit 0
172
+ - Bug fixed: Original symptom passes
173
+ - Requirements met: Checklist verified
174
+
175
+ **Red Flags:** "should"/"probably"/"seems to", satisfaction before verification, trusting agent reports
176
+
177
+ **Full protocol:** `references/verification-before-completion.md`
178
+
179
+ ## Integration with Workflows
180
+
181
+ - **Subagent-Driven:** Scout → Review → Verify before next task
182
+ - **Pull Requests:** Scout → Code quality → Verify → Merge
183
+ - **Task Pipeline:** Create review tasks with dependencies → auto-unblock through chain
184
+ - **PR Review:** `/code-review #123` → fetch diff → full review pipeline on PR changes
185
+ - **Commit Review:** `/code-review abc1234` → review specific commit with full pipeline
186
+
187
+ ## Codebase Analysis Subcommands
188
+
189
+ | Subcommand | Reference | Purpose |
190
+ |------------|-----------|---------|
191
+ | `/code-review codebase` | `references/codebase-scan-workflow.md` | Scan & analyze the codebase |
192
+ | `/code-review codebase parallel` | `references/parallel-review-workflow.md` | Ultrathink edge cases, then parallel verify |
193
+
194
+ ## Bottom Line
195
+
196
+ 1. Resolve input mode first — know WHAT you're reviewing
197
+ 2. Technical rigor over social performance
198
+ 3. Scout edge cases before review
199
+ 4. Evidence before claims
200
+
201
+ Verify. Scout. Question. Then implement. Evidence. Then claim.
@@ -0,0 +1,96 @@
1
+ # Checklist-Based Review Workflow
2
+
3
+ How to apply structured review checklists during code review.
4
+
5
+ ## When to Use
6
+
7
+ - Pre-landing review before merge
8
+ - Explicit request for checklist review
9
+ - Security audit before release
10
+ - Code-reviewer agent when reviewing significant changes (10+ files or security-sensitive)
11
+
12
+ ## Workflow
13
+
14
+ ### 1. Auto-Detect Project Type
15
+
16
+ ```bash
17
+ # Check for web app frameworks
18
+ if grep -qE '"(react|vue|svelte|next|nuxt|angular)"' package.json 2>/dev/null; then
19
+ echo "web-app"
20
+ # Check for API patterns
21
+ elif ls src/routes/ src/api/ src/controllers/ app/controllers/ 2>/dev/null | head -1; then
22
+ echo "api"
23
+ else
24
+ echo "base-only"
25
+ fi
26
+ ```
27
+
28
+ ### 2. Load Checklists
29
+
30
+ Always load: `checklists/base.md`
31
+
32
+ Overlay based on detection:
33
+ - `web-app` → also load `checklists/web-app.md`
34
+ - `api` → also load `checklists/api.md`
35
+ - Both detected → load both overlays
36
+
37
+ ### 3. Get the Diff
38
+
39
+ ```bash
40
+ git fetch origin main --quiet
41
+ git diff origin/main
42
+ ```
43
+
44
+ **CRITICAL:** Read the FULL diff before flagging anything. Checklist suppressions require full context.
45
+
46
+ ### 4. Two-Pass Review
47
+
48
+ **Pass 1 (CRITICAL) — Run first:**
49
+ - Scan diff against ALL critical categories (base + overlays)
50
+ - Each finding must include: `[file:line]`, problem, fix
51
+ - These block `/ship` pipeline
52
+
53
+ **Pass 2 (INFORMATIONAL) — Run second:**
54
+ - Scan diff against ALL informational categories (base + overlays)
55
+ - Same format: `[file:line]`, problem, fix
56
+ - Included in PR body but don't block
57
+
58
+ ### 5. Check Suppressions
59
+
60
+ Before reporting any finding, verify it's NOT in the suppressions list (bottom of `base.md`).
61
+
62
+ Key suppressions:
63
+ - Already addressed in the diff
64
+ - Readability-aiding redundancy
65
+ - Style/formatting issues
66
+ - "Consider using X" when Y works fine
67
+
68
+ ### 6. Output
69
+
70
+ ```
71
+ Pre-Landing Review: N issues (X critical, Y informational)
72
+
73
+ **CRITICAL** (blocking):
74
+ - [src/auth/login.ts:42] User input is interpolated directly into a query
75
+ Fix: Use the project's parameterized query helper before passing user input
76
+
77
+ **Issues** (non-blocking):
78
+ - [src/api/users.ts:88] Magic number 30 for pagination limit
79
+ Fix: Extract to constant `DEFAULT_PAGE_SIZE = 30`
80
+ ```
81
+
82
+ ### 7. Critical Issue Resolution
83
+
84
+ For each critical issue, use `AskUserQuestion`:
85
+ - Problem with `file:line`
86
+ - Recommended fix
87
+ - Options:
88
+ - A) Fix now (recommended)
89
+ - B) Acknowledge and proceed
90
+ - C) False positive — skip
91
+
92
+ If user chose A (fix): apply fixes, commit, then re-run tests before continuing.
93
+
94
+ ## Integration with /code-review
95
+
96
+ When invoked as part of standard code review, the checklist augments (not replaces) the existing scout → review → fix → verify pipeline. Checklist findings are merged with code-reviewer's own findings.
@@ -1,52 +1,52 @@
1
- # API Review Checklist (Overlay)
2
-
3
- Additive to `base.md`. Apply when project exposes REST/GraphQL/gRPC APIs.
4
-
5
- ## Detection
6
-
7
- Apply this overlay when any of these are true:
8
- - Project has route definitions (Express, FastAPI, NestJS, Django, Rails, Go chi/gin)
9
- - OpenAPI/Swagger spec file exists
10
- - `src/routes/`, `src/api/`, `src/controllers/` directories
11
- - GraphQL schema files in the diff
12
-
13
- ---
14
-
15
- ## Pass 1 — CRITICAL (additions to base)
16
-
17
- ### Auth & Rate Limiting
18
- - Public endpoints missing rate limiting (login, registration, password reset)
19
- - API keys or tokens exposed in URL query parameters (use headers)
20
- - Missing auth middleware on new routes
21
- - Batch/bulk endpoints without per-item authorization checks
22
-
23
- ### Input Validation
24
- - Request body accepted without schema validation (missing Zod, Joi, Pydantic, etc.)
25
- - Mass assignment: entire request body spread into database model
26
- - File upload without size/type restrictions
27
- - Array inputs without length limits (DoS via large payloads)
28
-
29
- ### Data Exposure
30
- - Sensitive fields in API responses (password hashes, internal IDs, tokens)
31
- - Stack traces or internal error details in production error responses
32
- - Verbose error messages that leak schema/implementation details
33
-
34
- ---
35
-
36
- ## Pass 2 — INFORMATIONAL (additions to base)
37
-
38
- ### API Design
39
- - List endpoints without pagination (LIMIT/OFFSET or cursor-based)
40
- - Missing consistent error response format across endpoints
41
- - Inconsistent naming conventions (camelCase vs snake_case in same API)
42
- - Missing request/response content-type headers
43
-
44
- ### Observability
45
- - New endpoints without logging/metrics
46
- - Error paths that swallow exceptions silently
47
- - Missing correlation/request IDs for tracing
48
-
49
- ### Versioning & Compatibility
50
- - Breaking changes to existing response shapes without version bump
51
- - Removed fields without deprecation notice
52
- - Changed field types (string → number) in existing responses
1
+ # API Review Checklist (Overlay)
2
+
3
+ Additive to `base.md`. Apply when project exposes REST/GraphQL/gRPC APIs.
4
+
5
+ ## Detection
6
+
7
+ Apply this overlay when any of these are true:
8
+ - Project has route definitions (Express, FastAPI, NestJS, Django, Rails, Go chi/gin)
9
+ - OpenAPI/Swagger spec file exists
10
+ - `src/routes/`, `src/api/`, `src/controllers/` directories
11
+ - GraphQL schema files in the diff
12
+
13
+ ---
14
+
15
+ ## Pass 1 — CRITICAL (additions to base)
16
+
17
+ ### Auth & Rate Limiting
18
+ - Public endpoints missing rate limiting (login, registration, password reset)
19
+ - API keys or tokens exposed in URL query parameters (use headers)
20
+ - Missing auth middleware on new routes
21
+ - Batch/bulk endpoints without per-item authorization checks
22
+
23
+ ### Input Validation
24
+ - Request body accepted without schema validation (missing Zod, Joi, Pydantic, etc.)
25
+ - Mass assignment: entire request body spread into database model
26
+ - File upload without size/type restrictions
27
+ - Array inputs without length limits (resource exhaustion via oversized requests)
28
+
29
+ ### Data Exposure
30
+ - Sensitive fields in API responses (password hashes, internal IDs, tokens)
31
+ - Stack traces or internal error details in production error responses
32
+ - Verbose error messages that leak schema/implementation details
33
+
34
+ ---
35
+
36
+ ## Pass 2 — INFORMATIONAL (additions to base)
37
+
38
+ ### API Design
39
+ - List endpoints without pagination (LIMIT/OFFSET or cursor-based)
40
+ - Missing consistent error response format across endpoints
41
+ - Inconsistent naming conventions (camelCase vs snake_case in same API)
42
+ - Missing request/response content-type headers
43
+
44
+ ### Observability
45
+ - New endpoints without logging/metrics
46
+ - Error paths that swallow exceptions silently
47
+ - Missing correlation/request IDs for tracing
48
+
49
+ ### Versioning & Compatibility
50
+ - Breaking changes to existing response shapes without version bump
51
+ - Removed fields without deprecation notice
52
+ - Changed field types (string → number) in existing responses
@@ -1,100 +1,100 @@
1
- # Base Review Checklist
2
-
3
- Universal checklist for all project types. Two-pass model: critical (blocking) + informational (non-blocking).
4
-
5
- ## Instructions
6
-
7
- Review `git diff origin/main` for the issues below. Be specific — cite `file:line` and suggest fixes. Skip anything that's fine. Only flag real problems.
8
-
9
- **Output format:**
10
-
11
- ```
12
- Pre-Landing Review: N issues (X critical, Y informational)
13
-
14
- **CRITICAL** (blocking):
15
- - [file:line] Problem description
16
- Fix: suggested fix
17
-
18
- **Issues** (non-blocking):
19
- - [file:line] Problem description
20
- Fix: suggested fix
21
- ```
22
-
23
- If no issues: `Pre-Landing Review: No issues found.`
24
-
25
- Be terse. One line problem, one line fix. No preamble.
26
-
27
- ---
28
-
29
- ## Pass 1 — CRITICAL (blocking)
30
-
31
- ### Injection & Data Safety
32
- - String interpolation in SQL/database queries (even with type casting — use parameterized queries)
33
- - Unsanitized user input written to database or rendered in HTML
34
- - Raw HTML output from user-controlled data (`innerHTML`, `dangerouslySetInnerHTML`, `html_safe`, `raw()`, `| safe`)
35
- - Command injection via string concatenation in shell commands (use argument arrays)
36
- - Path traversal via user input in file operations
37
-
38
- ### Race Conditions & Concurrency
39
- - Read-check-write without atomic operations (check-then-set should be atomic WHERE + UPDATE)
40
- - Find-or-create without unique database constraint (concurrent calls create duplicates)
41
- - Status transitions without atomic WHERE old_status + UPDATE new_status
42
- - Shared mutable state accessed without synchronization
43
-
44
- ### Security Boundaries
45
- - Missing authentication checks on new endpoints/routes
46
- - Privilege escalation paths (user can access/modify another user's data — IDOR)
47
- - Secrets in logs, error responses, or client-side code
48
- - LLM/AI output written to database or used in queries without validation
49
- - JWT/token comparison using `==` instead of constant-time comparison
50
-
51
- ### Auth & Access Control
52
- - New API endpoints without auth middleware
53
- - Missing authorization check (authenticated but not authorized)
54
- - Admin-only operations accessible to regular users
55
- - Session fixation or token reuse vulnerabilities
56
-
57
- ---
58
-
59
- ## Pass 2 — INFORMATIONAL (non-blocking)
60
-
61
- ### Conditional Side Effects
62
- - Code branches on condition but forgets side effect on one branch (e.g., sets status but not associated data)
63
- - Log messages claiming action happened but action was conditionally skipped
64
-
65
- ### Magic Numbers & String Coupling
66
- - Bare numeric literals used in multiple files — should be named constants
67
- - Error message strings used as query filters elsewhere (grep for the string)
68
-
69
- ### Dead Code & Consistency
70
- - Variables assigned but never read
71
- - Stale comments describing old behavior after code changed
72
- - Import/require statements for unused modules
73
-
74
- ### Test Gaps
75
- - Missing negative-path tests (error cases, validation failures)
76
- - Assertions on type/status but not side effects (e.g., checks status but not that email was sent)
77
- - Missing integration tests for security enforcement (auth, rate limiting, access control)
78
-
79
- ### Type Coercion at Boundaries
80
- - Values crossing language/system boundaries where type could change (string vs number)
81
- - Hash/digest inputs that don't normalize types before serialization
82
-
83
- ### Performance
84
- - O(n*m) lookups in views/templates (array search inside loops — use hash/map lookup)
85
- - Missing pagination on list endpoints returning unbounded results
86
- - N+1 queries: loading associations inside loops without eager loading
87
- - Unbounded queries without LIMIT
88
-
89
- ---
90
-
91
- ## Suppressions — DO NOT flag these
92
-
93
- - Redundancy that aids readability (e.g., `present?` redundant with length check)
94
- - "Add comment explaining why this threshold was chosen" — thresholds change, comments rot
95
- - "This assertion could be tighter" when assertion already covers the behavior
96
- - Consistency-only changes (wrapping a value to match how another constant is guarded)
97
- - Harmless no-ops (e.g., `.filter()` on array that never contains the filtered value)
98
- - ANYTHING already addressed in the diff being reviewed — read the FULL diff before commenting
99
- - Style/formatting issues (use a linter for that)
100
- - "Consider using X instead of Y" when Y works fine
1
+ # Base Review Checklist
2
+
3
+ Universal checklist for all project types. Two-pass model: critical (blocking) + informational (non-blocking).
4
+
5
+ ## Instructions
6
+
7
+ Review `git diff origin/main` for the issues below. Be specific — cite `file:line` and suggest fixes. Skip anything that's fine. Only flag real problems.
8
+
9
+ **Output format:**
10
+
11
+ ```
12
+ Pre-Landing Review: N issues (X critical, Y informational)
13
+
14
+ **CRITICAL** (blocking):
15
+ - [file:line] Problem description
16
+ Fix: suggested fix
17
+
18
+ **Issues** (non-blocking):
19
+ - [file:line] Problem description
20
+ Fix: suggested fix
21
+ ```
22
+
23
+ If no issues: `Pre-Landing Review: No issues found.`
24
+
25
+ Be terse. One line problem, one line fix. No preamble.
26
+
27
+ ---
28
+
29
+ ## Pass 1 — CRITICAL (blocking)
30
+
31
+ ### Injection & Data Safety
32
+ - String interpolation in SQL/database queries (even with type casting — use parameterized queries)
33
+ - Unsanitized user input written to database or rendered in HTML
34
+ - Raw HTML output from user-controlled data (`innerHTML`, `dangerouslySetInnerHTML`, `html_safe`, `raw()`, `| safe`)
35
+ - Command injection via string concatenation in shell commands (use argument arrays)
36
+ - Path traversal via user input in file operations
37
+
38
+ ### Race Conditions & Concurrency
39
+ - Read-check-write without atomic operations (check-then-set should be atomic WHERE + UPDATE)
40
+ - Find-or-create without unique database constraint (concurrent calls create duplicates)
41
+ - Status transitions without atomic WHERE old_status + UPDATE new_status
42
+ - Shared mutable state accessed without synchronization
43
+
44
+ ### Security Boundaries
45
+ - Missing authentication checks on new endpoints/routes
46
+ - Privilege escalation paths (user can access/modify another user's data — IDOR)
47
+ - Secrets in logs, error responses, or client-side code
48
+ - LLM/AI output written to database or used in queries without validation
49
+ - JWT/token comparison using `==` instead of constant-time comparison
50
+
51
+ ### Auth & Access Control
52
+ - New API endpoints without auth middleware
53
+ - Missing authorization check (authenticated but not authorized)
54
+ - Admin-only operations accessible to regular users
55
+ - Session fixation or token reuse defects
56
+
57
+ ---
58
+
59
+ ## Pass 2 — INFORMATIONAL (non-blocking)
60
+
61
+ ### Conditional Side Effects
62
+ - Code branches on condition but forgets side effect on one branch (e.g., sets status but not associated data)
63
+ - Log messages claiming action happened but action was conditionally skipped
64
+
65
+ ### Magic Numbers & String Coupling
66
+ - Bare numeric literals used in multiple files — should be named constants
67
+ - Error message strings used as query filters elsewhere (grep for the string)
68
+
69
+ ### Dead Code & Consistency
70
+ - Variables assigned but never read
71
+ - Stale comments describing old behavior after code changed
72
+ - Import/require statements for unused modules
73
+
74
+ ### Test Gaps
75
+ - Missing negative-path tests (error cases, validation failures)
76
+ - Assertions on type/status but not side effects (e.g., checks status but not that email was sent)
77
+ - Missing integration tests for security enforcement (auth, rate limiting, access control)
78
+
79
+ ### Type Coercion at Boundaries
80
+ - Values crossing language/system boundaries where type could change (string vs number)
81
+ - Hash/digest inputs that don't normalize types before serialization
82
+
83
+ ### Performance
84
+ - O(n*m) lookups in views/templates (array search inside loops — use hash/map lookup)
85
+ - Missing pagination on list endpoints returning unbounded results
86
+ - N+1 queries: loading associations inside loops without eager loading
87
+ - Unbounded queries without LIMIT
88
+
89
+ ---
90
+
91
+ ## Suppressions — DO NOT flag these
92
+
93
+ - Redundancy that aids readability (e.g., `present?` redundant with length check)
94
+ - "Add comment explaining why this threshold was chosen" — thresholds change, comments rot
95
+ - "This assertion could be tighter" when assertion already covers the behavior
96
+ - Consistency-only changes (wrapping a value to match how another constant is guarded)
97
+ - Harmless no-ops (e.g., `.filter()` on array that never contains the filtered value)
98
+ - ANYTHING already addressed in the diff being reviewed — read the FULL diff before commenting
99
+ - Style/formatting issues (use a linter for that)
100
+ - "Consider using X instead of Y" when Y works fine