@mrciphersmith/keryx 0.3.2 → 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +4634 -2445
- package/dist/core.js +66 -10
- package/package.json +1 -1
- package/src/gdskills/bundled/install-manifest.json +349 -2
- package/src/gdskills/bundled/rules/core/model-selection.mdc +18 -0
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/planning/interviewer/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/deploy/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/review/review-jev-contract/SKILL.md +193 -0
- package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.detail.md +81 -21
- package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +4 -4
- package/src/gdskills/bundled/stacks/c-cpp/agent-refs.json +4 -0
- package/src/gdskills/bundled/stacks/c-cpp/governance/eval.json +1777 -0
- package/src/gdskills/bundled/stacks/c-cpp/governance/scout.json +31 -0
- package/src/gdskills/bundled/stacks/c-cpp/pack.json +42 -0
- package/src/gdskills/bundled/stacks/c-cpp/rules/coding-style.mdc +80 -0
- package/src/gdskills/bundled/stacks/c-cpp/rules/patterns.mdc +87 -0
- package/src/gdskills/bundled/stacks/c-cpp/rules/security.mdc +90 -0
- package/src/gdskills/bundled/stacks/c-cpp/rules/testing.mdc +83 -0
- package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-build-fix/SKILL.md +153 -0
- package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-build-fix/evals.json +74 -0
- package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-code-review/SKILL.md +132 -0
- package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-code-review/evals.json +73 -0
- package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-implementation/SKILL.md +151 -0
- package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-implementation/evals.json +74 -0
- package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-testing/SKILL.md +152 -0
- package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-testing/evals.json +74 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/agent-refs.json +4 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/governance/eval.json +1295 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/governance/scout.json +26 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/pack.json +41 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/rules/patterns.mdc +77 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/rules/security.mdc +144 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-build-fix/SKILL.md +121 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-build-fix/evals.json +73 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-code-review/SKILL.md +139 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-code-review/evals.json +73 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-implementation/SKILL.md +147 -0
- package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-implementation/evals.json +74 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/agent-refs.json +4 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/governance/eval.json +865 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/governance/scout.json +16 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/pack.json +46 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/coding-style.mdc +74 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/patterns.mdc +81 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/security.mdc +146 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/testing.mdc +61 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-build-fix/SKILL.md +151 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-build-fix/evals.json +74 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-review/SKILL.md +135 -0
- package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-review/evals.json +76 -0
- package/src/gdskills/bundled/stacks/php-laravel/agent-refs.json +4 -0
- package/src/gdskills/bundled/stacks/php-laravel/governance/eval.json +1829 -0
- package/src/gdskills/bundled/stacks/php-laravel/governance/scout.json +33 -0
- package/src/gdskills/bundled/stacks/php-laravel/pack.json +41 -0
- package/src/gdskills/bundled/stacks/php-laravel/rules/coding-style.mdc +82 -0
- package/src/gdskills/bundled/stacks/php-laravel/rules/patterns.mdc +80 -0
- package/src/gdskills/bundled/stacks/php-laravel/rules/security.mdc +80 -0
- package/src/gdskills/bundled/stacks/php-laravel/rules/testing.mdc +82 -0
- package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-build-fix/SKILL.md +143 -0
- package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-build-fix/evals.json +74 -0
- package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-code-review/SKILL.md +126 -0
- package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-code-review/evals.json +76 -0
- package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-implementation/SKILL.md +140 -0
- package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-implementation/evals.json +75 -0
- package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-testing/SKILL.md +124 -0
- package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-testing/evals.json +74 -0
- package/src/gdskills/bundled/stacks/ruby-rails/agent-refs.json +4 -0
- package/src/gdskills/bundled/stacks/ruby-rails/governance/eval.json +1673 -0
- package/src/gdskills/bundled/stacks/ruby-rails/governance/scout.json +33 -0
- package/src/gdskills/bundled/stacks/ruby-rails/pack.json +42 -0
- package/src/gdskills/bundled/stacks/ruby-rails/rules/coding-style.mdc +69 -0
- package/src/gdskills/bundled/stacks/ruby-rails/rules/patterns.mdc +93 -0
- package/src/gdskills/bundled/stacks/ruby-rails/rules/security.mdc +90 -0
- package/src/gdskills/bundled/stacks/ruby-rails/rules/testing.mdc +89 -0
- package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-build-fix/SKILL.md +143 -0
- package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-build-fix/evals.json +73 -0
- package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-code-review/SKILL.md +134 -0
- package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-code-review/evals.json +71 -0
- package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-implementation/SKILL.md +141 -0
- package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-implementation/evals.json +72 -0
- package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-testing/SKILL.md +125 -0
- package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-testing/evals.json +72 -0
- package/src/gdskills/bundled/stacks/sql-db/agent-refs.json +4 -0
- package/src/gdskills/bundled/stacks/sql-db/governance/eval.json +1829 -0
- package/src/gdskills/bundled/stacks/sql-db/governance/scout.json +30 -0
- package/src/gdskills/bundled/stacks/sql-db/pack.json +40 -0
- package/src/gdskills/bundled/stacks/sql-db/rules/coding-style.mdc +69 -0
- package/src/gdskills/bundled/stacks/sql-db/rules/patterns.mdc +134 -0
- package/src/gdskills/bundled/stacks/sql-db/rules/security.mdc +74 -0
- package/src/gdskills/bundled/stacks/sql-db/rules/testing.mdc +83 -0
- package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-build-fix/SKILL.md +147 -0
- package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-build-fix/evals.json +72 -0
- package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-code-review/SKILL.md +132 -0
- package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-code-review/evals.json +73 -0
- package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-implementation/SKILL.md +153 -0
- package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-implementation/evals.json +77 -0
- package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-testing/SKILL.md +129 -0
- package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-testing/evals.json +73 -0
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: ruby-rails-code-review
|
|
3
|
+
description: "Use when reviewing a Rails change for framework-specific risks -- mass assignment gaps, N+1 queries, raw SQL interpolation, raw/html_safe XSS, missing auth/authorization filters, fat controllers/models, and non-idempotent ActiveJobs. Not for a change in another web framework's own MVC layer (use that framework's own code-review skill). Read-only, no edits."
|
|
4
|
+
triggers:
|
|
5
|
+
- "review this Rails diff for mass assignment issues"
|
|
6
|
+
- "check this Rails controller for N+1 queries"
|
|
7
|
+
- "review this Rails change for missing authorization"
|
|
8
|
+
- "any XSS risk from raw or html_safe in this Rails view"
|
|
9
|
+
- "review this Rails job for idempotency"
|
|
10
|
+
- "check this Rails diff for fat controller or fat model smells"
|
|
11
|
+
metadata:
|
|
12
|
+
origin: authored
|
|
13
|
+
category: review
|
|
14
|
+
version: "1.0.0"
|
|
15
|
+
compatible_harnesses: "claude,codex,cursor,zed,opencode"
|
|
16
|
+
license: "MIT"
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
# Ruby on Rails code review
|
|
20
|
+
|
|
21
|
+
Read-only review of a Rails change for framework-specific risks: mass
|
|
22
|
+
assignment, N+1 queries, raw SQL, XSS via `raw`/`html_safe`, missing
|
|
23
|
+
auth/authorization, fat controllers/models, and non-idempotent jobs.
|
|
24
|
+
This skill never edits code — it reports findings.
|
|
25
|
+
`rules/coding-style.mdc`, `rules/patterns.mdc`, and `rules/security.mdc`
|
|
26
|
+
are the rule set findings are checked against.
|
|
27
|
+
|
|
28
|
+
## Workflow
|
|
29
|
+
|
|
30
|
+
### Step 1: Scope the review
|
|
31
|
+
|
|
32
|
+
1. Identify the changed files (`git diff` against the review base) —
|
|
33
|
+
review only `*.rb` files (and templates, if in scope) in the diff,
|
|
34
|
+
not the whole repository.
|
|
35
|
+
2. Read enough of the surrounding, unchanged code to know whether a
|
|
36
|
+
flagged pattern is new in this diff or pre-existing; note
|
|
37
|
+
pre-existing issues separately from ones the diff introduces.
|
|
38
|
+
|
|
39
|
+
### Step 2: Check each changed file against the focus list
|
|
40
|
+
|
|
41
|
+
**Mass assignment**
|
|
42
|
+
- A model built/updated from `params` uses `params.expect`/
|
|
43
|
+
`params.require`+`permit` with an explicit attribute list — flag
|
|
44
|
+
`Model.new(params[:model])`, `record.update(params)`, or any
|
|
45
|
+
`permit!` on user-controlled params.
|
|
46
|
+
|
|
47
|
+
**N+1 queries**
|
|
48
|
+
- Any loop (explicit `each`/`map`, or implicit via a view iterating a
|
|
49
|
+
collection) that calls an association method per iteration — flag it
|
|
50
|
+
when the association was not eager-loaded (`includes`/`preload`/
|
|
51
|
+
`eager_load`) before the loop.
|
|
52
|
+
|
|
53
|
+
**SQL and data access**
|
|
54
|
+
- Flag any `where`, `find_by_sql`, `order`, or `pluck` call built with
|
|
55
|
+
string interpolation of a value that traces back to user input,
|
|
56
|
+
instead of parameterized conditions or `sanitize_sql`.
|
|
57
|
+
|
|
58
|
+
**XSS / output encoding**
|
|
59
|
+
- Flag `raw(...)` or `.html_safe` applied to a value that can contain
|
|
60
|
+
user-controlled or user-influenced content — it disables ERB's
|
|
61
|
+
automatic escaping for that value.
|
|
62
|
+
|
|
63
|
+
**Authentication and authorization**
|
|
64
|
+
- A controller action that should require login has the app's
|
|
65
|
+
`before_action` auth filter (not accidentally excluded via
|
|
66
|
+
`skip_before_action`) — flag one that doesn't.
|
|
67
|
+
- Flag an action that checks the user is logged in but never confirms
|
|
68
|
+
they're allowed to act on *this specific* record (an IDOR risk).
|
|
69
|
+
|
|
70
|
+
**CSRF**
|
|
71
|
+
- Flag `skip_before_action :verify_authenticity_token` or
|
|
72
|
+
`protect_from_forgery with: :null_session` added on an action
|
|
73
|
+
reachable from a standard HTML form, unless the diff itself
|
|
74
|
+
documents why (e.g. a signature-verified webhook).
|
|
75
|
+
|
|
76
|
+
**Fat controllers/models**
|
|
77
|
+
- A controller action with inline multi-step business logic, or a model
|
|
78
|
+
callback orchestrating unrelated side effects (email, third-party
|
|
79
|
+
call, another model's update) — flag as a candidate for extraction
|
|
80
|
+
into a service object per `rules/patterns.mdc`.
|
|
81
|
+
|
|
82
|
+
**Background jobs**
|
|
83
|
+
- An `ActiveJob#perform` whose side effect would be harmful if it ran
|
|
84
|
+
twice (charge, duplicate send) with no idempotency guard — flag it,
|
|
85
|
+
since most production Active Job adapters are at-least-once.
|
|
86
|
+
|
|
87
|
+
### Step 3: Report
|
|
88
|
+
|
|
89
|
+
For each finding: file:line, the pattern, why it matters (security
|
|
90
|
+
impact, correctness, maintainability), and the fix direction — but do
|
|
91
|
+
not apply it.
|
|
92
|
+
|
|
93
|
+
```
|
|
94
|
+
app/controllers/orders_controller.rb:18 — builds Order from raw
|
|
95
|
+
params[:order] with no strong parameters. Risk: mass assignment lets
|
|
96
|
+
any submitted key (including ones never meant to be user-settable)
|
|
97
|
+
through. Fix direction: require/permit an explicit attribute list via
|
|
98
|
+
params.expect(order: [...]) or params.require(:order).permit(...).
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
## Rules
|
|
102
|
+
|
|
103
|
+
- NEVER edit code — findings and fix direction only.
|
|
104
|
+
- Flag mass assignment gaps, N+1 queries, raw SQL interpolation,
|
|
105
|
+
raw/html_safe XSS risk, missing auth/authorization, disabled CSRF
|
|
106
|
+
protection, fat controller/model smells, and non-idempotent jobs; do
|
|
107
|
+
not report generic style nits already covered by `rubocop`.
|
|
108
|
+
- Distinguish a finding the diff introduces from a pre-existing one in
|
|
109
|
+
code the diff merely touches.
|
|
110
|
+
- When a suspected N+1 is not certain from reading alone (e.g. the
|
|
111
|
+
association might already be preloaded further up the call chain),
|
|
112
|
+
say so and recommend confirming with `bullet` or the query log rather
|
|
113
|
+
than asserting it with certainty.
|
|
114
|
+
|
|
115
|
+
## Red Flags
|
|
116
|
+
|
|
117
|
+
| Rationalization | Why it is wrong |
|
|
118
|
+
|---|---|
|
|
119
|
+
| "This form only has a few fields, mass assignment isn't really a risk here" | Any field in the raw params hash is a risk regardless of the form's apparent size; strong parameters cost nothing and close the gap categorically |
|
|
120
|
+
| "The N+1 here is only over 3-4 records in this view" | Reviews check the code path, not today's data volume — a small collection today is a large one after the feature ships and the table grows |
|
|
121
|
+
| "I'll just fix the missing before_action myself since it's one line" | This skill is read-only; report the finding and its fix direction, do not edit the file |
|
|
122
|
+
| "raw(user.bio) is fine, this is an internal admin tool" | User-controlled data marked pre-escaped is a genuine XSS vector regardless of who the internal audience is; flag it and let the fix direction note the sanitize option |
|
|
123
|
+
|
|
124
|
+
## Verification
|
|
125
|
+
|
|
126
|
+
Do not report the review done until all of the following hold:
|
|
127
|
+
|
|
128
|
+
- Every changed `*.rb` file in the diff was read, not just files named
|
|
129
|
+
in the PR description.
|
|
130
|
+
- Every finding names a concrete file:line, the specific risk category
|
|
131
|
+
from Step 2, and a fix direction.
|
|
132
|
+
- No source file was modified by this review.
|
|
133
|
+
- Findings distinguish diff-introduced issues from pre-existing ones in
|
|
134
|
+
touched files.
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
{
|
|
2
|
+
"triggers": {
|
|
3
|
+
"positive": [
|
|
4
|
+
"Review this Rails diff for a controller action that skips authorization",
|
|
5
|
+
"Check this Rails change for a query per row in the index view",
|
|
6
|
+
"Any XSS risk from marking this user bio safe in the view?",
|
|
7
|
+
"Review this Rails pull request for a fat model with too many callbacks",
|
|
8
|
+
"Check this Rails job diff for whether it's safe to run twice",
|
|
9
|
+
"Review this Rails controller for raw params interpolated into a where clause"
|
|
10
|
+
],
|
|
11
|
+
"negative": [
|
|
12
|
+
"Review this Go diff for goroutine leaks",
|
|
13
|
+
"Review this Django view for SQL injection",
|
|
14
|
+
"Implement the fix for the N+1 query you just found",
|
|
15
|
+
"Write a regression test for this authorization bug",
|
|
16
|
+
"Fix the rubocop offenses in this Rails file"
|
|
17
|
+
]
|
|
18
|
+
},
|
|
19
|
+
"scenarios": [
|
|
20
|
+
{
|
|
21
|
+
"id": "mass-assignment-finding",
|
|
22
|
+
"prompt": "Review this Rails diff: a create action does Order.new(params[:order]) with no permit call. What do you find?",
|
|
23
|
+
"strictness": "high",
|
|
24
|
+
"expected_behavior": [
|
|
25
|
+
{
|
|
26
|
+
"grader": "judge",
|
|
27
|
+
"rubric": "A correct answer identifies Order.new(params[:order]) with no strong parameters as a mass-assignment risk, explains that any key submitted under order gets assigned rather than only intended fields, and reports a fix direction (params.expect/params.require+permit with an explicit attribute list) as a finding only -- it never actually edits or claims to have edited the code, even partially, since this skill is read-only.",
|
|
28
|
+
"pass_criteria": [
|
|
29
|
+
"Identifies Order.new(params[:order]) with no strong parameters as the problem, naming it specifically from this diff.",
|
|
30
|
+
"Explains the concrete risk: any key submitted under :order gets mass-assigned, not just the fields the form is meant to expose.",
|
|
31
|
+
"States a concrete fix direction (params.expect or params.require+permit with an explicit attribute list) purely as a finding, with no code change applied or described as applied."
|
|
32
|
+
],
|
|
33
|
+
"fail_criteria": [
|
|
34
|
+
"Actually edits, patches, or claims to have already made even a partial or proof-of-concept code change (this skill is strictly read-only: findings and fix direction only, no edits of any size)."
|
|
35
|
+
]
|
|
36
|
+
}
|
|
37
|
+
],
|
|
38
|
+
"calibration": {
|
|
39
|
+
"known_right": "Finding: `Order.new(params[:order])` builds the model directly from the raw params hash with no `permit` call. That's a mass-assignment risk -- whatever the client submits under `order` gets assigned, not just the fields the form is supposed to expose, so an attacker or a stray client bug could set fields never meant to be user-controlled. Fix direction: require and permit an explicit attribute list, e.g. `params.expect(order: [:item_id, :quantity])` or `params.require(:order).permit(:item_id, :quantity)`, and pass that permitted hash into `Order.new` instead of the raw params. I'm reporting this as a finding rather than making the change, since this review is read-only.",
|
|
40
|
+
"known_wrong": "This is a known Rails anti-pattern so I went ahead and fixed it directly: I added `params.require(:order).permit(:item_id, :quantity)` above the `Order.new` call and updated it to use the permitted hash. The change is already in place and should be ready to merge as-is.",
|
|
41
|
+
"vague": "That params usage looks a little loose -- might be worth tightening up before this merges.",
|
|
42
|
+
"subtle_wrong": "Finding: `Order.new(params[:order])` skips strong parameters, which is worth fixing before merge. I went ahead and made a small proof-of-concept edit adding a `.permit(:item_id)` call so you can see the shape of the fix, though the rest of the form fields would still need to be added to the list before this actually works end to end."
|
|
43
|
+
}
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"id": "authorization-vs-authentication",
|
|
47
|
+
"prompt": "Review this Rails diff: a new PATCH /orders/:id action includes before_action :authenticate_user! but doesn't check that the order belongs to the current user. What do you find?",
|
|
48
|
+
"strictness": "high",
|
|
49
|
+
"expected_behavior": [
|
|
50
|
+
{
|
|
51
|
+
"grader": "judge",
|
|
52
|
+
"rubric": "A correct answer distinguishes authentication from authorization: it identifies that before_action :authenticate_user! only confirms someone is logged in, and flags the missing check that the order being updated actually belongs to that user (an IDOR risk letting any logged-in user modify another user's order), with a fix direction -- scoping the lookup to the current user or adding an explicit ownership check -- reported as a finding only, not applied.",
|
|
53
|
+
"pass_criteria": [
|
|
54
|
+
"States that before_action :authenticate_user! is authentication only, and that an ownership/authorization check on this specific order is separately missing.",
|
|
55
|
+
"Names the concrete risk: any logged-in user could update an order that isn't theirs (IDOR), not just an unauthenticated request.",
|
|
56
|
+
"States a concrete fix direction -- scoping the lookup to the current user (e.g. current_user.orders.find(params[:id])) or an explicit ownership check -- as a finding, with no code change applied."
|
|
57
|
+
],
|
|
58
|
+
"fail_criteria": [
|
|
59
|
+
"Treats before_action :authenticate_user! as sufficient on its own, without flagging the missing per-record ownership check. Mentioning that authentication alone is a first step but not the whole picture is not a failure -- concluding it fully covers this case is."
|
|
60
|
+
]
|
|
61
|
+
}
|
|
62
|
+
],
|
|
63
|
+
"calibration": {
|
|
64
|
+
"known_right": "Finding: the action requires login via `before_action :authenticate_user!`, but nothing confirms the order being updated actually belongs to the current user -- it looks like the order is likely loaded with a bare `Order.find(params[:id])`. That's an authorization gap, not an authentication one: any logged-in user can PATCH any order by guessing/incrementing the id (IDOR). Fix direction: scope the lookup to the current user, e.g. `current_user.orders.find(params[:id])`, or add an explicit check (`raise unless order.user == current_user`, or a policy object if the app has one) before allowing the update. Reporting this as a finding since this review is read-only.",
|
|
65
|
+
"known_wrong": "This looks fine -- the action already has `before_action :authenticate_user!`, so only logged-in users can hit it at all. That's the check that matters here; going further and verifying which specific order they're allowed to touch would be double-checking something the login requirement already covers.",
|
|
66
|
+
"vague": "Might want to double check the authorization here before merging, just to be safe.",
|
|
67
|
+
"subtle_wrong": "Finding: the action is authenticated via `before_action :authenticate_user!`. That said, since `params[:id]` comes from the route and not directly from a form field, and the id itself doesn't reveal anything sensitive on its own, the exposure here is pretty limited even without an explicit ownership check -- worth a follow-up ticket but not a blocker for this diff."
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
]
|
|
71
|
+
}
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: ruby-rails-implementation
|
|
3
|
+
description: "Use when implementing or extending a feature in a Ruby on Rails app -- MVC boundaries, strong parameters, ActiveRecord associations and scopes, service objects for fat controllers/models, ActiveJob, and modern Ruby 3.x idiom (pattern matching, endless methods, keyword args). Not for writing or fixing tests (use ruby-rails-testing)."
|
|
4
|
+
triggers:
|
|
5
|
+
- "implement this feature in Rails"
|
|
6
|
+
- "add a Rails controller action and model for this"
|
|
7
|
+
- "add strong parameters to this Rails controller"
|
|
8
|
+
- "extract this Rails controller logic into a service object"
|
|
9
|
+
- "add an ActiveRecord association and scope for this"
|
|
10
|
+
- "add an ActiveJob for this background task"
|
|
11
|
+
metadata:
|
|
12
|
+
origin: authored
|
|
13
|
+
category: implement
|
|
14
|
+
version: "1.0.0"
|
|
15
|
+
compatible_harnesses: "claude,codex,cursor,zed,opencode"
|
|
16
|
+
license: "MIT"
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
# Ruby on Rails implementation (Ruby 3.x, Rails)
|
|
20
|
+
|
|
21
|
+
Implement or extend a feature in a Rails codebase: MVC boundaries,
|
|
22
|
+
ActiveRecord modeling, strong parameters, service objects, ActiveJob, and
|
|
23
|
+
modern Ruby idiom. This is a combined Ruby+Rails pack (no separate
|
|
24
|
+
`lang:ruby` base) — `rules/coding-style.mdc`, `rules/patterns.mdc`, and
|
|
25
|
+
`rules/security.mdc` carry the full rule set this skill draws its
|
|
26
|
+
checklist from; read them before writing code, not just this summary.
|
|
27
|
+
|
|
28
|
+
## Workflow
|
|
29
|
+
|
|
30
|
+
### Step 1: Discover the project's own conventions
|
|
31
|
+
|
|
32
|
+
1. Read `Gemfile`/`Gemfile.lock` for the Rails and Ruby versions, and
|
|
33
|
+
whether the project already has a service-object gem or convention
|
|
34
|
+
(`app/services/`, a shared base class), FactoryBot, RSpec vs
|
|
35
|
+
Minitest, and any auth gem (Devise, has_secure_password).
|
|
36
|
+
2. Find the existing layout under `app/` for the area you're touching —
|
|
37
|
+
match its naming and directory conventions; do not invent a new
|
|
38
|
+
top-level directory for one feature.
|
|
39
|
+
3. Read 1-2 neighboring models/controllers for: validation style,
|
|
40
|
+
whether concerns are already used, how strong parameters are
|
|
41
|
+
structured, and whether callbacks or service objects are the
|
|
42
|
+
project's preferred way to keep controllers/models thin.
|
|
43
|
+
|
|
44
|
+
### Step 2: Design before writing
|
|
45
|
+
|
|
46
|
+
- Decide where each piece of new logic belongs: routing/params handling
|
|
47
|
+
in the controller, persistence/validation in the model, multi-step or
|
|
48
|
+
cross-model orchestration in a service object (`rules/patterns.mdc`)
|
|
49
|
+
— not stacked into one fat controller action or model callback.
|
|
50
|
+
- For a new ActiveRecord association, decide the query pattern up front:
|
|
51
|
+
will callers iterate the association in a loop (needs `includes` at
|
|
52
|
+
the call site to avoid N+1) or query it directly (a named `scope` may
|
|
53
|
+
belong on the model).
|
|
54
|
+
- For a new background job, decide what makes `perform` idempotent
|
|
55
|
+
before writing it — a uniqueness check, an idempotency key, or an
|
|
56
|
+
upsert — most Active Job queue adapters (Sidekiq, SQS) are
|
|
57
|
+
at-least-once, not exactly-once, though the actual guarantee depends
|
|
58
|
+
on the configured adapter (the inline/test adapters have none).
|
|
59
|
+
|
|
60
|
+
### Step 3: Implement
|
|
61
|
+
|
|
62
|
+
1. Strong parameters: require the permitted shape explicitly with
|
|
63
|
+
`params.expect(model: [:field, ...])` (current Rails idiom) or
|
|
64
|
+
`params.require(:model).permit(:field, ...)` — never build/update a
|
|
65
|
+
model from a raw `params` hash, and never call `permit!` on
|
|
66
|
+
user-controlled params.
|
|
67
|
+
2. Keep the controller action thin: params in, call the model/service,
|
|
68
|
+
render/redirect out. Extract a service object when the action's
|
|
69
|
+
logic spans multiple models, calls an external service, or has
|
|
70
|
+
enough branching to warrant its own tests.
|
|
71
|
+
3. Use `includes`/`preload`/`eager_load` before iterating an association
|
|
72
|
+
in a loop; push filtering into the database (`where`, a named
|
|
73
|
+
`scope`) instead of loading records and filtering in Ruby.
|
|
74
|
+
4. Reach for modern Ruby idiom where it genuinely clarifies: `case/in`
|
|
75
|
+
pattern matching for destructuring a hash/array-shaped value, an
|
|
76
|
+
endless method for a real one-expression method, keyword arguments
|
|
77
|
+
for a multi-parameter method — not as syntax for its own sake.
|
|
78
|
+
5. Every controller action that requires a logged-in user has the
|
|
79
|
+
app's `before_action` auth filter, and every action further checks
|
|
80
|
+
that the current user is authorized for *this* record, not just
|
|
81
|
+
logged in.
|
|
82
|
+
6. Add `# frozen_string_literal: true` to new files if the project's
|
|
83
|
+
existing files use it.
|
|
84
|
+
|
|
85
|
+
### Step 4: Verify
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
bundle exec rspec # or: bin/rails test
|
|
89
|
+
bundle exec rubocop
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Run `bin/rails db:test:prepare` if a migration changed the schema. Fix
|
|
93
|
+
findings at the root cause per `rules/security.mdc` and
|
|
94
|
+
`rules/coding-style.mdc`; a failing test here is a signal to fix the
|
|
95
|
+
implementation, not to reach for `ruby-rails-build-fix`'s scope unless
|
|
96
|
+
the failure is purely a dependency/migration/toolchain problem unrelated
|
|
97
|
+
to the feature logic.
|
|
98
|
+
|
|
99
|
+
### Step 5: Report
|
|
100
|
+
|
|
101
|
+
```
|
|
102
|
+
Implemented: app/models/order.rb, app/controllers/orders_controller.rb,
|
|
103
|
+
app/services/place_order.rb, spec/services/place_order_spec.rb
|
|
104
|
+
- New PlaceOrder service object consumed by OrdersController#create
|
|
105
|
+
- rspec/rubocop both pass
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
## Rules
|
|
109
|
+
|
|
110
|
+
- Never build or update a model from a raw `params` hash; always go
|
|
111
|
+
through strong parameters (`params.expect`/`params.require`+`permit`),
|
|
112
|
+
and never call `permit!` on user-controlled params.
|
|
113
|
+
- Never iterate an ActiveRecord association in a loop without eager
|
|
114
|
+
loading it first (`includes`) when the association is used inside
|
|
115
|
+
that loop.
|
|
116
|
+
- Never put multi-step orchestration or an external API call directly
|
|
117
|
+
in a controller action or a model callback — extract a service
|
|
118
|
+
object.
|
|
119
|
+
- Never write an `ActiveJob#perform` that assumes exactly-once delivery
|
|
120
|
+
for a side effect that would be harmful to repeat.
|
|
121
|
+
|
|
122
|
+
## Red Flags
|
|
123
|
+
|
|
124
|
+
| Rationalization | Why it is wrong |
|
|
125
|
+
|---|---|
|
|
126
|
+
| "I'll just do `Order.new(params[:order])`, it's a small internal form" | Bypasses strong parameters entirely; any key in `params[:order]`, including ones never meant to be user-settable, gets mass-assigned |
|
|
127
|
+
| "I'll call `.author` inside this `each` loop, it's just one extra query" | That "one extra query" happens once per row — N+1 queries; add `.includes(:author)` before the loop |
|
|
128
|
+
| "This callback also sends a welcome email and pings analytics, but it's still 'about' the model" | A model callback doing multi-step, cross-system work is the fat-model anti-pattern; extract a service object so it's testable in isolation and the model stays about persistence |
|
|
129
|
+
| "The job will basically only ever run once" | Most production Active Job adapters are at-least-once, not exactly-once; a retried or duplicated run has to be safe |
|
|
130
|
+
|
|
131
|
+
## Verification
|
|
132
|
+
|
|
133
|
+
Do not report the work done until all of the following hold:
|
|
134
|
+
|
|
135
|
+
- `bundle exec rspec` (or `bin/rails test`) and `bundle exec rubocop`
|
|
136
|
+
both exit 0.
|
|
137
|
+
- Every new/changed controller action that mutates a model goes through
|
|
138
|
+
strong parameters, with no `permit!` on user-controlled input.
|
|
139
|
+
- Every new loop over an ActiveRecord association eager-loads it first.
|
|
140
|
+
- Every new `ActiveJob#perform` is safe to run more than once for the
|
|
141
|
+
same logical unit of work.
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
{
|
|
2
|
+
"triggers": {
|
|
3
|
+
"positive": [
|
|
4
|
+
"Implement a PlaceOrder service object that a controller action can call",
|
|
5
|
+
"Add strong parameters to this create action so mass assignment isn't possible",
|
|
6
|
+
"Wire up an ActiveJob that charges a customer and needs to be safe to retry",
|
|
7
|
+
"Add an eager-loaded association so this index view doesn't run a query per row",
|
|
8
|
+
"Extract this fat controller action's business logic into its own class",
|
|
9
|
+
"Add a case/in pattern match to parse this webhook payload's JSON shape"
|
|
10
|
+
],
|
|
11
|
+
"negative": [
|
|
12
|
+
"Implement this feature in Django using its ORM",
|
|
13
|
+
"Add a POST endpoint in Express that validates the request body",
|
|
14
|
+
"Review this Rails diff for N+1 queries",
|
|
15
|
+
"Write RSpec tests for this Rails service object",
|
|
16
|
+
"Fix this failing bundle install error in the Rails app"
|
|
17
|
+
]
|
|
18
|
+
},
|
|
19
|
+
"scenarios": [
|
|
20
|
+
{
|
|
21
|
+
"id": "strong-parameters",
|
|
22
|
+
"prompt": "I'm building a create action for orders that saves whatever comes in from the form. How should I handle the params?",
|
|
23
|
+
"strictness": "high",
|
|
24
|
+
"expected_behavior": [
|
|
25
|
+
{
|
|
26
|
+
"grader": "judge",
|
|
27
|
+
"rubric": "A correct answer requires and permits an explicit list of attributes (params.expect or params.require+permit) before building/updating the model, rather than passing the raw params hash directly to the model or calling permit! to bypass the check.",
|
|
28
|
+
"pass_criteria": [
|
|
29
|
+
"Names params.expect(order: [...]) or params.require(:order).permit(...) explicitly as the mechanism, with a concrete attribute list shown.",
|
|
30
|
+
"States that building/updating the model directly from the raw params hash (e.g. Order.new(params[:order])) is unsafe because of mass assignment.",
|
|
31
|
+
"Does not recommend permit! as part of the fix."
|
|
32
|
+
],
|
|
33
|
+
"fail_criteria": [
|
|
34
|
+
"Recommends permit! on user-controlled params, or building/updating the model from the raw params hash (or by merging in unpermitted keys from it) instead of listing permitted attributes explicitly. Mentioning permit! only to warn against it is not a failure."
|
|
35
|
+
]
|
|
36
|
+
}
|
|
37
|
+
],
|
|
38
|
+
"calibration": {
|
|
39
|
+
"known_right": "Require the params under :order and permit only the fields the form actually submits, e.g. `params.expect(order: [:item_id, :quantity, :shipping_address])` (or `params.require(:order).permit(:item_id, :quantity, :shipping_address)` on an older Rails). Then pass the permitted hash to `Order.new(order_params)`. Don't build the record straight from `params[:order]` -- that hands mass assignment to whatever keys are in the submitted form, including ones nobody meant to be settable from outside, like a `total_price` override or an `admin` flag if one is ever added to the model.",
|
|
40
|
+
"known_wrong": "Simplest is `Order.new(params[:order])` and let ActiveRecord sort out which attributes it has -- Rails handles that fine. If you're getting a ForbiddenAttributesError anywhere in the chain, just call `.permit!` on the params hash before you use it; that clears the block globally so you don't have to keep listing out fields by hand as the form grows.",
|
|
41
|
+
"vague": "Make sure you're only accepting the fields you actually mean to accept before saving the order -- don't just trust the raw form input.",
|
|
42
|
+
"subtle_wrong": "Use `params.require(:order).permit(:item_id, :quantity)` for the known fields, and then merge in `params[:order].except(:item_id, :quantity)` for anything else the form might send, so you don't have to keep updating the permit list every time a new field gets added to the form."
|
|
43
|
+
},
|
|
44
|
+
"anti_patterns": ["permit!"]
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
"id": "n-plus-one",
|
|
48
|
+
"prompt": "This index view loops over @orders and calls order.customer.name for each one. Anything to watch out for?",
|
|
49
|
+
"strictness": "high",
|
|
50
|
+
"expected_behavior": [
|
|
51
|
+
{
|
|
52
|
+
"grader": "judge",
|
|
53
|
+
"rubric": "A correct answer identifies this as an N+1 query -- one extra query per order to load its customer -- and fixes it by eager-loading the association (includes(:customer), or preload/eager_load) where @orders is built, rather than leaving the per-row queries or fixing it only inside the view.",
|
|
54
|
+
"pass_criteria": [
|
|
55
|
+
"Identifies the per-row `.customer` call as an N+1 query risk, naming the mechanism (one extra query per order).",
|
|
56
|
+
"Names the concrete fix: add `.includes(:customer)` (or preload/eager_load) to the @orders query, in the controller/query-building code, not in the view.",
|
|
57
|
+
"States that this reduces the query count from one-per-row (N+1) to a small constant number."
|
|
58
|
+
],
|
|
59
|
+
"fail_criteria": [
|
|
60
|
+
"Dismisses the N+1 as acceptable because the table or collection is currently small, or proposes per-row caching of the association instead of eager-loading it at the point @orders is queried."
|
|
61
|
+
]
|
|
62
|
+
}
|
|
63
|
+
],
|
|
64
|
+
"calibration": {
|
|
65
|
+
"known_right": "That's an N+1: `order.customer.name` inside the loop runs one query per order to fetch its customer, on top of the query that loaded @orders in the first place -- so 50 orders means 51 queries. Fix it where @orders is built, in the controller: `@orders = Order.includes(:customer)` (composed with whatever the existing scope chain already is) rather than in the view -- Rails will pick preload or eager_load for you and pull all the customers in one or two queries total instead of one per row. The view code itself doesn't need to change.",
|
|
66
|
+
"known_wrong": "That's fine as long as the orders table stays small -- a handful of extra queries per page load isn't something to worry about unless you're seeing real performance problems in production. I'd leave the view as-is and only add includes if it actually shows up as slow in the logs.",
|
|
67
|
+
"vague": "That loop could be running more queries than it needs to -- might be worth eager loading the association at some point.",
|
|
68
|
+
"subtle_wrong": "Add `order.association(:customer).load_target` inside the loop instead of calling `.customer` directly -- that at least caches the association per order object so you're not re-triggering the query if the view happens to reference `.customer` more than once on the same row, without needing to touch how @orders is queried in the controller."
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
]
|
|
72
|
+
}
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: ruby-rails-testing
|
|
3
|
+
description: "Use when a Rails app's test suite needs writing, extending, or fixing -- RSpec request/model/job specs (or Minitest equivalents), FactoryBot fixtures, stubbing external HTTP calls, asserting on enqueued ActiveJobs, and avoiding sleep-based waits."
|
|
4
|
+
triggers:
|
|
5
|
+
- "write RSpec tests for this Rails model"
|
|
6
|
+
- "add a request spec for this Rails controller action"
|
|
7
|
+
- "test this ActiveJob with have_enqueued_job"
|
|
8
|
+
- "fix this failing Rails test"
|
|
9
|
+
- "write Minitest tests for this Rails model"
|
|
10
|
+
- "stub this external API call in a Rails test"
|
|
11
|
+
metadata:
|
|
12
|
+
origin: authored
|
|
13
|
+
category: test
|
|
14
|
+
version: "1.0.0"
|
|
15
|
+
compatible_harnesses: "claude,codex,cursor,zed,opencode"
|
|
16
|
+
license: "MIT"
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
# Ruby on Rails testing
|
|
20
|
+
|
|
21
|
+
Write, extend, or fix a Rails app's test suite: RSpec (or Minitest)
|
|
22
|
+
specs for models, controllers/requests, and jobs, with FactoryBot
|
|
23
|
+
fixtures, stubbed external boundaries, and deterministic waits.
|
|
24
|
+
`rules/testing.mdc` carries the full rule set this skill's checklist is
|
|
25
|
+
built from — read it, not just this summary, before writing tests.
|
|
26
|
+
|
|
27
|
+
## Workflow
|
|
28
|
+
|
|
29
|
+
### Step 1: Discover the project's test conventions
|
|
30
|
+
|
|
31
|
+
1. Check `Gemfile`/`Gemfile.lock` for RSpec vs Minitest, FactoryBot,
|
|
32
|
+
WebMock/VCR, and any time-freezing gem (Timecop or Rails' built-in
|
|
33
|
+
`travel_to`) — match whichever is already there, never introduce the
|
|
34
|
+
other test framework.
|
|
35
|
+
2. Find the layout: `spec/<mirror of app/>` (RSpec) or
|
|
36
|
+
`test/<mirror of app/>` (Minitest). Match the existing file for the
|
|
37
|
+
type of code under test (model spec, request spec, job spec).
|
|
38
|
+
3. Read 1-2 neighboring spec/test files for: fixture style
|
|
39
|
+
(`let`/`let!`/factories vs fixtures), whether request specs or
|
|
40
|
+
controller specs are used, and how external HTTP calls are already
|
|
41
|
+
stubbed in this project.
|
|
42
|
+
|
|
43
|
+
### Step 2: Plan test cases
|
|
44
|
+
|
|
45
|
+
**Models:** validations (presence, uniqueness, format), each
|
|
46
|
+
association's behavior relevant to the change, scopes, and any
|
|
47
|
+
non-trivial instance/class method.
|
|
48
|
+
|
|
49
|
+
**Requests (preferred over controller specs for new coverage):**
|
|
50
|
+
happy path (correct status/body), auth failure (no session — expect a
|
|
51
|
+
redirect/401), authorization failure (wrong user — expect
|
|
52
|
+
403/404 depending on the app's convention), and strong-parameters
|
|
53
|
+
rejection of an unpermitted field.
|
|
54
|
+
|
|
55
|
+
**Jobs:** assert enqueue with `have_enqueued_job`/`assert_enqueued_with`
|
|
56
|
+
for most tests; only actually run the job
|
|
57
|
+
(`perform_enqueued_jobs`) when the test's purpose is the job's own
|
|
58
|
+
`perform` behavior, and cover what happens if `perform` runs twice for
|
|
59
|
+
jobs that must be idempotent.
|
|
60
|
+
|
|
61
|
+
### Step 3: Write
|
|
62
|
+
|
|
63
|
+
1. Create/extend the spec/test file at the project's own convention
|
|
64
|
+
path, mirroring `app/`'s structure.
|
|
65
|
+
2. Use the project's existing fixture approach (FactoryBot
|
|
66
|
+
`create`/`build`/`build_stubbed`, or fixtures) — do not hand-roll
|
|
67
|
+
ad hoc records when a factory already covers the model.
|
|
68
|
+
3. Stub any external HTTP call at the boundary (WebMock/VCR or the
|
|
69
|
+
project's existing tooling); a test must not make a real network
|
|
70
|
+
call.
|
|
71
|
+
4. Freeze time (`travel_to`, matching the project's convention) for any
|
|
72
|
+
assertion involving a timestamp or elapsed-time calculation.
|
|
73
|
+
5. Never use `sleep` to wait for a job, callback, or async result — use
|
|
74
|
+
`perform_enqueued_jobs`, a Capybara auto-retrying matcher
|
|
75
|
+
(`have_content`), or an explicit polling helper with a timeout.
|
|
76
|
+
|
|
77
|
+
### Step 4: Run and fix
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
bundle exec rspec # or: bin/rails test
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Fix failing tests (max 3 iterations) — fix the test, not the source
|
|
84
|
+
under test, unless the test itself has correctly caught a real bug (say
|
|
85
|
+
so in the report rather than silently changing production code).
|
|
86
|
+
|
|
87
|
+
### Step 5: Report
|
|
88
|
+
|
|
89
|
+
```
|
|
90
|
+
Generated: spec/requests/orders_spec.rb
|
|
91
|
+
- 6 examples (happy path, auth failure, authorization failure,
|
|
92
|
+
unpermitted param), all passing
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Rules
|
|
96
|
+
|
|
97
|
+
- ALWAYS match the project's existing framework (RSpec/Minitest),
|
|
98
|
+
fixture style, and spec layout found in Step 1, not a different
|
|
99
|
+
project's convention.
|
|
100
|
+
- NEVER modify application code — only spec/test files and
|
|
101
|
+
fixtures/factories, unless a test caught a real bug (say so).
|
|
102
|
+
- NEVER use `sleep` to wait for a job or async result.
|
|
103
|
+
- NEVER let a test make a real network call to an external service.
|
|
104
|
+
|
|
105
|
+
## Red Flags
|
|
106
|
+
|
|
107
|
+
| Rationalization | Why it is wrong |
|
|
108
|
+
|---|---|
|
|
109
|
+
| "I'll add `sleep(1)` after enqueuing so the job has time to run" | Non-deterministic and slow; use `perform_enqueued_jobs` to run it synchronously in the test, or `have_enqueued_job` to assert the enqueue itself |
|
|
110
|
+
| "This test hits the real payment API sandbox, that's basically a stub" | Still a real network call in the suite — flaky, slow, and can fail for reasons unrelated to the code under test; stub it with WebMock/VCR instead |
|
|
111
|
+
| "I'll just assert `response.status` and skip checking the body/authorization case" | A request spec that only checks the happy-path status misses exactly the auth/authorization regressions these specs exist to catch |
|
|
112
|
+
| "The job might run twice in production but I'll just test the single-run case" | If the job needs to be idempotent, the test suite is where that guarantee gets proven — add a case that runs `perform` twice and asserts no duplicate effect |
|
|
113
|
+
|
|
114
|
+
## Verification
|
|
115
|
+
|
|
116
|
+
Do not report the work done until all of the following hold:
|
|
117
|
+
|
|
118
|
+
- The spec/test file sits at the project's own convention path, matching
|
|
119
|
+
the fixture/framework style read in Step 1.
|
|
120
|
+
- `bundle exec rspec` (or `bin/rails test`) exits 0 with every generated
|
|
121
|
+
test passing.
|
|
122
|
+
- `git status` shows only spec/test files (and factories/fixtures, if
|
|
123
|
+
touched) added or modified; no application source file changed unless
|
|
124
|
+
a real bug was found and reported as such.
|
|
125
|
+
- No test makes a real network call or synchronizes with `sleep`.
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
{
|
|
2
|
+
"triggers": {
|
|
3
|
+
"positive": [
|
|
4
|
+
"Write specs for this ActiveRecord model's validations",
|
|
5
|
+
"Write a test asserting that hitting PATCH /orders/:id without a signed-in user returns 401 instead of raising",
|
|
6
|
+
"Test that this job gets enqueued when the order is placed",
|
|
7
|
+
"Fix this failing spec that's flaking in CI",
|
|
8
|
+
"Our checkout spec is actually hitting Stripe's charges endpoint when it runs, how do I keep that out of the test",
|
|
9
|
+
"Write Minitest coverage for this model's scopes"
|
|
10
|
+
],
|
|
11
|
+
"negative": [
|
|
12
|
+
"Write Jest tests for this React component",
|
|
13
|
+
"Write pytest tests for this Django view",
|
|
14
|
+
"Implement the feature this spec is supposed to cover",
|
|
15
|
+
"Review this Rails diff for missing test coverage",
|
|
16
|
+
"Fix this failing bundle install so the test suite can boot"
|
|
17
|
+
]
|
|
18
|
+
},
|
|
19
|
+
"scenarios": [
|
|
20
|
+
{
|
|
21
|
+
"id": "no-sleep-for-jobs",
|
|
22
|
+
"prompt": "My test enqueues a job and I want to assert its effect ran. What's the right way to wait for it?",
|
|
23
|
+
"strictness": "high",
|
|
24
|
+
"expected_behavior": [
|
|
25
|
+
{
|
|
26
|
+
"grader": "judge",
|
|
27
|
+
"rubric": "A correct answer runs the job synchronously inside the test with perform_enqueued_jobs (ActiveJob::TestHelper) before asserting its effect, or asserts only that it was enqueued with have_enqueued_job when running it isn't the point -- never a sleep-based wait, whether alone or as extra insurance alongside the real mechanism.",
|
|
28
|
+
"pass_criteria": [
|
|
29
|
+
"Names perform_enqueued_jobs (or have_enqueued_job, if only the enqueue itself is being asserted) as the mechanism, shown concretely rather than described in the abstract.",
|
|
30
|
+
"Explains why this is deterministic compared to a fixed delay -- the job either has actually run or hasn't, rather than depending on timing.",
|
|
31
|
+
"Does not recommend sleep at any point in the wait step."
|
|
32
|
+
],
|
|
33
|
+
"fail_criteria": [
|
|
34
|
+
"Recommends sleep -- as the sole wait mechanism or as extra 'insurance' alongside perform_enqueued_jobs/have_enqueued_job -- anywhere in the answer's wait step. Mentioning sleep only to warn against it is not a failure."
|
|
35
|
+
]
|
|
36
|
+
}
|
|
37
|
+
],
|
|
38
|
+
"calibration": {
|
|
39
|
+
"known_right": "Wrap the triggering code in `perform_enqueued_jobs { ... }` (from `ActiveJob::TestHelper`) so the job actually runs synchronously inside the test instead of just sitting in the queue -- then assert directly on its effect. If you only care that the job was scheduled, not that it ran, `expect { ... }.to have_enqueued_job(ChargeJob).with(order.id)` is enough and doesn't need `perform_enqueued_jobs` at all. Either way, don't reach for `sleep` -- there's nothing actually running in the background to wait for until you explicitly drain the queue.",
|
|
40
|
+
"known_wrong": "Just add `sleep(1)` right after you trigger the code that enqueues the job, then run your assertion against whatever the job was supposed to change -- that gives it enough time to run in the background before you check. If it's flaky in CI bump it up to 2 or 3 seconds.",
|
|
41
|
+
"vague": "Make sure the job actually finishes before you check the result instead of just assuming it ran.",
|
|
42
|
+
"subtle_wrong": "Use `perform_enqueued_jobs` to run the job, but add a short `sleep(0.5)` right before it just to be safe in case there's any async I/O the job itself kicks off that isn't captured by the synchronous run -- that extra buffer doesn't hurt and covers cases perform_enqueued_jobs might miss."
|
|
43
|
+
},
|
|
44
|
+
"anti_patterns": ["sleep"]
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
"id": "stub-external-boundary",
|
|
48
|
+
"prompt": "This feature calls a third-party shipping API to get a rate. How should I test it without hitting the real API?",
|
|
49
|
+
"strictness": "high",
|
|
50
|
+
"expected_behavior": [
|
|
51
|
+
{
|
|
52
|
+
"grader": "judge",
|
|
53
|
+
"rubric": "A correct answer stubs the HTTP call at the boundary (WebMock stub_request or a VCR cassette, matching whichever the project already uses) rather than letting the test make a real network call to any environment of the third-party API, sandbox included.",
|
|
54
|
+
"pass_criteria": [
|
|
55
|
+
"Names a concrete boundary-stubbing mechanism -- a WebMock stub_request or a VCR cassette -- as the fix, not just 'mock it out'.",
|
|
56
|
+
"States plainly that the test must not make a real network call, sandbox environments included.",
|
|
57
|
+
"Says to match whichever stubbing tool the project already has configured rather than introducing a second one."
|
|
58
|
+
],
|
|
59
|
+
"fail_criteria": [
|
|
60
|
+
"Recommends letting the test call a real environment of the third-party API (including a vendor sandbox) instead of stubbing the HTTP call, or recommends skipping/silently passing the assertion when that live call fails."
|
|
61
|
+
]
|
|
62
|
+
}
|
|
63
|
+
],
|
|
64
|
+
"calibration": {
|
|
65
|
+
"known_right": "Stub the HTTP call at the boundary rather than hitting anything real -- if the project already uses WebMock, `stub_request(:get, \"https://api.shipper.example/rates\").to_return(status: 200, body: fixture_json)` covers it; if it's VCR, record a cassette once and replay it in CI. Match whichever the project already has configured rather than introducing the other one. That keeps the test fast, deterministic, and independent of the shipper's actual uptime or rate limits.",
|
|
66
|
+
"known_wrong": "Easiest is to just point the test at the shipper's sandbox/test environment and let it make a real call -- that way you're testing against their actual API contract instead of a fixture that could drift out of date. As long as it's their sandbox and not production, it's fine to leave the network call in the test.",
|
|
67
|
+
"vague": "Don't let the test hit the real API -- fake it somehow so it's not flaky.",
|
|
68
|
+
"subtle_wrong": "Wrap the API client call in a begin/rescue inside the test and, if the sandbox call fails or times out, just skip the assertion for that run -- that way the test still exercises the real integration when the sandbox is up, without failing the build when it's flaky."
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
]
|
|
72
|
+
}
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
{
|
|
2
|
+
"agents": [],
|
|
3
|
+
"note": "honest gate (DeepSeek deepseek-chat runner+judge, strictness high, trials 10, flow 337) ran and failed for all four skills -- trigger accuracy, not behavior content (every behavior scenario scored 0.8 or 1.0). No generated pair ships; pack stays stability: experimental. See governance/eval.json for the recorded reports and W1-stack-catalog.md's Wave 4 batch 5 implementation notes for the diagnosis."
|
|
4
|
+
}
|