@mrciphersmith/keryx 0.3.1 → 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/dist/cli.js +7310 -2471
  2. package/dist/core.js +116 -10
  3. package/package.json +1 -1
  4. package/src/gdskills/bundled/install-manifest.json +349 -2
  5. package/src/gdskills/bundled/rules/core/model-selection.mdc +18 -0
  6. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.md +1 -1
  7. package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.md +1 -1
  8. package/src/gdskills/bundled/skills/planning/interviewer/SKILL.md +1 -1
  9. package/src/gdskills/bundled/skills/quality/deploy/SKILL.md +1 -1
  10. package/src/gdskills/bundled/skills/review/review-jev-comments/SKILL.md +184 -0
  11. package/src/gdskills/bundled/skills/review/review-jev-contract/SKILL.md +193 -0
  12. package/src/gdskills/bundled/skills/review/review-jev-docs/SKILL.md +189 -0
  13. package/src/gdskills/bundled/skills/review/review-jev-risk/SKILL.md +190 -0
  14. package/src/gdskills/bundled/skills/review/review-jev-scenarios/SKILL.md +187 -0
  15. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.detail.md +88 -15
  16. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +4 -4
  17. package/src/gdskills/bundled/stacks/c-cpp/agent-refs.json +4 -0
  18. package/src/gdskills/bundled/stacks/c-cpp/governance/eval.json +1777 -0
  19. package/src/gdskills/bundled/stacks/c-cpp/governance/scout.json +31 -0
  20. package/src/gdskills/bundled/stacks/c-cpp/pack.json +42 -0
  21. package/src/gdskills/bundled/stacks/c-cpp/rules/coding-style.mdc +80 -0
  22. package/src/gdskills/bundled/stacks/c-cpp/rules/patterns.mdc +87 -0
  23. package/src/gdskills/bundled/stacks/c-cpp/rules/security.mdc +90 -0
  24. package/src/gdskills/bundled/stacks/c-cpp/rules/testing.mdc +83 -0
  25. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-build-fix/SKILL.md +153 -0
  26. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-build-fix/evals.json +74 -0
  27. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-code-review/SKILL.md +132 -0
  28. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-code-review/evals.json +73 -0
  29. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-implementation/SKILL.md +151 -0
  30. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-implementation/evals.json +74 -0
  31. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-testing/SKILL.md +152 -0
  32. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-testing/evals.json +74 -0
  33. package/src/gdskills/bundled/stacks/ci-github-gitlab/agent-refs.json +4 -0
  34. package/src/gdskills/bundled/stacks/ci-github-gitlab/governance/eval.json +1295 -0
  35. package/src/gdskills/bundled/stacks/ci-github-gitlab/governance/scout.json +26 -0
  36. package/src/gdskills/bundled/stacks/ci-github-gitlab/pack.json +41 -0
  37. package/src/gdskills/bundled/stacks/ci-github-gitlab/rules/patterns.mdc +77 -0
  38. package/src/gdskills/bundled/stacks/ci-github-gitlab/rules/security.mdc +144 -0
  39. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-build-fix/SKILL.md +121 -0
  40. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-build-fix/evals.json +73 -0
  41. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-code-review/SKILL.md +139 -0
  42. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-code-review/evals.json +73 -0
  43. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-implementation/SKILL.md +147 -0
  44. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-implementation/evals.json +74 -0
  45. package/src/gdskills/bundled/stacks/docker-k8s-terraform/agent-refs.json +4 -0
  46. package/src/gdskills/bundled/stacks/docker-k8s-terraform/governance/eval.json +865 -0
  47. package/src/gdskills/bundled/stacks/docker-k8s-terraform/governance/scout.json +16 -0
  48. package/src/gdskills/bundled/stacks/docker-k8s-terraform/pack.json +46 -0
  49. package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/coding-style.mdc +74 -0
  50. package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/patterns.mdc +81 -0
  51. package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/security.mdc +146 -0
  52. package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/testing.mdc +61 -0
  53. package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-build-fix/SKILL.md +151 -0
  54. package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-build-fix/evals.json +74 -0
  55. package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-review/SKILL.md +135 -0
  56. package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-review/evals.json +76 -0
  57. package/src/gdskills/bundled/stacks/php-laravel/agent-refs.json +4 -0
  58. package/src/gdskills/bundled/stacks/php-laravel/governance/eval.json +1829 -0
  59. package/src/gdskills/bundled/stacks/php-laravel/governance/scout.json +33 -0
  60. package/src/gdskills/bundled/stacks/php-laravel/pack.json +41 -0
  61. package/src/gdskills/bundled/stacks/php-laravel/rules/coding-style.mdc +82 -0
  62. package/src/gdskills/bundled/stacks/php-laravel/rules/patterns.mdc +80 -0
  63. package/src/gdskills/bundled/stacks/php-laravel/rules/security.mdc +80 -0
  64. package/src/gdskills/bundled/stacks/php-laravel/rules/testing.mdc +82 -0
  65. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-build-fix/SKILL.md +143 -0
  66. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-build-fix/evals.json +74 -0
  67. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-code-review/SKILL.md +126 -0
  68. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-code-review/evals.json +76 -0
  69. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-implementation/SKILL.md +140 -0
  70. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-implementation/evals.json +75 -0
  71. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-testing/SKILL.md +124 -0
  72. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-testing/evals.json +74 -0
  73. package/src/gdskills/bundled/stacks/ruby-rails/agent-refs.json +4 -0
  74. package/src/gdskills/bundled/stacks/ruby-rails/governance/eval.json +1673 -0
  75. package/src/gdskills/bundled/stacks/ruby-rails/governance/scout.json +33 -0
  76. package/src/gdskills/bundled/stacks/ruby-rails/pack.json +42 -0
  77. package/src/gdskills/bundled/stacks/ruby-rails/rules/coding-style.mdc +69 -0
  78. package/src/gdskills/bundled/stacks/ruby-rails/rules/patterns.mdc +93 -0
  79. package/src/gdskills/bundled/stacks/ruby-rails/rules/security.mdc +90 -0
  80. package/src/gdskills/bundled/stacks/ruby-rails/rules/testing.mdc +89 -0
  81. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-build-fix/SKILL.md +143 -0
  82. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-build-fix/evals.json +73 -0
  83. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-code-review/SKILL.md +134 -0
  84. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-code-review/evals.json +71 -0
  85. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-implementation/SKILL.md +141 -0
  86. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-implementation/evals.json +72 -0
  87. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-testing/SKILL.md +125 -0
  88. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-testing/evals.json +72 -0
  89. package/src/gdskills/bundled/stacks/sql-db/agent-refs.json +4 -0
  90. package/src/gdskills/bundled/stacks/sql-db/governance/eval.json +1829 -0
  91. package/src/gdskills/bundled/stacks/sql-db/governance/scout.json +30 -0
  92. package/src/gdskills/bundled/stacks/sql-db/pack.json +40 -0
  93. package/src/gdskills/bundled/stacks/sql-db/rules/coding-style.mdc +69 -0
  94. package/src/gdskills/bundled/stacks/sql-db/rules/patterns.mdc +134 -0
  95. package/src/gdskills/bundled/stacks/sql-db/rules/security.mdc +74 -0
  96. package/src/gdskills/bundled/stacks/sql-db/rules/testing.mdc +83 -0
  97. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-build-fix/SKILL.md +147 -0
  98. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-build-fix/evals.json +72 -0
  99. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-code-review/SKILL.md +132 -0
  100. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-code-review/evals.json +73 -0
  101. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-implementation/SKILL.md +153 -0
  102. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-implementation/evals.json +77 -0
  103. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-testing/SKILL.md +129 -0
  104. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-testing/evals.json +73 -0
@@ -0,0 +1,1829 @@
1
+ {
2
+ "schemaVersion": "1.0.0",
3
+ "reports": [
4
+ {
5
+ "schemaVersion": "1.0.0",
6
+ "skillId": "sql-db/sql-db-implementation",
7
+ "strictness": "high",
8
+ "trials": 10,
9
+ "triggerAccuracy": {
10
+ "truePositive": 4,
11
+ "falsePositive": 1,
12
+ "positives": 7,
13
+ "negatives": 7
14
+ },
15
+ "evidence": "authored",
16
+ "scenarios": [
17
+ {
18
+ "id": "trigger-positive-1",
19
+ "kind": "trigger-positive",
20
+ "prompt": "I need to add a required column to a table that already has millions of rows, what's the safe way to do this?",
21
+ "strictness": "high",
22
+ "trials": 1,
23
+ "passes": 1,
24
+ "passRate": 1,
25
+ "passAtK": 1,
26
+ "grader": "trigger-rank-fork-family",
27
+ "status": "ran",
28
+ "deterministic": true
29
+ },
30
+ {
31
+ "id": "trigger-positive-2",
32
+ "kind": "trigger-positive",
33
+ "prompt": "Design an index for a query that filters orders by status and sorts by created_at",
34
+ "strictness": "high",
35
+ "trials": 1,
36
+ "passes": 1,
37
+ "passRate": 1,
38
+ "passAtK": 1,
39
+ "grader": "trigger-rank-fork-family",
40
+ "status": "ran",
41
+ "deterministic": true
42
+ },
43
+ {
44
+ "id": "trigger-positive-3",
45
+ "kind": "trigger-positive",
46
+ "prompt": "I need a migration that adds a foreign key from payments.order_id to orders.id on a 20-million-row table, and it can't take the app down while it runs",
47
+ "strictness": "high",
48
+ "trials": 1,
49
+ "passes": 1,
50
+ "passRate": 1,
51
+ "passAtK": 1,
52
+ "grader": "trigger-rank-fork-family",
53
+ "status": "ran",
54
+ "deterministic": true
55
+ },
56
+ {
57
+ "id": "trigger-positive-4",
58
+ "kind": "trigger-positive",
59
+ "prompt": "This endpoint loops over orders and fetches the user for each one separately -- rewrite it as one query",
60
+ "strictness": "high",
61
+ "trials": 1,
62
+ "passes": 0,
63
+ "passRate": 0,
64
+ "passAtK": 0,
65
+ "grader": "trigger-rank-fork-family",
66
+ "status": "ran",
67
+ "deterministic": true
68
+ },
69
+ {
70
+ "id": "trigger-positive-5",
71
+ "kind": "trigger-positive",
72
+ "prompt": "I'm writing a query that filters by a value from an API request, how should I build it safely?",
73
+ "strictness": "high",
74
+ "trials": 1,
75
+ "passes": 0,
76
+ "passRate": 0,
77
+ "passAtK": 0,
78
+ "grader": "trigger-rank-fork-family",
79
+ "status": "ran",
80
+ "deterministic": true
81
+ },
82
+ {
83
+ "id": "trigger-positive-6",
84
+ "kind": "trigger-positive",
85
+ "prompt": "Add a unique constraint to this column and make sure it doesn't take the table down while it runs",
86
+ "strictness": "high",
87
+ "trials": 1,
88
+ "passes": 0,
89
+ "passRate": 0,
90
+ "passAtK": 0,
91
+ "grader": "trigger-rank-fork-family",
92
+ "status": "ran",
93
+ "deterministic": true
94
+ },
95
+ {
96
+ "id": "trigger-positive-7",
97
+ "kind": "trigger-positive",
98
+ "prompt": "Wrap these two related inserts into orders and wallets so a partial failure can't leave one without the other",
99
+ "strictness": "high",
100
+ "trials": 1,
101
+ "passes": 1,
102
+ "passRate": 1,
103
+ "passAtK": 1,
104
+ "grader": "trigger-rank-fork-family",
105
+ "status": "ran",
106
+ "deterministic": true
107
+ },
108
+ {
109
+ "id": "trigger-negative-1",
110
+ "kind": "trigger-negative",
111
+ "prompt": "Write a Django migration that adds a required field to this model",
112
+ "strictness": "high",
113
+ "trials": 1,
114
+ "passes": 1,
115
+ "passRate": 1,
116
+ "passAtK": 1,
117
+ "grader": "trigger-rank-fork-family",
118
+ "status": "ran",
119
+ "deterministic": true
120
+ },
121
+ {
122
+ "id": "trigger-negative-2",
123
+ "kind": "trigger-negative",
124
+ "prompt": "Review this Rails ActiveRecord migration for safety before we merge it",
125
+ "strictness": "high",
126
+ "trials": 1,
127
+ "passes": 1,
128
+ "passRate": 1,
129
+ "passAtK": 1,
130
+ "grader": "trigger-rank-fork-family",
131
+ "status": "ran",
132
+ "deterministic": true
133
+ },
134
+ {
135
+ "id": "trigger-negative-3",
136
+ "kind": "trigger-negative",
137
+ "prompt": "Review this Go function for SQL injection risk",
138
+ "strictness": "high",
139
+ "trials": 1,
140
+ "passes": 1,
141
+ "passRate": 1,
142
+ "passAtK": 1,
143
+ "grader": "trigger-rank-fork-family",
144
+ "status": "ran",
145
+ "deterministic": true
146
+ },
147
+ {
148
+ "id": "trigger-negative-4",
149
+ "kind": "trigger-negative",
150
+ "prompt": "Implement a new REST endpoint in this Express app that lists orders",
151
+ "strictness": "high",
152
+ "trials": 1,
153
+ "passes": 1,
154
+ "passRate": 1,
155
+ "passAtK": 1,
156
+ "grader": "trigger-rank-fork-family",
157
+ "status": "ran",
158
+ "deterministic": true
159
+ },
160
+ {
161
+ "id": "trigger-negative-5",
162
+ "kind": "trigger-negative",
163
+ "prompt": "Write a MongoDB aggregation pipeline that joins two collections",
164
+ "strictness": "high",
165
+ "trials": 1,
166
+ "passes": 1,
167
+ "passRate": 1,
168
+ "passAtK": 1,
169
+ "grader": "trigger-rank-fork-family",
170
+ "status": "ran",
171
+ "deterministic": true
172
+ },
173
+ {
174
+ "id": "trigger-negative-6",
175
+ "kind": "trigger-negative",
176
+ "prompt": "Write a Prisma schema migration that adds this field to the model",
177
+ "strictness": "high",
178
+ "trials": 1,
179
+ "passes": 0,
180
+ "passRate": 0,
181
+ "passAtK": 0,
182
+ "grader": "trigger-rank-fork-family",
183
+ "status": "ran",
184
+ "deterministic": true
185
+ },
186
+ {
187
+ "id": "trigger-negative-7",
188
+ "kind": "trigger-negative",
189
+ "prompt": "Fix this failing pytest test for the order validator",
190
+ "strictness": "high",
191
+ "trials": 1,
192
+ "passes": 1,
193
+ "passRate": 1,
194
+ "passAtK": 1,
195
+ "grader": "trigger-rank-fork-family",
196
+ "status": "ran",
197
+ "deterministic": true
198
+ },
199
+ {
200
+ "id": "safe-not-null-backfill",
201
+ "kind": "behavior",
202
+ "prompt": "I need to add a NOT NULL column to a Postgres table that already has millions of rows. What's the safe way to do this without an outage?",
203
+ "strictness": "high",
204
+ "trials": 10,
205
+ "passes": 8,
206
+ "passRate": 0.8,
207
+ "passAtK": 1,
208
+ "grader": "judge",
209
+ "status": "ran",
210
+ "trialRecords": [
211
+ {
212
+ "output": "The unsafe version, for contrast: `ALTER TABLE orders ADD COLUMN priority integer NOT NULL DEFAULT 1;`\nOn modern Postgres (11+) adding a column with a *constant* default is metadata-only and fast — but it still takes an `ACCESS EXCLUSIVE` lock for the instant of the change, and the bigger trap is when the default isn't constant, or when you then need to backfill real per-row values rather than one constant. The sequence below avoids both the lock-duration and backfill-bloat problems.\n\n**Step 1 — Add the column nullable, no default (fast, metadata-only):**\n```sql\nALTER TABLE orders ADD COLUMN priority integer;\n```\nThis is O(1). No table rewrite, lock held only momentarily.\n\n**Step 2 — Add the constraint `NOT VALID` (fast):**\n```sql\nALTER TABLE orders\n ADD CONSTRAINT orders_priority_not_null\n CHECK (priority IS NOT NULL) NOT VALID;\n```\n`NOT VALID` means Postgres records the constraint but does **not** scan existing rows. `ACCESS EXCLUSIVE` is held only briefly. New writes are checked against it immediately, so no new NULLs can sneak in.\n\n**Step 3 — Backfill in batches, not one giant `UPDATE`:**\n```sql\n-- repeat until 0 rows affected\nUPDATE orders\nSET priority = 1\nWHERE id IN (\n SELECT id FROM orders\n WHERE priority IS NULL\n LIMIT 10000\n);\n```\nA single `UPDATE ... WHERE priority IS NULL` over millions of rows holds row locks for the whole run and bloats the table with dead tuples. Batch it, and add a short sleep between batches so autovacuum keeps up. This must be its own migration file, separate from the schema change — see the \"never combine a blocking change with a slow backfill\" rule.\n\n**Step 4 — Validate the constraint (no blocking write lock):**\n```sql\nALTER TABLE orders VALIDATE CONSTRAINT orders_priority_not_null;\n```\nThis scans the table but takes only `SHARE UPDATE EXCLUSIVE` — normal reads and writes proceed concurrently. Now that it's validated, the planner also knows the column is non-null.\n\n**Step 5 — Optionally convert to a true `NOT NULL` (optional, faster now):**\n```sql\nALTER TABLE orders ALTER COLUMN priority SET NOT NULL;\n```\nBecause the validated `CHECK` constraint already proves no NULLs exist, Postgres skips the full-table scan here.\n\n**Important caveat:** steps 3 must finish inserting non-NULL values *before* any code path relies on the column. If the application still writes rows without setting `priority`, those inserts will fail once validation is attempted — coordinate the deploy: add column (deploy 1), start writing the value in application code, backfill, then validate (deploy 2).\n\n**A note on batching volume:** run the backfill during a lower-traffic window regardless, since each batch still generates WAL and dead tuples. Tune the `LIMIT` so a batch completes in well under a second on your hardware.\n\nTo verify: `EXPLAIN (ANALYZE, BUFFERS)` on your backfill query confirms it uses the PK index rather than a seq scan, and you can confirm the constraint state with `SELECT convalidated FROM pg_constraint WHERE conname = 'orders_priority_not_null';` — `true` means you're clear to proceed to step 5.",
213
+ "outputSha256": "629189aa4222f540b2177e2aa2755d58d801d1d7e69f03eb3e437303072fba90",
214
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
215
+ "deterministic": [],
216
+ "judge": {
217
+ "verdict": "pass",
218
+ "reason": "Step 1 adds column nullable (ALTER TABLE orders ADD COLUMN priority integer), Step 2 adds CHECK (priority IS NOT NULL) NOT VALID, Step 3 backfills in batches with LIMIT 10000, Step 4 VALIDATE CONSTRAINT. Caveat notes app code must write the value before validation. All pass criteria met; no fail criteria."
219
+ },
220
+ "passed": true
221
+ },
222
+ {
223
+ "output": "# Adding a `NOT NULL` column to a large Postgres table without an outage\n\nThe single-statement form `ALTER TABLE orders ADD COLUMN status text NOT NULL DEFAULT 'pending'` **rewrites the whole table under an `ACCESS EXCLUSIVE` lock** on Postgres before v11, and even on v11+ (where a constant default avoids the rewrite) the `SET NOT NULL` and any existing-row backfill still block. On a table with millions of rows that holds the lock long enough to stall every reader and writer — that's the outage.\n\nSplit it into a sequence of individually non-blocking steps. Run each step as its **own migration/transaction**, not all in one.\n\n## The sequence\n\n### 1. Add the column as nullable, with a constant default\n\n```sql\nALTER TABLE orders ADD COLUMN status text; -- no default yet\n```\n\nMetadata-only on all supported Postgres versions. Do not add `NOT NULL` or a default in this step.\n\nIf you want a default for *future* rows, add it once the column is in place — on Postgres 11+ this is metadata-only too:\n\n```sql\nALTER TABLE orders ALTER COLUMN status SET DEFAULT 'pending';\n```\n\n(A *volatile* default like `now()` or `gen_random_uuid()` forces a rewrite even on 11+ — use a constant, or handle the value in the backfill.)\n\n### 2. Backfill existing rows in batches\n\nDo **not** do `UPDATE orders SET status = 'pending'` in one statement — that takes a long-lived row lock on every row and bloats the table in one shot. Batch it, and commit between batches so locks release:\n\n```sql\n-- run repeatedly until 0 rows are affected\nWITH batch AS (\n SELECT id\n FROM orders\n WHERE status IS NULL\n ORDER BY id\n LIMIT 10000\n)\nUPDATE orders o\nSET status = 'pending'\nFROM batch b\nWHERE o.id = b.id;\n```\n\nBetween batches, pause briefly so autovacuum and normal traffic keep up. Make the loop idempotent (`WHERE status IS NULL`) so it can be safely re-run after an interruption.\n\n### 3. Enforce NOT NULL without a full-table lock\n\n```sql\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null\n NOT NULL (status) NOT VALID; -- fast: does not scan existing rows\n\nALTER TABLE orders\n VALIDATE CONSTRAINT orders_status_not_null; -- scans with only SHARE UPDATE EXCLUSIVE\n```\n\nThe `NOT VALID` form skips the initial scan; `VALIDATE CONSTRAINT` acquires the **weakest** lock (`SHARE UPDATE EXCLUSIVE`), which does not block reads or writes. This is the whole trick — it's how you get `NOT NULL` semantics without the `ACCESS EXCLUSIVE` scan.\n\n(On Postgres 12+, a plain `ALTER TABLE ... SET NOT NULL` can sometimes skip the scan if a *matching* `CHECK (col IS NOT NULL)` constraint already exists and has been validated — but the `NOT VALID` + `VALIDATE` chain is the reliable form across versions.)\n\n### 4. Add the index if the query needs one\n\nIf queries will filter on `status`, add the index **concurrently**, in its own step, outside a transaction:\n\n```sql\nCREATE INDEX CONCURRENTLY idx_orders_status_created_at\n ON orders (status, created_at);\n```\n\n`CREATE INDEX CONCURRENTLY` must run outside a transaction block and cannot be combined with other statements in the same migration.\n\n## Why each step is outage-safe\n\n| Step | Lock taken | Blocks reads/writes? |\n|---|---|---|\n| `ADD COLUMN` (nullable) | `ACCESS EXCLUSIVE`, metadata-only | No — held for microseconds |\n| `SET DEFAULT` (constant, PG11+) | `ACCESS EXCLUSIVE`, metadata-only | No |\n| Batched `UPDATE` | row locks on the batch only | No — other rows still writable |\n| `ADD CONSTRAINT ... NOT VALID` | brief `ACCESS EXCLUSIVE` | No — no scan performed |\n| `VALIDATE CONSTRAINT` | `SHARE UPDATE EXCLUSIVE` | No |\n| `CREATE INDEX CONCURRENTLY` | `SHARE UPDATE EXCLUSIVE` | No |\n\n## Ordering and safety notes\n\n- **Deploy order matters.** Make the column nullable and write both old (`NULL`) and new (`'pending'`) values from the app *before* you enforce `NOT NULL`, otherwise inserts from the still-running old code version will fail the constraint. Backfill, then flip `NOT NULL` once no writer can produce a `NULL`.\n- **Set a short `lock_timeout`** on the `ACCESS EXCLUSIVE` steps so that if the lock *is* momentarily held (e.g. a long open transaction), your migration aborts and retries rather than queuing behind it and blocking everything behind *you*:\n ```sql\n SET lock_timeout = '3s';\n ```\n- **Do not combine steps in one transaction.** A validated `NOT VALID` constraint and a `CONCURRENTLY` index both depend on *not* being wrapped with the earlier steps.\n- **Verify** with `EXPLAIN (ANALYZE, BUFFERS)` on the affected queries after the index lands, confirming an Index Scan rather than a Seq Scan.\n- On **MySQL** the equivalent is `ALTER TABLE ... ALGORITHM=INPLACE` (or the `gh-ost`/`pt-online-schema-change` tools for operations that can't be done in place); the mechanics differ, so pick the tooling that matches your engine rather than translating this sequence literally.",
224
+ "outputSha256": "9f9c48469db99624c39ac1810c70d514310d059782b5ce6f764bb9d288da7018",
225
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
226
+ "deterministic": [],
227
+ "judge": {
228
+ "verdict": "fail",
229
+ "reason": "The answer's step 3 uses invalid syntax: `ADD CONSTRAINT ... NOT NULL (status) NOT VALID` is not a real Postgres form; the correct non-blocking way is `ADD CONSTRAINT ... CHECK (status IS NOT NULL) NOT VALID` then `VALIDATE CONSTRAINT`. So pass criterion 3's concrete CHECK-constraint mechanism is not actually shown, even though the NOT VALID/VALIDATE idea is gestured at."
230
+ },
231
+ "passed": false
232
+ },
233
+ {
234
+ "output": "Short answer: never do `ALTER TABLE ... ADD COLUMN ... NOT NULL` in one shot on a big table, and never do it with a volatile default. On Postgres the `SET NOT NULL` step is the dangerous one — it takes an `ACCESS EXCLUSIVE` lock and full-table-scans to prove no row is null, and a volatile default (like `now()`) makes the `ADD COLUMN` itself rewrite the whole table. Split it into three migrations so no single step holds a long lock.\n\n**Migration 1 — add the column nullable, no default (or a constant default)**\n```sql\n-- Fast on modern Postgres (11+): a constant default is stored in catalog\n-- metadata, no table rewrite. A volatile default (now(), random()) forces a\n-- full rewrite — don't use one here.\nALTER TABLE orders ADD COLUMN status text; -- nullable for now\n```\nIf you *do* want a constant default, `ADD COLUMN status text NOT NULL DEFAULT 'pending'` is actually safe on PG 11+, but I'd still separate the backfill for a non-trivial default so the data path is explicit. Never put a volatile default in the ADD COLUMN.\n\n**Migration 2 — backfill in bounded batches (do not do one giant UPDATE)**\nA single `UPDATE orders SET status = 'pending' WHERE status IS NULL` takes row locks on millions of rows and bloats the table. Batch it:\n```sql\nDO $$\nDECLARE\n rows_updated int;\nBEGIN\n LOOP\n UPDATE orders\n SET status = 'pending'\n WHERE id IN (\n SELECT id FROM orders WHERE status IS NULL LIMIT 5000\n );\n GET DIAGNOSTICS rows_updated = ROW_COUNT;\n EXIT WHEN rows_updated = 0;\n COMMIT; -- requires running outside an implicit txn; see note\n END LOOP;\nEND $$;\n```\nRun this as a standalone job (e.g. a script with its own transaction per batch), not inside the migration transaction, and throttle it so it doesn't starve normal traffic. Add a partial index to make the \"find nulls\" scan cheap while this runs, then drop it:\n```sql\nCREATE INDEX CONCURRENTLY idx_orders_status_nulls ON orders (id) WHERE status IS NULL;\n```\n\n**Migration 3 — enforce NOT NULL without the long lock**\n```sql\n-- Add as NOT VALID: instant, takes only a brief lock, applies to new rows only.\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;\n\n-- Validate separately: takes a SHARE UPDATE EXCLUSIVE lock, allows reads/writes.\nALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n\n-- Now SET NOT NULL is a fast metadata-only change, because the validated\n-- CHECK constraint already proves it for the planner.\nALTER TABLE orders ALTER COLUMN status SET NOT NULL;\n\n-- Clean up the now-redundant CHECK constraint.\nALTER TABLE orders DROP CONSTRAINT orders_status_not_null;\n```\nThe `NOT VALID` + `VALIDATE CONSTRAINT` + `SET NOT NULL` trio is the key trick — it converts the expensive null-proof scan into a non-blocking operation.\n\n**After it's live:**\n```sql\nEXPLAIN (ANALYZE, BUFFERS) <your queries that filter/sort on status>;\n```\nConfirm the plan uses the intended index (Index Scan, not Seq Scan) and that none of your queries are doing per-row lookups in a loop — batch those into a single `JOIN`/`IN (...)`.\n\n**Sequencing/rollout notes**\n- Ship the three migrations as separate files/steps, not one transaction. Between step 2 and 3 the column is populated but not yet enforced — that's the intentionally safe window.\n- Do the backfill at low traffic, and watch replication lag / bloat while it runs.\n- If your app is deployed continuously, deploy code that tolerates `status IS NULL` before migration 1, and only rely on `NOT NULL` after migration 3 is fully applied.\n- Postgres 12+ has more fast-path defaults, but the `NOT VALID` → `VALIDATE` → `SET NOT NULL` pattern is the one that reliably avoids the lock regardless.\n- If you're on MySQL instead, the equivalent is adding the column as nullable, backfilling in batches, then `ALTER ... MODIFY ... NOT NULL ALGORITHM=INPLACE, LOCK=NONE` — and verify the engine reports the in-place form was actually used.\n\nIf you tell me the table name, the column, and whether you want a default, I can write the exact three migration files (up/down) for you.",
235
+ "outputSha256": "6d6414afa0718142dc117f1f3455918f0549630b4a229ac7c78cbaf640302893",
236
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
237
+ "deterministic": [],
238
+ "judge": {
239
+ "verdict": "pass",
240
+ "reason": "Answer shows nullable ADD COLUMN first (Migration 1) with app code tolerating NULL before enforcement; Migration 2 backfills in bounded batches (LIMIT 5000 loop, not one giant UPDATE); Migration 3 uses ADD CONSTRAINT ... CHECK (status IS NOT NULL) NOT VALID then separate VALIDATE CONSTRAINT. No direct SET NOT NULL as the enforcement mechanism; it only follows the validated CHECK."
241
+ },
242
+ "passed": true
243
+ },
244
+ {
245
+ "output": "Safe way to add a `NOT NULL` column to a large, live Postgres table: **split it into three separate migrations** — do not do it in one `ALTER TABLE`. A single `ADD COLUMN ... NOT NULL DEFAULT ...` (or `SET NOT NULL` on a backfilled column) takes an `ACCESS EXCLUSIVE` lock and, if there's a backfill, holds it for the duration of a full-table rewrite.\n\n## The three-step sequence\n\n**Step 1 — add the column nullable, with a default**\n\n```sql\nALTER TABLE orders\n ADD COLUMN status text;\n```\n\nIf every new row should get an automatic value going forward, add a *constant* default in the same statement — since Postgres 11 a constant (non-volatile) default is stored as table metadata and does **not** rewrite the table, so this is fast even on millions of rows:\n\n```sql\nALTER TABLE orders\n ADD COLUMN status text NOT NULL DEFAULT 'pending';\n```\n\nCaveat: this metadata-only fast path does **not** apply if you're on Postgres 10 or older, or if the \"default\" is actually a volatile expression (e.g. `now()`, a subquery). In those cases the default triggers a full rewrite — leave the column with no default, backfill explicitly, then add a separate default later.\n\n**Step 2 (separate migration) — backfill existing rows in batches**\n\nNever `UPDATE orders SET status = 'pending'` in one statement against a large table: it takes a long-running row lock pattern that blocks vacuum and can balloon WAL/bloat. Batch it, and commit between batches so no transaction stays open for the whole job:\n\n```sql\n-- run repeatedly until it affects 0 rows; commit between each run\nUPDATE orders\n SET status = 'pending'\n WHERE id IN (\n SELECT id FROM orders\n WHERE status IS NULL\n LIMIT 10000\n );\n```\n\nThen add a lightweight validity constraint as a guard, using the non-blocking form:\n\n```sql\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;\n\nALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n```\n\n`NOT VALID` is the key: adding it takes only a brief lock, and `VALIDATE CONSTRAINT` scans the table with a `SHARE UPDATE EXCLUSIVE` lock that does **not** block reads or writes. This lets you prove \"no nulls remain\" without a blocking `SET NOT NULL`.\n\n**Step 3 (separate migration) — set NOT NULL and drop the temporary check**\n\n```sql\nALTER TABLE orders ALTER COLUMN status SET NOT NULL;\nALTER TABLE orders DROP CONSTRAINT orders_status_not_null;\n```\n\nBecause the validated CHECK constraint already proves no nulls exist, `SET NOT NULL` now completes almost instantly instead of scanning the table.\n\n## Why not one statement\n\n- `ALTER TABLE ... SET NOT NULL` alone, without a validated CHECK constraint present, takes `ACCESS EXCLUSIVE` and full-scans the table to verify — blocking all reads and writes for the duration.\n- A backfill bundled into the same migration file as the `SET NOT NULL` means one long transaction holding a lock the whole time. Split files so each locks briefly.\n\n## Also apply here\n\n- **Index it if it's filtered/sorted on.** If `status` will appear in a `WHERE`/`JOIN`/`ORDER BY`, add the index in the same change — but on a live table use `CREATE INDEX CONCURRENTLY` (outside a transaction), not a plain `CREATE INDEX`, which blocks writes. For composite order, match the query shape (equality columns first, then range/sort column), e.g. `(status, created_at)`.\n- **Parameterize the backfill's value** if it's ever injected via code rather than a literal — no string interpolation.\n- **Verify.** After the backfill, `SELECT count(*) FROM orders WHERE status IS NULL` must be 0 before Step 3. After the index, run `EXPLAIN (ANALYZE, BUFFERS)` on the target query and confirm it uses an Index Scan, not a Seq Scan.\n\n**Report shape when done:**\n```\nImplemented (3 migrations):\n add_orders_status_column -- nullable/default add, no rewrite\n backfill_orders_status -- batched UPDATE + CHECK NOT VALID + VALIDATE\n enforce_orders_status_not_null -- SET NOT NULL + drop temp check\n + CREATE INDEX CONCURRENTLY (status, created_at)\nVerified: null count = 0; EXPLAIN ANALYZE shows Index Scan.\n```\n\nIf the table is genuinely new/empty in the same change, none of this applies — a single `ADD COLUMN ... NOT NULL` is fine. The multi-step sequence is for tables that already have live rows and ongoing traffic.",
246
+ "outputSha256": "10177372aa482ce806e4fd5a8b6bae32fa14b8a971b26ede1b873f4dc970c834",
247
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
248
+ "deterministic": [],
249
+ "judge": {
250
+ "verdict": "pass",
251
+ "reason": "Step 1 adds column nullable (ALTER TABLE orders ADD COLUMN status text) with app writing it going forward. Step 2 backfills in batches (LIMIT 10000, commit between runs), explicitly rejecting single UPDATE. Step 2 also names ADD CONSTRAINT ... CHECK (status IS NOT NULL) NOT VALID then VALIDATE CONSTRAINT. No fail criteria: SET NOT NULL only after validated check, not as enforcement mechanism."
252
+ },
253
+ "passed": true
254
+ },
255
+ {
256
+ "output": "# Adding a NOT NULL column to a large Postgres table\n\nThe one-shot `ALTER TABLE ... ADD COLUMN ... NOT NULL DEFAULT ...` is only safe in a narrow case; for millions of existing rows the safe path is a multi-step sequence. Here's the decision tree and the mechanics.\n\n## Step 0: Determine your Postgres version\n\n- **Postgres 11+**: `ADD COLUMN ... DEFAULT <constant>` no longer rewrites the table. A constant default is stored as metadata, so it's fast even on millions of rows. This changes the answer significantly.\n- **Postgres < 11**: adding a column with any default rewrites the whole table under an `ACCESS EXCLUSIVE` lock. You must not do this directly — you add the column nullable + no default, then backfill, then set the default.\n\nConfirm with `SELECT version();` before choosing the path.\n\n## The safe sequence\n\nAssume Postgres 11+ and a column that needs a default / needs to end up `NOT NULL`.\n\n### 1. Add the column as nullable, with a constant default (or no default)\n\n```sql\nALTER TABLE orders\n ADD COLUMN status text; -- no default yet; add separately if desired\n```\n\nThis takes an `ACCESS EXCLUSIVE` lock but is fast in 11+ (metadata only). Keep the statement alone in its own migration — do not combine it with a backfill.\n\nIf a default is needed, add it with a **constant** (not `now()`, not a subquery):\n\n```sql\nALTER TABLE orders ALTER COLUMN status SET DEFAULT 'pending';\n```\n\nA non-constant default forces a rewrite on every subsequent `ADD COLUMN`-with-default in some versions, and a volatile default on the column itself causes a rewrite on add in 11+ too. Use a constant, or backfill separately.\n\n### 2. Add the `NOT NULL` constraint as `NOT VALID`-equivalent\n\nPostgres does not allow `ADD CONSTRAINT ... NOT NULL ... NOT VALID` directly for a `NOT NULL` constraint — the idiomatic approach is a `CHECK` constraint added `NOT VALID`, validated separately, then promoted to `NOT NULL`:\n\n```sql\n-- fast: takes ACCESS EXCLUSIVE briefly, does not scan existing rows\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null\n CHECK (status IS NOT NULL) NOT VALID;\n\n-- slow part, but only takes SHARE UPDATE EXCLUSIVE:\n-- does not block reads or writes\nALTER TABLE orders\n VALIDATE CONSTRAINT orders_status_not_null;\n\n-- now promote to a real NOT NULL (fast, uses the validated check)\nALTER TABLE orders\n ALTER COLUMN status SET NOT NULL;\n\n-- optional cleanup\nALTER TABLE orders\n DROP CONSTRAINT orders_status_not_null;\n```\n\n`VALIDATE CONSTRAINT` scans the table but holds only `SHARE UPDATE EXCLUSIVE`, so concurrent reads and writes proceed. Splitting add-constraint from validate-constraint across two migrations is what keeps the window short.\n\n### 3. Backfill slowly, in batches\n\nBackfill **before** validation, in batches, outside a single giant transaction:\n\n```sql\n-- repeat until 0 rows affected\nUPDATE orders\nSET status = 'pending'\nWHERE status IS NULL\n AND id IN (\n SELECT id FROM orders WHERE status IS NULL LIMIT 10000\n );\n```\n\nRun in a loop from application code or a maintenance script. Key points:\n\n- **Batch by a key, not `LIMIT` alone** — a bare `LIMIT` without `ORDER BY` can rescan the same rows. Order by `id` or another indexed key and advance a cursor.\n- **Sleep between batches** to let autovacuum and replication keep up (e.g. 100–500ms).\n- **Watch replication lag** if you have replicas — large update storms cause WAL to pile up.\n- Each batch is its own transaction so a failure doesn't roll back the whole backfill and doesn't hold a lock for the entire duration.\n\n### 4. Repeat until no `NULL`s remain\n\nOnly once the backfill reports zero remaining `NULL`s should you run `VALIDATE CONSTRAINT` and `SET NOT NULL` from step 2. Validating with rows still `NULL` will fail.\n\n## If you're on Postgres < 11\n\nThe `DEFAULT` on `ADD COLUMN` is the rewrite trigger. Sequence:\n\n1. `ALTER TABLE orders ADD COLUMN status text;` (nullable, no default — fast).\n2. Backfill in batches as above.\n3. `ALTER TABLE orders ALTER COLUMN status SET DEFAULT 'pending';` (constant, cheap).\n4. Add `NOT VALID` CHECK → `VALIDATE CONSTRAINT` → `SET NOT NULL` → drop check, as above.\n\n## Indexing\n\nIf the new column needs to be filtered or sorted on, add the index concurrently, separate from the `ALTER TABLE`:\n\n```sql\nCREATE INDEX CONCURRENTLY idx_orders_status\n ON orders (status);\n```\n\n`CONCURRENTLY` avoids the `ACCESS EXCLUSIVE` lock that a plain `CREATE INDEX` holds for the whole build. It cannot run inside a transaction block — put it in its own migration. If it fails, it leaves an `INVALID` index; drop it (`DROP INDEX CONCURRENTLY`) and retry.\n\n## Transaction and migration-splitting discipline\n\n- **One blocking concern per migration file.** Add-column, backfill script, validate-constraint, and concurrent-index each go in their own file/step so a failure in one doesn't roll back a long backfill.\n- **Never hold a transaction open across the batch loop.** Each batch commits independently.\n- **Set a lock timeout on the `ALTER TABLE` statements** so they don't queue behind a long-running query and then block everything behind them:\n ```sql\n SET lock_timeout = '3s';\n ALTER TABLE orders ADD COLUMN status text;\n ```\n If it times out, retry — better than stalling the whole table's traffic.\n\n## Verification before calling it done\n\n```sql\n-- should show the constraint, validated\nSELECT conname, convalidated FROM pg_constraint\nWHERE conrelid = 'orders'::regclass;\n\n-- what actually uses the index you added\nEXPLAIN (ANALYZE, BUFFERS) SELECT ... FROM orders WHERE status = 'pending';\n```\n\nConfirm the `EXPLAIN` shows an `Index Scan`/`Index-Only Scan`, not a `Seq Scan`, and that `convalidated` is `true` on the promoted constraint.\n\n## Quick summary of the answer\n\nAdd the column **nullable** first. Backfill in **batched, autocommit** updates with sleeps. Add the `NOT NULL` as a **`NOT VALID` CHECK**, then **`VALIDATE CONSTRAINT`** (non-blocking), then **`SET NOT NULL`** (fast, since the check is already validated), then drop the check. Set **`lock_timeout`** on the DDL. Build any needed index with **`CREATE INDEX CONCURRENTLY`** in its own migration. Skip the long rewrite entirely on Postgres 11+.",
257
+ "outputSha256": "925027a2c07cffc609174832edad001dfa97021fd39e45db12b63acbb5c47bcd",
258
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
259
+ "deterministic": [],
260
+ "judge": {
261
+ "verdict": "pass",
262
+ "reason": "Answer adds column nullable first (step 1, 'ADD COLUMN status text'), has app write it going forward, backfills in batches (step 3, LIMIT 10000 loop with sleeps, not one UPDATE), and enforces NOT NULL via ADD CONSTRAINT CHECK (status IS NOT NULL) NOT VALID then separate VALIDATE CONSTRAINT then SET NOT NULL. No fail criteria: no plain SET NOT NULL as the enforcement path, no single unbatched UPDATE."
263
+ },
264
+ "passed": true
265
+ },
266
+ {
267
+ "output": "# Adding a NOT NULL column to a large Postgres table safely\n\nThe short answer: **add it nullable with no default, backfill it in batches, then add the constraint as `NOT VALID` and `VALIDATE` it separately.** Never do a single `ALTER TABLE ... ADD COLUMN ... NOT NULL ... DEFAULT`. On several million rows that statement takes an `ACCESS EXCLUSIVE` lock for the duration of the rewrite and blocks reads and writes — that is your outage.\n\nHere is the full sequence.\n\n## Step 0 — Decide whether you even need the full ceremony\n\n- **Adding a `NOT NULL` column with a *constant* default (Postgres 11+)** is already non-blocking and metadata-only — the engine stores the default and materializes it lazily, so existing rows are not rewritten. If your value is a constant and you're on PG 11+, a single `ALTER TABLE ... ADD COLUMN x int NOT NULL DEFAULT 0` is fine. Verify your version first.\n- **Adding a `NOT NULL` column whose value must be *computed per row*** (or whose default is volatile, like `now()`) is the case that needs everything below. This is the common real-world case, so I'll assume it.\n\n## Step 1 — Add the column nullable, with no default\n\n```sql\nALTER TABLE orders ADD COLUMN status text;\n```\n\nAdding a nullable column with no default is a metadata-only change in PostgreSQL — no table rewrite, and the `ACCESS EXCLUSIVE` lock is held for only the instant it takes to update the catalog. This is safe to run online.\n\n## Step 2 — Backfill in batches, outside the schema migration\n\nDo **not** run `UPDATE orders SET status = ...;` as a single statement. On millions of rows that takes a long transaction, holds row locks, generates enormous WAL, and blocks vacuum from cleaning up dead tuples — it can bloat the table badly and cause a runaway.\n\nInstead, backfill in bounded batches, each in its own short transaction:\n\n```sql\nUPDATE orders\nSET status = 'legacy'\nWHERE status IS NULL\n AND id IN (\n SELECT id FROM orders\n WHERE status IS NULL\n LIMIT 10000\n );\n```\n\nLoop this until it reports zero rows affected, committing between batches. A few notes that matter at scale:\n\n- **Batch by a keyed predicate, not `LIMIT` alone**, or you'll re-scan the head of the table each time. Keying on `id > last_seen_id` (or paging on an indexed column) keeps each batch's scan cheap.\n- **Pace it.** Add a short sleep between batches, or run it as a low-priority background job, so you don't saturate I/O or replication. On a primary with replicas, this backfill is now your replication lag driver — monitor it.\n- **Each batch commits independently**, so the table never holds one giant transaction and autovacuum can keep up.\n\nFor very large tables, a common refinement is a trigger on new inserts/updates to keep newly-written rows populated while the backfill catches up, so you don't race a moving target. Add that trigger before starting the backfill if the table is taking live writes.\n\n## Step 3 — Add the `NOT NULL` constraint as `NOT VALID`\n\n`ALTER TABLE ... SET NOT NULL` normally scans the whole table under `ACCESS EXCLUSIVE`. The non-blocking substitute is a `CHECK` constraint added `NOT VALID` (which skips the validation scan, so it's fast), then validated separately:\n\n```sql\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null\n CHECK (status IS NOT NULL) NOT VALID;\n```\n\n`NOT VALID` means the constraint is enforced for *new* rows immediately but existing rows are not yet certified — so the statement returns immediately without scanning.\n\n## Step 4 — Validate the constraint\n\n```sql\nALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n```\n\nValidation takes only a `SHARE UPDATE EXCLUSIVE` lock — it does **not** block reads or writes, so this can run online. It scans the table once to confirm every row satisfies the check. If step 2 wasn't fully finished, this fails loudly rather than silently letting bad data through, which is what you want.\n\n## Step 5 — (Optional) Promote to a true `NOT NULL`\n\nIn modern PostgreSQL (12+), a validated `CHECK (col IS NOT NULL)` is enough for the planner to treat the column as not-null for many purposes. If you specifically want the `attnotnull` catalog flag set, you can now run:\n\n```sql\nALTER TABLE orders ALTER COLUMN status SET NOT NULL;\n```\n\nBecause the validated check already proves no NULLs exist, this step is fast on PG 12+ (it can use the existing constraint's proof) rather than a full scan. Confirm your version's behavior — on older versions this still scans, in which case the validated `CHECK` is the better end state and you should just keep it.\n\n## If you also want a DEFAULT\n\nSet the default *after* the column exists and is backfilled, so no rewrite is triggered:\n\n```sql\nALTER TABLE orders ALTER COLUMN status SET DEFAULT 'active';\n```\n\nIn PG 11+ this is metadata-only.\n\n## Rollback / safety notes\n\n- **Each step is independently reversible.** Dropping the `CHECK` constraint, or the column, are both fast; nothing in this sequence is a one-way door until you drop the nullable column you replaced.\n- **Run each DDL step as its own migration file**, not bundled into one script. If step 4 is slow, you want it to be a distinct, observable unit of work you can retry without re-running steps 1–5.\n- **`SET lock_timeout` before each DDL statement** (e.g. `SET lock_timeout = '3s'`). If anything is holding a conflicting lock, you'd rather the migration fail fast and retry than queue up behind it and block every subsequent query — a queued `ACCESS EXCLUSIVE` request blocks everything behind it even before it acquires the lock.\n- **Verify before and after:** `SELECT count(*) FROM orders WHERE status IS NULL;` should be 0 after step 2, and step 4 should succeed. After deploying, run `EXPLAIN (ANALYZE, BUFFERS)` on whatever query actually filters on `status` to confirm the plan uses an index rather than a sequential scan — and add a partial or composite index on `status` in the same change if the query needs one.\n\n## The one-line summary\n\nNullable add (instant) → batched backfill (many short transactions) → `CHECK ... NOT VALID` (instant) → `VALIDATE CONSTRAINT` (non-blocking scan). The only steps that touch every row are steps 2 and 4, and both are online; nothing in this sequence ever holds an `ACCESS EXCLUSIVE` lock for longer than a catalog write, so there is no outage window.",
268
+ "outputSha256": "124b2e78be2133e7b016e0c9d1548e1284653d738a5e0a1f82333a86f0b0b95e",
269
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
270
+ "deterministic": [],
271
+ "judge": {
272
+ "verdict": "pass",
273
+ "reason": "Step 1 adds column nullable with no default; Step 2 backfills in bounded batches (LIMIT 10000, loop, commit between batches) explicitly rejecting single UPDATE; Step 3 shows ADD CONSTRAINT ... CHECK (status IS NOT NULL) NOT VALID; Step 4 shows separate VALIDATE CONSTRAINT. No fail criteria: SET NOT NULL only mentioned as optional after validation, not as the enforcement method."
274
+ },
275
+ "passed": true
276
+ },
277
+ {
278
+ "output": "The short answer: don't do it as one `ALTER TABLE`. On Postgres, adding a `NOT NULL` column in a single statement takes an `ACCESS EXCLUSIVE` lock and (for the default case) rewrites/scans the whole table, which on millions of rows means a multi-second-to-minute outage for any query touching that table. Split it into a sequence of short, individually-safe steps.\n\nFirst, though — the rule that changes everything: **in Postgres 11+, adding a column with a *constant* default no longer rewrites the table.** If your default is a literal (`0`, `''`, `'active'`, `now()` is *not* constant — see below), `ADD COLUMN ... NOT NULL DEFAULT <constant>` is already metadata-only and safe. So check that first:\n\n- **Constant default** → you can likely do it in one statement safely (still take the lock briefly, but it's metadata-only). Step 3 below is trivial.\n- **No default, or a non-constant default** (`now()`, `gen_random_uuid()`, another column) → you must use the multi-step backfill sequence.\n\nAssuming you need the real sequence (no default, or a non-constant one):\n\n**Step 1 — Add the column nullable, with no default.**\n```sql\nALTER TABLE orders ADD COLUMN status text;\n```\nThis is metadata-only: fast, brief lock, no rewrite.\n\n**Step 2 — Backfill in batches, out of band.**\nDo *not* backfill inside a single `UPDATE orders SET status = ...` — that's a long transaction with a big lock and huge bloat. Instead batch it:\n```sql\nUPDATE orders SET status = 'pending'\nWHERE id IN (\n SELECT id FROM orders WHERE status IS NULL LIMIT 10000\n);\n```\nLoop that until it affects zero rows. Between batches, commit so you're not holding one giant transaction. If another process (a deploy) needs to live-write to this table while you backfill, either add a trigger to keep new rows filled, or have the application write the value for new rows during this window.\n\n**Step 3 — Add the constraint without a long lock, then validate.**\n```sql\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null\n CHECK (status IS NOT NULL) NOT VALID;\n```\n`NOT VALID` takes only a brief lock and does not scan existing rows. Then, in a *separate* statement/transaction:\n```sql\nALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n```\n`VALIDATE` only takes a `SHARE UPDATE EXCLUSIVE` lock — it does not block reads or writes — and it scans the table to confirm every row satisfies the check. This is the important trick: the expensive scan happens under a lock that doesn't block traffic.\n\n**Step 4 — Attach the real `NOT NULL` (optional but clean).**\nIn Postgres 12+, once a validated `CHECK (col IS NOT NULL)` exists, `ALTER TABLE ... SET NOT NULL` can use it to skip the table scan:\n```sql\nALTER TABLE orders ALTER COLUMN status SET NOT NULL;\n```\nThen optionally drop the now-redundant check constraint. In PG 11 and earlier, `SET NOT NULL` always scans — in that case, just keep the validated check constraint as your enforcement and skip the formal `NOT NULL`, or accept the scan during a maintenance window.\n\n**Resulting deployment shape:** run steps 1, the constraint add, and the validate as separate migrations/files. Don't combine any of them with a slow backfill in the same migration.\n\nA few things that bite people here:\n\n- **`now()` / non-constant defaults are the trap.** `ADD COLUMN ts timestamptz NOT NULL DEFAULT now()` still rewrites in older versions and is a rewrite even in modern ones because it's not a constant. If you want \"one value for all existing rows,\" write a literal.\n- **Don't set a default on Step 1 if you're backfilling a per-row value** — a default that then gets changed is extra churn; add the default only after backfill if you want one going forward.\n- **Autovacuum / bloat:** the batched backfill leaves dead tuples; make sure autovacuum is keeping up, or run a manual `VACUUM` on the table after, so the table doesn't bloat.\n- **`CREATE INDEX` on the same column:** if you also need an index, use `CREATE INDEX CONCURRENTLY` in a separate, non-transactional migration — a plain `CREATE INDEX` takes a lock that blocks writes for the duration.\n\nSo concretely, for a millions-of-rows table with no constant default:\n\n1. `ADD COLUMN ... NULL` (fast).\n2. Batched `UPDATE ... WHERE col IS NULL LIMIT n` loop, committing per batch.\n3. `ADD CONSTRAINT ... CHECK (col IS NOT NULL) NOT VALID` (brief lock).\n4. `VALIDATE CONSTRAINT` (non-blocking scan).\n5. `SET NOT NULL` (skips scan in PG 12+, since the validated check satisfies it).\n6. Any index via `CREATE INDEX CONCURRENTLY` as its own migration.\n\nEach step is independently safe, so nothing holds a long exclusive lock and traffic keeps flowing.",
279
+ "outputSha256": "eb64839cbe4c1b6d26f7827f8ac31bd30c519ff963edbc3dd798dd0c097d70d4",
280
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
281
+ "deterministic": [],
282
+ "judge": {
283
+ "verdict": "pass",
284
+ "reason": "Step 1 adds column nullable (ALTER TABLE orders ADD COLUMN status text) with app writing new rows during backfill window; Step 2 batched UPDATE ... WHERE status IS NULL LIMIT 10000 looped with per-batch commits; Step 3 shows ADD CONSTRAINT ... CHECK (status IS NOT NULL) NOT VALID then separate VALIDATE CONSTRAINT. No fail criteria: SET NOT NULL only after validated check (PG12+ scan-skip), not as enforcement."
285
+ },
286
+ "passed": true
287
+ },
288
+ {
289
+ "output": "The safe sequence for Postgres splits the change across separate migrations so no single statement holds a long lock. Here it is, from the zero-downtime migration table:\n\n## The four-step sequence\n\n**Migration 1 — add the column nullable, with a default (if appropriate)**\n\n```sql\nALTER TABLE orders ADD COLUMN status text;\n```\n\nA nullable add is a fast metadata-only change on Postgres (11+). If the column needs a default, `ADD COLUMN ... DEFAULT 'pending'` is also metadata-only on PG 11+ because the default is stored in the catalog, not written to every row — so it does not rewrite the table.\n\n**Migration 2 — backfill in batches**\n\nDo not run a single `UPDATE orders SET status = 'pending';` — that takes a row lock on every matching row and can bloat the table and stall WAL. Batch it and commit between batches:\n\n```sql\n-- repeat until 0 rows affected; keep batch size small (e.g. 1000-10000)\nUPDATE orders\nSET status = 'pending'\nWHERE id IN (\n SELECT id FROM orders\n WHERE status IS NULL\n ORDER BY id\n LIMIT 5000\n FOR UPDATE SKIP LOCKED\n);\n```\n\nOr drive the batches from application code, committing each batch and pausing between them. This can even ship ahead of the constraint so the constraint migration is instant.\n\n**Migration 3 — add the constraint as NOT VALID**\n\n```sql\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null\n CHECK (status IS NOT NULL) NOT VALID;\n```\n\n`NOT VALID` only takes a brief `ACCESS EXCLUSIVE` lock to install the constraint — it does not scan existing rows, so it's fast. New inserts and updates are checked immediately, which stops the column from taking on new nulls while you finish.\n\n**Migration 4 — validate the constraint**\n\n```sql\nALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n```\n\nThis scans the table but only takes a `SHARE UPDATE EXCLUSIVE` lock — concurrent reads and writes proceed. Once it succeeds, the planner treats the column as effectively `NOT NULL`.\n\n## Then optionally convert to a real NOT NULL\n\nOn Postgres 12+, once a matching validated `CHECK (col IS NOT NULL)` exists, you can do:\n\n```sql\nALTER TABLE orders ALTER COLUMN status SET NOT NULL;\n```\n\nThe planner recognizes the validated check constraint and skips the full-table scan, so this is fast. On older versions, a plain `SET NOT NULL` does scan the whole table under an `ACCESS EXCLUSIVE` lock — in that case, keep the `CHECK` constraint as the enforcement mechanism instead of converting, or accept a maintenance window.\n\n## Things that will bite you\n\n- **Don't combine steps in one migration file.** The backfill can take minutes on millions of rows; keep it separate from the DDL so the DDL's brief lock isn't held for the backfill's duration.\n- **Set a `lock_timeout`** on the DDL statements (e.g. `SET lock_timeout = '3s';`) so a blocked `ALTER` fails fast instead of queueing behind a long-running query and then blocking everything behind it.\n- **The default-on-add trick only works on PG 11+.** On older versions, `ADD COLUMN ... DEFAULT` rewrites the whole table and locks it — backfill manually instead.\n- **Batch size matters.** Very small batches mean many round trips; very large ones hold locks and generate bursts of WAL. Size by measured rows/sec on a replica or staging dataset first.\n- **Verify with `EXPLAIN`** after the constraint is valid to confirm the planner is using it for any query that relies on the column being non-null.\n\nIf the column genuinely can't be backfilled to a single value (different default per row), the shape is the same — batch the per-row logic in step 2, just with a more complex `UPDATE`.\n\nWant me to also cover the reverse case (dropping/renaming a column) or how this differs on MySQL?",
290
+ "outputSha256": "1bf1fa33b80a726e7326c79288c32608cb0fab4b28aeab6c7942de9ce0637d9b",
291
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
292
+ "deterministic": [],
293
+ "judge": {
294
+ "verdict": "pass",
295
+ "reason": "Answer gives nullable ADD COLUMN first (Migration 1), batched backfill with LIMIT 5000 loop and warning against single UPDATE (Migration 2), then ADD CONSTRAINT ... CHECK (status IS NOT NULL) NOT VALID (Migration 3) and separate VALIDATE CONSTRAINT (Migration 4). All pass criteria concretely shown; no fail criteria."
296
+ },
297
+ "passed": true
298
+ },
299
+ {
300
+ "output": "# Adding a NOT NULL column to a large Postgres table without an outage\n\nThe critical fact: in Postgres 11+, `ALTER TABLE ... ADD COLUMN ... NOT NULL DEFAULT <constant>` is **metadata-only** — it does not rewrite the table and takes only a brief `ACCESS EXCLUSIVE` lock. That covers most real cases. The dangerous cases are a **non-constant default**, a **`NOT NULL` with no default** (must backfill), and adding a **CHECK/foreign key constraint**, all of which scan the table and block writes.\n\nPick your path based on which case you're in.\n\n## Case A — `NOT NULL` with a constant default (the common case)\n\nPostgres 11+:\n\n```sql\n-- Fast, metadata-only, brief lock. No table rewrite.\nALTER TABLE orders ADD COLUMN status text NOT NULL DEFAULT 'pending';\n```\n\nDo **not** wrap this in a transaction that also does other slow work — keep the lock hold time minimal. Set a short `lock_timeout` so this fails fast rather than queuing behind a long query and blocking everything behind it:\n\n```sql\nSET lock_timeout = '3s';\nALTER TABLE orders ADD COLUMN status text NOT NULL DEFAULT 'pending';\nRESET lock_timeout;\n```\n\nIf you're on Postgres 10 or older, this syntax **does** rewrite the table — use Case B instead.\n\n## Case B — `NOT NULL` requiring a backfill, or a non-constant default\n\nThe safe pattern is **add nullable → backfill in batches → add constraint using the non-blocking form**. Do each step as a **separate migration file**, so a failed/hung step doesn't roll back the schema change that already succeeded.\n\n**Step 1 — add the column, nullable, no default (metadata-only):**\n\n```sql\nALTER TABLE orders ADD COLUMN status text;\n```\n\nDo **not** set a `DEFAULT` yet if the value is non-constant — a volatile default in the `ADD COLUMN` forces a full rewrite in every PG version.\n\n**Step 2 — backfill in batches.** Never `UPDATE orders SET status = 'pending';` in one statement — it takes row locks on every row and bloats one giant transaction. Loop in chunks keyed on the primary key, committing each batch:\n\n```sql\n-- run repeatedly until 0 rows affected\nWITH batch AS (\n SELECT id FROM orders\n WHERE status IS NULL\n ORDER BY id\n LIMIT 10000\n FOR UPDATE SKIP LOCKED\n)\nUPDATE orders o\nSET status = 'pending'\nFROM batch\nWHERE o.id = batch.id;\n```\n\nBetween batches, pause briefly to let autovacuum and replicas keep up. If the value is genuinely constant for all existing rows, a plain `ALTER TABLE ... ADD COLUMN ... NOT NULL DEFAULT 'pending'` (Case A) is strictly simpler — reach for this batched loop only when the value varies per row or comes from a join.\n\n**Step 3 — add the `NOT NULL` constraint without a long full-table lock.** Add it as `NOT VALID`, then `VALIDATE` in a separate transaction that takes only a weak lock:\n\n```sql\n-- Migration file 2\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null\n CHECK (status IS NOT NULL) NOT VALID;\n```\n\n```sql\n-- Migration file 3: does the scan under a SHARE UPDATE EXCLUSIVE lock —\n-- concurrent reads AND writes stay unblocked\nALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n```\n\nNote: a `CHECK ... NOT VALID` **is** the non-blocking way to enforce `NOT NULL` on existing rows. There is no `ADD CONSTRAINT ... NOT NULL NOT VALID` form for a plain column constraint — but you can also just attach the default and convert:\n\n```sql\n-- Alternative in PG 12+: attach default, set NOT NULL — both are fast when\n-- the column has no NULLs already (the NOT NULL set only needs a scan,\n-- which VALIDATE-style operations do not block writes for in PG 12+).\nALTER TABLE orders ALTER COLUMN status SET DEFAULT 'pending';\nALTER TABLE orders ALTER COLUMN status SET NOT NULL; -- still scans; PG 12+ OK\n```\n\nThe `CHECK ... NOT VALID` + `VALIDATE` route is the safest across versions and is what I'd ship if the table is in the high millions.\n\n**Step 4 — drop the temporary check constraint (optional).** Once a real `NOT NULL` column constraint exists, the check is redundant:\n\n```sql\nALTER TABLE orders DROP CONSTRAINT orders_status_not_null;\n```\n\nOtherwise keep it — it's harmless and documents intent.\n\n## Case C — you also need an index on the new column\n\n```sql\n-- CONCURRENTLY: no lock on writes. Must be outside a transaction block.\nCREATE INDEX CONCURRENTLY idx_orders_status ON orders (status);\n```\n\n`CREATE INDEX CONCURRENTLY` cannot run inside a transaction — many migration tools (Rails, Django, some Flyway configs) wrap every migration in one by default. Disable that for this file or the statement will error. Also: a `CONCURRENTLY` index build can fail and leave an `INVALID` index behind; check `pg_index.indisvalid` afterward and drop/rebuild if needed.\n\n## Sequencing and operational safety\n\n- **Set `lock_timeout`** (a few seconds) on every migration that takes `ACCESS EXCLUSIVE`. If it can't get the lock immediately, it fails rather than stalling behind a long query and then blocking every write behind *it*. Retry.\n- **Separate files per step.** Add-column, backfill, validate, index are four migrations, not one transaction. A combined transaction holds locks across the slow backfill and can turn a 2-second change into a minutes-long outage.\n- **Watch replication.** A large backfill or index build generates WAL; on a replica with a single-threaded apply (or `max_worker_processes` unset for parallel apply) this can lag badly. Batch/pause the backfill if replica lag climbs.\n- **Add the app-side `NOT NULL` awareness last.** Deploy code that writes the column *before* the `NOT NULL` constraint lands, so no in-flight insert from old code violates it.\n\n## What to verify before calling it done\n\n```sql\n-- confirm the constraint landed and is valid\nSELECT conname, convalidated FROM pg_constraint WHERE conrelid = 'orders'::regclass;\n\n-- confirm no NULLs remain\nSELECT count(*) FROM orders WHERE status IS NULL;\n\n-- confirm the index is valid, not left INVALID by a failed CONCURRENTLY build\nSELECT indexrelid::regclass, indisvalid FROM pg_index WHERE indrelid = 'orders'::regclass;\n```\n\nRun the whole sequence against a **production-sized clone** first — a migration that's fast on an empty dev table can rewrite or lock for minutes at millions of rows, and that's exactly the failure this sequence exists to avoid.\n\n**The one-line summary:** in PG 11+ use a constant default and you're done in one metadata-only statement; for anything else, split into add-nullable → batched-backfill → `CHECK ... NOT VALID` → `VALIDATE`, and index with `CONCURRENTLY` outside a transaction.",
301
+ "outputSha256": "fa4159e44fa1255aec296dec6cc91c84a47c5da0388c6fcc8618bb8695fac361",
302
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
303
+ "deterministic": [],
304
+ "judge": {
305
+ "verdict": "fail",
306
+ "reason": "Pass 1 and 2 hold (nullable add + app writes before constraint; batched loop with LIMIT 10000). But pass 3 is undermined: the answer offers 'ALTER TABLE orders ALTER COLUMN status SET NOT NULL' as an acceptable alternative ('PG 12+ OK'), which is the plain SET NOT NULL the rubric forbids, and it also claims SET NOT NULL 'only needs a scan' that doesn't block writes. Fail criterion 1 holds."
307
+ },
308
+ "passed": false
309
+ },
310
+ {
311
+ "output": "Short answer: never issue a single `ALTER TABLE ... ADD COLUMN ... NOT NULL DEFAULT ...` (or `ADD COLUMN` + `SET NOT NULL` in one shot) against a large existing table. Split it into a multi-migration, zero-downtime sequence. Here's the mechanics.\n\n## Why the naive version hurts\n\nOn Postgres, `ALTER TABLE ... SET NOT NULL` (and a bare `ADD COLUMN ... NOT NULL` with no default) takes an `ACCESS EXCLUSIVE` lock and scans the whole table to verify no nulls exist. On millions of rows that lock can be held for the entire scan — the \"outage\" is every concurrent read and write blocking behind it. The safe path avoids a full-table verification scan under an exclusive lock by adding the constraint as `NOT VALID`, then validating it separately under a weaker lock.\n\n## The safe sequence (Postgres 11+)\n\n**Migration 1 — add the column, nullable, no lock held long**\n\n```sql\nALTER TABLE orders ADD COLUMN status text;\n```\n\nAdding a plain nullable column with no default is metadata-only (instant) on modern Postgres. Do **not** add a `DEFAULT` here unless it's a constant default (see the note below) — a volatile default forces a rewrite.\n\n**Migration 2 — backfill in bounded batches**\n\nDo not run one `UPDATE orders SET status = 'pending'` — that writes millions of rows in one transaction, bloats WAL, and holds locks/bloat for the whole run. Batch it:\n\n```sql\n-- Run repeatedly until 0 rows affected, in its own transaction each pass\nUPDATE orders\nSET status = 'pending'\nWHERE status IS NULL\n AND id IN (SELECT id FROM orders WHERE status IS NULL LIMIT 5000);\n```\n\nCommit between batches. If the table is being written to concurrently, either use a `WHERE status IS NULL` guard (as above) so no batch clobbers a newer write, or add a trigger that fills the column for new rows during the transition, then drop the trigger after validation.\n\n**Migration 3 — add the constraint WITHOUT scanning**\n\n```sql\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null\n CHECK (status IS NOT NULL) NOT VALID;\n```\n\n`NOT VALID` takes only a brief `ACCESS EXCLUSIVE` lock to register the constraint in the catalog — no scan, no long lock.\n\n**Migration 4 — validate under a weak lock**\n\n```sql\nALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n```\n\n`VALIDATE CONSTRAINT` scans the table but holds only a `SHARE UPDATE EXCLUSIVE` lock — concurrent reads and writes proceed. **Run this outside a transaction block** (`VALIDATE CONSTRAINT` cannot run inside an explicit transaction on some versions / behaves best standalone), and ideally in a low-traffic window since it is I/O heavy, though it no longer blocks traffic.\n\n**Migration 5 — promote to a true NOT NULL (optional but recommended)**\n\nA `CHECK (col IS NOT NULL)` constraint is not the same catalog object as a column-level `NOT NULL`. In Postgres 12+, you can promote it cheaply because the planner knows the check already holds:\n\n```sql\nALTER TABLE orders ALTER COLUMN status SET NOT NULL;\n```\n\nWith the validated check constraint in place, Postgres 12+ skips the full re-scan for `SET NOT NULL`, so this is now fast. Verify your version — on Postgres 11 and earlier this still scans, in which case you may leave the check constraint in place as the enforcement mechanism.\n\n**Migration 6 — drop the now-redundant check constraint**\n\n```sql\nALTER TABLE orders DROP CONSTRAINT orders_status_not_null;\n```\n\nOnly after confirming the column-level `NOT NULL` is in place and enforced.\n\n## Defaults: the special case\n\n- **Constant default** (`DEFAULT 'pending'`): on Postgres 11+, `ADD COLUMN ... DEFAULT 'pending' NOT NULL` is optimized — it does not rewrite the table, because the default is stored in the catalog for existing rows. This is genuinely a one-liner and often the *right* shortcut when the default is a literal and you want existing rows to read as that value.\n- **Volatile/expression default** (`DEFAULT now()`, `DEFAULT gen_random_uuid()`): forces a full table rewrite under an exclusive lock. Never do this directly — add nullable, backfill in batches, then set the default.\n\nIf your \"NOT NULL column\" is really \"existing rows should all get the same constant,\" the constant-default one-liner beats the whole sequence. If existing rows need distinct/computed values, or a later default differs from the backfill value, use the full sequence above.\n\n## Checklist before you ship\n\n- Each of the migrations above is its **own file/transaction** — never combine the blocking step with a slow backfill.\n- `SET NOT NULL` / `VALIDATE CONSTRAINT` run outside any long outer transaction; keep transaction lifetime to DB work only, commit before any else.\n- Confirm `EXPLAIN`-level behavior isn't your concern here (DDL), but confirm the migration on a table with realistic row count — not empty dev.\n- If writes are live during backfill, either guard each batch with `WHERE status IS NULL` or add the temporary fill trigger, and drop the trigger after validation.\n- Check your Postgres version — steps 1/5 lean on 11+ (constant-default optimization) and 12+ (fast `SET NOT NULL` after validated check). On older versions, keep the check constraint as the enforcement and skip step 5/6.\n\n## MySQL equivalence (if relevant)\n\nMySQL has no `NOT VALID` + `VALIDATE CONSTRAINT` pair. The analogous path: add the column nullable (`ALTER TABLE ... ADD COLUMN`, instant), backfill in batches with `LIMIT`, then `ALTER TABLE ... MODIFY col ... NOT NULL` — under MySQL 8 with `ALGORITHM=INPLACE, LOCK=NONE` where the engine supports it, otherwise expect a rebuild and schedule it. The batching/backfill discipline is identical.\n\nThe one-line version: **nullable add → batched backfill → `NOT VALID` check → `VALIDATE` → promote to `NOT NULL`**, each as its own migration, and use the constant-default one-liner only when the value is a literal.",
312
+ "outputSha256": "5791055e383443bb4a0fa732b50d1fe2b9b4ecc95e2ee8338ee247c2e94d5abb",
313
+ "promptSha256": "f238ec31a6960ba4681dedd32a40d0c2b44d40802c8fc74b34510c1530a4d035",
314
+ "deterministic": [],
315
+ "judge": {
316
+ "verdict": "pass",
317
+ "reason": "Answer shows nullable add (Migration 1: ALTER TABLE orders ADD COLUMN status text), app code writing going forward (trigger/guard for new rows), batched backfill (Migration 2: UPDATE ... WHERE status IS NULL AND id IN (... LIMIT 5000), commit between batches), and NOT VALID check + separate VALIDATE CONSTRAINT (Migrations 3-4). SET NOT NULL only after validated check, not as the enforcement mechanism."
318
+ },
319
+ "passed": true
320
+ }
321
+ ]
322
+ },
323
+ {
324
+ "id": "parameterized-status-filter",
325
+ "kind": "behavior",
326
+ "prompt": "Write a query that filters orders by a status value that comes in from an API request.",
327
+ "strictness": "high",
328
+ "trials": 10,
329
+ "passes": 10,
330
+ "passRate": 1,
331
+ "passAtK": 1,
332
+ "grader": "judge",
333
+ "status": "ran",
334
+ "trialRecords": [
335
+ {
336
+ "output": "```sql\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = $1\nORDER BY created_at DESC;\n```\n\n**Why this shape:**\n\n- **`$1` is a placeholder, never a literal or string-built value.** The status comes from an API request — that is exactly the untrusted input this position exists to handle. Binding it here is what makes the query immune to the `status = 'shipped'' OR '1'='1'` and `'shipped'; DROP TABLE orders; --` classes of attack without any escaping step of your own. Use `?` for MySQL, `:status` for a named-parameter API; the point is the driver sends the value over the wire as a parameter, separate from the SQL text.\n- **Bound parameter, not `WHERE status = '\\'' || $1 || '\\''`.** Concatenating or interpolating the value — including into a `format()` inside a stored procedure — is the failure this rule exists to prevent, so parameterize even when the caller \"looks internal.\"\n- **If `status` is actually a closed set**, validate it against the enum/allowed list in application code *before* the query as well; the bind protects against injection, but rejecting an unexpected value early gives a clean 400 instead of a silently empty result set.\n\nCall it from the application with the driver's parameter API, e.g.:\n\n```python\n# Python / psycopg\ncur.execute(\n \"SELECT id, customer_id, status, total_cents, created_at \"\n \"FROM orders WHERE status = %s ORDER BY created_at DESC\",\n (status_from_request,),\n)\n```\n\n**Index pairing:** if this filter runs against a table that can grow large, the query needs an index whose leading column matches the filter shape — `CREATE INDEX CONCURRENTLY idx_orders_status_created_at ON orders (status, created_at DESC);`. Composite order is `status` first (equality filter), `created_at` second (range/sort), which lets the scan both filter and return rows in the requested order without a sort step. Verify with `EXPLAIN ANALYZE` that the plan shows an Index Scan rather than a sequential scan on a table with meaningful row count.\n\n**If the endpoint also accepts a limit/cursor,** keep those bound too — `LIMIT $2` / `created_at < $3` — never spliced in as text; a `LIMIT` built by string formatting is just as injectable as a `WHERE` value.",
337
+ "outputSha256": "6be03ad5482d4b4be738178aa9e3b5eca9428bc6d1d3a4d2da01db1bd7d7f7b2",
338
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
339
+ "deterministic": [],
340
+ "judge": {
341
+ "verdict": "pass",
342
+ "reason": "Query text uses placeholder: `WHERE status = $1` (and `%s` in the psycopg example), not the literal value. Value bound separately: `cur.execute(\"... WHERE status = %s ...\", (status_from_request,))` passes the status as a bound argument distinct from the SQL text. No concatenation/interpolation of the value occurs."
343
+ },
344
+ "passed": true
345
+ },
346
+ {
347
+ "output": "Here's a parameterized query that filters orders by an API-supplied status value, using placeholder binding instead of string interpolation:\n\n```sql\n-- Postgres (node-postgres style placeholder)\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = $1\n AND created_at >= $2\nORDER BY created_at DESC\nLIMIT $3;\n```\n\nThe critical rule: the API-supplied `status` is passed as a bound parameter, never concatenated into the SQL text. In application code:\n\n```js\n// Good — value is bound, never interpolated\nconst { rows } = await pool.query(\n `SELECT id, customer_id, status, total_cents, created_at\n FROM orders\n WHERE status = $1\n AND created_at >= $2\n ORDER BY created_at DESC\n LIMIT $3`,\n [status, since, limit]\n);\n```\n\n```sql\n-- MySQL uses ? positional placeholders\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = ?\n AND created_at >= ?\nORDER BY created_at DESC\nLIMIT ?;\n```\n\n**Why not interpolation.** Do not do this, even though \"it's just a status enum\":\n\n```js\n// BAD — injectable the moment status becomes user-influenced\nconst sql = `SELECT ... FROM orders WHERE status = '${status}'`;\n```\n\nAn API value is user-influenced by definition. The moment the status isn't a fixed server-side constant — an admin panel, a filter dropdown, an import — `' OR 1=1 --` or worse is reachable.\n\n**Index support.** `WHERE status = $1 ... ORDER BY created_at DESC` wants a composite index with equality columns first, then the sort column:\n\n```sql\nCREATE INDEX CONCURRENTLY idx_orders_status_created_at\n ON orders (status, created_at DESC);\n```\n\nColumn order matters: equality-on-`status` narrows, then `created_at` satisfies the ordering so the planner avoids a sort. If the status set is small and you only ever query a few hot values, a partial index is leaner:\n\n```sql\nCREATE INDEX CONCURRENTLY idx_orders_pending_created_at\n ON orders (created_at DESC)\n WHERE status = 'pending';\n```\n\n**Verify the plan** before calling it done:\n\n```sql\nEXPLAIN (ANALYZE, BUFFERS)\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = 'pending'\n AND created_at >= now() - interval '30 days'\nORDER BY created_at DESC\nLIMIT 50;\n```\n\nConfirm an Index Scan using `idx_orders_status_created_at` (or the partial index), not a Seq Scan on a table with meaningful row count.\n\n**One caveat for the enum case.** If `status` is a fixed, small set, prefer validating it against an allowlist in the application layer *before* it reaches the query — not as a substitute for parameterization, but so an unexpected value fails fast with a clear error rather than silently returning zero rows. In Postgres you can also bind against an actual enum type so the database rejects invalid values itself. Parameterization and validation are complementary; neither replaces the other.",
348
+ "outputSha256": "b41e0186568ffd20719936e054a366eb61da8a668ef3d200c1fe675425461705",
349
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
350
+ "deterministic": [],
351
+ "judge": {
352
+ "verdict": "pass",
353
+ "reason": "Answer shows SQL with placeholder `WHERE status = $1` (and MySQL `?`), and binds value separately: `pool.query(..., [status, since, limit])`. The `${status}` interpolation appears only inside a 'BAD' example explicitly warned against, not as the solution. Both pass criteria hold; no fail criterion."
354
+ },
355
+ "passed": true
356
+ },
357
+ {
358
+ "output": "```sql\n-- Postgres / generic SQL — the status value arrives as a bound parameter,\n-- never concatenated into the SQL text.\nSELECT id, customer_id, status, total_cents, created_at\n FROM orders\n WHERE status = $1\n ORDER BY created_at DESC;\n```\n\n```python\n# Caller side (illustrative — the binding mechanism depends on the driver)\nrows = cursor.execute(\n \"SELECT id, customer_id, status, total_cents, created_at \"\n \"FROM orders WHERE status = %s ORDER BY created_at DESC\",\n (request_status,),\n).fetchall()\n```\n\nThe one thing that matters here: `request_status` is passed as a **bound parameter** (`$1` in Postgres, `%s`/`?` in MySQL and most drivers), not formatted into the string. Do **not** write anything like:\n\n```sql\n-- WRONG: SQL injection.\nWHERE status = '\"\"\" + request_status + \"\"\"'\n```\n\nEven if the API currently restricts `status` to a fixed enum, that value travels from outside your process and must be treated as untrusted. Parameter binding makes the value data, never SQL syntax, so a crafted input like `'; DROP TABLE orders; --` cannot change the statement's structure.\n\nIf the API does validate `status` against a fixed set of allowed values, do that validation too, at the boundary — but as defense in depth, not as a substitute for parameterization:\n\n```python\nALLOWED_STATUSES = {\"pending\", \"paid\", \"shipped\", \"cancelled\"}\n\nif request_status not in ALLOWED_STATUSES:\n raise ValueError(f\"invalid status: {request_status!r}\")\n\nrows = cursor.execute(\n \"SELECT id, customer_id, status, total_cents, created_at \"\n \"FROM orders WHERE status = %s ORDER BY created_at DESC\",\n (request_status,),\n).fetchall()\n```\n\nTwo implementation notes worth checking before this goes out:\n\n- **Index the filter.** If `orders` is more than trivially sized and this is a hot path, `WHERE status = ...` without an index is a sequential scan. Add a composite index matching the query shape, ordered equality-column first, then the sort column:\n ```sql\n CREATE INDEX CONCURRENTLY idx_orders_status_created_at\n ON orders (status, created_at DESC);\n ```\n `CONCURRENTLY` avoids the write lock `CREATE INDEX` would otherwise take on a live table.\n\n- **Watch for unbounded result sets.** `ORDER BY created_at DESC` with no `LIMIT` returns everything the status matches. If a status can accumulate many rows, add pagination (`... ORDER BY created_at DESC, id DESC LIMIT $2 OFFSET $3`, or keyset pagination on `(created_at, id)`) rather than letting the API response grow without bound.\n\nVerify the plan rather than assuming the index is used:\n\n```sql\nEXPLAIN (ANALYZE, BUFFERS)\nSELECT id, customer_id, status, total_cents, created_at\n FROM orders\n WHERE status = 'paid'\n ORDER BY created_at DESC;\n```\n\nConfirm it shows an Index Scan (or Index-Only Scan if the index covers the selected columns) rather than a Seq Scan on a table with meaningful row count.\n\nIf this lives inside an ORM, use its parameterized filter API (`Order.where(status=request_status)`) rather than its raw-SQL escape hatch — same binding guarantee, less room to slip into string interpolation.",
359
+ "outputSha256": "ae9b96a339e7c54ebffe41ba77e9bb9ffd3490a309f31aa862953e0be1879d34",
360
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
361
+ "deterministic": [],
362
+ "judge": {
363
+ "verdict": "pass",
364
+ "reason": "SQL shows `WHERE status = $1` (placeholder, not spliced value), and the Python call passes `(request_status,)` as a separate bound argument to cursor.execute. The concatenation example is explicitly labeled WRONG/anti-pattern, not used. Both pass criteria concretely met."
365
+ },
366
+ "passed": true
367
+ },
368
+ {
369
+ "output": "Here's the query, parameterized so the API-supplied status is never concatenated into SQL text:\n\n```sql\n-- Postgres / MySQL / most SQL engines (adjust placeholder style to engine/driver)\nSELECT\n o.id,\n o.customer_id,\n o.status,\n o.total_cents,\n o.created_at\nFROM orders AS o\nWHERE o.status = $1\nORDER BY o.created_at DESC;\n```\n\nKey points:\n\n- **`$1` is a bound parameter, not a substituted string.** The driver sends the status value out-of-band from the SQL text, so it cannot change the query's structure. The equivalent per engine/driver: `?` (MySQL, SQLite, JDBC/ODBC), `:status` (named-parameter drivers), `%(status)s` (psycopg), `@status` (SQL Server).\n- **Never do this:**\n ```sql\n -- WRONG -- injection point\n SELECT * FROM orders WHERE status = '[\" + status + \"]';\n ```\n Even if `status` \"comes from our own frontend,\" a user-controlled `status` value like `x' OR '1'='1` turns this into a full-table dump, and \"our own frontend\" is not a trust boundary.\n- **Index the filtered column** if it isn't already, in the same change, so the query doesn't do a sequential scan:\n\n ```sql\n -- Postgres: non-blocking on a live table\n CREATE INDEX CONCURRENTLY idx_orders_status_created_at\n ON orders (status, created_at DESC);\n\n -- MySQL 8.0 / InnoDB\n ALTER TABLE orders\n ADD INDEX idx_orders_status_created_at (status, created_at);\n ```\n\n The composite order — equality column first (`status`), then the sort column (`created_at`) — matches this query's `WHERE ... ORDER BY` shape, so a single index serves both the filter and the ordering.\n\n- **Verify the plan** after adding the index:\n ```sql\n EXPLAIN (ANALYZE, BUFFERS)\n SELECT o.id, o.customer_id, o.status, o.total_cents, o.created_at\n FROM orders AS o\n WHERE o.status = 'pending'\n ORDER BY o.created_at DESC;\n ```\n You want an Index Scan / Index-Only Scan, not a Seq Scan, and no separate sort step.\n\n- **If the API hands you a list of statuses rather than one value**, keep it parameterized as a batch rather than string-joining values:\n ```sql\n SELECT o.id, o.customer_id, o.status, o.total_cents, o.created_at\n FROM orders AS o\n WHERE o.status = ANY($1) -- $1 is the status array, bound as one param\n ORDER BY o.created_at DESC;\n ```\n or, on engines without array binding, generate the right number of placeholders (`WHERE o.status IN ($1, $2, $3)`) and bind each value positionally — never build the `IN` list by joining the raw values into a string.\n\n- **Validate the status at the boundary**, not in SQL. Constrain it to the known set (`pending`, `shipped`, ...) before it reaches the query, so an unexpected value fails fast rather than silently returning zero rows — the parameterization protects against injection; validation protects against nonsense input.",
370
+ "outputSha256": "be21761be94368205cbfd98f7286673b461a5ea38c3d38f043fa0940c2129781",
371
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
372
+ "deterministic": [],
373
+ "judge": {
374
+ "verdict": "pass",
375
+ "reason": "Query text uses placeholder: `WHERE o.status = $1` (not the literal value). Value bound separately: answer states `$1` is a bound parameter sent out-of-band, and lists driver equivalents (`?`, `:status`, `%(status)s`). No concatenation/interpolation of status into SQL; the concatenation example is explicitly labeled WRONG."
376
+ },
377
+ "passed": true
378
+ },
379
+ {
380
+ "output": "Here's a parameterized query that filters orders by a status value coming from an API request, following the rules above.\n\n## The query\n\n```sql\n-- Parameterized: $1 is bound by the driver, never interpolated into the text.\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = $1\nORDER BY created_at DESC\nLIMIT 100;\n```\n\nThe status value is passed to the driver's parameter array, not spliced into the SQL string:\n\n```js\n// node-postgres\nconst { rows } = await pool.query(\n `SELECT id, customer_id, status, total_cents, created_at\n FROM orders\n WHERE status = $1\n ORDER BY created_at DESC\n LIMIT 100`,\n [req.query.status] // bound parameter — not `${req.query.status}`\n);\n```\n\n```python\n# psycopg (Python)\ncur.execute(\n \"\"\"\n SELECT id, customer_id, status, total_cents, created_at\n FROM orders\n WHERE status = $1\n ORDER BY created_at DESC\n LIMIT 100\n \"\"\",\n (status_from_request,),\n)\n```\n\nMySQL uses `?` placeholders instead of `$1`:\n\n```sql\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = ?\nORDER BY created_at DESC\nLIMIT 100;\n```\n\n## Why it's written this way\n\n- **No string concatenation.** The single most important line here is that `req.query.status` / `status_from_request` goes into the parameter array, never into the SQL text. Building this as `\"WHERE status = '\" + status + \"'\"` would be a SQL injection hole — and the rationalization \"this value comes from our own API, not a user\" doesn't hold, since an API request parameter *is* user-influenced by definition.\n- **`ORDER BY created_at DESC` + `LIMIT`** is part of the query contract, so the index in the next section has to match it.\n\n## The index this needs\n\nThe `WHERE status = ? ORDER BY created_at DESC LIMIT 100` shape is a **composite** index, ordered equality-column first, then the range/sort column:\n\n```sql\nCREATE INDEX CONCURRENTLY idx_orders_status_created_at\n ON orders (status, created_at DESC);\n```\n\n- Column order matters: `status` (equality filter) comes first, `created_at` (sort) second. An index on `(created_at)` alone or `(created_at, status)` would not serve this query as well.\n- `CONCURRENTLY` (Postgres) / `ALGORITHM=INPLACE, LOCK=NONE` (MySQL 8) so adding it doesn't take a write lock on a live `orders` table — per the rule that a blocking index creation shouldn't go into a migration touching a table with existing rows.\n\n## Verification\n\nBefore reporting this done, confirm the plan actually uses the index rather than falling back to a sequential scan:\n\n```sql\nEXPLAIN (ANALYZE, BUFFERS)\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = 'pending'\nORDER BY created_at DESC\nLIMIT 100;\n```\n\nExpect an **Index Scan** (or Index-Only Scan if the index covers the selected columns) on `idx_orders_status_created_at`, not a `Seq Scan` on an `orders` table with meaningful row count. Run this against a production-sized dataset, not an empty dev database — a seq scan on 50 rows tells you nothing.\n\n## Two things worth flagging\n\n1. **Validate the status value** against the set of statuses your domain actually allows, *before* it reaches the database. Parameterization protects against injection; it does not stop `status=🤖garbage` from running a pointless query or returning a confusing empty result. An allow-list check at the API boundary is cheap and turns a silent empty result into a clear `400`.\n2. **If this list can grow beyond `LIMIT 100`**, use keyset pagination rather than `OFFSET` — `WHERE status = $1 AND (created_at, id) < ($2, $3) ORDER BY created_at DESC, id DESC LIMIT 100` — so deep pages don't degrade into scanning and discarding everything before them.",
381
+ "outputSha256": "d067dc202358e7a49f8ef952a57c7aa84a888d277bb290787331e1da1c2bc261",
382
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
383
+ "deterministic": [],
384
+ "judge": {
385
+ "verdict": "pass",
386
+ "reason": "Query text uses placeholder `WHERE status = $1` (and `?` for MySQL), not the literal value. Value is bound separately: node-postgres `[req.query.status]` passed as second arg to pool.query, and psycopg `(status_from_request,)` passed to cur.execute. No concatenation/interpolation of the status value into SQL; the answer explicitly warns against `${req.query.status}`."
387
+ },
388
+ "passed": true
389
+ },
390
+ {
391
+ "output": "```sql\n-- Parameterized: the status value is bound, never concatenated into the SQL text.\n-- Postgres / MySQL placeholder style shown; use the driver's parameter API\n-- (e.g. node-postgres $1, Python DB-API %s, MySQL ?) rather than any string formatting.\n\nSELECT\n o.id,\n o.customer_id,\n o.status,\n o.total_cents,\n o.created_at\nFROM orders AS o\nWHERE o.status = $1\nORDER BY o.created_at DESC;\n```\n\n**Why this shape**\n\n- `WHERE o.status = $1` — the API-supplied status is passed as a bound parameter. The value is sent to the server separately from the SQL text, so it can never alter the statement's structure (no injection, no quoting bugs). This holds equally for `IN`-style filters if the API ever accepts a list: keep the placeholder count fixed and bind each element, or use the driver's array-binding support — never build `(?, ?, ?)` by joining strings.\n- `ORDER BY o.created_at DESC` — ordering is applied server-side so the result is deterministic and paginatable (`LIMIT`/`OFFSET` or keyset pagination on `(created_at, id)`).\n\n**Index it in the same change**\n\nThe filter/order shape is a single equality followed by a range/sort, so a composite index in that column order serves the query directly:\n\n```sql\n-- Postgres\nCREATE INDEX CONCURRENTLY idx_orders_status_created_at\n ON orders (status, created_at DESC);\n\n-- MySQL (non-blocking on modern InnoDB)\nALTER TABLE orders\n ADD INDEX idx_orders_status_created_at (status, created_at DESC),\n ALGORITHM=INPLACE, LOCK=NONE;\n```\n\n`CONCURRENTLY` (Postgres) and `ALGORITHM=INPLACE, LOCK=NONE` (MySQL) keep the index build from blocking writes on a live table. Create the index in the same migration/change as the query that needs it.\n\n**Verify the index is actually used**\n\n```sql\n-- Postgres\nEXPLAIN (ANALYZE, BUFFERS)\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = 'shipped'\nORDER BY created_at DESC;\n\n-- MySQL\nEXPLAIN ANALYZE\nSELECT ... WHERE status = 'shipped' ORDER BY created_at DESC;\n```\n\nConfirm an Index Scan (or Index-Only Scan if the selected columns are covered) rather than a Seq Scan / full table scan on a table with meaningful row count. Run it with a representative status value and dataset, not an empty dev table.\n\n**If `status` is a low-cardinality column** (few distinct values, one of them dominant), the planner may still choose a sequential scan for a very common value — that is correct behavior for that value, not a broken index. In that case, add the value genuinely being filtered for to the index only if it is selective, or consider a partial index:\n\n```sql\nCREATE INDEX CONCURRENTLY idx_orders_status_created_at_active\n ON orders (created_at DESC)\n WHERE status IN ('pending', 'processing');\n```\n\n**Notes**\n\n- Keep the query lifetime to database work only: if the caller then makes a slow external call per order, commit/fetch first and do that work outside any transaction opened for this query.\n- Validate the incoming status against the allowed set at the API boundary before binding — not for injection safety (the placeholder already covers that), but so an unexpected value returns a clean `400` rather than an empty result set.",
392
+ "outputSha256": "eb182d512b64a4cc1990b7234b7e168fbe58775cca329b986f176eecab4d7e3b",
393
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
394
+ "deterministic": [],
395
+ "judge": {
396
+ "verdict": "pass",
397
+ "reason": "Query text uses placeholder: `WHERE o.status = $1` (no literal spliced in). Criterion 2: answer states the API-supplied status is passed as a bound parameter, sent separately from SQL text, and names driver APIs (node-postgres $1, DB-API %s, MySQL ?). No concatenation/interpolation of the value into SQL."
398
+ },
399
+ "passed": true
400
+ },
401
+ {
402
+ "output": "# Orders-by-status query\n\nBelow is the shape I'd write, in Postgres syntax, assuming the project's convention. The important part is the **parameter binding** for `status` — it must come in as a bind parameter, never concatenated/interpolated into the SQL text, since its value originates from an API request.\n\n```sql\n-- Filename: src/orders/queries.sql\n-- Purpose: list orders for a given status, newest first, with the customer\n-- name for display. Used by GET /orders?status=...\n--\n-- Parameters:\n-- $1 -> status (text), supplied by the API layer via a bound parameter\n-- $2 -> limit (int), optional pagination cap\n-- $3 -> offset (int), optional pagination offset\n\nSELECT\n o.id,\n o.customer_id,\n c.name AS customer_name,\n o.status,\n o.total_cents,\n o.created_at\nFROM orders AS o\nJOIN customers AS c\n ON c.id = o.customer_id\nWHERE o.status = $1\nORDER BY o.created_at DESC\nLIMIT $2\nOFFSET $3;\n```\n\n### Why it's written this way\n\n**1. Parameterized, not interpolated.**\nThe status value arrives from the request (query string, JSON body, header — doesn't matter). It goes into the placeholder `$1` (Postgres) / `?` (MySQL) / `:status` (named params), **never** into the SQL text:\n\n```sql\n-- NEVER do this, even if the value \"looks safe\" today:\n-- \"SELECT ... WHERE status = '\" || status || \"'\"\n```\n\nAn API value is user-influenced by definition. \"It's only a status enum\" is exactly the assumption SQL injection defeats — the moment a new feature (admin filter, saved-search import, batch job) pipes a less-trusted value through the same code path, string building becomes a hole. Bind it from day one.\n\n**2. Join, not a per-row lookup.**\nJoining `customers` in the same query avoids the classic N+1: fetching the orders, then issuing one `SELECT ... FROM customers WHERE id = ...` per order. `EXPLAIN ANALYZE` here should show a Hash Join / Nested Loop against the `customers` PK, not repeated round-trips.\n\n**3. The index this query needs.**\n`WHERE o.status = $1 ORDER BY o.created_at DESC` wants a composite index whose column order matches the filter shape — equality column first, then the sort column:\n\n```sql\n-- Filename: migrations/2026_09_25_add_orders_status_created_at_idx.sql\nCREATE INDEX CONCURRENTLY IF NOT EXISTS orders_status_created_at_idx\n ON orders (status, created_at DESC);\n```\n\n`CONCURRENTLY` so creating it doesn't take a write lock on `orders` (drop `CONCURRENTLY` on a brand-new table created in the same change, or if you're not on Postgres). Verify with:\n\n```sql\nEXPLAIN (ANALYZE, BUFFERS)\nSELECT o.id, o.customer_id, c.name, o.status, o.total_cents, o.created_at\nFROM orders AS o\nJOIN customers AS c ON c.id = o.customer_id\nWHERE o.status = 'pending'\nORDER BY o.created_at DESC\nLIMIT 50 OFFSET 0;\n```\n\nExpect an Index Scan (or Index-Only Scan on `orders`) plus a join to `customers` — **not** a Seq Scan on `orders`.\n\n### If the API allows \"all statuses\" (optional filter)\n\nDon't branch into two SQL strings. Either bind `NULL` and let the predicate short-circuit:\n\n```sql\nWHERE ($1::text IS NULL OR o.status = $1)\n```\n\n…or, if the parameter list itself is dynamic (multiple statuses from the request), still keep it as **bound placeholders**, not concatenated literals — e.g. build `status IN ($1, $2, $3)` and pass the array, or use Postgres's array form:\n\n```sql\nWHERE o.status = ANY($1::text[]) -- $1 bound as a text[] parameter\n```\n\nEither way: the *values* are bound; only the *number* of placeholders is assembled in code.\n\n### Call-site reminder\nAt the API layer, call it through the driver/ORM's parameterized API — e.g. `db.query(sql, [status, limit, offset])` in Node, `cursor.execute(sql, (status, limit, offset))` in Python, `PreparedStatement` with `setString(1, status)` in Java. No driver method should receive SQL with the status value already baked into the string.\n\n**Verification before reporting done:** every value bound (no concatenation of the incoming `status` anywhere in the path); `EXPLAIN ANALYZE` on the real query confirms the composite index is used and there's no Seq Scan on `orders`; the `customers` access is a single join, not a per-row loop.",
403
+ "outputSha256": "8101a6d7fe2a088c2d724ad917650491203969e693a6076055234ec29e3ef9e5",
404
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
405
+ "deterministic": [],
406
+ "judge": {
407
+ "verdict": "pass",
408
+ "reason": "Query text uses placeholder `WHERE o.status = $1` (not the literal value), and the value is bound separately: `db.query(sql, [status, limit, offset])`, `cursor.execute(sql, (status, limit, offset))`, `setString(1, status)`. No concatenation/interpolation of status into SQL; the concatenation example is explicitly shown as 'NEVER do this'."
409
+ },
410
+ "passed": true
411
+ },
412
+ {
413
+ "output": "```sql\n-- Placeholder syntax differs by engine:\n-- Postgres: $1\n-- MySQL: ?\n-- (Do not interpolate the status value into the SQL text.)\n\nSELECT\n o.id,\n o.customer_id,\n o.status,\n o.total_cents,\n o.created_at\nFROM orders AS o\nWHERE o.status = $1 -- MySQL: WHERE o.status = ?\nORDER BY o.created_at DESC;\n```\n\nUsage from application code (the value is bound, never concatenated):\n\n```python\n# Postgres via psycopg\ncur.execute(\n \"\"\"\n SELECT o.id, o.customer_id, o.status, o.total_cents, o.created_at\n FROM orders AS o\n WHERE o.status = %s\n ORDER BY o.created_at DESC\n \"\"\",\n (status_from_request,), # bound parameter, not f-string/format\n)\n```\n\n```javascript\n// Postgres via node-postgres\nconst { rows } = await pool.query(\n `SELECT id, customer_id, status, total_cents, created_at\n FROM orders\n WHERE status = $1\n ORDER BY created_at DESC`,\n [statusFromRequest] // value passed separately from SQL text\n);\n```\n\n```sql\n-- MySQL via a driver using ? placeholders\nSELECT id, customer_id, status, total_cents, created_at\nFROM orders\nWHERE status = ?\nORDER BY created_at DESC;\n```\n\nNotes that matter for this query:\n\n1. **Never build the `WHERE` with string formatting.** `status_from_request` is API-supplied and therefore user-influenced. `f\"...WHERE status = '{status}'\"` (or `+ status +`, or `.format(...)`) is a SQL-injection hole — a crafted value closes the quote and appends arbitrary SQL. Bind it as a parameter; the driver sends the value out-of-band and the engine never parses it as SQL.\n\n2. **Index the filtered/sorted columns if this is a hot path.** This query filters on `status` and sorts by `created_at`, so a composite index with equality column first, then the range/sort column:\n\n ```sql\n -- Postgres (use CONCURRENTLY on a live table)\n CREATE INDEX CONCURRENTLY idx_orders_status_created_at\n ON orders (status, created_at DESC);\n\n -- MySQL\n CREATE INDEX idx_orders_status_created_at\n ON orders (status, created_at);\n ```\n\n Then confirm with `EXPLAIN (ANALYZE, BUFFERS)` (Postgres) or `EXPLAIN ANALYZE` (MySQL) that the plan uses an index scan rather than a sequential/full table scan.\n\n3. **Paginate large result sets.** For an API endpoint, add `LIMIT $2 OFFSET $3` (both bound parameters too), or keyset pagination on `(created_at, id)` for large tables — an unbounded `SELECT` of matching orders is a load/latency risk once the table grows.\n\n4. **Validate the value too, but not as the primary defense.** Rejecting an unknown status before it reaches the query is good hygiene for error messages and logging, but parameter binding — not validation — is what makes the query injection-safe; the two are complementary, not alternatives.",
414
+ "outputSha256": "7320e826b558adedf98d28deb13b92c3d7684d69e2ab9bbfbdc593e9a587838d",
415
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
416
+ "deterministic": [],
417
+ "judge": {
418
+ "verdict": "pass",
419
+ "reason": "SQL uses placeholder `WHERE o.status = $1` (and `?` for MySQL) with no value spliced in. Python shows `cur.execute(\"...WHERE o.status = %s...\", (status_from_request,))` and JS shows `pool.query(..., [statusFromRequest])` — value bound separately from SQL text. No concatenation/interpolation of the status value occurs."
420
+ },
421
+ "passed": true
422
+ },
423
+ {
424
+ "output": "```sql\n-- Parameterized query: the status value is bound as a parameter, never\n-- concatenated or interpolated into the SQL text.\nSELECT\n o.id,\n o.customer_id,\n o.status,\n o.total_cents,\n o.created_at\nFROM orders AS o\nWHERE o.status = ? -- Postgres: $1 MySQL: ?\nORDER BY o.created_at DESC;\n```\n\nNotes for this specific query:\n\n- **Parameterization is the whole point here.** `status` arrives from an API request, so it is user-influenced — feed it to the placeholder (`$1` in Postgres, `?` in MySQL, or the parameter API of whatever driver/query builder the project uses). Do not build the string with `\"WHERE status = '\" + status + \"'\"` or a template literal, and do not use an ORM's raw-SQL escape hatch to concatenate it either.\n- **Validate against an allow-list on the way in.** Bind parameters prevent injection, but a free-text `status` also silently returns zero rows for a typo, or — if the column is an enum/int — can error. Check the incoming value against the set of known statuses at the request boundary before it reaches the query.\n- **Index support.** If `orders` can grow, `WHERE status = ? ORDER BY created_at DESC` is served by a composite index with equality column first, then the sort column:\n\n ```sql\n CREATE INDEX CONCURRENTLY idx_orders_status_created_at\n ON orders (status, created_at DESC);\n ```\n\n Use `CONCURRENTLY` (Postgres) / `ALGORITHM=INPLACE, LOCK=NONE` (MySQL) so creating it on a live table doesn't block writes. Add this in the same change as the query, and confirm with `EXPLAIN (ANALYZE, BUFFERS)` that the plan does an Index Scan rather than a Seq Scan.\n- **If the status set is small and skewed**, a partial index can be cheaper — e.g. `... ON orders (created_at DESC) WHERE status = 'pending'` for the one status that dominates hot queries.\n- **Pagination.** A list endpoint returning all matches will eventually return too many rows. If the API paginates, keep the ordering column in the index and page by `created_at`/`id` rather than a large `OFFSET`.",
425
+ "outputSha256": "69497b1524ce1f45d16758130c0d9d5aaf64ef67eaa2d831e0bc1851fa6ab3c0",
426
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
427
+ "deterministic": [],
428
+ "judge": {
429
+ "verdict": "pass",
430
+ "reason": "Query text uses placeholder: 'WHERE o.status = ?' with comment 'Postgres: $1 MySQL: ?' (criterion 1). Notes state the status value is bound as a parameter via the placeholder/driver parameter API, separate from SQL text (criterion 2). No concatenation/interpolation of the value; it explicitly warns against it."
431
+ },
432
+ "passed": true
433
+ },
434
+ {
435
+ "output": "Use a parameterized query — the status value is bound as a placeholder, never interpolated into the SQL text. That applies whether you're writing raw SQL directly or dropping into an ORM's raw-SQL escape hatch.\n\n**Raw SQL (Postgres, e.g. via `pg`):**\n\n```sql\nSELECT id, customer_id, status, created_at, total_cents\nFROM orders\nWHERE status = $1\nORDER BY created_at DESC;\n```\n\n```js\nconst { rows } = await pool.query(\n 'SELECT id, customer_id, status, created_at, total_cents ' +\n 'FROM orders WHERE status = $1 ORDER BY created_at DESC',\n [status] // bound, not concatenated\n);\n```\n\n**Raw SQL (MySQL, e.g. via `mysql2`):**\n\n```sql\nSELECT id, customer_id, status, created_at, total_cents\nFROM orders\nWHERE status = ?\nORDER BY created_at DESC;\n```\n\n```js\nconst [rows] = await pool.execute(\n 'SELECT id, customer_id, status, created_at, total_cents ' +\n 'FROM orders WHERE status = ? ORDER BY created_at DESC',\n [status]\n);\n```\n\n**ORM equivalent (parameter binding is handled by the ORM):**\n\n```js\nconst orders = await db('orders')\n .where({ status })\n .orderBy('created_at', 'desc');\n```\n\nKey points:\n\n- **Never** do `WHERE status = '${status}'` or `\"WHERE status = \" + status` — that's the SQL injection path, and it's the single red flag that disqualifies this query even if it \"works\" in dev.\n- `ORDER BY created_at DESC` benefits from an index with `created_at` in the right position. If this is a hot list endpoint, add a composite index `(status, created_at DESC)` so the filter and the sort are both served by one index — the equality column (`status`) first, then the sort/range column.\n- If you're filtering by \"status\" but the API can also pass a collection (e.g. `?status=open&status=paid`), keep it parameterized too — build `WHERE status = ANY($1)` on Postgres (pass an array) or `WHERE status IN (?, ?)` with a placeholder per element on MySQL; never join the values into the SQL string.\n- Validate the incoming `status` against your known enum/set before it reaches the query layer if you want a clean 400 instead of an empty result — but validate in the application, not by trusting it in the SQL.\n\nIf the table is large and this query is hot, run `EXPLAIN (ANALYZE, BUFFERS)` (Postgres) or `EXPLAIN ANALYZE` (MySQL) to confirm it's using the composite index and not falling back to a sequential scan before you ship it.",
436
+ "outputSha256": "8860b3b1c5dd66adfec2371bb4ddd0385acc98f6f7eb4571db19e0068a72f48e",
437
+ "promptSha256": "fa179c158398e42dc7ed9f50e9a60418064da5dccaf5c3c8a2a2965e0605063e",
438
+ "deterministic": [],
439
+ "judge": {
440
+ "verdict": "pass",
441
+ "reason": "Shows SQL with placeholder: 'WHERE status = $1' (Postgres) and 'WHERE status = ?' (MySQL), not the literal value. Binds value separately: pool.query(sql, [status]) and pool.execute(sql, [status]). ORM .where({status}) also binds. No concatenation/interpolation of status; the '${status}' example is only warned against."
442
+ },
443
+ "passed": true
444
+ }
445
+ ]
446
+ }
447
+ ],
448
+ "verdict": "fail",
449
+ "scope": "bundled",
450
+ "skillDigest": "8164fdf08b64d54ad55f9260bd578d308f495030daaf41895ba5743eb74259be",
451
+ "catalogDigest": "14504a0807a0089488b9cb690c4b13f20865cd7a7fb69a1e5d8dfea8bfd5fbd1",
452
+ "judgePromptVersion": "2026-09-25.1",
453
+ "runner": "deepseek",
454
+ "model": "deepseek-chat",
455
+ "runnerPromptVersion": "2026-09-25.1",
456
+ "recordedAt": "2026-09-25T18:14:14.367Z",
457
+ "judge": "deepseek",
458
+ "judgeModel": "deepseek-chat"
459
+ },
460
+ {
461
+ "schemaVersion": "1.0.0",
462
+ "skillId": "sql-db/sql-db-testing",
463
+ "strictness": "high",
464
+ "trials": 10,
465
+ "triggerAccuracy": {
466
+ "truePositive": 5,
467
+ "falsePositive": 0,
468
+ "positives": 7,
469
+ "negatives": 7
470
+ },
471
+ "evidence": "authored",
472
+ "scenarios": [
473
+ {
474
+ "id": "trigger-positive-1",
475
+ "kind": "trigger-positive",
476
+ "prompt": "Write a test that confirms this migration's rollback leaves the schema the way it was before",
477
+ "strictness": "high",
478
+ "trials": 1,
479
+ "passes": 1,
480
+ "passRate": 1,
481
+ "passAtK": 1,
482
+ "grader": "trigger-rank-fork-family",
483
+ "status": "ran",
484
+ "deterministic": true
485
+ },
486
+ {
487
+ "id": "trigger-positive-2",
488
+ "kind": "trigger-positive",
489
+ "prompt": "Add a test fixture for a user whose invite token expired two days ago",
490
+ "strictness": "high",
491
+ "trials": 1,
492
+ "passes": 0,
493
+ "passRate": 0,
494
+ "passAtK": 0,
495
+ "grader": "trigger-rank-fork-family",
496
+ "status": "ran",
497
+ "deterministic": true
498
+ },
499
+ {
500
+ "id": "trigger-positive-3",
501
+ "kind": "trigger-positive",
502
+ "prompt": "How should I isolate these database tests from each other so one doesn't affect the next?",
503
+ "strictness": "high",
504
+ "trials": 1,
505
+ "passes": 0,
506
+ "passRate": 0,
507
+ "passAtK": 0,
508
+ "grader": "trigger-rank-fork-family",
509
+ "status": "ran",
510
+ "deterministic": true
511
+ },
512
+ {
513
+ "id": "trigger-positive-4",
514
+ "kind": "trigger-positive",
515
+ "prompt": "Write a regression test proving this order-listing query no longer issues one query per row",
516
+ "strictness": "high",
517
+ "trials": 1,
518
+ "passes": 1,
519
+ "passRate": 1,
520
+ "passAtK": 1,
521
+ "grader": "trigger-rank-fork-family",
522
+ "status": "ran",
523
+ "deterministic": true
524
+ },
525
+ {
526
+ "id": "trigger-positive-5",
527
+ "kind": "trigger-positive",
528
+ "prompt": "Write a test proving the batched backfill script for orders.legacy_status touches every row with no row skipped or double-updated",
529
+ "strictness": "high",
530
+ "trials": 1,
531
+ "passes": 1,
532
+ "passRate": 1,
533
+ "passAtK": 1,
534
+ "grader": "trigger-rank-fork-family",
535
+ "status": "ran",
536
+ "deterministic": true
537
+ },
538
+ {
539
+ "id": "trigger-positive-6",
540
+ "kind": "trigger-positive",
541
+ "prompt": "Add test cases for this query's empty-result and boundary conditions",
542
+ "strictness": "high",
543
+ "trials": 1,
544
+ "passes": 1,
545
+ "passRate": 1,
546
+ "passAtK": 1,
547
+ "grader": "trigger-rank-fork-family",
548
+ "status": "ran",
549
+ "deterministic": true
550
+ },
551
+ {
552
+ "id": "trigger-positive-7",
553
+ "kind": "trigger-positive",
554
+ "prompt": "Set up a test that seeds fifty thousand rows to check this migration at realistic scale",
555
+ "strictness": "high",
556
+ "trials": 1,
557
+ "passes": 1,
558
+ "passRate": 1,
559
+ "passAtK": 1,
560
+ "grader": "trigger-rank-fork-family",
561
+ "status": "ran",
562
+ "deterministic": true
563
+ },
564
+ {
565
+ "id": "trigger-negative-1",
566
+ "kind": "trigger-negative",
567
+ "prompt": "Write pytest fixtures for this Django model's test suite",
568
+ "strictness": "high",
569
+ "trials": 1,
570
+ "passes": 1,
571
+ "passRate": 1,
572
+ "passAtK": 1,
573
+ "grader": "trigger-rank-fork-family",
574
+ "status": "ran",
575
+ "deterministic": true
576
+ },
577
+ {
578
+ "id": "trigger-negative-2",
579
+ "kind": "trigger-negative",
580
+ "prompt": "Add RSpec tests for this Rails ActiveRecord scope",
581
+ "strictness": "high",
582
+ "trials": 1,
583
+ "passes": 1,
584
+ "passRate": 1,
585
+ "passAtK": 1,
586
+ "grader": "trigger-rank-fork-family",
587
+ "status": "ran",
588
+ "deterministic": true
589
+ },
590
+ {
591
+ "id": "trigger-negative-3",
592
+ "kind": "trigger-negative",
593
+ "prompt": "Write Jest tests for this React component's rendering",
594
+ "strictness": "high",
595
+ "trials": 1,
596
+ "passes": 1,
597
+ "passRate": 1,
598
+ "passAtK": 1,
599
+ "grader": "trigger-rank-fork-family",
600
+ "status": "ran",
601
+ "deterministic": true
602
+ },
603
+ {
604
+ "id": "trigger-negative-4",
605
+ "kind": "trigger-negative",
606
+ "prompt": "Fix this failing Go test that uses testify assertions",
607
+ "strictness": "high",
608
+ "trials": 1,
609
+ "passes": 1,
610
+ "passRate": 1,
611
+ "passAtK": 1,
612
+ "grader": "trigger-rank-fork-family",
613
+ "status": "ran",
614
+ "deterministic": true
615
+ },
616
+ {
617
+ "id": "trigger-negative-5",
618
+ "kind": "trigger-negative",
619
+ "prompt": "Set up a mock server for this API integration test",
620
+ "strictness": "high",
621
+ "trials": 1,
622
+ "passes": 1,
623
+ "passRate": 1,
624
+ "passAtK": 1,
625
+ "grader": "trigger-rank-fork-family",
626
+ "status": "ran",
627
+ "deterministic": true
628
+ },
629
+ {
630
+ "id": "trigger-negative-6",
631
+ "kind": "trigger-negative",
632
+ "prompt": "Write end-to-end tests for this checkout flow using Playwright",
633
+ "strictness": "high",
634
+ "trials": 1,
635
+ "passes": 1,
636
+ "passRate": 1,
637
+ "passAtK": 1,
638
+ "grader": "trigger-rank-fork-family",
639
+ "status": "ran",
640
+ "deterministic": true
641
+ },
642
+ {
643
+ "id": "trigger-negative-7",
644
+ "kind": "trigger-negative",
645
+ "prompt": "Add unit tests for this Python function that formats currency",
646
+ "strictness": "high",
647
+ "trials": 1,
648
+ "passes": 1,
649
+ "passRate": 1,
650
+ "passAtK": 1,
651
+ "grader": "trigger-rank-fork-family",
652
+ "status": "ran",
653
+ "deterministic": true
654
+ },
655
+ {
656
+ "id": "fixed-clock-fixture",
657
+ "kind": "behavior",
658
+ "prompt": "I'm writing a test fixture for a token that should already be expired. How should I set the expiry timestamp?",
659
+ "strictness": "high",
660
+ "trials": 10,
661
+ "passes": 10,
662
+ "passRate": 1,
663
+ "passAtK": 1,
664
+ "grader": "judge",
665
+ "status": "ran",
666
+ "trialRecords": [
667
+ {
668
+ "output": "Set it from a fixed, injected reference time — not a live `NOW()`/`CURRENT_TIMESTAMP` evaluated when the test runs.\n\nConcretely:\n\n- Pick one reference instant for the test, e.g. `now = \"2024-01-15T12:00:00Z\"` (or a `datetime`/timestamp value the project's fixture helper accepts), and use that same value everywhere the test needs \"now.\" If the code under test takes a clock/time provider, inject this value into it rather than letting it read the system clock.\n- Set the expired token's expiry relative to that reference: `expires_at = now - interval '1 hour'` (or `now - 1h`/`timedelta(hours=1)` depending on stack). Pick a margin comfortably outside the validity window — 1 hour, not 1 second — so it isn't sensitive to scheduling jitter.\n- Name the fixture descriptively: `user_with_expired_token`, not `test_token_3`.\n\nWhy this matters: with a live `NOW()` and a tight margin, a row right at the expiry boundary can flip between \"expired\" and \"valid\" depending on exactly when the test run executes — a flaky test that passes on your machine and fails in CI. A fixed reference time plus a generous margin makes expiration deterministic.\n\nThe same fixed-reference-time rule applies to the sibling boundary case (a token that just barely should still be valid): assert on that row too, constructed from the same injected `now`, so the fixture proves the boundary is on the correct side rather than just sampling a point far from it.",
669
+ "outputSha256": "848b0f29432f15d94ed3b02f78136b59fd0520ba0421cf920ba87c9eacc8d6d2",
670
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
671
+ "deterministic": [],
672
+ "judge": {
673
+ "verdict": "pass",
674
+ "reason": "Answer sets expiry from a fixed injected reference time: 'now = \"2024-01-15T12:00:00Z\"' and 'expires_at = now - interval '1 hour'', explicitly not live NOW(). It explains live-clock boundary values flip between expired/valid depending on run timing, causing flakiness in CI. No fail criterion met."
675
+ },
676
+ "passed": true
677
+ },
678
+ {
679
+ "output": "Build it from a fixed, injected reference time, not a live clock call at test-run time.\n\nConcretely: if the code under test reads \"now\" from an injectable source (a `Clock`, a `now_fn`, a `TimeProvider`, whatever the project uses), inject a fixed reference instant into the test — e.g. `REFERENCE_TIME = 2024-01-15T12:00:00Z` — and set the fixture's expiry to something relative to that, like `REFERENCE_TIME - 1 hour`. Any boundary case (expires-at versus expires-before, by one second, by one microsecond) is then deterministic and reproducible forever, regardless of when or how many times the test runs.\n\nThe problem with `NOW()`, `CURRENT_TIMESTAMP`, `time.time()`, `datetime.utcnow()`, or equivalent evaluated at test-run time is that the expiry-vs-now comparison happens against two *different* instants: the fixture's construction time and the assertion's evaluation time. Between them the clock advances, and a \"just expired\" fixture can drift across the boundary — flaky in the direction that makes it look like it still has 999ms of life left. A fixture that's \"expired by a comfortable margin\" (e.g. one day) is *less* flaky but still wrong: it doesn't pin down the boundary behavior the test is presumably there to check, and it silently passes even if the expiry logic is off by hours.\n\nIf the project has no injectable time source, prefer (in order):\n\n1. Have the fixture helper set an explicit absolute timestamp — `expires_at = '2020-01-01T00:00:00Z'`, permanently in the past relative to any real run — rather than a relative-to-now expression. Absolute-and-old is at least deterministic in shape, though it only tests \"expired, by a lot,\" not the boundary.\n2. If the code under test truly hard-codes a live clock and can't be changed for this test, say so and narrow the test to \"expired\" rather than \"just expired,\" and note the missing boundary coverage in the report rather than pretending a live-relative fixture covers it.\n3. Do not paper over it by freezing the whole process's clock globally unless the project already does that elsewhere — match the project's existing convention (Step 1), don't introduce a second time-control mechanism.\n\nName the fixture for what it represents — `token_expired_before_reference_time`, not `test_token_3` — and, if the boundary matters, pair it with a sibling fixture just on the *other* side of it (`token_expiring_one_second_after_reference_time`) so the test proves the comparison is exclusive, not merely \"an old token is rejected.\"\n\nThis is just the setup half of the answer; the assertion half follows the same rule from `rules/testing.mdc` — assert on the actual outcome (rejected/accepted, and any error/time returned), not merely that validation \"ran without error.\"",
680
+ "outputSha256": "c325775a4fbb7a3b093f23c79583345e5e0a18785300ba9623fb93ef847fb8c1",
681
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
682
+ "deterministic": [],
683
+ "judge": {
684
+ "verdict": "pass",
685
+ "reason": "Answer explicitly says build expiry from a fixed injected reference time (REFERENCE_TIME = 2024-01-15T12:00:00Z, expiry = REFERENCE_TIME - 1 hour), not a live clock. It explains live NOW()/CURRENT_TIMESTAMP makes pass/fail depend on when the test runs (clock advances between construction and assertion, causing flakiness). No fail criterion: it warns against NOW()-based fixtures rather than recommending them."
686
+ },
687
+ "passed": true
688
+ },
689
+ {
690
+ "output": "Set the expiry timestamp from a fixed, injected reference time — not a live `NOW()`/`CURRENT_TIMESTAMP` evaluated at test-run time.\n\nConcretely, in the fixture:\n\n- Get a single reference time once (e.g. a `now`/`reference_time` value the test controls, or a frozen-clock helper the project already uses).\n- Derive the expired timestamp from it, e.g. `expired_at = reference_time - INTERVAL '1 hour'` (or whatever margin comfortably clears the \"expired\" boundary), and pass that literal into the insert/factory.\n- Keep the margin well away from the exact boundary so the test isn't sensitive to however long the fixture setup and assertions take.\n\nWhy not `NOW()`: if the expiry is computed against the live clock at run time, a token that \"just expired\" sits right at the boundary, and whether the query under test treats it as expired becomes a race with test-run timing — flaky, and it can pass or fail depending on machine load.\n\nTwo related notes:\n\n- If the test genuinely needs to exercise the boundary itself (a token that expired one instant ago vs. one still valid), that's where the fixed reference time matters most — construct both fixtures as offsets from the same injected `reference_time` and assert on the pair.\n- Name the fixtures descriptively, e.g. `user_with_expired_token`, not `test_user_3`, so the intent is legible at the assertion site.\n\nThen assert on the actual result — that the expired token is rejected / excluded — rather than only that the query ran without error.",
691
+ "outputSha256": "cdc629e1170d8b98a6f58e5c19edbdec208aa19a411017c9c206544b2b8ec67d",
692
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
693
+ "deterministic": [],
694
+ "judge": {
695
+ "verdict": "pass",
696
+ "reason": "Pass 1: answer says 'Set the expiry timestamp from a fixed, injected reference time' and shows 'expired_at = reference_time - INTERVAL '1 hour''. Pass 2: explains live NOW() makes token sit at boundary, 'a race with test-run timing — flaky'. No fail criterion: it explicitly rejects NOW()."
697
+ },
698
+ "passed": true
699
+ },
700
+ {
701
+ "output": "Set it from a fixed, injected reference time — not a live `NOW()`/`CURRENT_TIMESTAMP` evaluated at test-run time.\n\nConcretely:\n\n1. Pick a single reference time the test controls, e.g. `reference_time = parse(\"2024-01-01T00:00:00Z\")`, and use it for both the fixture's expiry and for any \"current time\" the code under test compares against (pass it in as a parameter / freeze the clock / inject a clock dependency — whichever the project's test conventions already use, per Step 1 of the workflow).\n\n2. Set the fixture's expiry relative to that reference, e.g. `expires_at = reference_time - timedelta(minutes=5)`. That gives you a token that is unambiguously expired, and it stays expired no matter when or how fast the test suite actually runs.\n\n3. If you also want a boundary case (token that just expired, or that expires in one second), build it from the same fixed reference: `expires_at = reference_time - timedelta(seconds=1)`. The whole point is that the delta between \"now\" and the expiry is exact and reproducible.\n\nWhy not just `NOW() - INTERVAL '5 minutes'`?\n\n- It looks fine most of the time, but the boundary case is flaky: whether a \"just expired\" row is actually seen as expired depends on the exact milliseconds between fixture insert and the assertion running, which varies run to run and machine to machine.\n- It also breaks down the moment the code under test caches or rounds timestamps — the gap between a live-`NOW()` fixture and a separately-evaluated live `NOW()` in the query is nonzero and uncontrolled.\n\nOne extra note for this specific fixture: name it descriptively (`user_with_expired_token`, not `test_user_3`) and seed only what the test needs, per the skill's fixture conventions. And this is exactly the kind of fixture where a live clock is the tempting shortcut and the fixed reference time is the correct one — it's called out directly in the Red Flags table:\n\n> \"I'll seed with `NOW()` for the expiry fixture, it's close enough\" → A live clock value makes boundary cases (a row that just expired) flaky depending on exact test-run timing; use a fixed reference time.",
702
+ "outputSha256": "a1039fbcd2e37fe840825011bcef4f69e57ecf22c423bf54bec4dbf38815a0fe",
703
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
704
+ "deterministic": [],
705
+ "judge": {
706
+ "verdict": "pass",
707
+ "reason": "Answer uses a fixed injected reference_time (parse(\"2024-01-01T00:00:00Z\")) and sets expires_at = reference_time - timedelta(minutes=5), explicitly rejecting live NOW()/CURRENT_TIMESTAMP. It explains the live-clock boundary is flaky because pass/fail depends on exact milliseconds between insert and assertion, varying run to run and machine to machine."
708
+ },
709
+ "passed": true
710
+ },
711
+ {
712
+ "output": "Use a fixed reference time captured once at the top of the test, and derive the expiry from it — never a live `NOW()` / `CURRENT_TIMESTAMP` evaluated at assertion time.\n\n```sql\n-- in the fixture setup\nSET @ref_time = '2024-01-15 12:00:00'; -- fixed reference, injected\n\nINSERT INTO tokens (user_id, token, expires_at)\nVALUES (1, 'expired-token', @ref_time - INTERVAL 1 HOUR);\n```\n\nor in a language-level fixture:\n\n```python\nREF_TIME = datetime(2024, 1, 15, 12, 0, 0)\n\ndef expired_token():\n return Token(user_id=1, token=\"expired-token\", expires_at=REF_TIME - timedelta(hours=1))\n```\n\nWhy this matters:\n\n- **A live clock makes the boundary flaky.** If the fixture writes `NOW() - interval '1 second'` and the test asserts something in the *same test run* that re-evaluates \"now\" a moment later, the token may or may not still be expired depending on exact timing — a classic intermittently-failing test.\n- **A fixed reference makes the case self-documenting.** `REF_TIME - 1 hour` says \"expired one hour before the reference moment,\" which is the intent; `NOW() - 1 hour` only says \"expired relative to whenever this happened to run.\"\n- **The implementation under test should also be fed the same reference clock.** If the query filters on `expires_at < :now`, pass `REF_TIME` in as that parameter rather than letting the query call the database's own `NOW()`. Otherwise the fixture is pinned but the comparison isn't, and you're back to a moving boundary.\n\nFor the boundary case specifically (a token that expires *exactly* at the reference instant), add a second fixture at `REF_TIME` itself, so you pin down whether the query treats the boundary as inclusive (`<`) or exclusive (`<=`) rather than leaving it to whichever side the clock happens to land on.",
713
+ "outputSha256": "06395eae6f148f4ebbeaacca2fc71c0db62d87f7976b9de0c320a0a94377ffd5",
714
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
715
+ "deterministic": [],
716
+ "judge": {
717
+ "verdict": "pass",
718
+ "reason": "Answer uses a fixed injected reference time: SET @ref_time = '2024-01-15 12:00:00' and REF_TIME = datetime(2024,1,15,12,0,0), deriving expiry as @ref_time - INTERVAL 1 HOUR / REF_TIME - timedelta(hours=1). It explains a live clock makes the boundary flaky ('may or may not still be expired depending on exact timing'). No live-clock fixture recommended."
719
+ },
720
+ "passed": true
721
+ },
722
+ {
723
+ "output": "Set the fixture's expiry relative to a **fixed, injected reference time** — not to a live `NOW()`/`CURRENT_TIMESTAMP` evaluated at test-run time.\n\nConcretely:\n\n- Define one reference instant once for the test (e.g. a constant `REFERENCE_NOW = '2026-01-01T00:00:00Z'`, or whatever the project's convention is — a frozen-clock/freezegun fixture, an injected `now` parameter, a explicit literal).\n- Derive the fixture from that reference: `expires_at = REFERENCE_NOW - interval '1 hour'` for \"already expired.\" If there's a small grace/skew window the app checks against, put the value clearly past it (e.g. minus an hour, not minus one second) so a boundary case isn't what you're accidentally testing.\n- If the code under test reads the current time itself, make it read the same injected reference — freeze/inject the clock rather than hoping the run happens to land on the right side of the boundary.\n\nWhy this matters: with a live `NOW()`, \"already expired\" is only true if the test run happens to execute after the expiry you wrote — run the suite fast enough, or run it right at a second boundary, and the token may still read as valid, producing a flaky pass/fail that has nothing to do with the code under test. A fixed reference makes the fact \"this token is expired\" true deterministically on every run.\n\nIf you also want to assert the *other* side, add a sibling fixture built the same way with `+ interval '1 hour'` (still-valid) from the same reference, so both cases resolve against one clock.",
724
+ "outputSha256": "d30a18854424b0e4d849e4a638bb7319920e676a4c646b0ebcd4c1b167c5eaa6",
725
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
726
+ "deterministic": [],
727
+ "judge": {
728
+ "verdict": "pass",
729
+ "reason": "Answer uses a fixed injected reference (REFERENCE_NOW constant / frozen clock) and derives expires_at = REFERENCE_NOW - interval '1 hour', explicitly not live NOW(). It explains a live NOW() makes 'already expired' depend on when the run executes, causing flaky pass/fail. No fail criterion: it never recommends NOW() as the basis."
730
+ },
731
+ "passed": true
732
+ },
733
+ {
734
+ "output": "Set it from a **fixed reference time**, not from a live `NOW()` / `CURRENT_TIMESTAMP` evaluated at test-run time — and definitely not something like `NOW() - INTERVAL '1 second'`, which is the flakiest version of all.\n\n**Why not a live clock:** a fixture built from the database's or the test process's current time makes the boundary case depend on exactly when the test happens to run. `NOW() - INTERVAL '1 second'` looks expired, but if anything in the test takes a moment — a slow fixture insert, a clock skew between the DB and the app server, a query that re-evaluates `NOW()` a second later — what you intended as \"already expired\" can end up on the wrong side of the boundary, or you get a test that passes locally and flakes in CI. Worse, it can't distinguish \"expired by a comfortable margin\" from \"expired by one second,\" so it doesn't actually pin down the boundary you're trying to test.\n\n**What to do instead:**\n\n1. **Generate the timestamp in the test language from a fixed, injected reference time**, then pass it into the fixture as a value. In Python-ish pseudocode:\n\n ```\n REFERENCE_NOW = datetime(2025, 1, 15, 12, 0, 0, tzinfo=timezone.utc)\n expires_at = REFERENCE_NOW - timedelta(hours=1) # comfortably, deliberately expired\n token = make_token(expires_at=expires_at)\n ```\n\n The exact value doesn't matter as long as it's fixed and expressed as a concrete offset from a fixed reference. Name the variable so the intent reads at a glance — `user_with_expired_token` and `expired_at = REFERENCE_NOW - timedelta(hours=1)`, not `test_token_3` with `ts = now() - 1`.\n\n2. **If the code under test takes a \"current time\" argument, inject the same fixed reference** so the expiry comparison is deterministic on both sides of the boundary. Don't let the app call the wall clock while the fixture is pinned to a literal.\n\n3. **If the code hard-codes the wall clock** and you can't inject it, then instead of racing a one-second-expired token, pick an offset comfortably past the expiry window (e.g. an hour, or a day) so ordinary timing jitter can't flip it — and add a separate, deliberately narrow boundary test only if the expiry logic itself is what you're validating.\n\n4. **Handle the timezone explicitly.** Store and compare in UTC (or whatever the schema standardizes on) so the fixed instant you wrote is the instant the database sees.\n\nThe same rule from the skill's Step 3 applies to any \"now\"-relative fixture: construct boundary values from a fixed, injected reference time, never from a live clock read. The payoff is a fixture that is genuinely expired every single run, not expired-by-luck.",
735
+ "outputSha256": "c37d0f63b78ca47f0d49b1215fb098dc99122ac048b617a5135368318a0418b8",
736
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
737
+ "deterministic": [],
738
+ "judge": {
739
+ "verdict": "pass",
740
+ "reason": "Answer uses a fixed injected reference time: REFERENCE_NOW = datetime(2025,1,15,12,0,0) with expires_at = REFERENCE_NOW - timedelta(hours=1), and explicitly rejects NOW()/CURRENT_TIMESTAMP and NOW() - INTERVAL '1 second'. It explains a live clock makes the boundary depend on when the test runs, causing flakes (slow insert, clock skew, CI)."
741
+ },
742
+ "passed": true
743
+ },
744
+ {
745
+ "output": "Use a fixed, injected reference time — never a live clock value like `NOW()` / `CURRENT_TIMESTAMP` evaluated at test-run time.\n\nFor a token that should already be expired, derive the expiry from a fixed reference timestamp and subtract a clear margin from it, rather than comparing against the wall clock. Two ways that look the same but are not:\n\n- **Fragile:** the fixture computes the expiry from `NOW()` at fixture-creation time. The test passes only because the assertion also happens \"right now,\" so the gap is tiny — but if the suite is slow, the test is retried, or the assertion happens a moment later, a boundary case flickers between pass and fail. Worse, it can accidentally pass or fail depending on how far past expiry you subtracted.\n- **Robust:** the fixture and the query/assertion both take the same injected reference timestamp (`@fixed_now` / a frozen-clock parameter), and the expiry is set to `fixed_now - <margin>` with a margin comfortably past the expiry boundary — e.g. `fixed_now - interval '1 hour'` for a token whose expiry is minutes-scale, so the \"already expired\" fact is unambiguous and time-independent.\n\nThe rule from the skill: construct any \"now\"-relative fixture value from a fixed, injected reference time — never a live `NOW()`/`CURRENT_TIMESTAMP` for a boundary case.\n\nConcretely, following the fixture conventions you found in Step 1 (use the project's existing factory/helper, name it descriptively):\n\n- `expired_token` → expiry = `fixed_now - interval '1 hour'`\n- `valid_token` (for the contrast case) → expiry = `fixed_now + interval '1 hour'`\n\nSeed both so the test proves the query actually discriminates expired from valid tokens — a test that only seeds the expired token can pass even if the expiry predicate is broken and rejects everything. Assert on the actual result rows returned, not just that the query ran.",
746
+ "outputSha256": "8291bc223185408dee6987b35e8c1778ae0e9a3384f9aee9ab669403a913ef26",
747
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
748
+ "deterministic": [],
749
+ "judge": {
750
+ "verdict": "pass",
751
+ "reason": "Answer uses fixed injected reference time: 'expiry = fixed_now - interval '1 hour'' and '@fixed_now / a frozen-clock parameter', explicitly rejecting live NOW()/CURRENT_TIMESTAMP. It explains the live-clock fragility: 'if the suite is slow, the test is retried, or the assertion happens a moment later, a boundary case flickers between pass and fail.' Both pass criteria concretely met; no fail criterion."
752
+ },
753
+ "passed": true
754
+ },
755
+ {
756
+ "output": "Set it from a fixed, injected reference time, not a live call to `NOW()`/`CURRENT_TIMESTAMP` evaluated when the test happens to run.\n\nConcretely: pick a frozen \"now\" for the test (the value the code under test and the fixture both read, e.g. an injected clock, a `frozen_time`/`freeze_time` helper, or just a constant passed into the fixture), and build the expiry as that reference time minus some clear margin:\n\n```\nEXPIRED_TOKEN = user_with_token(expires_at=REFERENCE_TIME - timedelta(hours=1))\n```\n\nWhy this matters: if you seed with a live `NOW()` minus one second, the row is only *just* expired at the instant of seeding. Under transaction wrapping, a slow CI box, or a test that advances time, the \"expired\" fixture can cross back over the boundary or race the assertion, producing a flaky test that fails intermittently and gets \"fixed\" by rerunning. A fixed reference time makes the relationship between the fixture and \"now\" deterministic and reproducible, which is exactly what a boundary case needs.\n\nTwo related points from the fixture rules:\n\n- Name it for what it asserts about the domain — `user_with_expired_token`, not `test_user_3` — so a reader sees the intent without decoding the timestamp.\n- If the test also needs a *non*-expired twin (e.g. asserting that a valid token is accepted and an expired one rejected in the same suite), derive both from the same reference time so they differ only in the one dimension under test:\n\n```\nREFERENCE_TIME = <fixed>\nVALID = user_with_token(expires_at=REFERENCE_TIME + timedelta(hours=1))\nEXPIRED = user_with_token(expires_at=REFERENCE_TIME - timedelta(hours=1))\n```\n\nOne caveat: if the code path under test reads the database's own clock rather than a passed-in reference (e.g. the validation is a SQL `WHERE expires_at > NOW()`), injecting a time into the fixture alone won't pin the comparison — the two clocks are independent. In that case either freeze the database's clock too (test-harness time control, a fixed `NOW()` function/mock where the harness supports it), or seed the expiry far enough in the past that the boundary can't be raced, and assert on the result rather than only that the query ran.\n\nSeeding only what the test needs, from a fixed reference time, is the whole point of Step 3.3 of the testing rules: construct \"now\"-relative fixture values from an injected reference time and never from a live `NOW()` for a boundary case.",
757
+ "outputSha256": "189a54cbbf57a976deafaccd399b120f73852f81dacccd9a4ce3057b18e29921",
758
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
759
+ "deterministic": [],
760
+ "judge": {
761
+ "verdict": "pass",
762
+ "reason": "Answer explicitly says to use a fixed injected reference time, not live NOW()/CURRENT_TIMESTAMP, and shows concrete code: EXPIRED_TOKEN = user_with_token(expires_at=REFERENCE_TIME - timedelta(hours=1)). It explains live-clock seeding makes the test flaky/race the boundary. No fail criterion: it warns against live NOW() rather than recommending it."
763
+ },
764
+ "passed": true
765
+ },
766
+ {
767
+ "output": "Set it from a fixed, injected reference time — never a live `NOW()`/`CURRENT_TIMESTAMP` evaluated at test-run time.\n\nConcretely, the fixture should take the \"current time\" as a parameter the test controls, then compute the expiry relative to that:\n\n```\ndef expired_token(now, ttl_seconds=3600):\n return token(expires_at=now - timedelta(seconds=1))\n```\n\nor, at the SQL level, insert `expires_at = :reference_time - interval '1 second'`\nrather than `expires_at = NOW() - interval '1 second'`.\n\nWhy this matters:\n\n- **Boundary cases are the point.** A test whose fixture says \"expired one second ago relative to wall-clock now\" is asserting against a moving target. If the query under test compares `expires_at < NOW()`, the row is expired at insert time and still expired a millisecond later — but a fixture built as \"expires in exactly 0 seconds\" or \"expires at the same millisecond as now\" can flip pass/fail based on test-run timing and CI load.\n- **Reproducibility.** A failing token-expiry test with a hardcoded or clock-derived timestamp isn't reproducible on a rerun; one built from a pinned reference time is. If you need the fixture to be deterministic across runs, pin the reference time to a fixed epoch in the test (e.g. `datetime(2024, 1, 1, tzinfo=timezone.utc)`) and pass that same value to both the fixture and the code path that reads \"now,\" if that path is injectable.\n- **Timezone correctness.** Build the reference time in UTC (or the app's canonical zone) and be explicit about it. A naive local-time `NOW()` in a fixture is a common source of tests that pass in one CI runner's timezone and fail in another.\n\nIf the application code under test itself calls `NOW()` internally and doesn't accept an injected clock, the fixture can still be built from a fixed reference time — just make the margin unambiguous and not zero. \"Expired by an hour\" from a pinned `now` is safe; \"expired by zero seconds\" from a live clock is flaky. And per the skill's rules, the fixture value should be named descriptively (`token_expired_an_hour_ago`) so the boundary intent is visible at the call site.\n\nOne caveat tied to the isolation rule: since this fixture is time-relative, make sure you're not mutating a shared row's `expires_at` in a way that leaks across tests — build it inside the transactional-rollback wrapper (or the project's equivalent) like any other fixture, and don't reuse a module-level token object across tests.",
768
+ "outputSha256": "0bb51759d2d342038471d9834eca0bbf4cc6404ae11a72fbebad0def48a9cbc8",
769
+ "promptSha256": "7e042636eb4d4df45e0d21e5fad6adb8cac3383bceff83ff936ee5b793c7e703",
770
+ "deterministic": [],
771
+ "judge": {
772
+ "verdict": "pass",
773
+ "reason": "Answer uses a fixed/injected reference time: `expired_token(now, ...)` with `expires_at=now - timedelta(seconds=1)` and SQL `:reference_time - interval '1 second'`, plus pinning to `datetime(2024,1,1,...)`. It explains live-clock boundary makes pass/fail depend on run timing ('moving target', 'flip pass/fail based on test-run timing and CI load'). No fail criterion: it explicitly rejects NOW()."
774
+ },
775
+ "passed": true
776
+ }
777
+ ]
778
+ },
779
+ {
780
+ "id": "n1-regression-query-count",
781
+ "kind": "behavior",
782
+ "prompt": "I fixed an N+1 query in the order-listing endpoint by switching to a single JOIN. How should I write the regression test so it doesn't silently regress back to N+1 later?",
783
+ "strictness": "high",
784
+ "trials": 10,
785
+ "passes": 10,
786
+ "passRate": 1,
787
+ "passAtK": 1,
788
+ "grader": "judge",
789
+ "status": "ran",
790
+ "trialRecords": [
791
+ {
792
+ "output": "# Regression test for the N+1 fix on the order-listing endpoint\n\nThe trap with an N+1 fix is that **the result set stays correct even when the N+1 comes back**. If someone later \"simplifies\" the JOIN away, or an ORM lazy-load sneaks in, every value-assertion test still passes green while the query count silently goes from 1 to 1+N. So the regression test has two independent assertions, and both must be present:\n\n1. **Result correctness** — the endpoint returns the right orders, with the right customer data joined in, in the right order.\n2. **Query count** — the endpoint (or repository function under test) issues a constant number of queries regardless of how many orders are returned.\n\nAssertion 2 is the one that actually guards against N+1. Assertion 1 alone does not.\n\n## Structure\n\nUse whatever query-counting hook the harness already exposes — most ORMs and test frameworks have a query log, a query counter, or a `listens_for`/`before_cursor_execute` event you can subscribe to. Do not hand-roll an instrumented connection if the project already has one; grep for an existing helper first (per the skill's Step 1).\n\n```\ntest_order_listing_query_count_is_constant:\n seed 3 orders, each with its own customer\n\n reset/clear query log\n response = client.get(\"/orders\")\n assert response.status_code == 200\n assert [o[\"id\"] for o in response.json()] == expected_ids\n assert len(query_log) == <constant> # e.g. 1, or a small fixed number\n\n seed 30 more orders, each with its own customer\n reset/clear query log\n response = client.get(\"/orders\")\n assert len(response.json()) == 33\n assert len(query_log) == <same constant> # <-- the actual regression guard\n```\n\nThe second half is the point. Comparing the query count at N=3 against the query count at N=33 (same constant, not scaled with row count) is what makes this a *counting* regression test rather than a one-off assertion that happens to be true today. A plain `assert len(query_log) == 1` on a single fixture size would pass today and could still be re-broken by a change that happens to also run one query at that fixture size.\n\nIf the ordering endpoint paginates, seed past one page boundary and assert the count is constant across pages too — a common place an N+1 reappears is the \"load relations for the current page\" loop.\n\n## Fixture notes\n\n- Seed each order with a **distinct** customer. If several orders share a customer, a broken N+1 that caches per-distinct-customer could still show a low query count and the test would pass. Distinct customers make the count assertion honest.\n- Add the fixtures through the project's existing factory/helper, not raw `INSERT`s, so schema changes don't silently break the fixture.\n- If the endpoint's order is time-dependent, pin the seed timestamps to fixed values and assert the ordering explicitly — don't rely on insertion order.\n\n## What not to do\n\n- Don't assert only on the response body. That is exactly the test that stays green through the regression.\n- Don't assert on a raw `EXPLAIN` plan or assume a specific JOIN. The test should guard the *observable property* (constant query count), not the specific SQL strategy — a future legitimate rewrite to a batched second query should still be allowed to pass if the count stays constant.\n- Don't modify the endpoint/query to make the test pass. If the new test fails because the query count is scaling with row count, that is the test correctly catching a regression, not a test bug.\n\n## Report shape\n\n```\nGenerated: tests/orders/test_order_listing_no_n_plus_one.py\n - Asserts result rows/ordering for 3 and 33 orders (distinct customers)\n - Asserts query count is the same constant at both sizes (regression guard)\n - One extra case seeding past the pagination boundary, count still constant\n```",
793
+ "outputSha256": "d13cc27d51f12400d08995232a22a89ac4e6afdd26fb0772b2d4128f7dc14399",
794
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
795
+ "deterministic": [],
796
+ "judge": {
797
+ "verdict": "pass",
798
+ "reason": "Answer asserts result correctness (assert [o[\"id\"] for o in response.json()] == expected_ids; assert len(response.json()) == 33) AND query count via query_log (assert len(query_log) == <constant> at both N=3 and N=33), explicitly noting result-only assertions stay green through regression."
799
+ },
800
+ "passed": true
801
+ },
802
+ {
803
+ "output": "To keep that N+1 fix from silently regressing, the test needs to assert two separate things, not just the result set:\n\n**1. Result correctness (the output the endpoint promises)**\nAssert on the actual rows, values, and ordering the endpoint returns — the same rows the JOIN produces today. This is what catches a later \"optimization\" that fixes the query count but mangles the result. If the endpoint specifies an `ORDER BY`, assert the ordering, not just the set.\n\n**2. Query count (the thing that actually regressed the first time)**\nThis is the part that specifically guards against N+1 coming back. Most ORMs and test harnesses expose either a query counter or a query log you can flush before the call and read after. The pattern:\n\n- Seed N orders that each have a related child (customer, line items — whatever the original N+1 was fetching per row).\n- Reset/clear the query counter right before hitting the endpoint.\n- Assert the endpoint with N=1 and N=large both issue the **same constant number of queries** — the classic N+1 signature is a count that scales with N. Asserting an absolute count is fine too, but asserting it doesn't grow with N is the assertion that survives a legitimate future refactor to two queries instead of one.\n\nConcretely, the count assertion should read like \"1 query for the listing, independent of the number of orders,\" not just \"≤ 2 queries.\" If someone later reintroduces a per-row lazy load, the count becomes `1 + N` and the test fails loudly.\n\nA few things that make this regression test actually bite:\n\n- **Seed the child rows so the per-row fetch is actually tempting.** If the fixture has childless orders (or a nullable relation), a reintroduced N+1 might not fire and the test passes falsely. Give every seeded order a related row.\n- **Test at a representative N, and specifically more than one.** N=3 can slip past a bad batch size that fails at N=100. Use a count well above any batch boundary — a few hundred is cheap and makes a per-row lazy load impossible to hide.\n- **Deterministic fixture times.** If the listing filters or sorts by time (recent orders, expiry), build the fixture from a fixed injected reference time, never live `NOW()` — otherwise the boundary rows flip in and out and the test flakes.\n- **Isolate with the project's existing transactional rollback** so the seeded rows don't leak into other tests.\n\nOne caution: don't make the count assertion brittle to legitimate changes. Asserting \"exactly 1 query\" is fine if that's genuinely the contract, but if a future feature legitimately needs two (a count + a fetch), that test fails for a good reason — which is usually exactly what you want, so it gets an explicit decision rather than a silent N+1. Either way, the query-count assertion is the load-bearing part; result-set-only tests let the N+1 walk right back in.",
804
+ "outputSha256": "7f8f60c2c7460de8948b545f3c076d583f79c29ff15ec0d78a4936d2c5fcce8f",
805
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
806
+ "deterministic": [],
807
+ "judge": {
808
+ "verdict": "pass",
809
+ "reason": "Answer asserts both: (1) result correctness — 'Assert on the actual rows, values, and ordering the endpoint returns'; (2) query count — 'Reset/clear the query counter right before hitting the endpoint' and 'Assert the endpoint with N=1 and N=large both issue the same constant number of queries', with count becoming '1 + N' on regression. Both pass criteria concretely met."
810
+ },
811
+ "passed": true
812
+ },
813
+ {
814
+ "output": "Assert the query count alongside the result set — a correct result set alone won't catch a future change that quietly reintroduces one query per row.\n\n**1. Assert the result set first (values and ordering).** A JOIN that's supposed to replace the per-row lookups must still return the same rows, same values, same `ORDER BY` order. Cover the empty-result case too — an inner JOIN instead of a LEFT JOIN silently drops orders with no matching row, and the happy-path-only test won't see it.\n\n**2. Assert the query count for a fixed fixture size.** Most ORMs/test harnesses expose a query counter or a query log you can snapshot. Seed enough orders to make the difference unambiguous — a handful isn't enough, since one-per-row at low N can look like a constant. Something in the tens-of-rows range makes \"1 query\" vs \"N+1 queries\" obvious and stable. Structure it as: run the listing, capture queries-executed, assert it equals the expected small constant (1, or 2 if there's a genuine second query like a count for pagination).\n\n**3. Make the count assertion a tight equality, not a ceiling.** `assert query_count == 1` catches a regression from 1 back to N-rows. `assert query_count <= 10` wouldn't catch an N+1 creeping in under the ceiling. If the exact constant is brittle (an unrelated extra metadata query), pin it to the observed non-N+1 value and update it deliberately when the endpoint legitimately changes shape — that deliberate update is exactly the friction you want, because it forces a human to notice the query shape changed.\n\n**4. Name the test for the invariant, not the implementation.** `test_listing_orders_does_not_issue_one_query_per_order` reads as the property under test; `test_orders_use_join` describes the implementation and invites someone to \"fix\" the test by reverting to per-row lookups when it fails.\n\n**5. Keep it isolated and self-contained.** Wrap it in the project's existing transactional-rollback convention, seed only the orders and related rows this test needs via the project's existing factory, and don't depend on data another test left behind — a flaky fixture turns this into a test someone deletes rather than a test someone trusts.\n\nOne more thing worth doing if the harness supports it: a companion assertion that the number of distinct tables/relations touched stays fixed, in case the count happens to stay constant while the shape regresses in a way the counter alone misses. That's optional; the count assertion is the one that earns its keep.\n\nNet shape: seed ~tens of orders → run listing → assert exact rows/values/order → assert exact query count → also assert the empty/edge case. Then confirm `git status` shows only the test file changed.",
815
+ "outputSha256": "987ffa2588bb63fd15ffde30fceb7e14174762a99e6ef405c95367e12f9d6b04",
816
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
817
+ "deterministic": [],
818
+ "judge": {
819
+ "verdict": "pass",
820
+ "reason": "Answer asserts result set values/ordering (point 1: 'same rows, same values, same ORDER BY order') and explicitly asserts query count (point 2: 'capture queries-executed, assert it equals the expected small constant (1, or 2...)'; point 3: 'assert query_count == 1'). Both pass criteria concretely satisfied."
821
+ },
822
+ "passed": true
823
+ },
824
+ {
825
+ "output": "Assert two things in the same test: the result set is still correct, **and** the query count stays bounded (constant, not one-per-row). The result assertion alone won't catch a future N+1 regression, because an N+1 loop still returns the right rows — that's the whole trap.\n\nConcretely:\n\n**1. Seed enough rows that N+1 and single-join differ in query count.**\nIf you seed 3 orders, \"3 queries\" and \"1 query\" are easy to confuse and the difference is small. Seed a row count where the old behavior would clearly produce N+1 (e.g. 20-50 orders, each with associated line items/customer), so a regression shows up as a count that scales with row count.\n\n**2. Assert the result set first — values and ordering.**\nThis is the behavior contract. If the endpoint specifies an order, assert the exact order; a JOIN rewrite is a classic place to accidentally change ordering or duplicate rows via a fan-out join. Assert the returned rows match what the pre-refactor code returned.\n\n**3. Assert the query count.**\nUse whatever query-counting mechanism your stack exposes:\n- Most ORMs / test harnesses expose a query counter or a query log/event subscriber you can drain after the call.\n- In raw drivers, wrap the connection in a counting proxy for the test.\nAssert the count is a small constant (e.g. `<= 2`: the main query plus maybe a lookup), and specifically assert it does **not** equal or scale with the number of orders. A strong form: run the same test with a larger seed and assert the count is unchanged — that's what actually proves it's not O(N).\n\nBetter still, assert the query count is **exactly** the expected constant rather than `<= N`. An upper bound alone can hide a partial regression; an exact count makes any extra query a failure.\n\n**4. Guard against false confidence in the count itself.**\n- Make sure the query counter is actually wired to the same connection the endpoint uses — a counter attached to the wrong connection silently reads zero and the test passes for the wrong reason.\n- Reset/drain the counter at the start of the test so setup queries aren't counted, and assert on the count of queries issued **during the endpoint call only**.\n- If the endpoint or ORM has a query cache / identity map, disabling it (or asserting it's off) prevents a second call from being masked.\n\n**5. Isolate it.**\nWrap in the project's transactional rollback so the seeded orders don't leak into other tests. Use realistic-but-descriptive fixture names.\n\nSketch:\n\n```\ndef test_order_listing_does_not_regress_to_n_plus_1(db):\n with db.transaction_rollback(): # project's isolation\n for i in range(30):\n make_order(customer=make_customer(), lines=2)\n\n qc = db.query_counter() # attached to the live connection\n qc.reset()\n\n result = client.get(\"/orders\") # the endpoint under test\n\n # 1. behavior contract: right rows, right shape, right order\n assert [o[\"id\"] for o in result.json()] == expected_ids_in_order()\n assert all(len(o[\"lines\"]) == 2 for o in result.json())\n\n # 2. the actual regression guard: constant, not one-per-row\n assert qc.count() == 1 # or the known constant for this endpoint\n assert qc.count() < len(orders) # explicit: it must not scale with rows\n```\n\nThe key discipline from the skill: **never assert only \"ran without error\" or only \"returned the right rows.\"** For an N+1 fix specifically, the query-count assertion is the regression test; the result assertion is what keeps the count assertion honest about *which* query it's counting.",
826
+ "outputSha256": "f4b587730707ae10df925f6522820f6fb748162e5ee4ddeac7b818b956120c8d",
827
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
828
+ "deterministic": [],
829
+ "judge": {
830
+ "verdict": "pass",
831
+ "reason": "Answer asserts result rows/values (step 2 and sketch: `assert [o[\"id\"] for o in result.json()] == expected_ids_in_order()`, line counts) AND asserts query count via counter (`qc = db.query_counter()`, `qc.reset()`, `assert qc.count() == 1`, `assert qc.count() < len(orders)`), explicitly noting result-only assertion won't catch N+1 regression."
832
+ },
833
+ "passed": true
834
+ },
835
+ {
836
+ "output": "The key point: asserting the endpoint's response is correct does **not** protect against an N+1 regression, because an N+1 query and a JOIN produce the *same rows*. You need a separate assertion on the query count.\n\nStructure the test in two parts:\n\n**1. Result assertion (the part everyone remembers).**\nSeed a handful of orders each with a known number of line items / related rows, hit the listing function, and assert the actual returned rows, values, and ordering — e.g. each order appears once with its items attached, in the order the query specifies. This catches the JOIN itself being wrong (fan-out duplicates, missing rows from an inner join that should be a left join, wrong ordering).\n\n**2. Query-count assertion (the part that catches the N+1 coming back).**\nWrap the call in the project's query counter / query log — most ORMs and test harnesses expose something like `assertNumQueries(n)` or a captured query list. Assert the count is a small constant **independent of how many orders the fixture has**. Two ways to make that robust:\n\n- Assert an exact count for a fixture with, say, 3 orders (e.g. `assertNumQueries(1)` for the listing query, plus whatever fixed setup overhead the harness counts).\n- Better, guard against future growth: run the same listing against a fixture of 3 orders and one of 30, and assert the query count is **identical** for both. That is the property that actually distinguishes a JOIN from an N+1 — a constant query count regardless of row count. An N+1 creeping back in makes the 30-order run issue 10× the queries and fails immediately.\n\nKeep the fixture realistic in shape, not just size: at least two orders with different item counts (including one with zero related rows) so a lazy-loading regression that happens to be skipped for single-item orders still trips.\n\nA few supporting points from the same workflow:\n\n- **Fixture values.** Naming the seeded rows descriptively (`order_with_three_items`, `order_with_no_items`) makes a failure message self-explanatory, versus `order_3`.\n- **Isolation.** Run the test inside the project's existing transaction-per-test wrapper (or its truncate-and-reseed equivalent) so the query log/state doesn't leak between tests — match whatever the suite already does rather than adding a second isolation strategy.\n- **Don't touch the query under test.** If this test fails, that's the signal the JOIN fix regressed — fix the endpoint, not the assertion.\n- **If the suite has no query counter.** Standardize on the harness's query log if one exists; if truly nothing exists, a lightweight counting connection wrapper or a query-log listener set up in the test fixture is enough — but prefer an existing helper so the whole suite shares one mechanism.\n\nSo the one-line answer: assert the rows are right, **and** assert the query count is (a) small and (b) unchanged as the fixture row count grows — the count-invariance assertion is what actually locks out the N+1.",
837
+ "outputSha256": "6f53b31bb1d0c15c137f5e5eb3faa1d21e2f0b47060f880209dadda54d456640",
838
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
839
+ "deterministic": [],
840
+ "judge": {
841
+ "verdict": "pass",
842
+ "reason": "Part 1 asserts actual returned rows/values/ordering (\"assert the actual returned rows, values, and ordering\"). Part 2 asserts query count via counter/log: names `assertNumQueries(n)`, exact count for 3 orders, and count-invariance across 3 vs 30 orders. Both pass criteria concretely met; no fail criterion holds."
843
+ },
844
+ "passed": true
845
+ },
846
+ {
847
+ "output": "Write two assertions in the same test, not just one:\n\n1. **Result assertion** — the query still returns the right rows, in the right order, with the right joined values (e.g. each order carries its customer's name). This catches a later refactor that \"fixes\" the query count but breaks the JOIN semantics, or that swaps the JOIN back and changes which rows match.\n\n2. **Query-count assertion** — capture the number of queries the endpoint/ORM issues during the call and assert it against the expected constant (typically a small fixed number — the base select plus a bounded set of lookups — not scaling with the number of orders returned). Most ORMs and test harnesses expose either a query counter or a query log you can count. Assert against a specific number, not \"less than N\" — a loose upper bound lets N+1 creep back in as long as the fixture stays small.\n\nThe reason the count assertion matters is that the result assertion alone can't catch the regression you're guarding against: an N+1 query returns exactly the same rows as the JOIN — it's just slower. So a future change that reintroduces per-row lookups would pass a result-only test silently.\n\nA few things that make the count assertion actually hold:\n\n- **Seed enough rows that N+1 is distinguishable from 1.** With 3 orders, \"3 plus base\" and \"1 plus base\" may be close enough to slip past a sloppy bound. Use a fixture row count clearly above the expected constant — say 10-20 orders — so one-per-row is unmistakably higher than the fixed count.\n- **Count at the right scope.** Assert on the queries issued for the single endpoint call, not the whole test (fixture setup queries will inflate the number). If the harness counts globally, reset the counter right before invoking the endpoint.\n- **Freeze/limit what else can query.** If the endpoint does other bounded work (a config lookup, an auth check), pin the expected constant to include those, so the test fails only when the order-dependent part changes.\n- **Keep the boundary case.** The empty-result case (filters matching nothing) should still issue the base query and zero per-row lookups — worth its own small assertion if an off-by-one in the JOIN or a missing guard is plausible.\n\nIf the fix you made was in code rather than in the migration/query itself, note that this skill's rule about not modifying the query under test still applies — the regression test asserts on the endpoint's behavior and its query count; it doesn't edit the JOIN. Only touch the endpoint/schema if the test caught a real bug, and say so explicitly if you do.\n\nPer the workflow's discovery step, first check whether the suite already has a query-count helper or convention (many projects do, for exactly this reason) and reuse it rather than introducing a second counting approach — matching the existing convention is the rule, not the suggestion.",
848
+ "outputSha256": "2ba4f9f166ceb2f954f65a2660d269443e143be92d0e39ba06885ca646ff00d9",
849
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
850
+ "deterministic": [],
851
+ "judge": {
852
+ "verdict": "pass",
853
+ "reason": "Answer explicitly requires two assertions: (1) result assertion — 'the query still returns the right rows, in the right order, with the right joined values'; (2) query-count assertion — 'capture the number of queries the endpoint/ORM issues... assert it against the expected constant', using ORM counter/query log. It also explains result-only tests miss N+1. Both pass criteria concretely met."
854
+ },
855
+ "passed": true
856
+ },
857
+ {
858
+ "output": "To make the regression test actually catch a future N+1, it needs to assert two independent things: the query *count*, and the *result set*. A result-only assertion won't catch a regression back to one-query-per-row, because the rows would still come out correct — just slowly. A count-only assertion can be fooled by a rewrite that keeps the count low but drops rows. You need both.\n\n**1. Assert on the query count.**\nMost frameworks expose a hook for this. Find what the project already has (a query log, a listener, a counter, a `connection.queries`-style list) and reuse it — don't hand-roll a second mechanism.\n\nThe assertion should be a fixed number, not a range or a \"less than N\" bound. Something like:\n\n```\nwith assert_query_count(1):\n response = client.get(\"/orders\")\n```\n\nThe key design point is that the fixture must contain **more than one parent row with children** — at least 3-5 orders each with a couple of line items. With a single order the N+1 is invisible: one parent plus one `IN` round trip happens to look the same as a JOIN. The count assertion only becomes meaningful once the fixture is big enough that the JOINed version (1 query) and the per-row version (1 + N queries) are clearly different numbers. `assert_query_count(1)` on a 5-order fixture fails loudly if someone reintroduces `order.items.each { ... }`.\n\nIf the endpoint legitimately needs more than one query (auth lookup, count, then the page of rows), assert the exact constant — `assert_query_count(3)` — so any *additional* per-row query trips it. \"≤ 3\" would let a regression to `3 + N` slip through.\n\n**2. Assert on the actual rows, values, and ordering.**\nBuild a small fixture with known, distinguishable data — e.g. orders with totals `[10, 20, 30]` and two items each — and assert the response returns exactly those orders in the query's `ORDER BY` order, with each order's items nested under it. This is what catches the other failure mode: someone \"fixes\" the N+1 by joining and accidentally collapsing line items, or drops orders that have zero items because the JOIN was inner instead of left. Include one order with **no items** in the fixture specifically so an inner-join regression shows up as a missing row.\n\n**3. Isolate and freeze.**\nWrap the test in the project's existing transactional rollback so it doesn't leak state. And if the endpoint filters by anything time-relative (recent orders, active status), build that fixture value from a fixed reference time, not a live `NOW()` — otherwise the rowset varies by run timing and the count/result assertions go flaky.\n\nPut a one-line comment on the test stating the intent, so the next person doesn't \"simplify\" the fixture down to one order and quietly neuter it:\n\n```\n# 5 orders x 2 items each: the JOIN must be 1 query, not 1+N.\n# Do not shrink this fixture — N+1 is invisible at N=1.\n```\n\nThat last point is the real risk. The count assertion is only as strong as the fixture underneath it, and the most common way this regression test dies is a later cleanup that trims the seed data back to a single row.",
859
+ "outputSha256": "70fa423769534ad9bd985620a7072cc82c92a83ca41f9230f1c52d688ed08360",
860
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
861
+ "deterministic": [],
862
+ "judge": {
863
+ "verdict": "pass",
864
+ "reason": "Answer asserts both: query count via assert_query_count(1) with exact constant (not ≤N), and result rows/values/ordering with fixture totals [10,20,30], nested items, and a no-items order. Explicitly notes result-only assertion won't catch N+1 regression."
865
+ },
866
+ "passed": true
867
+ },
868
+ {
869
+ "output": "Write two assertions in the same test, not just one:\n\n**1. The result assertion** — the joined query returns the same rows, values, and ordering the endpoint contract requires. This catches the JOIN itself being wrong (a `JOIN` where a `LEFT JOIN` belonged, rows dropped, sort changed). Have it cover at least the happy path plus the case that exercises the join condition you just changed (e.g. an order with no line items, if the N+1 loop used to naturally handle that).\n\n**2. The query-count assertion** — this is the one that actually guards against N+1. Use whatever counter your stack exposes: a query logger/hook, a connection-level statement counter, or a test-harness callback. The robust form is to assert the count is **constant in the row count**: seed, say, 3 orders, record the count, seed 30 orders, record the count again, and assert both are equal (or equal to a fixed expected number). A fixed \"assert exactly N queries\" test is brittle — it breaks every time an unrelated eager load is added — whereas \"count doesn't grow with the number of rows\" is precisely the property an N+1 regression violates and nothing else does.\n\nSo:\n\n```\ndef test_order_listing_does_not_regress_to_n_plus_one():\n seed_orders_with_items(3)\n count_small = count_queries(lambda: client.get(\"/orders\"))\n seed_orders_with_items(30)\n count_large = count_queries(lambda: client.get(\"/orders\"))\n\n assert count_small == count_large # query count independent of row count\n rows = client.get(\"/orders\").json()\n assert [r[\"id\"] for r in rows] == expected_ids_in_order # result still correct\n```\n\nA few specifics that make it not silently rot:\n\n- **Assert on the endpoint/query under test, not a re-implementation.** Drive the actual code path (the endpoint handler, or the repository method the endpoint calls), so a future refactor that reintroduces a per-row lazy load in that path is still exercised.\n- **Count only the queries you mean to constrain.** Reset/scope the counter around the call so setup, fixture seeding, and teardown queries don't pollute the count.\n- **Roll back via the project's transactional isolation** so the 30-row seed doesn't leak into the next test.\n- **Seed through the project's existing factory**, and make the extra rows go through the same relationship the N+1 would have triggered (the line-items/related collection), otherwise a reintroduced lazy load on that relationship won't show up as extra queries.\n- **Deterministic seeds, no live clock** — if ordering depends on a timestamp, inject a fixed reference time so both runs produce the same row order.\n\nAnd per the testing rules: don't touch the endpoint/query to make the test pass. If the count-invariance assertion fails after your JOIN change, that's the test correctly telling you the path still issues a per-row query — investigate rather than loosening the assertion.",
870
+ "outputSha256": "42b107be423329f7c29870eba14fc55164daec0038bf819880f1155bc9625f9e",
871
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
872
+ "deterministic": [],
873
+ "judge": {
874
+ "verdict": "pass",
875
+ "reason": "Answer asserts result rows/values (assert [r[\"id\"] for r in rows] == expected_ids_in_order) and query count via count_queries counter, asserting count_small == count_large (count independent of row count), explicitly guarding against N+1 regression."
876
+ },
877
+ "passed": true
878
+ },
879
+ {
880
+ "output": "Write the regression test so it asserts on **both** the result set and the query count — the result assertion alone won't catch a future refactor back to N+1, because a lazy-loaded version returns exactly the same rows.\n\n**1. Assert on the actual result set first.** Build a fixture with a known set of orders, each with a known number of related rows (line items, customer, whatever the N+1 was firing on), then assert the endpoint returns those exact orders with the exact related values. This is your correctness anchor — if someone \"fixes\" a count regression by dropping the JOIN and returning less data, this catches it.\n\n**2. Assert the query count, and make the assertion scale-sensitive.** This is the part that actually locks in the fix. Instrument the query log or the ORM's query counter for the duration of the request, run the endpoint, and assert the count is bounded by a small constant that does **not** grow with the number of orders:\n\n```\nrun the endpoint with N orders\nassert query_count <= 3 # some fixed ceiling, not a function of N\n```\n\nThe key move is to compare against a **second run with more orders**. Seed 3 orders, assert the count; then seed 20 orders in a separate test (or the same test with a second fixture batch), assert the count is the *same*. A JOIN version stays flat; an N+1 version scales linearly and fails the second assertion even if it happened to sneak under a loose absolute ceiling in the first.\n\n**3. Don't assert an exact tight number if the endpoint has incidental queries.** If the endpoint legitimately runs auth/session/count queries, an exact `== 1` will be brittle and get loosened to uselessness the first time something unrelated changes. Anchor the second half of the test on *invariance* (\"count(20 orders) == count(3 orders)\") rather than an absolute, and keep the absolute as a generous ceiling.\n\n**4. Seed through the project's existing fixture/factory helper**, not raw SQL inserts, and name fixtures descriptively (`order_with_three_line_items`).\n\nTwo failure modes to guard against explicitly:\n\n- **The eager-load silently detaches.** If the JOIN fix was really a `preload`/`includes` change and a later edit removes it, the row-count assertion still passes and only the query-count invariance test fails — so the invariance test is the one doing the work, don't skip it as \"redundant with the result check.\"\n- **The test passes because the fixture is too small.** N=1 or N=2 orders can't distinguish flat from linear. Seed enough rows that the N+1 version's count is unmistakably higher (10+ is usually plenty).\n\nFinally: wrap it in the project's transactional rollback isolation like every other DB test, and remember the rule from the workflow above — you're changing **test files only**, not the endpoint or query. If asserting the count surfaces that the \"fix\" is actually still firing per-row under some condition (say, when a related record is missing), that's a real bug: report it rather than loosening the assertion to make the test green.",
881
+ "outputSha256": "d0ad15ed40ad10bc0315aafb1cb3f4bcfb31bedd738dcee1ce25a5eee7c90422",
882
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
883
+ "deterministic": [],
884
+ "judge": {
885
+ "verdict": "pass",
886
+ "reason": "Answer asserts both: (1) exact result rows/values via fixture with known orders and related values ('assert the endpoint returns those exact orders with the exact related values'); (2) query count via ORM counter/query log ('Instrument the query log or the ORM's query counter... assert query_count <= 3'), plus scale-invariance count(20)==count(3). No fail criterion holds."
887
+ },
888
+ "passed": true
889
+ },
890
+ {
891
+ "output": "Write two assertions side by side in the same test, not just the one you'd reach for first:\n\n**1. The result assertion (proves the JOIN didn't change behavior).** Assert the exact rows, values, and ordering the endpoint is supposed to return — same shape the old per-row code returned. This is what catches a JOIN that accidentally drops orders with no related row (a `JOIN` that should have been a `LEFT JOIN`) or duplicates a row when a parent ends up with multiple children. Ordering matters if the query has an `ORDER BY`; assert the actual sequence, not just set membership.\n\n**2. The query-count assertion (this is the part that actually guards against regression).** A correct result set today does not prove the query stays a single JOIN after someone edits the query later — an accidental lazy-load or a re-introduced per-row lookup returns the identical rows while quietly issuing N+1 queries. So count the queries the endpoint issues and assert it's the small fixed number the JOIN version produces (typically 1, plus whatever the endpoint legitimately needs). Most ORMs and test harnesses expose a query counter or a log you can drain and count — reuse the project's existing query-count helper if one exists, otherwise add one.\n\nConcretely, for a listing endpoint:\n\n- Seed a row count large enough that N+1 is obvious — say 25 orders, not 2. With 2 orders a per-row regression issues ~3 queries and might slip past a loose bound; at 25 the difference between 1 query and 25 is unmistakable.\n- Make sure each seeded order has associated rows to load (line items, customer, etc.), because that's what the old N+1 actually iterated over. Seeding orders with no children won't reproduce the pattern.\n- Assert `query_count == 1` (or whatever the fixed number should be) rather than \"less than N\" — a loose upper bound tolerates a regression from 1 to 3.\n- Keep the result assertion immediately next to it so a future reader can't satisfy one and drop the other.\n\nRun the test once against the pre-fix code to confirm it actually fails (N queries) and against the post-fix code to confirm it passes (1 query) — a regression test that never demonstrated the failure is just decoration. Then wrap the whole thing in the project's existing transactional-rollback isolation so the seeded rows don't leak into other tests.\n\nOne caveat: watch that the query-count assertion isn't itself counting fixture setup. Seed via the project's factory/helper *before* you start the counter (or reset the counter after seeding), so the count reflects only the endpoint's queries. Otherwise your count is off by however many inserts the setup did, and you'll be tempted to loosen the bound — which is exactly how the guard gets neutered.",
892
+ "outputSha256": "ea401145b354764b75930525d2eca46a7a363e042bee65a00d1f2b8bf91dc1bf",
893
+ "promptSha256": "7339416a90c7c6ef868cafd86337e14a2b983162bc99873669c35942e36f8120",
894
+ "deterministic": [],
895
+ "judge": {
896
+ "verdict": "pass",
897
+ "reason": "Answer asserts exact result rows/values/ordering (\"Assert the exact rows, values, and ordering\") and separately asserts query count (\"Assert `query_count == 1`\", \"count the queries the endpoint issues\"), explicitly noting result-only is insufficient. Both pass criteria concretely met."
898
+ },
899
+ "passed": true
900
+ }
901
+ ]
902
+ }
903
+ ],
904
+ "verdict": "fail",
905
+ "scope": "bundled",
906
+ "skillDigest": "9945e5f29e699d7dce5cb955cc353757af6c1caa797e2710f7f195c0949fa4c3",
907
+ "catalogDigest": "14504a0807a0089488b9cb690c4b13f20865cd7a7fb69a1e5d8dfea8bfd5fbd1",
908
+ "judgePromptVersion": "2026-09-25.1",
909
+ "runner": "deepseek",
910
+ "model": "deepseek-chat",
911
+ "runnerPromptVersion": "2026-09-25.1",
912
+ "recordedAt": "2026-09-25T18:15:59.898Z",
913
+ "judge": "deepseek",
914
+ "judgeModel": "deepseek-chat"
915
+ },
916
+ {
917
+ "schemaVersion": "1.0.0",
918
+ "skillId": "sql-db/sql-db-code-review",
919
+ "strictness": "high",
920
+ "trials": 10,
921
+ "triggerAccuracy": {
922
+ "truePositive": 0,
923
+ "falsePositive": 3,
924
+ "positives": 7,
925
+ "negatives": 7
926
+ },
927
+ "evidence": "authored",
928
+ "scenarios": [
929
+ {
930
+ "id": "trigger-positive-1",
931
+ "kind": "trigger-positive",
932
+ "prompt": "Before this goes out, look over the diff that adds a partial unique index on orders.tracking_number and flag anything risky",
933
+ "strictness": "high",
934
+ "trials": 1,
935
+ "passes": 0,
936
+ "passRate": 0,
937
+ "passAtK": 0,
938
+ "grader": "trigger-rank-fork-family",
939
+ "status": "ran",
940
+ "deterministic": true
941
+ },
942
+ {
943
+ "id": "trigger-positive-2",
944
+ "kind": "trigger-positive",
945
+ "prompt": "This PR builds the invoice search filter by interpolating customer_id from the request straight into the SQL string -- is that exploitable?",
946
+ "strictness": "high",
947
+ "trials": 1,
948
+ "passes": 0,
949
+ "passRate": 0,
950
+ "passAtK": 0,
951
+ "grader": "trigger-rank-fork-family",
952
+ "status": "ran",
953
+ "deterministic": true
954
+ },
955
+ {
956
+ "id": "trigger-positive-3",
957
+ "kind": "trigger-positive",
958
+ "prompt": "We're about to run ALTER TABLE payments ADD COLUMN refunded_at against a 40-million-row table -- will that block writers?",
959
+ "strictness": "high",
960
+ "trials": 1,
961
+ "passes": 0,
962
+ "passRate": 0,
963
+ "passAtK": 0,
964
+ "grader": "trigger-rank-fork-family",
965
+ "status": "ran",
966
+ "deterministic": true
967
+ },
968
+ {
969
+ "id": "trigger-positive-4",
970
+ "kind": "trigger-positive",
971
+ "prompt": "This PR adds a query filtering invoices by customer_email but I don't see a matching index anywhere in the diff -- can you check?",
972
+ "strictness": "high",
973
+ "trials": 1,
974
+ "passes": 0,
975
+ "passRate": 0,
976
+ "passAtK": 0,
977
+ "grader": "trigger-rank-fork-family",
978
+ "status": "ran",
979
+ "deterministic": true
980
+ },
981
+ {
982
+ "id": "trigger-positive-5",
983
+ "kind": "trigger-positive",
984
+ "prompt": "The /api/invoices list handler in this PR loops and fetches the customer record per invoice -- does that pattern show up here?",
985
+ "strictness": "high",
986
+ "trials": 1,
987
+ "passes": 0,
988
+ "passRate": 0,
989
+ "passAtK": 0,
990
+ "grader": "trigger-rank-fork-family",
991
+ "status": "ran",
992
+ "deterministic": true
993
+ },
994
+ {
995
+ "id": "trigger-positive-6",
996
+ "kind": "trigger-positive",
997
+ "prompt": "This diff wraps a Stripe charge call inside the same DB transaction as the order insert -- is that going to hold locks too long?",
998
+ "strictness": "high",
999
+ "trials": 1,
1000
+ "passes": 0,
1001
+ "passRate": 0,
1002
+ "passAtK": 0,
1003
+ "grader": "trigger-rank-fork-family",
1004
+ "status": "ran",
1005
+ "deterministic": true
1006
+ },
1007
+ {
1008
+ "id": "trigger-positive-7",
1009
+ "kind": "trigger-positive",
1010
+ "prompt": "Look over this composite index and tell me if the column order is right for the query",
1011
+ "strictness": "high",
1012
+ "trials": 1,
1013
+ "passes": 0,
1014
+ "passRate": 0,
1015
+ "passAtK": 0,
1016
+ "grader": "trigger-rank-fork-family",
1017
+ "status": "ran",
1018
+ "deterministic": true
1019
+ },
1020
+ {
1021
+ "id": "trigger-negative-1",
1022
+ "kind": "trigger-negative",
1023
+ "prompt": "Review this Rails ActiveRecord migration diff for safety before merging",
1024
+ "strictness": "high",
1025
+ "trials": 1,
1026
+ "passes": 0,
1027
+ "passRate": 0,
1028
+ "passAtK": 0,
1029
+ "grader": "trigger-rank-fork-family",
1030
+ "status": "ran",
1031
+ "deterministic": true
1032
+ },
1033
+ {
1034
+ "id": "trigger-negative-2",
1035
+ "kind": "trigger-negative",
1036
+ "prompt": "Review this Django migration for missing indexes",
1037
+ "strictness": "high",
1038
+ "trials": 1,
1039
+ "passes": 0,
1040
+ "passRate": 0,
1041
+ "passAtK": 0,
1042
+ "grader": "trigger-rank-fork-family",
1043
+ "status": "ran",
1044
+ "deterministic": true
1045
+ },
1046
+ {
1047
+ "id": "trigger-negative-3",
1048
+ "kind": "trigger-negative",
1049
+ "prompt": "Review this Go function for SQL injection risk",
1050
+ "strictness": "high",
1051
+ "trials": 1,
1052
+ "passes": 0,
1053
+ "passRate": 0,
1054
+ "passAtK": 0,
1055
+ "grader": "trigger-rank-fork-family",
1056
+ "status": "ran",
1057
+ "deterministic": true
1058
+ },
1059
+ {
1060
+ "id": "trigger-negative-4",
1061
+ "kind": "trigger-negative",
1062
+ "prompt": "Review this diff and also fix the bugs you find",
1063
+ "strictness": "high",
1064
+ "trials": 1,
1065
+ "passes": 1,
1066
+ "passRate": 1,
1067
+ "passAtK": 1,
1068
+ "grader": "trigger-rank-fork-family",
1069
+ "status": "ran",
1070
+ "deterministic": true
1071
+ },
1072
+ {
1073
+ "id": "trigger-negative-5",
1074
+ "kind": "trigger-negative",
1075
+ "prompt": "Review this React component for accessibility issues",
1076
+ "strictness": "high",
1077
+ "trials": 1,
1078
+ "passes": 1,
1079
+ "passRate": 1,
1080
+ "passAtK": 1,
1081
+ "grader": "trigger-rank-fork-family",
1082
+ "status": "ran",
1083
+ "deterministic": true
1084
+ },
1085
+ {
1086
+ "id": "trigger-negative-6",
1087
+ "kind": "trigger-negative",
1088
+ "prompt": "Run a general security review on this whole codebase",
1089
+ "strictness": "high",
1090
+ "trials": 1,
1091
+ "passes": 1,
1092
+ "passRate": 1,
1093
+ "passAtK": 1,
1094
+ "grader": "trigger-rank-fork-family",
1095
+ "status": "ran",
1096
+ "deterministic": true
1097
+ },
1098
+ {
1099
+ "id": "trigger-negative-7",
1100
+ "kind": "trigger-negative",
1101
+ "prompt": "Review this Python function for a mutable default argument bug",
1102
+ "strictness": "high",
1103
+ "trials": 1,
1104
+ "passes": 1,
1105
+ "passRate": 1,
1106
+ "passAtK": 1,
1107
+ "grader": "trigger-rank-fork-family",
1108
+ "status": "ran",
1109
+ "deterministic": true
1110
+ },
1111
+ {
1112
+ "id": "read-only-injection-finding",
1113
+ "kind": "behavior",
1114
+ "prompt": "Review this diff: a query builds its WHERE clause by concatenating a status value from the request directly into the SQL string. What do you find?",
1115
+ "strictness": "high",
1116
+ "trials": 10,
1117
+ "passes": 10,
1118
+ "passRate": 1,
1119
+ "passAtK": 1,
1120
+ "grader": "judge",
1121
+ "status": "ran",
1122
+ "trialRecords": [
1123
+ {
1124
+ "output": "**Finding — SQL injection via string-concatenated request value**\n\n- **Pattern:** the query's `WHERE` clause is built by concatenating a `status` value that originates from the request directly into the SQL string (e.g. `\"... WHERE status = '\" + status + \"'\"`), rather than binding it as a parameter.\n- **Risk category (Step 2, Injection):** an external input (a request parameter) reaches SQL text by concatenation/interpolation rather than a placeholder. This is a classic injection sink: a caller controls a fragment of the SQL statement itself, not merely a bound value. A crafted `status` can terminate the intended predicate and append arbitrary SQL (a tautology to bypass filtering, a `UNION SELECT` to exfiltrate other columns/tables, a stacked statement where the driver permits it, or destructive DML).\n- **Why it matters:** this is not a correctness edge case or a performance nit — it lets an untrusted caller reshape the query. The immediate caller being \"internal\" does not exempt it; an internal caller can become externally reachable later without this query being revisited, so the string-built form must be flagged regardless.\n- **Fix direction (reported, not applied):** replace the concatenation with a bound parameter placeholder (e.g. `WHERE status = $1` / `?`) and pass `status` as a parameter value, so the database treats it strictly as data and never as SQL syntax. If `status` is ever used as an *identifier* (a column or table name chosen at runtime) rather than a value, a placeholder cannot bind an identifier — that case needs an explicit allowlist validation instead, not a parameter.\n- **Scope note:** this skill is read-only — the finding and fix direction above are reported, not edited into the file.\n\nOne caveat on the review process: a query can *also* be injection-safe only if every interpolated fragment is a bound value; if the diff interpolates more than just `status` (e.g. the column it filters on, or an `ORDER BY`/`LIMIT` clause), each such fragment is a separate finding and must be checked individually — a value placeholder does not protect an interpolated identifier.",
1125
+ "outputSha256": "0eccd6e18b67ee0a39f83dd3b7686f6e0c5fd00b3a4718d2dc137d0631cbf780",
1126
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1127
+ "deterministic": [],
1128
+ "judge": {
1129
+ "verdict": "pass",
1130
+ "reason": "Answer names the concatenated request `status` value in the WHERE clause as SQL injection (quotes `\"... WHERE status = '\" + status + \"'\"`), gives concrete fix direction (bind via `$1`/`?` placeholder, pass status as parameter value), and explicitly states it is reported not applied ('this skill is read-only — the finding and fix direction above are reported, not edited'). No code change claimed."
1131
+ },
1132
+ "passed": true
1133
+ },
1134
+ {
1135
+ "output": "Finding — SQL injection (Step 2, \"Injection\"), and this is the top-severity pattern in the list.\n\n**What it is:** the query text is assembled by concatenating a `status` value that originates from the request directly into the SQL string, e.g. something like `\"... WHERE status = '\" + status + \"'\"` (or an f-string / `%` / `.format` / template-literal equivalent). The request value lands in the query as SQL syntax, not as data.\n\n**Why it matters:** the `status` parameter is attacker-controlled. Because it is spliced in as text rather than bound as a value, the attacker controls a fragment of the SQL — not just that one field. With a value like `' OR '1'='1` the predicate is neutralized, and with `'; DROP TABLE ...; --` or a `UNION SELECT` the attacker can read or destroy arbitrary data the connection can reach. This is a full injection, not a quoting edge case, and it is reachable from a normal request, so it should block the merge.\n\n**Fix direction (do not apply — this skill is read-only):**\n- Bind the value as a parameter placeholder in the prepared statement — `WHERE status = ?` (or `$1` in Postgres, `%s` with a real parameterized driver in MySQL). The placeholder binds the *value*, so no request content can ever become SQL syntax. This is the correct fix and removes the injection entirely.\n- If (and only if) for some reason the field is being used as an *identifier* rather than a value — i.e. the request is choosing a column or table name — a placeholder cannot bind an identifier, so that case instead requires an explicit allowlist check against known-good column names. Do not treat this as a substitute for parameterizing the `status` value; it is a separate, narrower situation. From your description this is a value comparison, so parameterization is the right fix.\n\n**Caveat on scope:** I'm reviewing from your description of the diff, not the file on disk, so I can't give an exact file:line. Pin the finding to the specific concatenation site when reporting it. Also confirm whether this string-building exists only in the changed lines or is pre-existing in code the diff merely touches — if the latter, note it separately from anything the diff newly introduces.\n\nTwo related things worth checking in the same diff while you're here, since they often travel with this pattern:\n- Any other request-sourced value concatenated into the same query (a second field, an `ORDER BY` clause, a `LIMIT`) — same finding, same fix.\n- Whether this is happening inside an ORM's raw-SQL escape hatch rather than a hand-built connection — the fix is identical there; the escape hatch is exactly where this pattern hides.\n\nWhat I am **not** flagging: nothing about missing indexes, blocking DDL, N+1 shape, or transaction scope can be assessed from this description alone. If you want those checked, share the surrounding schema and the query's call site.",
1136
+ "outputSha256": "142acfcd6c8a3842c3c6675f8cbf363ac3a52688406683e3681348c0d544b133",
1137
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1138
+ "deterministic": [],
1139
+ "judge": {
1140
+ "verdict": "pass",
1141
+ "reason": "Answer names the concatenated request `status` value in the WHERE clause as SQL injection (\"WHERE status = '\" + status + \"'\"), gives the concrete fix of binding via placeholder (`?`, `$1`, `%s`), and explicitly says \"do not apply — this skill is read-only\" with no edit made or claimed."
1142
+ },
1143
+ "passed": true
1144
+ },
1145
+ {
1146
+ "output": "This is an injection finding, not a performance or style one.\n\n**Finding: SQL injection via concatenated `status` value**\n\n- **Pattern:** The query's `WHERE` clause is assembled by string concatenation/interpolation of a `status` value coming straight from the request (a parameter, form field, or message payload) — the classic injection shape from the focus list. The value crosses the boundary between untrusted external input and SQL text without passing through a bound placeholder.\n- **Why it matters:** The caller controls the text that lands in the `WHERE` clause, so they control the query's logic. Depending on the surrounding query form (e.g., `... WHERE status = '<value>'` or `... WHERE status IN (<value>)`), this ranges from bypassing a row-visibility filter to being able to terminate the string and append arbitrary SQL — reading, altering, or dropping data the query was never meant to reach. The fact that the parameter is named `status` and looks \"enum-like\" does not make it safe; a status value from the request is still external input.\n- **Fix direction (do not apply):** Bind the value as a parameter placeholder (`:status` / `?` / `$1`) using the driver/ORM's parameter binding, rather than concatenating it into the statement text. If the value is meant to select among a fixed set of statuses, additionally validate it against that allowlist server-side before use, and note that a set of statuses for an `IN (...)` list still needs one placeholder per element generated from a length the code controls — not from the raw string.\n\n**One thing to check before closing it out:** if the concatenated fragment is a dynamic *identifier* (e.g., a column or table name chosen from the request) rather than a value, a bound placeholder cannot fix it — identifiers cannot be parameterized, so that case needs an explicit allowlist check mapping the input to a known-safe identifier, not a placeholder. The report should say which of the two it is, since the fix direction differs.\n\n**Reporting:** cite the specific file:line where the string is built, name this as injection (diff-introduced, per the workflow's Step 2), and give the binding/validation fix direction above. Do not edit the file — this review is read-only.",
1147
+ "outputSha256": "ac36c67bcd68f00a4800be0d4c1e9a87fd1c3b00ecdc15ec5ae0e96d955e4e45",
1148
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1149
+ "deterministic": [],
1150
+ "judge": {
1151
+ "verdict": "pass",
1152
+ "reason": "Answer names the concatenated request `status` value in the WHERE clause as SQL injection (\"Pattern: ... assembled by string concatenation/interpolation of a `status` value coming straight from the request\"). Fix direction is concrete: \"Bind the value as a parameter placeholder (`:status` / `?` / `$1`) using the driver/ORM's parameter binding, rather than concatenating it into the statement text.\" It stays read-only: \"Fix direction (do not apply)\" and \"Do not edit the file — this review is read-only.\" No edit claimed."
1153
+ },
1154
+ "passed": true
1155
+ },
1156
+ {
1157
+ "output": "**Finding — SQL injection via string-concatenated request value**\n\nWhere: the query-building site in the diff (the function that assembles the WHERE clause string). The finding is on the line that concatenates the request's `status` value into the SQL text — cite it as `<file>:<line>`.\n\nPattern: SQL text is built by concatenating a value that traces directly back to external input (a request parameter). This is the injection category from the focus list — the classic \"value built into the SQL string instead of bound\" case. It is diff-introduced, not pre-existing, per Step 1.2.\n\nWhy it matters: the request value is not a bound parameter, so it is parsed as SQL, not as data. Even if the current caller only passes what it believes is a fixed set of status strings, the value is attacker-controlled at the request boundary; a crafted `status` can close the string/quote and append arbitrary SQL (or a tautology that widens the result set), reaching whatever the connection's role is permitted to do — reads of other rows/tables, and writes if the role allows them. Flag it regardless of who the immediate caller is: \"comes from an internal service\" does not neutralize it, since an internal caller can become externally reachable later without this query being revisited.\n\nFix direction:\n- Replace the concatenation with a bind placeholder — `WHERE status = $1` (Postgres) / `?` (MySQL) — and pass the status value through the parameter list, not into the string. The driver then treats it as data under all quoting.\n- If the value is genuinely off an enumerated allowlist, both defensively validate against that allowlist *and* still bind it. The allowlist check is the belt; the placeholder is the suspenders — do not use the allowlist as a substitute for binding.\n- If any part of the clause being concatenated is instead a dynamic *identifier* (a table or column name chosen at runtime), note that a placeholder cannot bind an identifier — that path needs an explicit allowlist of permitted identifiers, not a bind parameter. I can't tell from the one-line description whether the status value is used only as a comparison value or also to pick a column; call that out as the thing to confirm in the actual diff.\n\nNot reported here: the \"which query, exact line\" specifics, since this exercise describes the pattern rather than a concrete file; in a real review each finding names file:line, the risk category, and the fix direction, per the report format. No source was modified — this is findings only.\n\nOne related check worth pair-flagging while you are on this site, though it is a separate finding from the injection one: confirm whether this same query is issued inside a loop over rows from an earlier query (the N+1 shape). An N+1 built the same string-concatenation way compounds the exposure — the vulnerable builder runs once per row. If so, report it as its own finding with the batching/JOIN fix direction, not folded into the injection item.\n\nTo confirm the injection risk is real and not theoretical in your codebase, run the query with `EXPLAIN` against a literal `' OR 1=1 --` style value and observe whether it parses as a predicate rather than a string — but the fix is the same either way, so this is confirmation, not a gate on reporting.",
1158
+ "outputSha256": "7bb365021290dd6c4d53bd08a33d75a64fda94b14c86c15f73f1a501298f4dc7",
1159
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1160
+ "deterministic": [],
1161
+ "judge": {
1162
+ "verdict": "pass",
1163
+ "reason": "Answer names the concatenated request `status` value in the WHERE clause as SQL injection (\"concatenates the request's `status` value into the SQL text\"), gives the concrete fix (\"Replace the concatenation with a bind placeholder — `WHERE status = $1` (Postgres) / `?` (MySQL) — and pass the status value through the parameter list\"), and stays read-only (\"No source was modified — this is findings only\"). No edit claimed."
1164
+ },
1165
+ "passed": true
1166
+ },
1167
+ {
1168
+ "output": "This is a SQL injection finding, and it is the highest-severity issue in this diff.\n\n**Finding — migrations/queries: the dynamic `WHERE` clause**\n\n- **File:line** — the statement in the diff that builds `\"... WHERE status = '\" + status + \"'\"` (or its equivalent string concatenation / template interpolation of `status`).\n- **Pattern** — a value that traces back to external input (the request) is concatenated directly into SQL text rather than bound as a parameter.\n- **Why it matters** — the request-controlled `status` value is interpreted as SQL rather than as data. A caller can supply a payload (via a quote, comment, or `OR`/`UNION` construction) to alter or extend the statement, bypass the intended filter, and potentially read or modify rows outside the intended scope. Because the value arrives at the request boundary, this is directly reachable — not the \"internal caller today\" case the red-flag table excuses.\n- **Fix direction** — replace the concatenation with a bound parameter placeholder (`WHERE status = ?` / `$1` / `:status`) passed to the driver's parameter list, so the value can never be parsed as SQL. Do not apply this — report only.\n\n**Secondary finding — dynamic identifier, if present**\n\nIf the diff also picks a table or column name at runtime and concatenates it into the statement, that is a separate finding: a value placeholder cannot bind an identifier, so this needs an explicit allowlist check against the set of valid identifiers, not a parameter. Flag it as its own finding rather than folding it into the injection one above.\n\n**Rejection (do not report)**\n\n- If the diff merely *touches* a pre-existing concatenated `WHERE` (it was already there before this change), still report it, but mark it clearly as pre-existing in a touched file rather than diff-introduced, per the Rules.\n- Formatting/naming nits on the query are covered by `rules/coding-style.mdc` and are noise here — drop them.\n\n**Fix direction, not a fix**\n\nThis review is read-only. The remediation is: switch to a parameterized query at the call site; if the driver has an ORM raw-SQL escape hatch in play, bind parameters through it rather than interpolating the string. No files were modified.\n\n**Confidence note** — the injection finding does not depend on knowing the engine; parameter binding behaves the same way across Postgres, MySQL, and generic drivers, so I am asserting it without qualification. I am not asserting anything about index usage or a full scan here, since nothing in this diff is sufficient to judge the query plan — that would require `EXPLAIN ANALYZE`.",
1169
+ "outputSha256": "68148895942f022c87e726843de87932a76f81ebb590aeb2954ea50c7268413f",
1170
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1171
+ "deterministic": [],
1172
+ "judge": {
1173
+ "verdict": "pass",
1174
+ "reason": "Answer names the concatenated request value in the WHERE clause as SQL injection (\"WHERE status = '\" + status + \"'\"), gives concrete fix direction (bound placeholder ?/$1/:status passed to driver's parameter list), and explicitly stays read-only (\"Do not apply this — report only\", \"No files were modified\"). No edit or claimed change."
1175
+ },
1176
+ "passed": true
1177
+ },
1178
+ {
1179
+ "output": "Finding: SQL injection via string-concatenated request value.\n\nfile:line — the statement that builds the WHERE clause (e.g. `... \"WHERE status = '\" + request.status + \"'\" ...`). Pattern: a value that traces back to external input (a request parameter) is concatenated directly into the SQL text rather than passed as a bound placeholder. Risk category: injection. The `status` value is fully attacker-controlled at the SQL-parsing layer, so a payload like `' OR '1'='1`, a stacked `'; DROP TABLE ...; --`, or a UNION-based read can change the query's meaning or execute additional statements depending on the driver's multi-statement setting.\n\nFix direction: replace the concatenation with a bound parameter placeholder (`WHERE status = ?` / `$1`) and pass the value through the driver's parameter binding, not through the string. Do not attempt to sanitize by escaping quotes — binding is the fix, escaping is not. If the value is being used as an identifier (a column or table name chosen at runtime) rather than a value, a placeholder cannot bind it; that case needs an explicit allowlist check against the set of valid identifiers, not quoting.\n\nThat is the primary finding and it is a review blocker on its own.\n\nAdjacent items to confirm while you are in this file:\n\n- Query shape: if this string is built and issued inside a loop over rows from an earlier query, that is the N+1 shape — flag it separately with the batching/JOIN fix direction. Your diff description points at the concatenation, so treat N+1 as something to check, not a confirmed finding.\n- Transaction scope: if this query runs inside a transaction that is held open across a slow external call (an HTTP request, a queue publish), flag the lock-duration risk on top of the injection finding.\n- Indexing: if the `status` column this filters on has no supporting index and the table is large enough to matter, flag a missing index on the new WHERE column. Note this as \"run EXPLAIN ANALYZE to confirm\" rather than asserting a full scan without evidence.\n\nDo not edit the file — this is a read-only review; the finding and its fix direction above are the deliverable. No source file was modified. If the concatenation pattern exists in a broader helper that other call sites also use, the same finding applies to each call site that passes request-derived input into it, so check the other callers before calling the review done.",
1180
+ "outputSha256": "e98050d7c2146dd8b78af6ff9d5b7789a65919d5b55b0f21c0a7f94c6347162a",
1181
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1182
+ "deterministic": [],
1183
+ "judge": {
1184
+ "verdict": "pass",
1185
+ "reason": "Answer names the concatenated request value in the WHERE clause as SQL injection (\"WHERE status = '\" + request.status + \"'\"), gives the concrete fix (bound placeholder `WHERE status = ?` / `$1` via driver binding, not escaping), and stays read-only: \"Do not edit the file — this is a read-only review... No source file was modified.\" No edit claimed."
1186
+ },
1187
+ "passed": true
1188
+ },
1189
+ {
1190
+ "output": "**Finding: SQL injection via string-concatenated request value in the WHERE clause.**\n\n- **Pattern:** the WHERE clause is assembled by concatenating a status value taken directly from the request into the SQL text, rather than binding it as a parameter.\n- **Why it matters:** that status value traces straight back to external input. An attacker controls the literal characters that land in the SQL text, so they can break out of the intended value context (e.g. close a quote and append `OR 1=1`, `UNION SELECT ...`, stacked statements, or comment out the rest of the clause). This is textbook injection, not a hypothetical.\n- **Fix direction:** pass the status as a bound parameter / placeholder and build the query text as a fixed string (e.g. `... WHERE status = $1` with the value supplied separately). If the concatenated item is ever an identifier (a column or table name chosen at runtime rather than a value), a placeholder cannot bind an identifier — that case needs an explicit allowlist check plus quoting, not a placeholder.\n\n**Secondary consideration:** even a plain bound value can be misused if the surrounding clause is built from a dynamic identifier. So confirm which part is concatenated:\n\n- value concatenated → route through a placeholder;\n- identifier concatenated → route through an allowlist of permitted names, then quote.\n\nIf a placeholder-based fix changes the query's shape (e.g. the concatenation was also selecting which column to filter on), that is a separate identifier-validation finding.\n\n**Severity:** high — this is the injection category and should be treated as a merge blocker until the value is bound.\n\n**Scope note:** this review is read-only; I have not modified the file. The finding stands regardless of who calls the code today — an \"internal\" caller can become externally reachable later without this query being revisited, so flag it on the string-building pattern itself, not on the current caller's identity. If this same pattern exists in unchanged code the diff only touches, note it separately as pre-existing rather than introduced by this diff.\n\nOne thing I can't assert from the description alone: whether an index on the status column exists for the non-injected form of the query. If the WHERE clause filters on a status column with no supporting index and the table is large enough to matter, that's a separate indexing finding to raise — but I'd want to see the actual `WHERE`/`JOIN`/`ORDER BY` shape and the table's index definition before calling it a full scan rather than asking for `EXPLAIN ANALYZE` to confirm.\n\nBottom line: the concatenated request value in the WHERE clause is a concrete SQL injection, reported at the concatenation site, with the fix direction being a bound parameter (or an allowlist if an identifier is involved) — nothing edited.",
1191
+ "outputSha256": "881afe2b618a6156b961a05040e516dafc0e62d28a126e43ada7658320f1f93a",
1192
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1193
+ "deterministic": [],
1194
+ "judge": {
1195
+ "verdict": "pass",
1196
+ "reason": "Answer names the WHERE-clause concatenation of a request status value as SQL injection (\"Pattern: the WHERE clause is assembled by concatenating a status value taken directly from the request\"). Fix direction is concrete: \"pass the status as a bound parameter / placeholder ... WHERE status = $1 with the value supplied separately.\" It stays read-only: \"this review is read-only; I have not modified the file ... nothing edited.\" No edit claimed."
1197
+ },
1198
+ "passed": true
1199
+ },
1200
+ {
1201
+ "output": "**Finding — injection — request value concatenated into SQL string**\n\nFile:line (the changed query/query-builder statement, e.g. `src/reports/query.sql:1` or the application file issuing it) — the WHERE clause is assembled by string concatenation, splicing a `status` value taken from the request directly into the SQL text. Risk: SQL injection. Any caller-controlled value that reaches string-built SQL can terminate the intended predicate and append arbitrary SQL (`x' OR '1'='1`, stacked statements where the driver permits, or UNION-based extraction), so this is a data-exposure and data-destruction risk, not a performance nit. Fix direction: bind the value as a parameter (a `?`/`$1` placeholder in the concatenated text, or the query API's parameter argument) rather than splicing it. The literal SQL structure stays; only the value becomes a bound placeholder. Do not apply this — report only.\n\nTwo notes on scoping:\n\n- If the concatenated token is used as a **value** in a `WHERE status = ...` predicate, a placeholder fully fixes it (Step 2, injection: \"value placeholder\").\n- If any part of what is being concatenated is an **identifier** (a column or table name chosen at runtime from the same request), a placeholder cannot bind an identifier — that path needs an explicit allowlist check against known-valid identifier names instead. The diff shows a status value, but confirm whether any adjacent concatenation in the same statement is assembling an identifier; flag it separately if so.\n\n- Flag this regardless of who the immediate caller is. The \"it comes from an internal service, not a user\" rationalization does not hold: an internal caller today can become externally reachable later without this query being revisited.\n\n- If this string-built form was pre-existing and the diff only touches a nearby line in the same statement, say so — but the injection pattern still warrants flagging, and I am calling it out as a finding the diff's statement carries, not a new one this diff necessarily introduced.\n\nNo index, blocking-DDL, N+1, or transaction-scope issues are raised by the fragment described; this review is limited to the injection finding above — I have not seen the rest of the changed files, and per Step 1 the review is not complete until every changed `.sql` file and every changed application-code query/raw-SQL call site has been read. If you can share the full diff, I will check the remaining categories (blocking `NOT NULL`/index-creation, missing or misordered indexes on the new `WHERE` column, N+1 shape, transactions held across a slow call) against it.",
1202
+ "outputSha256": "6baceec7834fd6ea9abd6d61d52192734f436403625d3a9f0ddc57348eaaf4e1",
1203
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1204
+ "deterministic": [],
1205
+ "judge": {
1206
+ "verdict": "pass",
1207
+ "reason": "Answer names the WHERE clause built by string concatenation splicing a request `status` value into SQL text as SQL injection risk (\"splicing a `status` value taken from the request directly into the SQL text. Risk: SQL injection\"). Fix direction concretely given: bind as parameter with `?`/`$1` placeholder instead of splicing. Explicitly read-only: \"Do not apply this — report only.\" No edit claimed."
1208
+ },
1209
+ "passed": true
1210
+ },
1211
+ {
1212
+ "output": "Finding — SQL injection via string-built WHERE clause.\n\nThe diff concatenates a request-supplied status value directly into the SQL text. The value traces back to external input (a request parameter), and because it is concatenated into an identifier-or-value position of the query string rather than bound through a placeholder, this is the injection pattern flagged first under Injection in Step 2: any SQL text built by concatenating or interpolating a value that ultimately traces back to external input.\n\nWhy it matters: the attacker controls everything after the concatenation point. A status value like `' OR '1'='1`, or `' UNION SELECT ...`, or a stacked `'; DROP TABLE ...` depending on the driver's multi-statement setting, changes the query's meaning entirely. This is not limited to read exposure — a write-path query builder with the same shape gives an attacker write access. The immediate caller being \"internal\" or \"trusted today\" does not downgrade this; an internal caller can become externally reachable later without this query being revisited.\n\nFix direction (report only, do not apply): replace the concatenation with a bound parameter — the status becomes a placeholder (`WHERE status = $1` / `?`) and the value is passed through the parameter list, so it can never be parsed as SQL. If the concatenated piece is actually a dynamic identifier (a column or table name chosen at runtime) rather than a value, a placeholder cannot bind an identifier, so that case needs an explicit allowlist check against the set of permitted names instead. Confirm the surrounding code path routes the value through the binding layer rather than a raw-SQL escape hatch that re-introduces concatenation.\n\nNo other categories from Step 2 are determinable from the single statement described here — the engine, table size, surrounding schema, and transaction scope are not stated, so I am not asserting a missing-index, blocking-ALTER, N+1, or transaction-scope finding without evidence in the diff. If the diff also shows the surrounding schema, index definitions, or a loop issuing this query per row, those become reviewable; as described, the injection pattern is the finding.\n\nFor completeness: if this query is issued inside a loop over rows from an earlier query, the same call site is also an N+1 shape and should be flagged separately with a batching/JOIN fix direction — but that depends on the calling code, which this description does not include.\n\nI did not modify any file; this is a read-only review reporting the finding and fix direction only.",
1213
+ "outputSha256": "6e5a7ab92357759736e80bca2af19c1e529e40fdca6d768fa14225a735363311",
1214
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1215
+ "deterministic": [],
1216
+ "judge": {
1217
+ "verdict": "pass",
1218
+ "reason": "Criterion 1: names the concatenated request-supplied status value in the WHERE clause as SQL injection ('concatenates a request-supplied status value directly into the SQL text'). Criterion 2: concrete fix shown — 'the status becomes a placeholder (`WHERE status = $1` / `?`) and the value is passed through the parameter list'. Criterion 3: 'Fix direction (report only, do not apply)' and 'I did not modify any file'. No edit claimed, so fail criterion 1 does not hold."
1219
+ },
1220
+ "passed": true
1221
+ },
1222
+ {
1223
+ "output": "This is a SQL injection finding, and it is the highest-severity issue this skill checks for.\n\n**Finding**\n- Category: Injection (Step 2, first bullet — \"SQL text built by concatenating or interpolating a value that ultimately traces back to external input\").\n- The value is a request-supplied `status` interpolated directly into the SQL string that becomes the `WHERE` clause.\n\n**Why it matters**\nConcatenating request-controlled input into SQL text means the caller controls the shape of the query, not just a value. A crafted `status` can close the intended string/expression and append arbitrary SQL (or boolean logic that trivially bypasses the intended filter), turning a read into a read of the whole table or worse depending on what the surrounding statement does. This is not mitigated by the value currently being used only as a filter — the pattern is the defect, not the current value.\n\n**Fix direction (do not apply — report only)**\n- Bind the value as a parameter/placeholder (`WHERE status = $1` / `?`), and pass it through the driver's binding so it is never parsed as SQL.\n- If `status` instead selects a column, table, or an `ORDER BY` clause at runtime (a dynamic identifier), no placeholder can bind that — it needs an explicit allowlist check against the set of permitted identifiers. A value placeholder does not fix identifier injection.\n- If the string is being assembled inside an ORM's raw-SQL escape hatch, the same rule applies there; the ORM's normal query API does not exempt this call site.\n\n**Note on scoping**\nIf this query is new or changed in the diff, it is a diff-introduced finding. If the diff only touches nearby code and this concatenation was already present, report it as pre-existing but still flag it — an internal caller today can become externally reachable later without this query being revisited, so the pattern is worth surfacing either way.\n\n**What I would not do**\nNot flag it as a minor style nit, not assume the caller is trusted, and not edit the file — findings and fix direction only.\n\nTo be precise about file:line and whether this is diff-introduced versus pre-existing, I would need the actual diff content; the description above confirms the pattern but not the location. If you can share the diff, I can pin each finding to a concrete line and confirm whether any companion checks — missing index on the filtered column, N+1 shape around the caller, or transaction scope — also apply.",
1224
+ "outputSha256": "979569edff2999bb8307cbbb295d7f95295ce5b560ee78957b4d7cd6060a7a41",
1225
+ "promptSha256": "9f63404d2290818d3fa139eed21b12dd78a8f06abf073189d7cdfeaf7af5a764",
1226
+ "deterministic": [],
1227
+ "judge": {
1228
+ "verdict": "pass",
1229
+ "reason": "Answer names the request-supplied `status` concatenated into the SQL WHERE clause as a SQL injection risk (Finding section). Fix direction concretely given: bind as parameter/placeholder `WHERE status = $1` / `?` through driver binding. Explicitly read-only: 'do not apply — report only' and 'not edit the file — findings and fix direction only.' No code change claimed."
1230
+ },
1231
+ "passed": true
1232
+ }
1233
+ ]
1234
+ },
1235
+ {
1236
+ "id": "blocking-not-null-finding",
1237
+ "kind": "behavior",
1238
+ "prompt": "Review this diff: a migration runs ALTER TABLE orders ALTER COLUMN status SET NOT NULL directly against a table with production-scale row counts, no prior backfill or constraint staging. What's wrong?",
1239
+ "strictness": "high",
1240
+ "trials": 10,
1241
+ "passes": 10,
1242
+ "passRate": 1,
1243
+ "passAtK": 1,
1244
+ "grader": "judge",
1245
+ "status": "ran",
1246
+ "trialRecords": [
1247
+ {
1248
+ "output": "**Finding: blocking `NOT NULL` enforcement with no staged constraint**\n\n`migrations/.../alter_orders_status_not_null.sql — ALTER TABLE orders ALTER COLUMN status SET NOT NULL`, on a production-scale `orders` table, with no prior backfill and no `NOT VALID` + `VALIDATE CONSTRAINT` staging sequence.\n\n**Why it matters**\n\n- **Blocking lock on write traffic.** `SET NOT NULL` forces the engine to verify every existing row before the change commits. On a large table that verification scan is a full-table pass under an `ACCESS EXCLUSIVE` lock in Postgres (a `SHARE ROW EXCLUSIVE`/table-copy path depending on engine and version in MySQL). Concurrent reads and writes to `orders` stall for the entire duration of that scan — on production-scale row counts this is measured in minutes, not milliseconds, and can time out or cascade into application-level connection exhaustion.\n- **It will fail outright if any existing row violates it.** With no prior backfill, any row where `status IS NULL` — legacy rows, rows written before a default was added, rows from a partial-failure write path — aborts the migration mid-way. If that aborts inside an implicit transaction, there is no partial state; if the runner doesn't wrap it, you can be left on a half-applied version.\n- **No staging means the lock is held for the whole scan.** The safe pattern separates the \"declare the constraint\" step (fast, metadata-only) from the \"verify existing rows\" step (slow, can run with a weaker lock and be retried), so the exclusive-lock window stays small.\n\n**Fix direction (report only — not applied here)**\n\n1. **Backfill first, batched.** Populate `status` for existing `NULL` rows in row-range batches (e.g. by `id` range or `ctid`/primary-key chunk) rather than one unbatched `UPDATE`, so no single statement holds a long lock or generates an unbounded WAL/binlog burst. Confirm zero remaining `NULL` rows before step 2.\n2. **Add the constraint as `NOT VALID`, then validate separately.**\n - Postgres: `ALTER TABLE orders ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;` — this takes only a brief lock, does not scan the table, and immediately enforces the constraint for new writes.\n - Then, in a separate statement/transaction: `ALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;` — this scans existing rows under a weaker lock (`SHARE UPDATE EXCLUSIVE`) that does not block normal reads/writes.\n3. **Add a default / application-path guard** so new inserts cannot reintroduce `NULL` between the backfill and the constraint going live.\n4. **MySQL equivalent:** there is no `NOT VALID`; use online DDL (`ALGORITHM=INPLACE, LOCK=NONE` where supported for the column change) and still backfill first, or check whether a straight `MODIFY ... NOT NULL` can run online for the target version — otherwise plan a gh-ost/pt-online-schema-change style migration.\n\n**Verification before merge**\n\n- Confirm the migration was actually run against a production-scale copy (or an `EXPLAIN`/timing estimate), not just a dev table — current dev size is not evidence.\n- Confirm the backfill is batched and idempotent, and that a re-run is safe.\n- Confirm the `VALIDATE CONSTRAINT` step is a separate transaction from the `ADD CONSTRAINT ... NOT VALID` step, so the long scan never runs under the exclusive lock.\n\nThis finding stands regardless of the table's current size — the migration re-runs on every environment, including production once `orders` has grown.\n\n*(Read-only review: no files were modified. Fix directions above are reported, not applied.)*",
1249
+ "outputSha256": "ef4de572a7a79b7cba3adf899b0e10416c257d9394dc427ced6e268c1798dc3b",
1250
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1251
+ "deterministic": [],
1252
+ "judge": {
1253
+ "verdict": "pass",
1254
+ "reason": "Names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL and its ACCESS EXCLUSIVE full-table verification scan blocking reads/writes on production-scale rows. Gives concrete fix as a finding, not applied: ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID, then separate VALIDATE CONSTRAINT. No 'probably fast enough' hedge; explicitly rejects dev-size as evidence."
1255
+ },
1256
+ "passed": true
1257
+ },
1258
+ {
1259
+ "output": "**Finding: blocking `NOT NULL` change with no staged constraint path**\n\n`migrations/.../orders_status_not_null.sql` — `ALTER TABLE orders ALTER COLUMN status SET NOT NULL` run directly against a production-scale table, with no prior backfill and no `NOT VALID` + `VALIDATE CONSTRAINT` staging.\n\n**Why it matters**\n\n- On Postgres, `SET NOT NULL` takes an `ACCESS EXCLUSIVE` lock on `orders` and holds it while it scans the whole table to verify no existing row already has `status IS NULL`. On a production-scale table that scan is not instant, so concurrent reads and writes to `orders` are blocked for the whole duration — an outage-shaped migration, not just a slow one.\n- If any pre-existing row does have `status IS NULL`, the statement fails outright after holding that lock for the full scan. There is no backfill step in this diff, so if the migration order is (backfill elsewhere → this ALTER) and the backfill is not in this diff, that is a separate ordering gap to confirm; if there is no backfill at all, every NULL row is a latent failure on first run in each environment.\n- Because migrations re-run on every environment, a pattern that looks harmless against a small dev table still runs against production at full row count — \"the table is small right now\" is not a defense.\n\n**Fix direction (do not apply here — report only)**\n\nStage it instead of doing it in one statement:\n\n1. Backfill existing NULLs in batches (row-range or `ctid`/id-range chunks), pausing between batches, rather than one unbatched `UPDATE` — an unbatched backfill on a production-scale table holds locks and generates heavy WAL/binlog for its whole duration.\n2. Add the constraint in the non-blocking form: `ALTER TABLE orders ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;` — this takes only a brief lock to install, since it does not scan existing rows.\n3. Run `ALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;` as a separate statement — this scan takes a weaker lock that does not block writes.\n4. Only then, if a true `SET NOT NULL` is still wanted (e.g. to drop the CHECK constraint afterward), it becomes a cheap metadata-only change because Postgres can see the validated CHECK as proof.\n\n**Confirm before asserting the full scan**\n\nThe blocking-lock analysis above is standard Postgres behavior for `SET NOT NULL`. To confirm the actual cost against this table's real size and lock profile rather than assert it from reading alone, run `EXPLAIN ANALYZE` on the verification scan and check `pg_locks`/lock timeouts during a dry run on a production-sized clone.\n\n**Pre-existing vs. diff-introduced**\n\nThis finding is introduced by this diff. If `orders.status` was already being written as always-non-null by application code, that does not make the migration safe — the migration's lock and scan cost is independent of whether the data happens to satisfy the constraint.\n\nNo source file was modified by this review.",
1260
+ "outputSha256": "e6abf0476f6982f2862febe63fc101e50a22724ab63c1cec92777339156693cc",
1261
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1262
+ "deterministic": [],
1263
+ "judge": {
1264
+ "verdict": "pass",
1265
+ "reason": "Answer names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL, states it takes ACCESS EXCLUSIVE lock and scans whole table blocking reads/writes on production-scale table (outage-shaped). Gives concrete fix: ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID then separate VALIDATE CONSTRAINT, as report-only finding ('do not apply here'). No hedge dismissing it as probably fast."
1266
+ },
1267
+ "passed": true
1268
+ },
1269
+ {
1270
+ "output": "This is a blocking-`NOT NULL` schema change, the first item on the Step 2 blocking-schema-changes list.\n\n**What's wrong**\n\n`ALTER TABLE orders ALTER COLUMN status SET NOT NULL` is not a metadata-only change in Postgres. To enforce the constraint, Postgres must scan the entire table to verify no existing row already violates it. On a production-scale `orders` table that scan holds an `ACCESS EXCLUSIVE` lock for the duration, which blocks *reads as well as writes* — every concurrent query against `orders` queues behind the migration. Worse, because it's `ACCESS EXCLUSIVE`, any lock already held by a long-running transaction on `orders` blocks the migration itself, and behind it, every new query piles up. That is the classic \"one migration took the site down\" shape.\n\nThere is also a correctness gap independent of locking: if any existing row already has `status IS NULL`, the statement fails. The absence of a prior backfill means that outcome is possible rather than proven impossible, and either failure mode (constraint violation, or an unanticipated lock wait) surfaces only when the migration runs against real data.\n\n**What's missing from the diff, per the rule set**\n\n- No `NOT VALID` + `VALIDATE CONSTRAINT` staging. The safe Postgres sequence is two statements: `ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID`, which takes only a brief `ACCESS EXCLUSIVE` lock to record the constraint without scanning, then a separate `VALIDATE CONSTRAINT` statement, which takes a `SHARE UPDATE EXCLUSIVE` lock — concurrent reads and writes proceed during the scan.\n- No backfill. `NOT VALID` skips existing rows, so any NULLs must be backfilled *before* `VALIDATE CONSTRAINT` runs, or validation fails on exactly the rows the backfill was meant to fix. That backfill should itself be batched (Step 2's unbatched-backfill bullet) rather than a single unbounded `UPDATE` over `orders`.\n- No evidence the `status` column is being read as always-present by application code at the same moment — if this migration is meant to match a code change that assumes non-NULL, the database change and the code change need to be sequenced so there's no window where one is true and the other isn't.\n\n**Fix direction** — three ordered steps, not one statement:\n\n1. Batched backfill to eliminate NULLs in `status`.\n2. `ADD CONSTRAINT ... CHECK (status IS NOT NULL) NOT VALID` (brief lock only).\n3. `VALIDATE CONSTRAINT` (concurrent-safe scan; blocks nothing important).\n\nThen, and only if a true column-level `NOT NULL` attribute is still required for tooling reasons, note that Postgres 12+ can promote a validated `CHECK` to a real `NOT NULL` via `ALTER TABLE ... ALTER COLUMN ... SET NOT NULL`, reusing the validated constraint and skipping the full scan — but that promotion path depends on the constraint still being present and validated, so it must be sequenced after step 3, not run standalone as this diff does.\n\n**Caveat on assertion strength** — the rule set says not to claim a full scan without evidence. Here the underlying mechanic is documented Postgres behavior for `SET NOT NULL` absent a pre-validated constraint, so the scan is asserted from the engine's semantics, not guessed from query plans. What I cannot assert from the diff alone is the actual lock-wait severity, which depends on live row count, concurrent transaction length, and load — \"run this against a staging copy at production scale, or check `pg_stat_activity` for long-running transactions on `orders` before running it,\" is the confirmation step, not an assumption to skip.\n\nOne more thing worth flagging as a distinct finding rather than folding into the above: if this diff contains *only* the `SET NOT NULL` and no backfill, then the migration is also incomplete on its own terms — it presumes a state of the data it doesn't establish. That's a diff-introduced issue, not a pre-existing one, and it should be reported separately from the locking concern so the two fixes don't get conflated.",
1271
+ "outputSha256": "7d596aff2ae090a195ee3ef136778531cff19f2a03cd0facbe4f291362c39afd",
1272
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1273
+ "deterministic": [],
1274
+ "judge": {
1275
+ "verdict": "pass",
1276
+ "reason": "Names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL, explains it scans the whole table under ACCESS EXCLUSIVE lock blocking reads/writes on production-scale orders. Gives concrete safe sequence: ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID then separate VALIDATE CONSTRAINT, as fix direction (ordered steps), not applied. No hedge dismissing it as probably fast."
1277
+ },
1278
+ "passed": true
1279
+ },
1280
+ {
1281
+ "output": "The direct `ALTER TABLE orders ALTER COLUMN status SET NOT NULL` on a production-scale table is a blocking schema change with no staging. Specific issues:\n\n**What's wrong**\n\n1. **Blocking full-table scan under lock.** On Postgres, `SET NOT NULL` requires the engine to verify every existing row satisfies the constraint. On a production-scale table that is a full-table scan held under an `ACCESS EXCLUSIVE` lock on `orders`, blocking all concurrent reads and writes for the duration of the scan. On MySQL the equivalent DDL likewise forces a lock/rebuild depending on version and algorithm.\n\n2. **No backfill or constraint staging.** The diff runs the `NOT NULL` in one shot, so it depends on every existing row already having a non-null `status`. With no prior backfill (or a `CHECK ... NOT VALID` staging step), any pre-existing `NULL` row causes the statement to fail outright — and at production scale it fails only after doing all the scan work under the lock. Either way the migration is unsafe to run as written.\n\n3. **No rollback/partial-failure story.** If the ALTER fails partway on some backlog of `NULL` rows, the lock is released but the column is still nullable and the migration is left half-applied with no record of why.\n\n**Fix direction (do not apply — report only)**\n\nStage the constraint so no long lock is held and existing bad rows can be validated separately:\n\n- First backfill any `NULL` rows to a safe default (batched, e.g. range-limited, so each batch is a short transaction, not one unbounded `UPDATE`).\n- Then in Postgres add the constraint in the non-blocking form:\n `ALTER TABLE orders ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;`\n followed by a separate\n `ALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;`\n The `NOT VALID` add takes only a brief lock; the `VALIDATE` takes a weaker lock and can run without blocking writers.\n- On MySQL, confirm the DDL is an online/`ALGORITHM=INPLACE, LOCK=NONE` variant (or use a tool like gh-ost/pt-online-schema-change) before running anything table-rewriting; a plain blocking ALTER should not ship.\n\n**Note on scope:** the blocking-lock analysis depends on the target engine, and the diff as described doesn't say which one; the exact lock behavior (and therefore the severity) differs between Postgres and MySQL, so confirm the engine before treating the `NOT VALID`/`VALIDATE` sequence as directly applicable. Also worth checking whether `status` has an index — if the nullable state was masking a needs-backfill query, the supporting index question is a separate finding. Run `EXPLAIN ANALYZE` on the hot query against this column rather than assuming a full scan.\n\nReport this as a diff-introduced, blocking-schema-change finding with file:line and the staging fix direction above; do not edit the file.",
1282
+ "outputSha256": "569a9cff193356bc466cdb6ee9fb606804a4448c800552f611a8ebd38aa10887",
1283
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1284
+ "deterministic": [],
1285
+ "judge": {
1286
+ "verdict": "pass",
1287
+ "reason": "Answer names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL and states it takes ACCESS EXCLUSIVE lock with full-table verification scan on production-scale table (points 1-2). It gives the concrete safe sequence as a finding: ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID followed by separate VALIDATE CONSTRAINT, explicitly 'do not apply — report only'. No hedge dismissing the direct SET NOT NULL as probably fast."
1288
+ },
1289
+ "passed": true
1290
+ },
1291
+ {
1292
+ "output": "**Finding: `ALTER TABLE orders ALTER COLUMN status SET NOT NULL` on a production-scale table, with no backfill or constraint-staging step.**\n\n**Risk category:** Unsafe blocking schema change (Step 2).\n\n**What's wrong, at the DB-mechanics level:**\n\n1. **No prior backfill of existing NULLs.** If any existing row already has `status IS NULL`, this statement doesn't just block — it **fails outright** with an error (Postgres: `column \"status\" contains null values`; MySQL: `Invalid use of NULL value` / data-truncation error depending on sql_mode). So step one is a backfill (`UPDATE orders SET status = '<default>' WHERE status IS NULL`), batched by row-range if the table is large enough for an unbatched UPDATE to hold locks or generate excessive WAL/binlog for a meaningful duration.\n\n2. **The `SET NOT NULL` itself takes an `ACCESS EXCLUSIVE` lock and scans the whole table.** In Postgres, `ALTER COLUMN ... SET NOT NULL` is not a metadata-only catalog flip — it acquires `ACCESS EXCLUSIVE` and performs a full-table verification scan to prove no NULL exists. On a production-scale `orders` table, that means **all concurrent reads and writes to `orders` are blocked for the full duration of the scan**, which on a large table can be minutes. That is a write outage on one of the hottest tables in the schema.\n\n3. **It queues behind (and blocks) other traffic, and cascades.** `ACCESS EXCLUSIVE` is the strongest lock; it waits for every in-flight query on `orders` to drain, and while waiting, every *subsequent* query touching `orders` queues behind *it*. A lock-wait pileup on `orders` can take down unrelated features, not just the migration.\n\n**Fix direction (this skill does not apply it — report only):**\n\nUse the staged, non-blocking pattern instead of a direct `SET NOT NULL`:\n\n1. Backfill existing NULLs in **batches**, limited by row range / `ctid` / primary-key range, so no single statement holds a long lock or produces a WAL/binlog spike.\n2. Add the constraint as `NOT VALID` first, which is fast and takes only a brief `ACCESS EXCLUSIVE` lock to install the catalog entry without scanning the table:\n `ALTER TABLE orders ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;`\n3. Validate it in a **separate** statement, which takes only a `SHARE UPDATE EXCLUSIVE` lock (does not block concurrent reads/writes on Postgres):\n `ALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;`\n4. Only after validation succeeds, and once you've confirmed the check constraint is doing the enforcement, drop the redundant catalog step or migrate `SET NOT NULL` carefully — the `NOT VALID` + `VALIDATE` sequence is the safe path; the direct `SET NOT NULL` is not.\n\nOn MySQL, the equivalent is confirming the ALTER can run `ALGORITHM=INPLACE, LOCK=NONE`; a direct `SET NOT NULL` there also requires a table rebuild in many versions, so confirm the online-DDL path before running it at production scale.\n\n**Verification note:** If the exact lock behavior on your engine version is uncertain, run `EXPLAIN`/lock inspection or test the DDL against a production-sized clone before shipping — do not assert the exact lock mode without evidence, but the pattern (unbatched, un-staged, non-`NOT VALID` NOT NULL on a production-scale table) is a blocker regardless.\n\n**No source file was modified — this is a report only.**",
1293
+ "outputSha256": "7257c9e0663cc1f45ba30a1cfbe0340cb66def29e0d185e15d8ae30e1c32bc5e",
1294
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1295
+ "deterministic": [],
1296
+ "judge": {
1297
+ "verdict": "pass",
1298
+ "reason": "Answer names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL, states it takes ACCESS EXCLUSIVE lock and full-table verification scan blocking all reads/writes on a production-scale table (write outage). It gives the concrete fix as a finding: ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID, then separate ALTER TABLE orders VALIDATE CONSTRAINT, and states it is report-only, no file modified. No hedge dismissing the direct SET NOT NULL."
1299
+ },
1300
+ "passed": true
1301
+ },
1302
+ {
1303
+ "output": "This is a blocking schema change with no staging — the classic unsafe `SET NOT NULL`. Here is the review.\n\n## Finding\n\n```\nmigrations/<file>.sql:<line> — ALTER TABLE orders ALTER COLUMN status SET NOT NULL\n run directly against a production-scale `orders` table, with no prior\n backfill and no NOT VALID + VALIDATE CONSTRAINT staging.\n```\n\nCategory from the focus list: **blocking schema change** (Step 2). Risk and mechanics:\n\n- To set `NOT NULL`, the engine must verify every existing row actually satisfies the constraint. On Postgres this takes an `ACCESS EXCLUSIVE` lock on `orders` for the duration of the full-table scan. During that window every concurrent read and write on `orders` — including the order-placement path itself — blocks. On a production-scale table the scan is not instant, so this is an outage-shaped migration, not a maintenance-window nicety.\n- The `ALTER` is also atomic and all-or-nothing in the unsafe form: if it fails partway (statement timeout, a lock wait that exceeds `lock_timeout`), it rolls back entirely and you retry the same blocking scan. There is no incremental progress.\n- There is no prior backfill in the diff. If any existing row has `status IS NULL`, the statement fails outright after acquiring the lock — which is the *better* outcome, because the worse one is it succeeds and you've just taken a long exclusive lock on a hot table.\n\n## Why this is not a \"small table, fine\" case\n\nEven if `orders` is modest in your local/dev database, the migration runs again in every environment, including the one where `orders` has production-scale row counts (Red Flags: \"The table is small right now, the blocking ALTER is fine\"). The pattern is what gets flagged, regardless of the row count visible at review time.\n\n## Fix direction (do not apply — report only)\n\nSplit into staged steps, Postgres form:\n\n1. **Backfill first, batched.** Update any `status IS NULL` rows in limited ranges (e.g. `WHERE id BETWEEN n AND n+k`), so you are never issuing one unbounded `UPDATE` against the whole table — that is itself the \"single unbatched backfill\" finding from the index/backfill check.\n2. **Add the constraint as `NOT VALID`**, which takes only a brief lock and skips the full-table verification scan:\n ```sql\n ALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;\n ```\n3. **Validate in a separate statement**, on its own, so the long scan happens under a weaker lock and does not block concurrent reads/writes the way the original `ACCESS EXCLUSIVE` `SET NOT NULL` would:\n ```sql\n ALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n ```\n Run this in its own migration/commit so a timeout here does not roll back step 2.\n4. Optionally follow up later to convert to a true column `NOT NULL` once the validated check constraint is in place.\n\nMySQL equivalent: confirm the `ALTER` is `ALGORITHM=INPLACE, LOCK=NONE`-eligible for the engine/version before shipping, and backfill in batches — a naive `SET NOT NULL` there is the same class of blocking change.\n\n## Not asserted without evidence\n\nI am not claiming `status` actually has NULLs today; the diff shows no backfill either way, which is itself the gap. Confirm the current NULL count before choosing step 1's backfill vs. skip.\n\n## Scope notes\n\n- This finding is **diff-introduced** — the unsafe `ALTER` is the change under review, not a pre-existing statement the diff merely touches.\n- No source file was modified by this review; read-only.\n\n## Red-flag check\n\n- \"It's a small table, the blocking ALTER is fine\" — rejected; see above.\n- \"I'll just add the staging statements myself since it's an obvious fix\" — rejected; this skill reports the finding and fix direction only, it does not edit the migration.",
1304
+ "outputSha256": "abc5090170dc2019f4ea099ac3bf9e1e63c959ff0b1474ff7bbf1a170fb3577c",
1305
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1306
+ "deterministic": [],
1307
+ "judge": {
1308
+ "verdict": "pass",
1309
+ "reason": "Answer names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL, explains it takes ACCESS EXCLUSIVE lock for a full-table verification scan on production-scale orders (outage-shaped), and gives the concrete fix direction: ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID then separate ALTER TABLE orders VALIDATE CONSTRAINT, reported as finding only ('do not apply — report only'). It rejects the 'small table, fine' hedge."
1310
+ },
1311
+ "passed": true
1312
+ },
1313
+ {
1314
+ "output": "**Finding: blocking schema change on a production-scale table**\n\n`migrations/..._set_status_not_null.sql` — `ALTER TABLE orders ALTER COLUMN status SET NOT NULL`, run directly against a table with production-scale row counts and no prior backfill or staging step.\n\n**Why it matters (blocking schema-change risk, not a style nit)**\n\n- In Postgres, `SET NOT NULL` is not metadata-only in the way people assume: when there is no matching validated `CHECK (col IS NOT NULL)` constraint, Postgres scans the whole table to verify no NULL exists before it will mark the column `NOT NULL` (pre-v12 this takes an `ACCESS EXCLUSIVE` lock for the entire scan; more recent versions short-circuit to a `CHECK` constraint but still need the table verified). On a production-scale `orders` table that scan can run for minutes-to-hours, and for the duration concurrent reads and writes to `orders` are blocked. That's an outage risk, not a performance nit.\n- The diff also does not establish that the column *can* be `NOT NULL` yet. If any existing row has `status IS NULL`, the statement fails outright at deploy time — after having already held the lock for the duration of the failed scan. There's no backfill step in the diff to close that gap, and no `DEFAULT`/`CHECK` to define what should happen to future inserts that omit `status`.\n- A single unbatched operation against a production-scale table with no row-range/batch limiting compounds the lock/WAL duration.\n\n**Fix direction (do not apply here — this is reporting only)**\n\nStage it so no long-held lock ever blocks traffic:\n\n1. Backfill existing rows in bounded batches (e.g. by primary-key range), with a small sleep between batches, committing each batch — not one unbounded `UPDATE`.\n2. Add a `NOT VALID` check constraint that doesn't scan the table:\n `ALTER TABLE orders ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;`\n3. Validate it in a separate statement (`VALIDATE CONSTRAINT orders_status_not_null`), which takes only a `SHARE UPDATE EXCLUSIVE` lock and does not block reads/writes.\n4. Postgres then treats the validated constraint as proof for `SET NOT NULL`, so the eventual `ALTER COLUMN status SET NOT NULL` becomes a fast catalog-only change.\n\nFor MySQL, the equivalent concern is whether the `ALTER` runs with `ALGORITHM=INPLACE, LOCK=NONE` — a plain blocking `ALTER` on a large `orders` table has the same outage profile.\n\n**Also worth confirming while in this diff**\n\n- The application path that writes `status` (insert and update) actually sets a value on every write; a `SET NOT NULL` without that leaves a latent insert failure for callers that omit the column.\n- Whether the target engine version's `SET NOT NULL` behavior was checked, rather than assumed — if uncertain, say \"run `EXPLAIN`/check the engine version before deploy\" rather than asserting the exact lock level.\n\nThis is a diff-introduced finding: the migration file itself is new, so there's no pre-existing code to separate it from. And to be explicit — this review is read-only; I have not edited the migration.",
1315
+ "outputSha256": "66131bf8b55a1636ec0b95280945544c52419835577eefb59d0745296bad3199",
1316
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1317
+ "deterministic": [],
1318
+ "judge": {
1319
+ "verdict": "pass",
1320
+ "reason": "Criterion 1: names the exact statement 'ALTER TABLE orders ALTER COLUMN status SET NOT NULL' and explains the full-table verification scan holding ACCESS EXCLUSIVE lock, calling it an outage risk on production-scale orders. Criterion 2: gives concrete fix as a finding — 'ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID' then separate 'VALIDATE CONSTRAINT', explicitly 'do not apply here — this is reporting only'. No hedge dismissing the direct SET NOT NULL."
1321
+ },
1322
+ "passed": true
1323
+ },
1324
+ {
1325
+ "output": "This is the \"unsafe blocking NOT NULL\" pattern from Step 2's blocking-schema-changes checklist.\n\n**What the statement does**\n\n`ALTER TABLE orders ALTER COLUMN status SET NOT NULL` — setting a column NOT NULL requires the engine to verify that no existing row violates it. On Postgres, that verification is a full-table scan performed while holding an `ACCESS EXCLUSIVE` lock on `orders`. The lock is held for the entire duration of the scan, and it does not help that an earlier migration may have added a default — any row written before the constraint existed, or written by a path that skipped the default (a bulk load, a `COPY`, a direct `UPDATE ... SET status = NULL`, an older code version still running during the deploy window), can still be NULL.\n\n**Why each omitted step matters**\n\n- **No prior backfill** — if any existing row has `status IS NULL`, the ALTER fails outright. Worse, it fails *after* the scan, having already held the exclusive lock for the whole scan duration, so a large table pays the outage cost for nothing.\n- **No constraint staging** — the safe Postgres shape is two statements:\n 1. `ALTER TABLE orders ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;` — takes only a brief lock; marks the constraint unvalidated so it is not enforced against existing rows yet.\n 2. `ALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;` — scans the table while holding only a `SHARE UPDATE EXCLUSIVE` lock, which does **not** block reads or writes. Run it in its own transaction, outside a transaction block (`VALIDATE CONSTRAINT` inside a multi-statement txn escalates the lock and defeats the point).\n 3. Optionally, once validated, `ALTER COLUMN status SET NOT NULL` becomes cheap because Postgres can use the validated check constraint as proof instead of rescanning.\n\n On **MySQL**, the equivalent hazard is a table-rebuild copy during the ALTER; there is no `NOT VALID` equivalent, so this needs `ALGORITHM=INPLACE, LOCK=NONE` verified for the specific version, or an online-schema-change tool (pt-osc / gh-ost) — and `SET NOT NULL` is frequently not in-placeable at all, so a copy is the expected outcome on a large table.\n\n**Production-scale impact**\n\nOn a table with production row counts, the exclusive lock means concurrent reads and writes on `orders` block for the full scan — any query touching `orders` queues behind it, and lock queueing can cascade into connection-pool exhaustion and timeouts elsewhere. That is an outage, not a slow migration.\n\n**Fix direction**\n\nReplace the single `SET NOT NULL` with the staged form appropriate to the engine: backfill any remaining NULLs in bounded batches first, add the `CHECK ... NOT VALID` constraint, validate it separately, and only then set NOT NULL. Whatever the shape, none of this should run as one unbatched statement against a production-scale table.\n\n**Note for the review report**: this is a diff-introduced finding, not a pre-existing one. Report it as `migrations/<file>.sql:<line>` with the risk category (blocking schema change / lock duration), and flag the unbatched backfill separately if the diff also contains one — the missing backfill is implied here but should be stated as its own finding if a backfill statement is present in the diff.\n\nI did not modify any file; this is a finding and fix direction only.",
1326
+ "outputSha256": "fa3c0b5e86730e0b2c42d1d1ea311a2f4ad9163876a50cf088e46fc003d80032",
1327
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1328
+ "deterministic": [],
1329
+ "judge": {
1330
+ "verdict": "pass",
1331
+ "reason": "Names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL and its ACCESS EXCLUSIVE full-table scan lock, flags production-scale outage risk, and gives the concrete safe sequence: ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID then separate VALIDATE CONSTRAINT, explicitly as a finding ('I did not modify any file'). No hedge dismissing it as probably fast."
1332
+ },
1333
+ "passed": true
1334
+ },
1335
+ {
1336
+ "output": "This is a blocking schema change with no validation staging, and it has two separate problems.\n\n**1. The ALTER itself blocks.**\n\n`ALTER TABLE orders ALTER COLUMN status SET NOT NULL` is not an instant metadata-only change in Postgres when applied to a column that lacks a prior trusted constraint. To enforce `NOT NULL`, Postgres must scan the whole table to confirm no existing row violates it, and it holds an `ACCESS EXCLUSIVE` lock on `orders` for the duration of that scan. On a production-scale table that means concurrent reads and writes to `orders` are queued behind the ALTER until the scan finishes — not a brief metadata lock, a full-table-duration outage on your hottest table. On MySQL the story is engine-version dependent, but a plain `MODIFY COLUMN ... NOT NULL` on a large table is likewise not a free operation and can force a table rebuild under a blocking lock depending on the engine and version. Either way, \"runs directly against production-scale rows, no staging\" is the lock-duration risk this diff is introducing.\n\n**2. There is no backfill, so the migration will fail outright on any row that is already NULL.**\n\nEven setting the locking aside: if `status` currently has NULLs (which the absence of any preceding backfill strongly implies), the ALTER does not silently coerce them — it aborts with a constraint-violation error partway through. So this migration has a good chance of either hanging production under the lock while it scans, or failing, rolling back, and leaving you with a lock-held abort. And note the \"table is small right now\" temptation is a trap — a migration is written once and runs again on every environment, including production after the table has grown.\n\n**Fix direction (not applied — reporting only):**\n\n```sql\n-- Step 1: add the constraint as NOT VALID. This takes only a brief lock\n-- and does NOT scan the table.\nALTER TABLE orders\n ADD CONSTRAINT orders_status_not_null\n CHECK (status IS NOT NULL) NOT VALID;\n\n-- Step 2: backfill existing NULLs in batches, out of the migration path,\n-- so you don't hold a long write transaction. Run repeatedly until 0 rows.\nUPDATE orders SET status = 'unknown'\nWHERE id IN (\n SELECT id FROM orders WHERE status IS NULL LIMIT 10000\n);\n\n-- Step 3: validate separately. VALIDATE CONSTRAINT takes only a SHARE\n-- UPDATE EXCLUSIVE lock (does not block reads/writes) and can be\n-- interrupted/resumed.\nALTER TABLE orders VALIDATE CONSTRAINT orders_status_not_null;\n```\n\nKey points to flag in the report:\n\n- **Why the staging works:** `NOT VALID` skips the full scan on add, the constraint still enforces on new writes, and `VALIDATE CONSTRAINT` uses a weaker lock that permits concurrent reads and writes. Once validated, a `CHECK (status IS NOT NULL)` is logically equivalent to `NOT NULL` for the planner and can later be swapped to a true `SET NOT NULL` (which is now instant, because Postgres can use the validated constraint as proof).\n- **The backfill must be batched and run outside the migration**, not as a single unbatched `UPDATE` — an unbatched backfill on a production-scale table holds locks and generates excessive WAL/binlog for a meaningful duration, which is its own incident.\n- **If you're on MySQL**, confirm the engine/version and check `ALGORITHM=INPLACE` / `LOCK=NONE` behavior for the equivalent change before assuming the pattern transfers; a table rebuild under lock is the MySQL version of the same problem.\n\nOne caveat on the `EXPLAIN`-style claims: I'm asserting the Postgres locking behavior from the statement shape (SET NOT NULL with no pre-existing validated constraint = full scan + ACCESS EXCLUSIVE). If the schema already has a validated `CHECK (status IS NOT NULL)` from an earlier migration, the plan changes and this could be near-instant — worth confirming against the actual schema before treating the outage estimate as certain. But the absence of any constraint staging in this diff, plus the absence of a backfill, is the finding regardless.\n\nThis is a report only; I have not modified the migration.",
1337
+ "outputSha256": "db436e39cca3fffb8edfb061a9cd8fd3431f3014fd7088a5966119e219662697",
1338
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1339
+ "deterministic": [],
1340
+ "judge": {
1341
+ "verdict": "pass",
1342
+ "reason": "Answer names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL, states it holds ACCESS EXCLUSIVE lock for a full-table verification scan on production-scale rows (outage risk), and gives the concrete fix as a finding: ADD CONSTRAINT ... CHECK (status IS NOT NULL) NOT VALID then separate VALIDATE CONSTRAINT, explicitly 'not applied — reporting only'. No hedge dismissing it as probably fast."
1343
+ },
1344
+ "passed": true
1345
+ },
1346
+ {
1347
+ "output": "**Finding — blocking `SET NOT NULL` on a production-scale table**\n\n`ALTER TABLE orders ALTER COLUMN status SET NOT NULL` on a table with production-scale row counts is a blocking schema change and should not merge as written.\n\n**Why it matters**\n\n- In Postgres, `SET NOT NULL` is not a metadata-only change. Postgres must scan the entire table to verify no existing row violates the constraint, and that scan is held under an `ACCESS EXCLUSIVE` lock. Concurrent reads and writes against `orders` are blocked for the full duration of the scan — on a large table that is a production outage window, not a slow migration.\n- The diff has no prior backfill and no staged constraint, so if any existing row has `status IS NULL`, the `ALTER` fails outright after holding the lock for the length of the scan and then rolling back. Either outcome (failure or long blocking lock) is an incident on a production-scale table.\n- This is the shape flagged in Step 2 under \"Blocking schema changes\": a `NOT NULL` column or constraint added to a table that can have existing rows, with no `NOT VALID` + `VALIDATE CONSTRAINT` sequence.\n\n**Fix direction (not applied — read-only review)**\n\nSplit into staged, low-lock steps:\n\n1. Backfill in bounded batches, e.g. `UPDATE orders SET status = <default> WHERE status IS NULL` with a `LIMIT`/row-range loop rather than one unbatched `UPDATE`, so no single statement holds locks or generates excessive WAL for a meaningful duration.\n2. Add the constraint as `NOT VALID` so the initial `ADD CONSTRAINT` takes a brief lock and skips the full-table verification:\n `ALTER TABLE orders ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID;`\n3. Run `VALIDATE CONSTRAINT` as a separate statement, which takes a weaker lock (`SHARE UPDATE EXCLUSIVE`) and does not block ordinary reads/writes while it verifies existing rows.\n4. Only after that, in a later migration, `ALTER COLUMN status SET NOT NULL` — Postgres can then use the already-validated check constraint to skip the full scan.\n\nFor MySQL, confirm the equivalent online-DDL path (`ALGORITHM=INPLACE`/`LOCK=NONE` where supported for the column definition change plus backfill) rather than assuming the Postgres staging sequence maps directly.\n\n**Notes**\n\n- Flag is diff-introduced, not pre-existing.\n- Do not apply any of this — findings and fix direction only.\n- If you want certainty about the actual lock behavior and scan cost on your engine and row counts, run the staged form against a production-sized copy and observe lock acquisition, rather than assuming the scan is brief because the table \"seems fine.\"",
1348
+ "outputSha256": "bf1c699299d3c8c7fc2b7ac7328ea03264d85dee2b30b6e5261507027ef3ba0c",
1349
+ "promptSha256": "565bfbade5a111c5e0b85550cfa872b49cb5feb973ef2104e13a2304b3346049",
1350
+ "deterministic": [],
1351
+ "judge": {
1352
+ "verdict": "pass",
1353
+ "reason": "Answer names the exact statement ALTER TABLE orders ALTER COLUMN status SET NOT NULL, explains it takes ACCESS EXCLUSIVE lock plus full-table verification scan blocking reads/writes on production-scale table (outage window). Gives concrete fix: ADD CONSTRAINT orders_status_not_null CHECK (status IS NOT NULL) NOT VALID then separate VALIDATE CONSTRAINT, explicitly as read-only finding not applied. No hedge accepting the direct SET NOT NULL."
1354
+ },
1355
+ "passed": true
1356
+ }
1357
+ ]
1358
+ }
1359
+ ],
1360
+ "verdict": "fail",
1361
+ "scope": "bundled",
1362
+ "skillDigest": "24732066b7b858fef24afa7758fbc86b2cc12d2826b0b1c3e1b517d2a463c0af",
1363
+ "catalogDigest": "14504a0807a0089488b9cb690c4b13f20865cd7a7fb69a1e5d8dfea8bfd5fbd1",
1364
+ "judgePromptVersion": "2026-09-25.1",
1365
+ "runner": "deepseek",
1366
+ "model": "deepseek-chat",
1367
+ "runnerPromptVersion": "2026-09-25.1",
1368
+ "recordedAt": "2026-09-25T18:17:56.418Z",
1369
+ "judge": "deepseek",
1370
+ "judgeModel": "deepseek-chat"
1371
+ },
1372
+ {
1373
+ "schemaVersion": "1.0.0",
1374
+ "skillId": "sql-db/sql-db-build-fix",
1375
+ "strictness": "high",
1376
+ "trials": 10,
1377
+ "triggerAccuracy": {
1378
+ "truePositive": 3,
1379
+ "falsePositive": 1,
1380
+ "positives": 7,
1381
+ "negatives": 7
1382
+ },
1383
+ "evidence": "authored",
1384
+ "scenarios": [
1385
+ {
1386
+ "id": "trigger-positive-1",
1387
+ "kind": "trigger-positive",
1388
+ "prompt": "This migration keeps failing with a unique constraint violation, help me fix the root cause",
1389
+ "strictness": "high",
1390
+ "trials": 1,
1391
+ "passes": 1,
1392
+ "passRate": 1,
1393
+ "passAtK": 1,
1394
+ "grader": "trigger-rank-fork-family",
1395
+ "status": "ran",
1396
+ "deterministic": true
1397
+ },
1398
+ {
1399
+ "id": "trigger-positive-2",
1400
+ "kind": "trigger-positive",
1401
+ "prompt": "A query that used to hit the index now runs a full table scan, what changed?",
1402
+ "strictness": "high",
1403
+ "trials": 1,
1404
+ "passes": 1,
1405
+ "passRate": 1,
1406
+ "passAtK": 1,
1407
+ "grader": "trigger-rank-fork-family",
1408
+ "status": "ran",
1409
+ "deterministic": true
1410
+ },
1411
+ {
1412
+ "id": "trigger-positive-3",
1413
+ "kind": "trigger-positive",
1414
+ "prompt": "This migration is erroring that the index already exists on a re-run, fix it",
1415
+ "strictness": "high",
1416
+ "trials": 1,
1417
+ "passes": 0,
1418
+ "passRate": 0,
1419
+ "passAtK": 0,
1420
+ "grader": "trigger-rank-fork-family",
1421
+ "status": "ran",
1422
+ "deterministic": true
1423
+ },
1424
+ {
1425
+ "id": "trigger-positive-4",
1426
+ "kind": "trigger-positive",
1427
+ "prompt": "We're seeing a deadlock during this migration's backfill, help me resolve it",
1428
+ "strictness": "high",
1429
+ "trials": 1,
1430
+ "passes": 1,
1431
+ "passRate": 1,
1432
+ "passAtK": 1,
1433
+ "grader": "trigger-rank-fork-family",
1434
+ "status": "ran",
1435
+ "deterministic": true
1436
+ },
1437
+ {
1438
+ "id": "trigger-positive-5",
1439
+ "kind": "trigger-positive",
1440
+ "prompt": "This backfill is failing a NOT NULL check on some rows, figure out why",
1441
+ "strictness": "high",
1442
+ "trials": 1,
1443
+ "passes": 0,
1444
+ "passRate": 0,
1445
+ "passAtK": 0,
1446
+ "grader": "trigger-rank-fork-family",
1447
+ "status": "ran",
1448
+ "deterministic": true
1449
+ },
1450
+ {
1451
+ "id": "trigger-positive-6",
1452
+ "kind": "trigger-positive",
1453
+ "prompt": "Our migration tool errors that 0047_add_orders_index has the same timestamp prefix as a teammate's 0047_add_wallet_column, and only one of them applies",
1454
+ "strictness": "high",
1455
+ "trials": 1,
1456
+ "passes": 0,
1457
+ "passRate": 0,
1458
+ "passAtK": 0,
1459
+ "grader": "trigger-rank-fork-family",
1460
+ "status": "ran",
1461
+ "deterministic": true
1462
+ },
1463
+ {
1464
+ "id": "trigger-positive-7",
1465
+ "kind": "trigger-positive",
1466
+ "prompt": "The migration times out acquiring a lock in production, what's actually going on?",
1467
+ "strictness": "high",
1468
+ "trials": 1,
1469
+ "passes": 0,
1470
+ "passRate": 0,
1471
+ "passAtK": 0,
1472
+ "grader": "trigger-rank-fork-family",
1473
+ "status": "ran",
1474
+ "deterministic": true
1475
+ },
1476
+ {
1477
+ "id": "trigger-negative-1",
1478
+ "kind": "trigger-negative",
1479
+ "prompt": "npm install is failing with a peer dependency conflict",
1480
+ "strictness": "high",
1481
+ "trials": 1,
1482
+ "passes": 1,
1483
+ "passRate": 1,
1484
+ "passAtK": 1,
1485
+ "grader": "trigger-rank-fork-family",
1486
+ "status": "ran",
1487
+ "deterministic": true
1488
+ },
1489
+ {
1490
+ "id": "trigger-negative-2",
1491
+ "kind": "trigger-negative",
1492
+ "prompt": "This Django migration conflicts with another migration, please merge them",
1493
+ "strictness": "high",
1494
+ "trials": 1,
1495
+ "passes": 0,
1496
+ "passRate": 0,
1497
+ "passAtK": 0,
1498
+ "grader": "trigger-rank-fork-family",
1499
+ "status": "ran",
1500
+ "deterministic": true
1501
+ },
1502
+ {
1503
+ "id": "trigger-negative-3",
1504
+ "kind": "trigger-negative",
1505
+ "prompt": "cargo build is failing for this Rust crate",
1506
+ "strictness": "high",
1507
+ "trials": 1,
1508
+ "passes": 1,
1509
+ "passRate": 1,
1510
+ "passAtK": 1,
1511
+ "grader": "trigger-rank-fork-family",
1512
+ "status": "ran",
1513
+ "deterministic": true
1514
+ },
1515
+ {
1516
+ "id": "trigger-negative-4",
1517
+ "kind": "trigger-negative",
1518
+ "prompt": "Fix this failing Go test that uses testify assertions",
1519
+ "strictness": "high",
1520
+ "trials": 1,
1521
+ "passes": 1,
1522
+ "passRate": 1,
1523
+ "passAtK": 1,
1524
+ "grader": "trigger-rank-fork-family",
1525
+ "status": "ran",
1526
+ "deterministic": true
1527
+ },
1528
+ {
1529
+ "id": "trigger-negative-5",
1530
+ "kind": "trigger-negative",
1531
+ "prompt": "Write a new migration that adds a column to this table",
1532
+ "strictness": "high",
1533
+ "trials": 1,
1534
+ "passes": 1,
1535
+ "passRate": 1,
1536
+ "passAtK": 1,
1537
+ "grader": "trigger-rank-fork-family",
1538
+ "status": "ran",
1539
+ "deterministic": true
1540
+ },
1541
+ {
1542
+ "id": "trigger-negative-6",
1543
+ "kind": "trigger-negative",
1544
+ "prompt": "Review this migration diff for injection risk before merging",
1545
+ "strictness": "high",
1546
+ "trials": 1,
1547
+ "passes": 1,
1548
+ "passRate": 1,
1549
+ "passAtK": 1,
1550
+ "grader": "trigger-rank-fork-family",
1551
+ "status": "ran",
1552
+ "deterministic": true
1553
+ },
1554
+ {
1555
+ "id": "trigger-negative-7",
1556
+ "kind": "trigger-negative",
1557
+ "prompt": "Resolve this Terraform state lock error",
1558
+ "strictness": "high",
1559
+ "trials": 1,
1560
+ "passes": 1,
1561
+ "passRate": 1,
1562
+ "passAtK": 1,
1563
+ "grader": "trigger-rank-fork-family",
1564
+ "status": "ran",
1565
+ "deterministic": true
1566
+ },
1567
+ {
1568
+ "id": "no-constraint-drop-to-fix-violation",
1569
+ "kind": "behavior",
1570
+ "prompt": "A migration's backfill is failing with a NOT NULL constraint violation on some rows. How should I fix it?",
1571
+ "strictness": "high",
1572
+ "trials": 10,
1573
+ "passes": 10,
1574
+ "passRate": 1,
1575
+ "passAtK": 1,
1576
+ "grader": "judge",
1577
+ "status": "ran",
1578
+ "trialRecords": [
1579
+ {
1580
+ "output": "Find the actual rows that hit the violation and fix the logic or data that let a NULL reach that column — do not drop or loosen the NOT NULL constraint to make the error stop.\n\nConcretely:\n\n1. Reproduce and read the exact error — which column, which constraint, which statement in the backfill.\n2. Query for the violating rows directly, e.g. `SELECT id FROM <table> WHERE <column> IS NULL;`. Don't stop at the first row the error reported; find every row the backfill would set to NULL.\n3. Diagnose why the backfill's derivation produced a NULL for those rows. The common causes:\n - The backfill's `WHERE` clause doesn't match those rows (e.g. it selects `status IS NULL` but the legacy rows actually hold `''`), so the new column never gets populated.\n - The source expression is a correlated subquery/join that returns no row for those ids.\n - A `CASE`/`COALESCE` chain has no fallback branch.\n4. Fix the backfill expression (widen the `WHERE`, add a `COALESCE` fallback, correct the join) — or, if the data is genuinely wrong, fix the data first and then backfill.\n\nIf the intended resolution is truly that some rows must remain NULL, that is a different design decision (make the column nullable in a separate, deliberate migration) — not a fix for a backfill that is producing NULLs it shouldn't.\n\nThen verify: re-run the migration up (and down if it has one) against a full-size copy of the table, and confirm zero rows still violate the constraint across the whole backfill, not just the originally failing row.\n\nReport format:\n\n```\nFixed: migrations/2026_09_18_backfill_status.sql\n - Root cause: 340 legacy rows held status = '' (empty string, not NULL)\n from a prior import, which the backfill's WHERE status IS NULL clause\n never matched, so the new column stayed NULL\n - Widened the backfill condition to also match status = ''\n - Re-ran clean against a full-size copy of the table\n```\n\nNever delete or skip the failing migration step, and never drop the NOT NULL constraint as a shortcut — both leave the schema or the data silently wrong.",
1581
+ "outputSha256": "a6e4cccf3c43d559039e2816114613aabeacbf66c7b8571da002feaa2380de26",
1582
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1583
+ "deterministic": [],
1584
+ "judge": {
1585
+ "verdict": "pass",
1586
+ "reason": "Answer states investigating which rows fail and why backfill logic missed them: step 2 queries violating rows, step 3 diagnoses WHERE clause mismatch (e.g. selects status IS NULL but legacy rows hold ''), subquery returning no row, missing COALESCE fallback. Step 4 fixes backfill expression (widen WHERE, add COALESCE, correct join). Explicitly rejects dropping/loosening NOT NULL. No fail criterion holds."
1587
+ },
1588
+ "passed": true
1589
+ },
1590
+ {
1591
+ "output": "Find the actual rows that would violate the constraint before deciding how to fix the backfill — don't touch the constraint itself.\n\n**Diagnose first.** Run the backfill's `UPDATE`/`INSERT` as a `SELECT` against the same rows (or wrap it in a transaction and roll back) to see exactly which rows produce NULL for the constrained column, and what distinguishes them from the rows that don't. Common causes:\n\n- A `WHERE` clause that's too narrow — e.g. backfilling `WHERE status IS NULL` but legacy rows actually hold `''` or some sentinel instead of NULL.\n- A derivation expression that returns NULL for a case the author didn't anticipate (a join that misses, a `CASE` with no `ELSE`, a `NULLIF` on an already-NULL input).\n- A genuinely missing source value — the column's data simply wasn't populated for those rows in whatever created them.\n\n**Then fix the backfill, not the constraint.** The `NOT NULL` constraint is doing its job by catching exactly this. Depending on what you find:\n\n- Widen the backfill's condition or add the missing `CASE`/join branch so those rows get a real value.\n- If those rows legitimately have no value yet, give the column a documented, deliberate default as part of the migration — but decide that consciously and say so, rather than silently letting NULL through.\n- If the derivation requires a source row that truly doesn't exist, resolve that (fix the source, or make the migration explicitly handle the absent case) instead of loosening the schema.\n\n**Do not** drop the `NOT NULL`, make the column nullable, or add `COALESCE(...)` purely to stop the error without understanding why those rows have no value — that defeats the constraint's purpose and lets the same bad data through silently going forward.\n\n**Verify** by re-running the full migration end-to-end against a representative copy of the table (including a `down` if it has one), confirming zero violations across every affected row — not just the one that failed first — and confirm the constraint is still in place and unchanged. In your report, state the root cause in one sentence (e.g. \"legacy rows held `''` rather than NULL, which the backfill's `IS NULL` clause missed\"), not just \"it applies now.\"",
1592
+ "outputSha256": "94c860bb0cf39c76ce565d340112664f76c73e3400594bc3fc39df588bcfc164",
1593
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1594
+ "deterministic": [],
1595
+ "judge": {
1596
+ "verdict": "pass",
1597
+ "reason": "Answer diagnoses which rows fail and why the backfill missed them: 'WHERE status IS NULL but legacy rows actually hold '' or some sentinel', CASE with no ELSE, missed join. It fixes the backfill's own logic: 'Widen the backfill's condition or add the missing CASE/join branch so those rows get a real value.' It explicitly refuses to drop/loosen the NOT NULL constraint."
1598
+ },
1599
+ "passed": true
1600
+ },
1601
+ {
1602
+ "output": "The fix is to find why those rows produce a NULL — not to drop the `NOT NULL` constraint so the backfill succeeds.\n\nConcretely:\n\n1. **Identify the actual offending rows** before changing anything. Run the backfill's `UPDATE` as a `SELECT` against the same predicate and inspect the rows that would end up NULL:\n\n ```sql\n SELECT id, <source columns the backfill derives from>\n FROM target_table\n WHERE <backfill's WHERE clause>\n AND <the derived expression> IS NULL;\n ```\n\n The violation means the derivation logic returns NULL for some real case it didn't anticipate — a genuinely different row history, a legacy sentinel, a joined row that's missing, a `CASE` with no `ELSE`, and so on.\n\n2. **Fix the root cause in the backfill**, based on what those rows are:\n - A `CASE` missing an `ELSE` / an unhandled branch → handle the branch.\n - A derivation that reads a `LEFT JOIN`ed row that's sometimes absent → decide the correct value (COALESCE against a real default, or fix the join).\n - A legacy sentinel like `''` where the code assumed `NULL` → widen the predicate/derivation to match it.\n - Data that genuinely has no determinable value → decide the correct value for that class of row (a real default, or the row legitimately needs a different treatment), and encode that decision — don't leave NULL.\n\n3. **Do not** drop or relax the `NOT NULL` constraint to make the error go away, and do not skip the backfill step. The constraint is what caught the gap in the derivation; removing it lets invalid data through and silently hides the bug. The only time removing a constraint is legitimate is when the constraint itself was wrong to add — and then the report must say so and why, not just \"it stopped erroring.\"\n\n4. **Verify against the full dataset**, not just the row that originally failed: re-run the backfill end-to-end over a representative copy of the table and confirm zero rows remain that violate the constraint after the update.\n\n5. **Report the root cause in one sentence**, e.g.:\n\n ```\n Fixed: migrations/2026_09_18_backfill_status.sql\n - Root cause: 340 legacy rows had status = '' (empty string, not NULL)\n from a prior import, which the derivation's NULL-only branch missed\n - Extended the CASE to map '' to 'unknown'\n - Backfill re-run clean against a full-size copy of the table\n ```\n\nThe one-line test: if the fix makes the error disappear without changing which value those rows end up with, you fixed the symptom, not the cause.",
1603
+ "outputSha256": "8d6d8ec6176b8436d2c7be2954c9b55385345a596cc7d37bda5383ac76e15043",
1604
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1605
+ "deterministic": [],
1606
+ "judge": {
1607
+ "verdict": "pass",
1608
+ "reason": "Step 1 shows a SELECT to find offending rows and explains the derivation returns NULL for an unanticipated case (legacy sentinel, missing join, CASE with no ELSE). Step 2 fixes the backfill's own logic (handle branch, COALESCE, widen predicate for '' sentinel). Step 3 explicitly rejects dropping/relaxing NOT NULL. Concrete example: '' mapped to 'unknown'."
1609
+ },
1610
+ "passed": true
1611
+ },
1612
+ {
1613
+ "output": "Don't loosen or drop the `NOT NULL` constraint to make the insert succeed — the constraint is doing its job by catching rows where the backfill produced a NULL. Fix the actual source of the NULLs.\n\nWork through it like this:\n\n1. **Reproduce and find the actual offending rows.** Run the backfill's logic (or a `SELECT` mirroring it) filtering for the rows that end up with the NOT NULL column still NULL. Get the real data, not just the first error.\n\n2. **Diagnose why those rows produce NULL.** Usually one of a few things:\n - The backfill's `WHERE` clause is too narrow and misses a case (e.g. it matches `status IS NULL` but the legacy rows actually have `status = ''`, so they never get updated).\n - The default/derivation expression evaluates to NULL for a legitimate variant of the data (an unmatched `JOIN`, a `CASE` with no `ELSE`, a parse that fails on some format).\n - The rows genuinely have no correct value and need a human/business decision — in which case that decision is the fix, not a constraint change.\n\n3. **Fix the backfill, not the constraint.** Extend the `WHERE` condition, add an `ELSE` branch or `COALESCE` with a correct value, or fix the join. If truly-unfixable legacy rows exist, resolve them deliberately (correct the data or document a justified exception) — still not by dropping `NOT NULL`.\n\nOnly if the constraint itself was genuinely wrong for the schema would you change it, and then you'd say so explicitly and why. \"A NULL reached a NOT NULL column\" is normally a bug upstream.\n\n4. **Verify across the whole dataset, not just the failing row.** Re-run the migration up (and down if it has one) against a representative copy of the table and confirm zero rows end up NULL in a column that requires a value. Then report the root cause in one sentence, e.g.:\n\n```\nFixed: migrations/2026_09_18_backfill_status.sql\n - Root cause: 340 legacy rows had status = '' (not NULL) from a prior\n import, which the backfill's WHERE status IS NULL clause missed\n - Extended the backfill condition to also match status = ''\n - Migration re-run clean against a full-size copy of the table\n```\n\nThe one-line root cause matters — \"fixed the error\" without saying why it happened usually means the same class of failure comes back next backfill.",
1614
+ "outputSha256": "5e10378197a9fb5537c4956b01d3c46c88593a63b159d9ced9041583a3e2ad5d",
1615
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1616
+ "deterministic": [],
1617
+ "judge": {
1618
+ "verdict": "pass",
1619
+ "reason": "Step 1 says find the actual offending rows; step 2 diagnoses why backfill logic missed them, concretely naming WHERE clause too narrow (status IS NULL vs legacy status = ''), unmatched JOIN, CASE with no ELSE. Step 3 fixes the backfill condition/derivation (extend WHERE, add ELSE/COALESCE, fix join), explicitly refusing to drop NOT NULL. No fail criterion met."
1620
+ },
1621
+ "passed": true
1622
+ },
1623
+ {
1624
+ "output": "A `NOT NULL` violation on a backfill usually means the backfill's own default/derivation logic has a gap, not that the constraint is wrong. Per this skill, you must never drop or loosen the constraint to make the error stop — find and fix the actual cause.\n\n**Step 1: Reproduce and see the real rows.** Run the backfill against staging and capture the exact error text plus the offending rows. Something like:\n\n```sql\nSELECT id, <backfill_target_col>, <source_cols_you_derived_from>\nFROM <table>\nWHERE <backfill_target_col> IS NULL\n AND <your_backfill_WHERE_clause>;\n```\n\nA very common cause: the backfill only handles one \"empty\" representation but the data actually has several — e.g. `WHERE status IS NULL` misses rows where `status = ''`, or rows from a legacy import with a sentinel value, or rows whose join to the source table returned no match (so a `COALESCE`/`LEFT JOIN` yielded NULL).\n\n**Step 2: Fix the backfill logic, not the constraint.** Widen the condition to cover the real cases and supply a defensible value for each:\n\n```sql\nUPDATE <table>\nSET <target_col> = COALESCE(<derived_value>, <explicit_default>)\nWHERE <target_col> IS NULL\n OR <target_col> = ''; -- the case that was missed\n```\n\nWhere the correct value genuinely isn't derivable, decide it explicitly (a documented default, or a follow-up manual pass) rather than letting NULL through.\n\n**Step 3: Verify across the whole table, not just the row that errored.** Confirm zero remaining violations before re-running the migration:\n\n```sql\nSELECT COUNT(*) FROM <table> WHERE <target_col> IS NULL; -- expect 0\n```\n\nThen re-run the migration end-to-end (up, and down if it has one) against a representative dataset.\n\n**Step 4: Report the root cause in one sentence**, e.g.:\n\n```\nFixed: migrations/2026_09_18_backfill_status.sql\n - Root cause: 340 legacy rows had status = '' (empty string, not NULL)\n from a prior import, which the backfill's WHERE status IS NULL\n clause missed\n - Extended the backfill condition to also match status = ''\n - Migration re-run clean against a full-size copy of the table\n```\n\nThe constraint did its job here — it caught rows the backfill wasn't handling. The fix belongs in the backfill (or in the code path that produced those rows), never in weakening the constraint. Also resist \"just delete the failing migration\" or \"just bump the timeout\" — neither resolves the underlying data gap.",
1625
+ "outputSha256": "9cba7e54aaee4f6681866853d5fd5222992032a45ea1f59137051bacdff766a3",
1626
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1627
+ "deterministic": [],
1628
+ "judge": {
1629
+ "verdict": "pass",
1630
+ "reason": "Step 1 names the specific failing rows and why the backfill missed them: 'WHERE status IS NULL misses rows where status = ''' or legacy sentinel values or unmatched joins yielding NULL. Step 2 fixes the backfill condition itself: 'Widen the condition... OR <target_col> = ''' with COALESCE default. Explicitly refuses to drop/loosen the constraint."
1631
+ },
1632
+ "passed": true
1633
+ },
1634
+ {
1635
+ "output": "A `NOT NULL` violation during a backfill means the constraint is doing its job: it's catching rows your backfill logic didn't cover. The fix is to find those rows and correct the backfill (or the data), not to drop the constraint.\n\n**1. Reproduce and read the exact error.** Run the backfill against a copy of the failing data and capture the error text and the offending row(s). Then find *why* those rows are NULL:\n\n```sql\nSELECT id, <new_col>\nFROM <table>\nWHERE <new_col> IS NULL\nLIMIT 50;\n```\n\nLook at what's special about them. The usual causes:\n\n- **The backfill's WHERE clause missed a case.** Very common: the column was `''` (empty string, not NULL) from a legacy import, so `WHERE new_col IS NULL` never matched it. Or a join in a `UPDATE ... FROM` had no match for certain IDs, leaving the column NULL.\n- **A source column is itself NULL** for some legacy rows, so the derivation `new_col = f(old_col)` yields NULL.\n- **Rows inserted after the backfill ran** (e.g. by app traffic or a concurrent migration) that the backfill never saw.\n\n**2. Fix the backfill condition, not the constraint.** Once you know the actual case:\n\n- Empty-string case → extend the condition:\n ```sql\n UPDATE <table>\n SET new_col = <derivation>\n WHERE new_col IS NULL\n OR new_col = ''; -- the case the original backfill missed\n ```\n- Source column NULL / no join match → decide the correct value for those rows (a documented default, a sentinel like `'unknown'`, or a separate remediation) rather than silently letting them stay NULL. If a NULL is genuinely invalid, fix the data that produced it.\n- Concurrent inserts → make the backfill idempotent and re-runnable (a safe `WHERE new_col IS NULL`), run it after the writers are quiesced or in a way that catches late rows, then add the constraint.\n\nIf the backfill derives the value from a join, use a `LEFT JOIN` and inspect the NULL-right-side rows explicitly so you can see which IDs had no match — don't just switch to an inner join, which would silently skip them.\n\n**3. Order the constraint add after a clean backfill.** The sequence should be: backfill runs and reports zero remaining NULLs → *then* `ALTER TABLE ... SET NOT NULL`. If the constraint is already in place and blocking the backfill, that's the sign you added it before the backfill completed — fix the data, then let the backfill finish.\n\n**4. Verify.** Re-run the backfill against a full-size copy and confirm zero rows still violate the `NOT NULL` (not just the row that originally errored), then confirm the constraint is satisfied across the whole table.\n\n**What not to do:** dropping or loosening the `NOT NULL` (or making the column nullable) to make the error disappear. That just admits invalid data permanently and hides the bug that produced the NULLs — exactly the case the constraint existed to catch. Same for skipping the backfill step to get a \"green\" migration run.\n\n**Report it as:**\n```\nFixed: migrations/2026_09_18_backfill_status.sql\n - Root cause: 340 legacy rows had status = '' (empty string) from a\n prior import, which the backfill's WHERE status IS NULL missed\n - Extended the backfill condition to also match and repair status = ''\n - Backfill re-run clean against a full-size copy; 0 rows remaining NULL\n - NOT NULL constraint now applies successfully\n```",
1636
+ "outputSha256": "08e7512ffabfb23daab194808445b63d8d50be55ad7f8ccb84563d5d10909afa",
1637
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1638
+ "deterministic": [],
1639
+ "judge": {
1640
+ "verdict": "pass",
1641
+ "reason": "Answer investigates which rows fail and why the backfill missed them (empty-string legacy value vs WHERE IS NULL, NULL source column, no join match, concurrent inserts) and fixes the backfill condition (extending WHERE to match new_col=''), explicitly warning against dropping/loosening the NOT NULL constraint."
1642
+ },
1643
+ "passed": true
1644
+ },
1645
+ {
1646
+ "output": "Fix the actual cause, not the constraint. A `NOT NULL` violation during a backfill means the backfill's logic didn't cover every row that needs a value — the constraint is doing its job by catching exactly that.\n\n**Find the violating rows first.** Run the backfill's `UPDATE` as a `SELECT` against the same `WHERE` clause and inspect what it returns, then check the rows that *didn't* match — that's where the problem lives. Common causes:\n\n- The backfill condition is too narrow: it filters `WHERE status IS NULL`, but legacy rows actually hold an empty string `''` or a sentinel value from a prior import, so those rows never got updated.\n- A `LEFT JOIN` in the backfill sources its value from a table that has no matching row for some records, leaving the target column untouched.\n- A `CASE`/`COALESCE` expression falls through to `NULL` for an unhandled case.\n\n**Fix the backfill's coverage, not the schema.** Broaden the `WHERE` clause to catch the missed rows, add a `COALESCE` fallback that matches the intended derivation for the unhandled case, or fix the join so every row gets a source value. Re-run against a full-size copy of the table and confirm zero rows still violate the constraint afterward — not just the one row that originally errored.\n\n**What not to do:** don't drop or relax the `NOT NULL` constraint to make the error stop, and don't skip the migration step. If the constraint itself turns out to be genuinely wrong — for example the column is legitimately optional and the constraint was added by mistake — that's a different situation: state explicitly in your report that the constraint is the wrong thing, and why. That's a schema correction, not a workaround, and it needs to be justified on its own terms rather than used as a shortcut past a failing backfill.\n\n**Verify:** migration re-runs clean end-to-end (up, and down if it has one) against a representative dataset; no constraint was dropped or loosened; and your report names the root cause in one sentence — e.g. \"340 legacy rows held `status = ''` from a prior import, which the `WHERE status IS NULL` clause missed; extended the condition to also match `''`.\"",
1647
+ "outputSha256": "ac4e9a8356f97eaa48b0089036d7a4dce8c2ae06305eaa13f191651207a7deec",
1648
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1649
+ "deterministic": [],
1650
+ "judge": {
1651
+ "verdict": "pass",
1652
+ "reason": "Answer states investigating which rows fail and why backfill logic missed them: 'check the rows that didn't match', 'WHERE status IS NULL but legacy rows hold empty string '' or sentinel', LEFT JOIN no match, CASE/COALESCE falls to NULL. Fixes backfill logic: 'Broaden the WHERE clause', 'add a COALESCE fallback', 'fix the join'. Explicitly warns against dropping/relaxing NOT NULL. No fail criterion holds."
1653
+ },
1654
+ "passed": true
1655
+ },
1656
+ {
1657
+ "output": "Don't drop or relax the NOT NULL constraint to make the error stop. That constraint is telling you the backfill logic produced (or left) NULLs in a column that must not have them, and silencing it just moves the bad data downstream. Fix the actual cause instead.\n\n**Find the offending rows first.** Query for exactly which rows the backfill left/turned NULL and why:\n\n```sql\nSELECT id, <source columns>\nFROM <table>\nWHERE <new_column> IS NULL\n AND <the predicate the backfill was supposed to cover>;\n```\n\nThe usual suspects are:\n\n- **The backfill's WHERE/DEFAULT logic missed a case.** E.g. it did `WHERE status IS NULL` but the legacy data actually has `status = ''` (empty string, not NULL), or a column the backfill derives from is itself NULL for those rows. Widen the backfill's condition to cover the real data shape — that's the smallest correct fix.\n- **A row legitimately has no valid source value.** Then the default derivation is wrong for that population; decide what the correct value is (a documented sentinel, a value pulled from another column) rather than letting NULL through.\n- **The column should allow NULL after all** — only if the constraint itself is genuinely wrong for the domain. If so, that's a deliberate schema decision and must be stated explicitly in your report with the reason. This is rare; assume the constraint is right until you've proven otherwise.\n\n**Then fix the migration, not the constraint.** Adjust the backfill's predicate or derivation, re-run it, and confirm no NULLs remain:\n\n```sql\n-- verify the whole backfill, not just the row that first failed\nSELECT count(*) FROM <table> WHERE <new_column> IS NULL; -- must be 0\n```\n\nDepending on the engine/order, you may also need to make the migration idempotent (`IF NOT EXISTS`, a guarded update) so a partial re-run doesn't stumble on already-migrated rows.\n\n**Verify end-to-end:** re-run the full migration up (and down if it has one) against a representative copy of the table, confirm zero NULL violations across the entire backfill — not just the originally failing row — and confirm the NOT NULL constraint is still in place and enforced.\n\n**Report the root cause in one sentence,** e.g.:\n\n```\nFixed: migrations/2026_09_18_backfill_status.sql\n - Root cause: 340 legacy rows had status = '' (not NULL) from a prior\n import, which the backfill's WHERE status IS NULL clause missed\n - Extended the backfill condition to also match status = ''\n - Migration re-run clean against a full-size copy of the table; 0 NULLs remain\n```\n\nThe key discipline: the violation is evidence, and the fix belongs in the data or the backfill logic, never in removing the constraint that caught the problem.",
1658
+ "outputSha256": "5aa2447d254a862ffec4f934690213b30f381e2056610ff622878f1936e340bd",
1659
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1660
+ "deterministic": [],
1661
+ "judge": {
1662
+ "verdict": "pass",
1663
+ "reason": "Answer tells to query which rows are NULL and why (SELECT ... WHERE new_column IS NULL), names the concrete missed case: backfill did WHERE status IS NULL but legacy data has status = '' (empty string), and fixes the backfill condition to also match status = ''. Explicitly rejects dropping/relaxing the constraint. Pass criterion 1 and 2 both concretely satisfied."
1664
+ },
1665
+ "passed": true
1666
+ },
1667
+ {
1668
+ "output": "Find the actual rows violating the constraint first — don't drop or loosen the NOT NULL constraint to make the error stop.\n\nConcretely:\n\n1. **Reproduce and read the exact error.** Run the backfill against a copy of the failing data and get the specific rows and column that trip the constraint.\n\n2. **Diagnose why a NULL is reaching a NOT NULL column.** Typical causes:\n - The backfill's `WHERE`/`CASE` logic missed a case (e.g. it handled `status IS NULL` but legacy rows have `status = ''`, or a `COALESCE` chain doesn't cover every branch).\n - A join in the backfill produced no match, so the derived value came out NULL.\n - Rows legitimately predate the column and need a real default derived from other data, not a placeholder.\n\n3. **Fix the logic or the data — whichever is actually wrong:**\n - If the backfill derivation is incomplete, extend it so every row gets a valid value.\n - If specific rows have genuinely bad/missing source data, decide the correct resolution (derive from another column, merge, or a documented exception) rather than loosening the constraint to accept NULL.\n\n4. **For a large table, batch the backfill** into chunks that each commit, so a failure doesn't leave the table half-migrated and so you don't hold a long lock.\n\n5. **Verify:** re-run the migration end-to-end (up, and down if it has one) against a representative dataset — not just the originally failing row — and confirm zero constraint violations across the full backfill.\n\nThings NOT to do:\n- Don't `ALTER COLUMN ... DROP NOT NULL` just to make the error go away. The constraint is doing its job; the fix belongs in the backfill logic or the data.\n- Don't skip or delete the failing migration step to get a \"successful\" run.\n- Don't claim it's fixed until the full backfill runs clean on representative data.\n\nReport the root cause in one sentence, e.g.: \"Backfill's `WHERE status IS NULL` missed 340 legacy rows with `status = ''`; extended the condition to match both.\"",
1669
+ "outputSha256": "18f8cff78a539f28b6e12ee6690fb9fb841083c8e5cdfdd3f004f701372f9b92",
1670
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1671
+ "deterministic": [],
1672
+ "judge": {
1673
+ "verdict": "pass",
1674
+ "reason": "Step 2 diagnoses why backfill logic missed rows (WHERE/CASE missed a case, e.g. handled status IS NULL but legacy rows have status=''), and step 3 extends the derivation so every row gets a valid value. Explicitly warns against DROP NOT NULL. Concrete example given: 'WHERE status IS NULL missed 340 legacy rows with status = '''."
1675
+ },
1676
+ "passed": true
1677
+ },
1678
+ {
1679
+ "output": "A `NOT NULL` violation during a backfill almost always means the backfill's own logic missed a case — not that the constraint is wrong. So the fix belongs in the backfill (or the data), not in the constraint.\n\nConcretely:\n\n1. **Find the actual violating rows first.** Run the backfill's `SELECT`/`UPDATE` predicate as a plain `SELECT` and add the inverse condition — e.g. if you're doing `UPDATE t SET status = 'active' WHERE status IS NULL`, look for `WHERE status IS NOT NULL AND status = ''` or other non-`NULL` sentinel values that the backfill doesn't touch. A very common cause is legacy data that uses `''` (or `'unknown'`, `-1`, etc.) where the code now expects `NULL` or a real value.\n\n2. **Decide what the correct value is for those rows.** Options, in order of preference:\n - Extend the backfill condition to cover the missed case (e.g. `WHERE status IS NULL OR status = ''`).\n - If the missing value must be derived from another column/table, add that derivation.\n - If some rows genuinely have no valid value, that's a data-quality question to resolve explicitly (merge/dedupe/backfill from source), not something to paper over.\n\n3. **Do not** drop or `ALTER ... DROP NOT NULL` on the column to make the error stop. The constraint is catching exactly the case you need to understand; loosening it lets bad data into the table and hides the real problem.\n\n4. **Verify against a full-size copy of the table**, not just the row that originally failed — the whole point is that the backfill's predicate was too narrow, so you need to confirm *no* row is left unhandled.\n\n5. **Report the root cause**, e.g.:\n\n ```\n Fixed: migrations/2026_09_18_backfill_status.sql\n - Root cause: 340 legacy rows had status = '' (empty string, not NULL)\n from an earlier import; the backfill only matched `status IS NULL`\n - Extended the predicate to `status IS NULL OR status = ''`\n - Re-ran clean against a full-size copy of the table\n ```\n\nIf you paste the backfill statement and the exact error (including which column is `NOT NULL`), I can point at the specific predicate that needs widening.",
1680
+ "outputSha256": "dbe6824f665f6058ed889edb850a9cf203d095e24f0e30196a17be6b7ecab0b0",
1681
+ "promptSha256": "b58b051dc4cb6623e6e9b5cbac31e1c9393b8756ae7d95323f6d16b2d13ca92b",
1682
+ "deterministic": [],
1683
+ "judge": {
1684
+ "verdict": "pass",
1685
+ "reason": "Answer states investigating which rows fail and why: step 1 says run the predicate and look for non-NULL sentinels like '' that the backfill doesn't touch, citing legacy data using '' where code expects NULL. Step 2 fixes the backfill condition concretely: 'WHERE status IS NULL OR status = '''. Step 3 explicitly warns against DROP NOT NULL. No fail criterion holds."
1686
+ },
1687
+ "passed": true
1688
+ }
1689
+ ]
1690
+ },
1691
+ {
1692
+ "id": "planner-regression-not-a-hint",
1693
+ "kind": "behavior",
1694
+ "prompt": "A query that used to use an index now runs a sequential scan and got slow after the last deploy. How should I fix it?",
1695
+ "strictness": "high",
1696
+ "trials": 10,
1697
+ "passes": 10,
1698
+ "passRate": 1,
1699
+ "passAtK": 1,
1700
+ "grader": "judge",
1701
+ "status": "ran",
1702
+ "trialRecords": [
1703
+ {
1704
+ "output": "Short answer: don't reach for a planner hint first — find out *why* the index stopped being used, then fix that cause. Here's how to work it.\n\n## 1. Reproduce and classify\n\nRun `EXPLAIN (ANALYZE, BUFFERS)` (Postgres) or `EXPLAIN ANALYZE` (MySQL) on the slow query and read the actual plan. Confirm it's really a sequential/full scan and note the estimated vs. actual row counts. Then figure out which of these happened in \"the last deploy\":\n\n- **Statistics went stale** — a large data load or backfill changed the data distribution, but nobody re-ran `ANALYZE`, so the planner's row estimates are wrong and it *thinks* the index is a worse choice.\n- **The query's shape changed** — a deploy edit (a new `WHERE`, a different `ORDER BY`, a function wrapped around the indexed column like `WHERE date(col)` or `WHERE lower(col)`) means the query no longer matches the index's column order or no longer uses the leading column at all.\n- **The index itself changed or vanished** — a migration dropped it, renamed it, or reordered its columns so the leading column no longer matches the query's predicate.\n- **The data volume crossed a threshold** — the table genuinely grew enough that a sequential scan is now the *correct* plan for a low-selectivity predicate, and the \"regression\" is actually the planner being right.\n\n## 2. Fix the actual cause\n\n- **Stale statistics:** re-run `ANALYZE <table>` (Postgres) / `ANALYZE TABLE <table>` (MySQL). That alone resolves many regressions after a bulk load. Then confirm the plan flips back.\n- **Query shape changed:** look at the diff for this query/ORM call in the last deploy. If a new predicate was added that can't use the index, either give it its own index or rewrite the query so the indexed column is still usable as the leading column (avoid wrapping the indexed column in a function in the predicate, if possible). Per your composite-index ordering guidance, a composite index only helps if the query's leading predicate matches the index's leading column.\n- **Index changed:** check the migrations in this deploy for a dropped, renamed, or re-columned index on that table. If the correct fix is a new/corrected index, create it (`CREATE INDEX CONCURRENTLY` on Postgres to avoid locking a live table) matching the query's actual predicate and ordering.\n- **Planner is actually right:** if the predicate is genuinely low-selectivity and the table is large, a sequential scan may be the correct plan and the real \"slowness\" is downstream (e.g., a missing `LIMIT`, sorting, or an N+1). Fix that instead.\n\n## 3. Verify\n\nRe-run `EXPLAIN (ANALYZE, BUFFERS)` and confirm the plan now uses the index (or is the correct plan for the data) *and* that the actual runtime improved — not just that the query stopped being slow in one anecdotal run. Test against a dataset with a representative size and distribution, since a small dev dataset can hide exactly the statistics problem that caused this.\n\n## 4. Report\n\nState the root cause in one sentence, not just \"added an index.\" Example:\n\n```\nFixed: dashboard summary query\n - Root cause: last deploy added WHERE lower(email) = ..., which can't\n use the existing idx_users_email (plain column) — planner correctly\n fell back to a seq scan\n - Added idx_users_email_lower on lower(email); EXPLAIN now shows an\n index scan, query time 820ms -> 3ms on a full-size staging copy\n```\n\n## The rule that matters here\n\n**Never force a plan with a planner hint as your first move.** A hint papers over the real cause (stale stats, a changed index shape, a query that no longer matches the index) and will go stale itself the next time the data distribution shifts. Only consider a hint after you've ruled out stale statistics, a broken index shape, and a changed query — and even then, document why.",
1705
+ "outputSha256": "2bdaa1dfa53d8920759d43250d7d67750123a3292a3f3223bc4c82897c4d175c",
1706
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1707
+ "deterministic": [],
1708
+ "judge": {
1709
+ "verdict": "pass",
1710
+ "reason": "Step 1 explicitly recommends EXPLAIN (ANALYZE, BUFFERS) to confirm the scan and lists stale statistics (re-run ANALYZE), query shape change (function-wrapped column no longer matching index column order), and dropped/renamed/reordered index as root causes. Step 2 fixes each cause concretely (ANALYZE <table>, rewrite predicate, CREATE INDEX CONCURRENTLY). Hints only warned against, not proposed as fix."
1711
+ },
1712
+ "passed": true
1713
+ },
1714
+ {
1715
+ "output": "Start by confirming it's actually a plan change, not just more data. Run `EXPLAIN (ANALYZE, BUFFERS)` (Postgres) or `EXPLAIN ANALYZE` (MySQL) on the exact slow query and look at whether the index is still there and why the planner chose the scan.\n\nThe three common root causes, in the order I'd check them:\n\n**1. Stale statistics.** If the last deploy included a large data load or backfill, the planner's row estimates can be wildly off, so it concludes the index isn't worth using. Re-run `ANALYZE <table>` (Postgres) or `ANALYZE TABLE` (MySQL) and re-explains. If the plan flips back to the index, that's your whole fix — and it's the most common one after a big deploy.\n\n**2. The query's own shape changed.** If the deploy touched the query (added a function call on an indexed column, changed the column order in a composite index's matching WHERE/JOIN, added a `LIKE`/`OR` that defeats the index, or the index is now on `(a, b)` but the query filters on `b` alone), the index exists but can no longer be used. Fix the query or the index shape so they line up again — don't force the old plan with a hint.\n\n**3. The index was dropped or made invalid.** Check `pg_indexes`/`SHOW INDEX` for the table. A concurrent index build that failed, or a migration that recreated the table, can leave the index missing or invalid. If it's genuinely gone, recreate it (concurrently on Postgres, to avoid locking the table).\n\nIn all three cases, re-run `EXPLAIN ANALYZE` after the fix and confirm the plan actually uses the index and the timing is back to baseline. Don't reach for a planner hint first — a hint freezes today's plan but goes stale again the next time the data distribution shifts, and it hides the real cause rather than fixing it.",
1716
+ "outputSha256": "8aea2b9394e741f692758adaf5ceabd020f13bd2318a4782fb87a467d9069ce5",
1717
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1718
+ "deterministic": [],
1719
+ "judge": {
1720
+ "verdict": "pass",
1721
+ "reason": "Answer opens with EXPLAIN (ANALYZE, BUFFERS)/EXPLAIN ANALYZE to check if index still exists and why planner chose scan (criterion 1). It names three concrete root causes: stale statistics fixed via ANALYZE <table>/ANALYZE TABLE, query shape no longer matching index column order (e.g. index on (a,b) but query filters b alone) fixed by aligning query/index, and dropped/invalid index recreated. Explicitly warns against planner hints first, satisfying fail criterion avoidance."
1722
+ },
1723
+ "passed": true
1724
+ },
1725
+ {
1726
+ "output": "Start by confirming the regression is real, then work through the likely causes in order — stale statistics, changed query shape, or a missing/altered index — before you ever reach for a planner hint.\n\n## 1. Confirm the regression with EXPLAIN\n\nRun `EXPLAIN (ANALYZE, BUFFERS)` on Postgres (or `EXPLAIN ANALYZE` on MySQL) and read the actual plan and row estimates. Check:\n\n- Does the plan show a Sequential Scan / full table scan on the table you expect the index to serve?\n- Does the index still exist? (`\\d <table>` on Postgres, `SHOW INDEX FROM <table>` on MySQL.) A deploy may have dropped or renamed it.\n- Are the row estimates way off from the actual rows returned? A large mismatch points at stale statistics.\n\nDon't guess from \"it got slow\" alone — the plan tells you which of the causes below actually applies.\n\n## 2. Check statistics first (most common cause after a big data change)\n\nA deploy that also changed data volume or distribution can leave the planner's stats stale, so it mis-estimates selectivity and picks a seq scan.\n\n- Postgres: `ANALYZE <table>;`\n- MySQL: `ANALYZE TABLE <table>;`\n\nThen re-run the `EXPLAIN`. If the plan flips back to the index, stale statistics were the cause. For a table that changes a lot, check whether autovacuum/auto-analyze is keeping up — that's the real fix, not a one-off `ANALYZE`.\n\n## 3. If stats are fine, check the index shape against the current query\n\nA deploy often changes the query itself. Look at:\n\n- **Composite column order** — if the query's `WHERE`/`ORDER BY` columns changed, an existing composite index may no longer match. Per the composite-order guidance in `rules/patterns.mdc`, the index's leading column must match how the query filters.\n- **A function or cast wrapped around the indexed column** (e.g. `WHERE LOWER(email) = ...` or an implicit cast) — that makes a plain index unusable. Fix the query to be sargable, or add a matching expression/functional index.\n- **The query now selects/joins differently** — a new join or a wider `SELECT` can legitimately make a seq scan cheaper, in which case the \"fix\" is the query, not the index.\n\n## 4. Then verify\n\nRe-run `EXPLAIN (ANALYZE, BUFFERS)` and confirm the plan uses the index and the estimated-vs-actual rows line up. Don't call it fixed because it \"got faster\" — confirm the plan.\n\n## What not to do\n\n- **Don't force the old plan with a planner hint as the first move.** A hint papers over the actual cause (stale stats, a changed index shape) and can itself go stale the next time data distribution shifts. Fix the underlying cause; only hint if you've exhausted the real causes and documented why.\n- **Don't just `ANALYZE` once and walk away** if the table changes constantly — fix the stats maintenance, not the symptom.\n\nReport it with the root cause in one sentence, not just \"it's fast again\" — e.g. \"Root cause: stale statistics after a 40% row increase mis-estimated selectivity; re-running ANALYZE restored the index plan, and auto-analyze thresholds were tuned so it stays that way.\"",
1727
+ "outputSha256": "91ff0e07b20d3583c6384665badb01f1daf7c867adddd743a48c566999030cd6",
1728
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1729
+ "deterministic": [],
1730
+ "judge": {
1731
+ "verdict": "pass",
1732
+ "reason": "Step 1 recommends EXPLAIN (ANALYZE, BUFFERS) and checking whether the index still exists (\\d <table>) and whether row estimates are off (stale stats). Step 2 names stale statistics as root cause and fixes it with ANALYZE <table> plus autovacuum tuning. Step 3 covers query shape/composite column order mismatch. Hints explicitly deferred, not used as first fix."
1733
+ },
1734
+ "passed": true
1735
+ },
1736
+ {
1737
+ "output": "Start by confirming the plan actually changed, then find *why* before touching anything.\n\n**1. Confirm and localize with EXPLAIN.**\nRun `EXPLAIN (ANALYZE, BUFFERS) <query>` (Postgres) or `EXPLAIN ANALYZE` (MySQL). You want to see three things:\n- Is the index still there? (`\\d <table>` / `SHOW INDEX FROM <table>` — sometimes the last deploy dropped or renamed it.)\n- What does the planner *estimate* vs. what actually happened? A big row-count misestimate points at stale statistics, not a bad index.\n- Did the query's own shape change in the deploy (column list, a new function wrapping a column, an added OR/order change) so it no longer matches the index's column order?\n\n**2. Fix the actual cause, in order of likelihood:**\n\n- **Stale statistics** after a large data change: run `ANALYZE <table>` (Postgres) / `ANALYZE TABLE <table>` (MySQL). This is the most common \"it was fast, now it's not\" cause and the cheapest fix. Re-check `EXPLAIN` afterward.\n- **Query shape no longer matches the index.** If the deploy wrapped a column (`WHERE lower(email) = ...`) or changed predicate order, the existing index on `email` no longer applies — that's a real regression in the query, fix it there, or add a correctly-shaped index (e.g. on `lower(email)`). For composite indexes, confirm the column order still matches the query's equality-then-range pattern.\n- **Index genuinely missing:** re-create it, and check the migration that should have created it actually applied.\n- **Data distribution shifted** so the plan is legitimately correct now — verify with `ANALYZE` before assuming the old plan was right.\n\n**3. Only then consider a planner hint** — and if you use one, state in your report why the statistics/index shape could not be fixed instead. A hint goes stale itself the next time data distribution moves.\n\n**Do not** jump to forcing the old plan with a hint before checking statistics and index shape, and do not raise any timeout to make the symptom stop.\n\n**Report format** — root cause in one sentence, e.g.:\n```\nFixed: query in reports/summary.sql\n - Root cause: a 4M-row bulk import left pg_statistic stale, so the\n planner estimated 12 rows in the date range instead of 480k and\n chose a seq scan.\n - Ran ANALYZE orders; EXPLAIN ANALYZE now shows Index Scan.\n```\n\nDone only once `EXPLAIN`/`EXPLAIN ANALYZE` confirms the expected plan (not just that the query is faster), and the report names the actual cause.",
1738
+ "outputSha256": "97d03e9b9dc58a8e347c9e4353c92607c01eeedaae5a7144c44265555fb28b4c",
1739
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1740
+ "deterministic": [],
1741
+ "judge": {
1742
+ "verdict": "pass",
1743
+ "reason": "Step 1 uses EXPLAIN (ANALYZE, BUFFERS) to check if the index still exists and whether estimates vs actuals point to stale statistics; step 2 fixes root causes concretely: run ANALYZE <table>, fix query shape (lower(email) example) or add correctly-shaped index, re-create missing index. Hints only considered last, explicitly warning against forcing old plan first."
1744
+ },
1745
+ "passed": true
1746
+ },
1747
+ {
1748
+ "output": "Start by confirming what actually changed, since \"after the last deploy\" is the only clue you have right now and it pins the regression to a specific set of DDL/migration/code changes.\n\n**1. Confirm the plan and the index.**\nRun `EXPLAIN (ANALYZE, BUFFERS)` on Postgres (or `EXPLAIN ANALYZE` on MySQL) on the current build. You want three facts: (a) is it genuinely a seq/full-table scan now, (b) does the index still exist and is it valid — on Postgres a failed `CREATE INDEX CONCURRENTLY` can leave an **invalid** index that the planner silently ignores, and (c) how many rows is it actually scanning vs. the table size. An invalid index from a bad deploy is one of the most common causes of exactly this symptom.\n\n**2. Check for stale statistics.**\nIf the last deploy did a large backfill, bulk import, or a big delete, the planner's row estimates can be badly off even though the index is fine. On Postgres run `ANALYZE <table>` (or check whether autovacuum/autoanalyze last ran — `pg_stat_user_tables.last_autoanalyze`); on MySQL run `ANALYZE TABLE <table>`. If the plan flips back to the index after an `ANALYZE`, stale stats were the cause — but don't stop there, understand *why* they went stale (e.g. a backfill that bypassed the normal write path) so it doesn't recur on the next data load.\n\n**3. Check whether the query's shape changed.**\nLook at whether the deploy altered the query itself — an added `OR`, a function wrapped around the indexed column (`WHERE lower(email) = ...`), a changed parameter or `LIMIT`, a cast, or a different set of columns — in a way that no longer matches the index's column order or predicates. Composite index order matters: an index on `(a, b)` won't serve a `WHERE b = ?` predicate alone. If the deploy added a filter the index doesn't cover, the smallest correct fix is usually to correct the index shape (or a new covering index) to match the actual query, not to hint the planner.\n\n**4. Apply the smallest root-cause fix.**\n- Invalid index from a failed `CONCURRENTLY` → drop and recreate it correctly.\n- Stale statistics → `ANALYZE` the table, and fix whatever write path kept stats stale.\n- Index no longer matches the query → fix the index shape / composite order per the project's indexing conventions.\n- Planner genuinely underestimating (correlated columns, etc.) → that's a statistics-target or index-shape issue, still not a hint.\n\n**5. Verify.**\nRe-run `EXPLAIN (ANALYZE, BUFFERS)` and confirm the plan is back on the index and the estimated vs. actual row counts are close. Confirm against representative data volume, not just a dev-sized table.\n\n**What not to do:** do not reach for a planner hint (`/*+ INDEX(...) */`, `enable_seqscan = off`, a MySQL `FORCE INDEX`) as the first fix. A hint masks the real cause — stale stats or a stale/invalid index — and can go stale itself when data distribution shifts again. Also, don't assume the index is fine just because it's listed in `\\d` / `SHOW INDEX`; verify it's valid and actually usable.\n\nState the root cause in one sentence when you report it, e.g. \"The deploy's backfill left planner statistics stale; `ANALYZE` restored the index plan\" or \"A failed `CREATE INDEX CONCURRENTLY` left the index invalid, so the planner ignored it.\" \"I added a hint and it's fast now\" is not a root cause.",
1749
+ "outputSha256": "22952f6428cdb85f78214eaa8e0443221d0459da6e6b9e48a613c825c1e81726",
1750
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1751
+ "deterministic": [],
1752
+ "judge": {
1753
+ "verdict": "pass",
1754
+ "reason": "Step 1 runs EXPLAIN (ANALYZE, BUFFERS) to confirm the plan and whether the index still exists/is valid (invalid index from failed CREATE INDEX CONCURRENTLY). Step 2 checks stale statistics and runs ANALYZE <table>. Step 3 covers query shape no longer matching index column order (composite (a,b) vs WHERE b). Step 4 fixes each root cause. Step 5 warns against hints as first fix."
1755
+ },
1756
+ "passed": true
1757
+ },
1758
+ {
1759
+ "output": "A planner regression like this has a few distinct root causes, and the fix depends entirely on which one it actually is — so start by confirming the plan and the reason, not by forcing the old plan back.\n\nFirst, reproduce it and look at the actual plan. Run `EXPLAIN (ANALYZE, BUFFERS)` (Postgres) or `EXPLAIN ANALYZE <query>` (MySQL) and read which node the scan sits at. Then check, in order:\n\n1. **Is the index still there?** The last deploy may have dropped it as part of a migration, or a migration that was supposed to create it failed partway. If it's gone, that's the whole story — recreate it.\n\n2. **Are statistics stale?** If the deploy included a large data load or a big backfill, the planner's row estimates (`pg_statistic` on Postgres) may be out of date the planner now thinks the table is small relative to the filter, so it prefers a seq scan. Running `ANALYZE <table>` (Postgres) or `ANALYZE TABLE <table>` (MySQL) frequently restores the index plan by itself. This is the most common cause after a deploy that moved a lot of data.\n\n3. **Did the query's own shape change in the deploy?** If the query was edited — a new column added to the WHERE clause, an `OR` introduced, a function wrapped around an indexed column (`WHERE lower(email) = ...`), a cast, or the column order in a composite index no longer matching the query's equality/range pattern — the index may simply no longer be usable for this query. In that case fix the index shape (or the query) to match, rather than the other way around.\n\n4. **Has the data distribution shifted enough that the planner's choice is actually correct?** If the filter now matches a large fraction of the table, a seq scan can genuinely be cheaper and the \"regression\" is a schedule/volume problem, not a planner bug.\n\nOnly after you've ruled out the first three should you consider anything plan-forcing. Adding a planner hint or `pg_hint_plan`-style directive to pin the old plan is the last resort, not the first: it hides the actual cause, and it goes stale the next time data distribution shifts again — you'll be back here with a hint that now points at the wrong plan.\n\nSo the concrete path: `EXPLAIN ANALYZE` the query, confirm whether the index exists, re-run `ANALYZE` on the table(s) involved, and if the plan is still wrong, diff the query text and the index definition between the last-known-good deploy and now to find what changed shape. Fix that root cause. Then re-run `EXPLAIN ANALYZE` and confirm the plan now uses the index — not just that the query stopped being slow.",
1760
+ "outputSha256": "e79a8d2b0d07550d7c84cf421bc131820c7d3705caefeca15d5a6db9c5d680c1",
1761
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1762
+ "deterministic": [],
1763
+ "judge": {
1764
+ "verdict": "pass",
1765
+ "reason": "Answer tells user to run EXPLAIN (ANALYZE, BUFFERS)/EXPLAIN ANALYZE and check whether the index still exists and whether statistics are stale, recommending ANALYZE <table>. It names concrete root causes (dropped index, stale stats, query shape/column-order mismatch) and fixes each. Hints are explicitly last resort, not first response."
1766
+ },
1767
+ "passed": true
1768
+ },
1769
+ {
1770
+ "output": "Start by classifying this correctly, because \"add a hint\" is the wrong first move. Run `EXPLAIN (ANALYZE, BUFFERS)` on Postgres (or `EXPLAIN ANALYZE` on MySQL) and read the actual plan and the actual cause rather than forcing the old plan back.\n\nThe three most likely root causes, in order of how often they turn out to be the real one:\n\n1. **Stale statistics.** If the last deploy included a large data change (a bulk backfill, a big import, a partition swap), the planner's cached stats no longer reflect the table's actual distribution, so it estimates a sequential scan as cheaper than it is. Fix: run `ANALYZE <table>` (Postgres) / `ANALYZE TABLE <table>` (MySQL) and re-check the plan. On Postgres, also check whether autovacuum analyzed it and whether the table has a recent enough `last_analyze` in `pg_stat_user_tables`. This is the fix more often than any other single one.\n\n2. **The index shape no longer matches the query.** If the deploy changed the query — different columns in the `WHERE`, a new leading predicate, a changed join — the index's column order may no longer serve it. A composite index `(a, b)` only helps a predicate on `b` alone if the query can use a skip-scan or a separate index. Fix the index shape to match the query's actual access pattern (or add the index the new predicate needs), rather than hinting the planner onto an index that structurally can't serve the new query well.\n\n3. **The index is gone or unusable.** Confirm it still exists (`\\d <table>` on Postgres, `SHOW INDEX FROM <table>` on MySQL), that it isn't marked invalid, and — on Postgres — that it isn't a partial index whose predicate the query no longer satisfies.\n\nOnly after ruling those out should a planner hint even be on the table, and even then as a documented, temporary measure — a hint goes stale the next time the data distribution shifts and masks rather than fixes the real cause.\n\nSo the workflow is: reproduce with `EXPLAIN ANALYZE`, check statistics first, check index shape against the query's current form, check the index still exists and is usable, fix whichever is actually wrong, then re-run `EXPLAIN ANALYZE` to confirm the expected plan. Report the one-sentence root cause — e.g. \"an 8M-row backfill in the last deploy left statistics stale; `ANALYZE` restored the index scan\" — not just \"it's fast again.\"",
1771
+ "outputSha256": "f74ee905422297b166cc0a609760d2097265dcd86133e91d76f96136121ecf18",
1772
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1773
+ "deterministic": [],
1774
+ "judge": {
1775
+ "verdict": "pass",
1776
+ "reason": "Answer states running EXPLAIN (ANALYZE, BUFFERS)/EXPLAIN ANALYZE first and checking stats (ANALYZE <table>, pg_stat_user_tables last_analyze) before any fix. It names concrete root causes: stale statistics (fix: ANALYZE), index shape no longer matching query column order (fix: change/add index), and index dropped/invalid (fix: confirm via \\d). Hints only after ruling these out, as temporary."
1777
+ },
1778
+ "passed": true
1779
+ },
1780
+ {
1781
+ "output": "Don't guess — reproduce the plan change and find why the index stopped being chosen. The fix follows from the actual cause, and the order you check things in matters.\n\n**1. Confirm the change and the index's state**\n\nRun `EXPLAIN (ANALYZE, BUFFERS)` (Postgres) or `EXPLAIN ANALYZE` (MySQL) on the slow query. Confirm it's really a sequential/full-table scan, and that the index you expect still exists and isn't invalid/corrupt (`\\d <table>` in psql, or check `pg_index.indisvalid`). If the deploy dropped or renamed the index, that's the whole story and you're done — recreate it.\n\n**2. Check the cheap, common causes first**\n\n- **Stale statistics.** A large data change around the deploy can leave the planner's row estimates way off, so it prefers a scan. Run `ANALYZE <table>` (Postgres) or `ANALYZE TABLE <table>` (MySQL) and re-run `EXPLAIN`. This is the single most common cause and the cheapest fix.\n- **The query shape changed.** Read the diff of the deploy. If the code now wraps a column in a function (`WHERE lower(email) = ...`, `WHERE date(col) = ...`), casts it, or compares mismatched types, the planner can no longer use a plain index on that column. Fix the query to be sargable (e.g. a functional/expression index, or rewrite the predicate) rather than forcing a plan.\n- **Index shape no longer matches the predicate/order.** If the query's `WHERE`/`ORDER BY` changed such that a composite index's leading column or column order no longer lines up — per `rules/patterns.mdc`'s composite-order guidance — the planner can't use it effectively. Correct the index shape.\n\n**3. Only then consider plan forcing**\n\nA planner hint to force the old plan is a last resort, not the first move. It papers over the actual cause (stale stats, a changed index shape) and can itself go stale the next time the data distribution shifts.\n\n**4. Verify and report**\n\nRe-run `EXPLAIN`/`EXPLAIN ANALYZE` and confirm the expected plan is back — \"the query no longer errors\" isn't the bar here; the plan is the artifact. Then write it up with the root cause stated in one sentence, e.g.:\n\n```\nFixed: query in reports/active_users.sql\n - Root cause: statistics left stale after the 4M-row import in the\n same deploy; planner estimated ~50 rows, actual ~900k, chose seq scan\n - Ran ANALYZE users; EXPLAIN now shows Index Scan using users_status_idx\n```\n\nTwo things to hold the line on: don't drop or loosen a constraint to make a symptom go away (that's not in play here, but the same instinct applies to disabling the index), and don't reach for a hint before you've checked whether statistics are stale or the index shape drifted from the query — those are fixable causes, and a hint just hides them until the next data shift.",
1782
+ "outputSha256": "f0d251fd2a9081a7e89ce5afc3f91cebab0e52b7e00750fd691e9e44952baf6c",
1783
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1784
+ "deterministic": [],
1785
+ "judge": {
1786
+ "verdict": "pass",
1787
+ "reason": "Step 1 runs EXPLAIN (ANALYZE, BUFFERS) and checks index existence/validity (pg_index.indisvalid). Step 2 names stale statistics and prescribes ANALYZE <table>, plus query-shape mismatch (function-wrapped column) and composite index order drift, fixing each cause. Hints explicitly deferred to last resort, satisfying fail criterion avoidance."
1788
+ },
1789
+ "passed": true
1790
+ },
1791
+ {
1792
+ "output": "Start by confirming the regression rather than guessing. Run `EXPLAIN (ANALYZE, BUFFERS)` (Postgres) or `EXPLAIN ANALYZE` (MySQL) on the query and compare the plan to what you expect — you want to see whether the index still exists and is simply not being chosen, or whether it's gone. Then work through the likely causes in order:\n\n1. **Stale statistics.** If the last deploy included a large data change (big backfill, bulk load, mass update), the planner's stats may no longer reflect reality and it can decide a seq scan is cheaper when it isn't. For Postgres run `ANALYZE <table>`; for MySQL `ANALYZE TABLE <table>`. Re-run `EXPLAIN` and see if the index comes back. This is the single most common cause of \"it was fine until the last deploy\" and it's a zero-risk fix.\n\n2. **Changed query shape vs. index shape.** If the deploy changed the query — new WHERE columns, a function wrapped around an indexed column, a different column order in the predicate, a type cast that prevents index use — the existing index may genuinely no longer match. Per composite-index guidance, an index only helps if the leading columns match the query's equality/range pattern. If the query changed, the correct fix is usually to adjust the index (add/reorder columns) or rewrite the query to be index-friendly, not to force the old plan.\n\n3. **Index dropped or renamed** during the deploy's migrations. Check it still exists (`\\d <table>` in Postgres, `SHOW INDEX FROM <table>` in MySQL). If a migration dropped it, that's the root cause — restore it.\n\nOnly after ruling out the above should you consider planner hints or `SET enable_seqscan = off` — and even then treat them as a stopgap, not the fix, since a hint goes stale the next time the data distribution shifts and masks the underlying problem.\n\nAfter fixing, verify with `EXPLAIN ANALYZE` that the expected plan actually returns (index scan/bitmap scan, not seq scan), and check the timing improved under realistic data volume — not just that the query runs without error. State the root cause in one sentence in your report (e.g., \"stale statistics after a 4M-row backfill caused the planner to prefer a seq scan; re-running ANALYZE restored the index scan\"), because that's what lets the next person recognize the same pattern.\n\nOne caution: don't reach for dropping or loosening a constraint, skipping a migration step, or raising a lock/timeout to make a related error disappear — those silence symptoms and leave the real cause in place.",
1793
+ "outputSha256": "25ddb2a2c797ccdd6a5edf9c8f91c7e8d952737ed4f47b43c8d8e611ca11a77b",
1794
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1795
+ "deterministic": [],
1796
+ "judge": {
1797
+ "verdict": "pass",
1798
+ "reason": "Answer recommends EXPLAIN (ANALYZE, BUFFERS) to check if index still exists and stats stale, and running ANALYZE <table> before any fix (criterion 1). It names concrete root causes: stale statistics fixed via ANALYZE, changed query shape vs index column order fixed by adjusting index/rewriting query, and dropped index restored (criterion 2). Hints/enable_seqscan=off only mentioned as last-resort stopgap after investigation, so fail criterion 1 does not hold."
1799
+ },
1800
+ "passed": true
1801
+ },
1802
+ {
1803
+ "output": "Start by confirming the regression and finding its actual cause — don't reach for a planner hint first.\n\n**1. Confirm the plan changed**\n\nRun `EXPLAIN (ANALYZE, BUFFERS)` on Postgres (or `EXPLAIN ANALYZE` on MySQL) for the slow query. Confirm it's genuinely doing a seq scan where it used to use the index, and note the estimated vs. actual row counts.\n\n**2. Check the usual root causes, in order**\n\n- **Stale statistics.** If the last deploy loaded/changed a lot of data, the planner may be working off stale stats and mis-estimating. Fix: `ANALYZE <table>` (Postgres) or `ANALYZE TABLE <table>` (MySQL). Re-run `EXPLAIN` and see if the plan corrects itself.\n- **The index shape no longer matches the query.** If the deploy changed the query (added/removed a `WHERE` column, changed a join), an index whose column order no longer matches the new predicate order stops being usable. Fix the index to match the query's actual predicate/join shape per composite-key ordering guidance — not the query to match a stale index.\n- **The index was dropped or never migrated.** Check the index still exists (`\\d <table>` / `SHOW INDEX FROM <table>`); a migration may have dropped or renamed it.\n- **The data distribution genuinely changed** so the seq scan is now *correct* (e.g., the predicate now matches most of the table). In that case the \"regression\" is the planner being right and the query itself needs rethinking.\n\n**3. Only after those, consider a hint**\n\nA planner hint forces the old plan and can go stale itself when data distribution shifts again. It's a last resort, not the first fix.\n\n**4. Verify**\n\nRe-run `EXPLAIN ANALYZE` and confirm the expected plan (index scan, reasonable row estimates), not just that the query runs. Then report the root cause in one sentence — e.g. \"statistics were stale after a 2M-row backfill; `ANALYZE` restored the index scan.\"\n\nThe key rule: fix the underlying cause (stale stats, index shape mismatch, dropped index) before ever forcing a plan with a hint.",
1804
+ "outputSha256": "d023c1f276ae773efc5e395f1c732ae60da145bc5e49793610383b386f0f3166",
1805
+ "promptSha256": "b497cb38f8f45d11c22e2704d2e01c9ff6fa5d39d5e67ab8a085a4ee15b3de56",
1806
+ "deterministic": [],
1807
+ "judge": {
1808
+ "verdict": "pass",
1809
+ "reason": "Step 1 uses EXPLAIN (ANALYZE, BUFFERS) to confirm the plan; step 2 checks stale statistics with concrete fix ANALYZE <table>, index shape no longer matching query predicate order, and dropped index via \\d/SHOW INDEX. Hints are explicitly deferred to last resort, satisfying both pass criteria and avoiding the fail criterion."
1810
+ },
1811
+ "passed": true
1812
+ }
1813
+ ]
1814
+ }
1815
+ ],
1816
+ "verdict": "fail",
1817
+ "scope": "bundled",
1818
+ "skillDigest": "38321825b2ee3f6efefd4b44d2d83a1681b756119312c59e1dbb5808a5da2bac",
1819
+ "catalogDigest": "14504a0807a0089488b9cb690c4b13f20865cd7a7fb69a1e5d8dfea8bfd5fbd1",
1820
+ "judgePromptVersion": "2026-09-25.1",
1821
+ "runner": "deepseek",
1822
+ "model": "deepseek-chat",
1823
+ "runnerPromptVersion": "2026-09-25.1",
1824
+ "recordedAt": "2026-09-25T18:19:39.689Z",
1825
+ "judge": "deepseek",
1826
+ "judgeModel": "deepseek-chat"
1827
+ }
1828
+ ]
1829
+ }